1 // SPDX-License-Identifier: GPL-2.0-only 2 /* Copyright (c) 2011-2014 PLUMgrid, http://plumgrid.com 3 * Copyright (c) 2016 Facebook 4 * Copyright (c) 2018 Covalent IO, Inc. http://covalent.io 5 */ 6 #include <uapi/linux/btf.h> 7 #include <linux/bpf-cgroup.h> 8 #include <linux/kernel.h> 9 #include <linux/types.h> 10 #include <linux/slab.h> 11 #include <linux/bpf.h> 12 #include <linux/btf.h> 13 #include <linux/bpf_verifier.h> 14 #include <linux/filter.h> 15 #include <net/netlink.h> 16 #include <linux/file.h> 17 #include <linux/vmalloc.h> 18 #include <linux/stringify.h> 19 #include <linux/bsearch.h> 20 #include <linux/sort.h> 21 #include <linux/perf_event.h> 22 #include <linux/ctype.h> 23 #include <linux/error-injection.h> 24 #include <linux/bpf_lsm.h> 25 #include <linux/security.h> 26 #include <linux/verification.h> 27 #include <linux/btf_ids.h> 28 #include <linux/poison.h> 29 #include <linux/module.h> 30 #include <linux/cpumask.h> 31 #include <linux/cnum.h> 32 #include <linux/bpf_mem_alloc.h> 33 #include <net/xdp.h> 34 #include <linux/trace_events.h> 35 #include <linux/kallsyms.h> 36 37 #include "diagnostics.h" 38 #include "disasm.h" 39 40 static const struct bpf_verifier_ops * const bpf_verifier_ops[] = { 41 #define BPF_PROG_TYPE(_id, _name, prog_ctx_type, kern_ctx_type) \ 42 [_id] = & _name ## _verifier_ops, 43 #define BPF_MAP_TYPE(_id, _ops) 44 #define BPF_LINK_TYPE(_id, _name) 45 #include <linux/bpf_types.h> 46 #undef BPF_PROG_TYPE 47 #undef BPF_MAP_TYPE 48 #undef BPF_LINK_TYPE 49 }; 50 51 enum bpf_features { 52 BPF_FEAT_RDONLY_CAST_TO_VOID = 0, 53 BPF_FEAT_STREAMS = 1, 54 __MAX_BPF_FEAT, 55 }; 56 57 struct bpf_mem_alloc bpf_global_percpu_ma; 58 static bool bpf_global_percpu_ma_set; 59 60 /* bpf_check() is a static code analyzer that walks eBPF program 61 * instruction by instruction and updates register/stack state. 62 * All paths of conditional branches are analyzed until 'bpf_exit' insn. 63 * 64 * The first pass is depth-first-search to check that the program is a DAG. 65 * It rejects the following programs: 66 * - larger than BPF_MAXINSNS insns 67 * - if loop is present (detected via back-edge) 68 * - unreachable insns exist (shouldn't be a forest. program = one function) 69 * - out of bounds or malformed jumps 70 * The second pass is all possible path descent from the 1st insn. 71 * Since it's analyzing all paths through the program, the length of the 72 * analysis is limited to 64k insn, which may be hit even if total number of 73 * insn is less then 4K, but there are too many branches that change stack/regs. 74 * Number of 'branches to be analyzed' is limited to 1k 75 * 76 * On entry to each instruction, each register has a type, and the instruction 77 * changes the types of the registers depending on instruction semantics. 78 * If instruction is BPF_MOV64_REG(BPF_REG_1, BPF_REG_5), then type of R5 is 79 * copied to R1. 80 * 81 * All registers are 64-bit. 82 * R0 - return register 83 * R1-R5 argument passing registers 84 * R6-R9 callee saved registers 85 * R10 - frame pointer read-only 86 * 87 * At the start of BPF program the register R1 contains a pointer to bpf_context 88 * and has type PTR_TO_CTX. 89 * 90 * Verifier tracks arithmetic operations on pointers in case: 91 * BPF_MOV64_REG(BPF_REG_1, BPF_REG_10), 92 * BPF_ALU64_IMM(BPF_ADD, BPF_REG_1, -20), 93 * 1st insn copies R10 (which has FRAME_PTR) type into R1 94 * and 2nd arithmetic instruction is pattern matched to recognize 95 * that it wants to construct a pointer to some element within stack. 96 * So after 2nd insn, the register R1 has type PTR_TO_STACK 97 * (and -20 constant is saved for further stack bounds checking). 98 * Meaning that this reg is a pointer to stack plus known immediate constant. 99 * 100 * Most of the time the registers have SCALAR_VALUE type, which 101 * means the register has some value, but it's not a valid pointer. 102 * (like pointer plus pointer becomes SCALAR_VALUE type) 103 * 104 * When verifier sees load or store instructions the type of base register 105 * can be: PTR_TO_MAP_VALUE, PTR_TO_CTX, PTR_TO_STACK, PTR_TO_SOCKET. These are 106 * four pointer types recognized by check_mem_access() function. 107 * 108 * PTR_TO_MAP_VALUE means that this register is pointing to 'map element value' 109 * and the range of [ptr, ptr + map's value_size) is accessible. 110 * 111 * registers used to pass values to function calls are checked against 112 * function argument constraints. 113 * 114 * ARG_PTR_TO_MAP_KEY is one of such argument constraints. 115 * It means that the register type passed to this function must be 116 * PTR_TO_STACK and it will be used inside the function as 117 * 'pointer to map element key' 118 * 119 * For example the argument constraints for bpf_map_lookup_elem(): 120 * .ret_type = RET_PTR_TO_MAP_VALUE_OR_NULL, 121 * .arg1_type = ARG_CONST_MAP_PTR, 122 * .arg2_type = ARG_PTR_TO_MAP_KEY, 123 * 124 * ret_type says that this function returns 'pointer to map elem value or null' 125 * function expects 1st argument to be a const pointer to 'struct bpf_map' and 126 * 2nd argument should be a pointer to stack, which will be used inside 127 * the helper function as a pointer to map element key. 128 * 129 * On the kernel side the helper function looks like: 130 * u64 bpf_map_lookup_elem(u64 r1, u64 r2, u64 r3, u64 r4, u64 r5) 131 * { 132 * struct bpf_map *map = (struct bpf_map *) (unsigned long) r1; 133 * void *key = (void *) (unsigned long) r2; 134 * void *value; 135 * 136 * here kernel can access 'key' and 'map' pointers safely, knowing that 137 * [key, key + map->key_size) bytes are valid and were initialized on 138 * the stack of eBPF program. 139 * } 140 * 141 * Corresponding eBPF program may look like: 142 * BPF_MOV64_REG(BPF_REG_2, BPF_REG_10), // after this insn R2 type is FRAME_PTR 143 * BPF_ALU64_IMM(BPF_ADD, BPF_REG_2, -4), // after this insn R2 type is PTR_TO_STACK 144 * BPF_LD_MAP_FD(BPF_REG_1, map_fd), // after this insn R1 type is CONST_PTR_TO_MAP 145 * BPF_RAW_INSN(BPF_JMP | BPF_CALL, 0, 0, 0, BPF_FUNC_map_lookup_elem), 146 * here verifier looks at prototype of map_lookup_elem() and sees: 147 * .arg1_type == ARG_CONST_MAP_PTR and R1->type == CONST_PTR_TO_MAP, which is ok, 148 * Now verifier knows that this map has key of R1->map_ptr->key_size bytes 149 * 150 * Then .arg2_type == ARG_PTR_TO_MAP_KEY and R2->type == PTR_TO_STACK, ok so far, 151 * Now verifier checks that [R2, R2 + map's key_size) are within stack limits 152 * and were initialized prior to this call. 153 * If it's ok, then verifier allows this BPF_CALL insn and looks at 154 * .ret_type which is RET_PTR_TO_MAP_VALUE_OR_NULL, so it sets 155 * R0->type = PTR_TO_MAP_VALUE_OR_NULL which means bpf_map_lookup_elem() function 156 * returns either pointer to map value or NULL. 157 * 158 * When type PTR_TO_MAP_VALUE_OR_NULL passes through 'if (reg != 0) goto +off' 159 * insn, the register holding that pointer in the true branch changes state to 160 * PTR_TO_MAP_VALUE and the same register changes state to CONST_IMM in the false 161 * branch. See check_cond_jmp_op(). 162 * 163 * After the call R0 is set to return type of the function and registers R1-R5 164 * are set to NOT_INIT to indicate that they are no longer readable. 165 * 166 * The following reference types represent a potential reference to a kernel 167 * resource which, after first being allocated, must be checked and freed by 168 * the BPF program: 169 * - PTR_TO_SOCKET_OR_NULL, PTR_TO_SOCKET 170 * 171 * When the verifier sees a helper call return a reference type, it allocates a 172 * pointer id for the reference and stores it in the current function state. 173 * Similar to the way that PTR_TO_MAP_VALUE_OR_NULL is converted into 174 * PTR_TO_MAP_VALUE, PTR_TO_SOCKET_OR_NULL becomes PTR_TO_SOCKET when the type 175 * passes through a NULL-check conditional. For the branch wherein the state is 176 * changed to CONST_IMM, the verifier releases the reference. 177 * 178 * For each helper function that allocates a reference, such as 179 * bpf_sk_lookup_tcp(), there is a corresponding release function, such as 180 * bpf_sk_release(). When a reference type passes into the release function, 181 * the verifier also releases the reference. If any unchecked or unreleased 182 * reference remains at the end of the program, the verifier rejects it. 183 */ 184 185 /* verifier_state + insn_idx are pushed to stack when branch is encountered */ 186 struct bpf_verifier_stack_elem { 187 /* verifier state is 'st' 188 * before processing instruction 'insn_idx' 189 * and after processing instruction 'prev_insn_idx' 190 */ 191 struct bpf_verifier_state st; 192 int insn_idx; 193 int prev_insn_idx; 194 struct bpf_verifier_stack_elem *next; 195 /* length of verifier log at the time this state was pushed on stack */ 196 u32 log_pos; 197 u64 diag_log_pos; 198 }; 199 200 #define BPF_COMPLEXITY_LIMIT_JMP_SEQ 8192 201 #define BPF_COMPLEXITY_LIMIT_STATES 64 202 203 #define BPF_GLOBAL_PERCPU_MA_MAX_SIZE 512 204 205 #define BPF_PRIV_STACK_MIN_SIZE 64 206 207 static int acquire_reference(struct bpf_verifier_env *env, int insn_idx, int parent_id); 208 static int __release_reference_nomark(struct bpf_verifier_state *state, int id); 209 static int release_reference_nomark(struct bpf_verifier_env *env, int id); 210 static int release_reference(struct bpf_verifier_env *env, int id); 211 static void invalidate_non_owning_refs(struct bpf_verifier_env *env); 212 static void invalidate_rcu_protected_refs(struct bpf_verifier_env *env); 213 static bool in_rbtree_lock_required_cb(struct bpf_verifier_env *env); 214 static bool is_tracing_prog_type(enum bpf_prog_type type); 215 static int ref_set_non_owning(struct bpf_verifier_env *env, 216 struct bpf_reg_state *reg); 217 static bool is_trusted_reg(struct bpf_verifier_env *env, const struct bpf_reg_state *reg); 218 static inline bool in_sleepable_context(struct bpf_verifier_env *env); 219 static const char *non_sleepable_context_description(struct bpf_verifier_env *env); 220 static void scalar32_min_max_add(struct bpf_reg_state *dst_reg, struct bpf_reg_state *src_reg); 221 static void scalar_min_max_add(struct bpf_reg_state *dst_reg, struct bpf_reg_state *src_reg); 222 223 static void bpf_map_ptr_store(struct bpf_insn_aux_data *aux, 224 struct bpf_map *map, 225 bool unpriv, bool poison) 226 { 227 unpriv |= bpf_map_ptr_unpriv(aux); 228 aux->map_ptr_state.unpriv = unpriv; 229 aux->map_ptr_state.poison = poison; 230 aux->map_ptr_state.map_ptr = map; 231 } 232 233 static void bpf_map_key_store(struct bpf_insn_aux_data *aux, u64 state) 234 { 235 bool poisoned = bpf_map_key_poisoned(aux); 236 237 aux->map_key_state = state | BPF_MAP_KEY_SEEN | 238 (poisoned ? BPF_MAP_KEY_POISON : 0ULL); 239 } 240 241 static void update_ref_obj(struct ref_obj_desc *ref_obj, struct bpf_reg_state *reg) 242 { 243 ref_obj->id = reg->id; 244 ref_obj->parent_id = reg->parent_id; 245 ref_obj->cnt++; 246 } 247 248 static int validate_ref_obj(struct bpf_verifier_env *env, struct ref_obj_desc *ref_obj) 249 { 250 if (ref_obj->cnt > 1) { 251 verifier_bug(env, "function expects only one referenced object but got %d\n", 252 ref_obj->cnt); 253 return -EFAULT; 254 } 255 256 return 0; 257 } 258 259 struct bpf_kfunc_meta { 260 struct btf *btf; 261 const struct btf_type *proto; 262 const char *name; 263 const u32 *flags; 264 s32 id; 265 }; 266 267 struct btf *btf_vmlinux; 268 269 typedef struct argno { 270 int argno; 271 } argno_t; 272 273 static argno_t argno_from_reg(u32 regno) 274 { 275 return (argno_t){ .argno = regno }; 276 } 277 278 static argno_t argno_from_arg(u32 arg) 279 { 280 return (argno_t){ .argno = -arg }; 281 } 282 283 static int reg_from_argno(argno_t a) 284 { 285 if (a.argno >= 0) 286 return a.argno; 287 if (a.argno >= -MAX_BPF_FUNC_REG_ARGS) 288 return -a.argno; 289 return -1; 290 } 291 292 static int arg_from_argno(argno_t a) 293 { 294 if (a.argno < 0) 295 return -a.argno; 296 return -1; 297 } 298 299 static int arg_idx_from_argno(argno_t a) 300 { 301 return arg_from_argno(a) - 1; 302 } 303 304 static const char *btf_type_name(const struct btf *btf, u32 id) 305 { 306 return btf_name_by_offset(btf, btf_type_by_id(btf, id)->name_off); 307 } 308 309 static DEFINE_MUTEX(bpf_verifier_lock); 310 static DEFINE_MUTEX(btf_vmlinux_lock); 311 static DEFINE_MUTEX(bpf_percpu_ma_lock); 312 313 __printf(2, 3) static void verbose(void *private_data, const char *fmt, ...) 314 { 315 struct bpf_verifier_env *env = private_data; 316 va_list args; 317 318 if (!bpf_verifier_log_needed(&env->log)) 319 return; 320 321 va_start(args, fmt); 322 bpf_verifier_vlog(&env->log, fmt, args); 323 va_end(args); 324 } 325 326 static void verbose_invalid_scalar(struct bpf_verifier_env *env, 327 struct bpf_reg_state *reg, 328 struct bpf_retval_range range, const char *ctx, 329 const char *reg_name) 330 { 331 bool unknown = true; 332 333 verbose(env, "%s the register %s has", ctx, reg_name); 334 if (reg_smin(reg) > S64_MIN) { 335 verbose(env, " smin=%lld", reg_smin(reg)); 336 unknown = false; 337 } 338 if (reg_smax(reg) < S64_MAX) { 339 verbose(env, " smax=%lld", reg_smax(reg)); 340 unknown = false; 341 } 342 if (unknown) 343 verbose(env, " unknown scalar value"); 344 verbose(env, " should have been in [%d, %d]\n", range.minval, range.maxval); 345 } 346 347 static bool reg_not_null(struct bpf_verifier_env *env, const struct bpf_reg_state *reg) 348 { 349 enum bpf_reg_type type; 350 351 type = reg->type; 352 if (type_may_be_null(type)) 353 return false; 354 355 /* 356 * The types below guarantee a non-NULL base, an unbounded offset can 357 * still wrap base + offset to zero. 358 */ 359 if (reg_smin(reg) <= -BPF_MAX_VAR_OFF || reg_smax(reg) >= BPF_MAX_VAR_OFF) 360 return false; 361 362 type = base_type(type); 363 return type == PTR_TO_SOCKET || 364 type == PTR_TO_TCP_SOCK || 365 type == PTR_TO_XDP_SOCK || 366 type == PTR_TO_BUF || 367 type == PTR_TO_MAP_VALUE || 368 type == PTR_TO_MAP_KEY || 369 type == PTR_TO_SOCK_COMMON || 370 (type == PTR_TO_BTF_ID && is_trusted_reg(env, reg)) || 371 (type == PTR_TO_MEM && !(reg->type & PTR_UNTRUSTED)) || 372 type == CONST_PTR_TO_MAP; 373 } 374 375 static struct btf_record *reg_btf_record(const struct bpf_reg_state *reg) 376 { 377 struct btf_record *rec = NULL; 378 struct btf_struct_meta *meta; 379 380 if (reg->type == PTR_TO_MAP_VALUE) { 381 rec = reg->map_ptr->record; 382 } else if (type_is_ptr_alloc_obj(reg->type)) { 383 meta = btf_find_struct_meta(reg->btf, reg->btf_id); 384 if (meta) 385 rec = meta->record; 386 } 387 return rec; 388 } 389 390 bool bpf_subprog_is_global(const struct bpf_verifier_env *env, int subprog) 391 { 392 struct bpf_func_info_aux *aux = env->prog->aux->func_info_aux; 393 394 return aux && aux[subprog].linkage == BTF_FUNC_GLOBAL; 395 } 396 397 static bool subprog_returns_void(struct bpf_verifier_env *env, int subprog) 398 { 399 const struct btf_type *type, *func, *func_proto; 400 const struct btf *btf = env->prog->aux->btf; 401 u32 btf_id; 402 403 btf_id = env->prog->aux->func_info[subprog].type_id; 404 405 func = btf_type_by_id(btf, btf_id); 406 if (verifier_bug_if(!func, env, "btf_id %u not found", btf_id)) 407 return false; 408 409 func_proto = btf_type_by_id(btf, func->type); 410 if (!func_proto) 411 return false; 412 413 type = btf_type_skip_modifiers(btf, func_proto->type, NULL); 414 if (!type) 415 return false; 416 417 return btf_type_is_void(type); 418 } 419 420 const char *bpf_subprog_name(const struct bpf_verifier_env *env, int subprog) 421 { 422 struct bpf_func_info *info; 423 424 if (!env->prog->aux->func_info) 425 return ""; 426 427 info = &env->prog->aux->func_info[subprog]; 428 return btf_type_name(env->prog->aux->btf, info->type_id); 429 } 430 431 void bpf_mark_subprog_exc_cb(struct bpf_verifier_env *env, int subprog) 432 { 433 struct bpf_subprog_info *info = subprog_info(env, subprog); 434 435 info->is_cb = true; 436 info->is_async_cb = true; 437 info->is_exception_cb = true; 438 } 439 440 static bool subprog_is_exc_cb(struct bpf_verifier_env *env, int subprog) 441 { 442 return subprog_info(env, subprog)->is_exception_cb; 443 } 444 445 static bool reg_may_point_to_spin_lock(const struct bpf_reg_state *reg) 446 { 447 return btf_record_has_field(reg_btf_record(reg), BPF_SPIN_LOCK | BPF_RES_SPIN_LOCK); 448 } 449 450 static bool type_is_rdonly_mem(u32 type) 451 { 452 return type & MEM_RDONLY; 453 } 454 455 static bool is_acquire_function(enum bpf_func_id func_id, 456 const struct bpf_map *map) 457 { 458 enum bpf_map_type map_type = map ? map->map_type : BPF_MAP_TYPE_UNSPEC; 459 460 if (func_id == BPF_FUNC_sk_lookup_tcp || 461 func_id == BPF_FUNC_sk_lookup_udp || 462 func_id == BPF_FUNC_skc_lookup_tcp || 463 func_id == BPF_FUNC_ringbuf_reserve || 464 func_id == BPF_FUNC_kptr_xchg) 465 return true; 466 467 if (func_id == BPF_FUNC_map_lookup_elem && 468 (map_type == BPF_MAP_TYPE_SOCKMAP || 469 map_type == BPF_MAP_TYPE_SOCKHASH)) 470 return true; 471 472 return false; 473 } 474 475 static bool is_ptr_cast_function(enum bpf_func_id func_id) 476 { 477 return func_id == BPF_FUNC_tcp_sock || 478 func_id == BPF_FUNC_sk_fullsock || 479 func_id == BPF_FUNC_skc_to_tcp_sock || 480 func_id == BPF_FUNC_skc_to_tcp6_sock || 481 func_id == BPF_FUNC_skc_to_udp6_sock || 482 func_id == BPF_FUNC_skc_to_mptcp_sock || 483 func_id == BPF_FUNC_skc_to_tcp_timewait_sock || 484 func_id == BPF_FUNC_skc_to_tcp_request_sock; 485 } 486 487 static bool is_sync_callback_calling_kfunc(u32 btf_id); 488 static bool is_async_callback_calling_kfunc(u32 btf_id); 489 static bool is_callback_calling_kfunc(u32 btf_id); 490 491 static bool is_bpf_wq_set_callback_kfunc(u32 btf_id); 492 static bool is_task_work_add_kfunc(u32 func_id); 493 494 static bool is_sync_callback_calling_function(enum bpf_func_id func_id) 495 { 496 return func_id == BPF_FUNC_for_each_map_elem || 497 func_id == BPF_FUNC_find_vma || 498 func_id == BPF_FUNC_loop || 499 func_id == BPF_FUNC_user_ringbuf_drain; 500 } 501 502 static bool is_async_callback_calling_function(enum bpf_func_id func_id) 503 { 504 return func_id == BPF_FUNC_timer_set_callback; 505 } 506 507 static bool is_callback_calling_function(enum bpf_func_id func_id) 508 { 509 return is_sync_callback_calling_function(func_id) || 510 is_async_callback_calling_function(func_id); 511 } 512 513 bool bpf_is_sync_callback_calling_insn(struct bpf_insn *insn) 514 { 515 return (bpf_helper_call(insn) && is_sync_callback_calling_function(insn->imm)) || 516 (bpf_pseudo_kfunc_call(insn) && is_sync_callback_calling_kfunc(insn->imm)); 517 } 518 519 bool bpf_is_async_callback_calling_insn(struct bpf_insn *insn) 520 { 521 return (bpf_helper_call(insn) && is_async_callback_calling_function(insn->imm)) || 522 (bpf_pseudo_kfunc_call(insn) && is_async_callback_calling_kfunc(insn->imm)); 523 } 524 525 static bool is_async_cb_sleepable(struct bpf_verifier_env *env, struct bpf_insn *insn) 526 { 527 /* bpf_timer callbacks are never sleepable. */ 528 if (bpf_helper_call(insn) && insn->imm == BPF_FUNC_timer_set_callback) 529 return false; 530 531 /* bpf_wq and bpf_task_work callbacks are always sleepable. */ 532 if (bpf_pseudo_kfunc_call(insn) && insn->off == 0 && 533 (is_bpf_wq_set_callback_kfunc(insn->imm) || is_task_work_add_kfunc(insn->imm))) 534 return true; 535 536 verifier_bug(env, "unhandled async callback in is_async_cb_sleepable"); 537 return false; 538 } 539 540 bool bpf_is_may_goto_insn(struct bpf_insn *insn) 541 { 542 return insn->code == (BPF_JMP | BPF_JCOND) && insn->src_reg == BPF_MAY_GOTO; 543 } 544 545 static bool is_spi_bounds_valid(struct bpf_func_state *state, int spi, int nr_slots) 546 { 547 int allocated_slots = state->allocated_stack / BPF_REG_SIZE; 548 549 /* We need to check that slots between [spi - nr_slots + 1, spi] are 550 * within [0, allocated_stack). 551 * 552 * Please note that the spi grows downwards. For example, a dynptr 553 * takes the size of two stack slots; the first slot will be at 554 * spi and the second slot will be at spi - 1. 555 */ 556 return spi - nr_slots + 1 >= 0 && spi < allocated_slots; 557 } 558 559 static int stack_slot_obj_get_spi(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 560 const char *obj_kind, int nr_slots) 561 { 562 int off, spi; 563 564 if (!tnum_is_const(reg->var_off)) { 565 verbose(env, "%s has to be at a constant offset\n", obj_kind); 566 return -EINVAL; 567 } 568 569 off = reg->var_off.value; 570 if (off >= 0 || off % BPF_REG_SIZE) { 571 verbose(env, "cannot pass in %s at an offset=%d\n", obj_kind, off); 572 return -EINVAL; 573 } 574 575 spi = bpf_get_spi(off); 576 if (spi + 1 < nr_slots) { 577 verbose(env, "cannot pass in %s at an offset=%d\n", obj_kind, off); 578 return -EINVAL; 579 } 580 581 if (!is_spi_bounds_valid(bpf_func(env, reg), spi, nr_slots)) 582 return -ERANGE; 583 return spi; 584 } 585 586 static int dynptr_get_spi(struct bpf_verifier_env *env, struct bpf_reg_state *reg) 587 { 588 return stack_slot_obj_get_spi(env, reg, "dynptr", BPF_DYNPTR_NR_SLOTS); 589 } 590 591 static int iter_get_spi(struct bpf_verifier_env *env, struct bpf_reg_state *reg, int nr_slots) 592 { 593 return stack_slot_obj_get_spi(env, reg, "iter", nr_slots); 594 } 595 596 static int irq_flag_get_spi(struct bpf_verifier_env *env, struct bpf_reg_state *reg) 597 { 598 return stack_slot_obj_get_spi(env, reg, "irq_flag", 1); 599 } 600 601 static enum bpf_dynptr_type arg_to_dynptr_type(enum bpf_arg_type arg_type) 602 { 603 switch (arg_type & DYNPTR_TYPE_FLAG_MASK) { 604 case DYNPTR_TYPE_LOCAL: 605 return BPF_DYNPTR_TYPE_LOCAL; 606 case DYNPTR_TYPE_RINGBUF: 607 return BPF_DYNPTR_TYPE_RINGBUF; 608 case DYNPTR_TYPE_SKB: 609 return BPF_DYNPTR_TYPE_SKB; 610 case DYNPTR_TYPE_XDP: 611 return BPF_DYNPTR_TYPE_XDP; 612 case DYNPTR_TYPE_SKB_META: 613 return BPF_DYNPTR_TYPE_SKB_META; 614 case DYNPTR_TYPE_FILE: 615 return BPF_DYNPTR_TYPE_FILE; 616 default: 617 return BPF_DYNPTR_TYPE_INVALID; 618 } 619 } 620 621 static enum bpf_type_flag get_dynptr_type_flag(enum bpf_dynptr_type type) 622 { 623 switch (type) { 624 case BPF_DYNPTR_TYPE_LOCAL: 625 return DYNPTR_TYPE_LOCAL; 626 case BPF_DYNPTR_TYPE_RINGBUF: 627 return DYNPTR_TYPE_RINGBUF; 628 case BPF_DYNPTR_TYPE_SKB: 629 return DYNPTR_TYPE_SKB; 630 case BPF_DYNPTR_TYPE_XDP: 631 return DYNPTR_TYPE_XDP; 632 case BPF_DYNPTR_TYPE_SKB_META: 633 return DYNPTR_TYPE_SKB_META; 634 case BPF_DYNPTR_TYPE_FILE: 635 return DYNPTR_TYPE_FILE; 636 default: 637 return 0; 638 } 639 } 640 641 static bool dynptr_type_referenced(enum bpf_dynptr_type type) 642 { 643 return type == BPF_DYNPTR_TYPE_RINGBUF || type == BPF_DYNPTR_TYPE_FILE; 644 } 645 646 static void __mark_dynptr_reg(struct bpf_reg_state *reg, 647 enum bpf_dynptr_type type, 648 bool first_slot, int id, int parent_id); 649 650 static void mark_dynptr_stack_regs(struct bpf_verifier_env *env, 651 struct bpf_reg_state *sreg1, 652 struct bpf_reg_state *sreg2, 653 enum bpf_dynptr_type type, int parent_id) 654 { 655 int id = ++env->id_gen; 656 657 __mark_dynptr_reg(sreg1, type, true, id, parent_id); 658 __mark_dynptr_reg(sreg2, type, false, id, parent_id); 659 } 660 661 static void mark_dynptr_cb_reg(struct bpf_verifier_env *env, 662 struct bpf_reg_state *reg, 663 enum bpf_dynptr_type type) 664 { 665 __mark_dynptr_reg(reg, type, true, ++env->id_gen, 0); 666 } 667 668 static int destroy_if_dynptr_stack_slot(struct bpf_verifier_env *env, 669 struct bpf_func_state *state, int spi); 670 671 static int mark_stack_slots_dynptr(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 672 enum bpf_arg_type arg_type, int insn_idx, 673 struct ref_obj_desc *ref_obj, struct bpf_dynptr_desc *dynptr) 674 { 675 struct bpf_func_state *state = bpf_func(env, reg); 676 int spi, i, err, parent_id = 0; 677 enum bpf_dynptr_type type; 678 679 spi = dynptr_get_spi(env, reg); 680 if (spi < 0) 681 return spi; 682 683 /* We cannot assume both spi and spi - 1 belong to the same dynptr, 684 * hence we need to call destroy_if_dynptr_stack_slot twice for both, 685 * to ensure that for the following example: 686 * [d1][d1][d2][d2] 687 * spi 3 2 1 0 688 * So marking spi = 2 should lead to destruction of both d1 and d2. In 689 * case they do belong to same dynptr, second call won't see slot_type 690 * as STACK_DYNPTR and will simply skip destruction. 691 */ 692 err = destroy_if_dynptr_stack_slot(env, state, spi); 693 if (err) 694 return err; 695 err = destroy_if_dynptr_stack_slot(env, state, spi - 1); 696 if (err) 697 return err; 698 699 for (i = 0; i < BPF_REG_SIZE; i++) { 700 state->stack[spi].slot_type[i] = STACK_DYNPTR; 701 state->stack[spi - 1].slot_type[i] = STACK_DYNPTR; 702 } 703 704 type = arg_to_dynptr_type(arg_type); 705 if (type == BPF_DYNPTR_TYPE_INVALID) 706 return -EINVAL; 707 708 if (dynptr->type == BPF_DYNPTR_TYPE_INVALID) { /* dynptr constructors */ 709 err = validate_ref_obj(env, ref_obj); 710 if (err) 711 return err; 712 713 /* Track parent's id if the parent is a referenced object */ 714 parent_id = ref_obj->id; 715 716 if (dynptr_type_referenced(type)) { 717 int id; 718 719 /* 720 * Create an intermediate reference that tracks the referenced 721 * object for the referenced dynptr. Freeing a referenced dynptr 722 * through helpers/kfuncs will invalidate all clones. 723 */ 724 id = acquire_reference(env, insn_idx, parent_id); 725 if (id < 0) 726 return id; 727 728 parent_id = id; 729 } 730 } else { /* bpf_dynptr_clone() */ 731 parent_id = dynptr->parent_id; 732 } 733 734 mark_dynptr_stack_regs(env, &state->stack[spi].spilled_ptr, 735 &state->stack[spi - 1].spilled_ptr, type, parent_id); 736 737 return 0; 738 } 739 740 static void invalidate_dynptr(struct bpf_verifier_env *env, struct bpf_stack_state *stack) 741 { 742 int i; 743 744 for (i = 0; i < BPF_REG_SIZE; i++) { 745 stack[0].slot_type[i] = STACK_INVALID; 746 stack[1].slot_type[i] = STACK_INVALID; 747 } 748 749 bpf_mark_reg_not_init(env, &stack[0].spilled_ptr); 750 bpf_mark_reg_not_init(env, &stack[1].spilled_ptr); 751 } 752 753 static int unmark_stack_slots_dynptr(struct bpf_verifier_env *env, struct bpf_reg_state *reg) 754 { 755 struct bpf_func_state *state = bpf_func(env, reg); 756 int spi; 757 758 spi = dynptr_get_spi(env, reg); 759 if (spi < 0) 760 return spi; 761 762 /* 763 * For referenced dynptr, release the parent ref which cascades to 764 * all clones and derived slices. For non-referenced dynptr, only 765 * the dynptr and slices derived from it will be invalidated. 766 */ 767 reg = &state->stack[spi].spilled_ptr; 768 return release_reference(env, dynptr_type_referenced(reg->dynptr.type) 769 ? reg->parent_id 770 : reg->id); 771 } 772 773 static void __mark_reg_unknown(const struct bpf_verifier_env *env, 774 struct bpf_reg_state *reg); 775 776 static void mark_reg_invalid(const struct bpf_verifier_env *env, struct bpf_reg_state *reg) 777 { 778 if (!env->allow_ptr_leaks) 779 bpf_mark_reg_not_init(env, reg); 780 else 781 __mark_reg_unknown(env, reg); 782 } 783 784 static int dynptr_ref_cnt(struct bpf_verifier_env *env, int v_parent_id) 785 { 786 struct bpf_stack_state *stack; 787 struct bpf_func_state *state; 788 struct bpf_reg_state *reg; 789 int ref_cnt = 0; 790 791 bpf_for_each_reg_in_vstate_mask(env->cur_state, state, reg, stack, 1 << STACK_DYNPTR, ({ 792 if (!stack || stack->slot_type[0] != STACK_DYNPTR) 793 continue; 794 if (!stack->spilled_ptr.dynptr.first_slot) 795 continue; 796 if (stack->spilled_ptr.parent_id == v_parent_id) 797 ref_cnt++; 798 })); 799 800 return ref_cnt; 801 } 802 803 static int destroy_if_dynptr_stack_slot(struct bpf_verifier_env *env, 804 struct bpf_func_state *state, int spi) 805 { 806 int err = 0; 807 808 /* We always ensure that STACK_DYNPTR is never set partially, 809 * hence just checking for slot_type[0] is enough. This is 810 * different for STACK_SPILL, where it may be only set for 811 * 1 byte, so code has to use is_spilled_reg. 812 */ 813 if (state->stack[spi].slot_type[0] != STACK_DYNPTR) 814 return 0; 815 816 /* Reposition spi to first slot */ 817 if (!state->stack[spi].spilled_ptr.dynptr.first_slot) 818 spi = spi + 1; 819 820 /* 821 * A referenced dynptr can be overwritten only if there is at 822 * least one other dynptr sharing the same virtual ref parent, 823 * ensuring the reference can still be properly released. 824 */ 825 if (dynptr_type_referenced(state->stack[spi].spilled_ptr.dynptr.type) && 826 dynptr_ref_cnt(env, state->stack[spi].spilled_ptr.parent_id) <= 1) { 827 verbose(env, "cannot overwrite referenced dynptr\n"); 828 bpf_diag_res( 829 env, env->insn_idx, "referenced dynptr overwrite", 830 "This stack slot contains a dynptr that owns or protects a referenced resource. Overwriting the last dynptr for that resource would lose the verifier-tracked release path.", 831 "Release or clone the dynptr so another live dynptr still tracks the referenced resource before overwriting this stack slot."); 832 return -EINVAL; 833 } 834 835 /* Invalidate the dynptr and any derived slices */ 836 err = release_reference(env, state->stack[spi].spilled_ptr.id); 837 if (!err) { 838 mark_stack_slot_scratched(env, spi); 839 mark_stack_slot_scratched(env, spi - 1); 840 } 841 842 return err; 843 } 844 845 static bool is_dynptr_reg_valid_uninit(struct bpf_verifier_env *env, struct bpf_reg_state *reg) 846 { 847 int spi; 848 849 if (reg->type == CONST_PTR_TO_DYNPTR) 850 return false; 851 852 spi = dynptr_get_spi(env, reg); 853 854 /* -ERANGE (i.e. spi not falling into allocated stack slots) isn't an 855 * error because this just means the stack state hasn't been updated yet. 856 * We will do check_mem_access to check and update stack bounds later. 857 */ 858 if (spi < 0 && spi != -ERANGE) 859 return false; 860 861 /* We don't need to check if the stack slots are marked by previous 862 * dynptr initializations because we allow overwriting existing unreferenced 863 * STACK_DYNPTR slots, see mark_stack_slots_dynptr which calls 864 * destroy_if_dynptr_stack_slot to ensure dynptr objects at the slots we are 865 * touching are completely destructed before we reinitialize them for a new 866 * one. For referenced ones, destroy_if_dynptr_stack_slot returns an error early 867 * instead of delaying it until the end where the user will get "Unreleased 868 * reference" error. 869 */ 870 return true; 871 } 872 873 static bool is_dynptr_reg_valid_init(struct bpf_verifier_env *env, struct bpf_reg_state *reg) 874 { 875 struct bpf_func_state *state = bpf_func(env, reg); 876 int i, spi; 877 878 /* This already represents first slot of initialized bpf_dynptr. 879 * 880 * CONST_PTR_TO_DYNPTR already has fixed and var_off as 0 due to 881 * check_func_arg_reg_off's logic, so we don't need to check its 882 * offset and alignment. 883 */ 884 if (reg->type == CONST_PTR_TO_DYNPTR) 885 return true; 886 887 spi = dynptr_get_spi(env, reg); 888 if (spi < 0) 889 return false; 890 if (!state->stack[spi].spilled_ptr.dynptr.first_slot) 891 return false; 892 893 for (i = 0; i < BPF_REG_SIZE; i++) { 894 if (state->stack[spi].slot_type[i] != STACK_DYNPTR || 895 state->stack[spi - 1].slot_type[i] != STACK_DYNPTR) 896 return false; 897 } 898 899 return true; 900 } 901 902 static enum bpf_dynptr_type dynptr_reg_type(struct bpf_verifier_env *env, struct bpf_reg_state *reg) 903 { 904 struct bpf_func_state *state; 905 int spi; 906 907 if (reg->type == CONST_PTR_TO_DYNPTR) 908 return reg->dynptr.type; 909 910 spi = dynptr_get_spi(env, reg); 911 if (spi < 0) 912 return BPF_DYNPTR_TYPE_INVALID; 913 state = bpf_func(env, reg); 914 return state->stack[spi].spilled_ptr.dynptr.type; 915 } 916 917 static bool is_dynptr_type_expected(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 918 enum bpf_arg_type arg_type) 919 { 920 /* ARG_PTR_TO_DYNPTR takes any type of dynptr */ 921 if (arg_type == ARG_PTR_TO_DYNPTR) 922 return true; 923 924 return dynptr_reg_type(env, reg) == arg_to_dynptr_type(arg_type); 925 } 926 927 static void __mark_reg_known_zero(struct bpf_reg_state *reg); 928 929 static bool in_rcu_cs(struct bpf_verifier_env *env); 930 931 static bool is_kfunc_rcu_protected(struct bpf_call_arg_meta *meta); 932 933 static int mark_stack_slots_iter(struct bpf_verifier_env *env, 934 struct bpf_call_arg_meta *meta, 935 struct bpf_reg_state *reg, int insn_idx, 936 struct btf *btf, u32 btf_id, int nr_slots) 937 { 938 struct bpf_func_state *state = bpf_func(env, reg); 939 int spi, i, j, id; 940 941 spi = iter_get_spi(env, reg, nr_slots); 942 if (spi < 0) 943 return spi; 944 945 id = acquire_reference(env, insn_idx, 0); 946 if (id < 0) 947 return id; 948 949 for (i = 0; i < nr_slots; i++) { 950 struct bpf_stack_state *slot = &state->stack[spi - i]; 951 struct bpf_reg_state *st = &slot->spilled_ptr; 952 953 __mark_reg_known_zero(st); 954 st->type = PTR_TO_STACK; /* we don't have dedicated reg type */ 955 if (is_kfunc_rcu_protected(meta)) { 956 if (in_rcu_cs(env)) 957 st->type |= MEM_RCU; 958 else 959 st->type |= PTR_UNTRUSTED; 960 } 961 st->id = i == 0 ? id : 0; 962 st->iter.btf = btf; 963 st->iter.btf_id = btf_id; 964 st->iter.state = BPF_ITER_STATE_ACTIVE; 965 st->iter.depth = 0; 966 967 for (j = 0; j < BPF_REG_SIZE; j++) 968 slot->slot_type[j] = STACK_ITER; 969 970 mark_stack_slot_scratched(env, spi - i); 971 } 972 973 return 0; 974 } 975 976 static int unmark_stack_slots_iter(struct bpf_verifier_env *env, 977 struct bpf_reg_state *reg, int nr_slots) 978 { 979 struct bpf_func_state *state = bpf_func(env, reg); 980 int spi, i, j; 981 982 spi = iter_get_spi(env, reg, nr_slots); 983 if (spi < 0) 984 return spi; 985 986 for (i = 0; i < nr_slots; i++) { 987 struct bpf_stack_state *slot = &state->stack[spi - i]; 988 struct bpf_reg_state *st = &slot->spilled_ptr; 989 990 if (i == 0) 991 WARN_ON_ONCE(release_reference(env, st->id)); 992 993 bpf_mark_reg_not_init(env, st); 994 995 for (j = 0; j < BPF_REG_SIZE; j++) 996 slot->slot_type[j] = STACK_INVALID; 997 998 mark_stack_slot_scratched(env, spi - i); 999 } 1000 1001 return 0; 1002 } 1003 1004 static bool is_iter_reg_valid_uninit(struct bpf_verifier_env *env, 1005 struct bpf_reg_state *reg, int nr_slots) 1006 { 1007 struct bpf_func_state *state = bpf_func(env, reg); 1008 int spi, i, j; 1009 1010 /* For -ERANGE (i.e. spi not falling into allocated stack slots), we 1011 * will do check_mem_access to check and update stack bounds later, so 1012 * return true for that case. 1013 */ 1014 spi = iter_get_spi(env, reg, nr_slots); 1015 if (spi == -ERANGE) 1016 return true; 1017 if (spi < 0) 1018 return false; 1019 1020 for (i = 0; i < nr_slots; i++) { 1021 struct bpf_stack_state *slot = &state->stack[spi - i]; 1022 1023 for (j = 0; j < BPF_REG_SIZE; j++) 1024 if (slot->slot_type[j] == STACK_ITER) 1025 return false; 1026 } 1027 1028 return true; 1029 } 1030 1031 static int is_iter_reg_valid_init(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 1032 struct btf *btf, u32 btf_id, int nr_slots) 1033 { 1034 struct bpf_func_state *state = bpf_func(env, reg); 1035 int spi, i, j; 1036 1037 spi = iter_get_spi(env, reg, nr_slots); 1038 if (spi < 0) 1039 return -EINVAL; 1040 1041 for (i = 0; i < nr_slots; i++) { 1042 struct bpf_stack_state *slot = &state->stack[spi - i]; 1043 struct bpf_reg_state *st = &slot->spilled_ptr; 1044 1045 if (st->type & PTR_UNTRUSTED) 1046 return -EPROTO; 1047 /* only main (first) slot has id set */ 1048 if (i == 0 && !st->id) 1049 return -EINVAL; 1050 if (i != 0 && st->id) 1051 return -EINVAL; 1052 if (st->iter.btf != btf || st->iter.btf_id != btf_id) 1053 return -EINVAL; 1054 1055 for (j = 0; j < BPF_REG_SIZE; j++) 1056 if (slot->slot_type[j] != STACK_ITER) 1057 return -EINVAL; 1058 } 1059 1060 return 0; 1061 } 1062 1063 static int acquire_irq_state(struct bpf_verifier_env *env, int insn_idx); 1064 static int release_irq_state(struct bpf_verifier_env *env, int id); 1065 1066 static int mark_stack_slot_irq_flag(struct bpf_verifier_env *env, 1067 struct bpf_call_arg_meta *meta, 1068 struct bpf_reg_state *reg, int insn_idx, 1069 int kfunc_class) 1070 { 1071 struct bpf_func_state *state = bpf_func(env, reg); 1072 struct bpf_stack_state *slot; 1073 struct bpf_reg_state *st; 1074 int spi, i, id; 1075 1076 spi = irq_flag_get_spi(env, reg); 1077 if (spi < 0) 1078 return spi; 1079 1080 id = acquire_irq_state(env, insn_idx); 1081 if (id < 0) 1082 return id; 1083 1084 slot = &state->stack[spi]; 1085 st = &slot->spilled_ptr; 1086 1087 __mark_reg_known_zero(st); 1088 st->type = PTR_TO_STACK; /* we don't have dedicated reg type */ 1089 st->id = id; 1090 st->irq.kfunc_class = kfunc_class; 1091 1092 for (i = 0; i < BPF_REG_SIZE; i++) 1093 slot->slot_type[i] = STACK_IRQ_FLAG; 1094 1095 mark_stack_slot_scratched(env, spi); 1096 return 0; 1097 } 1098 1099 static int unmark_stack_slot_irq_flag(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 1100 int kfunc_class) 1101 { 1102 struct bpf_func_state *state = bpf_func(env, reg); 1103 struct bpf_stack_state *slot; 1104 struct bpf_reg_state *st; 1105 int spi, i, err; 1106 1107 spi = irq_flag_get_spi(env, reg); 1108 if (spi < 0) 1109 return spi; 1110 1111 slot = &state->stack[spi]; 1112 st = &slot->spilled_ptr; 1113 1114 if (st->irq.kfunc_class != kfunc_class) { 1115 const char *flag_kfunc = st->irq.kfunc_class == IRQ_NATIVE_KFUNC ? "native" : "lock"; 1116 const char *used_kfunc = kfunc_class == IRQ_NATIVE_KFUNC ? "native" : "lock"; 1117 const char *reason; 1118 1119 verbose(env, "irq flag acquired by %s kfuncs cannot be restored with %s kfuncs\n", 1120 flag_kfunc, used_kfunc); 1121 reason = bpf_diag_fmt(env, 1122 "This IRQ flag was saved by %s IRQ kfuncs, but the restore call " 1123 "belongs to the %s IRQ kfunc family. Save and restore operations " 1124 "must use the same family.", 1125 flag_kfunc, used_kfunc); 1126 bpf_diag_irq(env, env->insn_idx, "IRQ flag restore mismatch", reason, 1127 "Restore the flag with the matching IRQ restore kfunc for the save " 1128 "operation that created it.", 1129 bpf_diag_irq_depth(env->cur_state)); 1130 return -EINVAL; 1131 } 1132 1133 err = release_irq_state(env, st->id); 1134 WARN_ON_ONCE(err && err != -EACCES); 1135 if (err) { 1136 int insn_idx = 0; 1137 1138 for (int i = 0; i < env->cur_state->acquired_refs; i++) { 1139 if (env->cur_state->refs[i].id == env->cur_state->active_irq_id) { 1140 insn_idx = env->cur_state->refs[i].insn_idx; 1141 break; 1142 } 1143 } 1144 1145 verbose(env, "cannot restore irq state out of order, expected id=%d acquired at insn_idx=%d\n", 1146 env->cur_state->active_irq_id, insn_idx); 1147 bpf_diag_irq(env, env->insn_idx, "IRQ flag restore out of order", 1148 "IRQ-disabled regions must be restored in last-in, first-out order, " 1149 "but this restore does not match the currently active IRQ flag.", 1150 "Restore nested IRQ flags in the reverse order they were saved.", 1151 bpf_diag_irq_depth(env->cur_state)); 1152 return err; 1153 } 1154 1155 bpf_mark_reg_not_init(env, st); 1156 1157 for (i = 0; i < BPF_REG_SIZE; i++) 1158 slot->slot_type[i] = STACK_INVALID; 1159 1160 mark_stack_slot_scratched(env, spi); 1161 return 0; 1162 } 1163 1164 static bool is_irq_flag_reg_valid_uninit(struct bpf_verifier_env *env, struct bpf_reg_state *reg) 1165 { 1166 struct bpf_func_state *state = bpf_func(env, reg); 1167 struct bpf_stack_state *slot; 1168 int spi, i; 1169 1170 /* For -ERANGE (i.e. spi not falling into allocated stack slots), we 1171 * will do check_mem_access to check and update stack bounds later, so 1172 * return true for that case. 1173 */ 1174 spi = irq_flag_get_spi(env, reg); 1175 if (spi == -ERANGE) 1176 return true; 1177 if (spi < 0) 1178 return false; 1179 1180 slot = &state->stack[spi]; 1181 1182 for (i = 0; i < BPF_REG_SIZE; i++) 1183 if (slot->slot_type[i] == STACK_IRQ_FLAG) 1184 return false; 1185 return true; 1186 } 1187 1188 static int is_irq_flag_reg_valid_init(struct bpf_verifier_env *env, struct bpf_reg_state *reg) 1189 { 1190 struct bpf_func_state *state = bpf_func(env, reg); 1191 struct bpf_stack_state *slot; 1192 struct bpf_reg_state *st; 1193 int spi, i; 1194 1195 spi = irq_flag_get_spi(env, reg); 1196 if (spi < 0) 1197 return -EINVAL; 1198 1199 slot = &state->stack[spi]; 1200 st = &slot->spilled_ptr; 1201 1202 if (!st->id) 1203 return -EINVAL; 1204 1205 for (i = 0; i < BPF_REG_SIZE; i++) 1206 if (slot->slot_type[i] != STACK_IRQ_FLAG) 1207 return -EINVAL; 1208 return 0; 1209 } 1210 1211 /* Check if given stack slot is "special": 1212 * - spilled register state (STACK_SPILL); 1213 * - dynptr state (STACK_DYNPTR); 1214 * - iter state (STACK_ITER). 1215 * - irq flag state (STACK_IRQ_FLAG) 1216 */ 1217 static bool is_stack_slot_special(const struct bpf_stack_state *stack) 1218 { 1219 enum bpf_stack_slot_type type = stack->slot_type[BPF_REG_SIZE - 1]; 1220 1221 switch (type) { 1222 case STACK_SPILL: 1223 case STACK_DYNPTR: 1224 case STACK_ITER: 1225 case STACK_IRQ_FLAG: 1226 return true; 1227 case STACK_INVALID: 1228 case STACK_POISON: 1229 case STACK_MISC: 1230 case STACK_ZERO: 1231 return false; 1232 default: 1233 WARN_ONCE(1, "unknown stack slot type %d\n", type); 1234 return true; 1235 } 1236 } 1237 1238 /* The reg state of a pointer or a bounded scalar was saved when 1239 * it was spilled to the stack. 1240 */ 1241 1242 /* 1243 * Mark stack slot as STACK_MISC, unless it is already: 1244 * - STACK_INVALID, in which case they are equivalent. 1245 * - STACK_ZERO, in which case we preserve more precise STACK_ZERO. 1246 * - STACK_POISON, which truly forbids access to the slot. 1247 * Regardless of allow_ptr_leaks setting (i.e., privileged or unprivileged 1248 * mode), we won't promote STACK_INVALID to STACK_MISC. In privileged case it is 1249 * unnecessary as both are considered equivalent when loading data and pruning, 1250 * in case of unprivileged mode it will be incorrect to allow reads of invalid 1251 * slots. 1252 */ 1253 static void mark_stack_slot_misc(struct bpf_verifier_env *env, u8 *stype) 1254 { 1255 if (*stype == STACK_ZERO) 1256 return; 1257 if (*stype == STACK_INVALID || *stype == STACK_POISON) 1258 return; 1259 *stype = STACK_MISC; 1260 } 1261 1262 static void scrub_spilled_slot(u8 *stype) 1263 { 1264 if (*stype != STACK_INVALID && *stype != STACK_POISON) 1265 *stype = STACK_MISC; 1266 } 1267 1268 /* copy array src of length n * size bytes to dst. dst is reallocated if it's too 1269 * small to hold src. This is different from krealloc since we don't want to preserve 1270 * the contents of dst. 1271 * 1272 * Leaves dst untouched if src is NULL or length is zero. Returns NULL if memory could 1273 * not be allocated. 1274 */ 1275 static void *copy_array(void *dst, const void *src, size_t n, size_t size, gfp_t flags) 1276 { 1277 size_t alloc_bytes; 1278 void *orig = dst; 1279 size_t bytes; 1280 1281 if (ZERO_OR_NULL_PTR(src)) 1282 goto out; 1283 1284 if (unlikely(check_mul_overflow(n, size, &bytes))) 1285 return NULL; 1286 1287 alloc_bytes = max(ksize(orig), kmalloc_size_roundup(bytes)); 1288 dst = krealloc(orig, alloc_bytes, flags); 1289 if (!dst) { 1290 kfree(orig); 1291 return NULL; 1292 } 1293 1294 memcpy(dst, src, bytes); 1295 out: 1296 return dst ? dst : ZERO_SIZE_PTR; 1297 } 1298 1299 /* resize an array from old_n items to new_n items. the array is reallocated if it's too 1300 * small to hold new_n items. new items are zeroed out if the array grows. 1301 * 1302 * Contrary to krealloc_array, does not free arr if new_n is zero. 1303 */ 1304 static void *realloc_array(void *arr, size_t old_n, size_t new_n, size_t size) 1305 { 1306 size_t alloc_size; 1307 void *new_arr; 1308 1309 if (!new_n || old_n == new_n) 1310 goto out; 1311 1312 alloc_size = kmalloc_size_roundup(size_mul(new_n, size)); 1313 new_arr = krealloc(arr, alloc_size, GFP_KERNEL_ACCOUNT); 1314 if (!new_arr) { 1315 kfree(arr); 1316 return NULL; 1317 } 1318 arr = new_arr; 1319 1320 if (new_n > old_n) 1321 memset(arr + old_n * size, 0, (new_n - old_n) * size); 1322 1323 out: 1324 return arr ? arr : ZERO_SIZE_PTR; 1325 } 1326 1327 static int copy_reference_state(struct bpf_verifier_state *dst, const struct bpf_verifier_state *src) 1328 { 1329 dst->refs = copy_array(dst->refs, src->refs, src->acquired_refs, 1330 sizeof(struct bpf_reference_state), GFP_KERNEL_ACCOUNT); 1331 if (!dst->refs) 1332 return -ENOMEM; 1333 1334 dst->acquired_refs = src->acquired_refs; 1335 dst->active_locks = src->active_locks; 1336 dst->active_preempt_locks = src->active_preempt_locks; 1337 dst->active_rcu_locks = src->active_rcu_locks; 1338 dst->active_irq_id = src->active_irq_id; 1339 dst->active_lock_id = src->active_lock_id; 1340 dst->active_lock_ptr = src->active_lock_ptr; 1341 return 0; 1342 } 1343 1344 static int copy_stack_state(struct bpf_func_state *dst, const struct bpf_func_state *src) 1345 { 1346 size_t n = src->allocated_stack / BPF_REG_SIZE; 1347 1348 dst->stack = copy_array(dst->stack, src->stack, n, sizeof(struct bpf_stack_state), 1349 GFP_KERNEL_ACCOUNT); 1350 if (!dst->stack) 1351 return -ENOMEM; 1352 1353 dst->allocated_stack = src->allocated_stack; 1354 1355 /* copy stack args state */ 1356 n = src->out_stack_arg_cnt; 1357 if (n) { 1358 dst->stack_arg_regs = copy_array(dst->stack_arg_regs, src->stack_arg_regs, n, 1359 sizeof(struct bpf_reg_state), 1360 GFP_KERNEL_ACCOUNT); 1361 if (!dst->stack_arg_regs) 1362 return -ENOMEM; 1363 } 1364 1365 dst->out_stack_arg_cnt = src->out_stack_arg_cnt; 1366 return 0; 1367 } 1368 1369 static int resize_reference_state(struct bpf_verifier_state *state, size_t n) 1370 { 1371 state->refs = realloc_array(state->refs, state->acquired_refs, n, 1372 sizeof(struct bpf_reference_state)); 1373 if (!state->refs) 1374 return -ENOMEM; 1375 1376 state->acquired_refs = n; 1377 return 0; 1378 } 1379 1380 /* Possibly update state->allocated_stack to be at least size bytes. Also 1381 * possibly update the function's high-water mark in its bpf_subprog_info. 1382 */ 1383 static int grow_stack_state(struct bpf_verifier_env *env, struct bpf_func_state *state, int size) 1384 { 1385 size_t old_n = state->allocated_stack / BPF_REG_SIZE, n; 1386 1387 /* The stack size is always a multiple of BPF_REG_SIZE. */ 1388 size = round_up(size, BPF_REG_SIZE); 1389 n = size / BPF_REG_SIZE; 1390 1391 if (old_n >= n) 1392 return 0; 1393 1394 state->stack = realloc_array(state->stack, old_n, n, sizeof(struct bpf_stack_state)); 1395 if (!state->stack) 1396 return -ENOMEM; 1397 1398 state->allocated_stack = size; 1399 1400 /* update known max for given subprogram */ 1401 if (env->subprog_info[state->subprogno].stack_depth < size) 1402 env->subprog_info[state->subprogno].stack_depth = size; 1403 1404 return 0; 1405 } 1406 1407 static int grow_stack_arg_slots(struct bpf_verifier_env *env, 1408 struct bpf_func_state *state, int cnt) 1409 { 1410 size_t old_n = state->out_stack_arg_cnt; 1411 1412 if (old_n >= cnt) 1413 return 0; 1414 1415 state->stack_arg_regs = realloc_array(state->stack_arg_regs, old_n, cnt, 1416 sizeof(struct bpf_reg_state)); 1417 if (!state->stack_arg_regs) 1418 return -ENOMEM; 1419 1420 state->out_stack_arg_cnt = cnt; 1421 return 0; 1422 } 1423 1424 /* Acquire a pointer id from the env and update the state->refs to include 1425 * this new pointer reference. 1426 * On success, returns a valid pointer id to associate with the register 1427 * On failure, returns a negative errno. 1428 */ 1429 static struct bpf_reference_state *acquire_reference_state(struct bpf_verifier_env *env, int insn_idx) 1430 { 1431 struct bpf_verifier_state *state = env->cur_state; 1432 int new_ofs = state->acquired_refs; 1433 int err; 1434 1435 err = resize_reference_state(state, state->acquired_refs + 1); 1436 if (err) 1437 return NULL; 1438 state->refs[new_ofs].insn_idx = insn_idx; 1439 1440 return &state->refs[new_ofs]; 1441 } 1442 1443 static int acquire_reference(struct bpf_verifier_env *env, int insn_idx, int parent_id) 1444 { 1445 struct bpf_reference_state *s; 1446 1447 s = acquire_reference_state(env, insn_idx); 1448 if (!s) 1449 return -ENOMEM; 1450 s->type = REF_TYPE_PTR; 1451 s->id = ++env->id_gen; 1452 s->parent_id = parent_id; 1453 bpf_diag_record_ref_acquire(env, insn_idx, s->id); 1454 return s->id; 1455 } 1456 1457 static int acquire_lock_state(struct bpf_verifier_env *env, int insn_idx, enum ref_state_type type, 1458 int id, void *ptr) 1459 { 1460 struct bpf_verifier_state *state = env->cur_state; 1461 struct bpf_reference_state *s; 1462 1463 s = acquire_reference_state(env, insn_idx); 1464 if (!s) 1465 return -ENOMEM; 1466 s->type = type; 1467 s->id = id; 1468 s->ptr = ptr; 1469 1470 state->active_locks++; 1471 state->active_lock_id = id; 1472 state->active_lock_ptr = ptr; 1473 bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_LOCK, true, 1474 state->active_locks); 1475 return 0; 1476 } 1477 1478 static int acquire_irq_state(struct bpf_verifier_env *env, int insn_idx) 1479 { 1480 struct bpf_verifier_state *state = env->cur_state; 1481 struct bpf_reference_state *s; 1482 1483 s = acquire_reference_state(env, insn_idx); 1484 if (!s) 1485 return -ENOMEM; 1486 s->type = REF_TYPE_IRQ; 1487 s->id = ++env->id_gen; 1488 1489 state->active_irq_id = s->id; 1490 bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_IRQ, true, 1491 bpf_diag_irq_depth(state)); 1492 return s->id; 1493 } 1494 1495 static void release_reference_state(struct bpf_verifier_state *state, int idx) 1496 { 1497 int last_idx; 1498 size_t rem; 1499 1500 /* IRQ state requires the relative ordering of elements remaining the 1501 * same, since it relies on the refs array to behave as a stack, so that 1502 * it can detect out-of-order IRQ restore. Hence use memmove to shift 1503 * the array instead of swapping the final element into the deleted idx. 1504 */ 1505 last_idx = state->acquired_refs - 1; 1506 rem = state->acquired_refs - idx - 1; 1507 if (last_idx && idx != last_idx) 1508 memmove(&state->refs[idx], &state->refs[idx + 1], sizeof(*state->refs) * rem); 1509 memset(&state->refs[last_idx], 0, sizeof(*state->refs)); 1510 state->acquired_refs--; 1511 return; 1512 } 1513 1514 static bool find_reference_state(struct bpf_verifier_state *state, int id) 1515 { 1516 int i; 1517 1518 for (i = 0; i < state->acquired_refs; i++) { 1519 if (state->refs[i].type != REF_TYPE_PTR) 1520 continue; 1521 if (state->refs[i].id == id) 1522 return true; 1523 } 1524 1525 return false; 1526 } 1527 1528 static bool reg_is_referenced(struct bpf_verifier_env *env, const struct bpf_reg_state *reg) 1529 { 1530 return find_reference_state(env->cur_state, reg->id); 1531 } 1532 1533 static int release_lock_state(struct bpf_verifier_env *env, int type, int id, void *ptr) 1534 { 1535 struct bpf_verifier_state *state = env->cur_state; 1536 void *prev_ptr = NULL; 1537 u32 prev_id = 0; 1538 int i; 1539 1540 for (i = 0; i < state->acquired_refs; i++) { 1541 if (state->refs[i].type == type && state->refs[i].id == id && 1542 state->refs[i].ptr == ptr) { 1543 release_reference_state(state, i); 1544 state->active_locks--; 1545 /* Reassign active lock (id, ptr). */ 1546 state->active_lock_id = prev_id; 1547 state->active_lock_ptr = prev_ptr; 1548 bpf_diag_record_context(env, env->insn_idx, BPF_DIAG_CONTEXT_LOCK, 1549 false, state->active_locks); 1550 return 0; 1551 } 1552 if (state->refs[i].type & REF_TYPE_LOCK_MASK) { 1553 prev_id = state->refs[i].id; 1554 prev_ptr = state->refs[i].ptr; 1555 } 1556 } 1557 return -EINVAL; 1558 } 1559 1560 static int release_irq_state(struct bpf_verifier_env *env, int id) 1561 { 1562 struct bpf_verifier_state *state = env->cur_state; 1563 u32 prev_id = 0; 1564 int i; 1565 1566 if (id != state->active_irq_id) 1567 return -EACCES; 1568 1569 for (i = 0; i < state->acquired_refs; i++) { 1570 if (state->refs[i].type != REF_TYPE_IRQ) 1571 continue; 1572 if (state->refs[i].id == id) { 1573 release_reference_state(state, i); 1574 state->active_irq_id = prev_id; 1575 bpf_diag_record_context(env, env->insn_idx, BPF_DIAG_CONTEXT_IRQ, 1576 false, bpf_diag_irq_depth(state)); 1577 return 0; 1578 } else { 1579 prev_id = state->refs[i].id; 1580 } 1581 } 1582 return -EINVAL; 1583 } 1584 1585 static struct bpf_reference_state *find_lock_state(struct bpf_verifier_state *state, enum ref_state_type type, 1586 int id, void *ptr) 1587 { 1588 int i; 1589 1590 for (i = 0; i < state->acquired_refs; i++) { 1591 struct bpf_reference_state *s = &state->refs[i]; 1592 1593 if (!(s->type & type)) 1594 continue; 1595 1596 if (s->id == id && s->ptr == ptr) 1597 return s; 1598 } 1599 return NULL; 1600 } 1601 1602 static void free_func_state(struct bpf_func_state *state) 1603 { 1604 if (!state) 1605 return; 1606 kfree(state->stack_arg_regs); 1607 kfree(state->stack); 1608 kfree(state); 1609 } 1610 1611 void bpf_clear_jmp_history(struct bpf_verifier_state *state) 1612 { 1613 kfree(state->jmp_history); 1614 state->jmp_history = NULL; 1615 state->jmp_history_cnt = 0; 1616 } 1617 1618 void bpf_free_verifier_state(struct bpf_verifier_state *state, 1619 bool free_self) 1620 { 1621 int i; 1622 1623 for (i = 0; i <= state->curframe; i++) { 1624 free_func_state(state->frame[i]); 1625 state->frame[i] = NULL; 1626 } 1627 kfree(state->refs); 1628 bpf_clear_jmp_history(state); 1629 if (free_self) 1630 kfree(state); 1631 } 1632 1633 /* copy verifier state from src to dst growing dst stack space 1634 * when necessary to accommodate larger src stack 1635 */ 1636 static int copy_func_state(struct bpf_func_state *dst, 1637 const struct bpf_func_state *src) 1638 { 1639 memcpy(dst, src, offsetof(struct bpf_func_state, stack)); 1640 /* Instruction accounting is path-local, not part of verifier state. */ 1641 dst->insns_subtotal = 0; 1642 return copy_stack_state(dst, src); 1643 } 1644 1645 int bpf_copy_verifier_state(struct bpf_verifier_state *dst_state, 1646 const struct bpf_verifier_state *src) 1647 { 1648 struct bpf_func_state *dst; 1649 int i, err; 1650 1651 dst_state->jmp_history = copy_array(dst_state->jmp_history, src->jmp_history, 1652 src->jmp_history_cnt, sizeof(*dst_state->jmp_history), 1653 GFP_KERNEL_ACCOUNT); 1654 if (!dst_state->jmp_history) 1655 return -ENOMEM; 1656 dst_state->jmp_history_cnt = src->jmp_history_cnt; 1657 1658 /* if dst has more stack frames then src frame, free them, this is also 1659 * necessary in case of exceptional exits using bpf_throw. 1660 */ 1661 for (i = src->curframe + 1; i <= dst_state->curframe; i++) { 1662 free_func_state(dst_state->frame[i]); 1663 dst_state->frame[i] = NULL; 1664 } 1665 err = copy_reference_state(dst_state, src); 1666 if (err) 1667 return err; 1668 dst_state->speculative = src->speculative; 1669 dst_state->in_sleepable = src->in_sleepable; 1670 dst_state->curframe = src->curframe; 1671 dst_state->branches = src->branches; 1672 dst_state->parent = src->parent; 1673 dst_state->first_insn_idx = src->first_insn_idx; 1674 dst_state->last_insn_idx = src->last_insn_idx; 1675 dst_state->dfs_depth = src->dfs_depth; 1676 dst_state->callback_unroll_depth = src->callback_unroll_depth; 1677 dst_state->may_goto_depth = src->may_goto_depth; 1678 dst_state->equal_state = src->equal_state; 1679 for (i = 0; i <= src->curframe; i++) { 1680 dst = dst_state->frame[i]; 1681 if (!dst) { 1682 dst = kzalloc_obj(*dst, GFP_KERNEL_ACCOUNT); 1683 if (!dst) 1684 return -ENOMEM; 1685 dst_state->frame[i] = dst; 1686 } 1687 err = copy_func_state(dst, src->frame[i]); 1688 if (err) 1689 return err; 1690 } 1691 return 0; 1692 } 1693 1694 static u32 state_htab_size(struct bpf_verifier_env *env) 1695 { 1696 return env->prog->len; 1697 } 1698 1699 struct list_head *bpf_explored_state(struct bpf_verifier_env *env, int idx) 1700 { 1701 struct bpf_verifier_state *cur = env->cur_state; 1702 struct bpf_func_state *state = cur->frame[cur->curframe]; 1703 1704 return &env->explored_states[(idx ^ state->callsite) % state_htab_size(env)]; 1705 } 1706 1707 static bool same_callsites(struct bpf_verifier_state *a, struct bpf_verifier_state *b) 1708 { 1709 int fr; 1710 1711 if (a->curframe != b->curframe) 1712 return false; 1713 1714 for (fr = a->curframe; fr >= 0; fr--) 1715 if (a->frame[fr]->callsite != b->frame[fr]->callsite) 1716 return false; 1717 1718 return true; 1719 } 1720 1721 void bpf_free_backedges(struct bpf_scc_visit *visit) 1722 { 1723 struct bpf_scc_backedge *backedge, *next; 1724 1725 for (backedge = visit->backedges; backedge; backedge = next) { 1726 bpf_free_verifier_state(&backedge->state, false); 1727 next = backedge->next; 1728 kfree(backedge); 1729 } 1730 visit->backedges = NULL; 1731 } 1732 1733 static int pop_stack(struct bpf_verifier_env *env, int *prev_insn_idx, 1734 int *insn_idx, bool pop_log) 1735 { 1736 struct bpf_verifier_state *cur = env->cur_state; 1737 struct bpf_verifier_stack_elem *elem, *head = env->head; 1738 int err; 1739 1740 if (env->head == NULL) 1741 return -ENOENT; 1742 1743 if (cur) { 1744 err = bpf_copy_verifier_state(cur, &head->st); 1745 if (err) 1746 return err; 1747 bpf_diag_event_log_restore(env, head->diag_log_pos); 1748 } 1749 if (pop_log) 1750 bpf_vlog_reset(&env->log, head->log_pos); 1751 if (insn_idx) 1752 *insn_idx = head->insn_idx; 1753 if (prev_insn_idx) 1754 *prev_insn_idx = head->prev_insn_idx; 1755 elem = head->next; 1756 bpf_free_verifier_state(&head->st, false); 1757 kfree(head); 1758 env->head = elem; 1759 env->stack_size--; 1760 return 0; 1761 } 1762 1763 static bool error_recoverable_with_nospec(int err) 1764 { 1765 /* Should only return true for non-fatal errors that are allowed to 1766 * occur during speculative verification. For these we can insert a 1767 * nospec and the program might still be accepted. Do not include 1768 * something like ENOMEM because it is likely to re-occur for the next 1769 * architectural path once it has been recovered-from in all speculative 1770 * paths. 1771 */ 1772 return err == -EPERM || err == -EACCES || err == -EINVAL; 1773 } 1774 1775 static struct bpf_verifier_state *push_stack(struct bpf_verifier_env *env, 1776 int insn_idx, int prev_insn_idx, 1777 bool speculative) 1778 { 1779 struct bpf_verifier_state *cur = env->cur_state; 1780 struct bpf_verifier_stack_elem *elem; 1781 int err; 1782 1783 elem = kzalloc_obj(struct bpf_verifier_stack_elem, GFP_KERNEL_ACCOUNT); 1784 if (!elem) 1785 return ERR_PTR(-ENOMEM); 1786 1787 elem->insn_idx = insn_idx; 1788 elem->prev_insn_idx = prev_insn_idx; 1789 elem->next = env->head; 1790 elem->log_pos = env->log.end_pos; 1791 elem->diag_log_pos = bpf_diag_event_log_save(env); 1792 env->head = elem; 1793 env->stack_size++; 1794 err = bpf_copy_verifier_state(&elem->st, cur); 1795 if (err) 1796 return ERR_PTR(-ENOMEM); 1797 elem->st.speculative |= speculative; 1798 if (env->stack_size > BPF_COMPLEXITY_LIMIT_JMP_SEQ) { 1799 verbose(env, "The sequence of %d jumps is too complex.\n", 1800 env->stack_size); 1801 return ERR_PTR(-E2BIG); 1802 } 1803 if (elem->st.parent) { 1804 ++elem->st.parent->branches; 1805 /* WARN_ON(branches > 2) technically makes sense here, 1806 * but 1807 * 1. speculative states will bump 'branches' for non-branch 1808 * instructions 1809 * 2. is_state_visited() heuristics may decide not to create 1810 * a new state for a sequence of branches and all such current 1811 * and cloned states will be pointing to a single parent state 1812 * which might have large 'branches' count. 1813 */ 1814 } 1815 return &elem->st; 1816 } 1817 1818 static const char *reg_arg_name(struct bpf_verifier_env *env, argno_t argno) 1819 { 1820 char *buf = env->tmp_arg_name; 1821 int len = sizeof(env->tmp_arg_name); 1822 int arg, regno = reg_from_argno(argno); 1823 1824 if (regno >= 0) { 1825 snprintf(buf, len, "R%d", regno); 1826 } else { 1827 arg = arg_from_argno(argno); 1828 snprintf(buf, len, "*(R11-%u)", (arg - MAX_BPF_FUNC_REG_ARGS) * BPF_REG_SIZE); 1829 } 1830 1831 return buf; 1832 } 1833 1834 static const int caller_saved[CALLER_SAVED_REGS] = { 1835 BPF_REG_0, BPF_REG_1, BPF_REG_2, BPF_REG_3, BPF_REG_4, BPF_REG_5 1836 }; 1837 1838 static void bpf_diag_record_caller_saved(struct bpf_verifier_env *env, 1839 struct bpf_reg_state *regs) 1840 { 1841 int i; 1842 1843 for (i = 1; i < CALLER_SAVED_REGS; i++) { 1844 bpf_diag_record_scrub(env, ®s[caller_saved[i]], 1845 BPF_DIAG_MOD_CALLER_SAVED); 1846 } 1847 } 1848 1849 /* This helper doesn't clear reg->id */ 1850 static void ___mark_reg_known(struct bpf_reg_state *reg, u64 imm) 1851 { 1852 reg->var_off = tnum_const(imm); 1853 reg->r64 = cnum64_from_urange(imm, imm); 1854 reg->r32 = cnum32_from_urange((u32)imm, (u32)imm); 1855 } 1856 1857 /* Mark the unknown part of a register (variable offset or scalar value) as 1858 * known to have the value @imm. 1859 */ 1860 static void __mark_reg_known(struct bpf_reg_state *reg, u64 imm) 1861 { 1862 /* Clear off and union(map_ptr, range) */ 1863 memset(((u8 *)reg) + sizeof(reg->type), 0, 1864 offsetof(struct bpf_reg_state, var_off) - sizeof(reg->type)); 1865 reg->id = 0; 1866 reg->parent_id = 0; 1867 reg->map_uid = 0; 1868 ___mark_reg_known(reg, imm); 1869 } 1870 1871 static void __mark_reg32_known(struct bpf_reg_state *reg, u64 imm) 1872 { 1873 reg->var_off = tnum_const_subreg(reg->var_off, imm); 1874 reg->r32 = cnum32_from_urange((u32)imm, (u32)imm); 1875 } 1876 1877 /* Mark the 'variable offset' part of a register as zero. This should be 1878 * used only on registers holding a pointer type. 1879 */ 1880 static void __mark_reg_known_zero(struct bpf_reg_state *reg) 1881 { 1882 __mark_reg_known(reg, 0); 1883 } 1884 1885 static void __mark_reg_const_zero(const struct bpf_verifier_env *env, struct bpf_reg_state *reg) 1886 { 1887 __mark_reg_known(reg, 0); 1888 reg->type = SCALAR_VALUE; 1889 /* all scalars are assumed imprecise initially (unless unprivileged, 1890 * in which case everything is forced to be precise) 1891 */ 1892 reg->precise = !env->bpf_capable; 1893 } 1894 1895 static void mark_reg_known_zero(struct bpf_verifier_env *env, 1896 struct bpf_reg_state *regs, u32 regno) 1897 { 1898 __mark_reg_known_zero(regs + regno); 1899 } 1900 1901 static void __mark_dynptr_reg(struct bpf_reg_state *reg, enum bpf_dynptr_type type, 1902 bool first_slot, int id, int parent_id) 1903 { 1904 /* reg->type has no meaning for STACK_DYNPTR, but when we set reg for 1905 * callback arguments, it does need to be CONST_PTR_TO_DYNPTR, so simply 1906 * set it unconditionally as it is ignored for STACK_DYNPTR anyway. 1907 */ 1908 __mark_reg_known_zero(reg); 1909 reg->type = CONST_PTR_TO_DYNPTR; 1910 /* Give each dynptr a unique id to uniquely associate slices to it. */ 1911 reg->id = id; 1912 reg->parent_id = parent_id; 1913 reg->dynptr.type = type; 1914 reg->dynptr.first_slot = first_slot; 1915 } 1916 1917 /* 1918 * Refine the return type of the bpf_map_lookup_elem() for special map types: 1919 * map-in-map, xskmap, sockmap and sockhash. 1920 */ 1921 static void refine_map_lookup_value(struct bpf_reg_state *reg) 1922 { 1923 enum bpf_type_flag maybe_null = reg->type & PTR_MAYBE_NULL; 1924 const struct bpf_map *map = reg->map_ptr; 1925 1926 if (map->inner_map_meta) { 1927 reg->type = CONST_PTR_TO_MAP | maybe_null; 1928 reg->map_ptr = map->inner_map_meta; 1929 /* 1930 * transfer reg's id which is unique for every map_lookup_elem 1931 * as UID of the inner map. 1932 */ 1933 reg->map_uid = reg->id; 1934 } else if (map->map_type == BPF_MAP_TYPE_XSKMAP) { 1935 reg->type = PTR_TO_XDP_SOCK | maybe_null; 1936 reg->map_uid = 0; 1937 } else if (map->map_type == BPF_MAP_TYPE_SOCKMAP || 1938 map->map_type == BPF_MAP_TYPE_SOCKHASH) { 1939 reg->type = PTR_TO_SOCKET | maybe_null; 1940 reg->map_uid = 0; 1941 } 1942 } 1943 1944 static void mark_ptr_not_null_reg(struct bpf_reg_state *reg) 1945 { 1946 reg->type &= ~PTR_MAYBE_NULL; 1947 } 1948 1949 static void mark_reg_graph_node(struct bpf_reg_state *regs, u32 regno, 1950 struct btf_field_graph_root *ds_head) 1951 { 1952 __mark_reg_known(®s[regno], ds_head->node_offset); 1953 regs[regno].type = PTR_TO_BTF_ID | MEM_ALLOC; 1954 regs[regno].btf = ds_head->btf; 1955 regs[regno].btf_id = ds_head->value_btf_id; 1956 } 1957 1958 static bool reg_is_pkt_pointer(const struct bpf_reg_state *reg) 1959 { 1960 return type_is_pkt_pointer(reg->type); 1961 } 1962 1963 static bool reg_is_pkt_pointer_any(const struct bpf_reg_state *reg) 1964 { 1965 return reg_is_pkt_pointer(reg) || 1966 reg->type == PTR_TO_PACKET_END; 1967 } 1968 1969 static bool reg_is_dynptr_slice_pkt(const struct bpf_reg_state *reg) 1970 { 1971 return base_type(reg->type) == PTR_TO_MEM && 1972 (reg->type & 1973 (DYNPTR_TYPE_SKB | DYNPTR_TYPE_XDP | DYNPTR_TYPE_SKB_META)); 1974 } 1975 1976 /* Unmodified PTR_TO_PACKET[_META,_END] register from ctx access. */ 1977 static bool reg_is_init_pkt_pointer(const struct bpf_reg_state *reg, 1978 enum bpf_reg_type which) 1979 { 1980 /* The register can already have a range from prior markings. 1981 * This is fine as long as it hasn't been advanced from its 1982 * origin. 1983 */ 1984 return reg->type == which && 1985 reg->id == 0 && 1986 tnum_equals_const(reg->var_off, 0); 1987 } 1988 1989 static void __mark_reg32_unbounded(struct bpf_reg_state *reg) 1990 { 1991 reg->r32 = CNUM32_UNBOUNDED; 1992 } 1993 1994 static void __mark_reg64_unbounded(struct bpf_reg_state *reg) 1995 { 1996 reg->r64 = CNUM64_UNBOUNDED; 1997 } 1998 1999 /* Reset the min/max bounds of a register */ 2000 static void __mark_reg_unbounded(struct bpf_reg_state *reg) 2001 { 2002 __mark_reg64_unbounded(reg); 2003 __mark_reg32_unbounded(reg); 2004 } 2005 2006 static void reset_reg64_and_tnum(struct bpf_reg_state *reg) 2007 { 2008 __mark_reg64_unbounded(reg); 2009 reg->var_off = tnum_unknown; 2010 } 2011 2012 static void reset_reg32_and_tnum(struct bpf_reg_state *reg) 2013 { 2014 __mark_reg32_unbounded(reg); 2015 reg->var_off = tnum_unknown; 2016 } 2017 2018 static struct cnum32 cnum32_from_tnum(struct tnum tnum) 2019 { 2020 tnum = tnum_subreg(tnum); 2021 if ((tnum.mask & S32_MIN) || (tnum.value & S32_MIN)) 2022 /* min signed is max(sign bit) | min(other bits) */ 2023 /* max signed is min(sign bit) | max(other bits) */ 2024 return cnum32_from_srange(tnum.value | (tnum.mask & S32_MIN), 2025 tnum.value | (tnum.mask & S32_MAX)); 2026 else 2027 return cnum32_from_urange(tnum.value, (tnum.value | tnum.mask)); 2028 } 2029 2030 static struct cnum64 cnum64_from_tnum(struct tnum tnum) 2031 { 2032 if ((tnum.mask & S64_MIN) || (tnum.value & S64_MIN)) 2033 /* min signed is max(sign bit) | min(other bits) */ 2034 /* max signed is min(sign bit) | max(other bits) */ 2035 return cnum64_from_srange(tnum.value | (tnum.mask & S64_MIN), 2036 tnum.value | (tnum.mask & S64_MAX)); 2037 else 2038 return cnum64_from_urange(tnum.value, (tnum.value | tnum.mask)); 2039 } 2040 2041 static void __update_reg32_bounds(struct bpf_reg_state *reg) 2042 { 2043 cnum32_intersect_with(®->r32, cnum32_from_tnum(reg->var_off)); 2044 } 2045 2046 static void __update_reg64_bounds(struct bpf_reg_state *reg) 2047 { 2048 u64 tnum_next, tmax; 2049 bool umin_in_tnum; 2050 2051 cnum64_intersect_with(®->r64, cnum64_from_tnum(reg->var_off)); 2052 2053 /* Check if u64 and tnum overlap in a single value */ 2054 tnum_next = tnum_step(reg->var_off, reg_umin(reg)); 2055 umin_in_tnum = (reg_umin(reg) & ~reg->var_off.mask) == reg->var_off.value; 2056 tmax = reg->var_off.value | reg->var_off.mask; 2057 if (umin_in_tnum && tnum_next > reg_umax(reg)) { 2058 /* The u64 range and the tnum only overlap in umin. 2059 * u64: ---[xxxxxx]----- 2060 * tnum: --xx----------x- 2061 */ 2062 ___mark_reg_known(reg, reg_umin(reg)); 2063 } else if (!umin_in_tnum && tnum_next == tmax) { 2064 /* The u64 range and the tnum only overlap in the maximum value 2065 * represented by the tnum, called tmax. 2066 * u64: ---[xxxxxx]----- 2067 * tnum: xx-----x-------- 2068 */ 2069 ___mark_reg_known(reg, tmax); 2070 } else if (!umin_in_tnum && tnum_next <= reg_umax(reg) && 2071 tnum_step(reg->var_off, tnum_next) > reg_umax(reg)) { 2072 /* The u64 range and the tnum only overlap in between umin 2073 * (excluded) and umax. 2074 * u64: ---[xxxxxx]----- 2075 * tnum: xx----x-------x- 2076 */ 2077 ___mark_reg_known(reg, tnum_next); 2078 } 2079 } 2080 2081 static void __update_reg_bounds(struct bpf_reg_state *reg) 2082 { 2083 __update_reg32_bounds(reg); 2084 __update_reg64_bounds(reg); 2085 } 2086 2087 static void deduce_bounds_32_from_64(struct bpf_reg_state *reg) 2088 { 2089 cnum32_intersect_with(®->r32, cnum32_from_cnum64(reg->r64)); 2090 } 2091 2092 static void deduce_bounds_64_from_32(struct bpf_reg_state *reg) 2093 { 2094 reg->r64 = cnum64_cnum32_intersect(reg->r64, reg->r32); 2095 } 2096 2097 static void __reg_deduce_bounds(struct bpf_reg_state *reg) 2098 { 2099 deduce_bounds_32_from_64(reg); 2100 deduce_bounds_64_from_32(reg); 2101 } 2102 2103 /* Attempts to improve var_off based on unsigned min/max information */ 2104 static void __reg_bound_offset(struct bpf_reg_state *reg) 2105 { 2106 struct tnum var64_off = tnum_intersect(reg->var_off, 2107 tnum_range(reg_umin(reg), 2108 reg_umax(reg))); 2109 struct tnum var32_off = tnum_intersect(tnum_subreg(var64_off), 2110 tnum_range(reg_u32_min(reg), 2111 reg_u32_max(reg))); 2112 2113 reg->var_off = tnum_or(tnum_clear_subreg(var64_off), var32_off); 2114 } 2115 2116 static bool range_bounds_violation(struct bpf_reg_state *reg); 2117 2118 static void reg_bounds_sync(struct bpf_reg_state *reg) 2119 { 2120 /* If the input reg_state is invalid, we can exit early */ 2121 if (range_bounds_violation(reg)) 2122 return; 2123 /* We might have learned new bounds from the var_off. */ 2124 __update_reg_bounds(reg); 2125 /* We might have learned something about the sign bit. */ 2126 __reg_deduce_bounds(reg); 2127 __reg_deduce_bounds(reg); 2128 /* We might have learned some bits from the bounds. */ 2129 __reg_bound_offset(reg); 2130 /* Intersecting with the old var_off might have improved our bounds 2131 * slightly, e.g. if umax was 0x7f...f and var_off was (0; 0xf...fc), 2132 * then new var_off is (0; 0x7f...fc) which improves our umax. 2133 */ 2134 __update_reg_bounds(reg); 2135 } 2136 2137 static bool const_tnum_range_mismatch(struct bpf_reg_state *reg) 2138 { 2139 if (!tnum_is_const(reg->var_off)) 2140 return false; 2141 2142 return !cnum64_is_const(reg->r64) || reg->r64.base != reg->var_off.value; 2143 } 2144 2145 static bool const_tnum_range_mismatch_32(struct bpf_reg_state *reg) 2146 { 2147 if (!tnum_subreg_is_const(reg->var_off)) 2148 return false; 2149 2150 return !cnum32_is_const(reg->r32) || reg->r32.base != tnum_subreg(reg->var_off).value; 2151 } 2152 2153 static bool range_bounds_violation(struct bpf_reg_state *reg) 2154 { 2155 return cnum32_is_empty(reg->r32) || cnum64_is_empty(reg->r64); 2156 } 2157 2158 static int reg_bounds_sanity_check(struct bpf_verifier_env *env, 2159 struct bpf_reg_state *reg, const char *ctx) 2160 { 2161 const char *msg; 2162 2163 if (range_bounds_violation(reg)) { 2164 msg = "range bounds violation"; 2165 goto out; 2166 } 2167 2168 if (const_tnum_range_mismatch(reg)) { 2169 msg = "const tnum out of sync with range bounds"; 2170 goto out; 2171 } 2172 2173 if (const_tnum_range_mismatch_32(reg)) { 2174 msg = "const subreg tnum out of sync with range bounds"; 2175 goto out; 2176 } 2177 2178 return 0; 2179 out: 2180 verifier_bug(env, "REG INVARIANTS VIOLATION (%s): %s r64={.base=%#llx, .size=%#llx} " 2181 "r32={.base=%#x, .size=%#x} var_off=(%#llx, %#llx)", 2182 ctx, msg, 2183 reg->r64.base, reg->r64.size, 2184 reg->r32.base, reg->r32.size, 2185 reg->var_off.value, reg->var_off.mask); 2186 if (env->test_reg_invariants) 2187 return -EFAULT; 2188 __mark_reg_unbounded(reg); 2189 return 0; 2190 } 2191 2192 /* Mark a register as having a completely unknown (scalar) value. */ 2193 void bpf_mark_reg_unknown_imprecise(struct bpf_reg_state *reg) 2194 { 2195 memset(reg, 0, sizeof(*reg)); 2196 reg->type = SCALAR_VALUE; 2197 reg->var_off = tnum_unknown; 2198 __mark_reg_unbounded(reg); 2199 } 2200 2201 /* Mark a register as having a completely unknown (scalar) value, 2202 * initialize .precise as true when not bpf capable. 2203 */ 2204 static void __mark_reg_unknown(const struct bpf_verifier_env *env, 2205 struct bpf_reg_state *reg) 2206 { 2207 bpf_mark_reg_unknown_imprecise(reg); 2208 reg->precise = !env->bpf_capable; 2209 } 2210 2211 static void mark_reg_unknown(struct bpf_verifier_env *env, 2212 struct bpf_reg_state *regs, u32 regno) 2213 { 2214 __mark_reg_unknown(env, regs + regno); 2215 } 2216 2217 static int __mark_reg_s32_range(struct bpf_verifier_env *env, 2218 struct bpf_reg_state *regs, 2219 u32 regno, 2220 s32 s32_min, 2221 s32 s32_max) 2222 { 2223 struct bpf_reg_state *reg = regs + regno; 2224 2225 reg_set_srange32(reg, 2226 max_t(s32, reg_s32_min(reg), s32_min), 2227 min_t(s32, reg_s32_max(reg), s32_max)); 2228 reg_set_srange64(reg, 2229 max_t(s64, reg_smin(reg), s32_min), 2230 min_t(s64, reg_smax(reg), s32_max)); 2231 2232 reg_bounds_sync(reg); 2233 2234 return reg_bounds_sanity_check(env, reg, "s32_range"); 2235 } 2236 2237 void bpf_mark_reg_not_init(const struct bpf_verifier_env *env, 2238 struct bpf_reg_state *reg) 2239 { 2240 __mark_reg_unknown(env, reg); 2241 reg->type = NOT_INIT; 2242 } 2243 2244 static int mark_btf_ld_reg(struct bpf_verifier_env *env, 2245 struct bpf_reg_state *regs, u32 regno, 2246 enum bpf_reg_type reg_type, 2247 struct btf *btf, u32 btf_id, 2248 enum bpf_type_flag flag) 2249 { 2250 switch (reg_type) { 2251 case SCALAR_VALUE: 2252 mark_reg_unknown(env, regs, regno); 2253 return 0; 2254 case PTR_TO_BTF_ID: 2255 mark_reg_known_zero(env, regs, regno); 2256 regs[regno].type = PTR_TO_BTF_ID | flag; 2257 regs[regno].btf = btf; 2258 regs[regno].btf_id = btf_id; 2259 if (type_may_be_null(flag)) 2260 regs[regno].id = ++env->id_gen; 2261 return 0; 2262 case PTR_TO_MEM: 2263 mark_reg_known_zero(env, regs, regno); 2264 regs[regno].type = PTR_TO_MEM | flag; 2265 regs[regno].mem_size = 0; 2266 return 0; 2267 default: 2268 verifier_bug(env, "unexpected reg_type %d in %s\n", reg_type, __func__); 2269 return -EFAULT; 2270 } 2271 } 2272 2273 static void init_reg_state(struct bpf_verifier_env *env, 2274 struct bpf_func_state *state) 2275 { 2276 struct bpf_reg_state *regs = state->regs; 2277 int i; 2278 2279 for (i = 0; i < MAX_BPF_REG; i++) { 2280 bpf_mark_reg_not_init(env, ®s[i]); 2281 } 2282 2283 /* frame pointer */ 2284 regs[BPF_REG_FP].type = PTR_TO_STACK; 2285 mark_reg_known_zero(env, regs, BPF_REG_FP); 2286 regs[BPF_REG_FP].frameno = state->frameno; 2287 } 2288 2289 static struct bpf_retval_range retval_range(s32 minval, s32 maxval) 2290 { 2291 /* 2292 * return_32bit is set to false by default and set explicitly 2293 * by the caller when necessary. 2294 */ 2295 return (struct bpf_retval_range){ minval, maxval, false }; 2296 } 2297 2298 static void init_func_state(struct bpf_verifier_env *env, 2299 struct bpf_func_state *state, 2300 int callsite, int frameno, int subprogno) 2301 { 2302 state->callsite = callsite; 2303 state->frameno = frameno; 2304 bpf_diag_init_frame(env, state); 2305 state->subprogno = subprogno; 2306 state->callback_ret_range = retval_range(0, 0); 2307 init_reg_state(env, state); 2308 mark_verifier_state_scratched(env); 2309 } 2310 2311 /* Similar to push_stack(), but for async callbacks */ 2312 static struct bpf_verifier_state *push_async_cb(struct bpf_verifier_env *env, 2313 int insn_idx, int prev_insn_idx, 2314 int subprog, bool is_sleepable) 2315 { 2316 struct bpf_verifier_stack_elem *elem; 2317 struct bpf_func_state *frame; 2318 2319 elem = kzalloc_obj(struct bpf_verifier_stack_elem, GFP_KERNEL_ACCOUNT); 2320 if (!elem) 2321 return ERR_PTR(-ENOMEM); 2322 2323 elem->insn_idx = insn_idx; 2324 elem->prev_insn_idx = prev_insn_idx; 2325 elem->next = env->head; 2326 elem->log_pos = env->log.end_pos; 2327 elem->diag_log_pos = bpf_diag_event_log_save(env); 2328 env->head = elem; 2329 env->stack_size++; 2330 if (env->stack_size > BPF_COMPLEXITY_LIMIT_JMP_SEQ) { 2331 verbose(env, 2332 "The sequence of %d jumps is too complex for async cb.\n", 2333 env->stack_size); 2334 return ERR_PTR(-E2BIG); 2335 } 2336 /* Unlike push_stack() do not bpf_copy_verifier_state(). 2337 * The caller state doesn't matter. 2338 * This is async callback. It starts in a fresh stack. 2339 * Initialize it similar to do_check_common(). 2340 */ 2341 elem->st.branches = 1; 2342 elem->st.in_sleepable = is_sleepable; 2343 frame = kzalloc_obj(*frame, GFP_KERNEL_ACCOUNT); 2344 if (!frame) 2345 return ERR_PTR(-ENOMEM); 2346 init_func_state(env, frame, 2347 BPF_MAIN_FUNC /* callsite */, 2348 0 /* frameno within this callchain */, 2349 subprog /* subprog number within this prog */); 2350 elem->st.frame[0] = frame; 2351 return &elem->st; 2352 } 2353 2354 static int cmp_subprogs(const void *a, const void *b) 2355 { 2356 return ((struct bpf_subprog_info *)a)->start - 2357 ((struct bpf_subprog_info *)b)->start; 2358 } 2359 2360 /* Find subprogram that contains instruction at 'off' */ 2361 struct bpf_subprog_info *bpf_find_containing_subprog(struct bpf_verifier_env *env, int off) 2362 { 2363 struct bpf_subprog_info *vals = env->subprog_info; 2364 int l, r, m; 2365 2366 if (off >= env->prog->len || off < 0 || env->subprog_cnt == 0) 2367 return NULL; 2368 2369 l = 0; 2370 r = env->subprog_cnt - 1; 2371 while (l < r) { 2372 m = l + (r - l + 1) / 2; 2373 if (vals[m].start <= off) 2374 l = m; 2375 else 2376 r = m - 1; 2377 } 2378 return &vals[l]; 2379 } 2380 2381 /* Find subprogram that starts exactly at 'off' */ 2382 int bpf_find_subprog(struct bpf_verifier_env *env, int off) 2383 { 2384 struct bpf_subprog_info *p; 2385 2386 p = bpf_find_containing_subprog(env, off); 2387 if (!p || p->start != off) 2388 return -ENOENT; 2389 return p - env->subprog_info; 2390 } 2391 2392 static int add_subprog(struct bpf_verifier_env *env, int off) 2393 { 2394 int insn_cnt = env->prog->len; 2395 int ret; 2396 2397 if (off >= insn_cnt || off < 0) { 2398 verbose(env, "call to invalid destination\n"); 2399 return -EINVAL; 2400 } 2401 ret = bpf_find_subprog(env, off); 2402 if (ret >= 0) 2403 return ret; 2404 if (env->subprog_cnt >= BPF_MAX_SUBPROGS) { 2405 verbose(env, "too many subprograms\n"); 2406 return -E2BIG; 2407 } 2408 /* determine subprog starts. The end is one before the next starts */ 2409 env->subprog_info[env->subprog_cnt++].start = off; 2410 sort(env->subprog_info, env->subprog_cnt, 2411 sizeof(env->subprog_info[0]), cmp_subprogs, NULL); 2412 return env->subprog_cnt - 1; 2413 } 2414 2415 static int bpf_find_exception_callback_insn_off(struct bpf_verifier_env *env) 2416 { 2417 struct bpf_prog_aux *aux = env->prog->aux; 2418 struct btf *btf = aux->btf; 2419 const struct btf_type *t; 2420 u32 main_btf_id, id; 2421 const char *name; 2422 int ret, i; 2423 2424 /* Non-zero func_info_cnt implies valid btf */ 2425 if (!aux->func_info_cnt) 2426 return 0; 2427 main_btf_id = aux->func_info[0].type_id; 2428 2429 t = btf_type_by_id(btf, main_btf_id); 2430 if (!t) { 2431 verbose(env, "invalid btf id for main subprog in func_info\n"); 2432 return -EINVAL; 2433 } 2434 2435 name = btf_find_decl_tag_value(btf, t, -1, "exception_callback:"); 2436 if (IS_ERR(name)) { 2437 ret = PTR_ERR(name); 2438 /* If there is no tag present, there is no exception callback */ 2439 if (ret == -ENOENT) 2440 ret = 0; 2441 else if (ret == -EEXIST) 2442 verbose(env, "multiple exception callback tags for main subprog\n"); 2443 return ret; 2444 } 2445 2446 ret = btf_find_by_name_kind(btf, name, BTF_KIND_FUNC); 2447 if (ret < 0) { 2448 verbose(env, "exception callback '%s' could not be found in BTF\n", name); 2449 return ret; 2450 } 2451 id = ret; 2452 t = btf_type_by_id(btf, id); 2453 if (btf_func_linkage(t) != BTF_FUNC_GLOBAL) { 2454 verbose(env, "exception callback '%s' must have global linkage\n", name); 2455 return -EINVAL; 2456 } 2457 ret = 0; 2458 for (i = 0; i < aux->func_info_cnt; i++) { 2459 if (aux->func_info[i].type_id != id) 2460 continue; 2461 ret = aux->func_info[i].insn_off; 2462 /* Further func_info and subprog checks will also happen 2463 * later, so assume this is the right insn_off for now. 2464 */ 2465 if (!ret) { 2466 verbose(env, "invalid exception callback insn_off in func_info: 0\n"); 2467 ret = -EINVAL; 2468 } 2469 } 2470 if (!ret) { 2471 verbose(env, "exception callback type id not found in func_info\n"); 2472 ret = -EINVAL; 2473 } 2474 return ret; 2475 } 2476 2477 #define MAX_KFUNC_BTFS 256 2478 2479 struct bpf_kfunc_btf { 2480 struct btf *btf; 2481 struct module *module; 2482 u16 offset; 2483 }; 2484 2485 struct bpf_kfunc_btf_tab { 2486 struct bpf_kfunc_btf descs[MAX_KFUNC_BTFS]; 2487 u32 nr_descs; 2488 }; 2489 2490 static int kfunc_desc_cmp_by_id_off(const void *a, const void *b) 2491 { 2492 const struct bpf_kfunc_desc *d0 = a; 2493 const struct bpf_kfunc_desc *d1 = b; 2494 2495 /* func_id is not greater than BTF_MAX_TYPE */ 2496 return d0->func_id - d1->func_id ?: d0->offset - d1->offset; 2497 } 2498 2499 static int kfunc_btf_cmp_by_off(const void *a, const void *b) 2500 { 2501 const struct bpf_kfunc_btf *d0 = a; 2502 const struct bpf_kfunc_btf *d1 = b; 2503 2504 return d0->offset - d1->offset; 2505 } 2506 2507 static struct bpf_kfunc_desc * 2508 find_kfunc_desc(const struct bpf_prog *prog, u32 func_id, u16 offset) 2509 { 2510 struct bpf_kfunc_desc desc = { 2511 .func_id = func_id, 2512 .offset = offset, 2513 }; 2514 struct bpf_kfunc_desc_tab *tab; 2515 2516 tab = prog->aux->kfunc_tab; 2517 return bsearch(&desc, tab->descs, tab->nr_descs, 2518 sizeof(tab->descs[0]), kfunc_desc_cmp_by_id_off); 2519 } 2520 2521 int bpf_get_kfunc_addr(const struct bpf_prog *prog, u32 func_id, 2522 u16 btf_fd_idx, u8 **func_addr) 2523 { 2524 const struct bpf_kfunc_desc *desc; 2525 2526 desc = find_kfunc_desc(prog, func_id, btf_fd_idx); 2527 if (!desc) 2528 return -EFAULT; 2529 2530 *func_addr = (u8 *)desc->addr; 2531 return 0; 2532 } 2533 2534 #define BPF_FD_SLOT_BTF 1UL 2535 2536 static void fd_slot_set_map(struct bpf_fd_array *slot, struct bpf_map *map) 2537 { 2538 slot->val = (unsigned long)map; 2539 } 2540 2541 static void fd_slot_set_btf(struct bpf_fd_array *slot, struct btf *btf) 2542 { 2543 slot->val = (unsigned long)btf | BPF_FD_SLOT_BTF; 2544 } 2545 2546 static struct bpf_map *fd_slot_map(struct bpf_fd_array slot) 2547 { 2548 if (slot.val & BPF_FD_SLOT_BTF) 2549 return NULL; 2550 return (struct bpf_map *)slot.val; 2551 } 2552 2553 static struct btf *fd_slot_btf(struct bpf_fd_array slot) 2554 { 2555 if (!(slot.val & BPF_FD_SLOT_BTF)) 2556 return NULL; 2557 return (struct btf *)(slot.val & ~BPF_FD_SLOT_BTF); 2558 } 2559 2560 static struct btf * 2561 fd_array_get_btf_continuous(struct bpf_verifier_env *env, u32 idx) 2562 { 2563 struct btf *btf; 2564 2565 if (idx >= env->fd_array_cnt) { 2566 verbose(env, "kfunc fd_idx %u out of bounds, fd_array_cnt %u\n", 2567 idx, env->fd_array_cnt); 2568 return ERR_PTR(-EINVAL); 2569 } 2570 btf = fd_slot_btf(env->fd_array[idx]); 2571 if (!btf) { 2572 verbose(env, "kfunc fd_idx %u is not a module BTF\n", idx); 2573 return ERR_PTR(-EINVAL); 2574 } 2575 btf_get(btf); 2576 return btf; 2577 } 2578 2579 static struct btf * 2580 fd_array_get_btf_sparse(struct bpf_verifier_env *env, u32 idx) 2581 { 2582 struct btf *btf; 2583 int btf_fd; 2584 2585 if (copy_from_bpfptr_offset(&btf_fd, env->fd_array_raw, 2586 (size_t)idx * sizeof(btf_fd), sizeof(btf_fd))) 2587 return ERR_PTR(-EFAULT); 2588 btf = btf_get_by_fd(btf_fd); 2589 if (IS_ERR(btf)) { 2590 verbose(env, "invalid module BTF fd specified\n"); 2591 return btf; 2592 } 2593 return btf; 2594 } 2595 2596 static struct btf *fd_array_get_btf(struct bpf_verifier_env *env, u32 idx) 2597 { 2598 if (env->signature) { 2599 verbose(env, "signed program cannot bind any BTF\n"); 2600 return ERR_PTR(-EACCES); 2601 } 2602 if (env->fd_array) 2603 return fd_array_get_btf_continuous(env, idx); 2604 if (!bpfptr_is_null(env->fd_array_raw)) 2605 return fd_array_get_btf_sparse(env, idx); 2606 2607 verbose(env, "kfunc offset > 0 without fd_array is invalid\n"); 2608 return ERR_PTR(-EPROTO); 2609 } 2610 2611 static struct btf *__find_kfunc_desc_btf(struct bpf_verifier_env *env, 2612 s16 offset) 2613 { 2614 struct bpf_kfunc_btf kf_btf = { .offset = offset }; 2615 struct bpf_kfunc_btf_tab *tab; 2616 struct bpf_kfunc_btf *b; 2617 struct module *mod; 2618 struct btf *btf; 2619 2620 tab = env->prog->aux->kfunc_btf_tab; 2621 b = bsearch(&kf_btf, tab->descs, tab->nr_descs, 2622 sizeof(tab->descs[0]), kfunc_btf_cmp_by_off); 2623 if (!b) { 2624 if (tab->nr_descs == MAX_KFUNC_BTFS) { 2625 verbose(env, "too many different module BTFs\n"); 2626 return ERR_PTR(-E2BIG); 2627 } 2628 2629 btf = fd_array_get_btf(env, offset); 2630 if (IS_ERR(btf)) 2631 return btf; 2632 if (!btf_is_module(btf)) { 2633 verbose(env, "BTF fd for kfunc is not a module BTF\n"); 2634 btf_put(btf); 2635 return ERR_PTR(-EINVAL); 2636 } 2637 2638 mod = btf_try_get_module(btf); 2639 if (!mod) { 2640 btf_put(btf); 2641 return ERR_PTR(-ENXIO); 2642 } 2643 2644 b = &tab->descs[tab->nr_descs++]; 2645 b->btf = btf; 2646 b->module = mod; 2647 b->offset = offset; 2648 2649 /* sort() reorders entries by value, so b may no longer point 2650 * to the right entry after this 2651 */ 2652 sort(tab->descs, tab->nr_descs, sizeof(tab->descs[0]), 2653 kfunc_btf_cmp_by_off, NULL); 2654 } else { 2655 btf = b->btf; 2656 } 2657 2658 return btf; 2659 } 2660 2661 void bpf_free_kfunc_btf_tab(struct bpf_kfunc_btf_tab *tab) 2662 { 2663 if (!tab) 2664 return; 2665 2666 while (tab->nr_descs--) { 2667 module_put(tab->descs[tab->nr_descs].module); 2668 btf_put(tab->descs[tab->nr_descs].btf); 2669 } 2670 kfree(tab); 2671 } 2672 2673 static struct btf *find_kfunc_desc_btf(struct bpf_verifier_env *env, s16 offset) 2674 { 2675 if (offset) { 2676 if (offset < 0) { 2677 /* In the future, this can be allowed to increase limit 2678 * of fd index into fd_array, interpreted as u16. 2679 */ 2680 verbose(env, "negative offset disallowed for kernel module function call\n"); 2681 return ERR_PTR(-EINVAL); 2682 } 2683 2684 return __find_kfunc_desc_btf(env, offset); 2685 } 2686 return btf_vmlinux ?: ERR_PTR(-ENOENT); 2687 } 2688 2689 static struct btf *find_kfunc_desc_btf_cached(struct bpf_verifier_env *env, s16 offset) 2690 { 2691 struct bpf_kfunc_btf kf_btf = { .offset = offset }; 2692 struct bpf_kfunc_btf_tab *tab; 2693 struct bpf_kfunc_btf *b; 2694 2695 if (!offset) 2696 return btf_vmlinux ?: ERR_PTR(-ENOENT); 2697 if (offset < 0) 2698 return ERR_PTR(-EINVAL); 2699 2700 tab = env->prog->aux->kfunc_btf_tab; 2701 if (!tab) 2702 return ERR_PTR(-ENOENT); 2703 2704 b = bsearch(&kf_btf, tab->descs, tab->nr_descs, 2705 sizeof(tab->descs[0]), kfunc_btf_cmp_by_off); 2706 return b ? b->btf : ERR_PTR(-ENOENT); 2707 } 2708 2709 #define KF_IMPL_SUFFIX "_impl" 2710 2711 static const struct btf_type *find_kfunc_impl_proto(struct bpf_verifier_log *log, 2712 struct btf *btf, 2713 const char *func_name) 2714 { 2715 const struct btf_type *func; 2716 char buf[KSYM_NAME_LEN]; 2717 s32 impl_id; 2718 int len; 2719 2720 len = snprintf(buf, sizeof(buf), "%s%s", func_name, KF_IMPL_SUFFIX); 2721 if (len < 0 || len >= sizeof(buf)) { 2722 bpf_log(log, "function name %s%s is too long\n", 2723 func_name, KF_IMPL_SUFFIX); 2724 return NULL; 2725 } 2726 2727 impl_id = btf_find_by_name_kind(btf, buf, BTF_KIND_FUNC); 2728 if (impl_id <= 0) { 2729 bpf_log(log, "cannot find function %s in BTF\n", buf); 2730 return NULL; 2731 } 2732 2733 func = btf_type_by_id(btf, impl_id); 2734 2735 return btf_type_by_id(btf, func->type); 2736 } 2737 2738 static int fetch_kfunc_meta(struct bpf_verifier_env *env, 2739 s32 func_id, 2740 s16 offset, 2741 struct bpf_kfunc_meta *kfunc) 2742 { 2743 const struct btf_type *func, *func_proto; 2744 const char *func_name; 2745 u32 *kfunc_flags; 2746 struct btf *btf; 2747 2748 if (func_id <= 0) { 2749 verbose(env, "invalid kernel function btf_id %d\n", func_id); 2750 return -EINVAL; 2751 } 2752 2753 btf = find_kfunc_desc_btf(env, offset); 2754 if (IS_ERR(btf)) { 2755 verbose(env, "failed to find BTF for kernel function\n"); 2756 return PTR_ERR(btf); 2757 } 2758 2759 /* 2760 * Note that kfunc_flags may be NULL at this point, which 2761 * means that we couldn't find func_id in any relevant 2762 * kfunc_id_set. This most likely indicates an invalid kfunc 2763 * call. However we don't fail with an error here, 2764 * and let the caller decide what to do with NULL kfunc->flags. 2765 */ 2766 kfunc_flags = btf_kfunc_flags(btf, func_id, env->prog); 2767 2768 func = btf_type_by_id(btf, func_id); 2769 if (!func || !btf_type_is_func(func)) { 2770 verbose(env, "kernel btf_id %d is not a function\n", func_id); 2771 return -EINVAL; 2772 } 2773 2774 func_name = btf_name_by_offset(btf, func->name_off); 2775 2776 /* 2777 * An actual prototype of a kfunc with KF_IMPLICIT_ARGS flag 2778 * can be found through the counterpart _impl kfunc. 2779 */ 2780 if (kfunc_flags && (*kfunc_flags & KF_IMPLICIT_ARGS)) 2781 func_proto = find_kfunc_impl_proto(&env->log, btf, func_name); 2782 else 2783 func_proto = btf_type_by_id(btf, func->type); 2784 2785 if (!func_proto || !btf_type_is_func_proto(func_proto)) { 2786 verbose(env, "kernel function btf_id %d does not have a valid func_proto\n", 2787 func_id); 2788 return -EINVAL; 2789 } 2790 2791 memset(kfunc, 0, sizeof(*kfunc)); 2792 kfunc->btf = btf; 2793 kfunc->id = func_id; 2794 kfunc->name = func_name; 2795 kfunc->proto = func_proto; 2796 kfunc->flags = kfunc_flags; 2797 2798 return 0; 2799 } 2800 2801 static int gen_kfunc_arg_proto(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta, 2802 struct bpf_func_proto *proto); 2803 2804 int bpf_add_kfunc_call(struct bpf_verifier_env *env, u32 func_id, u16 offset) 2805 { 2806 struct bpf_call_arg_meta meta; 2807 struct bpf_kfunc_btf_tab *btf_tab; 2808 struct btf_func_model func_model; 2809 struct bpf_kfunc_desc_tab *tab; 2810 struct bpf_prog_aux *prog_aux; 2811 struct bpf_kfunc_meta kfunc; 2812 struct bpf_kfunc_desc *desc; 2813 unsigned long addr; 2814 int err; 2815 2816 prog_aux = env->prog->aux; 2817 tab = prog_aux->kfunc_tab; 2818 btf_tab = prog_aux->kfunc_btf_tab; 2819 if (!tab) { 2820 if (!btf_vmlinux) { 2821 verbose(env, "calling kernel function is not supported without CONFIG_DEBUG_INFO_BTF\n"); 2822 return -ENOTSUPP; 2823 } 2824 2825 if (!env->prog->jit_requested) { 2826 verbose(env, "JIT is required for calling kernel function\n"); 2827 return -ENOTSUPP; 2828 } 2829 2830 if (!bpf_jit_supports_kfunc_call()) { 2831 verbose(env, "JIT does not support calling kernel function\n"); 2832 return -ENOTSUPP; 2833 } 2834 2835 if (!env->prog->gpl_compatible) { 2836 verbose(env, "cannot call kernel function from non-GPL compatible program\n"); 2837 return -EINVAL; 2838 } 2839 2840 tab = kzalloc_obj(*tab, GFP_KERNEL_ACCOUNT); 2841 if (!tab) 2842 return -ENOMEM; 2843 prog_aux->kfunc_tab = tab; 2844 } 2845 2846 env->prog->jit_required = 1; 2847 2848 /* func_id == 0 is always invalid, but instead of returning an error, be 2849 * conservative and wait until the code elimination pass before returning 2850 * error, so that invalid calls that get pruned out can be in BPF programs 2851 * loaded from userspace. It is also required that offset be untouched 2852 * for such calls. 2853 */ 2854 if (!func_id && !offset) 2855 return 0; 2856 2857 if (!btf_tab && offset) { 2858 btf_tab = kzalloc_obj(*btf_tab, GFP_KERNEL_ACCOUNT); 2859 if (!btf_tab) 2860 return -ENOMEM; 2861 prog_aux->kfunc_btf_tab = btf_tab; 2862 } 2863 2864 if (find_kfunc_desc(env->prog, func_id, offset)) 2865 return 0; 2866 2867 if (tab->nr_descs == MAX_KFUNC_DESCS) { 2868 verbose(env, "too many different kernel function calls\n"); 2869 return -E2BIG; 2870 } 2871 2872 err = fetch_kfunc_meta(env, func_id, offset, &kfunc); 2873 if (err) 2874 return err; 2875 2876 addr = kallsyms_lookup_name(kfunc.name); 2877 if (!addr) { 2878 verbose(env, "cannot find address for kernel function %s\n", kfunc.name); 2879 return -EINVAL; 2880 } 2881 2882 if (bpf_dev_bound_kfunc_id(func_id)) { 2883 err = bpf_dev_bound_kfunc_check(&env->log, prog_aux); 2884 if (err) 2885 return err; 2886 } 2887 2888 err = btf_distill_func_proto(&env->log, kfunc.btf, kfunc.proto, kfunc.name, &func_model); 2889 if (err) 2890 return err; 2891 2892 memset(&meta, 0, sizeof(meta)); 2893 meta.btf = kfunc.btf; 2894 meta.func_id = kfunc.id; 2895 meta.func_proto = kfunc.proto; 2896 meta.func_name = kfunc.name; 2897 meta.kfunc_flags = kfunc.flags ? *kfunc.flags : 0; 2898 2899 tab = krealloc(tab, struct_size(tab, descs, tab->nr_descs + 1), GFP_KERNEL_ACCOUNT); 2900 if (!tab) 2901 return -ENOMEM; 2902 prog_aux->kfunc_tab = tab; 2903 2904 desc = &tab->descs[tab->nr_descs]; 2905 memset(desc, 0, sizeof(*desc)); 2906 2907 err = gen_kfunc_arg_proto(env, &meta, &desc->proto); 2908 if (err) 2909 return err; 2910 2911 desc->func_id = func_id; 2912 desc->offset = offset; 2913 desc->addr = addr; 2914 desc->func_model = func_model; 2915 tab->nr_descs++; 2916 sort(tab->descs, tab->nr_descs, sizeof(tab->descs[0]), 2917 kfunc_desc_cmp_by_id_off, NULL); 2918 return 0; 2919 } 2920 2921 static int add_subprogs(struct bpf_verifier_env *env) 2922 { 2923 struct bpf_subprog_info *subprog = env->subprog_info; 2924 int i, ret, insn_cnt = env->prog->len, ex_cb_insn; 2925 struct bpf_insn *insn = env->prog->insnsi; 2926 const char *operation, *suggestion; 2927 2928 /* Add entry function. */ 2929 ret = add_subprog(env, 0); 2930 if (ret) 2931 return ret; 2932 2933 for (i = 0; i < insn_cnt; i++, insn++) { 2934 if (!bpf_pseudo_func(insn) && !bpf_pseudo_call(insn)) 2935 continue; 2936 2937 if (!env->bpf_capable) { 2938 if (bpf_pseudo_func(insn)) { 2939 operation = "BPF function reference"; 2940 suggestion = "Load this program with the required capability, or avoid BPF function references in unprivileged programs."; 2941 } else { 2942 operation = "BPF-to-BPF function call"; 2943 suggestion = "Load this program with the required capability, or avoid BPF-to-BPF function calls in unprivileged programs."; 2944 } 2945 verbose(env, "loading/calling other bpf or kernel functions are allowed for CAP_BPF and CAP_SYS_ADMIN\n"); 2946 bpf_diag_policy( 2947 env, i, operation, 2948 "loading or calling other BPF functions requires CAP_BPF or CAP_SYS_ADMIN", 2949 suggestion); 2950 return -EPERM; 2951 } 2952 2953 ret = add_subprog(env, i + insn->imm + 1); 2954 if (ret < 0) 2955 return ret; 2956 } 2957 2958 ret = bpf_find_exception_callback_insn_off(env); 2959 if (ret < 0) 2960 return ret; 2961 ex_cb_insn = ret; 2962 2963 /* If ex_cb_insn > 0, this means that the main program has a subprog 2964 * marked using BTF decl tag to serve as the exception callback. 2965 */ 2966 if (ex_cb_insn) { 2967 ret = add_subprog(env, ex_cb_insn); 2968 if (ret < 0) 2969 return ret; 2970 for (i = 1; i < env->subprog_cnt; i++) { 2971 if (env->subprog_info[i].start != ex_cb_insn) 2972 continue; 2973 env->exception_callback_subprog = i; 2974 bpf_mark_subprog_exc_cb(env, i); 2975 break; 2976 } 2977 } 2978 2979 /* Add a fake 'exit' subprog which could simplify subprog iteration 2980 * logic. 'subprog_cnt' should not be increased. 2981 */ 2982 subprog[env->subprog_cnt].start = insn_cnt; 2983 2984 if (env->log.level & BPF_LOG_LEVEL2) 2985 for (i = 0; i < env->subprog_cnt; i++) 2986 verbose(env, "func#%d @%d\n", i, subprog[i].start); 2987 2988 return 0; 2989 } 2990 2991 static int add_kfuncs(struct bpf_verifier_env *env) 2992 { 2993 struct bpf_insn *insn = env->prog->insnsi; 2994 int i, ret, insn_cnt = env->prog->len; 2995 2996 for (i = 0; i < insn_cnt; i++, insn++) { 2997 if (!bpf_pseudo_kfunc_call(insn)) 2998 continue; 2999 3000 if (!env->bpf_capable) { 3001 verbose(env, "loading/calling other bpf or kernel functions are allowed for CAP_BPF and CAP_SYS_ADMIN\n"); 3002 bpf_diag_policy( 3003 env, i, "kernel function call", 3004 "calling kernel functions requires CAP_BPF or CAP_SYS_ADMIN", 3005 "Load this program with the required capability, or avoid kernel function calls in unprivileged programs."); 3006 return -EPERM; 3007 } 3008 3009 ret = bpf_add_kfunc_call(env, insn->imm, insn->off); 3010 if (ret < 0) 3011 return ret; 3012 } 3013 3014 return 0; 3015 } 3016 3017 static int check_subprogs(struct bpf_verifier_env *env) 3018 { 3019 int i, subprog_start, subprog_end, off, cur_subprog = 0; 3020 struct bpf_subprog_info *subprog = env->subprog_info; 3021 struct bpf_insn *insn = env->prog->insnsi; 3022 int insn_cnt = env->prog->len; 3023 3024 /* now check that all jumps are within the same subprog */ 3025 subprog_start = subprog[cur_subprog].start; 3026 subprog_end = subprog[cur_subprog + 1].start; 3027 for (i = 0; i < insn_cnt; i++) { 3028 u8 code = insn[i].code; 3029 3030 if (code == (BPF_JMP | BPF_CALL) && 3031 insn[i].src_reg == 0 && 3032 insn[i].imm == BPF_FUNC_tail_call) { 3033 subprog[cur_subprog].has_tail_call = true; 3034 subprog[cur_subprog].tail_call_reachable = true; 3035 } 3036 if (BPF_CLASS(code) == BPF_LD && 3037 (BPF_MODE(code) == BPF_ABS || BPF_MODE(code) == BPF_IND)) 3038 subprog[cur_subprog].has_ld_abs = true; 3039 if (BPF_CLASS(code) != BPF_JMP && BPF_CLASS(code) != BPF_JMP32) 3040 goto next; 3041 if (BPF_OP(code) == BPF_CALL) 3042 goto next; 3043 if (BPF_OP(code) == BPF_EXIT) { 3044 subprog[cur_subprog].exit_idx = i; 3045 goto next; 3046 } 3047 if (insn_is_gotox(&insn[i])) 3048 goto next; 3049 off = i + bpf_jmp_offset(&insn[i]) + 1; 3050 if (off < subprog_start || off >= subprog_end) { 3051 verbose(env, "jump out of range from insn %d to %d\n", i, off); 3052 bpf_diag_program_structure( 3053 env, i, "jump out of range", 3054 "Keep branch targets within the same subprogram, or use an explicit subprogram call.", 3055 "Instruction %d jumps to instruction %d, but subprogram %d only contains instructions %d through %d. " 3056 "A branch target must stay inside the same subprogram.", 3057 i, off, cur_subprog, subprog_start, subprog_end - 1); 3058 return -EINVAL; 3059 } 3060 next: 3061 if (i == subprog_end - 1) { 3062 /* to avoid fall-through from one subprog into another 3063 * the last insn of the subprog should be either exit 3064 * or unconditional jump back or bpf_throw call 3065 */ 3066 if (code != (BPF_JMP | BPF_EXIT) && 3067 code != (BPF_JMP32 | BPF_JA) && 3068 code != (BPF_JMP | BPF_JA) && 3069 !insn_is_gotox(&insn[i])) { 3070 verbose(env, "last insn is not an exit or jmp\n"); 3071 bpf_diag_program_structure( 3072 env, i, "subprogram can fall through", 3073 "End each subprogram with an exit or an explicit jump that keeps control flow inside the subprogram.", 3074 "Subprogram %d reaches its last instruction %d without an exit or jump, so control could continue into the next subprogram.", 3075 cur_subprog, i); 3076 return -EINVAL; 3077 } 3078 subprog_start = subprog_end; 3079 cur_subprog++; 3080 if (cur_subprog < env->subprog_cnt) 3081 subprog_end = subprog[cur_subprog + 1].start; 3082 } 3083 } 3084 return 0; 3085 } 3086 3087 /* 3088 * Sort subprogs in topological order so that leaf subprogs come first and 3089 * their callers come later. This is a DFS post-order traversal of the call 3090 * graph. Scan only reachable instructions (those in the computed postorder) of 3091 * the current subprog to discover callees (direct subprogs and sync 3092 * callbacks). 3093 */ 3094 static int sort_subprogs_topo(struct bpf_verifier_env *env) 3095 { 3096 struct bpf_subprog_info *si = env->subprog_info; 3097 int *insn_postorder = env->cfg.insn_postorder; 3098 struct bpf_insn *insn = env->prog->insnsi; 3099 int cnt = env->subprog_cnt; 3100 int *dfs_stack = NULL; 3101 int top = 0, order = 0; 3102 int i, ret = 0; 3103 u8 *color = NULL; 3104 3105 color = kvzalloc_objs(*color, cnt, GFP_KERNEL_ACCOUNT); 3106 dfs_stack = kvmalloc_objs(*dfs_stack, cnt, GFP_KERNEL_ACCOUNT); 3107 if (!color || !dfs_stack) { 3108 ret = -ENOMEM; 3109 goto out; 3110 } 3111 3112 /* 3113 * DFS post-order traversal. 3114 * Color values: 0 = unvisited, 1 = on stack, 2 = done. 3115 */ 3116 for (i = 0; i < cnt; i++) { 3117 if (color[i]) 3118 continue; 3119 color[i] = 1; 3120 dfs_stack[top++] = i; 3121 3122 while (top > 0) { 3123 int cur = dfs_stack[top - 1]; 3124 int po_start = si[cur].postorder_start; 3125 int po_end = si[cur + 1].postorder_start; 3126 bool pushed = false; 3127 int j; 3128 3129 for (j = po_start; j < po_end; j++) { 3130 int idx = insn_postorder[j]; 3131 int callee; 3132 3133 if (!bpf_pseudo_call(&insn[idx]) && !bpf_pseudo_func(&insn[idx])) 3134 continue; 3135 callee = bpf_find_subprog(env, idx + insn[idx].imm + 1); 3136 if (callee < 0) { 3137 ret = -EFAULT; 3138 goto out; 3139 } 3140 if (color[callee] == 2) 3141 continue; 3142 if (color[callee] == 1) { 3143 if (bpf_pseudo_func(&insn[idx])) 3144 continue; 3145 verbose(env, "recursive call from %s() to %s()\n", 3146 bpf_subprog_name(env, cur), 3147 bpf_subprog_name(env, callee)); 3148 bpf_diag_program_structure( 3149 env, idx, "recursive subprogram call", 3150 "Rewrite the recursion as an explicit bounded loop, or split the logic so subprogram calls do not form a cycle.", 3151 "This bpf2bpf call would make the subprogram call graph recursive. " 3152 "The verifier requires a finite, acyclic call graph so it can bound stack depth and analysis."); 3153 ret = -EINVAL; 3154 goto out; 3155 } 3156 color[callee] = 1; 3157 dfs_stack[top++] = callee; 3158 pushed = true; 3159 break; 3160 } 3161 3162 if (!pushed) { 3163 color[cur] = 2; 3164 env->subprog_topo_order[order++] = cur; 3165 top--; 3166 } 3167 } 3168 } 3169 3170 if (env->log.level & BPF_LOG_LEVEL2) 3171 for (i = 0; i < cnt; i++) 3172 verbose(env, "topo_order[%d] = %s\n", 3173 i, bpf_subprog_name(env, env->subprog_topo_order[i])); 3174 out: 3175 kvfree(dfs_stack); 3176 kvfree(color); 3177 return ret; 3178 } 3179 3180 static void mark_stack_slots_scratched(struct bpf_verifier_env *env, 3181 int spi, int nr_slots) 3182 { 3183 int i; 3184 3185 for (i = 0; i < nr_slots; i++) 3186 mark_stack_slot_scratched(env, spi - i); 3187 } 3188 3189 static int __check_reg_arg(struct bpf_verifier_env *env, struct bpf_reg_state *regs, u32 regno, 3190 enum bpf_reg_arg_type t) 3191 { 3192 struct bpf_reg_state *reg; 3193 3194 mark_reg_scratched(env, regno); 3195 3196 reg = ®s[regno]; 3197 if (t == SRC_OP) { 3198 /* check whether register used as source operand can be read */ 3199 if (reg->type == NOT_INIT) { 3200 verbose(env, "R%d !read_ok\n", regno); 3201 bpf_diag_unreadable_reg(env, env->insn_idx, regno); 3202 return -EACCES; 3203 } 3204 /* We don't need to worry about FP liveness because it's read-only */ 3205 if (regno == BPF_REG_FP) 3206 return 0; 3207 3208 return 0; 3209 } else { 3210 /* check whether register used as dest operand can be written to */ 3211 if (regno == BPF_REG_FP) { 3212 verbose(env, "frame pointer is read only\n"); 3213 return -EACCES; 3214 } 3215 if (t == DST_OP) 3216 mark_reg_unknown(env, regs, regno); 3217 } 3218 return 0; 3219 } 3220 3221 static int check_reg_arg(struct bpf_verifier_env *env, u32 regno, 3222 enum bpf_reg_arg_type t) 3223 { 3224 struct bpf_verifier_state *vstate = env->cur_state; 3225 struct bpf_func_state *state = vstate->frame[vstate->curframe]; 3226 3227 return __check_reg_arg(env, state->regs, regno, t); 3228 } 3229 3230 static void mark_indirect_target(struct bpf_verifier_env *env, int idx) 3231 { 3232 env->insn_aux_data[idx].indirect_target = true; 3233 } 3234 3235 #define LR_FRAMENO_BITS 4 3236 #define LR_SPI_BITS 6 3237 #define LR_ENTRY_BITS (LR_SPI_BITS + LR_FRAMENO_BITS + 1) 3238 #define LR_SIZE_BITS 4 3239 #define LR_FRAMENO_MASK ((1ull << LR_FRAMENO_BITS) - 1) 3240 #define LR_SPI_MASK ((1ull << LR_SPI_BITS) - 1) 3241 #define LR_SIZE_MASK ((1ull << LR_SIZE_BITS) - 1) 3242 #define LR_SPI_OFF LR_FRAMENO_BITS 3243 #define LR_IS_REG_OFF (LR_SPI_BITS + LR_FRAMENO_BITS) 3244 #define LINKED_REGS_MAX 5 3245 3246 static_assert(MAX_CALL_FRAMES <= (1 << LR_FRAMENO_BITS)); 3247 static_assert(LINKED_REGS_MAX < (1 << LR_SIZE_BITS)); 3248 static_assert(LINKED_REGS_MAX * LR_ENTRY_BITS + LR_SIZE_BITS <= 64); 3249 3250 struct linked_reg { 3251 u8 frameno; 3252 union { 3253 u8 spi; 3254 u8 regno; 3255 }; 3256 bool is_reg; 3257 }; 3258 3259 struct linked_regs { 3260 int cnt; 3261 struct linked_reg entries[LINKED_REGS_MAX]; 3262 }; 3263 3264 static struct linked_reg *linked_regs_push(struct linked_regs *s) 3265 { 3266 if (s->cnt < LINKED_REGS_MAX) 3267 return &s->entries[s->cnt++]; 3268 3269 return NULL; 3270 } 3271 3272 /* 3273 * Use u64 as a vector of 5 11-bit values, use first 4-bits to track 3274 * number of elements currently in stack. 3275 * Pack one history entry for linked registers as 11 bits in the following format: 3276 * - 4-bits frameno 3277 * - 6-bits spi_or_reg 3278 * - 1-bit is_reg 3279 */ 3280 static u64 linked_regs_pack(struct linked_regs *s) 3281 { 3282 u64 val = 0; 3283 int i; 3284 3285 for (i = 0; i < s->cnt; ++i) { 3286 struct linked_reg *e = &s->entries[i]; 3287 u64 tmp = 0; 3288 3289 tmp |= e->frameno; 3290 tmp |= e->spi << LR_SPI_OFF; 3291 tmp |= (e->is_reg ? 1 : 0) << LR_IS_REG_OFF; 3292 3293 val <<= LR_ENTRY_BITS; 3294 val |= tmp; 3295 } 3296 val <<= LR_SIZE_BITS; 3297 val |= s->cnt; 3298 return val; 3299 } 3300 3301 static void linked_regs_unpack(u64 val, struct linked_regs *s) 3302 { 3303 int i; 3304 3305 s->cnt = val & LR_SIZE_MASK; 3306 val >>= LR_SIZE_BITS; 3307 3308 for (i = 0; i < s->cnt; ++i) { 3309 struct linked_reg *e = &s->entries[i]; 3310 3311 e->frameno = val & LR_FRAMENO_MASK; 3312 e->spi = (val >> LR_SPI_OFF) & LR_SPI_MASK; 3313 e->is_reg = (val >> LR_IS_REG_OFF) & 0x1; 3314 val >>= LR_ENTRY_BITS; 3315 } 3316 } 3317 3318 const char *bpf_disasm_kfunc_name(void *data, const struct bpf_insn *insn) 3319 { 3320 const struct btf_type *func; 3321 struct btf *desc_btf; 3322 3323 if (insn->src_reg != BPF_PSEUDO_KFUNC_CALL) 3324 return NULL; 3325 3326 desc_btf = find_kfunc_desc_btf_cached(data, insn->off); 3327 if (IS_ERR(desc_btf)) 3328 return "<error>"; 3329 3330 func = btf_type_by_id(desc_btf, insn->imm); 3331 if (!func || !btf_type_is_func(func)) 3332 return "<error>"; 3333 return btf_name_by_offset(desc_btf, func->name_off); 3334 } 3335 3336 void bpf_verbose_insn(struct bpf_verifier_env *env, struct bpf_insn *insn) 3337 { 3338 const struct bpf_insn_cbs cbs = { 3339 .cb_call = bpf_disasm_kfunc_name, 3340 .cb_print = verbose, 3341 .private_data = env, 3342 }; 3343 3344 print_bpf_insn(&cbs, insn, env->allow_ptr_leaks); 3345 } 3346 3347 /* If any register R in hist->linked_regs is marked as precise in bt, 3348 * do bt_set_frame_{reg,slot}(bt, R) for all registers in hist->linked_regs. 3349 */ 3350 void bpf_bt_sync_linked_regs(struct backtrack_state *bt, struct bpf_jmp_history_entry *hist) 3351 { 3352 struct linked_regs linked_regs; 3353 bool some_precise = false; 3354 int i; 3355 3356 if (!hist || hist->linked_regs == 0) 3357 return; 3358 3359 linked_regs_unpack(hist->linked_regs, &linked_regs); 3360 for (i = 0; i < linked_regs.cnt; ++i) { 3361 struct linked_reg *e = &linked_regs.entries[i]; 3362 3363 if ((e->is_reg && bt_is_frame_reg_set(bt, e->frameno, e->regno)) || 3364 (!e->is_reg && bt_is_frame_slot_set(bt, e->frameno, e->spi))) { 3365 some_precise = true; 3366 break; 3367 } 3368 } 3369 3370 if (!some_precise) 3371 return; 3372 3373 for (i = 0; i < linked_regs.cnt; ++i) { 3374 struct linked_reg *e = &linked_regs.entries[i]; 3375 3376 if (e->is_reg) 3377 bpf_bt_set_frame_reg(bt, e->frameno, e->regno); 3378 else 3379 bpf_bt_set_frame_slot(bt, e->frameno, e->spi); 3380 } 3381 } 3382 3383 int mark_chain_precision(struct bpf_verifier_env *env, int regno) 3384 { 3385 return bpf_mark_chain_precision(env, env->cur_state, regno, NULL); 3386 } 3387 3388 /* mark_chain_precision_batch() assumes that env->bt is set in the caller to 3389 * desired reg and stack masks across all relevant frames 3390 */ 3391 static int mark_chain_precision_batch(struct bpf_verifier_env *env, 3392 struct bpf_verifier_state *starting_state) 3393 { 3394 return bpf_mark_chain_precision(env, starting_state, -1, NULL); 3395 } 3396 3397 /* check if register is a constant scalar value */ 3398 static bool is_reg_const(struct bpf_reg_state *reg, bool subreg32) 3399 { 3400 return reg->type == SCALAR_VALUE && 3401 tnum_is_const(subreg32 ? tnum_subreg(reg->var_off) : reg->var_off); 3402 } 3403 3404 /* assuming is_reg_const() is true, return constant value of a register */ 3405 static u64 reg_const_value(struct bpf_reg_state *reg, bool subreg32) 3406 { 3407 return subreg32 ? tnum_subreg(reg->var_off).value : reg->var_off.value; 3408 } 3409 3410 static bool is_pointer_regtype(enum bpf_reg_type type) 3411 { 3412 return type != SCALAR_VALUE && type != NOT_INIT; 3413 } 3414 3415 static bool __is_pointer_value(bool allow_ptr_leaks, 3416 const struct bpf_reg_state *reg) 3417 { 3418 if (allow_ptr_leaks) 3419 return false; 3420 3421 return is_pointer_regtype(reg->type); 3422 } 3423 3424 static void clear_scalar_id(struct bpf_reg_state *reg) 3425 { 3426 reg->id = 0; 3427 reg->delta = 0; 3428 } 3429 3430 static void assign_scalar_id_before_mov(struct bpf_verifier_env *env, 3431 struct bpf_reg_state *src_reg) 3432 { 3433 if (src_reg->type != SCALAR_VALUE) 3434 return; 3435 /* 3436 * The verifier is processing rX = rY insn and 3437 * rY->id has special linked register already. 3438 * Cleared it, since multiple rX += const are not supported. 3439 */ 3440 if (src_reg->id & BPF_ADD_CONST) 3441 clear_scalar_id(src_reg); 3442 /* 3443 * Ensure that src_reg has a valid ID that will be copied to 3444 * dst_reg and then will be used by sync_linked_regs() to 3445 * propagate min/max range. 3446 */ 3447 if (!src_reg->id && !tnum_is_const(src_reg->var_off)) 3448 src_reg->id = ++env->id_gen; 3449 } 3450 3451 static void save_register_state(struct bpf_verifier_env *env, 3452 struct bpf_func_state *state, 3453 int spi, struct bpf_reg_state *reg, 3454 int size) 3455 { 3456 int i; 3457 3458 bpf_diag_mod_begin(env, &state->stack[spi].spilled_ptr, reg, BPF_DIAG_MOD_SPILL); 3459 state->stack[spi].spilled_ptr = *reg; 3460 3461 for (i = BPF_REG_SIZE; i > BPF_REG_SIZE - size; i--) 3462 state->stack[spi].slot_type[i - 1] = STACK_SPILL; 3463 3464 /* size < 8 bytes spill */ 3465 for (; i; i--) 3466 mark_stack_slot_misc(env, &state->stack[spi].slot_type[i - 1]); 3467 3468 bpf_diag_mod_end(env); 3469 } 3470 3471 static bool is_bpf_st_mem(struct bpf_insn *insn) 3472 { 3473 return BPF_CLASS(insn->code) == BPF_ST && BPF_MODE(insn->code) == BPF_MEM; 3474 } 3475 3476 static int get_reg_width(struct bpf_reg_state *reg) 3477 { 3478 return fls64(reg_umax(reg)); 3479 } 3480 3481 /* See comment for mark_fastcall_pattern_for_call() */ 3482 static void check_fastcall_stack_contract(struct bpf_verifier_env *env, 3483 struct bpf_func_state *state, int insn_idx, int off) 3484 { 3485 struct bpf_subprog_info *subprog = &env->subprog_info[state->subprogno]; 3486 struct bpf_insn_aux_data *aux = env->insn_aux_data; 3487 int i; 3488 3489 if (subprog->fastcall_stack_off <= off || aux[insn_idx].fastcall_pattern) 3490 return; 3491 /* access to the region [max_stack_depth .. fastcall_stack_off) 3492 * from something that is not a part of the fastcall pattern, 3493 * disable fastcall rewrites for current subprogram by setting 3494 * fastcall_stack_off to a value smaller than any possible offset. 3495 */ 3496 subprog->fastcall_stack_off = S16_MIN; 3497 /* reset fastcall aux flags within subprogram, 3498 * happens at most once per subprogram 3499 */ 3500 for (i = subprog->start; i < (subprog + 1)->start; ++i) { 3501 aux[i].fastcall_spills_num = 0; 3502 aux[i].fastcall_pattern = 0; 3503 } 3504 } 3505 3506 static void scrub_special_slot(struct bpf_func_state *state, int spi) 3507 { 3508 int i; 3509 3510 /* regular write of data into stack destroys any spilled ptr */ 3511 state->stack[spi].spilled_ptr.type = NOT_INIT; 3512 /* Mark slots as STACK_MISC if they belonged to spilled ptr/dynptr/iter. */ 3513 if (is_stack_slot_special(&state->stack[spi])) 3514 for (i = 0; i < BPF_REG_SIZE; i++) 3515 scrub_spilled_slot(&state->stack[spi].slot_type[i]); 3516 } 3517 3518 /* check_stack_{read,write}_fixed_off functions track spill/fill of registers, 3519 * stack boundary and alignment are checked in check_mem_access() 3520 */ 3521 static int check_stack_write_fixed_off(struct bpf_verifier_env *env, 3522 /* stack frame we're writing to */ 3523 struct bpf_func_state *state, 3524 int off, int size, int value_regno, 3525 int insn_idx) 3526 { 3527 struct bpf_func_state *cur; /* state of the current function */ 3528 int i, slot = -off - 1, spi = slot / BPF_REG_SIZE, err; 3529 struct bpf_insn *insn = &env->prog->insnsi[insn_idx]; 3530 struct bpf_reg_state *reg = NULL; 3531 int insn_flags = INSN_F_STACK_ACCESS; 3532 int hist_spi = spi, hist_frame = state->frameno; 3533 3534 /* caller checked that off % size == 0 and -MAX_BPF_STACK <= off < 0, 3535 * so it's aligned access and [off, off + size) are within stack limits 3536 */ 3537 if (!env->allow_ptr_leaks && 3538 bpf_is_spilled_reg(&state->stack[spi]) && 3539 !bpf_is_spilled_scalar_reg(&state->stack[spi]) && 3540 size != BPF_REG_SIZE) { 3541 const char *reason; 3542 3543 verbose(env, "attempt to corrupt spilled pointer on stack\n"); 3544 reason = bpf_diag_fmt(env, 3545 "This store writes %d bytes at stack offset %d into a stack slot that currently holds a spilled pointer. " 3546 "Partial writes to spilled pointers are rejected because they can corrupt pointer metadata and leak kernel pointers.", 3547 size, off); 3548 bpf_diag_memory( 3549 env, insn_idx, "stack spill corruption", reason, 3550 "Write the full 8-byte spilled pointer slot, or use a separate stack slot for scalar data before overwriting only part of it."); 3551 return -EACCES; 3552 } 3553 3554 cur = env->cur_state->frame[env->cur_state->curframe]; 3555 if (value_regno >= 0) 3556 reg = &cur->regs[value_regno]; 3557 if (!env->bypass_spec_v4) { 3558 bool sanitize = reg && is_pointer_regtype(reg->type); 3559 3560 for (i = 0; i < size; i++) { 3561 u8 type = state->stack[spi].slot_type[(slot - i) % 3562 BPF_REG_SIZE]; 3563 3564 if (type != STACK_MISC && type != STACK_ZERO) { 3565 sanitize = true; 3566 break; 3567 } 3568 } 3569 3570 if (sanitize) 3571 env->insn_aux_data[insn_idx].nospec_result = true; 3572 } 3573 3574 err = destroy_if_dynptr_stack_slot(env, state, spi); 3575 if (err) 3576 return err; 3577 3578 check_fastcall_stack_contract(env, state, insn_idx, off); 3579 mark_stack_slot_scratched(env, spi); 3580 if (reg && !(off % BPF_REG_SIZE) && reg->type == SCALAR_VALUE && env->bpf_capable) { 3581 bool reg_value_fits; 3582 3583 reg_value_fits = get_reg_width(reg) <= BITS_PER_BYTE * size; 3584 /* Make sure that reg had an ID to build a relation on spill. */ 3585 if (reg_value_fits) 3586 assign_scalar_id_before_mov(env, reg); 3587 save_register_state(env, state, spi, reg, size); 3588 /* Break the relation on a narrowing spill. */ 3589 if (!reg_value_fits) 3590 clear_scalar_id(&state->stack[spi].spilled_ptr); 3591 } else if (!reg && !(off % BPF_REG_SIZE) && is_bpf_st_mem(insn) && 3592 env->bpf_capable) { 3593 struct bpf_reg_state *tmp_reg = &env->fake_reg[0]; 3594 3595 memset(tmp_reg, 0, sizeof(*tmp_reg)); 3596 __mark_reg_known(tmp_reg, insn->imm); 3597 tmp_reg->type = SCALAR_VALUE; 3598 save_register_state(env, state, spi, tmp_reg, size); 3599 } else if (reg && is_pointer_regtype(reg->type)) { 3600 /* register containing pointer is being spilled into stack */ 3601 if (size != BPF_REG_SIZE) { 3602 verbose_linfo(env, insn_idx, "; "); 3603 verbose(env, "invalid size of register spill\n"); 3604 return -EACCES; 3605 } 3606 if (state != cur && reg->type == PTR_TO_STACK) { 3607 verbose(env, "cannot spill pointers to stack into stack frame of the caller\n"); 3608 return -EINVAL; 3609 } 3610 save_register_state(env, state, spi, reg, size); 3611 } else { 3612 u8 type = STACK_MISC; 3613 3614 if (bpf_is_spilled_reg(&state->stack[spi])) 3615 bpf_diag_record_scrub(env, &state->stack[spi].spilled_ptr, 3616 BPF_DIAG_MOD_WRITE); 3617 scrub_special_slot(state, spi); 3618 3619 /* when we zero initialize stack slots mark them as such */ 3620 if ((reg && bpf_register_is_null(reg)) || 3621 (!reg && is_bpf_st_mem(insn) && insn->imm == 0)) { 3622 /* STACK_ZERO case happened because register spill 3623 * wasn't properly aligned at the stack slot boundary, 3624 * so it's not a register spill anymore; force 3625 * originating register to be precise to make 3626 * STACK_ZERO correct for subsequent states 3627 */ 3628 err = mark_chain_precision(env, value_regno); 3629 if (err) 3630 return err; 3631 type = STACK_ZERO; 3632 } 3633 3634 /* Mark slots affected by this stack write. */ 3635 for (i = 0; i < size; i++) 3636 state->stack[spi].slot_type[(slot - i) % BPF_REG_SIZE] = type; 3637 insn_flags = 0; /* not a register spill */ 3638 } 3639 3640 if (insn_flags) 3641 return bpf_push_jmp_history(env, env->cur_state, insn_flags, 3642 hist_spi, hist_frame, 0); 3643 return 0; 3644 } 3645 3646 /* Write the stack: 'stack[ptr_reg + off] = value_regno'. 'ptr_reg' is 3647 * known to contain a variable offset. 3648 * This function checks whether the write is permitted and conservatively 3649 * tracks the effects of the write, considering that each stack slot in the 3650 * dynamic range is potentially written to. 3651 * 3652 * 'value_regno' can be -1, meaning that an unknown value is being written to 3653 * the stack. 3654 * 3655 * Spilled pointers in range are not marked as written because we don't know 3656 * what's going to be actually written. This means that read propagation for 3657 * future reads cannot be terminated by this write. 3658 * 3659 * For privileged programs, uninitialized stack slots are considered 3660 * initialized by this write (even though we don't know exactly what offsets 3661 * are going to be written to). The idea is that we don't want the verifier to 3662 * reject future reads that access slots written to through variable offsets. 3663 */ 3664 static int check_stack_write_var_off(struct bpf_verifier_env *env, 3665 /* func where register points to */ 3666 struct bpf_func_state *state, 3667 struct bpf_reg_state *ptr_reg, int off, int size, 3668 int value_regno, int insn_idx) 3669 { 3670 struct bpf_func_state *cur; /* state of the current function */ 3671 int min_off, max_off; 3672 int i, err; 3673 struct bpf_reg_state *value_reg = NULL; 3674 struct bpf_insn *insn = &env->prog->insnsi[insn_idx]; 3675 bool writing_zero = false; 3676 /* set if the fact that we're writing a zero is used to let any 3677 * stack slots remain STACK_ZERO 3678 */ 3679 bool zero_used = false; 3680 3681 cur = env->cur_state->frame[env->cur_state->curframe]; 3682 min_off = reg_smin(ptr_reg) + off; 3683 max_off = reg_smax(ptr_reg) + off + size; 3684 if (value_regno >= 0) 3685 value_reg = &cur->regs[value_regno]; 3686 if ((value_reg && bpf_register_is_null(value_reg)) || 3687 (!value_reg && is_bpf_st_mem(insn) && insn->imm == 0)) 3688 writing_zero = true; 3689 3690 for (i = min_off; i < max_off; i++) { 3691 int spi; 3692 3693 spi = bpf_get_spi(i); 3694 err = destroy_if_dynptr_stack_slot(env, state, spi); 3695 if (err) 3696 return err; 3697 } 3698 3699 check_fastcall_stack_contract(env, state, insn_idx, min_off); 3700 /* Variable offset writes destroy any spilled pointers in range. */ 3701 for (i = min_off; i < max_off; i++) { 3702 u8 new_type, *stype; 3703 int slot, spi; 3704 3705 slot = -i - 1; 3706 spi = slot / BPF_REG_SIZE; 3707 stype = &state->stack[spi].slot_type[slot % BPF_REG_SIZE]; 3708 mark_stack_slot_scratched(env, spi); 3709 3710 if (!env->allow_ptr_leaks && *stype != STACK_MISC && *stype != STACK_ZERO) { 3711 /* Reject the write if range we may write to has not 3712 * been initialized beforehand. If we didn't reject 3713 * here, the ptr status would be erased below (even 3714 * though not all slots are actually overwritten), 3715 * possibly opening the door to leaks. 3716 * 3717 * We do however catch STACK_INVALID case below, and 3718 * only allow reading possibly uninitialized memory 3719 * later for CAP_PERFMON, as the write may not happen to 3720 * that slot. 3721 */ 3722 verbose(env, "spilled ptr in range of var-offset stack write; insn %d, ptr off: %d", 3723 insn_idx, i); 3724 return -EINVAL; 3725 } 3726 3727 /* If writing_zero and the spi slot contains a spill of value 0, 3728 * maintain the spill type. 3729 */ 3730 if (writing_zero && *stype == STACK_SPILL && 3731 bpf_is_spilled_scalar_reg(&state->stack[spi])) { 3732 struct bpf_reg_state *spill_reg = &state->stack[spi].spilled_ptr; 3733 3734 if (tnum_is_const(spill_reg->var_off) && spill_reg->var_off.value == 0) { 3735 zero_used = true; 3736 continue; 3737 } 3738 } 3739 3740 /* 3741 * Scrub slots if variable-offset stack write goes over spilled pointers. 3742 * Otherwise bpf_is_spilled_reg() may == true && spilled_ptr.type == NOT_INIT 3743 * and valid program is rejected by check_stack_read_fixed_off() 3744 * with obscure "invalid size of register fill" message. 3745 */ 3746 scrub_special_slot(state, spi); 3747 3748 /* Update the slot type. */ 3749 new_type = STACK_MISC; 3750 if (writing_zero && *stype == STACK_ZERO) { 3751 new_type = STACK_ZERO; 3752 zero_used = true; 3753 } 3754 /* If the slot is STACK_INVALID, we check whether it's OK to 3755 * pretend that it will be initialized by this write. The slot 3756 * might not actually be written to, and so if we mark it as 3757 * initialized future reads might leak uninitialized memory. 3758 * For privileged programs, we will accept such reads to slots 3759 * that may or may not be written because, if we're reject 3760 * them, the error would be too confusing. 3761 * Conservatively, treat STACK_POISON in a similar way. 3762 */ 3763 if ((*stype == STACK_INVALID || *stype == STACK_POISON) && 3764 !env->allow_uninit_stack) { 3765 verbose(env, "uninit stack in range of var-offset write prohibited for !root; insn %d, off: %d", 3766 insn_idx, i); 3767 return -EINVAL; 3768 } 3769 *stype = new_type; 3770 } 3771 if (zero_used) { 3772 /* backtracking doesn't work for STACK_ZERO yet. */ 3773 err = mark_chain_precision(env, value_regno); 3774 if (err) 3775 return err; 3776 } 3777 bpf_diag_record_scrub_stack(env, state, min_off, max_off, 3778 BPF_DIAG_MOD_VAR_WRITE); 3779 return 0; 3780 } 3781 3782 /* When register 'dst_regno' is assigned some values from stack[min_off, 3783 * max_off), we set the register's type according to the types of the 3784 * respective stack slots. If all the stack values are known to be zeros, then 3785 * so is the destination reg. Otherwise, the register is considered to be 3786 * SCALAR. This function does not deal with register filling; the caller must 3787 * ensure that all spilled registers in the stack range have been marked as 3788 * read. 3789 * 3790 * STACK_SPILL bytes backed by spilled scalar const zeroes are also considered 3791 * zero bytes. In that case, mark the contributing stack slots precise so 3792 * pruning cannot reuse a zero-spill state for a later non-zero spill state. 3793 * 3794 * Returns an error if precision backtracking fails. 3795 */ 3796 static int mark_reg_stack_read(struct bpf_verifier_env *env, 3797 /* func where src register points to */ 3798 struct bpf_func_state *ptr_state, 3799 int min_off, int max_off, int dst_regno) 3800 { 3801 struct bpf_verifier_state *vstate = env->cur_state; 3802 struct bpf_func_state *state = vstate->frame[vstate->curframe]; 3803 u64 zero_spill_mask = 0; 3804 int i, slot, spi; 3805 u8 *stype; 3806 int zeros = 0; 3807 3808 for (i = min_off; i < max_off; i++) { 3809 slot = -i - 1; 3810 spi = slot / BPF_REG_SIZE; 3811 mark_stack_slot_scratched(env, spi); 3812 stype = ptr_state->stack[spi].slot_type; 3813 if (stype[slot % BPF_REG_SIZE] == STACK_ZERO) { 3814 zeros++; 3815 continue; 3816 } 3817 if (stype[slot % BPF_REG_SIZE] == STACK_SPILL && 3818 bpf_register_is_null(&ptr_state->stack[spi].spilled_ptr)) { 3819 zero_spill_mask |= 1ull << spi; 3820 zeros++; 3821 continue; 3822 } 3823 break; 3824 } 3825 if (zeros == max_off - min_off) { 3826 /* Any access_size read into register is zero extended, 3827 * so the whole register == const_zero. 3828 */ 3829 __mark_reg_const_zero(env, &state->regs[dst_regno]); 3830 if (zero_spill_mask) { 3831 bpf_bt_set_frame_slot_mask(&env->bt, ptr_state->frameno, zero_spill_mask); 3832 return mark_chain_precision_batch(env, env->cur_state); 3833 } 3834 } else { 3835 /* have read misc data from the stack */ 3836 mark_reg_unknown(env, state->regs, dst_regno); 3837 } 3838 3839 return 0; 3840 } 3841 3842 static void bpf_diag_stack_read_uninit(struct bpf_verifier_env *env, int off, int i, 3843 int size) 3844 { 3845 const char *reason; 3846 3847 reason = bpf_diag_fmt(env, 3848 "This rejected read uses %d bytes at stack offset %d, but byte %d in that range is uninitialized on this path. " 3849 "Programs loaded with CAP_PERFMON can be allowed to read uninitialized stack bytes, but this program is being rejected without that allowance.", 3850 size, off, i); 3851 bpf_diag_memory( 3852 env, env->insn_idx, "uninitialized stack read", reason, 3853 "Initialize every byte in the stack range before reading it, adjust the offset and size so the read covers only initialized bytes, " 3854 "or load with CAP_PERFMON if uninitialized stack reads are intended."); 3855 } 3856 3857 /* Read the stack at 'off' and put the results into the register indicated by 3858 * 'dst_regno'. It handles reg filling if the addressed stack slot is a 3859 * spilled reg. 3860 * 3861 * 'dst_regno' can be -1, meaning that the read value is not going to a 3862 * register. 3863 * 3864 * The access is assumed to be within the current stack bounds. 3865 */ 3866 static int check_stack_read_fixed_off(struct bpf_verifier_env *env, 3867 /* func where src register points to */ 3868 struct bpf_func_state *reg_state, 3869 int off, int size, int dst_regno) 3870 { 3871 struct bpf_verifier_state *vstate = env->cur_state; 3872 struct bpf_func_state *state = vstate->frame[vstate->curframe]; 3873 int i, slot = -off - 1, spi = slot / BPF_REG_SIZE; 3874 struct bpf_reg_state *reg; 3875 u8 *stype, type; 3876 int err; 3877 int insn_flags = INSN_F_STACK_ACCESS; 3878 int hist_spi = spi, hist_frame = reg_state->frameno; 3879 3880 stype = reg_state->stack[spi].slot_type; 3881 reg = ®_state->stack[spi].spilled_ptr; 3882 3883 mark_stack_slot_scratched(env, spi); 3884 check_fastcall_stack_contract(env, state, env->insn_idx, off); 3885 3886 /* 3887 * Refine the in-progress load record's origin to the source stack slot. 3888 */ 3889 if (dst_regno >= 0) 3890 bpf_diag_mod_begin(env, &state->regs[dst_regno], reg, BPF_DIAG_MOD_WRITE); 3891 3892 if (bpf_is_spilled_reg(®_state->stack[spi])) { 3893 u8 spill_size = 1; 3894 3895 for (i = BPF_REG_SIZE - 1; i > 0 && stype[i - 1] == STACK_SPILL; i--) 3896 spill_size++; 3897 3898 if (size != BPF_REG_SIZE || spill_size != BPF_REG_SIZE) { 3899 if (reg->type != SCALAR_VALUE) { 3900 verbose_linfo(env, env->insn_idx, "; "); 3901 verbose(env, "invalid size of register fill\n"); 3902 return -EACCES; 3903 } 3904 3905 if (dst_regno < 0) 3906 return 0; 3907 3908 if (size <= spill_size && 3909 bpf_stack_narrow_access_ok(off, size, spill_size)) { 3910 if (env->bpf_capable && size == 4 && spill_size == 4 && 3911 get_reg_width(reg) <= 32) 3912 /* Ensure stack slot has an ID to build a relation 3913 * with the destination register on fill. 3914 */ 3915 assign_scalar_id_before_mov(env, reg); 3916 state->regs[dst_regno] = *reg; 3917 3918 /* Break the relation on a narrowing fill. 3919 * coerce_reg_to_size will adjust the boundaries. 3920 */ 3921 if (get_reg_width(reg) > size * BITS_PER_BYTE) 3922 clear_scalar_id(&state->regs[dst_regno]); 3923 } else { 3924 int spill_cnt = 0, zero_cnt = 0; 3925 3926 for (i = 0; i < size; i++) { 3927 type = stype[(slot - i) % BPF_REG_SIZE]; 3928 if (type == STACK_SPILL) { 3929 spill_cnt++; 3930 continue; 3931 } 3932 if (type == STACK_MISC) 3933 continue; 3934 if (type == STACK_ZERO) { 3935 zero_cnt++; 3936 continue; 3937 } 3938 if (type == STACK_INVALID && env->allow_uninit_stack) 3939 continue; 3940 if (type == STACK_POISON) { 3941 verbose(env, "reading from stack off %d+%d size %d, slot poisoned by dead code elimination\n", 3942 off, i, size); 3943 } else { 3944 verbose(env, "invalid read from stack off %d+%d size %d\n", 3945 off, i, size); 3946 bpf_diag_stack_read_uninit(env, off, i, size); 3947 } 3948 return -EACCES; 3949 } 3950 3951 if (spill_cnt == size && 3952 tnum_is_const(reg->var_off) && reg->var_off.value == 0) { 3953 __mark_reg_const_zero(env, &state->regs[dst_regno]); 3954 /* this IS register fill, so keep insn_flags */ 3955 } else if (zero_cnt == size) { 3956 /* similarly to mark_reg_stack_read(), preserve zeroes */ 3957 __mark_reg_const_zero(env, &state->regs[dst_regno]); 3958 insn_flags = 0; /* not restoring original register state */ 3959 } else { 3960 err = mark_reg_stack_read(env, reg_state, off, off + size, 3961 dst_regno); 3962 if (err) 3963 return err; 3964 insn_flags = 0; /* not restoring original register state */ 3965 } 3966 } 3967 } else if (dst_regno >= 0) { 3968 /* restore register state from stack */ 3969 if (env->bpf_capable) 3970 /* Ensure stack slot has an ID to build a relation 3971 * with the destination register on fill. 3972 */ 3973 assign_scalar_id_before_mov(env, reg); 3974 state->regs[dst_regno] = *reg; 3975 /* mark reg as written since spilled pointer state likely 3976 * has its liveness marks cleared by is_state_visited() 3977 * which resets stack/reg liveness for state transitions 3978 */ 3979 } else if (__is_pointer_value(env->allow_ptr_leaks, reg)) { 3980 /* If dst_regno==-1, the caller is asking us whether 3981 * it is acceptable to use this value as a SCALAR_VALUE 3982 * (e.g. for XADD). 3983 * We must not allow unprivileged callers to do that 3984 * with spilled pointers. 3985 */ 3986 verbose(env, "leaking pointer from stack off %d\n", 3987 off); 3988 return -EACCES; 3989 } 3990 } else { 3991 for (i = 0; i < size; i++) { 3992 type = stype[(slot - i) % BPF_REG_SIZE]; 3993 if (type == STACK_MISC) 3994 continue; 3995 if (type == STACK_ZERO) 3996 continue; 3997 if (type == STACK_INVALID && env->allow_uninit_stack) 3998 continue; 3999 if (type == STACK_POISON) { 4000 verbose(env, "reading from stack off %d+%d size %d, slot poisoned by dead code elimination\n", 4001 off, i, size); 4002 } else { 4003 verbose(env, "invalid read from stack off %d+%d size %d\n", 4004 off, i, size); 4005 bpf_diag_stack_read_uninit(env, off, i, size); 4006 } 4007 return -EACCES; 4008 } 4009 if (dst_regno >= 0) { 4010 err = mark_reg_stack_read(env, reg_state, off, off + size, dst_regno); 4011 if (err) 4012 return err; 4013 } 4014 insn_flags = 0; /* we are not restoring spilled register */ 4015 } 4016 if (insn_flags) 4017 return bpf_push_jmp_history(env, env->cur_state, insn_flags, 4018 hist_spi, hist_frame, 0); 4019 return 0; 4020 } 4021 4022 enum bpf_access_src { 4023 ACCESS_DIRECT = 1, /* the access is performed by an instruction */ 4024 ACCESS_HELPER = 2, /* the access is performed by a helper */ 4025 }; 4026 4027 static int check_stack_range_initialized(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 4028 argno_t argno, int off, int access_size, 4029 bool zero_size_allowed, 4030 enum bpf_access_type type, 4031 struct bpf_call_arg_meta *meta); 4032 4033 static struct bpf_reg_state *reg_state(struct bpf_verifier_env *env, int regno) 4034 { 4035 return cur_regs(env) + regno; 4036 } 4037 4038 /* Read the stack at 'reg + off' and put the result into the register 4039 * 'dst_regno'. 4040 * 'off' includes the pointer register's fixed offset(i.e. 'reg->off'), 4041 * but not its variable offset. 4042 * 'size' is assumed to be <= reg size and the access is assumed to be aligned. 4043 * 4044 * As opposed to check_stack_read_fixed_off, this function doesn't deal with 4045 * filling registers (i.e. reads of spilled register cannot be detected when 4046 * the offset is not fixed). We conservatively mark 'dst_regno' as containing 4047 * SCALAR_VALUE. That's why we assert that the 'reg' has a variable 4048 * offset; for a fixed offset check_stack_read_fixed_off should be used 4049 * instead. 4050 */ 4051 static int check_stack_read_var_off(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 4052 argno_t ptr_argno, int off, int size, int dst_regno) 4053 { 4054 struct bpf_func_state *ptr_state = bpf_func(env, reg); 4055 int err; 4056 int min_off, max_off; 4057 4058 /* Note that we pass a NULL meta, so raw access will not be permitted. 4059 */ 4060 err = check_stack_range_initialized(env, reg, ptr_argno, off, size, 4061 false, BPF_READ, NULL); 4062 if (err) 4063 return err; 4064 4065 min_off = reg_smin(reg) + off; 4066 max_off = reg_smax(reg) + off; 4067 err = mark_reg_stack_read(env, ptr_state, min_off, max_off + size, 4068 dst_regno); 4069 if (err) 4070 return err; 4071 check_fastcall_stack_contract(env, ptr_state, env->insn_idx, min_off); 4072 return 0; 4073 } 4074 4075 /* check_stack_read dispatches to check_stack_read_fixed_off or 4076 * check_stack_read_var_off. 4077 * 4078 * The caller must ensure that the offset falls within the allocated stack 4079 * bounds. 4080 * 4081 * 'dst_regno' is a register which will receive the value from the stack. It 4082 * can be -1, meaning that the read value is not going to a register. 4083 */ 4084 static int check_stack_read(struct bpf_verifier_env *env, 4085 struct bpf_reg_state *reg, argno_t ptr_argno, int off, int size, 4086 int dst_regno) 4087 { 4088 struct bpf_func_state *state = bpf_func(env, reg); 4089 int err; 4090 /* Some accesses are only permitted with a static offset. */ 4091 bool var_off = !tnum_is_const(reg->var_off); 4092 4093 /* The offset is required to be static when reads don't go to a 4094 * register, in order to not leak pointers (see 4095 * check_stack_read_fixed_off). 4096 */ 4097 if (dst_regno < 0 && var_off) { 4098 const char *reason; 4099 char tn_buf[48]; 4100 4101 tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off); 4102 verbose(env, "variable offset stack pointer cannot be passed into helper function; var_off=%s off=%d size=%d\n", 4103 tn_buf, off, size); 4104 reason = bpf_diag_fmt(env, 4105 "The helper would access the stack through variable offset %s plus fixed offset %d and size %d. " 4106 "Helper stack memory arguments require a constant stack offset and a precise initialized range.", 4107 tn_buf, off, size); 4108 bpf_diag_memory( 4109 env, env->insn_idx, "variable stack access", reason, 4110 "Use a fixed stack offset for helper memory arguments, or copy the needed bytes into a fixed stack slot first."); 4111 return -EACCES; 4112 } 4113 /* Variable offset is prohibited for unprivileged mode for simplicity 4114 * since it requires corresponding support in Spectre masking for stack 4115 * ALU. See also retrieve_ptr_limit(). The check in 4116 * check_stack_access_for_ptr_arithmetic() called by 4117 * adjust_ptr_min_max_vals() prevents users from creating stack pointers 4118 * with variable offsets, therefore no check is required here. Further, 4119 * just checking it here would be insufficient as speculative stack 4120 * writes could still lead to unsafe speculative behaviour. 4121 */ 4122 if (!var_off) { 4123 off += reg->var_off.value; 4124 err = check_stack_read_fixed_off(env, state, off, size, 4125 dst_regno); 4126 } else { 4127 /* Variable offset stack reads need more conservative handling 4128 * than fixed offset ones. Note that dst_regno >= 0 on this 4129 * branch. 4130 */ 4131 err = check_stack_read_var_off(env, reg, ptr_argno, off, size, 4132 dst_regno); 4133 } 4134 return err; 4135 } 4136 4137 /* check_stack_write dispatches to check_stack_write_fixed_off or 4138 * check_stack_write_var_off. 4139 * 4140 * 'reg' is the register used as a pointer into the stack. 4141 * 'value_regno' is the register whose value we're writing to the stack. It can 4142 * be -1, meaning that we're not writing from a register. 4143 * 4144 * The caller must ensure that the offset falls within the maximum stack size. 4145 */ 4146 static int check_stack_write(struct bpf_verifier_env *env, 4147 struct bpf_reg_state *reg, int off, int size, 4148 int value_regno, int insn_idx) 4149 { 4150 struct bpf_func_state *state = bpf_func(env, reg); 4151 int err; 4152 4153 if (tnum_is_const(reg->var_off)) { 4154 off += reg->var_off.value; 4155 err = check_stack_write_fixed_off(env, state, off, size, 4156 value_regno, insn_idx); 4157 } else { 4158 /* Variable offset stack reads need more conservative handling 4159 * than fixed offset ones. 4160 */ 4161 err = check_stack_write_var_off(env, state, 4162 reg, off, size, 4163 value_regno, insn_idx); 4164 } 4165 return err; 4166 } 4167 4168 /* 4169 * Write a value to the outgoing stack arg area. 4170 * off is a negative offset from r11 (e.g. -8 for arg6, -16 for arg7). 4171 */ 4172 static int check_stack_arg_write(struct bpf_verifier_env *env, struct bpf_func_state *state, 4173 int off, struct bpf_reg_state *value_reg) 4174 { 4175 int max_stack_arg_regs = MAX_BPF_FUNC_ARGS - MAX_BPF_FUNC_REG_ARGS; 4176 struct bpf_subprog_info *subprog = &env->subprog_info[state->subprogno]; 4177 int spi = -off / BPF_REG_SIZE - 1; 4178 struct bpf_reg_state *arg; 4179 int err; 4180 4181 if (spi >= max_stack_arg_regs) { 4182 verbose(env, "stack arg write offset %d exceeds max %d stack args\n", 4183 off, max_stack_arg_regs); 4184 return -EINVAL; 4185 } 4186 4187 err = grow_stack_arg_slots(env, state, spi + 1); 4188 if (err) 4189 return err; 4190 4191 /* Track the max outgoing stack arg slot count. */ 4192 if (spi + 1 > subprog->max_out_stack_arg_cnt) 4193 subprog->max_out_stack_arg_cnt = spi + 1; 4194 4195 arg = &state->stack_arg_regs[spi]; 4196 bpf_diag_mod_begin(env, arg, value_reg, BPF_DIAG_MOD_WRITE); 4197 4198 if (value_reg) { 4199 state->stack_arg_regs[spi] = *value_reg; 4200 } else { 4201 /* BPF_ST: store immediate, treat as scalar */ 4202 arg->type = SCALAR_VALUE; 4203 __mark_reg_known(arg, env->prog->insnsi[env->insn_idx].imm); 4204 } 4205 bpf_diag_mod_end(env); 4206 state->no_stack_arg_load = true; 4207 return bpf_push_jmp_history(env, env->cur_state, 4208 INSN_F_STACK_ARG_ACCESS, spi, 0, 0); 4209 } 4210 4211 /* 4212 * Read a value from the incoming stack arg area. 4213 * off is a positive offset from r11 (e.g. +8 for arg6, +16 for arg7). 4214 */ 4215 static int check_stack_arg_read(struct bpf_verifier_env *env, struct bpf_func_state *state, 4216 int off, int dst_regno) 4217 { 4218 struct bpf_subprog_info *subprog = &env->subprog_info[state->subprogno]; 4219 struct bpf_verifier_state *vstate = env->cur_state; 4220 int spi = off / BPF_REG_SIZE - 1; 4221 struct bpf_func_state *caller, *cur; 4222 struct bpf_reg_state *arg; 4223 4224 if (state->no_stack_arg_load) { 4225 verbose(env, "r11 load must be before any r11 store or call insn\n"); 4226 return -EINVAL; 4227 } 4228 4229 if (spi + 1 > bpf_in_stack_arg_cnt(subprog)) { 4230 verbose(env, "invalid read from stack arg off %d depth %d\n", 4231 off, bpf_in_stack_arg_cnt(subprog) * BPF_REG_SIZE); 4232 return -EACCES; 4233 } 4234 4235 caller = vstate->frame[vstate->curframe - 1]; 4236 arg = &caller->stack_arg_regs[spi]; 4237 cur = vstate->frame[vstate->curframe]; 4238 bpf_diag_mod_begin(env, &cur->regs[dst_regno], arg, BPF_DIAG_MOD_WRITE); 4239 cur->regs[dst_regno] = *arg; 4240 bpf_diag_mod_end(env); 4241 return bpf_push_jmp_history(env, env->cur_state, 4242 INSN_F_STACK_ARG_ACCESS, spi, 0, 0); 4243 } 4244 4245 static int mark_stack_arg_precision(struct bpf_verifier_env *env, int arg_idx) 4246 { 4247 struct bpf_func_state *caller = cur_func(env); 4248 int spi = arg_idx - MAX_BPF_FUNC_REG_ARGS; 4249 4250 bt_set_frame_stack_arg_slot(&env->bt, caller->frameno, spi); 4251 return mark_chain_precision_batch(env, env->cur_state); 4252 } 4253 4254 static int mark_arg_precision(struct bpf_verifier_env *env, argno_t argno) 4255 { 4256 int regno = reg_from_argno(argno); 4257 4258 if (regno >= 0) 4259 return mark_chain_precision(env, regno); 4260 return mark_stack_arg_precision(env, arg_idx_from_argno(argno)); 4261 } 4262 4263 static int check_outgoing_stack_args(struct bpf_verifier_env *env, struct bpf_func_state *caller, 4264 int nargs, const char *callee_name, const struct btf *btf, 4265 const struct btf_param *args) 4266 { 4267 int i, spi; 4268 4269 for (i = MAX_BPF_FUNC_REG_ARGS; i < nargs; i++) { 4270 spi = i - MAX_BPF_FUNC_REG_ARGS; 4271 if (spi >= caller->out_stack_arg_cnt || 4272 caller->stack_arg_regs[spi].type == NOT_INIT) { 4273 const char *arg_name = NULL; 4274 4275 if (args && args[i].name_off) 4276 arg_name = btf_name_by_offset(btf, args[i].name_off); 4277 verbose(env, "callee expects %d args, stack arg%d is not initialized\n", 4278 nargs, spi + 1); 4279 bpf_diag_stack_arg_uninit(env, env->insn_idx, nargs, spi, 4280 callee_name, arg_name); 4281 return -EFAULT; 4282 } 4283 } 4284 4285 return 0; 4286 } 4287 4288 static struct bpf_reg_state *get_func_arg_reg(struct bpf_func_state *caller, 4289 struct bpf_reg_state *regs, int arg) 4290 { 4291 if (arg < MAX_BPF_FUNC_REG_ARGS) 4292 return ®s[arg + 1]; 4293 4294 return &caller->stack_arg_regs[arg - MAX_BPF_FUNC_REG_ARGS]; 4295 } 4296 4297 static int check_map_access_type(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 4298 int off, int size, enum bpf_access_type type) 4299 { 4300 struct bpf_map *map = reg->map_ptr; 4301 u32 cap = bpf_map_flags_to_cap(map); 4302 4303 if (type == BPF_WRITE && !(cap & BPF_MAP_CAN_WRITE)) { 4304 verbose(env, "write into map forbidden, value_size=%d off=%lld size=%d\n", 4305 map->value_size, reg_smin(reg) + off, size); 4306 return -EACCES; 4307 } 4308 4309 if (type == BPF_READ && !(cap & BPF_MAP_CAN_READ)) { 4310 verbose(env, "read from map forbidden, value_size=%d off=%lld size=%d\n", 4311 map->value_size, reg_smin(reg) + off, size); 4312 return -EACCES; 4313 } 4314 4315 return 0; 4316 } 4317 4318 /* check read/write into memory region (e.g., map value, ringbuf sample, etc) */ 4319 static int __check_mem_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, 4320 int off, int size, u32 mem_size, 4321 bool zero_size_allowed) 4322 { 4323 bool size_ok = size > 0 || (size == 0 && zero_size_allowed); 4324 4325 if (off >= 0 && size_ok && (u64)off + size <= mem_size) 4326 return 0; 4327 4328 switch (reg->type) { 4329 case PTR_TO_MAP_KEY: 4330 verbose(env, "invalid access to map key, key_size=%d off=%d size=%d\n", 4331 mem_size, off, size); 4332 break; 4333 case PTR_TO_MAP_VALUE: 4334 verbose(env, "invalid access to map value, value_size=%d off=%d size=%d\n", 4335 mem_size, off, size); 4336 break; 4337 case PTR_TO_PACKET: 4338 case PTR_TO_PACKET_META: 4339 case PTR_TO_PACKET_END: 4340 verbose(env, "invalid access to packet, off=%d size=%d, %s(id=%d,off=%d,r=%d)\n", 4341 off, size, reg_arg_name(env, argno), reg->id, off, mem_size); 4342 break; 4343 case PTR_TO_CTX: 4344 verbose(env, "invalid access to context, ctx_size=%d off=%d size=%d\n", 4345 mem_size, off, size); 4346 break; 4347 case PTR_TO_MEM: 4348 default: 4349 verbose(env, "invalid access to memory, mem_size=%u off=%d size=%d\n", 4350 mem_size, off, size); 4351 } 4352 4353 return -EACCES; 4354 } 4355 4356 /* check read/write into a memory region with possible variable offset */ 4357 static int check_mem_region_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, 4358 int off, int size, u32 mem_size, 4359 bool zero_size_allowed) 4360 { 4361 const char *proof = ""; 4362 const char *start; 4363 s64 max_start, max_end; 4364 int err; 4365 4366 /* We may have adjusted the register pointing to memory region, so we 4367 * need to try adding each of min_value and max_value to off 4368 * to make sure our theoretical access will be safe. 4369 * 4370 * The minimum value is only important with signed 4371 * comparisons where we can't assume the floor of a 4372 * value is 0. If we are using signed variables for our 4373 * index'es we need to make sure that whatever we use 4374 * will have a set floor within our range. 4375 */ 4376 if (reg_smin(reg) < 0 && 4377 (reg_smin(reg) == S64_MIN || 4378 (off + reg_smin(reg) != (s64)(s32)(off + reg_smin(reg))) || 4379 reg_smin(reg) + off < 0)) { 4380 verbose(env, "%s min value is negative, either use unsigned index or do a if (index >=0) check.\n", 4381 reg_arg_name(env, argno)); 4382 err = -EACCES; 4383 if (bpf_diag_enabled(env)) { 4384 start = bpf_diag_fmt_s64_sum(env, reg_smin(reg), off); 4385 proof = bpf_diag_fmt( 4386 env, "the minimal bound for a memory access is a negative value: %s", 4387 start); 4388 } 4389 goto report_error; 4390 } 4391 4392 err = __check_mem_access(env, reg, argno, reg_smin(reg) + off, size, 4393 mem_size, zero_size_allowed); 4394 if (err) { 4395 verbose(env, "%s min value is outside of the allowed memory range\n", 4396 reg_arg_name(env, argno)); 4397 if (bpf_diag_enabled(env)) { 4398 start = bpf_diag_fmt_s64_sum(env, reg_smin(reg), off); 4399 proof = bpf_diag_fmt( 4400 env, "the minimal bound for a memory access is %s and is outside of the object of size %u", 4401 start, mem_size); 4402 } 4403 goto report_error; 4404 } 4405 4406 /* If we haven't set a max value then we need to bail since we can't be 4407 * sure we won't do bad things. 4408 * If reg_umax(reg) + off could overflow, treat that as unbounded too. 4409 */ 4410 if (reg_umax(reg) >= BPF_MAX_VAR_OFF) { 4411 verbose(env, "%s unbounded memory access, make sure to bounds check any such access\n", 4412 reg_arg_name(env, argno)); 4413 err = -EACCES; 4414 if (bpf_diag_enabled(env)) 4415 proof = bpf_diag_fmt( 4416 env, "the maximal bound for a memory access is %llu and exceeds maximum allowed offset of %u", 4417 reg_umax(reg), BPF_MAX_VAR_OFF); 4418 goto report_error; 4419 } 4420 4421 err = __check_mem_access(env, reg, argno, reg_umax(reg) + off, size, 4422 mem_size, zero_size_allowed); 4423 if (err) { 4424 verbose(env, "%s max value is outside of the allowed memory range\n", 4425 reg_arg_name(env, argno)); 4426 if (bpf_diag_enabled(env)) { 4427 max_start = (s64)reg_umax(reg) + off; 4428 max_end = max_start + size; 4429 proof = bpf_diag_fmt( 4430 env, "the maximal bound for a memory access is %lld: start %lld + access_size %d, beyond object_size %u", 4431 max_end, max_start, size, mem_size); 4432 } 4433 goto report_error; 4434 } 4435 4436 return 0; 4437 4438 report_error: 4439 bpf_diag_mem_bounds(env, env->insn_idx, reg_from_argno(argno), 4440 reg_arg_name(env, argno), reg_type_str(env, reg->type), proof, 4441 off, size, mem_size, reg); 4442 return err; 4443 } 4444 4445 static int __check_ptr_off_reg(struct bpf_verifier_env *env, 4446 const struct bpf_reg_state *reg, argno_t argno, 4447 bool fixed_off_ok) 4448 { 4449 /* Access to this pointer-typed register or passing it to a helper 4450 * is only allowed in its original, unmodified form. 4451 */ 4452 4453 if (!tnum_is_const(reg->var_off)) { 4454 char tn_buf[48]; 4455 4456 tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off); 4457 verbose(env, "variable %s access var_off=%s disallowed\n", 4458 reg_type_str(env, reg->type), tn_buf); 4459 return -EACCES; 4460 } 4461 4462 if (reg_smin(reg) < 0) { 4463 verbose(env, "negative offset %s ptr %s off=%lld disallowed\n", 4464 reg_type_str(env, reg->type), reg_arg_name(env, argno), reg->var_off.value); 4465 return -EACCES; 4466 } 4467 4468 if (!fixed_off_ok && reg->var_off.value != 0) { 4469 verbose(env, "dereference of modified %s ptr %s off=%lld disallowed\n", 4470 reg_type_str(env, reg->type), reg_arg_name(env, argno), reg->var_off.value); 4471 bpf_diag_invalid_deref(env, env->insn_idx, reg_from_argno(argno), 4472 reg_arg_name(env, argno), reg, 4473 BPF_DIAG_DEREF_MODIFIED_PTR, reg->var_off.value); 4474 return -EACCES; 4475 } 4476 4477 return 0; 4478 } 4479 4480 static int check_ptr_off_reg(struct bpf_verifier_env *env, 4481 const struct bpf_reg_state *reg, int regno) 4482 { 4483 return __check_ptr_off_reg(env, reg, argno_from_reg(regno), false); 4484 } 4485 4486 static int map_kptr_match_type(struct bpf_verifier_env *env, 4487 struct btf_field *kptr_field, 4488 struct bpf_reg_state *reg, u32 regno) 4489 { 4490 const char *targ_name = btf_type_name(kptr_field->kptr.btf, kptr_field->kptr.btf_id); 4491 int perm_flags; 4492 const char *reg_name = ""; 4493 4494 if (base_type(reg->type) != PTR_TO_BTF_ID) 4495 goto bad_type; 4496 4497 if (btf_is_kernel(reg->btf)) { 4498 perm_flags = PTR_MAYBE_NULL | PTR_TRUSTED | MEM_RCU; 4499 4500 /* Only unreferenced case accepts untrusted pointers */ 4501 if (kptr_field->type == BPF_KPTR_UNREF) 4502 perm_flags |= PTR_UNTRUSTED; 4503 } else { 4504 perm_flags = PTR_MAYBE_NULL | MEM_ALLOC; 4505 if (kptr_field->type == BPF_KPTR_PERCPU) 4506 perm_flags |= MEM_PERCPU; 4507 } 4508 4509 if (type_flag(reg->type) & ~perm_flags) 4510 goto bad_type; 4511 4512 /* 4513 * A BPF_KPTR_PERCPU field is read back as MEM_PERCPU, so the value 4514 * stored in it must carry the same flag. 4515 */ 4516 if ((kptr_field->type == BPF_KPTR_PERCPU) != !!(reg->type & MEM_PERCPU)) 4517 goto bad_type; 4518 4519 /* We need to verify reg->type and reg->btf, before accessing reg->btf */ 4520 reg_name = btf_type_name(reg->btf, reg->btf_id); 4521 4522 /* For ref_ptr case, release function check should ensure we get one 4523 * referenced PTR_TO_BTF_ID, and that its fixed offset is 0. For the 4524 * normal store of unreferenced kptr, we must ensure var_off is zero. 4525 * Since ref_ptr cannot be accessed directly by BPF insns, check for 4526 * reg->id is not needed here. 4527 */ 4528 if (__check_ptr_off_reg(env, reg, argno_from_reg(regno), true)) 4529 return -EACCES; 4530 4531 /* A full type match is needed, as BTF can be vmlinux, module or prog BTF, and 4532 * we also need to take into account the reg->var_off. 4533 * 4534 * We want to support cases like: 4535 * 4536 * struct foo { 4537 * struct bar br; 4538 * struct baz bz; 4539 * }; 4540 * 4541 * struct foo *v; 4542 * v = func(); // PTR_TO_BTF_ID 4543 * val->foo = v; // reg->var_off is zero, btf and btf_id match type 4544 * val->bar = &v->br; // reg->var_off is still zero, but we need to retry with 4545 * // first member type of struct after comparison fails 4546 * val->baz = &v->bz; // reg->var_off is non-zero, so struct needs to be walked 4547 * // to match type 4548 * 4549 * In the kptr_ref case, check_func_arg_reg_off already ensures reg->var_off 4550 * is zero. We must also ensure that btf_struct_ids_match does not walk 4551 * the struct to match type against first member of struct, i.e. reject 4552 * second case from above. Hence, when type is BPF_KPTR_REF, we set 4553 * strict mode to true for type match. 4554 */ 4555 if (!btf_struct_ids_match(&env->log, reg->btf, reg->btf_id, reg->var_off.value, 4556 kptr_field->kptr.btf, kptr_field->kptr.btf_id, 4557 kptr_field->type != BPF_KPTR_UNREF, 4558 !type_is_alloc(reg->type))) 4559 goto bad_type; 4560 return 0; 4561 bad_type: 4562 verbose(env, "invalid kptr access, R%d type=%s%s ", regno, 4563 reg_type_str(env, reg->type), reg_name); 4564 verbose(env, "expected=%s%s", reg_type_str(env, PTR_TO_BTF_ID), targ_name); 4565 if (kptr_field->type == BPF_KPTR_UNREF) 4566 verbose(env, " or %s%s\n", reg_type_str(env, PTR_TO_BTF_ID | PTR_UNTRUSTED), 4567 targ_name); 4568 else 4569 verbose(env, "\n"); 4570 return -EINVAL; 4571 } 4572 4573 static bool in_sleepable(struct bpf_verifier_env *env) 4574 { 4575 return env->cur_state->in_sleepable; 4576 } 4577 4578 /* The non-sleepable programs and sleepable programs with explicit bpf_rcu_read_lock() 4579 * can dereference RCU protected pointers and result is PTR_TRUSTED. 4580 */ 4581 static bool in_rcu_cs(struct bpf_verifier_env *env) 4582 { 4583 return env->cur_state->active_rcu_locks || 4584 env->cur_state->active_preempt_locks || 4585 env->cur_state->active_locks || 4586 env->cur_state->active_irq_id || 4587 !in_sleepable(env); 4588 } 4589 4590 /* Once GCC supports btf_type_tag the following mechanism will be replaced with tag check */ 4591 BTF_SET_START(rcu_protected_types) 4592 #ifdef CONFIG_NET 4593 BTF_ID(struct, prog_test_ref_kfunc) 4594 #endif 4595 #ifdef CONFIG_CGROUPS 4596 BTF_ID(struct, cgroup) 4597 #endif 4598 #ifdef CONFIG_BPF_JIT 4599 BTF_ID(struct, bpf_cpumask) 4600 #endif 4601 BTF_ID(struct, task_struct) 4602 #ifdef CONFIG_CRYPTO 4603 BTF_ID(struct, bpf_crypto_ctx) 4604 #endif 4605 #ifdef CONFIG_INET 4606 BTF_ID(struct, bpf_ksock) 4607 #endif 4608 BTF_SET_END(rcu_protected_types) 4609 4610 static bool rcu_protected_object(const struct btf *btf, u32 btf_id) 4611 { 4612 if (!btf_is_kernel(btf)) 4613 return true; 4614 return btf_id_set_contains(&rcu_protected_types, btf_id); 4615 } 4616 4617 static struct btf_record *kptr_pointee_btf_record(struct btf_field *kptr_field) 4618 { 4619 struct btf_struct_meta *meta; 4620 4621 if (btf_is_kernel(kptr_field->kptr.btf)) 4622 return NULL; 4623 4624 meta = btf_find_struct_meta(kptr_field->kptr.btf, 4625 kptr_field->kptr.btf_id); 4626 4627 return meta ? meta->record : NULL; 4628 } 4629 4630 static bool rcu_safe_kptr(const struct btf_field *field) 4631 { 4632 const struct btf_field_kptr *kptr = &field->kptr; 4633 4634 return field->type == BPF_KPTR_PERCPU || 4635 (field->type == BPF_KPTR_REF && rcu_protected_object(kptr->btf, kptr->btf_id)); 4636 } 4637 4638 static u32 btf_ld_kptr_type(struct bpf_verifier_env *env, struct btf_field *kptr_field) 4639 { 4640 struct btf_record *rec; 4641 u32 ret; 4642 4643 ret = PTR_MAYBE_NULL; 4644 if (rcu_safe_kptr(kptr_field) && in_rcu_cs(env)) { 4645 ret |= MEM_RCU; 4646 if (kptr_field->type == BPF_KPTR_PERCPU) 4647 ret |= MEM_PERCPU; 4648 else if (!btf_is_kernel(kptr_field->kptr.btf)) 4649 ret |= MEM_ALLOC; 4650 4651 rec = kptr_pointee_btf_record(kptr_field); 4652 if (rec && btf_record_has_field(rec, BPF_GRAPH_NODE)) 4653 ret |= NON_OWN_REF; 4654 } else { 4655 ret |= PTR_UNTRUSTED; 4656 } 4657 4658 return ret; 4659 } 4660 4661 static int mark_uptr_ld_reg(struct bpf_verifier_env *env, u32 regno, 4662 struct btf_field *field) 4663 { 4664 struct bpf_reg_state *reg; 4665 const struct btf_type *t; 4666 4667 t = btf_type_by_id(field->kptr.btf, field->kptr.btf_id); 4668 mark_reg_known_zero(env, cur_regs(env), regno); 4669 reg = reg_state(env, regno); 4670 reg->type = PTR_TO_MEM | PTR_MAYBE_NULL; 4671 reg->mem_size = t->size; 4672 reg->id = ++env->id_gen; 4673 4674 return 0; 4675 } 4676 4677 static int check_map_kptr_access(struct bpf_verifier_env *env, 4678 int value_regno, int insn_idx, 4679 struct btf_field *kptr_field) 4680 { 4681 struct bpf_insn *insn = &env->prog->insnsi[insn_idx]; 4682 int class = BPF_CLASS(insn->code); 4683 struct bpf_reg_state *val_reg; 4684 int ret; 4685 4686 /* Things we already checked for in check_map_access and caller: 4687 * - Reject cases where variable offset may touch kptr 4688 * - size of access (must be BPF_DW) 4689 * - tnum_is_const(reg->var_off) 4690 * - kptr_field->offset == off + reg->var_off.value 4691 */ 4692 /* Only BPF_[LDX,STX,ST] | BPF_MEM | BPF_DW is supported */ 4693 if (BPF_MODE(insn->code) != BPF_MEM) { 4694 verbose(env, "kptr in map can only be accessed using BPF_MEM instruction mode\n"); 4695 return -EACCES; 4696 } 4697 4698 /* We only allow loading referenced kptr, since it will be marked as 4699 * untrusted, similar to unreferenced kptr. 4700 */ 4701 if (class != BPF_LDX && 4702 (kptr_field->type == BPF_KPTR_REF || kptr_field->type == BPF_KPTR_PERCPU)) { 4703 verbose(env, "store to referenced kptr disallowed\n"); 4704 return -EACCES; 4705 } 4706 if (class != BPF_LDX && kptr_field->type == BPF_UPTR) { 4707 verbose(env, "store to uptr disallowed\n"); 4708 return -EACCES; 4709 } 4710 4711 if (class == BPF_LDX) { 4712 if (kptr_field->type == BPF_UPTR) 4713 return mark_uptr_ld_reg(env, value_regno, kptr_field); 4714 4715 /* We can simply mark the value_regno receiving the pointer 4716 * value from map as PTR_TO_BTF_ID, with the correct type. 4717 */ 4718 ret = mark_btf_ld_reg(env, cur_regs(env), value_regno, PTR_TO_BTF_ID, 4719 kptr_field->kptr.btf, kptr_field->kptr.btf_id, 4720 btf_ld_kptr_type(env, kptr_field)); 4721 if (ret < 0) 4722 return ret; 4723 } else if (class == BPF_STX) { 4724 val_reg = reg_state(env, value_regno); 4725 if (bpf_register_is_null(val_reg)) { 4726 /* 4727 * This store is valid only because the scalar is known to be 4728 * zero. Mark it precise so another scalar cannot be pruned 4729 * against this state. 4730 */ 4731 return mark_chain_precision(env, value_regno); 4732 } 4733 if (map_kptr_match_type(env, kptr_field, val_reg, value_regno)) 4734 return -EACCES; 4735 } else if (class == BPF_ST) { 4736 if (insn->imm) { 4737 verbose(env, "BPF_ST imm must be 0 when storing to kptr at off=%u\n", 4738 kptr_field->offset); 4739 return -EACCES; 4740 } 4741 } else { 4742 verbose(env, "kptr in map can only be accessed using BPF_LDX/BPF_STX/BPF_ST\n"); 4743 return -EACCES; 4744 } 4745 return 0; 4746 } 4747 4748 /* 4749 * Return the size of the memory region accessible from a pointer to map value. 4750 * For INSN_ARRAY maps whole bpf_insn_array->ips array is accessible. 4751 */ 4752 static u32 map_mem_size(const struct bpf_map *map) 4753 { 4754 if (map->map_type == BPF_MAP_TYPE_INSN_ARRAY) 4755 return map->max_entries * sizeof(long); 4756 4757 return map->value_size; 4758 } 4759 4760 /* check read/write into a map element with possible variable offset */ 4761 static int check_map_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, 4762 int off, int size, bool zero_size_allowed, 4763 enum bpf_access_src src) 4764 { 4765 struct bpf_map *map = reg->map_ptr; 4766 u32 mem_size = map_mem_size(map); 4767 struct btf_record *rec; 4768 int err, i; 4769 4770 err = check_mem_region_access(env, reg, argno, off, size, mem_size, zero_size_allowed); 4771 if (err) 4772 return err; 4773 4774 if (IS_ERR_OR_NULL(map->record)) 4775 return 0; 4776 rec = map->record; 4777 for (i = 0; i < rec->cnt; i++) { 4778 struct btf_field *field = &rec->fields[i]; 4779 u32 p = field->offset; 4780 4781 /* If any part of a field can be touched by load/store, reject 4782 * this program. To check that [x1, x2) overlaps with [y1, y2), 4783 * it is sufficient to check x1 < y2 && y1 < x2. 4784 */ 4785 if (reg_smin(reg) + off < p + field->size && 4786 p < reg_umax(reg) + off + size) { 4787 switch (field->type) { 4788 case BPF_KPTR_UNREF: 4789 case BPF_KPTR_REF: 4790 case BPF_KPTR_PERCPU: 4791 case BPF_UPTR: 4792 if (src != ACCESS_DIRECT) { 4793 verbose(env, "%s cannot be accessed indirectly by helper\n", 4794 btf_field_type_name(field->type)); 4795 return -EACCES; 4796 } 4797 if (!tnum_is_const(reg->var_off)) { 4798 verbose(env, "%s access cannot have variable offset\n", 4799 btf_field_type_name(field->type)); 4800 return -EACCES; 4801 } 4802 if (p != off + reg->var_off.value) { 4803 verbose(env, "%s access misaligned expected=%u off=%llu\n", 4804 btf_field_type_name(field->type), 4805 p, off + reg->var_off.value); 4806 return -EACCES; 4807 } 4808 if (size != bpf_size_to_bytes(BPF_DW)) { 4809 verbose(env, "%s access size must be BPF_DW\n", 4810 btf_field_type_name(field->type)); 4811 return -EACCES; 4812 } 4813 break; 4814 default: 4815 verbose(env, "%s cannot be accessed directly by load/store\n", 4816 btf_field_type_name(field->type)); 4817 return -EACCES; 4818 } 4819 } 4820 } 4821 return 0; 4822 } 4823 4824 static bool may_access_direct_pkt_data(struct bpf_verifier_env *env, 4825 const struct bpf_func_proto *fn, 4826 enum bpf_access_type t) 4827 { 4828 enum bpf_prog_type prog_type = resolve_prog_type(env->prog); 4829 4830 switch (prog_type) { 4831 /* Program types only with direct read access go here! */ 4832 case BPF_PROG_TYPE_LWT_IN: 4833 case BPF_PROG_TYPE_LWT_OUT: 4834 case BPF_PROG_TYPE_LWT_SEG6LOCAL: 4835 case BPF_PROG_TYPE_SK_REUSEPORT: 4836 case BPF_PROG_TYPE_FLOW_DISSECTOR: 4837 case BPF_PROG_TYPE_CGROUP_SKB: 4838 if (t == BPF_WRITE) 4839 return false; 4840 fallthrough; 4841 4842 /* Program types with direct read + write access go here! */ 4843 case BPF_PROG_TYPE_SCHED_CLS: 4844 case BPF_PROG_TYPE_SCHED_ACT: 4845 case BPF_PROG_TYPE_XDP: 4846 case BPF_PROG_TYPE_LWT_XMIT: 4847 case BPF_PROG_TYPE_SK_SKB: 4848 case BPF_PROG_TYPE_SK_MSG: 4849 if (fn) 4850 return fn->pkt_access; 4851 4852 env->seen_direct_write = true; 4853 return true; 4854 4855 case BPF_PROG_TYPE_CGROUP_SOCKOPT: 4856 if (t == BPF_WRITE) 4857 env->seen_direct_write = true; 4858 4859 return true; 4860 4861 default: 4862 return false; 4863 } 4864 } 4865 4866 static int check_packet_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, int off, 4867 int size, bool zero_size_allowed) 4868 { 4869 int err; 4870 4871 if (reg->range < 0) { 4872 verbose(env, "%s offset is outside of the packet\n", reg_arg_name(env, argno)); 4873 return -EINVAL; 4874 } 4875 4876 err = check_mem_region_access(env, reg, argno, off, size, reg->range, zero_size_allowed); 4877 if (err) 4878 return err; 4879 4880 /* __check_mem_access has made sure "off + size - 1" is within u16. 4881 * reg_umax(reg) can't be bigger than MAX_PACKET_OFF which is 0xffff, 4882 * otherwise find_good_pkt_pointers would have refused to set range info 4883 * that __check_mem_access would have rejected this pkt access. 4884 * Therefore, "off + reg_umax(reg) + size - 1" won't overflow u32. 4885 */ 4886 env->prog->aux->max_pkt_offset = 4887 max_t(u32, env->prog->aux->max_pkt_offset, 4888 off + reg_umax(reg) + size - 1); 4889 4890 return 0; 4891 } 4892 4893 static bool is_var_ctx_off_allowed(struct bpf_prog *prog) 4894 { 4895 return resolve_prog_type(prog) == BPF_PROG_TYPE_SYSCALL; 4896 } 4897 4898 /* check access to 'struct bpf_context' fields. Supports fixed offsets only */ 4899 static int __check_ctx_access(struct bpf_verifier_env *env, int insn_idx, int off, int size, 4900 enum bpf_access_type t, struct bpf_insn_access_aux *info) 4901 { 4902 if (env->ops->is_valid_access && 4903 env->ops->is_valid_access(off, size, t, env->prog, info)) { 4904 /* A non zero info.ctx_field_size indicates that this field is a 4905 * candidate for later verifier transformation to load the whole 4906 * field and then apply a mask when accessed with a narrower 4907 * access than actual ctx access size. A zero info.ctx_field_size 4908 * will only allow for whole field access and rejects any other 4909 * type of narrower access. 4910 */ 4911 if (base_type(info->reg_type) == PTR_TO_BTF_ID) { 4912 if (info->ref_id && 4913 !find_reference_state(env->cur_state, info->ref_id)) { 4914 verbose(env, "invalid bpf_context access off=%d. Reference may already be released\n", 4915 off); 4916 return -EACCES; 4917 } 4918 } else { 4919 env->insn_aux_data[insn_idx].ctx_field_size = info->ctx_field_size; 4920 } 4921 /* remember the offset of last byte accessed in ctx */ 4922 if (env->prog->aux->max_ctx_offset < off + size) 4923 env->prog->aux->max_ctx_offset = off + size; 4924 return 0; 4925 } 4926 4927 verbose(env, "invalid bpf_context access off=%d size=%d\n", off, size); 4928 return -EACCES; 4929 } 4930 4931 static int check_ctx_access(struct bpf_verifier_env *env, int insn_idx, struct bpf_reg_state *reg, argno_t argno, 4932 int off, int access_size, enum bpf_access_type t, 4933 struct bpf_insn_access_aux *info) 4934 { 4935 /* 4936 * Program types that don't rewrite ctx accesses can safely 4937 * dereference ctx pointers with fixed offsets. 4938 */ 4939 bool var_off_ok = is_var_ctx_off_allowed(env->prog); 4940 bool fixed_off_ok = !env->ops->convert_ctx_access; 4941 int err; 4942 4943 if (var_off_ok) 4944 err = check_mem_region_access(env, reg, argno, off, access_size, U16_MAX, false); 4945 else 4946 err = __check_ptr_off_reg(env, reg, argno, fixed_off_ok); 4947 if (err) 4948 return err; 4949 off += reg_umax(reg); 4950 4951 err = __check_ctx_access(env, insn_idx, off, access_size, t, info); 4952 if (err) 4953 verbose_linfo(env, insn_idx, "; "); 4954 return err; 4955 } 4956 4957 static int check_flow_keys_access(struct bpf_verifier_env *env, 4958 struct bpf_reg_state *reg, argno_t argno, 4959 int off, int size) 4960 { 4961 /* Only a constant offset is allowed here; fold it into off. */ 4962 if (!tnum_is_const(reg->var_off)) { 4963 char tn_buf[48]; 4964 4965 tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off); 4966 verbose(env, "%s invalid variable offset to flow keys: off=%d, var_off=%s\n", 4967 reg_arg_name(env, argno), off, tn_buf); 4968 return -EACCES; 4969 } 4970 off += reg->var_off.value; 4971 4972 if (size < 0 || off < 0 || 4973 (u64)off + size > sizeof(struct bpf_flow_keys)) { 4974 verbose(env, "invalid access to flow keys off=%d size=%d\n", 4975 off, size); 4976 return -EACCES; 4977 } 4978 return 0; 4979 } 4980 4981 static int check_sock_access(struct bpf_verifier_env *env, int insn_idx, 4982 struct bpf_reg_state *reg, argno_t argno, int off, int size, 4983 enum bpf_access_type t) 4984 { 4985 struct bpf_insn_access_aux info = {}; 4986 bool valid; 4987 4988 if (reg_smin(reg) < 0) { 4989 verbose(env, "%s min value is negative, either use unsigned index or do a if (index >=0) check.\n", 4990 reg_arg_name(env, argno)); 4991 return -EACCES; 4992 } 4993 4994 switch (reg->type) { 4995 case PTR_TO_SOCK_COMMON: 4996 valid = bpf_sock_common_is_valid_access(off, size, t, &info); 4997 break; 4998 case PTR_TO_SOCKET: 4999 valid = bpf_sock_is_valid_access(off, size, t, &info); 5000 break; 5001 case PTR_TO_TCP_SOCK: 5002 valid = bpf_tcp_sock_is_valid_access(off, size, t, &info); 5003 break; 5004 case PTR_TO_XDP_SOCK: 5005 valid = bpf_xdp_sock_is_valid_access(off, size, t, &info); 5006 break; 5007 default: 5008 valid = false; 5009 } 5010 5011 if (valid) { 5012 env->insn_aux_data[insn_idx].ctx_field_size = 5013 info.ctx_field_size; 5014 return 0; 5015 } 5016 5017 verbose(env, "%s invalid %s access off=%d size=%d\n", 5018 reg_arg_name(env, argno), reg_type_str(env, reg->type), off, size); 5019 5020 return -EACCES; 5021 } 5022 5023 static bool is_pointer_value(struct bpf_verifier_env *env, int regno) 5024 { 5025 return __is_pointer_value(env->allow_ptr_leaks, reg_state(env, regno)); 5026 } 5027 5028 static bool is_ctx_reg(struct bpf_verifier_env *env, int regno) 5029 { 5030 const struct bpf_reg_state *reg = reg_state(env, regno); 5031 5032 return reg->type == PTR_TO_CTX; 5033 } 5034 5035 static bool is_sk_reg(struct bpf_verifier_env *env, int regno) 5036 { 5037 const struct bpf_reg_state *reg = reg_state(env, regno); 5038 5039 return type_is_sk_pointer(reg->type); 5040 } 5041 5042 static bool is_pkt_reg(struct bpf_verifier_env *env, int regno) 5043 { 5044 const struct bpf_reg_state *reg = reg_state(env, regno); 5045 5046 return type_is_pkt_pointer(reg->type); 5047 } 5048 5049 static bool is_flow_key_reg(struct bpf_verifier_env *env, int regno) 5050 { 5051 const struct bpf_reg_state *reg = reg_state(env, regno); 5052 5053 /* Separate to is_ctx_reg() since we still want to allow BPF_ST here. */ 5054 return reg->type == PTR_TO_FLOW_KEYS; 5055 } 5056 5057 static bool is_arena_reg(struct bpf_verifier_env *env, int regno) 5058 { 5059 const struct bpf_reg_state *reg = reg_state(env, regno); 5060 5061 return reg->type == PTR_TO_ARENA; 5062 } 5063 5064 static bool is_load_acq_unsafe(struct bpf_verifier_env *env, int regno, 5065 struct bpf_insn *insn) 5066 { 5067 const struct bpf_reg_state *reg = reg_state(env, regno); 5068 5069 /* 5070 * A BPF_LOAD_ACQ is not rewritten to a BPF_PROBE_MEM load by the 5071 * verifier, unlike a regular BPF_LDX. The JIT would emit a plain load 5072 * with no exception table entry, so a fault (e.g. NULL deref) crashes 5073 * the kernel instead of being handled. Reject the source pointer types 5074 * that would have needed that protection, the remaining ones stay 5075 * allowed. 5076 */ 5077 return insn->imm == BPF_LOAD_ACQ && bpf_may_fault_on_deref(reg->type); 5078 } 5079 5080 /* Return false if @regno contains a pointer whose type isn't supported for 5081 * atomic instruction @insn. 5082 */ 5083 static bool atomic_ptr_type_ok(struct bpf_verifier_env *env, int regno, 5084 struct bpf_insn *insn) 5085 { 5086 if (is_ctx_reg(env, regno)) 5087 return false; 5088 if (is_pkt_reg(env, regno)) 5089 return false; 5090 if (is_flow_key_reg(env, regno)) 5091 return false; 5092 if (is_sk_reg(env, regno)) 5093 return false; 5094 if (is_arena_reg(env, regno)) 5095 return bpf_jit_supports_insn(insn, true); 5096 if (is_load_acq_unsafe(env, regno, insn)) 5097 return false; 5098 return true; 5099 } 5100 5101 static u32 *reg2btf_ids[__BPF_REG_TYPE_MAX] = { 5102 #ifdef CONFIG_NET 5103 [PTR_TO_SOCKET] = &btf_sock_ids[BTF_SOCK_TYPE_SOCK], 5104 [PTR_TO_SOCK_COMMON] = &btf_sock_ids[BTF_SOCK_TYPE_SOCK_COMMON], 5105 [PTR_TO_TCP_SOCK] = &btf_sock_ids[BTF_SOCK_TYPE_TCP], 5106 #endif 5107 [CONST_PTR_TO_MAP] = btf_bpf_map_id, 5108 }; 5109 5110 static enum bpf_reg_type lookup_reg2btf_ids(u32 ref_id) 5111 { 5112 enum bpf_reg_type type; 5113 5114 for (type = 0; type < __BPF_REG_TYPE_MAX; type++) { 5115 if (reg2btf_ids[type] && *reg2btf_ids[type] == ref_id) 5116 return type; 5117 } 5118 5119 return NOT_INIT; 5120 } 5121 5122 static bool is_trusted_reg(struct bpf_verifier_env *env, const struct bpf_reg_state *reg) 5123 { 5124 /* A referenced register is always trusted. */ 5125 if (reg_is_referenced(env, reg)) 5126 return true; 5127 5128 /* Types listed in the reg2btf_ids are always trusted */ 5129 if (reg2btf_ids[base_type(reg->type)] && 5130 !bpf_type_has_unsafe_modifiers(reg->type)) 5131 return true; 5132 5133 /* If a register is not referenced, it is trusted if it has the 5134 * MEM_ALLOC or PTR_TRUSTED type modifiers, and no others. Some of the 5135 * other type modifiers may be safe, but we elect to take an opt-in 5136 * approach here as some (e.g. PTR_UNTRUSTED and PTR_MAYBE_NULL) are 5137 * not. 5138 * 5139 * Eventually, we should make PTR_TRUSTED the single source of truth 5140 * for whether a register is trusted. 5141 */ 5142 return type_flag(reg->type) & BPF_REG_TRUSTED_MODIFIERS && 5143 !bpf_type_has_unsafe_modifiers(reg->type); 5144 } 5145 5146 static bool is_rcu_reg(const struct bpf_reg_state *reg) 5147 { 5148 return reg->type & MEM_RCU; 5149 } 5150 5151 static void clear_trusted_flags(enum bpf_type_flag *flag) 5152 { 5153 *flag &= ~(BPF_REG_TRUSTED_MODIFIERS | MEM_RCU); 5154 } 5155 5156 static int check_pkt_ptr_alignment(struct bpf_verifier_env *env, 5157 const struct bpf_reg_state *reg, 5158 int off, int size, bool strict) 5159 { 5160 struct tnum reg_off; 5161 int ip_align; 5162 5163 /* Byte size accesses are always allowed. */ 5164 if (!strict || size == 1) 5165 return 0; 5166 5167 /* For platforms that do not have a Kconfig enabling 5168 * CONFIG_HAVE_EFFICIENT_UNALIGNED_ACCESS the value of 5169 * NET_IP_ALIGN is universally set to '2'. And on platforms 5170 * that do set CONFIG_HAVE_EFFICIENT_UNALIGNED_ACCESS, we get 5171 * to this code only in strict mode where we want to emulate 5172 * the NET_IP_ALIGN==2 checking. Therefore use an 5173 * unconditional IP align value of '2'. 5174 */ 5175 ip_align = 2; 5176 5177 reg_off = tnum_add(reg->var_off, tnum_const(ip_align + off)); 5178 if (!tnum_is_aligned(reg_off, size)) { 5179 char tn_buf[48]; 5180 5181 tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off); 5182 verbose(env, 5183 "misaligned packet access off %d+%s+%d size %d\n", 5184 ip_align, tn_buf, off, size); 5185 return -EACCES; 5186 } 5187 5188 return 0; 5189 } 5190 5191 static int check_generic_ptr_alignment(struct bpf_verifier_env *env, 5192 const struct bpf_reg_state *reg, 5193 const char *pointer_desc, 5194 int off, int size, bool strict) 5195 { 5196 struct tnum reg_off; 5197 5198 /* Byte size accesses are always allowed. */ 5199 if (!strict || size == 1) 5200 return 0; 5201 5202 reg_off = tnum_add(reg->var_off, tnum_const(off)); 5203 if (!tnum_is_aligned(reg_off, size)) { 5204 char tn_buf[48]; 5205 5206 tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off); 5207 verbose(env, "misaligned %saccess off %s+%d size %d\n", 5208 pointer_desc, tn_buf, off, size); 5209 return -EACCES; 5210 } 5211 5212 return 0; 5213 } 5214 5215 static int check_ptr_alignment(struct bpf_verifier_env *env, 5216 const struct bpf_reg_state *reg, int off, 5217 int size, bool strict_alignment_once) 5218 { 5219 bool strict = env->strict_alignment || strict_alignment_once; 5220 const char *pointer_desc = ""; 5221 5222 switch (reg->type) { 5223 case PTR_TO_PACKET: 5224 case PTR_TO_PACKET_META: 5225 /* Special case, because of NET_IP_ALIGN. Given metadata sits 5226 * right in front, treat it the very same way. 5227 */ 5228 return check_pkt_ptr_alignment(env, reg, off, size, strict); 5229 case PTR_TO_FLOW_KEYS: 5230 pointer_desc = "flow keys "; 5231 break; 5232 case PTR_TO_MAP_KEY: 5233 pointer_desc = "key "; 5234 break; 5235 case PTR_TO_MAP_VALUE: 5236 pointer_desc = "value "; 5237 if (reg->map_ptr->map_type == BPF_MAP_TYPE_INSN_ARRAY) 5238 strict = true; 5239 break; 5240 case PTR_TO_CTX: 5241 pointer_desc = "context "; 5242 break; 5243 case PTR_TO_STACK: 5244 pointer_desc = "stack "; 5245 /* The stack spill tracking logic in check_stack_write_fixed_off() 5246 * and check_stack_read_fixed_off() relies on stack accesses being 5247 * aligned. 5248 */ 5249 strict = true; 5250 break; 5251 case PTR_TO_SOCKET: 5252 pointer_desc = "sock "; 5253 break; 5254 case PTR_TO_SOCK_COMMON: 5255 pointer_desc = "sock_common "; 5256 break; 5257 case PTR_TO_TCP_SOCK: 5258 pointer_desc = "tcp_sock "; 5259 break; 5260 case PTR_TO_XDP_SOCK: 5261 pointer_desc = "xdp_sock "; 5262 break; 5263 case PTR_TO_ARENA: 5264 return 0; 5265 default: 5266 break; 5267 } 5268 return check_generic_ptr_alignment(env, reg, pointer_desc, off, size, 5269 strict); 5270 } 5271 5272 static enum priv_stack_mode bpf_enable_priv_stack(struct bpf_prog *prog) 5273 { 5274 if (!bpf_jit_supports_private_stack()) 5275 return NO_PRIV_STACK; 5276 5277 /* bpf_prog_check_recur() checks all prog types that use bpf trampoline 5278 * while kprobe/tp/perf_event/raw_tp don't use trampoline hence checked 5279 * explicitly. 5280 */ 5281 switch (prog->type) { 5282 case BPF_PROG_TYPE_KPROBE: 5283 case BPF_PROG_TYPE_TRACEPOINT: 5284 case BPF_PROG_TYPE_PERF_EVENT: 5285 case BPF_PROG_TYPE_RAW_TRACEPOINT: 5286 return PRIV_STACK_ADAPTIVE; 5287 case BPF_PROG_TYPE_TRACING: 5288 case BPF_PROG_TYPE_LSM: 5289 case BPF_PROG_TYPE_STRUCT_OPS: 5290 if (prog->aux->priv_stack_requested || bpf_prog_check_recur(prog)) 5291 return PRIV_STACK_ADAPTIVE; 5292 fallthrough; 5293 default: 5294 break; 5295 } 5296 5297 return NO_PRIV_STACK; 5298 } 5299 5300 static int round_up_stack_depth(struct bpf_verifier_env *env, int stack_depth) 5301 { 5302 if (env->prog->jit_requested) 5303 return round_up(stack_depth, 16); 5304 5305 /* round up to 32-bytes, since this is granularity 5306 * of interpreter stack size 5307 */ 5308 return round_up(max_t(u32, stack_depth, 1), 32); 5309 } 5310 5311 /* temporary state used for call frame depth calculation */ 5312 struct bpf_subprog_call_depth_info { 5313 int ret_insn; /* caller instruction where we return to. */ 5314 int caller; /* caller subprogram idx */ 5315 int frame; /* # of consecutive static call stack frames on top of stack */ 5316 }; 5317 5318 /* starting from main bpf function walk all instructions of the function 5319 * and recursively walk all callees that given function can call. 5320 * Ignore jump and exit insns. 5321 */ 5322 static int check_max_stack_depth_subprog(struct bpf_verifier_env *env, int idx, 5323 struct bpf_subprog_call_depth_info *dinfo, 5324 bool priv_stack_supported) 5325 { 5326 struct bpf_subprog_info *subprog = env->subprog_info; 5327 struct bpf_insn *insn = env->prog->insnsi; 5328 int depth = 0, frame = 0, i, subprog_end, subprog_depth; 5329 bool tail_call_reachable = false; 5330 int total; 5331 int tmp; 5332 5333 /* no caller idx */ 5334 dinfo[idx].caller = -1; 5335 5336 i = subprog[idx].start; 5337 if (!priv_stack_supported) 5338 subprog[idx].priv_stack_mode = NO_PRIV_STACK; 5339 process_func: 5340 if (subprog[idx].has_ld_abs) { 5341 for (tmp = idx; tmp >= 0; tmp = dinfo[tmp].caller) { 5342 if (subprog[tmp].is_cb) { 5343 verbose(env, "cannot use BPF_LD_[ABS|IND] within callback\n"); 5344 return -EINVAL; 5345 } 5346 } 5347 } 5348 5349 /* protect against potential stack overflow that might happen when 5350 * bpf2bpf calls get combined with tailcalls. Limit the caller's stack 5351 * depth for such case down to 256 so that the worst case scenario 5352 * would result in 8k stack size (32 which is tailcall limit * 256 = 5353 * 8k). 5354 * 5355 * To get the idea what might happen, see an example: 5356 * func1 -> sub rsp, 128 5357 * subfunc1 -> sub rsp, 256 5358 * tailcall1 -> add rsp, 256 5359 * func2 -> sub rsp, 192 (total stack size = 128 + 192 = 320) 5360 * subfunc2 -> sub rsp, 64 5361 * subfunc22 -> sub rsp, 128 5362 * tailcall2 -> add rsp, 128 5363 * func3 -> sub rsp, 32 (total stack size 128 + 192 + 64 + 32 = 416) 5364 * 5365 * tailcall will unwind the current stack frame but it will not get rid 5366 * of caller's stack as shown on the example above. 5367 */ 5368 if (idx && subprog[idx].has_tail_call && depth >= 256) { 5369 verbose(env, 5370 "tail_calls are not allowed when call stack of previous frames is %d bytes. Too large\n", 5371 depth); 5372 return -EACCES; 5373 } 5374 5375 subprog_depth = round_up_stack_depth(env, subprog[idx].stack_depth); 5376 if (IS_ENABLED(CONFIG_X86_64) && subprog[idx].stack_arg_cnt) { 5377 /* x86-64 uses R9 for both private stack frame pointer and arg6. */ 5378 subprog[idx].priv_stack_mode = NO_PRIV_STACK; 5379 } else if (priv_stack_supported) { 5380 /* Request private stack support only if the subprog stack 5381 * depth is no less than BPF_PRIV_STACK_MIN_SIZE. This is to 5382 * avoid jit penalty if the stack usage is small. 5383 */ 5384 if (subprog[idx].priv_stack_mode == PRIV_STACK_UNKNOWN && 5385 subprog_depth >= BPF_PRIV_STACK_MIN_SIZE) 5386 subprog[idx].priv_stack_mode = PRIV_STACK_ADAPTIVE; 5387 } 5388 5389 if (subprog[idx].priv_stack_mode == PRIV_STACK_ADAPTIVE) { 5390 if (subprog_depth > env->max_stack_depth) 5391 env->max_stack_depth = subprog_depth; 5392 if (subprog_depth > MAX_BPF_STACK) { 5393 verbose(env, "stack size of subprog %d is %d. Too large\n", 5394 idx, subprog_depth); 5395 return -EACCES; 5396 } 5397 } else { 5398 depth += subprog_depth; 5399 if (depth > env->max_stack_depth) 5400 env->max_stack_depth = depth; 5401 if (depth > MAX_BPF_STACK) { 5402 total = 0; 5403 for (tmp = idx; tmp >= 0; tmp = dinfo[tmp].caller) 5404 total++; 5405 5406 verbose(env, "combined stack size of %d calls is %d. Too large\n", 5407 total, depth); 5408 return -EACCES; 5409 } 5410 } 5411 continue_func: 5412 subprog_end = subprog[idx + 1].start; 5413 for (; i < subprog_end; i++) { 5414 int next_insn, sidx; 5415 5416 if (bpf_pseudo_kfunc_call(insn + i) && !insn[i].off) { 5417 bool err = false; 5418 5419 if (!bpf_is_throw_kfunc(insn + i)) 5420 continue; 5421 for (tmp = idx; tmp >= 0 && !err; tmp = dinfo[tmp].caller) { 5422 if (subprog[tmp].is_cb) { 5423 err = true; 5424 break; 5425 } 5426 } 5427 if (!err) 5428 continue; 5429 verbose(env, 5430 "bpf_throw kfunc (insn %d) cannot be called from callback subprog %d\n", 5431 i, idx); 5432 return -EINVAL; 5433 } 5434 5435 if (!bpf_pseudo_call(insn + i) && !bpf_pseudo_func(insn + i)) 5436 continue; 5437 /* remember insn and function to return to */ 5438 5439 /* find the callee */ 5440 next_insn = i + insn[i].imm + 1; 5441 sidx = bpf_find_subprog(env, next_insn); 5442 if (verifier_bug_if(sidx < 0, env, "callee not found at insn %d", next_insn)) 5443 return -EFAULT; 5444 if (subprog[sidx].is_async_cb) { 5445 /* async callbacks don't increase bpf prog stack size unless called directly */ 5446 if (!bpf_pseudo_call(insn + i)) 5447 continue; 5448 if (subprog[sidx].is_exception_cb) { 5449 verbose(env, "insn %d cannot call exception cb directly", i); 5450 return -EINVAL; 5451 } 5452 } 5453 5454 /* store caller info for after we return from callee */ 5455 dinfo[idx].frame = frame; 5456 dinfo[idx].ret_insn = i + 1; 5457 5458 /* push caller idx into callee's dinfo */ 5459 dinfo[sidx].caller = idx; 5460 5461 i = next_insn; 5462 5463 idx = sidx; 5464 if (!priv_stack_supported) 5465 subprog[idx].priv_stack_mode = NO_PRIV_STACK; 5466 5467 /* sync tail_call_reachable with callee state on entry */ 5468 tail_call_reachable = subprog[idx].has_tail_call; 5469 5470 frame = bpf_subprog_is_global(env, idx) ? 0 : frame + 1; 5471 if (frame >= MAX_CALL_FRAMES) { 5472 verbose(env, "the call stack of %d frames is too deep !\n", 5473 frame); 5474 return -E2BIG; 5475 } 5476 goto process_func; 5477 } 5478 /* if tail call got detected across bpf2bpf calls then mark each of the 5479 * currently present subprog frames as tail call reachable subprogs; 5480 * this info will be utilized by JIT so that we will be preserving the 5481 * tail call counter throughout bpf2bpf calls combined with tailcalls 5482 */ 5483 if (tail_call_reachable) { 5484 for (tmp = idx; tmp >= 0; tmp = dinfo[tmp].caller) { 5485 if (subprog[tmp].is_cb) { 5486 verbose(env, "cannot tail call within callback\n"); 5487 return -EINVAL; 5488 } 5489 if (subprog[tmp].stack_arg_cnt) { 5490 verbose(env, "tail_calls are not allowed in programs with stack args\n"); 5491 return -EINVAL; 5492 } 5493 subprog[tmp].tail_call_reachable = true; 5494 } 5495 } else if (!idx && subprog[0].has_tail_call && subprog[0].stack_arg_cnt) { 5496 verbose(env, "tail_calls are not allowed in programs with stack args\n"); 5497 return -EINVAL; 5498 } 5499 5500 if (subprog[0].tail_call_reachable) 5501 env->prog->aux->tail_call_reachable = true; 5502 5503 /* end of for() loop means the last insn of the 'subprog' 5504 * was reached. Doesn't matter whether it was JA or EXIT 5505 */ 5506 if (frame == 0 && dinfo[idx].caller < 0) 5507 return 0; 5508 if (subprog[idx].priv_stack_mode != PRIV_STACK_ADAPTIVE) 5509 depth -= round_up_stack_depth(env, subprog[idx].stack_depth); 5510 5511 /* pop caller idx from callee */ 5512 idx = dinfo[idx].caller; 5513 5514 /* retrieve caller state from its frame */ 5515 frame = dinfo[idx].frame; 5516 i = dinfo[idx].ret_insn; 5517 5518 /* reset tail_call_reachable to the parent's actual state */ 5519 tail_call_reachable = subprog[idx].tail_call_reachable; 5520 5521 goto continue_func; 5522 } 5523 5524 static int check_max_stack_depth(struct bpf_verifier_env *env) 5525 { 5526 enum priv_stack_mode priv_stack_mode = PRIV_STACK_UNKNOWN; 5527 struct bpf_subprog_call_depth_info *dinfo; 5528 struct bpf_subprog_info *si = env->subprog_info; 5529 bool priv_stack_supported; 5530 int ret; 5531 5532 dinfo = kvzalloc_objs(*dinfo, env->subprog_cnt, GFP_KERNEL_ACCOUNT); 5533 if (!dinfo) 5534 return -ENOMEM; 5535 5536 for (int i = 0; i < env->subprog_cnt; i++) { 5537 if (si[i].has_tail_call) { 5538 priv_stack_mode = NO_PRIV_STACK; 5539 break; 5540 } 5541 } 5542 5543 if (priv_stack_mode == PRIV_STACK_UNKNOWN) 5544 priv_stack_mode = bpf_enable_priv_stack(env->prog); 5545 5546 /* All async_cb subprogs use normal kernel stack. If a particular 5547 * subprog appears in both main prog and async_cb subtree, that 5548 * subprog will use normal kernel stack to avoid potential nesting. 5549 * The reverse subprog traversal ensures when main prog subtree is 5550 * checked, the subprogs appearing in async_cb subtrees are already 5551 * marked as using normal kernel stack, so stack size checking can 5552 * be done properly. 5553 */ 5554 for (int i = env->subprog_cnt - 1; i >= 0; i--) { 5555 if (!i || si[i].is_async_cb) { 5556 priv_stack_supported = !i && priv_stack_mode == PRIV_STACK_ADAPTIVE; 5557 ret = check_max_stack_depth_subprog(env, i, dinfo, 5558 priv_stack_supported); 5559 if (ret < 0) { 5560 kvfree(dinfo); 5561 return ret; 5562 } 5563 } 5564 } 5565 5566 for (int i = 0; i < env->subprog_cnt; i++) { 5567 if (si[i].priv_stack_mode == PRIV_STACK_ADAPTIVE) { 5568 env->prog->aux->jits_use_priv_stack = true; 5569 break; 5570 } 5571 } 5572 5573 kvfree(dinfo); 5574 5575 return 0; 5576 } 5577 5578 static int __check_buffer_access(struct bpf_verifier_env *env, 5579 const char *buf_info, 5580 const struct bpf_reg_state *reg, 5581 argno_t argno, int off, int size, 5582 u32 *access_end) 5583 { 5584 s64 start; 5585 5586 if (!tnum_is_const(reg->var_off)) { 5587 char tn_buf[48]; 5588 5589 tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off); 5590 verbose(env, 5591 "%s invalid variable buffer offset: off=%d, var_off=%s\n", 5592 reg_arg_name(env, argno), off, tn_buf); 5593 return -EACCES; 5594 } 5595 5596 start = (s64)reg->var_off.value + off; 5597 if (start < 0) { 5598 verbose(env, 5599 "%s invalid negative %s buffer offset: off=%d, var_off=%lld\n", 5600 reg_arg_name(env, argno), buf_info, off, (s64)reg->var_off.value); 5601 return -EACCES; 5602 } 5603 5604 *access_end = start + size; 5605 return 0; 5606 } 5607 5608 static int check_tp_buffer_access(struct bpf_verifier_env *env, 5609 const struct bpf_reg_state *reg, 5610 argno_t argno, int off, int size) 5611 { 5612 u32 access_end; 5613 int err; 5614 5615 err = __check_buffer_access(env, "tracepoint", reg, argno, off, size, &access_end); 5616 if (err) 5617 return err; 5618 5619 env->prog->aux->max_tp_access = max(access_end, env->prog->aux->max_tp_access); 5620 5621 return 0; 5622 } 5623 5624 static int check_buffer_access(struct bpf_verifier_env *env, 5625 const struct bpf_reg_state *reg, 5626 argno_t argno, int off, int size, 5627 bool zero_size_allowed, 5628 u32 *max_access) 5629 { 5630 const char *buf_info = type_is_rdonly_mem(reg->type) ? "rdonly" : "rdwr"; 5631 u32 access_end; 5632 int err; 5633 5634 err = __check_buffer_access(env, buf_info, reg, argno, off, size, &access_end); 5635 if (err) 5636 return err; 5637 5638 *max_access = max(access_end, *max_access); 5639 5640 return 0; 5641 } 5642 5643 /* BPF architecture zero extends alu32 ops into 64-bit registesr */ 5644 static void zext_32_to_64(struct bpf_reg_state *reg) 5645 { 5646 reg->var_off = tnum_subreg(reg->var_off); 5647 reg_set_urange64(reg, reg_u32_min(reg), reg_u32_max(reg)); 5648 } 5649 5650 /* truncate register to smaller size (in bytes) 5651 * must be called with size < BPF_REG_SIZE 5652 */ 5653 static void coerce_reg_to_size(struct bpf_reg_state *reg, int size) 5654 { 5655 u64 mask; 5656 5657 /* clear high bits in bit representation */ 5658 reg->var_off = tnum_cast(reg->var_off, size); 5659 5660 /* fix arithmetic bounds */ 5661 mask = ((u64)1 << (size * 8)) - 1; 5662 if ((reg_umin(reg) & ~mask) == (reg_umax(reg) & ~mask)) 5663 reg_set_urange64(reg, reg_umin(reg) & mask, reg_umax(reg) & mask); 5664 else 5665 reg_set_urange64(reg, 0, mask); 5666 5667 /* If size is smaller than 32bit register the 32bit register 5668 * values are also truncated so we push 64-bit bounds into 5669 * 32-bit bounds. Above were truncated < 32-bits already. 5670 */ 5671 if (size < 4) 5672 __mark_reg32_unbounded(reg); 5673 5674 reg_bounds_sync(reg); 5675 } 5676 5677 static void set_sext64_default_val(struct bpf_reg_state *reg, int size) 5678 { 5679 if (size == 1) { 5680 reg_set_srange64(reg, S8_MIN, S8_MAX); 5681 reg_set_srange32(reg, S8_MIN, S8_MAX); 5682 } else if (size == 2) { 5683 reg_set_srange64(reg, S16_MIN, S16_MAX); 5684 reg_set_srange32(reg, S16_MIN, S16_MAX); 5685 } else { 5686 /* size == 4 */ 5687 reg_set_srange64(reg, S32_MIN, S32_MAX); 5688 reg_set_srange32(reg, S32_MIN, S32_MAX); 5689 } 5690 reg->var_off = tnum_unknown; 5691 } 5692 5693 static void coerce_reg_to_size_sx(struct bpf_reg_state *reg, int size) 5694 { 5695 s64 init_s64_max, init_s64_min, s64_max, s64_min, u64_cval; 5696 u64 top_smax_value, top_smin_value; 5697 u64 num_bits = size * 8; 5698 5699 if (tnum_is_const(reg->var_off)) { 5700 u64_cval = reg->var_off.value; 5701 if (size == 1) 5702 reg->var_off = tnum_const((s8)u64_cval); 5703 else if (size == 2) 5704 reg->var_off = tnum_const((s16)u64_cval); 5705 else 5706 /* size == 4 */ 5707 reg->var_off = tnum_const((s32)u64_cval); 5708 5709 u64_cval = reg->var_off.value; 5710 reg->r64 = cnum64_from_urange(u64_cval, u64_cval); 5711 reg->r32 = cnum32_from_urange((u32)u64_cval, (u32)u64_cval); 5712 return; 5713 } 5714 5715 top_smax_value = ((u64)reg_smax(reg) >> num_bits) << num_bits; 5716 top_smin_value = ((u64)reg_smin(reg) >> num_bits) << num_bits; 5717 5718 if (top_smax_value != top_smin_value) 5719 goto out; 5720 5721 /* find the s64_min and s64_min after sign extension */ 5722 if (size == 1) { 5723 init_s64_max = (s8)reg_smax(reg); 5724 init_s64_min = (s8)reg_smin(reg); 5725 } else if (size == 2) { 5726 init_s64_max = (s16)reg_smax(reg); 5727 init_s64_min = (s16)reg_smin(reg); 5728 } else { 5729 init_s64_max = (s32)reg_smax(reg); 5730 init_s64_min = (s32)reg_smin(reg); 5731 } 5732 5733 s64_max = max(init_s64_max, init_s64_min); 5734 s64_min = min(init_s64_max, init_s64_min); 5735 5736 /* both of s64_max/s64_min positive or negative */ 5737 if ((s64_max >= 0) == (s64_min >= 0)) { 5738 reg_set_srange64(reg, s64_min, s64_max); 5739 reg_set_srange32(reg, s64_min, s64_max); 5740 reg->var_off = tnum_range(s64_min, s64_max); 5741 return; 5742 } 5743 5744 out: 5745 set_sext64_default_val(reg, size); 5746 } 5747 5748 static void set_sext32_default_val(struct bpf_reg_state *reg, int size) 5749 { 5750 if (size == 1) 5751 reg_set_srange32(reg, S8_MIN, S8_MAX); 5752 else 5753 /* size == 2 */ 5754 reg_set_srange32(reg, S16_MIN, S16_MAX); 5755 reg->var_off = tnum_subreg(tnum_unknown); 5756 } 5757 5758 static void coerce_subreg_to_size_sx(struct bpf_reg_state *reg, int size) 5759 { 5760 s32 init_s32_max, init_s32_min, s32_max, s32_min, u32_val; 5761 u32 top_smax_value, top_smin_value; 5762 u32 num_bits = size * 8; 5763 5764 if (tnum_is_const(reg->var_off)) { 5765 u32_val = reg->var_off.value; 5766 if (size == 1) 5767 reg->var_off = tnum_const((s8)u32_val); 5768 else 5769 reg->var_off = tnum_const((s16)u32_val); 5770 5771 u32_val = reg->var_off.value; 5772 reg_set_srange32(reg, u32_val, u32_val); 5773 return; 5774 } 5775 5776 top_smax_value = ((u32)reg_s32_max(reg) >> num_bits) << num_bits; 5777 top_smin_value = ((u32)reg_s32_min(reg) >> num_bits) << num_bits; 5778 5779 if (top_smax_value != top_smin_value) 5780 goto out; 5781 5782 /* find the s32_min and s32_min after sign extension */ 5783 if (size == 1) { 5784 init_s32_max = (s8)reg_s32_max(reg); 5785 init_s32_min = (s8)reg_s32_min(reg); 5786 } else { 5787 /* size == 2 */ 5788 init_s32_max = (s16)reg_s32_max(reg); 5789 init_s32_min = (s16)reg_s32_min(reg); 5790 } 5791 s32_max = max(init_s32_max, init_s32_min); 5792 s32_min = min(init_s32_max, init_s32_min); 5793 5794 if ((s32_min >= 0) == (s32_max >= 0)) { 5795 reg_set_srange32(reg, s32_min, s32_max); 5796 reg->var_off = tnum_subreg(tnum_range(s32_min, s32_max)); 5797 return; 5798 } 5799 5800 out: 5801 set_sext32_default_val(reg, size); 5802 } 5803 5804 bool bpf_map_is_rdonly(const struct bpf_map *map) 5805 { 5806 /* A map is considered read-only if the following condition are true: 5807 * 5808 * 1) BPF program side cannot change any of the map content. The 5809 * BPF_F_RDONLY_PROG flag is throughout the lifetime of a map 5810 * and was set at map creation time. 5811 * 2) The map value(s) have been initialized from user space by a 5812 * loader and then "frozen", such that no new map update/delete 5813 * operations from syscall side are possible for the rest of 5814 * the map's lifetime from that point onwards. 5815 * 3) Any parallel/pending map update/delete operations from syscall 5816 * side have been completed. Only after that point, it's safe to 5817 * assume that map value(s) are immutable. 5818 */ 5819 return (map->map_flags & BPF_F_RDONLY_PROG) && 5820 READ_ONCE(map->frozen) && 5821 !bpf_map_write_active(map); 5822 } 5823 5824 int bpf_map_direct_read(struct bpf_map *map, int off, int size, u64 *val, 5825 bool is_ldsx) 5826 { 5827 void *ptr; 5828 u64 addr; 5829 int err; 5830 5831 if (map->map_type == BPF_MAP_TYPE_INSN_ARRAY || map->map_type == BPF_MAP_TYPE_PERCPU_ARRAY) 5832 return -EINVAL; 5833 err = map->ops->map_direct_value_addr(map, &addr, off); 5834 if (err) 5835 return err; 5836 ptr = (void *)(long)addr + off; 5837 5838 switch (size) { 5839 case sizeof(u8): 5840 *val = is_ldsx ? (s64)*(s8 *)ptr : (u64)*(u8 *)ptr; 5841 break; 5842 case sizeof(u16): 5843 *val = is_ldsx ? (s64)*(s16 *)ptr : (u64)*(u16 *)ptr; 5844 break; 5845 case sizeof(u32): 5846 *val = is_ldsx ? (s64)*(s32 *)ptr : (u64)*(u32 *)ptr; 5847 break; 5848 case sizeof(u64): 5849 *val = *(u64 *)ptr; 5850 break; 5851 default: 5852 return -EINVAL; 5853 } 5854 return 0; 5855 } 5856 5857 #define BTF_TYPE_SAFE_RCU(__type) __PASTE(__type, __safe_rcu) 5858 #define BTF_TYPE_SAFE_RCU_OR_NULL(__type) __PASTE(__type, __safe_rcu_or_null) 5859 #define BTF_TYPE_SAFE_TRUSTED(__type) __PASTE(__type, __safe_trusted) 5860 #define BTF_TYPE_SAFE_TRUSTED_OR_NULL(__type) __PASTE(__type, __safe_trusted_or_null) 5861 5862 /* 5863 * Allow list few fields as RCU trusted or full trusted. 5864 * This logic doesn't allow mix tagging and will be removed once GCC supports 5865 * btf_type_tag. 5866 */ 5867 5868 /* RCU trusted: these fields are trusted in RCU CS and never NULL */ 5869 BTF_TYPE_SAFE_RCU(struct task_struct) { 5870 const cpumask_t *cpus_ptr; 5871 struct css_set __rcu *cgroups; 5872 struct task_struct __rcu *real_parent; 5873 struct task_struct *group_leader; 5874 }; 5875 5876 BTF_TYPE_SAFE_RCU(struct cgroup) { 5877 /* cgrp->kn is always accessible as documented in kernel/cgroup/cgroup.c */ 5878 struct kernfs_node *kn; 5879 }; 5880 5881 BTF_TYPE_SAFE_RCU(struct css_set) { 5882 struct cgroup *dfl_cgrp; 5883 }; 5884 5885 BTF_TYPE_SAFE_RCU(struct cgroup_subsys_state) { 5886 struct cgroup *cgroup; 5887 }; 5888 5889 /* RCU trusted: these fields are trusted in RCU CS and can be NULL */ 5890 BTF_TYPE_SAFE_RCU_OR_NULL(struct mm_struct) { 5891 struct file __rcu *exe_file; 5892 #ifdef CONFIG_MEMCG 5893 struct task_struct __rcu *owner; 5894 #endif 5895 }; 5896 5897 /* skb->sk, req->sk are not RCU protected, but we mark them as such 5898 * because bpf prog accessible sockets are SOCK_RCU_FREE. 5899 */ 5900 BTF_TYPE_SAFE_RCU_OR_NULL(struct sk_buff) { 5901 struct sock *sk; 5902 }; 5903 5904 BTF_TYPE_SAFE_RCU_OR_NULL(struct request_sock) { 5905 struct sock *sk; 5906 }; 5907 5908 /* full trusted: these fields are trusted even outside of RCU CS and never NULL */ 5909 BTF_TYPE_SAFE_TRUSTED(struct bpf_iter_meta) { 5910 struct seq_file *seq; 5911 }; 5912 5913 BTF_TYPE_SAFE_TRUSTED(struct bpf_iter__task) { 5914 struct bpf_iter_meta *meta; 5915 struct task_struct *task; 5916 }; 5917 5918 BTF_TYPE_SAFE_TRUSTED(struct linux_binprm) { 5919 struct file *file; 5920 }; 5921 5922 BTF_TYPE_SAFE_TRUSTED(struct file) { 5923 struct inode *f_inode; 5924 }; 5925 5926 BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct dentry) { 5927 struct inode *d_inode; 5928 }; 5929 5930 BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct socket) { 5931 struct sock *sk; 5932 }; 5933 5934 BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct vm_area_struct) { 5935 struct mm_struct *vm_mm; 5936 struct file *vm_file; 5937 }; 5938 5939 static bool type_is_rcu(struct bpf_verifier_env *env, 5940 struct bpf_reg_state *reg, 5941 const char *field_name, u32 btf_id) 5942 { 5943 BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU(struct task_struct)); 5944 BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU(struct cgroup)); 5945 BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU(struct css_set)); 5946 BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU(struct cgroup_subsys_state)); 5947 5948 return btf_nested_type_is_trusted(&env->log, reg, field_name, btf_id, "__safe_rcu"); 5949 } 5950 5951 static bool type_is_rcu_or_null(struct bpf_verifier_env *env, 5952 struct bpf_reg_state *reg, 5953 const char *field_name, u32 btf_id) 5954 { 5955 BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU_OR_NULL(struct mm_struct)); 5956 BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU_OR_NULL(struct sk_buff)); 5957 BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU_OR_NULL(struct request_sock)); 5958 5959 return btf_nested_type_is_trusted(&env->log, reg, field_name, btf_id, "__safe_rcu_or_null"); 5960 } 5961 5962 static bool type_is_trusted(struct bpf_verifier_env *env, 5963 struct bpf_reg_state *reg, 5964 const char *field_name, u32 btf_id) 5965 { 5966 BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED(struct bpf_iter_meta)); 5967 BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED(struct bpf_iter__task)); 5968 BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED(struct linux_binprm)); 5969 BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED(struct file)); 5970 5971 return btf_nested_type_is_trusted(&env->log, reg, field_name, btf_id, "__safe_trusted"); 5972 } 5973 5974 static bool type_is_trusted_or_null(struct bpf_verifier_env *env, 5975 struct bpf_reg_state *reg, 5976 const char *field_name, u32 btf_id) 5977 { 5978 BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct socket)); 5979 BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct dentry)); 5980 BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct vm_area_struct)); 5981 5982 return btf_nested_type_is_trusted(&env->log, reg, field_name, btf_id, 5983 "__safe_trusted_or_null"); 5984 } 5985 5986 static int check_ptr_to_btf_access(struct bpf_verifier_env *env, 5987 struct bpf_reg_state *regs, struct bpf_reg_state *reg, 5988 argno_t argno, int off, int size, 5989 enum bpf_access_type atype, 5990 int value_regno) 5991 { 5992 const struct btf_type *t = btf_type_by_id(reg->btf, reg->btf_id); 5993 const char *tname = btf_name_by_offset(reg->btf, t->name_off); 5994 const char *field_name = NULL; 5995 enum bpf_type_flag flag = 0; 5996 u32 btf_id = 0; 5997 int ret; 5998 5999 if (!env->allow_ptr_leaks) { 6000 verbose(env, 6001 "'struct %s' access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n", 6002 tname); 6003 return -EPERM; 6004 } 6005 if (!env->prog->gpl_compatible && btf_is_kernel(reg->btf)) { 6006 verbose(env, 6007 "Cannot access kernel 'struct %s' from non-GPL compatible program\n", 6008 tname); 6009 return -EINVAL; 6010 } 6011 6012 if (!tnum_is_const(reg->var_off)) { 6013 char tn_buf[48]; 6014 6015 tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off); 6016 verbose(env, 6017 "%s is ptr_%s invalid variable offset: off=%d, var_off=%s\n", 6018 reg_arg_name(env, argno), tname, off, tn_buf); 6019 return -EACCES; 6020 } 6021 6022 off += reg->var_off.value; 6023 6024 if (off < 0) { 6025 verbose(env, 6026 "%s is ptr_%s invalid negative access: off=%d\n", 6027 reg_arg_name(env, argno), tname, off); 6028 return -EACCES; 6029 } 6030 6031 if (reg->type & MEM_USER) { 6032 verbose(env, 6033 "%s is ptr_%s access user memory: off=%d\n", 6034 reg_arg_name(env, argno), tname, off); 6035 return -EACCES; 6036 } 6037 6038 if (reg->type & MEM_PERCPU) { 6039 verbose(env, 6040 "%s is ptr_%s access percpu memory: off=%d\n", 6041 reg_arg_name(env, argno), tname, off); 6042 return -EACCES; 6043 } 6044 6045 if (atype != BPF_READ && bpf_may_fault_on_deref(reg->type)) { 6046 verbose(env, "only read is supported\n"); 6047 return -EACCES; 6048 } 6049 6050 if (env->ops->btf_struct_access && !type_is_alloc(reg->type) && atype == BPF_WRITE) { 6051 if (!btf_is_kernel(reg->btf)) { 6052 verifier_bug(env, "reg->btf must be kernel btf"); 6053 return -EFAULT; 6054 } 6055 ret = env->ops->btf_struct_access(&env->log, reg, off, size); 6056 if (ret < 0) 6057 verbose(env, 6058 "%s cannot write into ptr_%s at off=%d size=%d\n", 6059 reg_arg_name(env, argno), tname, off, size); 6060 } else { 6061 /* Writes are permitted with default btf_struct_access for 6062 * program allocated objects (which always have id > 0). 6063 */ 6064 if (atype != BPF_READ && !type_is_ptr_alloc_obj(reg->type)) { 6065 verbose(env, "only read is supported\n"); 6066 return -EACCES; 6067 } 6068 6069 /* 6070 * A fault-prone allocated object may still be read through a 6071 * BPF_PROBE_MEM load after its lifetime protection ends. Writes 6072 * through such pointers were rejected above. 6073 */ 6074 if (type_is_alloc(reg->type) && !bpf_may_fault_on_deref(reg->type) && 6075 !type_is_non_owning_ref(reg->type) && 6076 !(reg->type & MEM_RCU) && !reg_is_referenced(env, reg)) { 6077 verifier_bug(env, "allocated object must have a referenced id"); 6078 return -EFAULT; 6079 } 6080 6081 ret = btf_struct_access(&env->log, reg, off, size, atype, &btf_id, &flag, &field_name); 6082 } 6083 6084 if (ret < 0) 6085 return ret; 6086 6087 if (ret != PTR_TO_BTF_ID) { 6088 /* just mark; */ 6089 6090 } else if (type_flag(reg->type) & PTR_UNTRUSTED) { 6091 /* If this is an untrusted pointer, all pointers formed by walking it 6092 * also inherit the untrusted flag. 6093 */ 6094 flag = PTR_UNTRUSTED; 6095 6096 } else if (is_trusted_reg(env, reg) || is_rcu_reg(reg)) { 6097 /* By default any pointer obtained from walking a trusted pointer is no 6098 * longer trusted, unless the field being accessed has explicitly been 6099 * marked as inheriting its parent's state of trust (either full or RCU). 6100 * For example: 6101 * 'cgroups' pointer is untrusted if task->cgroups dereference 6102 * happened in a sleepable program outside of bpf_rcu_read_lock() 6103 * section. In a non-sleepable program it's trusted while in RCU CS (aka MEM_RCU). 6104 * Note bpf_rcu_read_unlock() converts MEM_RCU pointers to PTR_UNTRUSTED. 6105 * 6106 * A regular RCU-protected pointer with __rcu tag can also be deemed 6107 * trusted if we are in an RCU CS. Such pointer can be NULL. 6108 */ 6109 if (type_is_trusted(env, reg, field_name, btf_id)) { 6110 flag |= PTR_TRUSTED; 6111 } else if (type_is_trusted_or_null(env, reg, field_name, btf_id)) { 6112 flag |= PTR_TRUSTED | PTR_MAYBE_NULL; 6113 } else if (in_rcu_cs(env) && !type_may_be_null(reg->type)) { 6114 if (type_is_rcu(env, reg, field_name, btf_id)) { 6115 /* ignore __rcu tag and mark it MEM_RCU */ 6116 flag |= MEM_RCU; 6117 } else if (flag & MEM_RCU || 6118 type_is_rcu_or_null(env, reg, field_name, btf_id)) { 6119 /* __rcu tagged pointers can be NULL */ 6120 flag |= MEM_RCU | PTR_MAYBE_NULL; 6121 6122 /* We always trust them */ 6123 if (type_is_rcu_or_null(env, reg, field_name, btf_id) && 6124 flag & PTR_UNTRUSTED) 6125 flag &= ~PTR_UNTRUSTED; 6126 } else if (flag & (MEM_PERCPU | MEM_USER)) { 6127 /* keep as-is */ 6128 } else { 6129 /* walking unknown pointers yields old deprecated PTR_TO_BTF_ID */ 6130 clear_trusted_flags(&flag); 6131 } 6132 } else { 6133 /* 6134 * If not in RCU CS or MEM_RCU pointer can be NULL then 6135 * aggressively mark as untrusted otherwise such 6136 * pointers will be plain PTR_TO_BTF_ID without flags 6137 * and will be allowed to be passed into helpers for 6138 * compat reasons. 6139 */ 6140 flag = PTR_UNTRUSTED; 6141 } 6142 } else { 6143 /* Old compat. Deprecated */ 6144 clear_trusted_flags(&flag); 6145 } 6146 6147 if (atype == BPF_READ && value_regno >= 0) { 6148 ret = mark_btf_ld_reg(env, regs, value_regno, ret, reg->btf, btf_id, flag); 6149 if (ret < 0) 6150 return ret; 6151 } 6152 6153 return 0; 6154 } 6155 6156 static int check_ptr_to_map_access(struct bpf_verifier_env *env, 6157 struct bpf_reg_state *regs, struct bpf_reg_state *reg, 6158 argno_t argno, int off, int size, 6159 enum bpf_access_type atype, 6160 int value_regno) 6161 { 6162 struct bpf_map *map = reg->map_ptr; 6163 struct bpf_reg_state map_reg; 6164 enum bpf_type_flag flag = 0; 6165 const struct btf_type *t; 6166 const char *tname; 6167 u32 btf_id; 6168 int ret; 6169 6170 if (!btf_vmlinux) { 6171 verbose(env, "map_ptr access not supported without CONFIG_DEBUG_INFO_BTF\n"); 6172 return -ENOTSUPP; 6173 } 6174 6175 if (!map->ops->map_btf_id || !*map->ops->map_btf_id) { 6176 verbose(env, "map_ptr access not supported for map type %d\n", 6177 map->map_type); 6178 return -ENOTSUPP; 6179 } 6180 6181 t = btf_type_by_id(btf_vmlinux, *map->ops->map_btf_id); 6182 tname = btf_name_by_offset(btf_vmlinux, t->name_off); 6183 6184 if (!env->allow_ptr_leaks) { 6185 verbose(env, 6186 "'struct %s' access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n", 6187 tname); 6188 return -EPERM; 6189 } 6190 6191 if (off < 0) { 6192 verbose(env, "%s is %s invalid negative access: off=%d\n", 6193 reg_arg_name(env, argno), tname, off); 6194 return -EACCES; 6195 } 6196 6197 if (atype != BPF_READ) { 6198 verbose(env, "only read from %s is supported\n", tname); 6199 return -EACCES; 6200 } 6201 6202 /* Simulate access to a PTR_TO_BTF_ID */ 6203 memset(&map_reg, 0, sizeof(map_reg)); 6204 ret = mark_btf_ld_reg(env, &map_reg, 0, PTR_TO_BTF_ID, 6205 btf_vmlinux, *map->ops->map_btf_id, 0); 6206 if (ret < 0) 6207 return ret; 6208 ret = btf_struct_access(&env->log, &map_reg, off, size, atype, &btf_id, &flag, NULL); 6209 if (ret < 0) 6210 return ret; 6211 6212 if (value_regno >= 0) { 6213 ret = mark_btf_ld_reg(env, regs, value_regno, ret, btf_vmlinux, btf_id, flag); 6214 if (ret < 0) 6215 return ret; 6216 } 6217 6218 return 0; 6219 } 6220 6221 /* Check that the stack access at the given offset is within bounds. The 6222 * maximum valid offset is -1. 6223 * 6224 * The minimum valid offset is -MAX_BPF_STACK for writes, and 6225 * -state->allocated_stack for reads. 6226 */ 6227 static int check_stack_slot_within_bounds(struct bpf_verifier_env *env, 6228 s64 off, 6229 struct bpf_func_state *state, 6230 enum bpf_access_type t) 6231 { 6232 int min_valid_off; 6233 6234 if (t == BPF_WRITE || env->allow_uninit_stack) 6235 min_valid_off = -MAX_BPF_STACK; 6236 else 6237 min_valid_off = -state->allocated_stack; 6238 6239 if (off < min_valid_off || off > -1) 6240 return -EACCES; 6241 return 0; 6242 } 6243 6244 /* Check that the stack access at 'regno + off' falls within the maximum stack 6245 * bounds. 6246 * 6247 * 'off' includes `regno->offset`, but not its dynamic part (if any). 6248 */ 6249 static int check_stack_access_within_bounds( 6250 struct bpf_verifier_env *env, struct bpf_reg_state *reg, 6251 argno_t argno, int off, int access_size, 6252 enum bpf_access_type type) 6253 { 6254 struct bpf_func_state *state = bpf_func(env, reg); 6255 s64 min_off, max_off; 6256 int err; 6257 char *err_extra; 6258 6259 if (type == BPF_READ) 6260 err_extra = " read from"; 6261 else 6262 err_extra = " write to"; 6263 6264 if (tnum_is_const(reg->var_off)) { 6265 min_off = (s64)reg->var_off.value + off; 6266 max_off = min_off + access_size; 6267 } else { 6268 if (reg_smax(reg) >= BPF_MAX_VAR_OFF || 6269 reg_smin(reg) <= -BPF_MAX_VAR_OFF) { 6270 verbose(env, "invalid unbounded variable-offset%s stack %s\n", 6271 err_extra, reg_arg_name(env, argno)); 6272 return -EACCES; 6273 } 6274 min_off = reg_smin(reg) + off; 6275 max_off = reg_smax(reg) + off + access_size; 6276 } 6277 6278 err = check_stack_slot_within_bounds(env, min_off, state, type); 6279 if (!err && max_off > 0) 6280 err = -EINVAL; /* out of stack access into non-negative offsets */ 6281 if (!err && access_size < 0) 6282 /* access_size should not be negative (or overflow an int); others checks 6283 * along the way should have prevented such an access. 6284 */ 6285 err = -EFAULT; /* invalid negative access size; integer overflow? */ 6286 6287 if (err) { 6288 if (tnum_is_const(reg->var_off)) { 6289 verbose(env, "invalid%s stack %s off=%lld size=%d\n", 6290 err_extra, reg_arg_name(env, argno), min_off, access_size); 6291 } else { 6292 char tn_buf[48]; 6293 6294 tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off); 6295 verbose(env, "invalid variable-offset%s stack %s var_off=%s off=%d size=%d\n", 6296 err_extra, reg_arg_name(env, argno), tn_buf, off, access_size); 6297 } 6298 return err; 6299 } 6300 6301 /* Note that there is no stack access with offset zero, so the needed stack 6302 * size is -min_off, not -min_off+1. 6303 */ 6304 return grow_stack_state(env, state, -min_off /* size */); 6305 } 6306 6307 static bool get_func_retval_range(struct bpf_prog *prog, 6308 struct bpf_retval_range *range) 6309 { 6310 if (prog->type == BPF_PROG_TYPE_LSM && 6311 prog->expected_attach_type == BPF_LSM_MAC && 6312 !bpf_lsm_get_retval_range(prog, range)) { 6313 return true; 6314 } 6315 return false; 6316 } 6317 6318 static void add_scalar_to_reg(struct bpf_reg_state *dst_reg, s64 val) 6319 { 6320 struct bpf_reg_state fake_reg; 6321 6322 if (!val) 6323 return; 6324 6325 fake_reg.type = SCALAR_VALUE; 6326 __mark_reg_known(&fake_reg, val); 6327 6328 scalar32_min_max_add(dst_reg, &fake_reg); 6329 scalar_min_max_add(dst_reg, &fake_reg); 6330 dst_reg->var_off = tnum_add(dst_reg->var_off, fake_reg.var_off); 6331 6332 reg_bounds_sync(dst_reg); 6333 } 6334 6335 static int check_map_mem_read(struct bpf_verifier_env *env, struct bpf_reg_state *reg, int off, 6336 int bpf_size, int value_regno, bool is_ldsx) 6337 { 6338 struct bpf_reg_state *regs = cur_regs(env); 6339 int size = bpf_size_to_bytes(bpf_size); 6340 struct bpf_map *map = reg->map_ptr; 6341 6342 switch (map->map_type) { 6343 case BPF_MAP_TYPE_INSN_ARRAY: 6344 if (bpf_size != BPF_DW) { 6345 verbose(env, "Invalid read of %d bytes from insn_array\n", size); 6346 return -EACCES; 6347 } 6348 regs[value_regno] = *reg; 6349 add_scalar_to_reg(®s[value_regno], off); 6350 regs[value_regno].type = PTR_TO_INSN; 6351 return 0; 6352 case BPF_MAP_TYPE_PERCPU_ARRAY: 6353 goto reg_unknown; 6354 default: 6355 break; 6356 } 6357 6358 /* If map is read-only, track its contents as scalars. */ 6359 if (tnum_is_const(reg->var_off) && 6360 bpf_map_is_rdonly(map) && 6361 map->ops->map_direct_value_addr) { 6362 int map_off = off + reg->var_off.value; 6363 u64 val = 0; 6364 int err; 6365 6366 err = bpf_map_direct_read(map, map_off, size, &val, is_ldsx); 6367 if (err) 6368 return err; 6369 6370 regs[value_regno].type = SCALAR_VALUE; 6371 __mark_reg_known(®s[value_regno], val); 6372 return 0; 6373 } 6374 6375 reg_unknown: 6376 mark_reg_unknown(env, regs, value_regno); 6377 return 0; 6378 } 6379 6380 /* check whether memory at (regno + off) is accessible for t = (read | write) 6381 * if t==write, value_regno is a register which value is stored into memory 6382 * if t==read, value_regno is a register which will receive the value from memory 6383 * if t==write && value_regno==-1, some unknown value is stored into memory 6384 * if t==read && value_regno==-1, don't care what we read from memory 6385 */ 6386 static int check_mem_access(struct bpf_verifier_env *env, int insn_idx, struct bpf_reg_state *reg, argno_t argno, 6387 int off, int bpf_size, enum bpf_access_type t, 6388 int value_regno, bool strict_alignment_once, bool is_ldsx) 6389 { 6390 struct bpf_reg_state *regs = cur_regs(env); 6391 int size, err = 0; 6392 6393 size = bpf_size_to_bytes(bpf_size); 6394 if (size < 0) 6395 return size; 6396 6397 err = check_ptr_alignment(env, reg, off, size, strict_alignment_once); 6398 if (err) 6399 return err; 6400 6401 if (reg->type == PTR_TO_MAP_KEY) { 6402 if (t == BPF_WRITE) { 6403 verbose(env, "write to change key %s not allowed\n", 6404 reg_arg_name(env, argno)); 6405 return -EACCES; 6406 } 6407 6408 err = check_mem_region_access(env, reg, argno, off, size, 6409 reg->map_ptr->key_size, false); 6410 if (err) 6411 return err; 6412 if (value_regno >= 0) 6413 mark_reg_unknown(env, regs, value_regno); 6414 } else if (reg->type == PTR_TO_MAP_VALUE) { 6415 struct btf_field *kptr_field = NULL; 6416 6417 if (t == BPF_WRITE && value_regno >= 0 && 6418 is_pointer_value(env, value_regno)) { 6419 verbose(env, "R%d leaks addr into map\n", value_regno); 6420 return -EACCES; 6421 } 6422 err = check_map_access_type(env, reg, off, size, t); 6423 if (err) 6424 return err; 6425 err = check_map_access(env, reg, argno, off, size, false, ACCESS_DIRECT); 6426 if (err) 6427 return err; 6428 if (tnum_is_const(reg->var_off)) 6429 kptr_field = btf_record_find(reg->map_ptr->record, 6430 off + reg->var_off.value, BPF_KPTR | BPF_UPTR); 6431 if (kptr_field) { 6432 err = check_map_kptr_access(env, value_regno, insn_idx, kptr_field); 6433 } else if (t == BPF_READ && value_regno >= 0) { 6434 err = check_map_mem_read(env, reg, off, bpf_size, value_regno, is_ldsx); 6435 } 6436 } else if (base_type(reg->type) == PTR_TO_MEM) { 6437 bool rdonly_mem = type_is_rdonly_mem(reg->type); 6438 bool rdonly_untrusted = rdonly_mem && (reg->type & PTR_UNTRUSTED); 6439 6440 if (type_may_be_null(reg->type)) { 6441 verbose(env, "%s invalid mem access '%s'\n", reg_arg_name(env, argno), 6442 reg_type_str(env, reg->type)); 6443 bpf_diag_invalid_deref(env, insn_idx, reg_from_argno(argno), 6444 reg_arg_name(env, argno), reg, 6445 BPF_DIAG_DEREF_NULLABLE_PTR, 0); 6446 return -EACCES; 6447 } 6448 6449 if (t == BPF_WRITE && rdonly_mem) { 6450 verbose(env, "%s cannot write into %s\n", 6451 reg_arg_name(env, argno), reg_type_str(env, reg->type)); 6452 return -EACCES; 6453 } 6454 6455 if (t == BPF_WRITE && value_regno >= 0 && 6456 is_pointer_value(env, value_regno)) { 6457 verbose(env, "R%d leaks addr into mem\n", value_regno); 6458 return -EACCES; 6459 } 6460 6461 if (rdonly_untrusted && !env->allow_ptr_leaks) { 6462 verbose(env, "%s access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n", 6463 reg_type_str(env, reg->type)); 6464 bpf_diag_policy(env, insn_idx, "read from untrusted read-only memory", 6465 "the access requires CAP_PERFMON", 6466 "Load the program with CAP_PERFMON, or avoid dereferencing untrusted pointers."); 6467 return -EPERM; 6468 } 6469 6470 /* 6471 * Accesses to untrusted PTR_TO_MEM are done through probe 6472 * instructions, hence no need to check bounds in that case. 6473 */ 6474 if (!rdonly_untrusted) 6475 err = check_mem_region_access(env, reg, argno, off, size, 6476 reg->mem_size, false); 6477 if (!err && value_regno >= 0 && (t == BPF_READ || rdonly_mem)) 6478 mark_reg_unknown(env, regs, value_regno); 6479 } else if (reg->type == PTR_TO_CTX) { 6480 struct bpf_insn_access_aux info = { 6481 .reg_type = SCALAR_VALUE, 6482 .is_ldsx = is_ldsx, 6483 .log = &env->log, 6484 }; 6485 struct bpf_retval_range range; 6486 6487 if (t == BPF_WRITE && value_regno >= 0 && 6488 is_pointer_value(env, value_regno)) { 6489 verbose(env, "R%d leaks addr into ctx\n", value_regno); 6490 return -EACCES; 6491 } 6492 6493 err = check_ctx_access(env, insn_idx, reg, argno, off, size, t, &info); 6494 if (!err && t == BPF_READ && value_regno >= 0) { 6495 /* ctx access returns either a scalar, or a 6496 * PTR_TO_PACKET[_META,_END]. In the latter 6497 * case, we know the offset is zero. 6498 */ 6499 if (info.reg_type == SCALAR_VALUE) { 6500 if (info.is_retval && get_func_retval_range(env->prog, &range)) { 6501 mark_reg_unknown(env, regs, value_regno); 6502 err = __mark_reg_s32_range(env, regs, value_regno, 6503 range.minval, range.maxval); 6504 if (err) 6505 return err; 6506 } else { 6507 mark_reg_unknown(env, regs, value_regno); 6508 } 6509 } else { 6510 mark_reg_known_zero(env, regs, 6511 value_regno); 6512 if (base_type(info.reg_type) == PTR_TO_BTF_ID) { 6513 regs[value_regno].btf = info.btf; 6514 regs[value_regno].btf_id = info.btf_id; 6515 regs[value_regno].id = info.ref_id; 6516 } 6517 if (type_may_be_null(info.reg_type) && !regs[value_regno].id) 6518 regs[value_regno].id = ++env->id_gen; 6519 } 6520 regs[value_regno].type = info.reg_type; 6521 } 6522 6523 } else if (reg->type == PTR_TO_STACK) { 6524 /* Basic bounds checks. */ 6525 err = check_stack_access_within_bounds(env, reg, argno, off, size, t); 6526 if (err) 6527 return err; 6528 6529 if (t == BPF_READ) 6530 err = check_stack_read(env, reg, argno, off, size, 6531 value_regno); 6532 else 6533 err = check_stack_write(env, reg, off, size, 6534 value_regno, insn_idx); 6535 } else if (reg_is_pkt_pointer(reg)) { 6536 if (t == BPF_WRITE && !may_access_direct_pkt_data(env, NULL, t)) { 6537 verbose(env, "cannot write into packet\n"); 6538 return -EACCES; 6539 } 6540 if (t == BPF_WRITE && value_regno >= 0 && 6541 is_pointer_value(env, value_regno)) { 6542 verbose(env, "R%d leaks addr into packet\n", 6543 value_regno); 6544 return -EACCES; 6545 } 6546 err = check_packet_access(env, reg, argno, off, size, false); 6547 if (!err && t == BPF_READ && value_regno >= 0) 6548 mark_reg_unknown(env, regs, value_regno); 6549 } else if (reg->type == PTR_TO_FLOW_KEYS) { 6550 if (t == BPF_WRITE && value_regno >= 0 && 6551 is_pointer_value(env, value_regno)) { 6552 verbose(env, "R%d leaks addr into flow keys\n", 6553 value_regno); 6554 return -EACCES; 6555 } 6556 6557 err = check_flow_keys_access(env, reg, argno, off, size); 6558 if (!err && t == BPF_READ && value_regno >= 0) 6559 mark_reg_unknown(env, regs, value_regno); 6560 } else if (type_is_sk_pointer(reg->type)) { 6561 if (t == BPF_WRITE) { 6562 verbose(env, "%s cannot write into %s\n", 6563 reg_arg_name(env, argno), reg_type_str(env, reg->type)); 6564 return -EACCES; 6565 } 6566 err = check_sock_access(env, insn_idx, reg, argno, off, size, t); 6567 if (!err && value_regno >= 0) 6568 mark_reg_unknown(env, regs, value_regno); 6569 } else if (reg->type == PTR_TO_TP_BUFFER) { 6570 err = check_tp_buffer_access(env, reg, argno, off, size); 6571 if (!err && t == BPF_READ && value_regno >= 0) 6572 mark_reg_unknown(env, regs, value_regno); 6573 } else if (base_type(reg->type) == PTR_TO_BTF_ID && 6574 !type_may_be_null(reg->type)) { 6575 err = check_ptr_to_btf_access(env, regs, reg, argno, off, size, t, 6576 value_regno); 6577 } else if (reg->type == CONST_PTR_TO_MAP) { 6578 err = check_ptr_to_map_access(env, regs, reg, argno, off, size, t, 6579 value_regno); 6580 } else if (base_type(reg->type) == PTR_TO_BUF && 6581 !type_may_be_null(reg->type)) { 6582 bool rdonly_mem = type_is_rdonly_mem(reg->type); 6583 u32 *max_access; 6584 6585 if (rdonly_mem) { 6586 if (t == BPF_WRITE) { 6587 verbose(env, "%s cannot write into %s\n", 6588 reg_arg_name(env, argno), reg_type_str(env, reg->type)); 6589 return -EACCES; 6590 } 6591 max_access = &env->prog->aux->max_rdonly_access; 6592 } else { 6593 max_access = &env->prog->aux->max_rdwr_access; 6594 } 6595 6596 err = check_buffer_access(env, reg, argno, off, size, false, 6597 max_access); 6598 6599 if (!err && value_regno >= 0 && (rdonly_mem || t == BPF_READ)) 6600 mark_reg_unknown(env, regs, value_regno); 6601 } else if (reg->type == PTR_TO_ARENA) { 6602 if (t == BPF_READ && value_regno >= 0) 6603 mark_reg_unknown(env, regs, value_regno); 6604 } else { 6605 enum bpf_diag_invalid_deref_kind kind = BPF_DIAG_DEREF_INVALID_PTR; 6606 6607 verbose(env, "%s invalid mem access '%s'\n", reg_arg_name(env, argno), 6608 reg_type_str(env, reg->type)); 6609 if (reg->type == SCALAR_VALUE) 6610 kind = BPF_DIAG_DEREF_SCALAR; 6611 else if (type_may_be_null(reg->type)) 6612 kind = BPF_DIAG_DEREF_NULLABLE_PTR; 6613 bpf_diag_invalid_deref(env, insn_idx, reg_from_argno(argno), 6614 reg_arg_name(env, argno), reg, kind, 0); 6615 return -EACCES; 6616 } 6617 6618 if (!err && size < BPF_REG_SIZE && value_regno >= 0 && t == BPF_READ && 6619 regs[value_regno].type == SCALAR_VALUE) { 6620 if (!is_ldsx) { 6621 /* b/h/w load zero-extends, mark upper bits as known 0 */ 6622 coerce_reg_to_size(®s[value_regno], size); 6623 } else { 6624 /* 6625 * Sign-extension can change the register value relative 6626 * to a scalar it is linked with by id (e.g. a zero- 6627 * extending fill of the same spilled stack slot), thus 6628 * drop the shared id in that case. 6629 */ 6630 bool no_sext = reg_umax(®s[value_regno]) < 6631 (1ULL << (size * BITS_PER_BYTE - 1)); 6632 6633 coerce_reg_to_size_sx(®s[value_regno], size); 6634 if (!no_sext) 6635 clear_scalar_id(®s[value_regno]); 6636 } 6637 } 6638 return err; 6639 } 6640 6641 static int save_aux_ptr_type(struct bpf_verifier_env *env, enum bpf_reg_type type, 6642 bool allow_trust_mismatch); 6643 6644 static int check_load_mem(struct bpf_verifier_env *env, struct bpf_insn *insn, 6645 bool strict_alignment_once, bool is_ldsx, 6646 bool allow_trust_mismatch, const char *ctx) 6647 { 6648 struct bpf_verifier_state *vstate = env->cur_state; 6649 struct bpf_func_state *state = vstate->frame[vstate->curframe]; 6650 struct bpf_reg_state *regs = cur_regs(env); 6651 enum bpf_reg_type src_reg_type; 6652 int err; 6653 6654 /* Handle stack arg read */ 6655 if (is_stack_arg_ldx(insn)) { 6656 err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK); 6657 if (err) 6658 return err; 6659 return check_stack_arg_read(env, state, insn->off, insn->dst_reg); 6660 } 6661 6662 /* check src operand */ 6663 err = check_reg_arg(env, insn->src_reg, SRC_OP); 6664 if (err) 6665 return err; 6666 6667 /* check dst operand */ 6668 err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK); 6669 if (err) 6670 return err; 6671 6672 src_reg_type = regs[insn->src_reg].type; 6673 6674 /* 6675 * check_stack_read_fixed_off() may refine the modification's origin to 6676 * the source stack slot. 6677 */ 6678 bpf_diag_mod_begin(env, ®s[insn->dst_reg], NULL, BPF_DIAG_MOD_WRITE); 6679 err = check_mem_access(env, env->insn_idx, regs + insn->src_reg, argno_from_reg(insn->src_reg), insn->off, 6680 BPF_SIZE(insn->code), BPF_READ, insn->dst_reg, 6681 strict_alignment_once, is_ldsx); 6682 err = err ?: save_aux_ptr_type(env, src_reg_type, 6683 allow_trust_mismatch); 6684 err = err ?: reg_bounds_sanity_check(env, ®s[insn->dst_reg], ctx); 6685 if (!err) 6686 bpf_diag_mod_end(env); 6687 6688 return err; 6689 } 6690 6691 static int check_store_reg(struct bpf_verifier_env *env, struct bpf_insn *insn, 6692 bool strict_alignment_once) 6693 { 6694 struct bpf_verifier_state *vstate = env->cur_state; 6695 struct bpf_func_state *state = vstate->frame[vstate->curframe]; 6696 struct bpf_reg_state *regs = cur_regs(env); 6697 enum bpf_reg_type dst_reg_type; 6698 int err; 6699 6700 /* Handle stack arg write */ 6701 if (is_stack_arg_stx(insn)) { 6702 err = check_reg_arg(env, insn->src_reg, SRC_OP); 6703 if (err) 6704 return err; 6705 return check_stack_arg_write(env, state, insn->off, regs + insn->src_reg); 6706 } 6707 6708 /* check src1 operand */ 6709 err = check_reg_arg(env, insn->src_reg, SRC_OP); 6710 if (err) 6711 return err; 6712 6713 /* check src2 operand */ 6714 err = check_reg_arg(env, insn->dst_reg, SRC_OP); 6715 if (err) 6716 return err; 6717 6718 dst_reg_type = regs[insn->dst_reg].type; 6719 6720 /* Check if (dst_reg + off) is writeable. */ 6721 err = check_mem_access(env, env->insn_idx, regs + insn->dst_reg, argno_from_reg(insn->dst_reg), insn->off, 6722 BPF_SIZE(insn->code), BPF_WRITE, insn->src_reg, 6723 strict_alignment_once, false); 6724 err = err ?: save_aux_ptr_type(env, dst_reg_type, false); 6725 6726 return err; 6727 } 6728 6729 static int check_atomic_rmw(struct bpf_verifier_env *env, 6730 struct bpf_insn *insn) 6731 { 6732 struct bpf_reg_state *dst_reg; 6733 int load_reg; 6734 int err; 6735 6736 if (BPF_SIZE(insn->code) != BPF_W && BPF_SIZE(insn->code) != BPF_DW) { 6737 verbose(env, "invalid atomic operand size\n"); 6738 return -EINVAL; 6739 } 6740 6741 /* check src1 operand */ 6742 err = check_reg_arg(env, insn->src_reg, SRC_OP); 6743 if (err) 6744 return err; 6745 6746 /* check src2 operand */ 6747 err = check_reg_arg(env, insn->dst_reg, SRC_OP); 6748 if (err) 6749 return err; 6750 6751 if (insn->imm == BPF_CMPXCHG) { 6752 /* Check comparison of R0 with memory location */ 6753 const u32 aux_reg = BPF_REG_0; 6754 6755 err = check_reg_arg(env, aux_reg, SRC_OP); 6756 if (err) 6757 return err; 6758 6759 if (is_pointer_value(env, aux_reg)) { 6760 verbose(env, "R%d leaks addr into mem\n", aux_reg); 6761 return -EACCES; 6762 } 6763 } 6764 6765 if (is_pointer_value(env, insn->src_reg)) { 6766 verbose(env, "R%d leaks addr into mem\n", insn->src_reg); 6767 return -EACCES; 6768 } 6769 6770 if (!atomic_ptr_type_ok(env, insn->dst_reg, insn)) { 6771 verbose(env, "BPF_ATOMIC stores into R%d %s is not allowed\n", 6772 insn->dst_reg, 6773 reg_type_str(env, reg_state(env, insn->dst_reg)->type)); 6774 return -EACCES; 6775 } 6776 6777 load_reg = bpf_atomic_load_reg(insn); 6778 if (load_reg >= 0) { 6779 /* check and record load of old value */ 6780 err = check_reg_arg(env, load_reg, DST_OP); 6781 if (err) 6782 return err; 6783 } 6784 6785 dst_reg = cur_regs(env) + insn->dst_reg; 6786 6787 /* Check whether we can read the memory, with second call for fetch 6788 * case to simulate the register fill. 6789 */ 6790 err = check_mem_access(env, env->insn_idx, dst_reg, argno_from_reg(insn->dst_reg), insn->off, 6791 BPF_SIZE(insn->code), BPF_READ, -1, true, false); 6792 if (!err && load_reg >= 0) { 6793 bpf_diag_mod_begin(env, cur_regs(env) + load_reg, NULL, BPF_DIAG_MOD_WRITE); 6794 err = check_mem_access(env, env->insn_idx, dst_reg, argno_from_reg(insn->dst_reg), 6795 insn->off, BPF_SIZE(insn->code), 6796 BPF_READ, load_reg, true, false); 6797 if (!err) 6798 bpf_diag_mod_end(env); 6799 } 6800 if (err) 6801 return err; 6802 6803 err = save_aux_ptr_type(env, dst_reg->type, false); 6804 if (err) 6805 return err; 6806 /* Check whether we can write into the same memory. */ 6807 err = check_mem_access(env, env->insn_idx, dst_reg, argno_from_reg(insn->dst_reg), insn->off, 6808 BPF_SIZE(insn->code), BPF_WRITE, -1, true, false); 6809 if (err) 6810 return err; 6811 return 0; 6812 } 6813 6814 static int check_atomic_load(struct bpf_verifier_env *env, 6815 struct bpf_insn *insn) 6816 { 6817 int err; 6818 6819 err = check_reg_arg(env, insn->src_reg, SRC_OP); 6820 if (err) 6821 return err; 6822 6823 if (!atomic_ptr_type_ok(env, insn->src_reg, insn)) { 6824 verbose(env, "BPF_ATOMIC loads from R%d %s is not allowed\n", 6825 insn->src_reg, 6826 reg_type_str(env, reg_state(env, insn->src_reg)->type)); 6827 return -EACCES; 6828 } 6829 6830 return check_load_mem(env, insn, true, false, false, "atomic_load"); 6831 } 6832 6833 static int check_atomic_store(struct bpf_verifier_env *env, 6834 struct bpf_insn *insn) 6835 { 6836 int err; 6837 6838 err = check_store_reg(env, insn, true); 6839 if (err) 6840 return err; 6841 6842 if (!atomic_ptr_type_ok(env, insn->dst_reg, insn)) { 6843 verbose(env, "BPF_ATOMIC stores into R%d %s is not allowed\n", 6844 insn->dst_reg, 6845 reg_type_str(env, reg_state(env, insn->dst_reg)->type)); 6846 return -EACCES; 6847 } 6848 6849 return 0; 6850 } 6851 6852 static int check_atomic(struct bpf_verifier_env *env, struct bpf_insn *insn) 6853 { 6854 switch (insn->imm) { 6855 case BPF_ADD: 6856 case BPF_ADD | BPF_FETCH: 6857 case BPF_AND: 6858 case BPF_AND | BPF_FETCH: 6859 case BPF_OR: 6860 case BPF_OR | BPF_FETCH: 6861 case BPF_XOR: 6862 case BPF_XOR | BPF_FETCH: 6863 case BPF_XCHG: 6864 case BPF_CMPXCHG: 6865 return check_atomic_rmw(env, insn); 6866 case BPF_LOAD_ACQ: 6867 if (BPF_SIZE(insn->code) == BPF_DW && BITS_PER_LONG != 64) { 6868 verbose(env, 6869 "64-bit load-acquires are only supported on 64-bit arches\n"); 6870 return -EOPNOTSUPP; 6871 } 6872 return check_atomic_load(env, insn); 6873 case BPF_STORE_REL: 6874 if (BPF_SIZE(insn->code) == BPF_DW && BITS_PER_LONG != 64) { 6875 verbose(env, 6876 "64-bit store-releases are only supported on 64-bit arches\n"); 6877 return -EOPNOTSUPP; 6878 } 6879 return check_atomic_store(env, insn); 6880 default: 6881 verbose(env, "BPF_ATOMIC uses invalid atomic opcode %02x\n", 6882 insn->imm); 6883 return -EINVAL; 6884 } 6885 } 6886 6887 /* When register 'regno' is used to read the stack (either directly or through 6888 * a helper function) make sure that it's within stack boundary and, depending 6889 * on the access type and privileges, that all elements of the stack are 6890 * initialized. 6891 * 6892 * All registers that have been spilled on the stack in the slots within the 6893 * read offsets are marked as read. 6894 */ 6895 static int check_stack_range_initialized( 6896 struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, int off, 6897 int access_size, bool zero_size_allowed, 6898 enum bpf_access_type type, struct bpf_call_arg_meta *meta) 6899 { 6900 struct bpf_func_state *state = bpf_func(env, reg); 6901 int err, min_off, max_off, i, j, slot, spi; 6902 /* Some accesses can write anything into the stack, others are 6903 * read-only. 6904 */ 6905 bool clobber = type == BPF_WRITE; 6906 /* 6907 * Negative access_size signals global subprog arg check where 6908 * STACK_POISON slots are acceptable. static stack liveness 6909 * might have determined that subprog doesn't read them, 6910 * but BTF based global subprog validation isn't accurate enough. 6911 */ 6912 bool allow_poison = access_size < 0 || clobber; 6913 /* The call will initialize the memory; uninitialized stack allowed */ 6914 bool raw_mode = meta && meta->arg_raw_mem.regno == reg_from_argno(argno); 6915 6916 access_size = abs(access_size); 6917 6918 if (access_size == 0 && !zero_size_allowed) { 6919 verbose(env, "invalid zero-sized read\n"); 6920 return -EACCES; 6921 } 6922 6923 err = check_stack_access_within_bounds(env, reg, argno, off, access_size, type); 6924 if (err) 6925 return err; 6926 6927 if (tnum_is_const(reg->var_off)) { 6928 min_off = max_off = reg->var_off.value + off; 6929 } else { 6930 /* Variable offset is prohibited for unprivileged mode for 6931 * simplicity since it requires corresponding support in 6932 * Spectre masking for stack ALU. 6933 * See also retrieve_ptr_limit(). 6934 */ 6935 if (!env->bypass_spec_v1) { 6936 char tn_buf[48]; 6937 6938 tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off); 6939 verbose(env, "%s variable offset stack access prohibited for !root, var_off=%s\n", 6940 reg_arg_name(env, argno), tn_buf); 6941 return -EACCES; 6942 } 6943 /* Only initialized buffer on stack is allowed to be accessed 6944 * with variable offset. With uninitialized buffer it's hard to 6945 * guarantee that whole memory is marked as initialized on 6946 * helper return since specific bounds are unknown what may 6947 * cause uninitialized stack leaking. 6948 */ 6949 raw_mode = false; 6950 6951 min_off = reg_smin(reg) + off; 6952 max_off = reg_smax(reg) + off; 6953 } 6954 6955 if (raw_mode) { 6956 meta->arg_raw_mem.size = access_size; 6957 return 0; 6958 } 6959 6960 for (i = min_off; i < max_off + access_size; i++) { 6961 u8 *stype; 6962 6963 slot = -i - 1; 6964 spi = slot / BPF_REG_SIZE; 6965 if (state->allocated_stack <= slot) { 6966 verbose(env, "allocated_stack too small\n"); 6967 return -EFAULT; 6968 } 6969 6970 stype = &state->stack[spi].slot_type[slot % BPF_REG_SIZE]; 6971 if (*stype == STACK_MISC) 6972 goto mark; 6973 if ((*stype == STACK_ZERO) || 6974 (*stype == STACK_INVALID && env->allow_uninit_stack)) { 6975 if (clobber) { 6976 /* helper can write anything into the stack */ 6977 *stype = STACK_MISC; 6978 } 6979 goto mark; 6980 } 6981 6982 if (bpf_is_spilled_reg(&state->stack[spi]) && 6983 (state->stack[spi].spilled_ptr.type == SCALAR_VALUE || 6984 env->allow_ptr_leaks)) { 6985 if (clobber) { 6986 __mark_reg_unknown(env, &state->stack[spi].spilled_ptr); 6987 for (j = 0; j < BPF_REG_SIZE; j++) 6988 scrub_spilled_slot(&state->stack[spi].slot_type[j]); 6989 } 6990 goto mark; 6991 } 6992 6993 if (*stype == STACK_POISON) { 6994 if (allow_poison) 6995 goto mark; 6996 verbose(env, "reading from stack %s off %d+%d size %d, slot poisoned by dead code elimination\n", 6997 reg_arg_name(env, argno), min_off, i - min_off, access_size); 6998 } else if (tnum_is_const(reg->var_off)) { 6999 verbose(env, "invalid read from stack %s off %d+%d size %d\n", 7000 reg_arg_name(env, argno), min_off, i - min_off, access_size); 7001 } else { 7002 char tn_buf[48]; 7003 7004 tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off); 7005 verbose(env, "invalid read from stack %s var_off %s+%d size %d\n", 7006 reg_arg_name(env, argno), tn_buf, i - min_off, access_size); 7007 } 7008 return -EACCES; 7009 mark: 7010 ; 7011 } 7012 return 0; 7013 } 7014 7015 static int check_helper_mem_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 7016 argno_t argno, int access_size, 7017 enum bpf_access_type access_type, bool zero_size_allowed, 7018 struct bpf_call_arg_meta *meta, bool *known_memory) 7019 { 7020 struct bpf_reg_state *regs = cur_regs(env); 7021 u32 *max_access; 7022 7023 if (known_memory) 7024 *known_memory = true; 7025 7026 switch (base_type(reg->type)) { 7027 case PTR_TO_PACKET: 7028 case PTR_TO_PACKET_META: 7029 return check_packet_access(env, reg, argno, 0, access_size, 7030 zero_size_allowed); 7031 case PTR_TO_MAP_KEY: 7032 if (access_type == BPF_WRITE) { 7033 verbose(env, "%s cannot write into %s\n", 7034 reg_arg_name(env, argno), reg_type_str(env, reg->type)); 7035 return -EACCES; 7036 } 7037 return check_mem_region_access(env, reg, argno, 0, access_size, 7038 reg->map_ptr->key_size, false); 7039 case PTR_TO_MAP_VALUE: 7040 if (check_map_access_type(env, reg, 0, access_size, access_type)) 7041 return -EACCES; 7042 return check_map_access(env, reg, argno, 0, access_size, 7043 zero_size_allowed, ACCESS_HELPER); 7044 case PTR_TO_MEM: 7045 if (type_is_rdonly_mem(reg->type)) { 7046 if (access_type == BPF_WRITE) { 7047 verbose(env, "%s cannot write into %s\n", 7048 reg_arg_name(env, argno), reg_type_str(env, reg->type)); 7049 return -EACCES; 7050 } 7051 } 7052 return check_mem_region_access(env, reg, argno, 0, 7053 access_size, reg->mem_size, 7054 zero_size_allowed); 7055 case PTR_TO_BUF: 7056 if (type_is_rdonly_mem(reg->type)) { 7057 if (access_type == BPF_WRITE) { 7058 verbose(env, "%s cannot write into %s\n", 7059 reg_arg_name(env, argno), reg_type_str(env, reg->type)); 7060 return -EACCES; 7061 } 7062 7063 max_access = &env->prog->aux->max_rdonly_access; 7064 } else { 7065 max_access = &env->prog->aux->max_rdwr_access; 7066 } 7067 return check_buffer_access(env, reg, argno, 0, 7068 access_size, zero_size_allowed, 7069 max_access); 7070 case PTR_TO_STACK: 7071 return check_stack_range_initialized( 7072 env, reg, 7073 argno, 0, access_size, 7074 zero_size_allowed, access_type, meta); 7075 case PTR_TO_BTF_ID: 7076 return check_ptr_to_btf_access(env, regs, reg, argno, 0, 7077 access_size, access_type, -1); 7078 case PTR_TO_CTX: 7079 /* Only permit reading or writing syscall context using helper calls. */ 7080 if (is_var_ctx_off_allowed(env->prog)) { 7081 int err = check_mem_region_access(env, reg, argno, 0, access_size, U16_MAX, 7082 zero_size_allowed); 7083 if (err) 7084 return err; 7085 if (env->prog->aux->max_ctx_offset < reg_umax(reg) + access_size) 7086 env->prog->aux->max_ctx_offset = reg_umax(reg) + access_size; 7087 return 0; 7088 } 7089 fallthrough; 7090 default: /* scalar_value or invalid ptr */ 7091 /* Allow zero-byte read from NULL, regardless of pointer type */ 7092 if (zero_size_allowed && access_size == 0 && 7093 bpf_register_is_null(reg)) 7094 return 0; 7095 if (known_memory && base_type(reg->type) != PTR_TO_CTX) 7096 *known_memory = false; 7097 7098 verbose(env, "%s type=%s ", reg_arg_name(env, argno), 7099 reg_type_str(env, reg->type)); 7100 verbose(env, "expected=%s\n", reg_type_str(env, PTR_TO_STACK)); 7101 return -EACCES; 7102 } 7103 } 7104 7105 enum bpf_mem_size_failure { 7106 BPF_MEM_SIZE_FAIL_NONE, 7107 BPF_MEM_SIZE_FAIL_MEMORY, 7108 BPF_MEM_SIZE_FAIL_SIZE, 7109 }; 7110 7111 /* verify arguments to helpers or kfuncs consisting of a pointer and an access 7112 * size. 7113 * 7114 * @mem_reg contains the pointer, @size_reg contains the access size. 7115 */ 7116 static int check_mem_size_reg(struct bpf_verifier_env *env, 7117 struct bpf_reg_state *mem_reg, 7118 struct bpf_reg_state *size_reg, argno_t mem_argno, 7119 argno_t size_argno, u32 access_type, 7120 bool zero_size_allowed, 7121 struct bpf_call_arg_meta *meta, 7122 enum bpf_mem_size_failure *failure) 7123 { 7124 int err = 0; 7125 7126 if (failure) 7127 *failure = BPF_MEM_SIZE_FAIL_NONE; 7128 7129 /* This is used to refine r0 return value bounds for helpers 7130 * that enforce this value as an upper bound on return values. 7131 * See do_refine_retval_range() for helpers that can refine 7132 * the return value. C type of helper is u32 so we pull register 7133 * bound from umax_value however, if negative verifier errors 7134 * out. Only upper bounds can be learned because retval is an 7135 * int type and negative retvals are allowed. 7136 */ 7137 meta->msize_max_value = reg_umax(size_reg); 7138 7139 /* The register is SCALAR_VALUE; the access check happens using 7140 * its boundaries. For unprivileged variable accesses, disable 7141 * raw mode so that the program is required to initialize all 7142 * the memory that the helper could just partially fill up. 7143 */ 7144 if (!tnum_is_const(size_reg->var_off)) 7145 meta = NULL; 7146 7147 if (reg_smin(size_reg) < 0) { 7148 verbose(env, "%s min value is negative, either use unsigned or 'var &= const'\n", 7149 reg_arg_name(env, size_argno)); 7150 err = -EACCES; 7151 goto size_error; 7152 } 7153 7154 if (reg_umin(size_reg) == 0 && !zero_size_allowed) { 7155 verbose(env, "%s invalid zero-sized read: u64=[%lld,%lld]\n", 7156 reg_arg_name(env, size_argno), reg_umin(size_reg), reg_umax(size_reg)); 7157 err = -EACCES; 7158 goto size_error; 7159 } 7160 7161 if (reg_umax(size_reg) >= BPF_MAX_VAR_SIZ) { 7162 verbose(env, "%s unbounded memory access, use 'var &= const' or 'if (var < const)'\n", 7163 reg_arg_name(env, size_argno)); 7164 err = -EACCES; 7165 goto size_error; 7166 } 7167 7168 if (access_type & BPF_READ) 7169 err = check_helper_mem_access(env, mem_reg, mem_argno, reg_umax(size_reg), 7170 BPF_READ, zero_size_allowed, meta, NULL); 7171 if (!err && access_type & BPF_WRITE) 7172 err = check_helper_mem_access(env, mem_reg, mem_argno, reg_umax(size_reg), 7173 BPF_WRITE, zero_size_allowed, meta, NULL); 7174 if (err && failure) 7175 *failure = BPF_MEM_SIZE_FAIL_MEMORY; 7176 7177 if (!err) 7178 err = mark_arg_precision(env, size_argno); 7179 7180 return err; 7181 7182 size_error: 7183 if (failure) 7184 *failure = BPF_MEM_SIZE_FAIL_SIZE; 7185 return err; 7186 } 7187 7188 static int check_mem_reg(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 7189 argno_t argno, u32 mem_size, enum bpf_access_type access_type, 7190 struct bpf_call_arg_meta *meta, bool *known_memory) 7191 { 7192 int size, err = 0; 7193 7194 if (bpf_register_is_null(reg)) 7195 return mark_arg_precision(env, argno); 7196 if (known_memory) 7197 *known_memory = true; 7198 7199 if (mem_size > S32_MAX) { 7200 verbose(env, "%s memory size %u is too large\n", 7201 reg_arg_name(env, argno), mem_size); 7202 return -EACCES; 7203 } 7204 7205 /* 7206 * Only a global subprog (meta == NULL) may read poisoned stack slots: 7207 * its static stack liveness proved the callee body skips them. 7208 */ 7209 size = (!meta && base_type(reg->type) == PTR_TO_STACK) ? -(int)mem_size : mem_size; 7210 7211 if (access_type & BPF_READ) 7212 err = check_helper_mem_access(env, reg, argno, size, BPF_READ, true, meta, 7213 known_memory); 7214 if (!err && (access_type & BPF_WRITE)) 7215 err = check_helper_mem_access(env, reg, argno, size, BPF_WRITE, true, meta, 7216 known_memory); 7217 7218 return err; 7219 } 7220 7221 static int process_const_alloc_mem_size(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 7222 argno_t argno, struct ret_mem_desc *ret_mem) 7223 { 7224 int regno = reg_from_argno(argno); 7225 int err; 7226 7227 if (ret_mem->found) { 7228 verifier_bug(env, "only one allocation size argument permitted"); 7229 return -EFAULT; 7230 } 7231 7232 if (!tnum_is_const(reg->var_off)) { 7233 verbose(env, "%s is not a const\n", reg_arg_name(env, argno)); 7234 return -EINVAL; 7235 } 7236 7237 if (reg->var_off.value > U32_MAX) { 7238 verbose(env, "%s allocation size exceeds u32 max\n", reg_arg_name(env, argno)); 7239 return -EINVAL; 7240 } 7241 7242 if (regno >= 0) 7243 err = mark_chain_precision(env, regno); 7244 else 7245 err = mark_stack_arg_precision(env, arg_idx_from_argno(argno)); 7246 if (err) 7247 return err; 7248 7249 ret_mem->size = reg->var_off.value; 7250 ret_mem->found = true; 7251 7252 return 0; 7253 } 7254 7255 static int process_const_arg(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 7256 argno_t argno, struct bpf_call_arg_meta *meta) 7257 { 7258 int regno = reg_from_argno(argno); 7259 int err; 7260 7261 if (meta->arg_constant.found) { 7262 verifier_bug(env, "only one constant argument permitted"); 7263 return -EFAULT; 7264 } 7265 7266 if (!tnum_is_const(reg->var_off)) { 7267 verbose(env, "%s must be a known constant\n", reg_arg_name(env, argno)); 7268 return -EINVAL; 7269 } 7270 7271 if (regno >= 0) 7272 err = mark_chain_precision(env, regno); 7273 else 7274 err = mark_stack_arg_precision(env, arg_idx_from_argno(argno)); 7275 if (err < 0) 7276 return err; 7277 7278 meta->arg_constant.found = true; 7279 meta->arg_constant.value = reg->var_off.value; 7280 7281 return 0; 7282 } 7283 7284 enum { 7285 PROCESS_SPIN_LOCK = (1 << 0), 7286 PROCESS_RES_LOCK = (1 << 1), 7287 PROCESS_LOCK_IRQ = (1 << 2), 7288 }; 7289 7290 /* Implementation details: 7291 * bpf_map_lookup returns PTR_TO_MAP_VALUE_OR_NULL. 7292 * bpf_obj_new returns PTR_TO_BTF_ID | MEM_ALLOC | PTR_MAYBE_NULL. 7293 * Two bpf_map_lookups (even with the same key) will have different reg->id. 7294 * Two separate bpf_obj_new will also have different reg->id. 7295 * For traditional PTR_TO_MAP_VALUE or PTR_TO_BTF_ID | MEM_ALLOC, the verifier 7296 * clears reg->id after value_or_null->value transition, since the verifier only 7297 * cares about the range of access to valid map value pointer and doesn't care 7298 * about actual address of the map element. 7299 * For maps with 'struct bpf_spin_lock' inside map value the verifier keeps 7300 * reg->id > 0 after value_or_null->value transition. By doing so 7301 * two bpf_map_lookups will be considered two different pointers that 7302 * point to different bpf_spin_locks. Likewise for pointers to allocated objects 7303 * returned from bpf_obj_new. 7304 * The verifier allows taking only one bpf_spin_lock at a time to avoid 7305 * dead-locks. 7306 * Since only one bpf_spin_lock is allowed the checks are simpler than 7307 * reg_is_refcounted() logic. The verifier needs to remember only 7308 * one spin_lock instead of array of acquired_refs. 7309 * env->cur_state->active_locks remembers which map value element or allocated 7310 * object got locked and clears it after bpf_spin_unlock. 7311 */ 7312 static int process_spin_lock(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, int flags) 7313 { 7314 bool is_lock = flags & PROCESS_SPIN_LOCK, is_res_lock = flags & PROCESS_RES_LOCK; 7315 const char *lock_str = is_res_lock ? "bpf_res_spin" : "bpf_spin"; 7316 struct bpf_verifier_state *cur = env->cur_state; 7317 struct bpf_reference_state *lock; 7318 bool is_const = tnum_is_const(reg->var_off); 7319 bool is_irq = flags & PROCESS_LOCK_IRQ; 7320 u64 val = reg->var_off.value; 7321 struct bpf_map *map = NULL; 7322 struct btf *btf = NULL; 7323 struct btf_record *rec; 7324 u32 spin_lock_off; 7325 int err; 7326 7327 if (!is_const) { 7328 verbose(env, 7329 "%s doesn't have constant offset. %s_lock has to be at the constant offset\n", 7330 reg_arg_name(env, argno), lock_str); 7331 return -EINVAL; 7332 } 7333 if (reg->type == PTR_TO_MAP_VALUE) { 7334 map = reg->map_ptr; 7335 if (!map->btf) { 7336 verbose(env, 7337 "map '%s' has to have BTF in order to use %s_lock\n", 7338 map->name, lock_str); 7339 return -EINVAL; 7340 } 7341 } else { 7342 btf = reg->btf; 7343 } 7344 7345 rec = reg_btf_record(reg); 7346 if (!btf_record_has_field(rec, is_res_lock ? BPF_RES_SPIN_LOCK : BPF_SPIN_LOCK)) { 7347 verbose(env, "%s '%s' has no valid %s_lock\n", map ? "map" : "local", 7348 map ? map->name : "kptr", lock_str); 7349 return -EINVAL; 7350 } 7351 spin_lock_off = is_res_lock ? rec->res_spin_lock_off : rec->spin_lock_off; 7352 if (spin_lock_off != val) { 7353 verbose(env, "off %lld doesn't point to 'struct %s_lock' that is at %d\n", 7354 val, lock_str, spin_lock_off); 7355 return -EINVAL; 7356 } 7357 if (is_lock) { 7358 void *ptr; 7359 int type; 7360 7361 if (map) 7362 ptr = map; 7363 else 7364 ptr = btf; 7365 7366 if (!is_res_lock && cur->active_locks) { 7367 lock = find_lock_state(cur, REF_TYPE_LOCK, 0, NULL); 7368 if (lock) { 7369 verbose(env, 7370 "Locking two bpf_spin_locks are not allowed\n"); 7371 bpf_diag_lock( 7372 env, env->insn_idx, "nested spin lock", 7373 "This path already holds a bpf_spin_lock. The verifier allows only one regular BPF spin lock at a time.", 7374 "Unlock the current bpf_spin_lock before taking another one.", lock); 7375 return -EINVAL; 7376 } 7377 } else if (is_res_lock && cur->active_locks) { 7378 lock = find_lock_state(cur, REF_TYPE_RES_LOCK | REF_TYPE_RES_LOCK_IRQ, 7379 reg->id, ptr); 7380 if (lock) { 7381 verbose(env, "Acquiring the same lock again, AA deadlock detected\n"); 7382 bpf_diag_lock( 7383 env, env->insn_idx, "recursive resource spin lock", 7384 "This path already holds the same resource spin lock. Taking it again would deadlock.", 7385 "Avoid reacquiring the same resource spin lock before it is unlocked.", lock); 7386 return -EINVAL; 7387 } 7388 } 7389 7390 if (is_res_lock && is_irq) 7391 type = REF_TYPE_RES_LOCK_IRQ; 7392 else if (is_res_lock) 7393 type = REF_TYPE_RES_LOCK; 7394 else 7395 type = REF_TYPE_LOCK; 7396 err = acquire_lock_state(env, env->insn_idx, type, reg->id, ptr); 7397 if (err < 0) { 7398 verbose(env, "Failed to acquire lock state\n"); 7399 return err; 7400 } 7401 } else { 7402 void *ptr; 7403 int type; 7404 7405 if (map) 7406 ptr = map; 7407 else 7408 ptr = btf; 7409 7410 if (!cur->active_locks) { 7411 verbose(env, "%s_unlock without taking a lock\n", lock_str); 7412 bpf_diag_res( 7413 env, env->insn_idx, "unlock without lock", 7414 "This unlock operation has no matching active lock on the current path.", 7415 "Take the matching lock before this unlock, or remove the unmatched unlock path."); 7416 return -EINVAL; 7417 } 7418 7419 if (is_res_lock && is_irq) 7420 type = REF_TYPE_RES_LOCK_IRQ; 7421 else if (is_res_lock) 7422 type = REF_TYPE_RES_LOCK; 7423 else 7424 type = REF_TYPE_LOCK; 7425 7426 lock = find_lock_state(cur, type, reg->id, ptr); 7427 if (!lock) { 7428 verbose(env, "%s_unlock of different lock\n", lock_str); 7429 lock = find_lock_state(cur, REF_TYPE_LOCK_MASK, cur->active_lock_id, 7430 cur->active_lock_ptr); 7431 bpf_diag_lock( 7432 env, env->insn_idx, "unlock of a different lock", 7433 "This unlock does not match any active lock with the same tracked identity on the current path.", 7434 "Unlock the same lock object that was most recently acquired.", lock); 7435 return -EINVAL; 7436 } 7437 if (reg->id != cur->active_lock_id || ptr != cur->active_lock_ptr) { 7438 verbose(env, "%s_unlock cannot be out of order\n", lock_str); 7439 lock = find_lock_state(cur, REF_TYPE_LOCK_MASK, cur->active_lock_id, 7440 cur->active_lock_ptr); 7441 bpf_diag_lock( 7442 env, env->insn_idx, "unlock out of order", 7443 "Locks must be released in last-in, first-out order, but this unlock does not match the currently active lock.", 7444 "Release nested locks in the reverse order they were acquired.", lock); 7445 return -EINVAL; 7446 } 7447 if (release_lock_state(env, type, reg->id, ptr)) { 7448 verbose(env, "%s_unlock of different lock\n", lock_str); 7449 bpf_diag_lock( 7450 env, env->insn_idx, "unlock of a different lock", 7451 "The verifier could not release a lock state matching this unlock operation.", 7452 "Pass the same lock object and lock kind that were used for the matching lock operation.", 7453 lock); 7454 return -EINVAL; 7455 } 7456 /* 7457 * Invalidate non-owning refs before RCU demotion clears their 7458 * NON_OWN_REF flag. 7459 */ 7460 invalidate_non_owning_refs(env); 7461 7462 if (!in_rcu_cs(env)) 7463 invalidate_rcu_protected_refs(env); 7464 } 7465 return 0; 7466 } 7467 7468 /* Check if @regno is a pointer to a specific field in a map value */ 7469 static int check_map_field_pointer(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, 7470 enum btf_field_type field_type, 7471 struct bpf_map_desc *map_desc) 7472 { 7473 bool is_const = tnum_is_const(reg->var_off); 7474 struct bpf_map *map = reg->map_ptr; 7475 u64 val = reg->var_off.value; 7476 const char *struct_name = btf_field_type_name(field_type); 7477 int field_off = -1; 7478 7479 if (!is_const) { 7480 verbose(env, 7481 "%s doesn't have constant offset. %s has to be at the constant offset\n", 7482 reg_arg_name(env, argno), struct_name); 7483 return -EINVAL; 7484 } 7485 if (!map->btf) { 7486 verbose(env, "map '%s' has to have BTF in order to use %s\n", map->name, 7487 struct_name); 7488 return -EINVAL; 7489 } 7490 if (!btf_record_has_field(map->record, field_type)) { 7491 verbose(env, "map '%s' has no valid %s\n", map->name, struct_name); 7492 return -EINVAL; 7493 } 7494 switch (field_type) { 7495 case BPF_TIMER: 7496 field_off = map->record->timer_off; 7497 break; 7498 case BPF_TASK_WORK: 7499 field_off = map->record->task_work_off; 7500 break; 7501 case BPF_WORKQUEUE: 7502 field_off = map->record->wq_off; 7503 break; 7504 default: 7505 verifier_bug(env, "unsupported BTF field type: %s\n", struct_name); 7506 return -EINVAL; 7507 } 7508 if (field_off != val) { 7509 verbose(env, "off %lld doesn't point to 'struct %s' that is at %d\n", 7510 val, struct_name, field_off); 7511 return -EINVAL; 7512 } 7513 if (map_desc->ptr) { 7514 verifier_bug(env, "Two map pointers in a %s helper", struct_name); 7515 return -EFAULT; 7516 } 7517 map_desc->uid = reg->map_uid; 7518 map_desc->ptr = map; 7519 return 0; 7520 } 7521 7522 static int process_timer_func(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, 7523 struct bpf_map_desc *map) 7524 { 7525 if (IS_ENABLED(CONFIG_PREEMPT_RT)) { 7526 verbose(env, "bpf_timer cannot be used for PREEMPT_RT.\n"); 7527 return -EOPNOTSUPP; 7528 } 7529 return check_map_field_pointer(env, reg, argno, BPF_TIMER, map); 7530 } 7531 7532 static int process_kptr_func(struct bpf_verifier_env *env, int regno, 7533 struct bpf_call_arg_meta *meta) 7534 { 7535 struct bpf_reg_state *reg = reg_state(env, regno); 7536 struct btf_field *kptr_field; 7537 struct bpf_map *map_ptr; 7538 struct btf_record *rec; 7539 u32 kptr_off; 7540 7541 if (type_is_ptr_alloc_obj(reg->type)) { 7542 rec = reg_btf_record(reg); 7543 } else { /* PTR_TO_MAP_VALUE */ 7544 map_ptr = reg->map_ptr; 7545 if (!map_ptr->btf) { 7546 verbose(env, "map '%s' has to have BTF in order to use bpf_kptr_xchg\n", 7547 map_ptr->name); 7548 return -EINVAL; 7549 } 7550 rec = map_ptr->record; 7551 meta->map.ptr = map_ptr; 7552 } 7553 7554 if (!tnum_is_const(reg->var_off)) { 7555 verbose(env, 7556 "R%d doesn't have constant offset. kptr has to be at the constant offset\n", 7557 regno); 7558 return -EINVAL; 7559 } 7560 7561 if (!btf_record_has_field(rec, BPF_KPTR)) { 7562 verbose(env, "R%d has no valid kptr\n", regno); 7563 return -EINVAL; 7564 } 7565 7566 kptr_off = reg->var_off.value; 7567 kptr_field = btf_record_find(rec, kptr_off, BPF_KPTR); 7568 if (!kptr_field) { 7569 verbose(env, "off=%d doesn't point to kptr\n", kptr_off); 7570 return -EACCES; 7571 } 7572 if (kptr_field->type != BPF_KPTR_REF && kptr_field->type != BPF_KPTR_PERCPU) { 7573 verbose(env, "off=%d kptr isn't referenced kptr\n", kptr_off); 7574 return -EACCES; 7575 } 7576 meta->kptr_field = kptr_field; 7577 return 0; 7578 } 7579 7580 static void bpf_diag_call_arg(struct bpf_verifier_env *env, u32 insn_idx, argno_t argno, 7581 const char *call_name, const char *reason, const char *suggestion); 7582 __printf(6, 7) static void bpf_diag_call_arg_fmt(struct bpf_verifier_env *env, u32 insn_idx, 7583 argno_t argno, const char *call_name, 7584 const char *suggestion, const char *fmt, ...); 7585 7586 /* 7587 * Validate dynptr arguments for helper, kfunc and subprog. 7588 * 7589 * @dynptr is both input and output. It is populated when the argument is 7590 * tagged with MEM_UNINIT (i.e., the dynptr argument that will be constructed) 7591 * and consumed when the argument is expecting to be an initialized dynptr. 7592 * @parent_id is used to track the referenced parent object (e.g., file or skb in 7593 * qdisc program) when constructing a dynptr. 7594 * 7595 * There are two register types representing a bpf_dynptr, one is PTR_TO_STACK 7596 * which points to a stack slot, and the other is CONST_PTR_TO_DYNPTR. 7597 * 7598 * In both cases we deal with the first 8 bytes, but need to mark the next 8 7599 * bytes as STACK_DYNPTR in case of PTR_TO_STACK. In case of 7600 * CONST_PTR_TO_DYNPTR, we are guaranteed to get the beginning of the object. 7601 * 7602 * Mutability of bpf_dynptr is at two levels: the dynptr and the memory the 7603 * dynptr points to. At the first level, the verifier will make sure a 7604 * CONST_PTR_TO_DYNPTR cannot be reinitialized or destroyed. The mutability of 7605 * a dynptr's view (i.e., start and offset) is not tracked as there is not such 7606 * use case. The second level is tracked using the upper bit of bpf_dynptr->size 7607 * and checked dynamically during runtime. 7608 */ 7609 static int process_dynptr_func(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 7610 argno_t argno, int insn_idx, const char *call_name, 7611 enum bpf_arg_type arg_type, 7612 struct ref_obj_desc *ref_obj, struct bpf_dynptr_desc *dynptr) 7613 { 7614 int spi, err = 0; 7615 7616 if (reg->type != PTR_TO_STACK && reg->type != CONST_PTR_TO_DYNPTR) { 7617 verbose(env, 7618 "%s expected pointer to stack or const struct bpf_dynptr\n", 7619 reg_arg_name(env, argno)); 7620 bpf_diag_call_arg_fmt( 7621 env, insn_idx, argno, call_name, 7622 "Pass the address of a stack dynptr object, or use a const dynptr pointer returned by the verifier-supported path.", 7623 "a dynptr argument must be a pointer to a dynptr stack slot or a verifier-provided const struct bpf_dynptr, but %s is %s", 7624 reg_arg_name(env, argno), bpf_diag_reg_type_plain(env, reg->type)); 7625 return -EINVAL; 7626 } 7627 7628 /* MEM_UNINIT - Points to memory that is an appropriate candidate for 7629 * constructing a mutable bpf_dynptr object. 7630 * 7631 * Currently, this is only possible with PTR_TO_STACK 7632 * pointing to a region of at least 16 bytes which doesn't 7633 * contain an existing bpf_dynptr. 7634 * 7635 * OBJ_RELEASE - Points to a initialized bpf_dynptr that will be 7636 * destroyed. 7637 * 7638 * None - Points to a initialized dynptr that cannot be 7639 * reinitialized or destroyed. However, the view of the 7640 * dynptr and the memory it points to may be mutated. 7641 */ 7642 if (arg_type & MEM_UNINIT) { 7643 int i; 7644 7645 if (!is_dynptr_reg_valid_uninit(env, reg)) { 7646 verbose(env, "Dynptr has to be an uninitialized dynptr\n"); 7647 bpf_diag_res( 7648 env, insn_idx, "dynptr is already initialized", 7649 "This kfunc constructs a dynptr and requires an uninitialized dynptr stack slot, but the selected slot already holds dynptr state.", 7650 "Use a fresh stack dynptr slot, or release/destroy the existing dynptr before reusing the slot."); 7651 return -EINVAL; 7652 } 7653 7654 /* we write BPF_DW bits (8 bytes) at a time */ 7655 for (i = 0; i < BPF_DYNPTR_SIZE; i += 8) { 7656 err = check_mem_access(env, insn_idx, reg, argno, 7657 i, BPF_DW, BPF_WRITE, -1, false, false); 7658 if (err) 7659 return err; 7660 } 7661 7662 err = mark_stack_slots_dynptr(env, reg, arg_type, insn_idx, ref_obj, dynptr); 7663 } else /* OBJ_RELEASE and None case from above */ { 7664 /* For the reg->type == PTR_TO_STACK case, bpf_dynptr is never const */ 7665 if (reg->type == CONST_PTR_TO_DYNPTR && (arg_type & OBJ_RELEASE)) { 7666 verbose(env, "CONST_PTR_TO_DYNPTR cannot be released\n"); 7667 bpf_diag_res( 7668 env, insn_idx, "const dynptr release", 7669 "This release operation was given a const dynptr. Const dynptr values are verifier-provided views and cannot be released by the program.", 7670 "Release only mutable dynptrs that the program initialized or reserved."); 7671 return -EINVAL; 7672 } 7673 7674 if (!is_dynptr_reg_valid_init(env, reg)) { 7675 verbose(env, "Expected an initialized dynptr as %s\n", 7676 reg_arg_name(env, argno)); 7677 bpf_diag_res( 7678 env, insn_idx, "uninitialized dynptr use", 7679 "This operation requires an initialized dynptr, but the stack slot does not currently hold a valid dynptr on this path.", 7680 "Initialize the dynptr on every path before this call, and avoid overwriting or releasing it before this use."); 7681 return -EINVAL; 7682 } 7683 7684 /* Fold modifiers (in this case, OBJ_RELEASE) when checking expected type */ 7685 if (!is_dynptr_type_expected(env, reg, arg_type & ~OBJ_RELEASE)) { 7686 enum bpf_dynptr_type expected_type = arg_to_dynptr_type(arg_type); 7687 enum bpf_dynptr_type actual_type = dynptr_reg_type(env, reg); 7688 7689 verbose(env, "Expected a dynptr of type %s as %s\n", 7690 dynptr_type_str(expected_type), reg_arg_name(env, argno)); 7691 bpf_diag_call_arg_fmt( 7692 env, insn_idx, argno, call_name, 7693 "Use a dynptr constructor that matches this operation, or call an operation that accepts the dynptr's current type.", 7694 "the dynptr is initialized with backing object type %s, but this operation expects dynptr type %s", 7695 dynptr_type_str(actual_type), dynptr_type_str(expected_type)); 7696 return -EINVAL; 7697 } 7698 7699 if (reg->type != CONST_PTR_TO_DYNPTR) { 7700 struct bpf_func_state *state = bpf_func(env, reg); 7701 7702 spi = dynptr_get_spi(env, reg); 7703 if (spi < 0) 7704 return spi; 7705 7706 mark_stack_slots_scratched(env, spi, BPF_DYNPTR_NR_SLOTS); 7707 7708 reg = &state->stack[spi].spilled_ptr; 7709 } 7710 7711 if (dynptr) { 7712 dynptr->type = reg->dynptr.type; 7713 dynptr->id = reg->id; 7714 dynptr->parent_id = reg->parent_id; 7715 } 7716 } 7717 return err; 7718 } 7719 7720 static bool is_iter_kfunc(struct bpf_call_arg_meta *meta) 7721 { 7722 return meta->kfunc_flags & (KF_ITER_NEW | KF_ITER_NEXT | KF_ITER_DESTROY); 7723 } 7724 7725 static bool is_iter_new_kfunc(struct bpf_call_arg_meta *meta) 7726 { 7727 return meta->kfunc_flags & KF_ITER_NEW; 7728 } 7729 7730 static bool is_iter_destroy_kfunc(struct bpf_call_arg_meta *meta) 7731 { 7732 return meta->kfunc_flags & KF_ITER_DESTROY; 7733 } 7734 7735 static bool is_kfunc_arg_iter(struct bpf_call_arg_meta *meta, int arg_idx, 7736 const struct btf_param *arg) 7737 { 7738 /* btf_check_iter_kfuncs() guarantees that first argument of any iter 7739 * kfunc is iter state pointer 7740 */ 7741 if (is_iter_kfunc(meta)) 7742 return arg_idx == 0; 7743 7744 /* iter passed as an argument to a generic kfunc */ 7745 return btf_param_match_suffix(meta->btf, arg, "__iter"); 7746 } 7747 7748 static int process_iter_arg(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, int insn_idx, 7749 struct bpf_call_arg_meta *meta) 7750 { 7751 struct bpf_func_state *state = bpf_func(env, reg); 7752 const struct btf_type *t; 7753 u32 arg_idx = arg_idx_from_argno(argno); 7754 int spi, err, i, nr_slots, btf_id; 7755 7756 if (reg->type != PTR_TO_STACK) { 7757 verbose(env, "%s expected pointer to an iterator on stack\n", 7758 reg_arg_name(env, argno)); 7759 bpf_diag_call_arg_fmt( 7760 env, insn_idx, argno, meta->func_name, 7761 "Pass the address of a stack iterator object for iterator new, next, and destroy calls.", 7762 "iterator state must live in verifier-tracked stack memory, but %s is %s", 7763 reg_arg_name(env, argno), bpf_diag_reg_type_plain(env, reg->type)); 7764 return -EINVAL; 7765 } 7766 7767 /* For iter_{new,next,destroy} functions, btf_check_iter_kfuncs() 7768 * ensures struct convention, so we wouldn't need to do any BTF 7769 * validation here. But given iter state can be passed as a parameter 7770 * to any kfunc, if arg has "__iter" suffix, we need to be a bit more 7771 * conservative here. 7772 */ 7773 btf_id = btf_check_iter_arg(meta->btf, meta->func_proto, arg_idx); 7774 if (btf_id < 0) { 7775 verbose(env, "expected valid iter pointer as %s\n", 7776 reg_arg_name(env, argno)); 7777 bpf_diag_call_arg( 7778 env, insn_idx, argno, meta->func_name, 7779 "the kfunc expects a recognized iterator state pointer, but this argument does not match a valid iterator type", 7780 "Pass the exact iterator state type expected by this kfunc."); 7781 return -EINVAL; 7782 } 7783 t = btf_type_by_id(meta->btf, btf_id); 7784 nr_slots = t->size / BPF_REG_SIZE; 7785 7786 if (is_iter_new_kfunc(meta)) { 7787 /* bpf_iter_<type>_new() expects pointer to uninit iter state */ 7788 if (!is_iter_reg_valid_uninit(env, reg, nr_slots)) { 7789 verbose(env, "expected uninitialized iter_%s as %s\n", 7790 iter_type_str(meta->btf, btf_id), reg_arg_name(env, argno)); 7791 bpf_diag_res( 7792 env, insn_idx, "iterator is already initialized", 7793 "Iterator creation requires an uninitialized iterator stack object, but this stack range already contains iterator state.", 7794 "Use a fresh iterator stack slot, or destroy the existing iterator before reusing the slot."); 7795 return -EINVAL; 7796 } 7797 7798 for (i = 0; i < nr_slots * 8; i += BPF_REG_SIZE) { 7799 err = check_mem_access(env, insn_idx, reg, argno, 7800 i, BPF_DW, BPF_WRITE, -1, false, false); 7801 if (err) 7802 return err; 7803 } 7804 7805 err = mark_stack_slots_iter(env, meta, reg, insn_idx, meta->btf, btf_id, nr_slots); 7806 if (err) 7807 return err; 7808 } else { 7809 /* iter_next() or iter_destroy(), as well as any kfunc 7810 * accepting iter argument, expect initialized iter state 7811 */ 7812 err = is_iter_reg_valid_init(env, reg, meta->btf, btf_id, nr_slots); 7813 switch (err) { 7814 case 0: 7815 break; 7816 case -EINVAL: 7817 verbose(env, "expected an initialized iter_%s as %s\n", 7818 iter_type_str(meta->btf, btf_id), reg_arg_name(env, argno)); 7819 bpf_diag_res( 7820 env, insn_idx, "uninitialized iterator use", 7821 "This iterator operation requires an initialized iterator state object, but the stack range does not contain a live iterator on this path.", 7822 "Call the matching iterator new kfunc on every path before calling next or destroy, and do not destroy the iterator before this use."); 7823 return err; 7824 case -EPROTO: 7825 verbose(env, "expected an RCU CS when using %s\n", meta->func_name); 7826 bpf_diag_ctx_required( 7827 env, insn_idx, meta->func_name, BPF_DIAG_CONTEXT_RCU, 7828 "Wrap iterator use in bpf_rcu_read_lock() and bpf_rcu_read_unlock(), keeping all exit paths balanced."); 7829 return err; 7830 default: 7831 return err; 7832 } 7833 7834 spi = iter_get_spi(env, reg, nr_slots); 7835 if (spi < 0) 7836 return spi; 7837 7838 mark_stack_slots_scratched(env, spi, nr_slots); 7839 7840 /* remember meta->iter info for process_iter_next_call() */ 7841 meta->iter.spi = spi; 7842 meta->iter.frameno = reg->frameno; 7843 update_ref_obj(&meta->ref_obj, &state->stack[spi].spilled_ptr); 7844 7845 if (is_iter_destroy_kfunc(meta)) { 7846 err = unmark_stack_slots_iter(env, reg, nr_slots); 7847 if (err) 7848 return err; 7849 } 7850 } 7851 7852 return 0; 7853 } 7854 7855 /* Look for a previous loop entry at insn_idx: nearest parent state 7856 * stopped at insn_idx with callsites matching those in cur->frame. 7857 */ 7858 static struct bpf_verifier_state *find_prev_entry(struct bpf_verifier_env *env, 7859 struct bpf_verifier_state *cur, 7860 int insn_idx) 7861 { 7862 struct bpf_verifier_state_list *sl; 7863 struct bpf_verifier_state *st; 7864 struct list_head *pos, *head; 7865 7866 /* Explored states are pushed in stack order, most recent states come first */ 7867 head = bpf_explored_state(env, insn_idx); 7868 list_for_each(pos, head) { 7869 sl = container_of(pos, struct bpf_verifier_state_list, node); 7870 /* If st->branches != 0 state is a part of current DFS verification path, 7871 * hence cur & st for a loop. 7872 */ 7873 st = &sl->state; 7874 if (st->insn_idx == insn_idx && st->branches && same_callsites(st, cur) && 7875 st->dfs_depth < cur->dfs_depth) 7876 return st; 7877 } 7878 7879 return NULL; 7880 } 7881 7882 /* 7883 * Check if scalar registers are exact for the purpose of not widening. 7884 * More lenient than regs_exact() 7885 */ 7886 static bool scalars_exact_for_widen(const struct bpf_reg_state *rold, 7887 const struct bpf_reg_state *rcur) 7888 { 7889 return !memcmp(rold, rcur, offsetof(struct bpf_reg_state, id)); 7890 } 7891 7892 static void maybe_widen_reg(struct bpf_verifier_env *env, 7893 struct bpf_reg_state *rold, struct bpf_reg_state *rcur) 7894 { 7895 if (rold->type != SCALAR_VALUE) 7896 return; 7897 if (rold->type != rcur->type) 7898 return; 7899 if (rold->precise || rcur->precise || scalars_exact_for_widen(rold, rcur)) 7900 return; 7901 __mark_reg_unknown(env, rcur); 7902 } 7903 7904 static int widen_imprecise_scalars(struct bpf_verifier_env *env, 7905 struct bpf_verifier_state *old, 7906 struct bpf_verifier_state *cur) 7907 { 7908 struct bpf_func_state *fold, *fcur; 7909 int i, fr, num_slots; 7910 7911 for (fr = old->curframe; fr >= 0; fr--) { 7912 fold = old->frame[fr]; 7913 fcur = cur->frame[fr]; 7914 7915 for (i = 0; i < MAX_BPF_REG; i++) 7916 maybe_widen_reg(env, 7917 &fold->regs[i], 7918 &fcur->regs[i]); 7919 7920 num_slots = min(fold->allocated_stack / BPF_REG_SIZE, 7921 fcur->allocated_stack / BPF_REG_SIZE); 7922 for (i = 0; i < num_slots; i++) { 7923 if (!bpf_is_spilled_reg(&fold->stack[i]) || 7924 !bpf_is_spilled_reg(&fcur->stack[i])) 7925 continue; 7926 7927 maybe_widen_reg(env, 7928 &fold->stack[i].spilled_ptr, 7929 &fcur->stack[i].spilled_ptr); 7930 } 7931 } 7932 return 0; 7933 } 7934 7935 static struct bpf_reg_state *get_iter_from_state(struct bpf_verifier_state *cur_st, 7936 struct bpf_call_arg_meta *meta) 7937 { 7938 int iter_frameno = meta->iter.frameno; 7939 int iter_spi = meta->iter.spi; 7940 7941 return &cur_st->frame[iter_frameno]->stack[iter_spi].spilled_ptr; 7942 } 7943 7944 /* process_iter_next_call() is called when verifier gets to iterator's next 7945 * "method" (e.g., bpf_iter_num_next() for numbers iterator) call. We'll refer 7946 * to it as just "iter_next()" in comments below. 7947 * 7948 * BPF verifier relies on a crucial contract for any iter_next() 7949 * implementation: it should *eventually* return NULL, and once that happens 7950 * it should keep returning NULL. That is, once iterator exhausts elements to 7951 * iterate, it should never reset or spuriously return new elements. 7952 * 7953 * With the assumption of such contract, process_iter_next_call() simulates 7954 * a fork in the verifier state to validate loop logic correctness and safety 7955 * without having to simulate infinite amount of iterations. 7956 * 7957 * In current state, we first assume that iter_next() returned NULL and 7958 * iterator state is set to DRAINED (BPF_ITER_STATE_DRAINED). In such 7959 * conditions we should not form an infinite loop and should eventually reach 7960 * exit. 7961 * 7962 * Besides that, we also fork current state and enqueue it for later 7963 * verification. In a forked state we keep iterator state as ACTIVE 7964 * (BPF_ITER_STATE_ACTIVE) and assume non-NULL return from iter_next(). We 7965 * also bump iteration depth to prevent erroneous infinite loop detection 7966 * later on (see iter_active_depths_differ() comment for details). In this 7967 * state we assume that we'll eventually loop back to another iter_next() 7968 * calls (it could be in exactly same location or in some other instruction, 7969 * it doesn't matter, we don't make any unnecessary assumptions about this, 7970 * everything revolves around iterator state in a stack slot, not which 7971 * instruction is calling iter_next()). When that happens, we either will come 7972 * to iter_next() with equivalent state and can conclude that next iteration 7973 * will proceed in exactly the same way as we just verified, so it's safe to 7974 * assume that loop converges. If not, we'll go on another iteration 7975 * simulation with a different input state, until all possible starting states 7976 * are validated or we reach maximum number of instructions limit. 7977 * 7978 * This way, we will either exhaustively discover all possible input states 7979 * that iterator loop can start with and eventually will converge, or we'll 7980 * effectively regress into bounded loop simulation logic and either reach 7981 * maximum number of instructions if loop is not provably convergent, or there 7982 * is some statically known limit on number of iterations (e.g., if there is 7983 * an explicit `if n > 100 then break;` statement somewhere in the loop). 7984 * 7985 * Iteration convergence logic in is_state_visited() relies on exact 7986 * states comparison, which ignores read and precision marks. 7987 * This is necessary because read and precision marks are not finalized 7988 * while in the loop. Exact comparison might preclude convergence for 7989 * simple programs like below: 7990 * 7991 * i = 0; 7992 * while(iter_next(&it)) 7993 * i++; 7994 * 7995 * At each iteration step i++ would produce a new distinct state and 7996 * eventually instruction processing limit would be reached. 7997 * 7998 * To avoid such behavior speculatively forget (widen) range for 7999 * imprecise scalar registers, if those registers were not precise at the 8000 * end of the previous iteration and do not match exactly. 8001 * 8002 * This is a conservative heuristic that allows to verify wide range of programs, 8003 * however it precludes verification of programs that conjure an 8004 * imprecise value on the first loop iteration and use it as precise on a second. 8005 * For example, the following safe program would fail to verify: 8006 * 8007 * struct bpf_num_iter it; 8008 * int arr[10]; 8009 * int i = 0, a = 0; 8010 * bpf_iter_num_new(&it, 0, 10); 8011 * while (bpf_iter_num_next(&it)) { 8012 * if (a == 0) { 8013 * a = 1; 8014 * i = 7; // Because i changed verifier would forget 8015 * // it's range on second loop entry. 8016 * } else { 8017 * arr[i] = 42; // This would fail to verify. 8018 * } 8019 * } 8020 * bpf_iter_num_destroy(&it); 8021 */ 8022 static int process_iter_next_call(struct bpf_verifier_env *env, int insn_idx, 8023 struct bpf_call_arg_meta *meta) 8024 { 8025 struct bpf_verifier_state *cur_st = env->cur_state, *queued_st, *prev_st; 8026 struct bpf_func_state *cur_fr = cur_st->frame[cur_st->curframe], *queued_fr; 8027 struct bpf_reg_state *cur_iter, *queued_iter; 8028 8029 BTF_TYPE_EMIT(struct bpf_iter); 8030 8031 cur_iter = get_iter_from_state(cur_st, meta); 8032 8033 if (cur_iter->iter.state != BPF_ITER_STATE_ACTIVE && 8034 cur_iter->iter.state != BPF_ITER_STATE_DRAINED) { 8035 verifier_bug(env, "unexpected iterator state %d (%s)", 8036 cur_iter->iter.state, iter_state_str(cur_iter->iter.state)); 8037 return -EFAULT; 8038 } 8039 8040 if (cur_iter->iter.state == BPF_ITER_STATE_ACTIVE) { 8041 /* Because iter_next() call is a checkpoint is_state_visitied() 8042 * should guarantee parent state with same call sites and insn_idx. 8043 */ 8044 if (!cur_st->parent || cur_st->parent->insn_idx != insn_idx || 8045 !same_callsites(cur_st->parent, cur_st)) { 8046 verifier_bug(env, "bad parent state for iter next call"); 8047 return -EFAULT; 8048 } 8049 /* Note cur_st->parent in the call below, it is necessary to skip 8050 * checkpoint created for cur_st by is_state_visited() 8051 * right at this instruction. 8052 */ 8053 prev_st = find_prev_entry(env, cur_st->parent, insn_idx); 8054 /* branch out active iter state */ 8055 queued_st = push_stack(env, insn_idx + 1, insn_idx, false); 8056 if (IS_ERR(queued_st)) 8057 return PTR_ERR(queued_st); 8058 8059 queued_iter = get_iter_from_state(queued_st, meta); 8060 queued_iter->iter.state = BPF_ITER_STATE_ACTIVE; 8061 queued_iter->iter.depth++; 8062 if (prev_st) 8063 widen_imprecise_scalars(env, prev_st, queued_st); 8064 8065 queued_fr = queued_st->frame[queued_st->curframe]; 8066 mark_ptr_not_null_reg(&queued_fr->regs[BPF_REG_0]); 8067 } 8068 8069 /* switch to DRAINED state, but keep the depth unchanged */ 8070 /* mark current iter state as drained and assume returned NULL */ 8071 cur_iter->iter.state = BPF_ITER_STATE_DRAINED; 8072 __mark_reg_const_zero(env, &cur_fr->regs[BPF_REG_0]); 8073 8074 return 0; 8075 } 8076 8077 static bool arg_type_is_mem_size(enum bpf_arg_type type) 8078 { 8079 return type == ARG_MEM_SIZE || type == ARG_MEM_SIZE_OR_ZERO; 8080 } 8081 8082 static bool arg_type_is_raw_mem(enum bpf_arg_type type) 8083 { 8084 /* 8085 * A map value output buffer (e.g. bpf_map_pop_elem) is also a raw 8086 * (uninitialized) memory argument, and like ARG_PTR_TO_MEM it may be 8087 * passed as a PTR_TO_STACK that reaches check_stack_range_initialized(). 8088 */ 8089 return (base_type(type) == ARG_PTR_TO_MEM || 8090 base_type(type) == ARG_PTR_TO_MAP_VALUE) && 8091 type & MEM_UNINIT; 8092 } 8093 8094 static bool arg_type_is_release(enum bpf_arg_type type) 8095 { 8096 return type & OBJ_RELEASE; 8097 } 8098 8099 static bool arg_type_is_dynptr(enum bpf_arg_type type) 8100 { 8101 return base_type(type) == ARG_PTR_TO_DYNPTR; 8102 } 8103 8104 static int resolve_map_arg_type(struct bpf_verifier_env *env, 8105 const struct bpf_call_arg_meta *meta, 8106 enum bpf_arg_type *arg_type) 8107 { 8108 if (!meta->map.ptr) { 8109 /* kernel subsystem misconfigured verifier */ 8110 verifier_bug(env, "invalid map_ptr to access map->type"); 8111 return -EFAULT; 8112 } 8113 8114 switch (meta->map.ptr->map_type) { 8115 case BPF_MAP_TYPE_SOCKMAP: 8116 case BPF_MAP_TYPE_SOCKHASH: 8117 if (*arg_type == ARG_PTR_TO_MAP_VALUE) { 8118 *arg_type = ARG_PTR_TO_BTF_ID_SOCK_COMMON; 8119 } else { 8120 verbose(env, "invalid arg_type for sockmap/sockhash\n"); 8121 return -EINVAL; 8122 } 8123 break; 8124 case BPF_MAP_TYPE_BLOOM_FILTER: 8125 if (meta->func_id == BPF_FUNC_map_peek_elem) 8126 *arg_type = ARG_PTR_TO_MAP_VALUE; 8127 break; 8128 default: 8129 break; 8130 } 8131 return 0; 8132 } 8133 8134 struct bpf_reg_types { 8135 const enum bpf_reg_type types[10]; 8136 u32 *btf_id; 8137 }; 8138 8139 static const struct bpf_reg_types sock_types = { 8140 .types = { 8141 PTR_TO_SOCK_COMMON, 8142 PTR_TO_SOCKET, 8143 PTR_TO_TCP_SOCK, 8144 PTR_TO_XDP_SOCK, 8145 }, 8146 }; 8147 8148 #ifdef CONFIG_NET 8149 static const struct bpf_reg_types btf_id_sock_common_types = { 8150 .types = { 8151 PTR_TO_SOCK_COMMON, 8152 PTR_TO_SOCKET, 8153 PTR_TO_TCP_SOCK, 8154 PTR_TO_XDP_SOCK, 8155 PTR_TO_BTF_ID, 8156 PTR_TO_BTF_ID | PTR_TRUSTED, 8157 }, 8158 .btf_id = &btf_sock_ids[BTF_SOCK_TYPE_SOCK_COMMON], 8159 }; 8160 #endif 8161 8162 static const struct bpf_reg_types mem_types = { 8163 .types = { 8164 PTR_TO_STACK, 8165 PTR_TO_PACKET, 8166 PTR_TO_PACKET_META, 8167 PTR_TO_MAP_KEY, 8168 PTR_TO_MAP_VALUE, 8169 PTR_TO_MEM, 8170 PTR_TO_MEM | MEM_RINGBUF, 8171 PTR_TO_BUF, 8172 PTR_TO_BTF_ID | PTR_TRUSTED, 8173 PTR_TO_CTX, 8174 }, 8175 }; 8176 8177 static const struct bpf_reg_types spin_lock_types = { 8178 .types = { 8179 PTR_TO_MAP_VALUE, 8180 PTR_TO_BTF_ID | MEM_ALLOC, 8181 } 8182 }; 8183 8184 static const struct bpf_reg_types fullsock_types = { .types = { PTR_TO_SOCKET } }; 8185 static const struct bpf_reg_types scalar_types = { .types = { SCALAR_VALUE } }; 8186 static const struct bpf_reg_types context_types = { .types = { PTR_TO_CTX } }; 8187 static const struct bpf_reg_types ringbuf_mem_types = { .types = { PTR_TO_MEM | MEM_RINGBUF } }; 8188 static const struct bpf_reg_types const_map_ptr_types = { .types = { CONST_PTR_TO_MAP } }; 8189 static const struct bpf_reg_types btf_ptr_types = { 8190 .types = { 8191 PTR_TO_BTF_ID, 8192 PTR_TO_BTF_ID | PTR_TRUSTED, 8193 PTR_TO_BTF_ID | MEM_RCU, 8194 }, 8195 }; 8196 static const struct bpf_reg_types percpu_btf_ptr_types = { 8197 .types = { 8198 PTR_TO_BTF_ID | MEM_PERCPU, 8199 PTR_TO_BTF_ID | MEM_PERCPU | MEM_RCU, 8200 PTR_TO_BTF_ID | MEM_PERCPU | PTR_TRUSTED, 8201 } 8202 }; 8203 static const struct bpf_reg_types func_ptr_types = { .types = { PTR_TO_FUNC } }; 8204 static const struct bpf_reg_types stack_ptr_types = { .types = { PTR_TO_STACK } }; 8205 static const struct bpf_reg_types const_str_ptr_types = { .types = { PTR_TO_MAP_VALUE } }; 8206 static const struct bpf_reg_types timer_types = { .types = { PTR_TO_MAP_VALUE } }; 8207 static const struct bpf_reg_types kptr_xchg_dest_types = { 8208 .types = { 8209 PTR_TO_MAP_VALUE, 8210 PTR_TO_BTF_ID | MEM_ALLOC, 8211 PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF, 8212 PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF | MEM_RCU, 8213 } 8214 }; 8215 static const struct bpf_reg_types dynptr_types = { 8216 .types = { 8217 PTR_TO_STACK, 8218 CONST_PTR_TO_DYNPTR, 8219 } 8220 }; 8221 8222 static const struct bpf_reg_types *compatible_reg_types[__BPF_ARG_TYPE_MAX] = { 8223 [ARG_PTR_TO_MAP_KEY] = &mem_types, 8224 [ARG_PTR_TO_MAP_VALUE] = &mem_types, 8225 [ARG_MEM_SIZE] = &scalar_types, 8226 [ARG_MEM_SIZE_OR_ZERO] = &scalar_types, 8227 [ARG_CONST_ALLOC_SIZE_OR_ZERO] = &scalar_types, 8228 [ARG_SCALAR] = &scalar_types, 8229 [ARG_CONST_MAP_PTR] = &const_map_ptr_types, 8230 [ARG_PTR_TO_CTX] = &context_types, 8231 [ARG_PTR_TO_SOCK_COMMON] = &sock_types, 8232 #ifdef CONFIG_NET 8233 [ARG_PTR_TO_BTF_ID_SOCK_COMMON] = &btf_id_sock_common_types, 8234 #endif 8235 [ARG_PTR_TO_SOCKET] = &fullsock_types, 8236 [ARG_PTR_TO_BTF_ID] = &btf_ptr_types, 8237 [ARG_PTR_TO_SPIN_LOCK] = &spin_lock_types, 8238 [ARG_PTR_TO_MEM] = &mem_types, 8239 [ARG_PTR_TO_RINGBUF_MEM] = &ringbuf_mem_types, 8240 [ARG_PTR_TO_PERCPU_BTF_ID] = &percpu_btf_ptr_types, 8241 [ARG_PTR_TO_FUNC] = &func_ptr_types, 8242 [ARG_PTR_TO_STACK] = &stack_ptr_types, 8243 [ARG_PTR_TO_CONST_STR] = &const_str_ptr_types, 8244 [ARG_PTR_TO_TIMER] = &timer_types, 8245 [ARG_KPTR_XCHG_DEST] = &kptr_xchg_dest_types, 8246 [ARG_PTR_TO_DYNPTR] = &dynptr_types, 8247 }; 8248 8249 static void bpf_diag_call_arg(struct bpf_verifier_env *env, u32 insn_idx, argno_t argno, 8250 const char *call_name, const char *reason, 8251 const char *suggestion) 8252 { 8253 int arg = arg_from_argno(argno); 8254 int regno = reg_from_argno(argno); 8255 int stack_slot = -1; 8256 8257 if (arg < 0 && regno >= BPF_REG_1 && regno <= BPF_REG_5) 8258 arg = regno; 8259 if (arg > MAX_BPF_FUNC_REG_ARGS) 8260 stack_slot = arg - MAX_BPF_FUNC_REG_ARGS - 1; 8261 8262 bpf_diag_call_type(env, insn_idx, arg, regno, stack_slot, 8263 call_name && *call_name ? call_name : "call", 8264 reg_arg_name(env, argno), reason, suggestion); 8265 } 8266 8267 static const char *bpf_diag_arg_name(struct bpf_verifier_env *env, argno_t argno) 8268 { 8269 return bpf_diag_fmt(env, "%s", reg_arg_name(env, argno)); 8270 } 8271 8272 __printf(6, 7) static void bpf_diag_call_arg_fmt(struct bpf_verifier_env *env, u32 insn_idx, 8273 argno_t argno, const char *call_name, 8274 const char *suggestion, const char *fmt, ...) 8275 { 8276 const char *reason; 8277 va_list args; 8278 8279 va_start(args, fmt); 8280 reason = bpf_diag_vfmt(env, fmt, args); 8281 va_end(args); 8282 8283 bpf_diag_call_arg(env, insn_idx, argno, call_name, reason, suggestion); 8284 } 8285 8286 static const char *bpf_diag_expected_reg_types(struct bpf_verifier_env *env, 8287 const enum bpf_reg_type *types, int count) 8288 { 8289 size_t len = 0, size = 1; 8290 char *buf; 8291 int i; 8292 8293 for (i = 0; i < count; i++) 8294 size += strlen(reg_type_str(env, types[i])) + (i ? 2 : 0); 8295 8296 buf = bpf_diag_fmt_buf(env, size); 8297 if (!buf) 8298 return ""; 8299 8300 for (i = 0; i < count; i++) 8301 len += scnprintf(buf + len, size - len, "%s%s", i ? ", " : "", 8302 reg_type_str(env, types[i])); 8303 return buf; 8304 } 8305 8306 static int check_reg_type(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, 8307 enum bpf_arg_type arg_type, const u32 *arg_btf_id, 8308 struct bpf_call_arg_meta *meta, const char *call_name) 8309 { 8310 enum bpf_reg_type expected, type = reg->type; 8311 const struct bpf_reg_types *compatible; 8312 const char *actual, *accepted; 8313 int i, j, err; 8314 8315 compatible = compatible_reg_types[base_type(arg_type)]; 8316 if (!compatible) { 8317 verifier_bug(env, "unsupported arg type %d", arg_type); 8318 return -EFAULT; 8319 } 8320 8321 /* ARG_PTR_TO_MEM + RDONLY is compatible with PTR_TO_MEM and PTR_TO_MEM + RDONLY, 8322 * but ARG_PTR_TO_MEM is compatible only with PTR_TO_MEM and NOT with PTR_TO_MEM + RDONLY 8323 * 8324 * Same for MAYBE_NULL: 8325 * 8326 * ARG_PTR_TO_MEM + MAYBE_NULL is compatible with PTR_TO_MEM and PTR_TO_MEM + MAYBE_NULL, 8327 * but ARG_PTR_TO_MEM is compatible only with PTR_TO_MEM but NOT with PTR_TO_MEM + MAYBE_NULL 8328 * 8329 * ARG_PTR_TO_MEM is compatible with PTR_TO_MEM that is tagged with a dynptr type. 8330 * 8331 * Therefore we fold these flags depending on the arg_type before comparison. 8332 */ 8333 if (arg_type & MEM_RDONLY) 8334 type &= ~MEM_RDONLY; 8335 if (arg_type & PTR_MAYBE_NULL) 8336 type &= ~PTR_MAYBE_NULL; 8337 if (base_type(arg_type) == ARG_PTR_TO_MEM) 8338 type &= ~DYNPTR_TYPE_FLAG_MASK; 8339 8340 /* Local kptr types are allowed as the source argument of bpf_kptr_xchg */ 8341 if (meta->func_id == BPF_FUNC_kptr_xchg && type_is_alloc(type) && reg_from_argno(argno) == BPF_REG_2) { 8342 type &= ~MEM_ALLOC; 8343 type &= ~MEM_PERCPU; 8344 } 8345 8346 for (i = 0; i < ARRAY_SIZE(compatible->types); i++) { 8347 expected = compatible->types[i]; 8348 if (expected == NOT_INIT) 8349 break; 8350 8351 if (type == expected) 8352 goto found; 8353 } 8354 8355 verbose(env, "%s type=%s expected=", reg_arg_name(env, argno), reg_type_str(env, reg->type)); 8356 for (j = 0; j + 1 < i; j++) 8357 verbose(env, "%s, ", reg_type_str(env, compatible->types[j])); 8358 verbose(env, "%s\n", reg_type_str(env, compatible->types[j])); 8359 actual = bpf_diag_fmt(env, "%s", reg_type_str(env, reg->type)); 8360 accepted = bpf_diag_expected_reg_types(env, compatible->types, i); 8361 bpf_diag_call_arg_fmt(env, env->insn_idx, argno, call_name, 8362 "Pass a value with one of the accepted pointer or scalar types for this call.", 8363 "it has type %s, but this argument accepts %s", 8364 actual, accepted); 8365 return -EACCES; 8366 8367 found: 8368 if (base_type(reg->type) != PTR_TO_BTF_ID) 8369 return 0; 8370 8371 if (compatible == &mem_types) { 8372 if (!(arg_type & MEM_RDONLY)) { 8373 verbose(env, 8374 "%s() may write into memory pointed by %s type=%s\n", 8375 func_id_name(meta->func_id), 8376 reg_arg_name(env, argno), reg_type_str(env, reg->type)); 8377 return -EACCES; 8378 } 8379 return 0; 8380 } 8381 8382 switch ((int)reg->type) { 8383 case PTR_TO_BTF_ID: 8384 case PTR_TO_BTF_ID | PTR_TRUSTED: 8385 case PTR_TO_BTF_ID | PTR_TRUSTED | PTR_MAYBE_NULL: 8386 case PTR_TO_BTF_ID | MEM_RCU: 8387 case PTR_TO_BTF_ID | PTR_MAYBE_NULL: 8388 case PTR_TO_BTF_ID | PTR_MAYBE_NULL | MEM_RCU: 8389 { 8390 /* For bpf_sk_release, it needs to match against first member 8391 * 'struct sock_common', hence make an exception for it. This 8392 * allows bpf_sk_release to work for multiple socket types. 8393 */ 8394 bool strict_type_match = arg_type_is_release(arg_type) && 8395 meta->func_id != BPF_FUNC_sk_release; 8396 8397 if (type_may_be_null(reg->type) && 8398 (!type_may_be_null(arg_type) || arg_type_is_release(arg_type))) { 8399 verbose(env, "Possibly NULL pointer passed to helper %s\n", 8400 reg_arg_name(env, argno)); 8401 bpf_diag_call_arg( 8402 env, env->insn_idx, argno, call_name, 8403 "the pointer may be NULL, but this call requires a non-NULL pointer", 8404 "Add a NULL check and make the call only on the non-NULL path."); 8405 return -EACCES; 8406 } 8407 8408 if (!arg_btf_id) { 8409 if (!compatible->btf_id) { 8410 verifier_bug(env, "missing arg compatible BTF ID"); 8411 return -EFAULT; 8412 } 8413 arg_btf_id = compatible->btf_id; 8414 } 8415 8416 if (meta->func_id == BPF_FUNC_kptr_xchg) { 8417 if (map_kptr_match_type(env, meta->kptr_field, reg, reg_from_argno(argno))) 8418 return -EACCES; 8419 } else { 8420 if (arg_btf_id == BPF_PTR_POISON) { 8421 verbose(env, "verifier internal error:"); 8422 verbose(env, "%s has non-overwritten BPF_PTR_POISON type\n", 8423 reg_arg_name(env, argno)); 8424 return -EACCES; 8425 } 8426 8427 err = __check_ptr_off_reg(env, reg, argno, true); 8428 if (err) 8429 return err; 8430 8431 if (!btf_struct_ids_match(&env->log, reg->btf, reg->btf_id, 8432 reg->var_off.value, btf_vmlinux, *arg_btf_id, 8433 strict_type_match, !type_is_alloc(reg->type))) { 8434 verbose(env, "%s is of type %s but %s is expected\n", 8435 reg_arg_name(env, argno), 8436 btf_type_name(reg->btf, reg->btf_id), 8437 btf_type_name(btf_vmlinux, *arg_btf_id)); 8438 return -EACCES; 8439 } 8440 } 8441 break; 8442 } 8443 case PTR_TO_BTF_ID | MEM_ALLOC: 8444 case PTR_TO_BTF_ID | MEM_PERCPU | MEM_ALLOC: 8445 case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF: 8446 case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF | MEM_RCU: 8447 if (meta->func_id != BPF_FUNC_spin_lock && meta->func_id != BPF_FUNC_spin_unlock && 8448 meta->func_id != BPF_FUNC_kptr_xchg) { 8449 verifier_bug(env, "unimplemented handling of MEM_ALLOC"); 8450 return -EFAULT; 8451 } 8452 /* Check if local kptr in src arg matches kptr in dst arg */ 8453 if (meta->func_id == BPF_FUNC_kptr_xchg) { 8454 int regno = reg_from_argno(argno); 8455 8456 if (regno == BPF_REG_2 && 8457 map_kptr_match_type(env, meta->kptr_field, reg, regno)) 8458 return -EACCES; 8459 } 8460 break; 8461 case PTR_TO_BTF_ID | MEM_PERCPU: 8462 case PTR_TO_BTF_ID | MEM_PERCPU | MEM_RCU: 8463 case PTR_TO_BTF_ID | MEM_PERCPU | PTR_TRUSTED: 8464 /* Handled by helper specific checks */ 8465 break; 8466 default: 8467 verifier_bug(env, "invalid PTR_TO_BTF_ID register for type match"); 8468 return -EFAULT; 8469 } 8470 return 0; 8471 } 8472 8473 static struct btf_field * 8474 reg_find_field_offset(const struct bpf_reg_state *reg, s32 off, u32 fields) 8475 { 8476 struct btf_field *field; 8477 struct btf_record *rec; 8478 8479 rec = reg_btf_record(reg); 8480 if (!rec) 8481 return NULL; 8482 8483 field = btf_record_find(rec, off, fields); 8484 if (!field) 8485 return NULL; 8486 8487 return field; 8488 } 8489 8490 static int __check_func_arg_reg_off(struct bpf_verifier_env *env, 8491 const struct bpf_reg_state *reg, argno_t argno, 8492 enum bpf_arg_type arg_type, 8493 bool btf_id_fixed_off_ok) 8494 { 8495 u32 type = reg->type; 8496 8497 /* When referenced register is passed to release function, its fixed 8498 * offset must be 0. 8499 * 8500 * We will check arg_type_is_release reg has id when storing 8501 * meta->release_regno. 8502 */ 8503 if (arg_type_is_release(arg_type)) { 8504 /* ARG_PTR_TO_DYNPTR with OBJ_RELEASE is a bit special, as it 8505 * may not directly point to the object being released, but to 8506 * dynptr pointing to such object, which might be at some offset 8507 * on the stack. In that case, we simply to fallback to the 8508 * default handling. 8509 */ 8510 if (arg_type_is_dynptr(arg_type) && type == PTR_TO_STACK) 8511 return 0; 8512 8513 /* Doing check_ptr_off_reg check for the offset will catch this 8514 * because fixed_off_ok is false, but checking here allows us 8515 * to give the user a better error message. 8516 */ 8517 if (!tnum_is_const(reg->var_off) || reg->var_off.value != 0) { 8518 verbose(env, "%s must have zero offset when passed to release func or trusted arg to kfunc\n", 8519 reg_arg_name(env, argno)); 8520 return -EINVAL; 8521 } 8522 } 8523 8524 switch (type) { 8525 /* Pointer types where both fixed and variable offset is explicitly allowed: */ 8526 case PTR_TO_STACK: 8527 case PTR_TO_PACKET: 8528 case PTR_TO_PACKET_META: 8529 case PTR_TO_MAP_KEY: 8530 case PTR_TO_MAP_VALUE: 8531 case PTR_TO_MEM: 8532 case PTR_TO_MEM | MEM_RDONLY: 8533 case PTR_TO_MEM | MEM_RINGBUF: 8534 case PTR_TO_BUF: 8535 case PTR_TO_BUF | MEM_RDONLY: 8536 case PTR_TO_ARENA: 8537 case SCALAR_VALUE: 8538 return 0; 8539 /* All the rest must be rejected, except PTR_TO_BTF_ID which allows 8540 * fixed offset. 8541 */ 8542 case PTR_TO_BTF_ID: 8543 case PTR_TO_BTF_ID | MEM_ALLOC: 8544 case PTR_TO_BTF_ID | PTR_TRUSTED: 8545 case PTR_TO_BTF_ID | MEM_RCU: 8546 case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF: 8547 case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF | MEM_RCU: 8548 /* When referenced PTR_TO_BTF_ID is passed to release function, 8549 * its fixed offset must be 0. In the other cases, fixed offset 8550 * can be non-zero unless the caller requires otherwise. 8551 * var_off always must be 0 for PTR_TO_BTF_ID, hence we still 8552 * need to do checks instead of returning. 8553 */ 8554 return __check_ptr_off_reg(env, reg, argno, btf_id_fixed_off_ok); 8555 case PTR_TO_CTX: 8556 /* 8557 * Allow fixed and variable offsets for syscall context, but 8558 * only when the argument is passed as memory, not ctx, 8559 * otherwise we may get modified ctx in tail called programs and 8560 * global subprogs (that may act as extension prog hooks). 8561 */ 8562 if (arg_type != ARG_PTR_TO_CTX && is_var_ctx_off_allowed(env->prog)) 8563 return 0; 8564 fallthrough; 8565 default: 8566 return __check_ptr_off_reg(env, reg, argno, false); 8567 } 8568 } 8569 8570 static int check_func_arg_reg_off(struct bpf_verifier_env *env, 8571 const struct bpf_reg_state *reg, argno_t argno, 8572 enum bpf_arg_type arg_type) 8573 { 8574 return __check_func_arg_reg_off(env, reg, argno, arg_type, true); 8575 } 8576 8577 static int check_arg_const_str(struct bpf_verifier_env *env, 8578 struct bpf_reg_state *reg, argno_t argno) 8579 { 8580 struct bpf_map *map = reg->map_ptr; 8581 int err; 8582 int map_off; 8583 u64 map_addr; 8584 char *str_ptr; 8585 8586 if (reg->type != PTR_TO_MAP_VALUE) 8587 return -EINVAL; 8588 8589 if (map->map_type == BPF_MAP_TYPE_INSN_ARRAY) { 8590 verbose(env, "%s points to insn_array map which cannot be used as const string\n", 8591 reg_arg_name(env, argno)); 8592 return -EACCES; 8593 } 8594 8595 if (map->map_type == BPF_MAP_TYPE_PERCPU_ARRAY) { 8596 verbose(env, "%s points to percpu_array map which cannot be used as const string\n", 8597 reg_arg_name(env, argno)); 8598 return -EACCES; 8599 } 8600 8601 if (!bpf_map_is_rdonly(map)) { 8602 verbose(env, "%s does not point to a readonly map'\n", reg_arg_name(env, argno)); 8603 return -EACCES; 8604 } 8605 8606 if (!tnum_is_const(reg->var_off)) { 8607 verbose(env, "%s is not a constant address'\n", reg_arg_name(env, argno)); 8608 return -EACCES; 8609 } 8610 8611 if (!map->ops->map_direct_value_addr) { 8612 verbose(env, "no direct value access support for this map type\n"); 8613 return -EACCES; 8614 } 8615 8616 err = check_map_access(env, reg, argno, 0, 8617 map->value_size - reg->var_off.value, false, 8618 ACCESS_HELPER); 8619 if (err) 8620 return err; 8621 8622 map_off = reg->var_off.value; 8623 err = map->ops->map_direct_value_addr(map, &map_addr, map_off); 8624 if (err) { 8625 verbose(env, "direct value access on string failed\n"); 8626 return err; 8627 } 8628 8629 str_ptr = (char *)(long)(map_addr); 8630 if (!strnchr(str_ptr + map_off, map->value_size - map_off, 0)) { 8631 verbose(env, "string is not zero-terminated\n"); 8632 return -EINVAL; 8633 } 8634 return 0; 8635 } 8636 8637 /* Returns constant key value in `value` if possible, else negative error */ 8638 static int get_constant_map_key(struct bpf_verifier_env *env, 8639 struct bpf_reg_state *key, 8640 u32 key_size, 8641 s64 *value) 8642 { 8643 struct bpf_func_state *state = bpf_func(env, key); 8644 struct bpf_reg_state *reg; 8645 int slot, spi, off; 8646 int spill_size = 0; 8647 int zero_size = 0; 8648 int stack_off; 8649 int i, err; 8650 u8 *stype; 8651 8652 if (!env->bpf_capable) 8653 return -EOPNOTSUPP; 8654 if (key->type != PTR_TO_STACK) 8655 return -EOPNOTSUPP; 8656 if (!tnum_is_const(key->var_off)) 8657 return -EOPNOTSUPP; 8658 8659 stack_off = key->var_off.value; 8660 slot = -stack_off - 1; 8661 spi = slot / BPF_REG_SIZE; 8662 off = slot % BPF_REG_SIZE; 8663 stype = state->stack[spi].slot_type; 8664 8665 /* First handle precisely tracked STACK_ZERO */ 8666 for (i = off; i >= 0 && stype[i] == STACK_ZERO; i--) 8667 zero_size++; 8668 if (zero_size >= key_size) { 8669 *value = 0; 8670 return 0; 8671 } 8672 8673 /* Check that stack contains a scalar spill of expected size */ 8674 if (!bpf_is_spilled_scalar_reg(&state->stack[spi])) 8675 return -EOPNOTSUPP; 8676 for (i = off; i >= 0 && stype[i] == STACK_SPILL; i--) 8677 spill_size++; 8678 if (spill_size != key_size) 8679 return -EOPNOTSUPP; 8680 8681 reg = &state->stack[spi].spilled_ptr; 8682 if (!tnum_is_const(reg->var_off)) 8683 /* Stack value not statically known */ 8684 return -EOPNOTSUPP; 8685 8686 /* We are relying on a constant value. So mark as precise 8687 * to prevent pruning on it. 8688 */ 8689 bpf_bt_set_frame_slot(&env->bt, key->frameno, spi); 8690 err = mark_chain_precision_batch(env, env->cur_state); 8691 if (err < 0) 8692 return err; 8693 8694 *value = reg->var_off.value; 8695 return 0; 8696 } 8697 8698 static bool can_elide_value_nullness(const struct bpf_map *map); 8699 8700 static int process_map_ptr_arg(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 8701 argno_t argno, struct bpf_call_arg_meta *meta) 8702 { 8703 /* Use map_uid (which is unique id of inner map) to reject: 8704 * inner_map1 = bpf_map_lookup_elem(outer_map, key1) 8705 * inner_map2 = bpf_map_lookup_elem(outer_map, key2) 8706 * if (inner_map1 && inner_map2) { 8707 * timer = bpf_map_lookup_elem(inner_map1); 8708 * if (timer) 8709 * // mismatch would have been allowed 8710 * bpf_timer_init(timer, inner_map2); 8711 * } 8712 * 8713 * Comparing map_ptr is enough to distinguish normal and outer maps. 8714 */ 8715 if (meta->map.ptr && 8716 (meta->map.ptr != reg->map_ptr || meta->map.uid != reg->map_uid)) { 8717 argno_t obj_argno = argno_from_reg(reg_from_argno(argno) - 1); 8718 struct btf_record *rec = meta->map.ptr->record; 8719 const char *obj_name = "workqueue"; 8720 8721 if (rec->timer_off >= 0) 8722 obj_name = "timer"; 8723 else if (rec->task_work_off >= 0) 8724 obj_name = "bpf_task_work"; 8725 8726 verbose(env, "%s pointer in %s map_uid=%d ", 8727 obj_name, reg_arg_name(env, obj_argno), meta->map.uid); 8728 verbose(env, "doesn't match map pointer in %s map_uid=%d\n", 8729 reg_arg_name(env, argno), reg->map_uid); 8730 return -EINVAL; 8731 } 8732 8733 meta->map.ptr = reg->map_ptr; 8734 meta->map.uid = reg->map_uid; 8735 return 0; 8736 } 8737 8738 static int check_func_arg(struct bpf_verifier_env *env, u32 arg, 8739 struct bpf_call_arg_meta *meta, 8740 int insn_idx) 8741 { 8742 const struct bpf_func_proto *fn = meta->fn; 8743 u32 regno = BPF_REG_1 + arg; 8744 struct bpf_reg_state *reg = reg_state(env, regno); 8745 enum bpf_arg_type arg_type = fn->arg_type[arg]; 8746 argno_t argno = argno_from_reg(regno); 8747 enum bpf_reg_type type = reg->type; 8748 u32 *arg_btf_id = NULL; 8749 u32 key_size; 8750 int err = 0; 8751 8752 if (arg_type == ARG_DONTCARE) 8753 return 0; 8754 8755 err = check_reg_arg(env, regno, SRC_OP); 8756 if (err) 8757 return err; 8758 8759 if (arg_type == ARG_ANYTHING) { 8760 if (is_pointer_value(env, regno)) { 8761 verbose(env, "R%d leaks addr into helper function\n", 8762 regno); 8763 return -EACCES; 8764 } 8765 return 0; 8766 } 8767 8768 if (type_is_pkt_pointer(type) && 8769 !may_access_direct_pkt_data(env, fn, BPF_READ)) { 8770 verbose(env, "helper access to the packet is not allowed\n"); 8771 return -EACCES; 8772 } 8773 8774 if (base_type(arg_type) == ARG_PTR_TO_MAP_VALUE) { 8775 err = resolve_map_arg_type(env, meta, &arg_type); 8776 if (err) 8777 return err; 8778 } 8779 8780 if (bpf_register_is_null(reg) && type_may_be_null(arg_type)) { 8781 /* A NULL register has a SCALAR_VALUE type, so skip 8782 * type checking. 8783 */ 8784 err = mark_chain_precision(env, regno); 8785 if (err) 8786 return err; 8787 goto skip_type_check; 8788 } 8789 8790 /* arg_btf_id and arg_size are in a union. */ 8791 if (base_type(arg_type) == ARG_PTR_TO_BTF_ID || 8792 base_type(arg_type) == ARG_PTR_TO_SPIN_LOCK) 8793 arg_btf_id = fn->arg_btf_id[arg]; 8794 8795 err = check_reg_type(env, reg, argno, arg_type, arg_btf_id, meta, 8796 func_id_name(meta->func_id)); 8797 if (err) 8798 return err; 8799 8800 err = check_func_arg_reg_off(env, reg, argno, arg_type); 8801 if (err) 8802 return err; 8803 8804 skip_type_check: 8805 if (arg_type_is_release(arg_type) && !arg_type_is_dynptr(arg_type) && 8806 !reg_is_referenced(env, reg) && !bpf_register_is_null(reg)) { 8807 verbose(env, "release helper %s expects referenced PTR_TO_BTF_ID passed to %s\n", 8808 func_id_name(meta->func_id), reg_arg_name(env, argno)); 8809 bpf_diag_call_arg( 8810 env, insn_idx, argno, func_id_name(meta->func_id), 8811 "release helpers require a value that owns a live resource returned by a matching acquire helper", 8812 "Pass the resource-owning pointer returned by the matching acquire helper, and avoid calling the release helper after ownership has already been transferred or released."); 8813 return -EINVAL; 8814 } 8815 8816 if (reg_is_referenced(env, reg)) 8817 update_ref_obj(&meta->ref_obj, reg); 8818 8819 switch (base_type(arg_type)) { 8820 case ARG_CONST_MAP_PTR: 8821 /* bpf_map_xxx(map_ptr) call: remember that map_ptr */ 8822 err = process_map_ptr_arg(env, reg, argno, meta); 8823 if (err) 8824 return err; 8825 break; 8826 case ARG_PTR_TO_MAP_KEY: 8827 /* bpf_map_xxx(..., map_ptr, ..., key) call: 8828 * check that [key, key + map->key_size) are within 8829 * stack limits and initialized 8830 */ 8831 if (!meta->map.ptr) { 8832 /* in function declaration map_ptr must come before 8833 * map_key, so that it's verified and known before 8834 * we have to check map_key here. Otherwise it means 8835 * that kernel subsystem misconfigured verifier 8836 */ 8837 verifier_bug(env, "invalid map_ptr to access map->key"); 8838 return -EFAULT; 8839 } 8840 key_size = meta->map.ptr->key_size; 8841 err = check_helper_mem_access(env, reg, argno, key_size, BPF_READ, false, NULL, 8842 NULL); 8843 if (err) 8844 return err; 8845 if (can_elide_value_nullness(meta->map.ptr)) { 8846 err = get_constant_map_key(env, reg, key_size, &meta->const_map_key); 8847 if (err < 0) { 8848 meta->const_map_key = -1; 8849 if (err == -EOPNOTSUPP) 8850 err = 0; 8851 else 8852 return err; 8853 } 8854 } 8855 break; 8856 case ARG_PTR_TO_MAP_VALUE: 8857 if (type_may_be_null(arg_type) && bpf_register_is_null(reg)) 8858 return 0; 8859 8860 /* bpf_map_xxx(..., map_ptr, ..., value) call: 8861 * check [value, value + map->value_size) validity 8862 */ 8863 if (!meta->map.ptr) { 8864 /* kernel subsystem misconfigured verifier */ 8865 verifier_bug(env, "invalid map_ptr to access map->value"); 8866 return -EFAULT; 8867 } 8868 8869 /* 8870 * Disable raw mode for bpf_map_peek_elem() on a bloom filter. The helper reads 8871 * the value buffer as an input rather than filling it. 8872 */ 8873 if (meta->func_id == BPF_FUNC_map_peek_elem && 8874 meta->map.ptr->map_type == BPF_MAP_TYPE_BLOOM_FILTER) 8875 meta->arg_raw_mem.regno = 0; 8876 8877 err = check_helper_mem_access(env, reg, argno, meta->map.ptr->value_size, 8878 arg_type & MEM_WRITE ? BPF_WRITE : BPF_READ, 8879 false, meta, NULL); 8880 break; 8881 case ARG_PTR_TO_PERCPU_BTF_ID: 8882 if (!reg->btf_id) { 8883 verbose(env, "Helper has invalid btf_id in R%d\n", regno); 8884 return -EACCES; 8885 } 8886 meta->ret_btf = reg->btf; 8887 meta->ret_btf_id = reg->btf_id; 8888 break; 8889 case ARG_PTR_TO_SPIN_LOCK: 8890 if (in_rbtree_lock_required_cb(env)) { 8891 verbose(env, "can't spin_{lock,unlock} in rbtree cb\n"); 8892 return -EACCES; 8893 } 8894 if (meta->func_id == BPF_FUNC_spin_lock) { 8895 err = process_spin_lock(env, reg, argno, PROCESS_SPIN_LOCK); 8896 if (err) 8897 return err; 8898 } else if (meta->func_id == BPF_FUNC_spin_unlock) { 8899 err = process_spin_lock(env, reg, argno, 0); 8900 if (err) 8901 return err; 8902 } else { 8903 verifier_bug(env, "spin lock arg on unexpected helper"); 8904 return -EFAULT; 8905 } 8906 break; 8907 case ARG_PTR_TO_TIMER: 8908 err = process_timer_func(env, reg, argno, &meta->map); 8909 if (err) 8910 return err; 8911 break; 8912 case ARG_PTR_TO_FUNC: 8913 meta->subprogno = reg->subprogno; 8914 break; 8915 case ARG_PTR_TO_MEM: 8916 /* The access to this pointer is only checked when we hit the 8917 * next is_mem_size argument below. 8918 */ 8919 if (arg_type & MEM_FIXED_SIZE) { 8920 err = check_mem_reg(env, reg, argno_from_reg(regno), fn->arg_size[arg], 8921 arg_type & MEM_WRITE ? BPF_WRITE : BPF_READ, meta, NULL); 8922 if (err) 8923 return err; 8924 if (arg_type & MEM_ALIGNED) 8925 err = check_ptr_alignment(env, reg, 0, fn->arg_size[arg], true); 8926 } 8927 break; 8928 case ARG_MEM_SIZE: 8929 err = check_mem_size_reg(env, reg_state(env, regno - 1), reg, 8930 argno_from_reg(regno - 1), argno, 8931 fn->arg_type[arg - 1] & MEM_WRITE ? BPF_WRITE : BPF_READ, 8932 false, meta, NULL); 8933 break; 8934 case ARG_MEM_SIZE_OR_ZERO: 8935 err = check_mem_size_reg(env, reg_state(env, regno - 1), reg, 8936 argno_from_reg(regno - 1), argno, 8937 fn->arg_type[arg - 1] & MEM_WRITE ? BPF_WRITE : BPF_READ, 8938 true, meta, NULL); 8939 break; 8940 case ARG_PTR_TO_DYNPTR: 8941 err = process_dynptr_func(env, reg, argno, insn_idx, func_id_name(meta->func_id), 8942 arg_type, &meta->ref_obj, &meta->dynptr); 8943 if (err) 8944 return err; 8945 break; 8946 case ARG_CONST_ALLOC_SIZE_OR_ZERO: 8947 err = process_const_alloc_mem_size(env, reg, argno, &meta->ret_mem); 8948 if (err) 8949 return err; 8950 break; 8951 case ARG_PTR_TO_CONST_STR: 8952 { 8953 err = check_arg_const_str(env, reg, argno); 8954 if (err) 8955 return err; 8956 break; 8957 } 8958 case ARG_KPTR_XCHG_DEST: 8959 err = process_kptr_func(env, regno, meta); 8960 if (err) 8961 return err; 8962 break; 8963 } 8964 8965 return err; 8966 } 8967 8968 static bool may_update_sockmap(struct bpf_verifier_env *env, int func_id) 8969 { 8970 enum bpf_attach_type eatype = env->prog->expected_attach_type; 8971 enum bpf_prog_type type = resolve_prog_type(env->prog); 8972 8973 if (func_id != BPF_FUNC_map_update_elem && 8974 func_id != BPF_FUNC_map_delete_elem) 8975 return false; 8976 8977 /* It's not possible to get access to a locked struct sock in these 8978 * contexts, so updating is safe. 8979 */ 8980 switch (type) { 8981 case BPF_PROG_TYPE_TRACING: 8982 if (eatype == BPF_TRACE_ITER) 8983 return true; 8984 break; 8985 case BPF_PROG_TYPE_SOCK_OPS: 8986 /* map_update allowed only via dedicated helpers with event type checks */ 8987 if (func_id == BPF_FUNC_map_delete_elem) 8988 return true; 8989 break; 8990 case BPF_PROG_TYPE_SK_REUSEPORT: 8991 case BPF_PROG_TYPE_SK_LOOKUP: 8992 return true; 8993 default: 8994 break; 8995 } 8996 8997 verbose(env, "cannot update sockmap in this context\n"); 8998 return false; 8999 } 9000 9001 bool bpf_allow_tail_call_in_subprogs(struct bpf_verifier_env *env) 9002 { 9003 return env->prog->jit_requested && 9004 bpf_jit_supports_subprog_tailcalls(); 9005 } 9006 9007 static int check_map_func_compatibility(struct bpf_verifier_env *env, 9008 struct bpf_map *map, int func_id) 9009 { 9010 if (!map) 9011 return 0; 9012 9013 /* We need a two way check, first is from map perspective ... */ 9014 switch (map->map_type) { 9015 case BPF_MAP_TYPE_PROG_ARRAY: 9016 if (func_id != BPF_FUNC_tail_call) 9017 goto error; 9018 break; 9019 case BPF_MAP_TYPE_PERF_EVENT_ARRAY: 9020 if (func_id != BPF_FUNC_perf_event_read && 9021 func_id != BPF_FUNC_perf_event_output && 9022 func_id != BPF_FUNC_skb_output && 9023 func_id != BPF_FUNC_perf_event_read_value && 9024 func_id != BPF_FUNC_xdp_output) 9025 goto error; 9026 break; 9027 case BPF_MAP_TYPE_RINGBUF: 9028 if (func_id != BPF_FUNC_ringbuf_output && 9029 func_id != BPF_FUNC_ringbuf_reserve && 9030 func_id != BPF_FUNC_ringbuf_query && 9031 func_id != BPF_FUNC_ringbuf_reserve_dynptr && 9032 func_id != BPF_FUNC_ringbuf_submit_dynptr && 9033 func_id != BPF_FUNC_ringbuf_discard_dynptr) 9034 goto error; 9035 break; 9036 case BPF_MAP_TYPE_USER_RINGBUF: 9037 if (func_id != BPF_FUNC_user_ringbuf_drain) 9038 goto error; 9039 break; 9040 case BPF_MAP_TYPE_STACK_TRACE: 9041 if (func_id != BPF_FUNC_get_stackid) 9042 goto error; 9043 break; 9044 case BPF_MAP_TYPE_CGROUP_ARRAY: 9045 if (func_id != BPF_FUNC_skb_under_cgroup && 9046 func_id != BPF_FUNC_current_task_under_cgroup) 9047 goto error; 9048 break; 9049 case BPF_MAP_TYPE_CGROUP_STORAGE: 9050 case BPF_MAP_TYPE_PERCPU_CGROUP_STORAGE: 9051 if (func_id != BPF_FUNC_get_local_storage) 9052 goto error; 9053 break; 9054 case BPF_MAP_TYPE_DEVMAP: 9055 case BPF_MAP_TYPE_DEVMAP_HASH: 9056 if (func_id != BPF_FUNC_redirect_map && 9057 func_id != BPF_FUNC_map_lookup_elem) 9058 goto error; 9059 break; 9060 /* Restrict bpf side of cpumap and xskmap, open when use-cases 9061 * appear. 9062 */ 9063 case BPF_MAP_TYPE_CPUMAP: 9064 if (func_id != BPF_FUNC_redirect_map) 9065 goto error; 9066 break; 9067 case BPF_MAP_TYPE_XSKMAP: 9068 if (func_id != BPF_FUNC_redirect_map && 9069 func_id != BPF_FUNC_map_lookup_elem) 9070 goto error; 9071 break; 9072 case BPF_MAP_TYPE_ARRAY_OF_MAPS: 9073 case BPF_MAP_TYPE_HASH_OF_MAPS: 9074 if (func_id != BPF_FUNC_map_lookup_elem) 9075 goto error; 9076 break; 9077 case BPF_MAP_TYPE_SOCKMAP: 9078 if (func_id != BPF_FUNC_sk_redirect_map && 9079 func_id != BPF_FUNC_sock_map_update && 9080 func_id != BPF_FUNC_msg_redirect_map && 9081 func_id != BPF_FUNC_sk_select_reuseport && 9082 func_id != BPF_FUNC_map_lookup_elem && 9083 !may_update_sockmap(env, func_id)) 9084 goto error; 9085 break; 9086 case BPF_MAP_TYPE_SOCKHASH: 9087 if (func_id != BPF_FUNC_sk_redirect_hash && 9088 func_id != BPF_FUNC_sock_hash_update && 9089 func_id != BPF_FUNC_msg_redirect_hash && 9090 func_id != BPF_FUNC_sk_select_reuseport && 9091 func_id != BPF_FUNC_map_lookup_elem && 9092 !may_update_sockmap(env, func_id)) 9093 goto error; 9094 break; 9095 case BPF_MAP_TYPE_REUSEPORT_SOCKARRAY: 9096 if (func_id != BPF_FUNC_sk_select_reuseport) 9097 goto error; 9098 break; 9099 case BPF_MAP_TYPE_QUEUE: 9100 case BPF_MAP_TYPE_STACK: 9101 if (func_id != BPF_FUNC_map_peek_elem && 9102 func_id != BPF_FUNC_map_pop_elem && 9103 func_id != BPF_FUNC_map_push_elem) 9104 goto error; 9105 break; 9106 case BPF_MAP_TYPE_SK_STORAGE: 9107 if (func_id != BPF_FUNC_sk_storage_get && 9108 func_id != BPF_FUNC_sk_storage_delete && 9109 func_id != BPF_FUNC_kptr_xchg) 9110 goto error; 9111 break; 9112 case BPF_MAP_TYPE_INODE_STORAGE: 9113 if (func_id != BPF_FUNC_inode_storage_get && 9114 func_id != BPF_FUNC_inode_storage_delete && 9115 func_id != BPF_FUNC_kptr_xchg) 9116 goto error; 9117 break; 9118 case BPF_MAP_TYPE_TASK_STORAGE: 9119 if (func_id != BPF_FUNC_task_storage_get && 9120 func_id != BPF_FUNC_task_storage_delete && 9121 func_id != BPF_FUNC_kptr_xchg) 9122 goto error; 9123 break; 9124 case BPF_MAP_TYPE_CGRP_STORAGE: 9125 if (func_id != BPF_FUNC_cgrp_storage_get && 9126 func_id != BPF_FUNC_cgrp_storage_delete && 9127 func_id != BPF_FUNC_kptr_xchg) 9128 goto error; 9129 break; 9130 case BPF_MAP_TYPE_BLOOM_FILTER: 9131 if (func_id != BPF_FUNC_map_peek_elem && 9132 func_id != BPF_FUNC_map_push_elem) 9133 goto error; 9134 break; 9135 case BPF_MAP_TYPE_INSN_ARRAY: 9136 goto error; 9137 default: 9138 break; 9139 } 9140 9141 /* ... and second from the function itself. */ 9142 switch (func_id) { 9143 case BPF_FUNC_tail_call: 9144 if (map->map_type != BPF_MAP_TYPE_PROG_ARRAY) 9145 goto error; 9146 if (env->subprog_cnt > 1 && !bpf_allow_tail_call_in_subprogs(env)) { 9147 verbose(env, "mixing of tail_calls and bpf-to-bpf calls is not supported\n"); 9148 return -EINVAL; 9149 } 9150 break; 9151 case BPF_FUNC_perf_event_read: 9152 case BPF_FUNC_perf_event_output: 9153 case BPF_FUNC_perf_event_read_value: 9154 case BPF_FUNC_skb_output: 9155 case BPF_FUNC_xdp_output: 9156 if (map->map_type != BPF_MAP_TYPE_PERF_EVENT_ARRAY) 9157 goto error; 9158 break; 9159 case BPF_FUNC_ringbuf_output: 9160 case BPF_FUNC_ringbuf_reserve: 9161 case BPF_FUNC_ringbuf_query: 9162 case BPF_FUNC_ringbuf_reserve_dynptr: 9163 case BPF_FUNC_ringbuf_submit_dynptr: 9164 case BPF_FUNC_ringbuf_discard_dynptr: 9165 if (map->map_type != BPF_MAP_TYPE_RINGBUF) 9166 goto error; 9167 break; 9168 case BPF_FUNC_user_ringbuf_drain: 9169 if (map->map_type != BPF_MAP_TYPE_USER_RINGBUF) 9170 goto error; 9171 break; 9172 case BPF_FUNC_get_stackid: 9173 if (map->map_type != BPF_MAP_TYPE_STACK_TRACE) 9174 goto error; 9175 break; 9176 case BPF_FUNC_current_task_under_cgroup: 9177 case BPF_FUNC_skb_under_cgroup: 9178 if (map->map_type != BPF_MAP_TYPE_CGROUP_ARRAY) 9179 goto error; 9180 break; 9181 case BPF_FUNC_redirect_map: 9182 if (map->map_type != BPF_MAP_TYPE_DEVMAP && 9183 map->map_type != BPF_MAP_TYPE_DEVMAP_HASH && 9184 map->map_type != BPF_MAP_TYPE_CPUMAP && 9185 map->map_type != BPF_MAP_TYPE_XSKMAP) 9186 goto error; 9187 break; 9188 case BPF_FUNC_sk_redirect_map: 9189 case BPF_FUNC_msg_redirect_map: 9190 case BPF_FUNC_sock_map_update: 9191 if (map->map_type != BPF_MAP_TYPE_SOCKMAP) 9192 goto error; 9193 break; 9194 case BPF_FUNC_sk_redirect_hash: 9195 case BPF_FUNC_msg_redirect_hash: 9196 case BPF_FUNC_sock_hash_update: 9197 if (map->map_type != BPF_MAP_TYPE_SOCKHASH) 9198 goto error; 9199 break; 9200 case BPF_FUNC_get_local_storage: 9201 if (map->map_type != BPF_MAP_TYPE_CGROUP_STORAGE && 9202 map->map_type != BPF_MAP_TYPE_PERCPU_CGROUP_STORAGE) 9203 goto error; 9204 break; 9205 case BPF_FUNC_sk_select_reuseport: 9206 if (map->map_type != BPF_MAP_TYPE_REUSEPORT_SOCKARRAY && 9207 map->map_type != BPF_MAP_TYPE_SOCKMAP && 9208 map->map_type != BPF_MAP_TYPE_SOCKHASH) 9209 goto error; 9210 break; 9211 case BPF_FUNC_map_pop_elem: 9212 if (map->map_type != BPF_MAP_TYPE_QUEUE && 9213 map->map_type != BPF_MAP_TYPE_STACK) 9214 goto error; 9215 break; 9216 case BPF_FUNC_map_peek_elem: 9217 case BPF_FUNC_map_push_elem: 9218 if (map->map_type != BPF_MAP_TYPE_QUEUE && 9219 map->map_type != BPF_MAP_TYPE_STACK && 9220 map->map_type != BPF_MAP_TYPE_BLOOM_FILTER) 9221 goto error; 9222 break; 9223 case BPF_FUNC_map_lookup_percpu_elem: 9224 if (map->map_type != BPF_MAP_TYPE_PERCPU_ARRAY && 9225 map->map_type != BPF_MAP_TYPE_PERCPU_HASH && 9226 map->map_type != BPF_MAP_TYPE_LRU_PERCPU_HASH) 9227 goto error; 9228 break; 9229 case BPF_FUNC_sk_storage_get: 9230 case BPF_FUNC_sk_storage_delete: 9231 if (map->map_type != BPF_MAP_TYPE_SK_STORAGE) 9232 goto error; 9233 break; 9234 case BPF_FUNC_inode_storage_get: 9235 case BPF_FUNC_inode_storage_delete: 9236 if (map->map_type != BPF_MAP_TYPE_INODE_STORAGE) 9237 goto error; 9238 break; 9239 case BPF_FUNC_task_storage_get: 9240 case BPF_FUNC_task_storage_delete: 9241 if (map->map_type != BPF_MAP_TYPE_TASK_STORAGE) 9242 goto error; 9243 break; 9244 case BPF_FUNC_cgrp_storage_get: 9245 case BPF_FUNC_cgrp_storage_delete: 9246 if (map->map_type != BPF_MAP_TYPE_CGRP_STORAGE) 9247 goto error; 9248 break; 9249 default: 9250 break; 9251 } 9252 9253 return 0; 9254 error: 9255 verbose(env, "cannot pass map_type %d into func %s#%d\n", 9256 map->map_type, func_id_name(func_id), func_id); 9257 return -EINVAL; 9258 } 9259 9260 static bool check_raw_mode_ok(const struct bpf_func_proto *fn, struct bpf_call_arg_meta *meta) 9261 { 9262 int i; 9263 9264 for (i = 0; i < ARRAY_SIZE(fn->arg_type); i++) { 9265 if (fn->arg_type[i] == ARG_DONTCARE) 9266 break; 9267 if (!arg_type_is_raw_mem(fn->arg_type[i])) 9268 continue; 9269 if (meta->arg_raw_mem.regno) 9270 return false; 9271 meta->arg_raw_mem.regno = i + 1; 9272 } 9273 9274 return true; 9275 } 9276 9277 static bool check_args_pair_invalid(const struct bpf_func_proto *fn, int arg) 9278 { 9279 bool is_fixed = fn->arg_type[arg] & MEM_FIXED_SIZE; 9280 bool has_size = fn->arg_size[arg] != 0; 9281 bool is_next_size = false; 9282 9283 if (arg + 1 < ARRAY_SIZE(fn->arg_type)) 9284 is_next_size = arg_type_is_mem_size(fn->arg_type[arg + 1]); 9285 9286 if (base_type(fn->arg_type[arg]) != ARG_PTR_TO_MEM) 9287 return is_next_size; 9288 9289 return has_size == is_next_size || is_next_size == is_fixed; 9290 } 9291 9292 static bool check_arg_pair_ok(const struct bpf_func_proto *fn) 9293 { 9294 /* bpf_xxx(..., buf, len) call will access 'len' 9295 * bytes from memory 'buf'. Both arg types need 9296 * to be paired, so make sure there's no buggy 9297 * helper function specification. 9298 */ 9299 if (arg_type_is_mem_size(fn->arg1_type) || 9300 check_args_pair_invalid(fn, 0) || 9301 check_args_pair_invalid(fn, 1) || 9302 check_args_pair_invalid(fn, 2) || 9303 check_args_pair_invalid(fn, 3) || 9304 check_args_pair_invalid(fn, 4)) 9305 return false; 9306 9307 return true; 9308 } 9309 9310 static bool check_btf_id_ok(const struct bpf_func_proto *fn) 9311 { 9312 int i; 9313 9314 for (i = 0; i < ARRAY_SIZE(fn->arg_type); i++) { 9315 if (fn->arg_type[i] == ARG_DONTCARE) 9316 break; 9317 if (base_type(fn->arg_type[i]) == ARG_PTR_TO_BTF_ID) 9318 return !!fn->arg_btf_id[i]; 9319 if (base_type(fn->arg_type[i]) == ARG_PTR_TO_SPIN_LOCK) 9320 return fn->arg_btf_id[i] == BPF_PTR_POISON; 9321 if (base_type(fn->arg_type[i]) != ARG_PTR_TO_BTF_ID && fn->arg_btf_id[i] && 9322 /* arg_btf_id and arg_size are in a union. */ 9323 (base_type(fn->arg_type[i]) != ARG_PTR_TO_MEM || 9324 !(fn->arg_type[i] & MEM_FIXED_SIZE))) 9325 return false; 9326 } 9327 9328 return true; 9329 } 9330 9331 static bool check_mem_arg_rw_flag_ok(const struct bpf_func_proto *fn) 9332 { 9333 int i; 9334 9335 for (i = 0; i < ARRAY_SIZE(fn->arg_type); i++) { 9336 enum bpf_arg_type arg_type = fn->arg_type[i]; 9337 9338 if (arg_type == ARG_DONTCARE) 9339 break; 9340 if (base_type(arg_type) != ARG_PTR_TO_MEM) 9341 continue; 9342 if (!(arg_type & (MEM_WRITE | MEM_RDONLY))) 9343 return false; 9344 } 9345 9346 return true; 9347 } 9348 9349 static bool check_proto_release_reg(const struct bpf_func_proto *fn, struct bpf_call_arg_meta *meta) 9350 { 9351 int i; 9352 9353 for (i = 0; i < ARRAY_SIZE(fn->arg_type); i++) { 9354 enum bpf_arg_type arg_type = fn->arg_type[i]; 9355 9356 if (arg_type == ARG_DONTCARE) 9357 break; 9358 if (arg_type_is_release(arg_type)) { 9359 if (meta->release_regno) 9360 return false; 9361 meta->release_regno = i + 1; 9362 } 9363 } 9364 9365 return true; 9366 } 9367 9368 static int check_func_proto(const struct bpf_func_proto *fn, struct bpf_call_arg_meta *meta) 9369 { 9370 return check_raw_mode_ok(fn, meta) && 9371 check_arg_pair_ok(fn) && 9372 check_mem_arg_rw_flag_ok(fn) && 9373 check_proto_release_reg(fn, meta) && 9374 check_btf_id_ok(fn) ? 0 : -EINVAL; 9375 } 9376 9377 /* Packet data might have moved, any old PTR_TO_PACKET[_META,_END] 9378 * are now invalid, so turn them into unknown SCALAR_VALUE. 9379 * 9380 * This also applies to dynptr slices belonging to skb and xdp dynptrs, 9381 * since these slices point to packet data. 9382 */ 9383 static void clear_all_pkt_pointers(struct bpf_verifier_env *env) 9384 { 9385 struct bpf_func_state *state; 9386 struct bpf_reg_state *reg; 9387 9388 bpf_for_each_reg_in_vstate(env->cur_state, state, reg, ({ 9389 if (reg_is_pkt_pointer_any(reg) || reg_is_dynptr_slice_pkt(reg)) { 9390 bpf_diag_record_scrub(env, reg, BPF_DIAG_MOD_PKT_DATA_CHANGE); 9391 mark_reg_invalid(env, reg); 9392 } 9393 })); 9394 } 9395 9396 enum { 9397 AT_PKT_END = -1, 9398 BEYOND_PKT_END = -2, 9399 }; 9400 9401 static void mark_pkt_end(struct bpf_verifier_state *vstate, int regn, bool range_open) 9402 { 9403 struct bpf_func_state *state = vstate->frame[vstate->curframe]; 9404 struct bpf_reg_state *reg = &state->regs[regn]; 9405 9406 if (reg->type != PTR_TO_PACKET) 9407 /* PTR_TO_PACKET_META is not supported yet */ 9408 return; 9409 9410 /* The 'reg' is pkt > pkt_end or pkt >= pkt_end. 9411 * How far beyond pkt_end it goes is unknown. 9412 * if (!range_open) it's the case of pkt >= pkt_end 9413 * if (range_open) it's the case of pkt > pkt_end 9414 * hence this pointer is at least 1 byte bigger than pkt_end 9415 */ 9416 if (range_open) 9417 reg->range = BEYOND_PKT_END; 9418 else 9419 reg->range = AT_PKT_END; 9420 } 9421 9422 static int __release_reference_nomark(struct bpf_verifier_state *state, int id) 9423 { 9424 int i; 9425 9426 for (i = 0; i < state->acquired_refs; i++) { 9427 if (state->refs[i].type != REF_TYPE_PTR) 9428 continue; 9429 if (state->refs[i].id == id) { 9430 release_reference_state(state, i); 9431 return 0; 9432 } 9433 } 9434 return -EINVAL; 9435 } 9436 9437 static int release_reference_nomark(struct bpf_verifier_env *env, int id) 9438 { 9439 int err; 9440 9441 err = __release_reference_nomark(env->cur_state, id); 9442 if (!err) 9443 bpf_diag_record_ref_release(env, env->insn_idx, id); 9444 return err; 9445 } 9446 9447 static int idstack_push(struct bpf_idmap *idmap, u32 id) 9448 { 9449 int i; 9450 9451 if (!id) 9452 return 0; 9453 9454 for (i = 0; i < idmap->cnt; i++) 9455 if (idmap->map[i].old == id) 9456 return 0; 9457 9458 if (WARN_ON_ONCE(idmap->cnt >= BPF_ID_MAP_SIZE)) 9459 return -EFAULT; 9460 9461 idmap->map[idmap->cnt++].old = id; 9462 return 0; 9463 } 9464 9465 static int idstack_pop(struct bpf_idmap *idmap) 9466 { 9467 if (!idmap->cnt) 9468 return 0; 9469 9470 return idmap->map[--idmap->cnt].old; 9471 } 9472 9473 /* Release id and objects derived from it iteratively in a DFS manner */ 9474 static int release_reference(struct bpf_verifier_env *env, int id) 9475 { 9476 u32 mask = (1 << STACK_SPILL) | (1 << STACK_DYNPTR); 9477 struct bpf_verifier_state *vstate = env->cur_state; 9478 struct bpf_idmap *idstack = &env->idmap_scratch; 9479 struct bpf_stack_state *stack; 9480 struct bpf_func_state *state; 9481 struct bpf_reg_state *reg; 9482 int i, err; 9483 9484 idstack->cnt = 0; 9485 err = idstack_push(idstack, id); 9486 if (err) 9487 return err; 9488 9489 if (find_reference_state(vstate, id)) { 9490 err = release_reference_nomark(env, id); 9491 WARN_ON_ONCE(err); 9492 } 9493 9494 while ((id = idstack_pop(idstack))) { 9495 /* 9496 * Child references are inaccessible after parent is released, 9497 * any child references that exist at this point are a leak. 9498 */ 9499 for (i = 0; i < vstate->acquired_refs; i++) { 9500 if (vstate->refs[i].type != REF_TYPE_PTR) 9501 continue; 9502 if (vstate->refs[i].parent_id != id) 9503 continue; 9504 verbose(env, "Leaking reference id=%d alloc_insn=%d. Release it first.\n", 9505 vstate->refs[i].id, vstate->refs[i].insn_idx); 9506 return -EINVAL; 9507 } 9508 9509 bpf_for_each_reg_in_vstate_mask(vstate, state, reg, stack, mask, ({ 9510 if (reg->id != id && reg->parent_id != id) 9511 continue; 9512 9513 /* Free objects derived from the current object */ 9514 if (reg->parent_id == id) { 9515 err = idstack_push(idstack, reg->id); 9516 if (err) 9517 return err; 9518 } 9519 9520 /* 9521 * A dynptr occupies two stack slots that invalidate_dynptr() 9522 * clears together. Record both scrubs before invalidating it. 9523 */ 9524 if (stack && stack->slot_type[BPF_REG_SIZE - 1] == STACK_DYNPTR) { 9525 struct bpf_stack_state *dyn_stack = stack; 9526 9527 if (reg->dynptr.first_slot) 9528 dyn_stack--; 9529 bpf_diag_record_scrub(env, &dyn_stack[0].spilled_ptr, 9530 BPF_DIAG_MOD_REF_RELEASE); 9531 bpf_diag_record_scrub(env, &dyn_stack[1].spilled_ptr, 9532 BPF_DIAG_MOD_REF_RELEASE); 9533 invalidate_dynptr(env, dyn_stack); 9534 continue; 9535 } 9536 bpf_diag_record_scrub(env, reg, BPF_DIAG_MOD_REF_RELEASE); 9537 if (!stack || stack->slot_type[BPF_REG_SIZE - 1] == STACK_SPILL) 9538 mark_reg_invalid(env, reg); 9539 })); 9540 } 9541 9542 return 0; 9543 } 9544 9545 static void invalidate_non_owning_refs(struct bpf_verifier_env *env) 9546 { 9547 struct bpf_func_state *unused; 9548 struct bpf_reg_state *reg; 9549 9550 bpf_for_each_reg_in_vstate(env->cur_state, unused, reg, ({ 9551 if (type_is_non_owning_ref(reg->type)) { 9552 bpf_diag_record_scrub(env, reg, BPF_DIAG_MOD_NON_OWN_REF); 9553 mark_reg_invalid(env, reg); 9554 } 9555 })); 9556 } 9557 9558 static void invalidate_rcu_protected_refs(struct bpf_verifier_env *env) 9559 { 9560 struct bpf_stack_state *stack; 9561 struct bpf_func_state *state; 9562 struct bpf_reg_state *reg; 9563 u32 clear_mask = (1 << STACK_SPILL) | (1 << STACK_ITER); 9564 9565 bpf_for_each_reg_in_vstate_mask(env->cur_state, state, reg, stack, clear_mask, ({ 9566 if (reg->type & MEM_RCU) { 9567 bpf_diag_mod_begin(env, reg, NULL, BPF_DIAG_MOD_WRITE); 9568 reg->type &= ~(MEM_RCU | PTR_MAYBE_NULL | NON_OWN_REF); 9569 reg->type |= PTR_UNTRUSTED; 9570 bpf_diag_mod_end(env); 9571 } 9572 })); 9573 } 9574 9575 static int ref_convert_alloc_rcu_protected(struct bpf_verifier_env *env, u32 id) 9576 { 9577 struct bpf_func_state *state; 9578 struct bpf_reg_state *reg; 9579 int err; 9580 9581 err = release_reference_nomark(env, id); 9582 if (err) 9583 return err; 9584 9585 bpf_for_each_reg_in_vstate(env->cur_state, state, reg, ({ 9586 if (reg->id != id) 9587 continue; 9588 if ((reg->type & MEM_ALLOC) && (reg->type & MEM_PERCPU)) { 9589 bpf_diag_mod_begin(env, reg, NULL, BPF_DIAG_MOD_WRITE); 9590 reg->id = 0; 9591 reg->type &= ~MEM_ALLOC; 9592 reg->type |= MEM_RCU; 9593 bpf_diag_mod_end(env); 9594 } 9595 })); 9596 9597 return err; 9598 } 9599 9600 static void clear_caller_saved_regs(struct bpf_verifier_env *env, 9601 struct bpf_reg_state *regs) 9602 { 9603 int i; 9604 9605 bpf_diag_record_caller_saved(env, regs); 9606 9607 /* after the call registers r0 - r5 were scratched */ 9608 for (i = 0; i < CALLER_SAVED_REGS; i++) { 9609 bpf_mark_reg_not_init(env, ®s[caller_saved[i]]); 9610 __check_reg_arg(env, regs, caller_saved[i], DST_OP_NO_MARK); 9611 } 9612 } 9613 9614 static void invalidate_outgoing_stack_args(struct bpf_verifier_env *env, 9615 struct bpf_func_state *state) 9616 { 9617 int i, nslots = state->out_stack_arg_cnt; 9618 9619 for (i = 0; i < nslots; i++) { 9620 bpf_diag_record_scrub(env, &state->stack_arg_regs[i], BPF_DIAG_MOD_CALLER_SAVED); 9621 bpf_mark_reg_not_init(env, &state->stack_arg_regs[i]); 9622 } 9623 } 9624 9625 typedef int (*set_callee_state_fn)(struct bpf_verifier_env *env, 9626 struct bpf_func_state *caller, 9627 struct bpf_func_state *callee, 9628 int insn_idx); 9629 9630 static int set_callee_state(struct bpf_verifier_env *env, 9631 struct bpf_func_state *caller, 9632 struct bpf_func_state *callee, int insn_idx); 9633 9634 static int setup_func_entry(struct bpf_verifier_env *env, int subprog, int callsite, 9635 set_callee_state_fn set_callee_state_cb, 9636 struct bpf_verifier_state *state) 9637 { 9638 struct bpf_func_state *caller, *callee; 9639 int err; 9640 9641 if (state->curframe + 1 >= MAX_CALL_FRAMES) { 9642 verbose(env, "the call stack of %d frames is too deep\n", 9643 state->curframe + 2); 9644 return -E2BIG; 9645 } 9646 9647 if (state->frame[state->curframe + 1]) { 9648 verifier_bug(env, "Frame %d already allocated", state->curframe + 1); 9649 return -EFAULT; 9650 } 9651 9652 caller = state->frame[state->curframe]; 9653 callee = kzalloc_obj(*callee, GFP_KERNEL_ACCOUNT); 9654 if (!callee) 9655 return -ENOMEM; 9656 state->frame[state->curframe + 1] = callee; 9657 9658 /* callee cannot access r0, r6 - r9 for reading and has to write 9659 * into its own stack before reading from it. 9660 * callee can read/write into caller's stack 9661 */ 9662 init_func_state(env, callee, 9663 /* remember the callsite, it will be used by bpf_exit */ 9664 callsite, 9665 state->curframe + 1 /* frameno within this callchain */, 9666 subprog /* subprog number within this prog */); 9667 err = set_callee_state_cb(env, caller, callee, callsite); 9668 if (err) 9669 goto err_out; 9670 9671 /* only increment it after check_reg_arg() finished */ 9672 state->curframe++; 9673 9674 return 0; 9675 9676 err_out: 9677 free_func_state(callee); 9678 state->frame[state->curframe + 1] = NULL; 9679 return err; 9680 } 9681 9682 static int btf_check_func_arg_match(struct bpf_verifier_env *env, int subprog, 9683 const struct btf *btf, 9684 struct bpf_reg_state *regs) 9685 { 9686 struct bpf_subprog_info *sub = subprog_info(env, subprog); 9687 struct bpf_func_state *caller = cur_func(env); 9688 struct bpf_verifier_log *log = &env->log; 9689 struct ref_obj_desc ref_obj = {}; 9690 const struct btf_param *args; 9691 const struct btf_type *func, *func_proto; 9692 u32 i; 9693 int ret, err; 9694 9695 ret = btf_prepare_func_args(env, subprog); 9696 if (ret) { 9697 if (bpf_in_stack_arg_cnt(sub) > 0) { 9698 err = check_outgoing_stack_args(env, caller, sub->arg_cnt, 9699 bpf_subprog_name(env, subprog), 9700 NULL, NULL); 9701 if (err) 9702 return err; 9703 } 9704 return ret; 9705 } 9706 9707 func = btf_type_by_id(btf, env->prog->aux->func_info[subprog].type_id); 9708 func_proto = btf_type_by_id(btf, func->type); 9709 args = btf_params(func_proto); 9710 ret = check_outgoing_stack_args(env, caller, sub->arg_cnt, 9711 bpf_subprog_name(env, subprog), btf, args); 9712 if (ret) 9713 return ret; 9714 9715 /* check that BTF function arguments match actual types that the 9716 * verifier sees. 9717 */ 9718 for (i = 0; i < sub->arg_cnt; i++) { 9719 argno_t argno = argno_from_arg(i + 1); 9720 struct bpf_reg_state *reg = get_func_arg_reg(caller, regs, i); 9721 struct bpf_subprog_arg_info *arg = &sub->args[i]; 9722 9723 if (arg->arg_type == ARG_ANYTHING) { 9724 if (reg->type != SCALAR_VALUE) { 9725 bpf_log(log, "%s is not a scalar\n", reg_arg_name(env, argno)); 9726 return -EINVAL; 9727 } 9728 } else if (arg->arg_type & PTR_UNTRUSTED) { 9729 /* 9730 * Anything is allowed for untrusted arguments, as these are 9731 * read-only and probe read instructions would protect against 9732 * invalid memory access. 9733 */ 9734 } else if (arg->arg_type == ARG_PTR_TO_CTX) { 9735 ret = check_func_arg_reg_off(env, reg, argno, ARG_PTR_TO_CTX); 9736 if (ret < 0) 9737 return ret; 9738 /* If function expects ctx type in BTF check that caller 9739 * is passing PTR_TO_CTX. 9740 */ 9741 if (reg->type != PTR_TO_CTX) { 9742 bpf_log(log, "%s expects pointer to ctx\n", 9743 reg_arg_name(env, argno)); 9744 return -EINVAL; 9745 } 9746 } else if (base_type(arg->arg_type) == ARG_PTR_TO_MEM) { 9747 ret = check_func_arg_reg_off(env, reg, argno, ARG_DONTCARE); 9748 if (ret < 0) 9749 return ret; 9750 if (check_mem_reg(env, reg, argno, arg->mem_size, BPF_READ | BPF_WRITE, NULL, 9751 NULL)) 9752 return -EINVAL; 9753 /* 9754 * PTR_TO_PACKET get passed as PTR_TO_MEM, preventing 9755 * us from adjusting bounds tracking info. 9756 */ 9757 if ((reg_is_pkt_pointer_any(reg) || reg_is_dynptr_slice_pkt(reg)) && 9758 sub->changes_pkt_data) { 9759 bpf_log(log, "%s is a packet pointer, but func#%d may change packet data\n", 9760 reg_arg_name(env, argno), subprog); 9761 return -EINVAL; 9762 } 9763 if (!(arg->arg_type & PTR_MAYBE_NULL) && 9764 (type_may_be_null(reg->type) || bpf_register_is_null(reg))) { 9765 bpf_log(log, "%s is expected to be non-NULL\n", 9766 reg_arg_name(env, argno)); 9767 return -EINVAL; 9768 } 9769 } else if (base_type(arg->arg_type) == ARG_PTR_TO_ARENA) { 9770 /* 9771 * Can pass any value and the kernel won't crash, but 9772 * only PTR_TO_ARENA or SCALAR make sense. Everything 9773 * else is a bug in the bpf program. Point it out to 9774 * the user at the verification time instead of 9775 * run-time debug nightmare. 9776 */ 9777 if (reg->type != PTR_TO_ARENA && reg->type != SCALAR_VALUE) { 9778 bpf_log(log, "%s is not a pointer to arena or scalar.\n", 9779 reg_arg_name(env, argno)); 9780 return -EINVAL; 9781 } 9782 } else if (arg->arg_type == ARG_PTR_TO_DYNPTR) { 9783 ret = check_func_arg_reg_off(env, reg, argno, ARG_PTR_TO_DYNPTR); 9784 if (ret) 9785 return ret; 9786 9787 ret = process_dynptr_func(env, reg, argno, env->insn_idx, 9788 bpf_subprog_name(env, subprog), arg->arg_type, 9789 &ref_obj, NULL); 9790 if (ret) 9791 return ret; 9792 } else if (base_type(arg->arg_type) == ARG_PTR_TO_BTF_ID) { 9793 struct bpf_call_arg_meta meta; 9794 int err; 9795 9796 if (bpf_register_is_null(reg) && type_may_be_null(arg->arg_type)) { 9797 err = mark_arg_precision(env, argno); 9798 if (err) 9799 return err; 9800 continue; 9801 } 9802 9803 memset(&meta, 0, sizeof(meta)); /* leave func_id as zero */ 9804 err = check_reg_type(env, reg, argno, arg->arg_type, &arg->btf_id, &meta, 9805 bpf_subprog_name(env, subprog)); 9806 err = err ?: check_func_arg_reg_off(env, reg, argno, arg->arg_type); 9807 if (err) 9808 return err; 9809 } else { 9810 verifier_bug(env, "unrecognized %s type %d", 9811 reg_arg_name(env, argno), arg->arg_type); 9812 return -EFAULT; 9813 } 9814 } 9815 9816 return 0; 9817 } 9818 9819 /* Compare BTF of a function call with given bpf_reg_state. 9820 * Returns: 9821 * EFAULT - there is a verifier bug. Abort verification. 9822 * EINVAL - there is a type mismatch or BTF is not available. 9823 * 0 - BTF matches with what bpf_reg_state expects. 9824 * Only PTR_TO_CTX and SCALAR_VALUE states are recognized. 9825 */ 9826 static int btf_check_subprog_call(struct bpf_verifier_env *env, int subprog, 9827 struct bpf_reg_state *regs) 9828 { 9829 struct bpf_prog *prog = env->prog; 9830 struct btf *btf = prog->aux->btf; 9831 u32 btf_id; 9832 int err; 9833 9834 if (!prog->aux->func_info) 9835 return -EINVAL; 9836 9837 btf_id = prog->aux->func_info[subprog].type_id; 9838 if (!btf_id) 9839 return -EFAULT; 9840 9841 if (prog->aux->func_info_aux[subprog].unreliable) 9842 return -EINVAL; 9843 9844 err = btf_check_func_arg_match(env, subprog, btf, regs); 9845 /* Compiler optimizations can remove arguments from static functions 9846 * or mismatched type can be passed into a global function. 9847 * In such cases mark the function as unreliable from BTF point of view. 9848 */ 9849 if (err) 9850 prog->aux->func_info_aux[subprog].unreliable = true; 9851 return err; 9852 } 9853 9854 static int push_callback_call(struct bpf_verifier_env *env, struct bpf_insn *insn, 9855 int insn_idx, int subprog, 9856 set_callee_state_fn set_callee_state_cb) 9857 { 9858 struct bpf_verifier_state *state = env->cur_state, *callback_state; 9859 struct bpf_func_state *caller, *callee; 9860 int err; 9861 9862 caller = state->frame[state->curframe]; 9863 err = btf_check_subprog_call(env, subprog, caller->regs); 9864 if (err == -EFAULT) 9865 return err; 9866 9867 /* set_callee_state is used for direct subprog calls, but we are 9868 * interested in validating only BPF helpers that can call subprogs as 9869 * callbacks 9870 */ 9871 env->subprog_info[subprog].is_cb = true; 9872 if (bpf_pseudo_kfunc_call(insn) && 9873 !is_callback_calling_kfunc(insn->imm)) { 9874 verifier_bug(env, "kfunc %s#%d not marked as callback-calling", 9875 func_id_name(insn->imm), insn->imm); 9876 return -EFAULT; 9877 } else if (!bpf_pseudo_kfunc_call(insn) && 9878 !is_callback_calling_function(insn->imm)) { /* helper */ 9879 verifier_bug(env, "helper %s#%d not marked as callback-calling", 9880 func_id_name(insn->imm), insn->imm); 9881 return -EFAULT; 9882 } 9883 9884 if (bpf_is_async_callback_calling_insn(insn)) { 9885 struct bpf_verifier_state *async_cb; 9886 9887 /* there is no real recursion here. timer and workqueue callbacks are async */ 9888 env->subprog_info[subprog].is_async_cb = true; 9889 async_cb = push_async_cb(env, env->subprog_info[subprog].start, 9890 insn_idx, subprog, 9891 is_async_cb_sleepable(env, insn)); 9892 if (IS_ERR(async_cb)) 9893 return PTR_ERR(async_cb); 9894 callee = async_cb->frame[0]; 9895 callee->async_entry_cnt = caller->async_entry_cnt + 1; 9896 9897 /* Convert bpf_timer_set_callback() args into timer callback args */ 9898 err = set_callee_state_cb(env, caller, callee, insn_idx); 9899 if (err) 9900 return err; 9901 9902 return 0; 9903 } 9904 9905 /* for callback functions enqueue entry to callback and 9906 * proceed with next instruction within current frame. 9907 */ 9908 callback_state = push_stack(env, env->subprog_info[subprog].start, insn_idx, false); 9909 if (IS_ERR(callback_state)) 9910 return PTR_ERR(callback_state); 9911 9912 err = setup_func_entry(env, subprog, insn_idx, set_callee_state_cb, 9913 callback_state); 9914 if (err) 9915 return err; 9916 9917 callback_state->callback_unroll_depth++; 9918 callback_state->frame[callback_state->curframe - 1]->callback_depth++; 9919 caller->callback_depth = 0; 9920 return 0; 9921 } 9922 9923 static int process_bpf_exit_full(struct bpf_verifier_env *env, 9924 bool *do_print_state, bool exception_exit); 9925 9926 static int check_func_call(struct bpf_verifier_env *env, struct bpf_insn *insn, 9927 int *insn_idx) 9928 { 9929 struct bpf_verifier_state *state = env->cur_state; 9930 struct bpf_subprog_info *caller_info; 9931 u16 callee_incoming, stack_arg_cnt; 9932 struct bpf_func_state *caller; 9933 int err, subprog, target_insn; 9934 9935 target_insn = *insn_idx + insn->imm + 1; 9936 subprog = bpf_find_subprog(env, target_insn); 9937 if (verifier_bug_if(subprog < 0, env, "target of func call at insn %d is not a program", 9938 target_insn)) 9939 return -EFAULT; 9940 9941 caller = state->frame[state->curframe]; 9942 err = btf_check_subprog_call(env, subprog, caller->regs); 9943 if (err == -EFAULT) 9944 return err; 9945 if (bpf_subprog_is_global(env, subprog)) { 9946 struct bpf_func_info_aux *sub_aux = subprog_aux(env, subprog); 9947 const char *sub_name = bpf_subprog_name(env, subprog); 9948 const char *operation; 9949 bool returns_void; 9950 9951 if (env->cur_state->active_locks) { 9952 verbose(env, "global function calls are not allowed while holding a lock,\n" 9953 "use static function instead\n"); 9954 operation = bpf_diag_fmt(env, "global function %s()", sub_name); 9955 bpf_diag_ctx_active(env, *insn_idx, operation, BPF_DIAG_CONTEXT_LOCK, 9956 "Release the lock before calling the global function, or use a static function instead."); 9957 return -EINVAL; 9958 } 9959 9960 if (env->subprog_info[subprog].might_sleep && !in_sleepable_context(env)) { 9961 verbose(env, "sleepable global function %s() called in %s\n", 9962 sub_name, non_sleepable_context_description(env)); 9963 operation = bpf_diag_fmt(env, "sleepable global function %s()", sub_name); 9964 bpf_diag_ctx_forbidden(env, *insn_idx, operation, 9965 "Move the call outside the critical section, or use a non-sleepable function."); 9966 return -EINVAL; 9967 } 9968 9969 if (err) { 9970 verbose(env, "Caller passes invalid args into func#%d ('%s')\n", 9971 subprog, sub_name); 9972 return err; 9973 } 9974 9975 if (env->log.level & BPF_LOG_LEVEL) 9976 verbose(env, "Func#%d ('%s') is global and assumed valid.\n", 9977 subprog, sub_name); 9978 sub_aux->called[in_sleepable_context(env)] = true; 9979 returns_void = subprog_returns_void(env, subprog); 9980 if (env->subprog_info[subprog].changes_pkt_data) 9981 clear_all_pkt_pointers(env); 9982 if (returns_void) 9983 bpf_diag_record_scrub(env, &caller->regs[BPF_REG_0], BPF_DIAG_MOD_CALLER_SAVED); 9984 else 9985 bpf_diag_mod_begin(env, &caller->regs[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE); 9986 clear_caller_saved_regs(env, caller->regs); 9987 invalidate_outgoing_stack_args(env, cur_func(env)); 9988 9989 /* All non-void global functions return a 64-bit SCALAR_VALUE. */ 9990 if (!returns_void) { 9991 mark_reg_unknown(env, caller->regs, BPF_REG_0); 9992 bpf_diag_mod_end(env); 9993 } 9994 9995 if (env->subprog_info[subprog].might_throw) { 9996 struct bpf_verifier_state *branch; 9997 9998 branch = push_stack(env, *insn_idx + 1, *insn_idx, false); 9999 if (IS_ERR(branch)) { 10000 verbose(env, "failed to push state for global subprog exception path\n"); 10001 return PTR_ERR(branch); 10002 } 10003 return process_bpf_exit_full(env, NULL, true); 10004 } 10005 10006 /* continue with next insn after call */ 10007 return 0; 10008 } 10009 10010 /* 10011 * Track caller's total stack arg count (incoming + max outgoing). 10012 * This is needed so the JIT knows how much stack arg space to allocate. 10013 */ 10014 caller_info = &env->subprog_info[caller->subprogno]; 10015 callee_incoming = bpf_in_stack_arg_cnt(&env->subprog_info[subprog]); 10016 stack_arg_cnt = bpf_in_stack_arg_cnt(caller_info) + callee_incoming; 10017 if (stack_arg_cnt > caller_info->stack_arg_cnt) 10018 caller_info->stack_arg_cnt = stack_arg_cnt; 10019 10020 /* for regular function entry setup new frame and continue 10021 * from that frame. 10022 */ 10023 err = setup_func_entry(env, subprog, *insn_idx, set_callee_state, state); 10024 if (err) 10025 return err; 10026 10027 bpf_diag_record_scrub(env, &caller->regs[BPF_REG_0], BPF_DIAG_MOD_CALLER_SAVED); 10028 clear_caller_saved_regs(env, caller->regs); 10029 10030 /* and go analyze first insn of the callee */ 10031 *insn_idx = env->subprog_info[subprog].start - 1; 10032 10033 if (env->log.level & BPF_LOG_LEVEL) { 10034 verbose(env, "caller:\n"); 10035 print_verifier_state(env, state, caller->frameno, true); 10036 verbose(env, "callee:\n"); 10037 print_verifier_state(env, state, state->curframe, true); 10038 } 10039 10040 return 0; 10041 } 10042 10043 int map_set_for_each_callback_args(struct bpf_verifier_env *env, 10044 struct bpf_func_state *caller, 10045 struct bpf_func_state *callee) 10046 { 10047 /* bpf_for_each_map_elem(struct bpf_map *map, void *callback_fn, 10048 * void *callback_ctx, u64 flags); 10049 * callback_fn(struct bpf_map *map, void *key, void *value, 10050 * void *callback_ctx); 10051 */ 10052 callee->regs[BPF_REG_1] = caller->regs[BPF_REG_1]; 10053 10054 callee->regs[BPF_REG_2].type = PTR_TO_MAP_KEY; 10055 __mark_reg_known_zero(&callee->regs[BPF_REG_2]); 10056 callee->regs[BPF_REG_2].map_ptr = caller->regs[BPF_REG_1].map_ptr; 10057 callee->regs[BPF_REG_2].map_uid = caller->regs[BPF_REG_1].map_uid; 10058 10059 callee->regs[BPF_REG_3].type = PTR_TO_MAP_VALUE; 10060 __mark_reg_known_zero(&callee->regs[BPF_REG_3]); 10061 callee->regs[BPF_REG_3].map_ptr = caller->regs[BPF_REG_1].map_ptr; 10062 callee->regs[BPF_REG_3].map_uid = caller->regs[BPF_REG_1].map_uid; 10063 callee->regs[BPF_REG_3].id = ++env->id_gen; 10064 10065 /* pointer to stack or null */ 10066 callee->regs[BPF_REG_4] = caller->regs[BPF_REG_3]; 10067 10068 /* unused */ 10069 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]); 10070 return 0; 10071 } 10072 10073 static int set_callee_state(struct bpf_verifier_env *env, 10074 struct bpf_func_state *caller, 10075 struct bpf_func_state *callee, int insn_idx) 10076 { 10077 int i; 10078 10079 /* copy r1 - r5 args that callee can access. The copy includes parent 10080 * pointers, which connects us up to the liveness chain 10081 */ 10082 for (i = BPF_REG_1; i <= BPF_REG_5; i++) 10083 callee->regs[i] = caller->regs[i]; 10084 return 0; 10085 } 10086 10087 static int set_map_elem_callback_state(struct bpf_verifier_env *env, 10088 struct bpf_func_state *caller, 10089 struct bpf_func_state *callee, 10090 int insn_idx) 10091 { 10092 struct bpf_insn_aux_data *insn_aux = &env->insn_aux_data[insn_idx]; 10093 struct bpf_map *map; 10094 int err; 10095 10096 /* valid map_ptr and poison value does not matter */ 10097 map = insn_aux->map_ptr_state.map_ptr; 10098 if (!map->ops->map_set_for_each_callback_args || 10099 !map->ops->map_for_each_callback) { 10100 verbose(env, "callback function not allowed for map\n"); 10101 return -ENOTSUPP; 10102 } 10103 10104 err = map->ops->map_set_for_each_callback_args(env, caller, callee); 10105 if (err) 10106 return err; 10107 10108 callee->in_callback_fn = true; 10109 callee->callback_ret_range = retval_range(0, 1); 10110 return 0; 10111 } 10112 10113 static int set_loop_callback_state(struct bpf_verifier_env *env, 10114 struct bpf_func_state *caller, 10115 struct bpf_func_state *callee, 10116 int insn_idx) 10117 { 10118 /* bpf_loop(u32 nr_loops, void *callback_fn, void *callback_ctx, 10119 * u64 flags); 10120 * callback_fn(u64 index, void *callback_ctx); 10121 */ 10122 callee->regs[BPF_REG_1].type = SCALAR_VALUE; 10123 callee->regs[BPF_REG_2] = caller->regs[BPF_REG_3]; 10124 10125 /* unused */ 10126 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_3]); 10127 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]); 10128 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]); 10129 10130 callee->in_callback_fn = true; 10131 callee->callback_ret_range = retval_range(0, 1); 10132 return 0; 10133 } 10134 10135 static int set_timer_callback_state(struct bpf_verifier_env *env, 10136 struct bpf_func_state *caller, 10137 struct bpf_func_state *callee, 10138 int insn_idx) 10139 { 10140 struct bpf_map *map_ptr = caller->regs[BPF_REG_1].map_ptr; 10141 u32 map_uid = caller->regs[BPF_REG_1].map_uid; 10142 10143 /* bpf_timer_set_callback(struct bpf_timer *timer, void *callback_fn); 10144 * callback_fn(struct bpf_map *map, void *key, void *value); 10145 */ 10146 callee->regs[BPF_REG_1].type = CONST_PTR_TO_MAP; 10147 __mark_reg_known_zero(&callee->regs[BPF_REG_1]); 10148 callee->regs[BPF_REG_1].map_ptr = map_ptr; 10149 callee->regs[BPF_REG_1].map_uid = map_uid; 10150 10151 callee->regs[BPF_REG_2].type = PTR_TO_MAP_KEY; 10152 __mark_reg_known_zero(&callee->regs[BPF_REG_2]); 10153 callee->regs[BPF_REG_2].map_ptr = map_ptr; 10154 callee->regs[BPF_REG_2].map_uid = map_uid; 10155 10156 callee->regs[BPF_REG_3].type = PTR_TO_MAP_VALUE; 10157 __mark_reg_known_zero(&callee->regs[BPF_REG_3]); 10158 callee->regs[BPF_REG_3].map_ptr = map_ptr; 10159 callee->regs[BPF_REG_3].map_uid = map_uid; 10160 callee->regs[BPF_REG_3].id = ++env->id_gen; 10161 10162 /* unused */ 10163 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]); 10164 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]); 10165 callee->in_async_callback_fn = true; 10166 callee->callback_ret_range = retval_range(0, 0); 10167 return 0; 10168 } 10169 10170 static int set_find_vma_callback_state(struct bpf_verifier_env *env, 10171 struct bpf_func_state *caller, 10172 struct bpf_func_state *callee, 10173 int insn_idx) 10174 { 10175 /* bpf_find_vma(struct task_struct *task, u64 addr, 10176 * void *callback_fn, void *callback_ctx, u64 flags) 10177 * (callback_fn)(struct task_struct *task, 10178 * struct vm_area_struct *vma, void *callback_ctx); 10179 */ 10180 callee->regs[BPF_REG_1] = caller->regs[BPF_REG_1]; 10181 10182 callee->regs[BPF_REG_2].type = PTR_TO_BTF_ID; 10183 __mark_reg_known_zero(&callee->regs[BPF_REG_2]); 10184 callee->regs[BPF_REG_2].btf = btf_vmlinux; 10185 callee->regs[BPF_REG_2].btf_id = btf_tracing_ids[BTF_TRACING_TYPE_VMA]; 10186 10187 /* pointer to stack or null */ 10188 callee->regs[BPF_REG_3] = caller->regs[BPF_REG_4]; 10189 10190 /* unused */ 10191 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]); 10192 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]); 10193 callee->in_callback_fn = true; 10194 callee->callback_ret_range = retval_range(0, 1); 10195 return 0; 10196 } 10197 10198 static int set_user_ringbuf_callback_state(struct bpf_verifier_env *env, 10199 struct bpf_func_state *caller, 10200 struct bpf_func_state *callee, 10201 int insn_idx) 10202 { 10203 /* bpf_user_ringbuf_drain(struct bpf_map *map, void *callback_fn, void 10204 * callback_ctx, u64 flags); 10205 * callback_fn(const struct bpf_dynptr_t* dynptr, void *callback_ctx); 10206 */ 10207 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_0]); 10208 mark_dynptr_cb_reg(env, &callee->regs[BPF_REG_1], BPF_DYNPTR_TYPE_LOCAL); 10209 callee->regs[BPF_REG_2] = caller->regs[BPF_REG_3]; 10210 10211 /* unused */ 10212 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_3]); 10213 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]); 10214 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]); 10215 10216 callee->in_callback_fn = true; 10217 callee->callback_ret_range = retval_range(0, 1); 10218 return 0; 10219 } 10220 10221 static int set_rbtree_add_callback_state(struct bpf_verifier_env *env, 10222 struct bpf_func_state *caller, 10223 struct bpf_func_state *callee, 10224 int insn_idx) 10225 { 10226 /* void bpf_rbtree_add_impl(struct bpf_rb_root *root, struct bpf_rb_node *node, 10227 * bool (less)(struct bpf_rb_node *a, const struct bpf_rb_node *b)); 10228 * 10229 * 'struct bpf_rb_node *node' arg to bpf_rbtree_add_impl is the same PTR_TO_BTF_ID w/ offset 10230 * that 'less' callback args will be receiving. However, 'node' arg was release_reference'd 10231 * by this point, so look at 'root' 10232 */ 10233 struct btf_field *field; 10234 10235 field = reg_find_field_offset(&caller->regs[BPF_REG_1], 10236 caller->regs[BPF_REG_1].var_off.value, 10237 BPF_RB_ROOT); 10238 if (!field || !field->graph_root.value_btf_id) 10239 return -EFAULT; 10240 10241 mark_reg_graph_node(callee->regs, BPF_REG_1, &field->graph_root); 10242 ref_set_non_owning(env, &callee->regs[BPF_REG_1]); 10243 mark_reg_graph_node(callee->regs, BPF_REG_2, &field->graph_root); 10244 ref_set_non_owning(env, &callee->regs[BPF_REG_2]); 10245 10246 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_3]); 10247 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]); 10248 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]); 10249 callee->in_callback_fn = true; 10250 callee->callback_ret_range = retval_range(0, 1); 10251 return 0; 10252 } 10253 10254 static int set_task_work_schedule_callback_state(struct bpf_verifier_env *env, 10255 struct bpf_func_state *caller, 10256 struct bpf_func_state *callee, 10257 int insn_idx) 10258 { 10259 struct bpf_map *map_ptr = caller->regs[BPF_REG_3].map_ptr; 10260 u32 map_uid = caller->regs[BPF_REG_3].map_uid; 10261 10262 /* 10263 * callback_fn(struct bpf_map *map, void *key, void *value); 10264 */ 10265 callee->regs[BPF_REG_1].type = CONST_PTR_TO_MAP; 10266 __mark_reg_known_zero(&callee->regs[BPF_REG_1]); 10267 callee->regs[BPF_REG_1].map_ptr = map_ptr; 10268 callee->regs[BPF_REG_1].map_uid = map_uid; 10269 10270 callee->regs[BPF_REG_2].type = PTR_TO_MAP_KEY; 10271 __mark_reg_known_zero(&callee->regs[BPF_REG_2]); 10272 callee->regs[BPF_REG_2].map_ptr = map_ptr; 10273 callee->regs[BPF_REG_2].map_uid = map_uid; 10274 10275 callee->regs[BPF_REG_3].type = PTR_TO_MAP_VALUE; 10276 __mark_reg_known_zero(&callee->regs[BPF_REG_3]); 10277 callee->regs[BPF_REG_3].map_ptr = map_ptr; 10278 callee->regs[BPF_REG_3].map_uid = map_uid; 10279 callee->regs[BPF_REG_3].id = ++env->id_gen; 10280 10281 /* unused */ 10282 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]); 10283 bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]); 10284 callee->in_async_callback_fn = true; 10285 callee->callback_ret_range = retval_range(S32_MIN, S32_MAX); 10286 return 0; 10287 } 10288 10289 static bool is_rbtree_lock_required_kfunc(u32 btf_id); 10290 10291 static void account_processed_insn(struct bpf_verifier_env *env) 10292 { 10293 struct bpf_func_state *frame = cur_func(env); 10294 10295 env->insn_processed++; 10296 frame->insns_subtotal++; 10297 env->subprog_info[frame->subprogno].insns_self++; 10298 } 10299 10300 static void account_processed_insns(struct bpf_verifier_env *env, 10301 struct bpf_func_state *callee, 10302 struct bpf_func_state *caller) 10303 { 10304 u32 insns; 10305 10306 if (!callee) 10307 return; 10308 10309 insns = callee->insns_subtotal; 10310 10311 env->subprog_info[callee->subprogno].insns_total += insns; 10312 if (caller) 10313 caller->insns_subtotal += insns; 10314 callee->insns_subtotal = 0; 10315 } 10316 10317 static void account_current_path(struct bpf_verifier_env *env) 10318 { 10319 struct bpf_verifier_state *state = env->cur_state; 10320 int frame; 10321 10322 for (frame = state->curframe; frame >= 0; frame--) 10323 account_processed_insns(env, state->frame[frame], 10324 frame ? state->frame[frame - 1] : NULL); 10325 } 10326 10327 /* 10328 * Are we currently verifying the callback for an rbtree kfunc that must 10329 * be called with a lock held, or one of that callback's subprogs? If so, 10330 * no need to complain about an unreleased lock. 10331 */ 10332 static bool in_rbtree_lock_required_cb(struct bpf_verifier_env *env) 10333 { 10334 struct bpf_verifier_state *state = env->cur_state; 10335 struct bpf_insn *insn = env->prog->insnsi; 10336 struct bpf_func_state *callee; 10337 int kfunc_btf_id; 10338 u32 frame; 10339 10340 for (frame = state->curframe; frame; frame--) { 10341 callee = state->frame[frame]; 10342 if (!callee->in_callback_fn) 10343 continue; 10344 10345 kfunc_btf_id = insn[callee->callsite].imm; 10346 if (is_rbtree_lock_required_kfunc(kfunc_btf_id)) 10347 return true; 10348 } 10349 10350 return false; 10351 } 10352 10353 static bool retval_range_within(struct bpf_retval_range range, const struct bpf_reg_state *reg) 10354 { 10355 if (range.return_32bit) 10356 return range.minval <= reg_s32_min(reg) && reg_s32_max(reg) <= range.maxval; 10357 else 10358 return range.minval <= reg_smin(reg) && reg_smax(reg) <= range.maxval; 10359 } 10360 10361 static int prepare_func_exit(struct bpf_verifier_env *env, int *insn_idx) 10362 { 10363 struct bpf_verifier_state *state = env->cur_state, *prev_st; 10364 struct bpf_func_state *caller, *callee; 10365 struct bpf_reg_state *r0; 10366 bool in_callback_fn; 10367 int err; 10368 10369 callee = state->frame[state->curframe]; 10370 r0 = &callee->regs[BPF_REG_0]; 10371 if (r0->type == PTR_TO_STACK) { 10372 /* technically it's ok to return caller's stack pointer 10373 * (or caller's caller's pointer) back to the caller, 10374 * since these pointers are valid. Only current stack 10375 * pointer will be invalid as soon as function exits, 10376 * but let's be conservative 10377 */ 10378 verbose(env, "cannot return stack pointer to the caller\n"); 10379 return -EINVAL; 10380 } 10381 10382 caller = state->frame[state->curframe - 1]; 10383 if (callee->in_callback_fn) { 10384 if (r0->type != SCALAR_VALUE) { 10385 verbose(env, "R0 not a scalar value\n"); 10386 return -EACCES; 10387 } 10388 10389 /* we are going to rely on register's precise value */ 10390 err = mark_chain_precision(env, BPF_REG_0); 10391 if (err) 10392 return err; 10393 10394 /* enforce R0 return value range, and bpf_callback_t returns 64bit */ 10395 if (!retval_range_within(callee->callback_ret_range, r0)) { 10396 verbose_invalid_scalar(env, r0, callee->callback_ret_range, 10397 "At callback return", "R0"); 10398 return -EINVAL; 10399 } 10400 if (!bpf_calls_callback(env, callee->callsite)) { 10401 verifier_bug(env, "in callback at %d, callsite %d !calls_callback", 10402 *insn_idx, callee->callsite); 10403 return -EFAULT; 10404 } 10405 } else { 10406 /* return to the caller whatever r0 had in the callee */ 10407 bpf_diag_mod_begin(env, &caller->regs[BPF_REG_0], r0, BPF_DIAG_MOD_WRITE); 10408 caller->regs[BPF_REG_0] = *r0; 10409 bpf_diag_mod_end(env); 10410 } 10411 10412 /* for callbacks like bpf_loop or bpf_for_each_map_elem go back to callsite, 10413 * there function call logic would reschedule callback visit. If iteration 10414 * converges is_state_visited() would prune that visit eventually. 10415 */ 10416 in_callback_fn = callee->in_callback_fn; 10417 if (in_callback_fn) 10418 *insn_idx = callee->callsite; 10419 else 10420 *insn_idx = callee->callsite + 1; 10421 10422 if (env->log.level & BPF_LOG_LEVEL) { 10423 verbose(env, "returning from callee:\n"); 10424 print_verifier_state(env, state, callee->frameno, true); 10425 verbose(env, "to caller at %d:\n", *insn_idx); 10426 print_verifier_state(env, state, caller->frameno, true); 10427 } 10428 account_processed_insns(env, callee, caller); 10429 /* clear everything in the callee. In case of exceptional exits using 10430 * bpf_throw, this will be done by copy_verifier_state for extra frames. */ 10431 free_func_state(callee); 10432 state->frame[state->curframe--] = NULL; 10433 invalidate_outgoing_stack_args(env, caller); 10434 10435 /* for callbacks widen imprecise scalars to make programs like below verify: 10436 * 10437 * struct ctx { int i; } 10438 * void cb(int idx, struct ctx *ctx) { ctx->i++; ... } 10439 * ... 10440 * struct ctx = { .i = 0; } 10441 * bpf_loop(100, cb, &ctx, 0); 10442 * 10443 * This is similar to what is done in process_iter_next_call() for open 10444 * coded iterators. 10445 */ 10446 prev_st = in_callback_fn ? find_prev_entry(env, state, *insn_idx) : NULL; 10447 if (prev_st) { 10448 err = widen_imprecise_scalars(env, prev_st, state); 10449 if (err) 10450 return err; 10451 } 10452 return 0; 10453 } 10454 10455 static int do_refine_retval_range(struct bpf_verifier_env *env, 10456 struct bpf_reg_state *regs, int ret_type, 10457 int func_id, 10458 struct bpf_call_arg_meta *meta) 10459 { 10460 struct bpf_retval_range range; 10461 struct bpf_reg_state *ret_reg = ®s[BPF_REG_0]; 10462 enum bpf_prog_type prog_type = resolve_prog_type(env->prog); 10463 10464 if (ret_type != RET_INTEGER) 10465 return 0; 10466 10467 switch (func_id) { 10468 case BPF_FUNC_get_stack: 10469 case BPF_FUNC_get_task_stack: 10470 case BPF_FUNC_probe_read_str: 10471 case BPF_FUNC_probe_read_kernel_str: 10472 case BPF_FUNC_probe_read_user_str: 10473 reg_set_srange64(ret_reg, -MAX_ERRNO, meta->msize_max_value); 10474 reg_set_srange32(ret_reg, -MAX_ERRNO, meta->msize_max_value); 10475 reg_bounds_sync(ret_reg); 10476 break; 10477 case BPF_FUNC_get_smp_processor_id: 10478 reg_set_urange64(ret_reg, 0, nr_cpu_ids - 1); 10479 reg_set_urange32(ret_reg, 0, nr_cpu_ids - 1); 10480 reg_bounds_sync(ret_reg); 10481 break; 10482 case BPF_FUNC_get_retval: 10483 /* 10484 * bpf_get_retval may see arbitrary value passed by bpf_prog_run_array_cg for 10485 * CGROUP_GETSOCKOPT type. 10486 */ 10487 if (prog_type == BPF_PROG_TYPE_CGROUP_SOCKOPT && 10488 env->prog->expected_attach_type == BPF_CGROUP_GETSOCKOPT) 10489 break; 10490 10491 if (prog_type == BPF_PROG_TYPE_LSM && 10492 env->prog->expected_attach_type == BPF_LSM_CGROUP) { 10493 if (!env->prog->aux->attach_func_proto->type) 10494 break; 10495 bpf_lsm_get_retval_range(env->prog, &range); 10496 } else { 10497 range.minval = -MAX_ERRNO; 10498 range.maxval = 0; 10499 } 10500 10501 reg_set_srange64(ret_reg, range.minval, range.maxval); 10502 reg_set_srange32(ret_reg, range.minval, range.maxval); 10503 reg_bounds_sync(ret_reg); 10504 break; 10505 } 10506 10507 return reg_bounds_sanity_check(env, ret_reg, "retval"); 10508 } 10509 10510 static int 10511 record_func_map(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta, 10512 int func_id, int insn_idx) 10513 { 10514 struct bpf_insn_aux_data *aux = &env->insn_aux_data[insn_idx]; 10515 struct bpf_map *map = meta->map.ptr; 10516 10517 if (func_id != BPF_FUNC_tail_call && 10518 func_id != BPF_FUNC_map_lookup_elem && 10519 func_id != BPF_FUNC_map_update_elem && 10520 func_id != BPF_FUNC_map_delete_elem && 10521 func_id != BPF_FUNC_map_push_elem && 10522 func_id != BPF_FUNC_map_pop_elem && 10523 func_id != BPF_FUNC_map_peek_elem && 10524 func_id != BPF_FUNC_for_each_map_elem && 10525 func_id != BPF_FUNC_redirect_map && 10526 func_id != BPF_FUNC_map_lookup_percpu_elem) 10527 return 0; 10528 10529 if (map == NULL) { 10530 verifier_bug(env, "expected map for helper call"); 10531 return -EFAULT; 10532 } 10533 10534 /* In case of read-only, some additional restrictions 10535 * need to be applied in order to prevent altering the 10536 * state of the map from program side. 10537 */ 10538 if ((map->map_flags & BPF_F_RDONLY_PROG) && 10539 (func_id == BPF_FUNC_map_delete_elem || 10540 func_id == BPF_FUNC_map_update_elem || 10541 func_id == BPF_FUNC_map_push_elem || 10542 func_id == BPF_FUNC_map_pop_elem)) { 10543 verbose(env, "write into map forbidden\n"); 10544 return -EACCES; 10545 } 10546 10547 if (!aux->map_ptr_state.map_ptr) 10548 bpf_map_ptr_store(aux, meta->map.ptr, 10549 !meta->map.ptr->bypass_spec_v1, false); 10550 else if (aux->map_ptr_state.map_ptr != meta->map.ptr) 10551 bpf_map_ptr_store(aux, meta->map.ptr, 10552 !meta->map.ptr->bypass_spec_v1, true); 10553 return 0; 10554 } 10555 10556 static int 10557 record_func_key(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta, 10558 int func_id, int insn_idx) 10559 { 10560 struct bpf_insn_aux_data *aux = &env->insn_aux_data[insn_idx]; 10561 struct bpf_reg_state *reg; 10562 struct bpf_map *map = meta->map.ptr; 10563 u64 val, max; 10564 int err; 10565 10566 if (func_id != BPF_FUNC_tail_call) 10567 return 0; 10568 if (!map || map->map_type != BPF_MAP_TYPE_PROG_ARRAY) { 10569 verbose(env, "expected prog array map for tail call"); 10570 return -EINVAL; 10571 } 10572 10573 reg = reg_state(env, BPF_REG_3); 10574 val = reg->var_off.value; 10575 max = map->max_entries; 10576 10577 if (!(is_reg_const(reg, false) && val < max)) { 10578 bpf_map_key_store(aux, BPF_MAP_KEY_POISON); 10579 return 0; 10580 } 10581 10582 err = mark_chain_precision(env, BPF_REG_3); 10583 if (err) 10584 return err; 10585 if (bpf_map_key_unseen(aux)) 10586 bpf_map_key_store(aux, val); 10587 else if (!bpf_map_key_poisoned(aux) && 10588 bpf_map_key_immediate(aux) != val) 10589 bpf_map_key_store(aux, BPF_MAP_KEY_POISON); 10590 return 0; 10591 } 10592 10593 static int check_reference_leak(struct bpf_verifier_env *env, bool exception_exit) 10594 { 10595 struct bpf_verifier_state *state = env->cur_state; 10596 enum bpf_prog_type type = resolve_prog_type(env->prog); 10597 struct bpf_reg_state *reg = reg_state(env, BPF_REG_0); 10598 bool refs_lingering = false; 10599 int i; 10600 10601 if (!exception_exit && cur_func(env)->frameno) 10602 return 0; 10603 10604 for (i = 0; i < state->acquired_refs; i++) { 10605 if (state->refs[i].type != REF_TYPE_PTR) 10606 continue; 10607 /* Allow struct_ops programs to return a referenced kptr back to 10608 * kernel. Type checks are performed later in check_return_code. 10609 */ 10610 if (type == BPF_PROG_TYPE_STRUCT_OPS && !exception_exit && 10611 reg->id == state->refs[i].id) 10612 continue; 10613 verbose(env, "Unreleased reference id=%d alloc_insn=%d\n", 10614 state->refs[i].id, state->refs[i].insn_idx); 10615 bpf_diag_leak(env, state->refs[i].id, state->refs[i].insn_idx, env->insn_idx); 10616 refs_lingering = true; 10617 } 10618 return refs_lingering ? -EINVAL : 0; 10619 } 10620 10621 static int check_resource_leak(struct bpf_verifier_env *env, bool exception_exit, bool check_lock, const char *prefix) 10622 { 10623 int err; 10624 10625 if (check_lock && env->cur_state->active_locks) { 10626 verbose(env, "%s cannot be used inside bpf_spin_lock-ed region\n", prefix); 10627 bpf_diag_ctx_active(env, env->insn_idx, prefix, BPF_DIAG_CONTEXT_LOCK, 10628 "Release the BPF spin lock before this operation on every path."); 10629 return -EINVAL; 10630 } 10631 10632 err = check_reference_leak(env, exception_exit); 10633 if (err) { 10634 verbose(env, "%s would lead to reference leak\n", prefix); 10635 return err; 10636 } 10637 10638 if (check_lock && env->cur_state->active_irq_id) { 10639 verbose(env, "%s cannot be used inside bpf_local_irq_save-ed region\n", prefix); 10640 bpf_diag_ctx_active(env, env->insn_idx, prefix, BPF_DIAG_CONTEXT_IRQ, 10641 "Restore the saved IRQ state before this operation on every path."); 10642 return -EINVAL; 10643 } 10644 10645 if (check_lock && env->cur_state->active_rcu_locks) { 10646 verbose(env, "%s cannot be used inside bpf_rcu_read_lock-ed region\n", prefix); 10647 bpf_diag_ctx_active(env, env->insn_idx, prefix, BPF_DIAG_CONTEXT_RCU, 10648 "Call bpf_rcu_read_unlock() before this operation on every path."); 10649 return -EINVAL; 10650 } 10651 10652 if (check_lock && env->cur_state->active_preempt_locks) { 10653 verbose(env, "%s cannot be used inside bpf_preempt_disable-ed region\n", prefix); 10654 bpf_diag_ctx_active( 10655 env, env->insn_idx, prefix, BPF_DIAG_CONTEXT_PREEMPT, 10656 "Call bpf_preempt_enable() before this operation on every path."); 10657 return -EINVAL; 10658 } 10659 10660 return 0; 10661 } 10662 10663 static int check_bpf_snprintf_call(struct bpf_verifier_env *env, 10664 struct bpf_reg_state *regs) 10665 { 10666 struct bpf_reg_state *fmt_reg = ®s[BPF_REG_3]; 10667 struct bpf_reg_state *data_len_reg = ®s[BPF_REG_5]; 10668 struct bpf_map *fmt_map = fmt_reg->map_ptr; 10669 struct bpf_bprintf_data data = {}; 10670 int err, fmt_map_off, num_args; 10671 u64 fmt_addr; 10672 char *fmt; 10673 10674 /* data must be an array of u64 */ 10675 if (data_len_reg->var_off.value % 8) 10676 return -EINVAL; 10677 num_args = data_len_reg->var_off.value / 8; 10678 10679 /* fmt being ARG_PTR_TO_CONST_STR guarantees that var_off is const 10680 * and map_direct_value_addr is set. 10681 */ 10682 fmt_map_off = fmt_reg->var_off.value; 10683 err = fmt_map->ops->map_direct_value_addr(fmt_map, &fmt_addr, 10684 fmt_map_off); 10685 if (err) { 10686 verbose(env, "failed to retrieve map value address\n"); 10687 return -EFAULT; 10688 } 10689 fmt = (char *)(long)fmt_addr + fmt_map_off; 10690 10691 /* We are also guaranteed that fmt+fmt_map_off is NULL terminated, we 10692 * can focus on validating the format specifiers. 10693 */ 10694 err = bpf_bprintf_prepare(fmt, UINT_MAX, NULL, num_args, &data); 10695 if (err < 0) 10696 verbose(env, "Invalid format string\n"); 10697 10698 return err; 10699 } 10700 10701 static int check_get_func_ip(struct bpf_verifier_env *env) 10702 { 10703 enum bpf_prog_type type = resolve_prog_type(env->prog); 10704 int func_id = BPF_FUNC_get_func_ip; 10705 10706 if (type == BPF_PROG_TYPE_TRACING) { 10707 if (!bpf_prog_has_trampoline(env->prog)) { 10708 verbose(env, "func %s#%d supported only for fentry/fexit/fsession/fmod_ret programs\n", 10709 func_id_name(func_id), func_id); 10710 return -ENOTSUPP; 10711 } 10712 return 0; 10713 } else if (type == BPF_PROG_TYPE_KPROBE) { 10714 return 0; 10715 } 10716 10717 verbose(env, "func %s#%d not supported for program type %d\n", 10718 func_id_name(func_id), func_id, type); 10719 return -ENOTSUPP; 10720 } 10721 10722 static struct bpf_insn_aux_data *cur_aux(const struct bpf_verifier_env *env) 10723 { 10724 return &env->insn_aux_data[env->insn_idx]; 10725 } 10726 10727 /* Returns 1 if R4 is a known zero, 0 if it is not, a negative errno on error. */ 10728 static int loop_flag_is_zero(struct bpf_verifier_env *env) 10729 { 10730 struct bpf_reg_state *reg = reg_state(env, BPF_REG_4); 10731 int err; 10732 10733 if (!bpf_register_is_null(reg)) 10734 return 0; 10735 10736 err = mark_chain_precision(env, BPF_REG_4); 10737 if (err) 10738 return err; 10739 return 1; 10740 } 10741 10742 static int update_loop_inline_state(struct bpf_verifier_env *env, u32 subprogno) 10743 { 10744 struct bpf_loop_inline_state *state = &cur_aux(env)->loop_inline_state; 10745 int flag_is_zero; 10746 10747 if (!state->initialized) { 10748 flag_is_zero = loop_flag_is_zero(env); 10749 if (flag_is_zero < 0) 10750 return flag_is_zero; 10751 state->initialized = 1; 10752 state->fit_for_inline = flag_is_zero; 10753 state->callback_subprogno = subprogno; 10754 return 0; 10755 } 10756 10757 if (!state->fit_for_inline) 10758 return 0; 10759 10760 flag_is_zero = loop_flag_is_zero(env); 10761 if (flag_is_zero < 0) 10762 return flag_is_zero; 10763 state->fit_for_inline = (flag_is_zero && 10764 state->callback_subprogno == subprogno); 10765 return 0; 10766 } 10767 10768 /* Returns whether or not the given map can potentially elide 10769 * lookup return value nullness check. This is possible if the key 10770 * is statically known. 10771 */ 10772 static bool can_elide_value_nullness(const struct bpf_map *map) 10773 { 10774 if (map->map_flags & BPF_F_INNER_MAP) 10775 return false; 10776 10777 switch (map->map_type) { 10778 case BPF_MAP_TYPE_ARRAY: 10779 case BPF_MAP_TYPE_PERCPU_ARRAY: 10780 return true; 10781 default: 10782 return false; 10783 } 10784 } 10785 10786 int bpf_get_helper_proto(struct bpf_verifier_env *env, int func_id, 10787 const struct bpf_func_proto **ptr) 10788 { 10789 if (func_id < 0 || func_id >= __BPF_FUNC_MAX_ID) 10790 return -ERANGE; 10791 10792 if (!env->ops->get_func_proto) 10793 return -EINVAL; 10794 10795 *ptr = env->ops->get_func_proto(func_id, env->prog); 10796 return *ptr && (*ptr)->func ? 0 : -EINVAL; 10797 } 10798 10799 /* Check if we're in a sleepable context. */ 10800 static inline bool in_sleepable_context(struct bpf_verifier_env *env) 10801 { 10802 return !in_rcu_cs(env); 10803 } 10804 10805 static const char *non_sleepable_context_description(struct bpf_verifier_env *env) 10806 { 10807 if (env->cur_state->active_rcu_locks) 10808 return "rcu_read_lock region"; 10809 if (env->cur_state->active_preempt_locks) 10810 return "non-preemptible region"; 10811 if (env->cur_state->active_irq_id) 10812 return "IRQ-disabled region"; 10813 if (env->cur_state->active_locks) 10814 return "lock region"; 10815 return "non-sleepable prog"; 10816 } 10817 10818 static int release_reg(struct bpf_verifier_env *env, struct bpf_reg_state *reg, 10819 bool convert_rcu, bool release_dynptr) 10820 { 10821 int err = -EINVAL; 10822 10823 if (bpf_register_is_null(reg)) 10824 return 0; 10825 10826 if (release_dynptr) 10827 err = unmark_stack_slots_dynptr(env, reg); 10828 else if (convert_rcu) 10829 err = ref_convert_alloc_rcu_protected(env, reg->id); 10830 else if (reg_is_referenced(env, reg)) 10831 err = release_reference(env, reg->id); 10832 10833 return err; 10834 } 10835 10836 static int check_helper_call(struct bpf_verifier_env *env, struct bpf_insn *insn, 10837 int *insn_idx_p) 10838 { 10839 enum bpf_prog_type prog_type = resolve_prog_type(env->prog); 10840 bool returns_cpu_specific_alloc_ptr = false; 10841 const struct bpf_func_proto *fn = NULL; 10842 enum bpf_return_type ret_type; 10843 enum bpf_type_flag ret_flag; 10844 struct bpf_reg_state *regs; 10845 struct bpf_call_arg_meta meta; 10846 const char *operation; 10847 int insn_idx = *insn_idx_p; 10848 bool changes_data; 10849 int i, err, func_id; 10850 10851 /* find function prototype */ 10852 func_id = insn->imm; 10853 err = bpf_get_helper_proto(env, insn->imm, &fn); 10854 if (err == -ERANGE) { 10855 verbose(env, "invalid func %s#%d\n", func_id_name(func_id), func_id); 10856 return -EINVAL; 10857 } 10858 10859 if (err) { 10860 verbose(env, "program of this type cannot use helper %s#%d\n", 10861 func_id_name(func_id), func_id); 10862 operation = bpf_diag_fmt(env, "helper %s#%d", func_id_name(func_id), func_id); 10863 bpf_diag_policy( 10864 env, insn_idx, operation, "this program type does not allow the helper", 10865 "Use a helper allowed for this program type, or move the logic to a compatible program type."); 10866 return err; 10867 } 10868 10869 /* eBPF programs must be GPL compatible to use GPL-ed functions */ 10870 if (!env->prog->gpl_compatible && fn->gpl_only) { 10871 verbose(env, "cannot call GPL-restricted function from non-GPL compatible program\n"); 10872 operation = bpf_diag_fmt(env, "helper %s#%d", func_id_name(func_id), func_id); 10873 bpf_diag_policy( 10874 env, insn_idx, operation, 10875 "this helper is restricted to GPL-compatible programs", 10876 "Use a GPL-compatible license, or replace the helper with one that is available to non-GPL programs."); 10877 return -EINVAL; 10878 } 10879 10880 if (fn->allowed && !fn->allowed(env->prog)) { 10881 verbose(env, "helper call is not allowed in probe\n"); 10882 operation = bpf_diag_fmt(env, "helper %s#%d", func_id_name(func_id), func_id); 10883 bpf_diag_policy( 10884 env, insn_idx, operation, 10885 "the helper-specific policy callback rejected this program", 10886 "Use the helper only from an allowed attach point or program configuration."); 10887 return -EINVAL; 10888 } 10889 10890 /* With LD_ABS/IND some JITs save/restore skb from r1. */ 10891 changes_data = bpf_helper_changes_pkt_data(func_id); 10892 if (changes_data && fn->arg1_type != ARG_PTR_TO_CTX) { 10893 verifier_bug(env, "func %s#%d: r1 != ctx", func_id_name(func_id), func_id); 10894 return -EFAULT; 10895 } 10896 10897 memset(&meta, 0, sizeof(meta)); 10898 10899 err = check_func_proto(fn, &meta); 10900 if (err) { 10901 verifier_bug(env, "incorrect func proto %s#%d", func_id_name(func_id), func_id); 10902 return err; 10903 } 10904 10905 if (fn->might_sleep && !in_sleepable_context(env)) { 10906 verbose(env, "sleepable helper %s#%d in %s\n", func_id_name(func_id), func_id, 10907 non_sleepable_context_description(env)); 10908 operation = bpf_diag_fmt(env, "sleepable helper %s#%d", 10909 func_id_name(func_id), func_id); 10910 bpf_diag_ctx_forbidden(env, insn_idx, operation, 10911 "Move the helper call outside the critical section, or use a non-sleepable helper."); 10912 return -EINVAL; 10913 } 10914 10915 /* Track non-sleepable context for helpers. */ 10916 if (!in_sleepable_context(env)) 10917 env->insn_aux_data[insn_idx].non_sleepable = true; 10918 10919 meta.func_id = func_id; 10920 meta.fn = fn; 10921 /* check args */ 10922 for (i = 0; i < MAX_BPF_FUNC_REG_ARGS; i++) { 10923 err = check_func_arg(env, i, &meta, insn_idx); 10924 if (err) 10925 return err; 10926 } 10927 10928 err = record_func_map(env, &meta, func_id, insn_idx); 10929 if (err) 10930 return err; 10931 10932 err = record_func_key(env, &meta, func_id, insn_idx); 10933 if (err) 10934 return err; 10935 10936 regs = cur_regs(env); 10937 10938 /* Mark slots with STACK_MISC in case of raw mode, stack offset 10939 * is inferred from register state. 10940 */ 10941 for (i = 0; i < meta.arg_raw_mem.size; i++) { 10942 err = check_mem_access(env, insn_idx, regs + meta.arg_raw_mem.regno, 10943 argno_from_reg(meta.arg_raw_mem.regno), i, BPF_B, 10944 BPF_WRITE, -1, false, false); 10945 if (err) 10946 return err; 10947 } 10948 10949 if (meta.release_regno) { 10950 struct bpf_reg_state *reg = ®s[meta.release_regno]; 10951 bool convert_rcu = (func_id == BPF_FUNC_kptr_xchg) && in_rcu_cs(env) && 10952 (reg->type & MEM_ALLOC) && (reg->type & MEM_PERCPU); 10953 10954 err = release_reg(env, reg, convert_rcu, !!meta.dynptr.id); 10955 if (err) 10956 return err; 10957 } 10958 10959 switch (func_id) { 10960 case BPF_FUNC_tail_call: 10961 err = check_resource_leak(env, false, true, "tail_call"); 10962 if (err) 10963 return err; 10964 break; 10965 case BPF_FUNC_get_local_storage: 10966 /* check that flags argument in get_local_storage(map, flags) is 0, 10967 * this is required because get_local_storage() can't return an error. 10968 */ 10969 if (!bpf_register_is_null(®s[BPF_REG_2])) { 10970 verbose(env, "get_local_storage() doesn't support non-zero flags\n"); 10971 return -EINVAL; 10972 } 10973 err = mark_chain_precision(env, BPF_REG_2); 10974 if (err) 10975 return err; 10976 break; 10977 case BPF_FUNC_for_each_map_elem: 10978 err = push_callback_call(env, insn, insn_idx, meta.subprogno, 10979 set_map_elem_callback_state); 10980 break; 10981 case BPF_FUNC_timer_set_callback: 10982 err = push_callback_call(env, insn, insn_idx, meta.subprogno, 10983 set_timer_callback_state); 10984 break; 10985 case BPF_FUNC_find_vma: 10986 err = push_callback_call(env, insn, insn_idx, meta.subprogno, 10987 set_find_vma_callback_state); 10988 break; 10989 case BPF_FUNC_snprintf: 10990 err = check_bpf_snprintf_call(env, regs); 10991 break; 10992 case BPF_FUNC_loop: 10993 err = update_loop_inline_state(env, meta.subprogno); 10994 if (err) 10995 return err; 10996 /* Verifier relies on R1 value to determine if bpf_loop() iteration 10997 * is finished, thus mark it precise. 10998 */ 10999 err = mark_chain_precision(env, BPF_REG_1); 11000 if (err) 11001 return err; 11002 if (cur_func(env)->callback_depth < reg_umax(®s[BPF_REG_1])) { 11003 err = push_callback_call(env, insn, insn_idx, meta.subprogno, 11004 set_loop_callback_state); 11005 } else { 11006 cur_func(env)->callback_depth = 0; 11007 if (env->log.level & BPF_LOG_LEVEL2) 11008 verbose(env, "frame%d bpf_loop iteration limit reached\n", 11009 env->cur_state->curframe); 11010 } 11011 break; 11012 case BPF_FUNC_dynptr_from_mem: 11013 if (regs[BPF_REG_1].type != PTR_TO_MAP_VALUE) { 11014 verbose(env, "Unsupported reg type %s for bpf_dynptr_from_mem data\n", 11015 reg_type_str(env, regs[BPF_REG_1].type)); 11016 return -EACCES; 11017 } 11018 break; 11019 case BPF_FUNC_set_retval: 11020 { 11021 struct bpf_retval_range range = { 11022 .minval = -MAX_ERRNO, 11023 .maxval = 0, 11024 .return_32bit = true 11025 }; 11026 struct bpf_reg_state *r1 = ®s[BPF_REG_1]; 11027 11028 if (r1->type != SCALAR_VALUE) { 11029 verbose(env, "R1 is not a scalar\n"); 11030 return -EINVAL; 11031 } 11032 11033 /* CGROUP_GETSOCKOPT is allowed to return arbitrary value */ 11034 if (prog_type == BPF_PROG_TYPE_CGROUP_SOCKOPT && 11035 env->prog->expected_attach_type == BPF_CGROUP_GETSOCKOPT) 11036 break; 11037 11038 if (prog_type == BPF_PROG_TYPE_LSM && 11039 env->prog->expected_attach_type == BPF_LSM_CGROUP) { 11040 if (!env->prog->aux->attach_func_proto->type) { 11041 /* Make sure programs that attach to void 11042 * hooks don't try to modify return value. 11043 */ 11044 verbose(env, "BPF_LSM_CGROUP that attach to void LSM hooks can't modify return value!\n"); 11045 return -EINVAL; 11046 } 11047 bpf_lsm_get_retval_range(env->prog, &range); 11048 } 11049 11050 err = mark_chain_precision(env, BPF_REG_1); 11051 if (err) 11052 return err; 11053 11054 if (!retval_range_within(range, r1)) { 11055 verbose_invalid_scalar(env, r1, range, "At bpf_set_retval", "R1"); 11056 return -EINVAL; 11057 } 11058 11059 break; 11060 } 11061 case BPF_FUNC_dynptr_write: 11062 { 11063 enum bpf_dynptr_type dynptr_type = meta.dynptr.type; 11064 11065 if (dynptr_type == BPF_DYNPTR_TYPE_INVALID) 11066 return -EFAULT; 11067 11068 if (dynptr_type == BPF_DYNPTR_TYPE_SKB || 11069 dynptr_type == BPF_DYNPTR_TYPE_SKB_META) 11070 /* this will trigger clear_all_pkt_pointers(), which will 11071 * invalidate all dynptr slices associated with the skb 11072 */ 11073 changes_data = true; 11074 11075 break; 11076 } 11077 case BPF_FUNC_per_cpu_ptr: 11078 case BPF_FUNC_this_cpu_ptr: 11079 { 11080 struct bpf_reg_state *reg = ®s[BPF_REG_1]; 11081 const struct btf_type *type; 11082 11083 if (reg->type & MEM_RCU) { 11084 type = btf_type_by_id(reg->btf, reg->btf_id); 11085 if (!type || !btf_type_is_struct(type)) { 11086 verbose(env, "Helper has invalid btf/btf_id in R1\n"); 11087 return -EFAULT; 11088 } 11089 returns_cpu_specific_alloc_ptr = true; 11090 env->insn_aux_data[insn_idx].call_with_percpu_alloc_ptr = true; 11091 } 11092 break; 11093 } 11094 case BPF_FUNC_user_ringbuf_drain: 11095 err = push_callback_call(env, insn, insn_idx, meta.subprogno, 11096 set_user_ringbuf_callback_state); 11097 break; 11098 } 11099 11100 if (err) 11101 return err; 11102 11103 /* reset caller saved regs */ 11104 bpf_diag_record_caller_saved(env, regs); 11105 bpf_diag_mod_begin(env, ®s[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE); 11106 for (i = 0; i < CALLER_SAVED_REGS; i++) { 11107 bpf_mark_reg_not_init(env, ®s[caller_saved[i]]); 11108 check_reg_arg(env, caller_saved[i], DST_OP_NO_MARK); 11109 } 11110 invalidate_outgoing_stack_args(env, cur_func(env)); 11111 11112 /* update return register (already marked as written above) */ 11113 ret_type = fn->ret_type; 11114 ret_flag = type_flag(ret_type); 11115 11116 switch (base_type(ret_type)) { 11117 case RET_INTEGER: 11118 /* sets type to SCALAR_VALUE */ 11119 mark_reg_unknown(env, regs, BPF_REG_0); 11120 break; 11121 case RET_VOID: 11122 regs[BPF_REG_0].type = NOT_INIT; 11123 break; 11124 case RET_PTR_TO_MAP_VALUE: 11125 /* There is no offset yet applied, variable or fixed */ 11126 mark_reg_known_zero(env, regs, BPF_REG_0); 11127 /* remember map_ptr, so that check_map_access() 11128 * can check 'value_size' boundary of memory access 11129 * to map element returned from bpf_map_lookup_elem() 11130 */ 11131 if (meta.map.ptr == NULL) { 11132 verifier_bug(env, "unexpected null map_ptr"); 11133 return -EFAULT; 11134 } 11135 11136 if (func_id == BPF_FUNC_map_lookup_elem && 11137 can_elide_value_nullness(meta.map.ptr) && 11138 meta.const_map_key >= 0 && 11139 meta.const_map_key < meta.map.ptr->max_entries) 11140 ret_flag &= ~PTR_MAYBE_NULL; 11141 11142 regs[BPF_REG_0].map_ptr = meta.map.ptr; 11143 regs[BPF_REG_0].map_uid = meta.map.uid; 11144 regs[BPF_REG_0].type = PTR_TO_MAP_VALUE | ret_flag; 11145 if (type_may_be_null(ret_flag) || 11146 btf_record_has_field(meta.map.ptr->record, BPF_SPIN_LOCK | BPF_RES_SPIN_LOCK)) { 11147 regs[BPF_REG_0].id = ++env->id_gen; 11148 } 11149 /* requires regs[BPF_REG_0].id to be set because of the map-in-map case */ 11150 refine_map_lookup_value(®s[BPF_REG_0]); 11151 break; 11152 case RET_PTR_TO_SOCKET: 11153 mark_reg_known_zero(env, regs, BPF_REG_0); 11154 regs[BPF_REG_0].type = PTR_TO_SOCKET | ret_flag; 11155 break; 11156 case RET_PTR_TO_SOCK_COMMON: 11157 mark_reg_known_zero(env, regs, BPF_REG_0); 11158 regs[BPF_REG_0].type = PTR_TO_SOCK_COMMON | ret_flag; 11159 break; 11160 case RET_PTR_TO_TCP_SOCK: 11161 mark_reg_known_zero(env, regs, BPF_REG_0); 11162 regs[BPF_REG_0].type = PTR_TO_TCP_SOCK | ret_flag; 11163 break; 11164 case RET_PTR_TO_MEM: 11165 mark_reg_known_zero(env, regs, BPF_REG_0); 11166 regs[BPF_REG_0].type = PTR_TO_MEM | ret_flag; 11167 regs[BPF_REG_0].mem_size = meta.ret_mem.size; 11168 break; 11169 case RET_PTR_TO_MEM_OR_BTF_ID: 11170 { 11171 const struct btf_type *t; 11172 11173 mark_reg_known_zero(env, regs, BPF_REG_0); 11174 t = btf_type_skip_modifiers(meta.ret_btf, meta.ret_btf_id, NULL); 11175 if (!btf_type_is_struct(t)) { 11176 u32 tsize; 11177 const struct btf_type *ret; 11178 const char *tname; 11179 11180 /* resolve the type size of ksym. */ 11181 ret = btf_resolve_size(meta.ret_btf, t, &tsize); 11182 if (IS_ERR(ret)) { 11183 tname = btf_name_by_offset(meta.ret_btf, t->name_off); 11184 verbose(env, "unable to resolve the size of type '%s': %ld\n", 11185 tname, PTR_ERR(ret)); 11186 return -EINVAL; 11187 } 11188 regs[BPF_REG_0].type = PTR_TO_MEM | ret_flag; 11189 regs[BPF_REG_0].mem_size = tsize; 11190 } else { 11191 if (returns_cpu_specific_alloc_ptr) { 11192 regs[BPF_REG_0].type = PTR_TO_BTF_ID | MEM_ALLOC | MEM_RCU; 11193 } else { 11194 /* MEM_RDONLY may be carried from ret_flag, but it 11195 * doesn't apply on PTR_TO_BTF_ID. Fold it, otherwise 11196 * it will confuse the check of PTR_TO_BTF_ID in 11197 * check_mem_access(). 11198 */ 11199 ret_flag &= ~MEM_RDONLY; 11200 regs[BPF_REG_0].type = PTR_TO_BTF_ID | ret_flag; 11201 } 11202 11203 regs[BPF_REG_0].btf = meta.ret_btf; 11204 regs[BPF_REG_0].btf_id = meta.ret_btf_id; 11205 } 11206 break; 11207 } 11208 case RET_PTR_TO_BTF_ID: 11209 { 11210 struct btf *ret_btf; 11211 int ret_btf_id; 11212 11213 mark_reg_known_zero(env, regs, BPF_REG_0); 11214 regs[BPF_REG_0].type = PTR_TO_BTF_ID | ret_flag; 11215 if (func_id == BPF_FUNC_kptr_xchg) { 11216 ret_btf = meta.kptr_field->kptr.btf; 11217 ret_btf_id = meta.kptr_field->kptr.btf_id; 11218 if (!btf_is_kernel(ret_btf)) { 11219 regs[BPF_REG_0].type |= MEM_ALLOC; 11220 if (meta.kptr_field->type == BPF_KPTR_PERCPU) 11221 regs[BPF_REG_0].type |= MEM_PERCPU; 11222 } 11223 } else { 11224 if (fn->ret_btf_id == BPF_PTR_POISON) { 11225 verifier_bug(env, "func %s has non-overwritten BPF_PTR_POISON return type", 11226 func_id_name(func_id)); 11227 return -EFAULT; 11228 } 11229 ret_btf = btf_vmlinux; 11230 ret_btf_id = *fn->ret_btf_id; 11231 } 11232 if (ret_btf_id == 0) { 11233 verbose(env, "invalid return type %u of func %s#%d\n", 11234 base_type(ret_type), func_id_name(func_id), 11235 func_id); 11236 return -EINVAL; 11237 } 11238 regs[BPF_REG_0].btf = ret_btf; 11239 regs[BPF_REG_0].btf_id = ret_btf_id; 11240 break; 11241 } 11242 default: 11243 verbose(env, "unknown return type %u of func %s#%d\n", 11244 base_type(ret_type), func_id_name(func_id), func_id); 11245 return -EINVAL; 11246 } 11247 11248 if (type_may_be_null(regs[BPF_REG_0].type) && !regs[BPF_REG_0].id) 11249 regs[BPF_REG_0].id = ++env->id_gen; 11250 11251 if (is_ptr_cast_function(func_id) && 11252 find_reference_state(env->cur_state, meta.ref_obj.id)) { 11253 struct bpf_verifier_state *branch; 11254 struct bpf_reg_state *r0; 11255 11256 err = validate_ref_obj(env, &meta.ref_obj); 11257 if (err) 11258 return err; 11259 11260 bpf_diag_mod_end(env); 11261 11262 /* 11263 * In order for a release of any of the original or cast pointers 11264 * to invalidate all other pointers, reuse the same reference id for 11265 * the cast result. 11266 * This reference id can't be used for nullness propagation, 11267 * as cast might return NULL for a non-NULL input. 11268 * Hence, explore the NULL case as a separate branch. 11269 */ 11270 branch = push_stack(env, env->insn_idx + 1, env->insn_idx, false); 11271 if (IS_ERR(branch)) 11272 return PTR_ERR(branch); 11273 11274 r0 = &branch->frame[branch->curframe]->regs[BPF_REG_0]; 11275 __mark_reg_known_zero(r0); 11276 r0->type = SCALAR_VALUE; 11277 11278 bpf_diag_mod_begin(env, ®s[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE); 11279 regs[BPF_REG_0].type &= ~PTR_MAYBE_NULL; 11280 regs[BPF_REG_0].id = meta.ref_obj.id; 11281 } else if (is_acquire_function(func_id, meta.map.ptr)) { 11282 int id = acquire_reference(env, insn_idx, 0); 11283 11284 if (id < 0) 11285 return id; 11286 11287 regs[BPF_REG_0].id = id; 11288 } 11289 11290 if (func_id == BPF_FUNC_dynptr_data) 11291 regs[BPF_REG_0].parent_id = meta.dynptr.id; 11292 11293 err = do_refine_retval_range(env, regs, fn->ret_type, func_id, &meta); 11294 if (err) 11295 return err; 11296 11297 bpf_diag_mod_end(env); 11298 11299 err = check_map_func_compatibility(env, meta.map.ptr, func_id); 11300 if (err) 11301 return err; 11302 11303 if ((func_id == BPF_FUNC_get_stack || 11304 func_id == BPF_FUNC_get_task_stack) && 11305 !env->prog->has_callchain_buf) { 11306 const char *err_str; 11307 11308 #ifdef CONFIG_PERF_EVENTS 11309 err = get_callchain_buffers(sysctl_perf_event_max_stack); 11310 err_str = "cannot get callchain buffer for func %s#%d\n"; 11311 #else 11312 err = -ENOTSUPP; 11313 err_str = "func %s#%d not supported without CONFIG_PERF_EVENTS\n"; 11314 #endif 11315 if (err) { 11316 verbose(env, err_str, func_id_name(func_id), func_id); 11317 return err; 11318 } 11319 11320 env->prog->has_callchain_buf = true; 11321 } 11322 11323 if (func_id == BPF_FUNC_get_stackid || func_id == BPF_FUNC_get_stack) 11324 env->prog->call_get_stack = true; 11325 11326 if (func_id == BPF_FUNC_get_func_ip) { 11327 if (check_get_func_ip(env)) 11328 return -ENOTSUPP; 11329 env->prog->call_get_func_ip = true; 11330 } 11331 11332 if (func_id == BPF_FUNC_tail_call) { 11333 if (env->cur_state->curframe) { 11334 struct bpf_verifier_state *branch; 11335 11336 /* 11337 * A taken tail call is modeled as a return from the current 11338 * frame. A callback frame cannot be left that way because 11339 * prepare_func_exit() would apply its return contract to the 11340 * unknown R0 synthesized below. Stack-depth validation rejects 11341 * this construct anyway. 11342 */ 11343 if (cur_func(env)->in_callback_fn) { 11344 verbose(env, "cannot tail call within callback\n"); 11345 return -EINVAL; 11346 } 11347 mark_reg_scratched(env, BPF_REG_0); 11348 branch = push_stack(env, env->insn_idx + 1, env->insn_idx, false); 11349 if (IS_ERR(branch)) 11350 return PTR_ERR(branch); 11351 clear_all_pkt_pointers(env); 11352 mark_reg_unknown(env, regs, BPF_REG_0); 11353 err = prepare_func_exit(env, &env->insn_idx); 11354 if (err) 11355 return err; 11356 env->insn_idx--; 11357 } else { 11358 changes_data = false; 11359 } 11360 } 11361 11362 if (changes_data) 11363 clear_all_pkt_pointers(env); 11364 return 0; 11365 } 11366 11367 static bool is_kfunc_acquire(struct bpf_call_arg_meta *meta) 11368 { 11369 return meta->kfunc_flags & KF_ACQUIRE; 11370 } 11371 11372 static bool is_kfunc_release(struct bpf_call_arg_meta *meta) 11373 { 11374 return meta->kfunc_flags & KF_RELEASE; 11375 } 11376 11377 static bool is_kfunc_destructive(struct bpf_call_arg_meta *meta) 11378 { 11379 return meta->kfunc_flags & KF_DESTRUCTIVE; 11380 } 11381 11382 static bool is_kfunc_perfmon(struct bpf_call_arg_meta *meta) 11383 { 11384 return meta->kfunc_flags & KF_PERFMON; 11385 } 11386 11387 static bool is_kfunc_rcu(struct bpf_call_arg_meta *meta) 11388 { 11389 return meta->kfunc_flags & KF_RCU; 11390 } 11391 11392 static bool is_kfunc_rcu_protected(struct bpf_call_arg_meta *meta) 11393 { 11394 return meta->kfunc_flags & KF_RCU_PROTECTED; 11395 } 11396 11397 static bool is_kfunc_arg_mem_size(const struct btf *btf, 11398 const struct btf_param *arg) 11399 { 11400 const struct btf_type *t; 11401 11402 t = btf_type_skip_modifiers(btf, arg->type, NULL); 11403 if (!btf_type_is_scalar(t)) 11404 return false; 11405 11406 return btf_param_match_suffix(btf, arg, "__sz"); 11407 } 11408 11409 static bool is_kfunc_arg_const_mem_size(const struct btf *btf, 11410 const struct btf_param *arg) 11411 { 11412 const struct btf_type *t; 11413 11414 t = btf_type_skip_modifiers(btf, arg->type, NULL); 11415 if (!btf_type_is_scalar(t)) 11416 return false; 11417 11418 return btf_param_match_suffix(btf, arg, "__szk"); 11419 } 11420 11421 static bool is_kfunc_arg_constant(const struct btf *btf, const struct btf_param *arg) 11422 { 11423 return btf_param_match_suffix(btf, arg, "__k"); 11424 } 11425 11426 static bool is_kfunc_arg_ignore(const struct btf *btf, const struct btf_param *arg) 11427 { 11428 return btf_param_match_suffix(btf, arg, "__ign"); 11429 } 11430 11431 static bool is_kfunc_arg_map(const struct btf *btf, const struct btf_param *arg) 11432 { 11433 return btf_param_match_suffix(btf, arg, "__map"); 11434 } 11435 11436 static bool is_kfunc_arg_const_map(const struct btf *btf, const struct btf_param *arg) 11437 { 11438 return btf_param_match_suffix(btf, arg, "__const_map"); 11439 } 11440 11441 static bool is_kfunc_arg_alloc_obj(const struct btf *btf, const struct btf_param *arg) 11442 { 11443 return btf_param_match_suffix(btf, arg, "__alloc"); 11444 } 11445 11446 static bool is_kfunc_arg_uninit(const struct btf *btf, const struct btf_param *arg) 11447 { 11448 return btf_param_match_suffix(btf, arg, "__uninit"); 11449 } 11450 11451 static bool is_kfunc_arg_refcounted_kptr(const struct btf *btf, const struct btf_param *arg) 11452 { 11453 return btf_param_match_suffix(btf, arg, "__refcounted_kptr"); 11454 } 11455 11456 static bool is_kfunc_arg_nullable(const struct btf *btf, const struct btf_param *arg) 11457 { 11458 return btf_param_match_suffix(btf, arg, "__nullable") || 11459 btf_param_match_suffix(btf, arg, "__arena"); 11460 } 11461 11462 static bool is_kfunc_arg_nonown_allowed(const struct btf *btf, const struct btf_param *arg) 11463 { 11464 return btf_param_match_suffix(btf, arg, "__nonown_allowed"); 11465 } 11466 11467 static bool is_kfunc_arg_const_str(const struct btf *btf, const struct btf_param *arg) 11468 { 11469 return btf_param_match_suffix(btf, arg, "__str"); 11470 } 11471 11472 static bool is_kfunc_arg_irq_flag(const struct btf *btf, const struct btf_param *arg) 11473 { 11474 return btf_param_match_suffix(btf, arg, "__irq_flag"); 11475 } 11476 11477 static bool is_kfunc_arg_arena(const struct btf *btf, const struct btf_param *arg) 11478 { 11479 return btf_param_match_suffix(btf, arg, "__arena__nullable") || 11480 btf_param_match_suffix(btf, arg, "__arena"); 11481 } 11482 11483 static bool is_kfunc_arg_scalar_with_name(const struct btf *btf, 11484 const struct btf_param *arg, 11485 const char *name) 11486 { 11487 int len, target_len = strlen(name); 11488 const char *param_name; 11489 11490 param_name = btf_name_by_offset(btf, arg->name_off); 11491 if (str_is_empty(param_name)) 11492 return false; 11493 len = strlen(param_name); 11494 if (len != target_len) 11495 return false; 11496 if (strcmp(param_name, name)) 11497 return false; 11498 11499 return true; 11500 } 11501 11502 enum { 11503 KF_ARG_DYNPTR_ID, 11504 KF_ARG_LIST_HEAD_ID, 11505 KF_ARG_LIST_NODE_ID, 11506 KF_ARG_RB_ROOT_ID, 11507 KF_ARG_RB_NODE_ID, 11508 KF_ARG_WORKQUEUE_ID, 11509 KF_ARG_RES_SPIN_LOCK_ID, 11510 KF_ARG_TASK_WORK_ID, 11511 KF_ARG_PROG_AUX_ID, 11512 KF_ARG_TIMER_ID 11513 }; 11514 11515 BTF_ID_LIST(kf_arg_btf_ids) 11516 BTF_ID(struct, bpf_dynptr) 11517 BTF_ID(struct, bpf_list_head) 11518 BTF_ID(struct, bpf_list_node) 11519 BTF_ID(struct, bpf_rb_root) 11520 BTF_ID(struct, bpf_rb_node) 11521 BTF_ID(struct, bpf_wq) 11522 BTF_ID(struct, bpf_res_spin_lock) 11523 BTF_ID(struct, bpf_task_work) 11524 BTF_ID(struct, bpf_prog_aux) 11525 BTF_ID(struct, bpf_timer) 11526 11527 static bool __is_kfunc_ptr_arg_type(const struct btf *btf, 11528 const struct btf_param *arg, int type) 11529 { 11530 const struct btf_type *t; 11531 u32 res_id; 11532 11533 t = btf_type_skip_modifiers(btf, arg->type, NULL); 11534 if (!t) 11535 return false; 11536 if (!btf_type_is_ptr(t)) 11537 return false; 11538 t = btf_type_skip_modifiers(btf, t->type, &res_id); 11539 if (!t) 11540 return false; 11541 return btf_types_are_same(btf, res_id, btf_vmlinux, kf_arg_btf_ids[type]); 11542 } 11543 11544 static bool is_kfunc_arg_dynptr(const struct btf *btf, const struct btf_param *arg) 11545 { 11546 return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_DYNPTR_ID); 11547 } 11548 11549 static bool is_kfunc_arg_list_head(const struct btf *btf, const struct btf_param *arg) 11550 { 11551 return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_LIST_HEAD_ID); 11552 } 11553 11554 static bool is_kfunc_arg_list_node(const struct btf *btf, const struct btf_param *arg) 11555 { 11556 return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_LIST_NODE_ID); 11557 } 11558 11559 static bool is_kfunc_arg_rbtree_root(const struct btf *btf, const struct btf_param *arg) 11560 { 11561 return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_RB_ROOT_ID); 11562 } 11563 11564 static bool is_kfunc_arg_rbtree_node(const struct btf *btf, const struct btf_param *arg) 11565 { 11566 return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_RB_NODE_ID); 11567 } 11568 11569 static bool is_kfunc_arg_timer(const struct btf *btf, const struct btf_param *arg) 11570 { 11571 return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_TIMER_ID); 11572 } 11573 11574 static bool is_kfunc_arg_wq(const struct btf *btf, const struct btf_param *arg) 11575 { 11576 return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_WORKQUEUE_ID); 11577 } 11578 11579 static bool is_kfunc_arg_task_work(const struct btf *btf, const struct btf_param *arg) 11580 { 11581 return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_TASK_WORK_ID); 11582 } 11583 11584 static bool is_kfunc_arg_res_spin_lock(const struct btf *btf, const struct btf_param *arg) 11585 { 11586 return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_RES_SPIN_LOCK_ID); 11587 } 11588 11589 static bool is_rbtree_node_type(const struct btf_type *t) 11590 { 11591 return t == btf_type_by_id(btf_vmlinux, kf_arg_btf_ids[KF_ARG_RB_NODE_ID]); 11592 } 11593 11594 static bool is_list_node_type(const struct btf_type *t) 11595 { 11596 return t == btf_type_by_id(btf_vmlinux, kf_arg_btf_ids[KF_ARG_LIST_NODE_ID]); 11597 } 11598 11599 static bool is_kfunc_arg_callback(struct bpf_verifier_env *env, const struct btf *btf, 11600 const struct btf_param *arg) 11601 { 11602 const struct btf_type *t; 11603 11604 t = btf_type_resolve_func_ptr(btf, arg->type, NULL); 11605 if (!t) 11606 return false; 11607 11608 return true; 11609 } 11610 11611 static bool is_kfunc_arg_prog_aux(const struct btf *btf, const struct btf_param *arg) 11612 { 11613 return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_PROG_AUX_ID); 11614 } 11615 11616 /* 11617 * A kfunc with KF_IMPLICIT_ARGS has two prototypes in BTF: 11618 * - the _impl prototype with full arg list (meta->func_proto) 11619 * - the BPF API prototype w/o implicit args (func->type in BTF) 11620 * To determine whether an argument is implicit, we compare its position 11621 * against the number of arguments in the prototype w/o implicit args. 11622 */ 11623 static bool is_kfunc_arg_implicit(const struct bpf_call_arg_meta *meta, u32 arg_idx) 11624 { 11625 const struct btf_type *func, *func_proto; 11626 u32 argn; 11627 11628 if (!(meta->kfunc_flags & KF_IMPLICIT_ARGS)) 11629 return false; 11630 11631 func = btf_type_by_id(meta->btf, meta->func_id); 11632 func_proto = btf_type_by_id(meta->btf, func->type); 11633 argn = btf_type_vlen(func_proto); 11634 11635 return argn <= arg_idx; 11636 } 11637 11638 /* Returns true if struct is composed of scalars, 4 levels of nesting allowed */ 11639 static bool __btf_type_is_scalar_struct(struct bpf_verifier_env *env, 11640 const struct btf *btf, 11641 const struct btf_type *t, int rec) 11642 { 11643 const struct btf_type *member_type; 11644 const struct btf_member *member; 11645 u32 i; 11646 11647 if (!btf_type_is_struct(t)) 11648 return false; 11649 11650 for_each_member(i, t, member) { 11651 const struct btf_array *array; 11652 11653 member_type = btf_type_skip_modifiers(btf, member->type, NULL); 11654 if (btf_type_is_struct(member_type)) { 11655 if (rec >= 3) { 11656 verbose(env, "max struct nesting depth exceeded\n"); 11657 return false; 11658 } 11659 if (!__btf_type_is_scalar_struct(env, btf, member_type, rec + 1)) 11660 return false; 11661 continue; 11662 } 11663 if (btf_type_is_array(member_type)) { 11664 array = btf_array(member_type); 11665 if (!array->nelems) 11666 return false; 11667 member_type = btf_type_skip_modifiers(btf, array->type, NULL); 11668 if (!btf_type_is_scalar(member_type)) 11669 return false; 11670 continue; 11671 } 11672 if (!btf_type_is_scalar(member_type)) 11673 return false; 11674 } 11675 return true; 11676 } 11677 11678 enum kfunc_ptr_arg_type { 11679 KF_ARG_CONST_MEM_SIZE, 11680 KF_ARG_MEM_SIZE, 11681 KF_ARG_CONST, 11682 KF_ARG_CONST_ALLOC_SIZE_OR_ZERO, 11683 KF_ARG_ANYTHING, 11684 KF_ARG_PTR_TO_CTX, 11685 KF_ARG_PTR_TO_ALLOC_BTF_ID, /* Allocated object */ 11686 KF_ARG_PTR_TO_REFCOUNTED_KPTR, /* Refcounted local kptr */ 11687 KF_ARG_PTR_TO_DYNPTR, 11688 KF_ARG_PTR_TO_ITER, 11689 KF_ARG_PTR_TO_LIST_HEAD, 11690 KF_ARG_PTR_TO_LIST_NODE, 11691 KF_ARG_PTR_TO_BTF_ID, /* Also covers reg2btf_ids conversions */ 11692 KF_ARG_PTR_TO_MEM, 11693 KF_ARG_PTR_TO_CALLBACK, 11694 KF_ARG_PTR_TO_RB_ROOT, 11695 KF_ARG_PTR_TO_RB_NODE, 11696 KF_ARG_PTR_TO_CONST_STR, 11697 KF_ARG_CONST_MAP_PTR, 11698 KF_ARG_PTR_TO_TIMER, 11699 KF_ARG_PTR_TO_WORKQUEUE, 11700 KF_ARG_PTR_TO_IRQ_FLAG, 11701 KF_ARG_PTR_TO_RES_SPIN_LOCK, 11702 KF_ARG_PTR_TO_TASK_WORK, 11703 KF_ARG_PTR_TO_ARENA, 11704 }; 11705 11706 enum special_kfunc_type { 11707 KF_bpf_obj_new_impl, 11708 KF_bpf_obj_new, 11709 KF_bpf_obj_drop_impl, 11710 KF_bpf_obj_drop, 11711 KF_bpf_refcount_acquire_impl, 11712 KF_bpf_refcount_acquire, 11713 KF_bpf_list_push_front_impl, 11714 KF_bpf_list_push_front, 11715 KF_bpf_list_push_back_impl, 11716 KF_bpf_list_push_back, 11717 KF_bpf_list_add, 11718 KF_bpf_list_pop_front, 11719 KF_bpf_list_pop_back, 11720 KF_bpf_list_del, 11721 KF_bpf_list_front, 11722 KF_bpf_list_back, 11723 KF_bpf_list_is_first, 11724 KF_bpf_list_is_last, 11725 KF_bpf_list_empty, 11726 KF_bpf_cast_to_kern_ctx, 11727 KF_bpf_rdonly_cast, 11728 KF_bpf_rcu_read_lock, 11729 KF_bpf_rcu_read_unlock, 11730 KF_bpf_rbtree_remove, 11731 KF_bpf_rbtree_add_impl, 11732 KF_bpf_rbtree_add, 11733 KF_bpf_rbtree_first, 11734 KF_bpf_rbtree_root, 11735 KF_bpf_rbtree_left, 11736 KF_bpf_rbtree_right, 11737 KF_bpf_dynptr_from_skb, 11738 KF_bpf_dynptr_from_xdp, 11739 KF_bpf_dynptr_from_skb_meta, 11740 KF_bpf_xdp_pull_data, 11741 KF_bpf_dynptr_slice, 11742 KF_bpf_dynptr_slice_rdwr, 11743 KF_bpf_dynptr_clone, 11744 KF_bpf_percpu_obj_new_impl, 11745 KF_bpf_percpu_obj_new, 11746 KF_bpf_percpu_obj_drop_impl, 11747 KF_bpf_percpu_obj_drop, 11748 KF_bpf_throw, 11749 KF_bpf_wq_set_callback, 11750 KF_bpf_preempt_disable, 11751 KF_bpf_preempt_enable, 11752 KF_bpf_iter_css_task_new, 11753 KF_bpf_session_cookie, 11754 KF_bpf_get_kmem_cache, 11755 KF_bpf_local_irq_save, 11756 KF_bpf_local_irq_restore, 11757 KF_bpf_iter_num_new, 11758 KF_bpf_iter_num_next, 11759 KF_bpf_iter_num_destroy, 11760 KF_bpf_set_dentry_xattr, 11761 KF_bpf_remove_dentry_xattr, 11762 KF_bpf_res_spin_lock, 11763 KF_bpf_res_spin_unlock, 11764 KF_bpf_res_spin_lock_irqsave, 11765 KF_bpf_res_spin_unlock_irqrestore, 11766 KF_bpf_dynptr_from_file, 11767 KF_bpf_dynptr_file_discard, 11768 KF___bpf_trap, 11769 KF_bpf_task_work_schedule_signal, 11770 KF_bpf_task_work_schedule_resume, 11771 KF_bpf_arena_alloc_pages, 11772 KF_bpf_arena_free_pages, 11773 KF_bpf_session_is_return, 11774 }; 11775 11776 BTF_ID_LIST(special_kfunc_list) 11777 BTF_ID(func, bpf_obj_new_impl) 11778 BTF_ID(func, bpf_obj_new) 11779 BTF_ID(func, bpf_obj_drop_impl) 11780 BTF_ID(func, bpf_obj_drop) 11781 BTF_ID(func, bpf_refcount_acquire_impl) 11782 BTF_ID(func, bpf_refcount_acquire) 11783 BTF_ID(func, bpf_list_push_front_impl) 11784 BTF_ID(func, bpf_list_push_front) 11785 BTF_ID(func, bpf_list_push_back_impl) 11786 BTF_ID(func, bpf_list_push_back) 11787 BTF_ID(func, bpf_list_add) 11788 BTF_ID(func, bpf_list_pop_front) 11789 BTF_ID(func, bpf_list_pop_back) 11790 BTF_ID(func, bpf_list_del) 11791 BTF_ID(func, bpf_list_front) 11792 BTF_ID(func, bpf_list_back) 11793 BTF_ID(func, bpf_list_is_first) 11794 BTF_ID(func, bpf_list_is_last) 11795 BTF_ID(func, bpf_list_empty) 11796 BTF_ID(func, bpf_cast_to_kern_ctx) 11797 BTF_ID(func, bpf_rdonly_cast) 11798 BTF_ID(func, bpf_rcu_read_lock) 11799 BTF_ID(func, bpf_rcu_read_unlock) 11800 BTF_ID(func, bpf_rbtree_remove) 11801 BTF_ID(func, bpf_rbtree_add_impl) 11802 BTF_ID(func, bpf_rbtree_add) 11803 BTF_ID(func, bpf_rbtree_first) 11804 BTF_ID(func, bpf_rbtree_root) 11805 BTF_ID(func, bpf_rbtree_left) 11806 BTF_ID(func, bpf_rbtree_right) 11807 #ifdef CONFIG_NET 11808 BTF_ID(func, bpf_dynptr_from_skb) 11809 BTF_ID(func, bpf_dynptr_from_xdp) 11810 BTF_ID(func, bpf_dynptr_from_skb_meta) 11811 BTF_ID(func, bpf_xdp_pull_data) 11812 #else 11813 BTF_ID_UNUSED 11814 BTF_ID_UNUSED 11815 BTF_ID_UNUSED 11816 BTF_ID_UNUSED 11817 #endif 11818 BTF_ID(func, bpf_dynptr_slice) 11819 BTF_ID(func, bpf_dynptr_slice_rdwr) 11820 BTF_ID(func, bpf_dynptr_clone) 11821 BTF_ID(func, bpf_percpu_obj_new_impl) 11822 BTF_ID(func, bpf_percpu_obj_new) 11823 BTF_ID(func, bpf_percpu_obj_drop_impl) 11824 BTF_ID(func, bpf_percpu_obj_drop) 11825 BTF_ID(func, bpf_throw) 11826 BTF_ID(func, bpf_wq_set_callback) 11827 BTF_ID(func, bpf_preempt_disable) 11828 BTF_ID(func, bpf_preempt_enable) 11829 #ifdef CONFIG_CGROUPS 11830 BTF_ID(func, bpf_iter_css_task_new) 11831 #else 11832 BTF_ID_UNUSED 11833 #endif 11834 #ifdef CONFIG_BPF_EVENTS 11835 BTF_ID(func, bpf_session_cookie) 11836 #else 11837 BTF_ID_UNUSED 11838 #endif 11839 BTF_ID(func, bpf_get_kmem_cache) 11840 BTF_ID(func, bpf_local_irq_save) 11841 BTF_ID(func, bpf_local_irq_restore) 11842 BTF_ID(func, bpf_iter_num_new) 11843 BTF_ID(func, bpf_iter_num_next) 11844 BTF_ID(func, bpf_iter_num_destroy) 11845 #ifdef CONFIG_BPF_LSM 11846 BTF_ID(func, bpf_set_dentry_xattr) 11847 BTF_ID(func, bpf_remove_dentry_xattr) 11848 #else 11849 BTF_ID_UNUSED 11850 BTF_ID_UNUSED 11851 #endif 11852 BTF_ID(func, bpf_res_spin_lock) 11853 BTF_ID(func, bpf_res_spin_unlock) 11854 BTF_ID(func, bpf_res_spin_lock_irqsave) 11855 BTF_ID(func, bpf_res_spin_unlock_irqrestore) 11856 BTF_ID(func, bpf_dynptr_from_file) 11857 BTF_ID(func, bpf_dynptr_file_discard) 11858 BTF_ID(func, __bpf_trap) 11859 BTF_ID(func, bpf_task_work_schedule_signal) 11860 BTF_ID(func, bpf_task_work_schedule_resume) 11861 BTF_ID(func, bpf_arena_alloc_pages) 11862 BTF_ID(func, bpf_arena_free_pages) 11863 #ifdef CONFIG_BPF_EVENTS 11864 BTF_ID(func, bpf_session_is_return) 11865 #else 11866 BTF_ID_UNUSED 11867 #endif 11868 11869 static bool is_bpf_obj_new_kfunc(u32 func_id) 11870 { 11871 return func_id == special_kfunc_list[KF_bpf_obj_new] || 11872 func_id == special_kfunc_list[KF_bpf_obj_new_impl]; 11873 } 11874 11875 static bool is_bpf_percpu_obj_new_kfunc(u32 func_id) 11876 { 11877 return func_id == special_kfunc_list[KF_bpf_percpu_obj_new] || 11878 func_id == special_kfunc_list[KF_bpf_percpu_obj_new_impl]; 11879 } 11880 11881 static bool is_bpf_obj_drop_kfunc(u32 func_id) 11882 { 11883 return func_id == special_kfunc_list[KF_bpf_obj_drop] || 11884 func_id == special_kfunc_list[KF_bpf_obj_drop_impl]; 11885 } 11886 11887 static bool is_bpf_percpu_obj_drop_kfunc(u32 func_id) 11888 { 11889 return func_id == special_kfunc_list[KF_bpf_percpu_obj_drop] || 11890 func_id == special_kfunc_list[KF_bpf_percpu_obj_drop_impl]; 11891 } 11892 11893 static bool is_bpf_refcount_acquire_kfunc(u32 func_id) 11894 { 11895 return func_id == special_kfunc_list[KF_bpf_refcount_acquire] || 11896 func_id == special_kfunc_list[KF_bpf_refcount_acquire_impl]; 11897 } 11898 11899 static bool is_bpf_list_push_kfunc(u32 func_id) 11900 { 11901 return func_id == special_kfunc_list[KF_bpf_list_push_front] || 11902 func_id == special_kfunc_list[KF_bpf_list_push_front_impl] || 11903 func_id == special_kfunc_list[KF_bpf_list_push_back] || 11904 func_id == special_kfunc_list[KF_bpf_list_push_back_impl] || 11905 func_id == special_kfunc_list[KF_bpf_list_add]; 11906 } 11907 11908 static bool is_bpf_rbtree_add_kfunc(u32 func_id) 11909 { 11910 return func_id == special_kfunc_list[KF_bpf_rbtree_add] || 11911 func_id == special_kfunc_list[KF_bpf_rbtree_add_impl]; 11912 } 11913 11914 static bool is_task_work_add_kfunc(u32 func_id) 11915 { 11916 return func_id == special_kfunc_list[KF_bpf_task_work_schedule_signal] || 11917 func_id == special_kfunc_list[KF_bpf_task_work_schedule_resume]; 11918 } 11919 11920 static bool is_kfunc_ret_null(struct bpf_call_arg_meta *meta) 11921 { 11922 if (is_bpf_refcount_acquire_kfunc(meta->func_id) && meta->arg_owning_ref) 11923 return false; 11924 11925 return meta->kfunc_flags & KF_RET_NULL; 11926 } 11927 11928 static bool is_kfunc_bpf_rcu_read_lock(struct bpf_call_arg_meta *meta) 11929 { 11930 return meta->func_id == special_kfunc_list[KF_bpf_rcu_read_lock]; 11931 } 11932 11933 static bool is_kfunc_bpf_rcu_read_unlock(struct bpf_call_arg_meta *meta) 11934 { 11935 return meta->func_id == special_kfunc_list[KF_bpf_rcu_read_unlock]; 11936 } 11937 11938 static bool is_kfunc_bpf_preempt_disable(struct bpf_call_arg_meta *meta) 11939 { 11940 return meta->func_id == special_kfunc_list[KF_bpf_preempt_disable]; 11941 } 11942 11943 static bool is_kfunc_bpf_preempt_enable(struct bpf_call_arg_meta *meta) 11944 { 11945 return meta->func_id == special_kfunc_list[KF_bpf_preempt_enable]; 11946 } 11947 11948 bool bpf_is_kfunc_pkt_changing(struct bpf_call_arg_meta *meta) 11949 { 11950 return meta->func_id == special_kfunc_list[KF_bpf_xdp_pull_data]; 11951 } 11952 11953 static int 11954 get_kfunc_arg_type(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta, 11955 const struct btf_param *args, int arg, int nargs) 11956 { 11957 const struct btf_type *t, *ref_t = NULL; 11958 argno_t argno = argno_from_arg(arg + 1); 11959 const char *ref_tname = NULL; 11960 int arg_type; 11961 11962 t = btf_type_skip_modifiers(meta->btf, args[arg].type, NULL); 11963 11964 /* Scalar arguments are classified from their BTF suffix/name alone. */ 11965 if (btf_type_is_scalar(t)) { 11966 if (is_kfunc_arg_constant(meta->btf, &args[arg])) 11967 return KF_ARG_CONST; 11968 if (is_kfunc_arg_const_mem_size(meta->btf, &args[arg])) 11969 return KF_ARG_CONST_MEM_SIZE; 11970 if (is_kfunc_arg_mem_size(meta->btf, &args[arg])) 11971 return KF_ARG_MEM_SIZE; 11972 if (is_kfunc_arg_scalar_with_name(meta->btf, &args[arg], "rdonly_buf_size") || 11973 is_kfunc_arg_scalar_with_name(meta->btf, &args[arg], "rdwr_buf_size")) 11974 return KF_ARG_CONST_ALLOC_SIZE_OR_ZERO; 11975 return KF_ARG_ANYTHING; 11976 } 11977 11978 if (!btf_type_is_ptr(t)) { 11979 verbose(env, "Unrecognized %s type %s\n", 11980 reg_arg_name(env, argno), btf_type_str(t)); 11981 return -EINVAL; 11982 } 11983 ref_t = btf_type_skip_modifiers(meta->btf, t->type, NULL); 11984 ref_tname = btf_name_by_offset(meta->btf, ref_t->name_off); 11985 11986 /* In this function, we verify the kfunc's BTF as per the argument type, 11987 * leaving the rest of the verification with respect to the register 11988 * type to our caller. When a set of conditions hold in the BTF type of 11989 * arguments, we resolve it to a known kfunc_ptr_arg_type. 11990 */ 11991 if (meta->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx] || 11992 meta->func_id == special_kfunc_list[KF_bpf_session_is_return] || 11993 meta->func_id == special_kfunc_list[KF_bpf_session_cookie]) 11994 arg_type = KF_ARG_PTR_TO_CTX; 11995 else if (btf_is_prog_ctx_type(&env->log, meta->btf, t, resolve_prog_type(env->prog), arg)) 11996 arg_type = KF_ARG_PTR_TO_CTX; 11997 else if (is_kfunc_arg_alloc_obj(meta->btf, &args[arg])) 11998 arg_type = KF_ARG_PTR_TO_ALLOC_BTF_ID; 11999 else if (is_kfunc_arg_refcounted_kptr(meta->btf, &args[arg])) 12000 arg_type = KF_ARG_PTR_TO_REFCOUNTED_KPTR; 12001 else if (is_kfunc_arg_dynptr(meta->btf, &args[arg])) 12002 arg_type = KF_ARG_PTR_TO_DYNPTR; 12003 else if (is_kfunc_arg_iter(meta, arg, &args[arg])) 12004 arg_type = KF_ARG_PTR_TO_ITER; 12005 else if (is_kfunc_arg_list_head(meta->btf, &args[arg])) 12006 arg_type = KF_ARG_PTR_TO_LIST_HEAD; 12007 else if (is_kfunc_arg_list_node(meta->btf, &args[arg])) 12008 arg_type = KF_ARG_PTR_TO_LIST_NODE; 12009 else if (is_kfunc_arg_rbtree_root(meta->btf, &args[arg])) 12010 arg_type = KF_ARG_PTR_TO_RB_ROOT; 12011 else if (is_kfunc_arg_rbtree_node(meta->btf, &args[arg])) 12012 arg_type = KF_ARG_PTR_TO_RB_NODE; 12013 else if (is_kfunc_arg_const_str(meta->btf, &args[arg])) 12014 arg_type = KF_ARG_PTR_TO_CONST_STR; 12015 else if (is_kfunc_arg_const_map(meta->btf, &args[arg])) 12016 arg_type = KF_ARG_CONST_MAP_PTR; 12017 else if (is_kfunc_arg_map(meta->btf, &args[arg])) 12018 arg_type = KF_ARG_PTR_TO_BTF_ID; 12019 else if (is_kfunc_arg_wq(meta->btf, &args[arg])) 12020 arg_type = KF_ARG_PTR_TO_WORKQUEUE; 12021 else if (is_kfunc_arg_timer(meta->btf, &args[arg])) 12022 arg_type = KF_ARG_PTR_TO_TIMER; 12023 else if (is_kfunc_arg_task_work(meta->btf, &args[arg])) 12024 arg_type = KF_ARG_PTR_TO_TASK_WORK; 12025 else if (is_kfunc_arg_irq_flag(meta->btf, &args[arg])) 12026 arg_type = KF_ARG_PTR_TO_IRQ_FLAG; 12027 else if (is_kfunc_arg_res_spin_lock(meta->btf, &args[arg])) 12028 arg_type = KF_ARG_PTR_TO_RES_SPIN_LOCK; 12029 else if (is_kfunc_arg_callback(env, meta->btf, &args[arg])) 12030 arg_type = KF_ARG_PTR_TO_CALLBACK; 12031 else if (is_kfunc_arg_arena(meta->btf, &args[arg])) { 12032 if (!bpf_jit_supports_arena_args()) { 12033 verbose(env, "JIT does not support kfunc %s() with arena pointer arguments\n", 12034 meta->func_name); 12035 return -ENOTSUPP; 12036 } 12037 if (!env->prog->aux->arena) { 12038 verbose(env, 12039 "%s arena pointer requires a program with an associated arena\n", 12040 reg_arg_name(env, argno)); 12041 return -EINVAL; 12042 } 12043 if (reg_from_argno(argno) < 0) { 12044 verbose(env, "%s arena pointer cannot be a stack argument\n", 12045 reg_arg_name(env, argno)); 12046 return -EINVAL; 12047 } 12048 /* 12049 * Both suffixes accept a constant zero. The function model determines 12050 * whether the JIT rebases it to the arena base or preserves NULL. 12051 * The common nullable path below records that verifier property. 12052 */ 12053 arg_type = KF_ARG_PTR_TO_ARENA; 12054 } else if (arg + 1 < nargs && 12055 (is_kfunc_arg_mem_size(meta->btf, &args[arg + 1]) || 12056 is_kfunc_arg_const_mem_size(meta->btf, &args[arg + 1]))) { 12057 if (!btf_type_is_void(ref_t) && !btf_type_is_scalar(ref_t) && 12058 !__btf_type_is_scalar_struct(env, meta->btf, ref_t, 0)) { 12059 verbose(env, "%s pointer type %s %s must point to void, scalar, or struct with scalar\n", 12060 reg_arg_name(env, argno), btf_type_str(ref_t), ref_tname); 12061 return -EINVAL; 12062 } 12063 arg_type = KF_ARG_PTR_TO_MEM; 12064 } else if (btf_type_is_struct(ref_t)) 12065 /* A pointer to a struct without a size argument is classified as KF_ARG_PTR_TO_BTF_ID */ 12066 arg_type = KF_ARG_PTR_TO_BTF_ID; 12067 else { 12068 /* 12069 * Otherwise this is a fixed-size memory buffer supported by 12070 * check_helper_mem_access(): a pointer to a scalar or a struct of 12071 * scalars. The access size is derived from the pointed-to BTF type. 12072 */ 12073 if (!btf_type_is_scalar(ref_t) && 12074 !__btf_type_is_scalar_struct(env, meta->btf, ref_t, 0)) { 12075 verbose(env, "%s pointer type %s %s must point to scalar, or struct with scalar\n", 12076 reg_arg_name(env, argno), btf_type_str(ref_t), ref_tname); 12077 return -EINVAL; 12078 } 12079 arg_type = KF_ARG_PTR_TO_MEM | MEM_FIXED_SIZE; 12080 } 12081 12082 if (is_kfunc_arg_nullable(meta->btf, &args[arg])) 12083 arg_type |= PTR_MAYBE_NULL; 12084 12085 return arg_type; 12086 } 12087 12088 static int gen_kfunc_arg_proto(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta, 12089 struct bpf_func_proto *proto) 12090 { 12091 const struct btf *btf = meta->btf; 12092 const struct btf_param *args; 12093 u32 i, nargs; 12094 int arg_type; 12095 12096 args = (const struct btf_param *)(meta->func_proto + 1); 12097 nargs = btf_type_vlen(meta->func_proto); 12098 if (nargs > MAX_BPF_FUNC_ARGS) { 12099 verbose(env, "Function %s has %d > %d args\n", meta->func_name, 12100 nargs, MAX_BPF_FUNC_ARGS); 12101 return -EINVAL; 12102 } 12103 if (nargs > MAX_BPF_FUNC_REG_ARGS && !bpf_jit_supports_stack_args()) { 12104 verbose(env, "JIT does not support kfunc %s() with %d args\n", 12105 meta->func_name, nargs); 12106 return -ENOTSUPP; 12107 } 12108 12109 for (i = 0; i < nargs; i++) { 12110 if (is_kfunc_arg_prog_aux(btf, &args[i]) || 12111 is_kfunc_arg_ignore(btf, &args[i]) || 12112 is_kfunc_arg_implicit(meta, i)) 12113 continue; 12114 12115 arg_type = get_kfunc_arg_type(env, meta, args, i, nargs); 12116 if (arg_type < 0) 12117 return arg_type; 12118 12119 proto->arg_type[i] = arg_type; 12120 } 12121 12122 return 0; 12123 } 12124 12125 static int process_kf_arg_ptr_to_btf_id(struct bpf_verifier_env *env, 12126 struct bpf_reg_state *reg, 12127 const struct btf_type *ref_t, 12128 const char *ref_tname, u32 ref_id, 12129 struct bpf_call_arg_meta *meta, 12130 int arg, argno_t argno) 12131 { 12132 const struct btf_type *reg_ref_t; 12133 bool strict_type_match = false; 12134 const struct btf *reg_btf; 12135 const char *reg_ref_tname; 12136 bool taking_projection; 12137 bool struct_same; 12138 u32 reg_ref_id; 12139 12140 if (base_type(reg->type) == PTR_TO_BTF_ID) { 12141 reg_btf = reg->btf; 12142 reg_ref_id = reg->btf_id; 12143 } else { 12144 reg_btf = btf_vmlinux; 12145 reg_ref_id = *reg2btf_ids[base_type(reg->type)]; 12146 } 12147 12148 /* Enforce strict type matching for calls to kfuncs that are acquiring 12149 * or releasing a reference, or are no-cast aliases. We do _not_ 12150 * enforce strict matching for kfuncs by default, 12151 * as we want to enable BPF programs to pass types that are bitwise 12152 * equivalent without forcing them to explicitly cast with something 12153 * like bpf_cast_to_kern_ctx(). 12154 * 12155 * For example, say we had a type like the following: 12156 * 12157 * struct bpf_cpumask { 12158 * cpumask_t cpumask; 12159 * refcount_t usage; 12160 * }; 12161 * 12162 * Note that as specified in <linux/cpumask.h>, cpumask_t is typedef'ed 12163 * to a struct cpumask, so it would be safe to pass a struct 12164 * bpf_cpumask * to a kfunc expecting a struct cpumask *. 12165 * 12166 * The philosophy here is similar to how we allow scalars of different 12167 * types to be passed to kfuncs as long as the size is the same. The 12168 * only difference here is that we're simply allowing 12169 * btf_struct_ids_match() to walk the struct at the 0th offset, and 12170 * resolve types. 12171 */ 12172 if ((is_kfunc_release(meta) && reg_is_referenced(env, reg)) || 12173 btf_type_ids_nocast_alias(&env->log, reg_btf, reg_ref_id, meta->btf, ref_id)) 12174 strict_type_match = true; 12175 12176 WARN_ON_ONCE(is_kfunc_release(meta) && !tnum_is_const(reg->var_off)); 12177 12178 reg_ref_t = btf_type_skip_modifiers(reg_btf, reg_ref_id, ®_ref_id); 12179 reg_ref_tname = btf_name_by_offset(reg_btf, reg_ref_t->name_off); 12180 struct_same = btf_struct_ids_match(&env->log, reg_btf, reg_ref_id, reg->var_off.value, 12181 meta->btf, ref_id, strict_type_match, 12182 !type_is_alloc(reg->type)); 12183 /* If kfunc is accepting a projection type (ie. __sk_buff), it cannot 12184 * actually use it -- it must cast to the underlying type. So we allow 12185 * caller to pass in the underlying type. 12186 */ 12187 taking_projection = btf_is_projection_of(ref_tname, reg_ref_tname); 12188 if (!taking_projection && !struct_same) { 12189 verbose(env, "kernel function %s %s expected pointer to %s %s but %s has a pointer to %s %s\n", 12190 meta->func_name, reg_arg_name(env, argno), 12191 btf_type_str(ref_t), ref_tname, reg_arg_name(env, argno), 12192 btf_type_str(reg_ref_t), reg_ref_tname); 12193 return -EINVAL; 12194 } 12195 return 0; 12196 } 12197 12198 static int process_irq_flag(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, 12199 struct bpf_call_arg_meta *meta) 12200 { 12201 int err, spi, kfunc_class = IRQ_NATIVE_KFUNC; 12202 bool irq_save; 12203 12204 if (meta->func_id == special_kfunc_list[KF_bpf_local_irq_save] || 12205 meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave]) { 12206 irq_save = true; 12207 if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave]) 12208 kfunc_class = IRQ_LOCK_KFUNC; 12209 } else if (meta->func_id == special_kfunc_list[KF_bpf_local_irq_restore] || 12210 meta->func_id == special_kfunc_list[KF_bpf_res_spin_unlock_irqrestore]) { 12211 irq_save = false; 12212 if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_unlock_irqrestore]) 12213 kfunc_class = IRQ_LOCK_KFUNC; 12214 } else { 12215 verifier_bug(env, "unknown irq flags kfunc"); 12216 return -EFAULT; 12217 } 12218 12219 if (irq_save) { 12220 if (!is_irq_flag_reg_valid_uninit(env, reg)) { 12221 verbose(env, "expected uninitialized irq flag as %s\n", 12222 reg_arg_name(env, argno)); 12223 bpf_diag_res(env, env->insn_idx, "IRQ flag is already initialized", 12224 "Saving IRQ state requires an uninitialized stack slot for " 12225 "the IRQ flag, but this slot already contains tracked IRQ " 12226 "flag state.", 12227 "Use a fresh stack slot for this save operation, or restore " 12228 "the existing IRQ flag before reusing the slot."); 12229 return -EINVAL; 12230 } 12231 12232 err = check_mem_access(env, env->insn_idx, reg, argno, 0, BPF_DW, 12233 BPF_WRITE, -1, false, false); 12234 if (err) 12235 return err; 12236 12237 err = mark_stack_slot_irq_flag(env, meta, reg, env->insn_idx, kfunc_class); 12238 if (err) 12239 return err; 12240 } else { 12241 err = is_irq_flag_reg_valid_init(env, reg); 12242 if (err) { 12243 verbose(env, "expected an initialized irq flag as %s\n", 12244 reg_arg_name(env, argno)); 12245 bpf_diag_res(env, env->insn_idx, "uninitialized IRQ flag restore", 12246 "Restoring IRQ state requires a stack slot that was " 12247 "initialized by a matching IRQ save operation on this path.", 12248 "Pass the same stack slot that was previously initialized by " 12249 "the matching IRQ save kfunc."); 12250 return err; 12251 } 12252 12253 spi = irq_flag_get_spi(env, reg); 12254 if (spi < 0) 12255 return spi; 12256 12257 mark_stack_slots_scratched(env, spi, 1); 12258 12259 err = unmark_stack_slot_irq_flag(env, reg, kfunc_class); 12260 if (err) 12261 return err; 12262 12263 if (!in_rcu_cs(env)) 12264 invalidate_rcu_protected_refs(env); 12265 } 12266 return 0; 12267 } 12268 12269 static int ref_set_non_owning(struct bpf_verifier_env *env, struct bpf_reg_state *reg) 12270 { 12271 struct btf_record *rec = reg_btf_record(reg); 12272 12273 if (!env->cur_state->active_locks) { 12274 verifier_bug(env, "%s w/o active lock", __func__); 12275 return -EFAULT; 12276 } 12277 12278 if (type_flag(reg->type) & NON_OWN_REF) { 12279 verifier_bug(env, "NON_OWN_REF already set"); 12280 return -EFAULT; 12281 } 12282 12283 reg->type |= NON_OWN_REF; 12284 if (rec->refcount_off >= 0) 12285 reg->type |= MEM_RCU; 12286 12287 return 0; 12288 } 12289 12290 static void ref_convert_owning_non_owning(struct bpf_verifier_env *env, u32 id) 12291 { 12292 struct bpf_func_state *unused; 12293 struct bpf_reg_state *reg; 12294 int err; 12295 12296 err = release_reference_nomark(env, id); 12297 WARN_ON_ONCE(err); 12298 12299 bpf_for_each_reg_in_vstate(env->cur_state, unused, reg, ({ 12300 if (reg->id == id) { 12301 reg->id = 0; 12302 ref_set_non_owning(env, reg); 12303 } 12304 })); 12305 12306 return; 12307 } 12308 12309 /* Implementation details: 12310 * 12311 * Each register points to some region of memory, which we define as an 12312 * allocation. Each allocation may embed a bpf_spin_lock which protects any 12313 * special BPF objects (bpf_list_head, bpf_rb_root, etc.) part of the same 12314 * allocation. The lock and the data it protects are colocated in the same 12315 * memory region. 12316 * 12317 * Hence, everytime a register holds a pointer value pointing to such 12318 * allocation, the verifier preserves a unique reg->id for it. 12319 * 12320 * The verifier remembers the lock 'ptr' and the lock 'id' whenever 12321 * bpf_spin_lock is called. 12322 * 12323 * To enable this, lock state in the verifier captures two values: 12324 * active_lock.ptr = Register's type specific pointer 12325 * active_lock.id = A unique ID for each register pointer value 12326 * 12327 * Currently, PTR_TO_MAP_VALUE and PTR_TO_BTF_ID | MEM_ALLOC are the two 12328 * supported register types. 12329 * 12330 * The active_lock.ptr in case of map values is the reg->map_ptr, and in case of 12331 * allocated objects is the reg->btf pointer. 12332 * 12333 * The active_lock.id is non-unique for maps supporting direct_value_addr, as we 12334 * can establish the provenance of the map value statically for each distinct 12335 * lookup into such maps. They always contain a single map value hence unique 12336 * IDs for each pseudo load pessimizes the algorithm and rejects valid programs. 12337 * 12338 * So, in case of global variables, they use array maps with max_entries = 1, 12339 * hence their active_lock.ptr becomes map_ptr and id = 0 (since they all point 12340 * into the same map value as max_entries is 1, as described above). 12341 * 12342 * In case of inner map lookups, the inner map pointer has same map_ptr as the 12343 * outer map pointer (in verifier context), but each lookup into an inner map 12344 * assigns a fresh reg->id to the lookup, so while lookups into distinct inner 12345 * maps from the same outer map share the same map_ptr as active_lock.ptr, they 12346 * will get different reg->id assigned to each lookup, hence different 12347 * active_lock.id. 12348 * 12349 * In case of allocated objects, active_lock.ptr is the reg->btf, and the 12350 * reg->id is a unique ID preserved after the NULL pointer check on the pointer 12351 * returned from bpf_obj_new. Each allocation receives a new reg->id. 12352 */ 12353 static int check_reg_allocation_locked(struct bpf_verifier_env *env, struct bpf_reg_state *reg) 12354 { 12355 struct bpf_reference_state *s; 12356 void *ptr; 12357 u32 id; 12358 12359 switch ((int)reg->type) { 12360 case PTR_TO_MAP_VALUE: 12361 ptr = reg->map_ptr; 12362 break; 12363 case PTR_TO_BTF_ID | MEM_ALLOC: 12364 ptr = reg->btf; 12365 break; 12366 default: 12367 verifier_bug(env, "unknown reg type for lock check"); 12368 return -EFAULT; 12369 } 12370 id = reg->id; 12371 12372 if (!env->cur_state->active_locks) 12373 return -EINVAL; 12374 s = find_lock_state(env->cur_state, REF_TYPE_LOCK_MASK, id, ptr); 12375 if (!s) { 12376 verbose(env, "held lock and object are not in the same allocation\n"); 12377 return -EINVAL; 12378 } 12379 return 0; 12380 } 12381 12382 static bool is_bpf_list_api_kfunc(u32 btf_id) 12383 { 12384 return is_bpf_list_push_kfunc(btf_id) || 12385 btf_id == special_kfunc_list[KF_bpf_list_pop_front] || 12386 btf_id == special_kfunc_list[KF_bpf_list_pop_back] || 12387 btf_id == special_kfunc_list[KF_bpf_list_del] || 12388 btf_id == special_kfunc_list[KF_bpf_list_front] || 12389 btf_id == special_kfunc_list[KF_bpf_list_back] || 12390 btf_id == special_kfunc_list[KF_bpf_list_is_first] || 12391 btf_id == special_kfunc_list[KF_bpf_list_is_last] || 12392 btf_id == special_kfunc_list[KF_bpf_list_empty]; 12393 } 12394 12395 static bool is_bpf_rbtree_api_kfunc(u32 btf_id) 12396 { 12397 return is_bpf_rbtree_add_kfunc(btf_id) || 12398 btf_id == special_kfunc_list[KF_bpf_rbtree_remove] || 12399 btf_id == special_kfunc_list[KF_bpf_rbtree_first] || 12400 btf_id == special_kfunc_list[KF_bpf_rbtree_root] || 12401 btf_id == special_kfunc_list[KF_bpf_rbtree_left] || 12402 btf_id == special_kfunc_list[KF_bpf_rbtree_right]; 12403 } 12404 12405 static bool is_bpf_res_spin_lock_kfunc(u32 btf_id) 12406 { 12407 return btf_id == special_kfunc_list[KF_bpf_res_spin_lock] || 12408 btf_id == special_kfunc_list[KF_bpf_res_spin_unlock] || 12409 btf_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave] || 12410 btf_id == special_kfunc_list[KF_bpf_res_spin_unlock_irqrestore]; 12411 } 12412 12413 static bool kfunc_spin_allowed(struct bpf_verifier_env *env, s32 func_id, s16 offset) 12414 { 12415 struct bpf_kfunc_meta kfunc; 12416 int err; 12417 12418 err = fetch_kfunc_meta(env, func_id, offset, &kfunc); 12419 if (err || !kfunc.flags) 12420 return false; 12421 12422 return *kfunc.flags & KF_SPINLOCK_SAFE; 12423 } 12424 12425 static bool is_sync_callback_calling_kfunc(u32 btf_id) 12426 { 12427 return is_bpf_rbtree_add_kfunc(btf_id); 12428 } 12429 12430 static bool is_async_callback_calling_kfunc(u32 btf_id) 12431 { 12432 return is_bpf_wq_set_callback_kfunc(btf_id) || 12433 is_task_work_add_kfunc(btf_id); 12434 } 12435 12436 bool bpf_is_throw_kfunc(struct bpf_insn *insn) 12437 { 12438 return bpf_pseudo_kfunc_call(insn) && insn->off == 0 && 12439 insn->imm == special_kfunc_list[KF_bpf_throw]; 12440 } 12441 12442 static bool is_bpf_wq_set_callback_kfunc(u32 btf_id) 12443 { 12444 return btf_id == special_kfunc_list[KF_bpf_wq_set_callback]; 12445 } 12446 12447 static bool is_callback_calling_kfunc(u32 btf_id) 12448 { 12449 return is_sync_callback_calling_kfunc(btf_id) || 12450 is_async_callback_calling_kfunc(btf_id); 12451 } 12452 12453 static bool is_rbtree_lock_required_kfunc(u32 btf_id) 12454 { 12455 return is_bpf_rbtree_api_kfunc(btf_id); 12456 } 12457 12458 static bool check_kfunc_is_graph_root_api(struct bpf_verifier_env *env, 12459 enum btf_field_type head_field_type, 12460 u32 kfunc_btf_id) 12461 { 12462 bool ret; 12463 12464 switch (head_field_type) { 12465 case BPF_LIST_HEAD: 12466 ret = is_bpf_list_api_kfunc(kfunc_btf_id); 12467 break; 12468 case BPF_RB_ROOT: 12469 ret = is_bpf_rbtree_api_kfunc(kfunc_btf_id); 12470 break; 12471 default: 12472 verbose(env, "verifier internal error: unexpected graph root argument type %s\n", 12473 btf_field_type_name(head_field_type)); 12474 return false; 12475 } 12476 12477 if (!ret) 12478 verbose(env, "verifier internal error: %s head arg for unknown kfunc\n", 12479 btf_field_type_name(head_field_type)); 12480 return ret; 12481 } 12482 12483 static bool check_kfunc_is_graph_node_api(struct bpf_verifier_env *env, 12484 enum btf_field_type node_field_type, 12485 u32 kfunc_btf_id) 12486 { 12487 bool ret; 12488 12489 switch (node_field_type) { 12490 case BPF_LIST_NODE: 12491 ret = is_bpf_list_push_kfunc(kfunc_btf_id) || 12492 kfunc_btf_id == special_kfunc_list[KF_bpf_list_del] || 12493 kfunc_btf_id == special_kfunc_list[KF_bpf_list_is_first] || 12494 kfunc_btf_id == special_kfunc_list[KF_bpf_list_is_last]; 12495 break; 12496 case BPF_RB_NODE: 12497 ret = (is_bpf_rbtree_add_kfunc(kfunc_btf_id) || 12498 kfunc_btf_id == special_kfunc_list[KF_bpf_rbtree_remove] || 12499 kfunc_btf_id == special_kfunc_list[KF_bpf_rbtree_left] || 12500 kfunc_btf_id == special_kfunc_list[KF_bpf_rbtree_right]); 12501 break; 12502 default: 12503 verbose(env, "verifier internal error: unexpected graph node argument type %s\n", 12504 btf_field_type_name(node_field_type)); 12505 return false; 12506 } 12507 12508 if (!ret) 12509 verbose(env, "verifier internal error: %s node arg for unknown kfunc\n", 12510 btf_field_type_name(node_field_type)); 12511 return ret; 12512 } 12513 12514 static int 12515 __process_kf_arg_ptr_to_graph_root(struct bpf_verifier_env *env, 12516 struct bpf_reg_state *reg, argno_t argno, 12517 struct bpf_call_arg_meta *meta, 12518 enum btf_field_type head_field_type, 12519 struct btf_field **head_field) 12520 { 12521 const char *head_type_name; 12522 struct btf_field *field; 12523 struct btf_record *rec; 12524 u32 head_off; 12525 12526 if (meta->btf != btf_vmlinux) { 12527 verifier_bug(env, "unexpected btf mismatch in kfunc call"); 12528 return -EFAULT; 12529 } 12530 12531 if (!check_kfunc_is_graph_root_api(env, head_field_type, meta->func_id)) 12532 return -EFAULT; 12533 12534 head_type_name = btf_field_type_name(head_field_type); 12535 if (!tnum_is_const(reg->var_off)) { 12536 verbose(env, 12537 "%s doesn't have constant offset. %s has to be at the constant offset\n", 12538 reg_arg_name(env, argno), head_type_name); 12539 return -EINVAL; 12540 } 12541 12542 rec = reg_btf_record(reg); 12543 head_off = reg->var_off.value; 12544 field = btf_record_find(rec, head_off, head_field_type); 12545 if (!field) { 12546 verbose(env, "%s not found at offset=%u\n", head_type_name, head_off); 12547 return -EINVAL; 12548 } 12549 12550 /* All functions require bpf_list_head to be protected using a bpf_spin_lock */ 12551 if (check_reg_allocation_locked(env, reg)) { 12552 verbose(env, "bpf_spin_lock at off=%d must be held for %s\n", 12553 rec->spin_lock_off, head_type_name); 12554 return -EINVAL; 12555 } 12556 12557 if (*head_field) { 12558 verifier_bug(env, "repeating %s arg", head_type_name); 12559 return -EFAULT; 12560 } 12561 *head_field = field; 12562 return 0; 12563 } 12564 12565 static int process_kf_arg_ptr_to_list_head(struct bpf_verifier_env *env, 12566 struct bpf_reg_state *reg, argno_t argno, 12567 struct bpf_call_arg_meta *meta) 12568 { 12569 return __process_kf_arg_ptr_to_graph_root(env, reg, argno, meta, BPF_LIST_HEAD, 12570 &meta->arg_list_head.field); 12571 } 12572 12573 static int process_kf_arg_ptr_to_rbtree_root(struct bpf_verifier_env *env, 12574 struct bpf_reg_state *reg, argno_t argno, 12575 struct bpf_call_arg_meta *meta) 12576 { 12577 return __process_kf_arg_ptr_to_graph_root(env, reg, argno, meta, BPF_RB_ROOT, 12578 &meta->arg_rbtree_root.field); 12579 } 12580 12581 static int 12582 __process_kf_arg_ptr_to_graph_node(struct bpf_verifier_env *env, 12583 struct bpf_reg_state *reg, argno_t argno, 12584 struct bpf_call_arg_meta *meta, 12585 enum btf_field_type head_field_type, 12586 enum btf_field_type node_field_type, 12587 struct btf_field **node_field) 12588 { 12589 const char *node_type_name; 12590 const struct btf_type *et, *t; 12591 struct btf_field *field; 12592 u32 node_off; 12593 12594 if (meta->btf != btf_vmlinux) { 12595 verifier_bug(env, "unexpected btf mismatch in kfunc call"); 12596 return -EFAULT; 12597 } 12598 12599 if (!check_kfunc_is_graph_node_api(env, node_field_type, meta->func_id)) 12600 return -EFAULT; 12601 12602 node_type_name = btf_field_type_name(node_field_type); 12603 if (!tnum_is_const(reg->var_off)) { 12604 verbose(env, 12605 "%s doesn't have constant offset. %s has to be at the constant offset\n", 12606 reg_arg_name(env, argno), node_type_name); 12607 return -EINVAL; 12608 } 12609 12610 node_off = reg->var_off.value; 12611 field = reg_find_field_offset(reg, node_off, node_field_type); 12612 if (!field) { 12613 verbose(env, "%s not found at offset=%u\n", node_type_name, node_off); 12614 return -EINVAL; 12615 } 12616 12617 field = *node_field; 12618 12619 et = btf_type_by_id(field->graph_root.btf, field->graph_root.value_btf_id); 12620 t = btf_type_by_id(reg->btf, reg->btf_id); 12621 if (!btf_struct_ids_match(&env->log, reg->btf, reg->btf_id, 0, field->graph_root.btf, 12622 field->graph_root.value_btf_id, true, 12623 !type_is_alloc(reg->type))) { 12624 verbose(env, "operation on %s expects arg#1 %s at offset=%d " 12625 "in struct %s, but arg is at offset=%d in struct %s\n", 12626 btf_field_type_name(head_field_type), 12627 btf_field_type_name(node_field_type), 12628 field->graph_root.node_offset, 12629 btf_name_by_offset(field->graph_root.btf, et->name_off), 12630 node_off, btf_name_by_offset(reg->btf, t->name_off)); 12631 return -EINVAL; 12632 } 12633 meta->arg_btf = reg->btf; 12634 meta->arg_btf_id = reg->btf_id; 12635 12636 if (node_off != field->graph_root.node_offset) { 12637 verbose(env, "arg#1 offset=%d, but expected %s at offset=%d in struct %s\n", 12638 node_off, btf_field_type_name(node_field_type), 12639 field->graph_root.node_offset, 12640 btf_name_by_offset(field->graph_root.btf, et->name_off)); 12641 return -EINVAL; 12642 } 12643 12644 return 0; 12645 } 12646 12647 static int process_kf_arg_ptr_to_list_node(struct bpf_verifier_env *env, 12648 struct bpf_reg_state *reg, argno_t argno, 12649 struct bpf_call_arg_meta *meta) 12650 { 12651 return __process_kf_arg_ptr_to_graph_node(env, reg, argno, meta, 12652 BPF_LIST_HEAD, BPF_LIST_NODE, 12653 &meta->arg_list_head.field); 12654 } 12655 12656 static int process_kf_arg_ptr_to_rbtree_node(struct bpf_verifier_env *env, 12657 struct bpf_reg_state *reg, argno_t argno, 12658 struct bpf_call_arg_meta *meta) 12659 { 12660 return __process_kf_arg_ptr_to_graph_node(env, reg, argno, meta, 12661 BPF_RB_ROOT, BPF_RB_NODE, 12662 &meta->arg_rbtree_root.field); 12663 } 12664 12665 /* 12666 * css_task iter allowlist is needed to avoid dead locking on css_set_lock. 12667 * LSM hooks and iters (both sleepable and non-sleepable) are safe. 12668 * Any sleepable progs are also safe since bpf_check_attach_target() enforce 12669 * them can only be attached to some specific hook points. 12670 */ 12671 static bool check_css_task_iter_allowlist(struct bpf_verifier_env *env) 12672 { 12673 enum bpf_prog_type prog_type = resolve_prog_type(env->prog); 12674 12675 switch (prog_type) { 12676 case BPF_PROG_TYPE_LSM: 12677 return true; 12678 case BPF_PROG_TYPE_TRACING: 12679 if (env->prog->expected_attach_type == BPF_TRACE_ITER) 12680 return true; 12681 fallthrough; 12682 default: 12683 return in_sleepable(env); 12684 } 12685 } 12686 12687 static int check_kfunc_args(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta, 12688 int insn_idx) 12689 { 12690 const char *func_name = meta->func_name, *ref_tname; 12691 struct bpf_func_state *caller = cur_func(env); 12692 struct bpf_reg_state *regs = cur_regs(env); 12693 const struct btf *btf = meta->btf; 12694 const struct btf_param *args; 12695 struct btf_record *rec; 12696 u32 i, nargs; 12697 int ret; 12698 12699 args = (const struct btf_param *)(meta->func_proto + 1); 12700 nargs = btf_type_vlen(meta->func_proto); 12701 12702 ret = check_outgoing_stack_args(env, caller, nargs, func_name, btf, args); 12703 if (ret) 12704 return ret; 12705 12706 /* Check that BTF function arguments match actual types that the 12707 * verifier sees. 12708 */ 12709 for (i = 0; i < nargs; i++) { 12710 struct bpf_reg_state *reg = get_func_arg_reg(caller, regs, i); 12711 const struct btf_type *t, *ref_t, *resolve_ret; 12712 enum bpf_arg_type arg_type = ARG_DONTCARE; 12713 argno_t argno = argno_from_arg(i + 1); 12714 int regno = reg_from_argno(argno); 12715 bool btf_id_fixed_off_ok = true; 12716 u32 ref_id = args[i].type, type_size; 12717 int kf_arg_type = meta->fn->arg_type[i]; 12718 12719 if (is_kfunc_arg_prog_aux(btf, &args[i])) { 12720 /* Reject repeated use bpf_prog_aux */ 12721 if (meta->arg_prog) { 12722 verifier_bug(env, "Only 1 prog->aux argument supported per-kfunc"); 12723 return -EFAULT; 12724 } 12725 if (regno < 0) { 12726 verbose(env, "%s prog->aux cannot be a stack argument\n", 12727 reg_arg_name(env, argno)); 12728 return -EINVAL; 12729 } 12730 meta->arg_prog = true; 12731 cur_aux(env)->arg_prog = regno; 12732 continue; 12733 } 12734 12735 if (is_kfunc_arg_ignore(btf, &args[i]) || is_kfunc_arg_implicit(meta, i)) 12736 continue; 12737 12738 t = btf_type_skip_modifiers(btf, args[i].type, NULL); 12739 12740 if (btf_type_is_ptr(t)) { 12741 ref_t = btf_type_skip_modifiers(btf, t->type, &ref_id); 12742 ref_tname = btf_name_by_offset(btf, ref_t->name_off); 12743 } 12744 12745 if (btf_type_is_ptr(t) && 12746 (bpf_register_is_null(reg) || type_may_be_null(reg->type)) && 12747 !type_may_be_null(kf_arg_type)) { 12748 const char *expected_type; 12749 12750 expected_type = bpf_diag_fmt_btf_type(env, btf, args[i].type); 12751 verbose(env, "Possibly NULL pointer passed to trusted %s\n", 12752 reg_arg_name(env, argno)); 12753 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 12754 "Add a NULL check and call the kfunc only on the non-NULL path.", 12755 "the pointer may be NULL, but this kfunc requires a non-NULL value of type %s", 12756 expected_type); 12757 return -EACCES; 12758 } 12759 12760 if (regno == meta->release_regno && !is_kfunc_arg_dynptr(meta->btf, &args[i]) && 12761 !reg_is_referenced(env, reg) && !bpf_register_is_null(reg)) { 12762 const char *expected_type; 12763 12764 expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id); 12765 verbose(env, "release kfunc %s expects referenced PTR_TO_BTF_ID passed to %s\n", 12766 func_name, reg_arg_name(env, argno)); 12767 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 12768 "Pass the resource-owning pointer returned by the matching acquire kfunc, and avoid calling the release kfunc after ownership has already been transferred or released.", 12769 "release kfuncs require a resource-owning value of type %s returned by a matching acquire kfunc", 12770 expected_type); 12771 return -EINVAL; 12772 } 12773 12774 if (reg_is_referenced(env, reg)) 12775 update_ref_obj(&meta->ref_obj, reg); 12776 12777 if (bpf_register_is_null(reg) && type_may_be_null(kf_arg_type)) { 12778 ret = mark_arg_precision(env, argno); 12779 if (ret) 12780 return ret; 12781 continue; 12782 } 12783 12784 if (is_kfunc_arg_map(btf, &args[i])) { 12785 ref_id = *reg2btf_ids[CONST_PTR_TO_MAP]; 12786 ref_t = btf_type_by_id(btf_vmlinux, ref_id); 12787 ref_tname = btf_name_by_offset(btf, ref_t->name_off); 12788 } 12789 12790 switch (base_type(kf_arg_type)) { 12791 case KF_ARG_CONST: 12792 case KF_ARG_CONST_MEM_SIZE: 12793 case KF_ARG_MEM_SIZE: 12794 case KF_ARG_ANYTHING: 12795 case KF_ARG_CONST_ALLOC_SIZE_OR_ZERO: 12796 case KF_ARG_PTR_TO_ALLOC_BTF_ID: 12797 case KF_ARG_PTR_TO_BTF_ID: 12798 case KF_ARG_CONST_MAP_PTR: 12799 case KF_ARG_PTR_TO_ITER: 12800 case KF_ARG_PTR_TO_LIST_HEAD: 12801 case KF_ARG_PTR_TO_LIST_NODE: 12802 case KF_ARG_PTR_TO_RB_ROOT: 12803 case KF_ARG_PTR_TO_RB_NODE: 12804 case KF_ARG_PTR_TO_MEM: 12805 case KF_ARG_PTR_TO_CALLBACK: 12806 case KF_ARG_PTR_TO_CONST_STR: 12807 case KF_ARG_PTR_TO_WORKQUEUE: 12808 case KF_ARG_PTR_TO_TIMER: 12809 case KF_ARG_PTR_TO_TASK_WORK: 12810 case KF_ARG_PTR_TO_IRQ_FLAG: 12811 case KF_ARG_PTR_TO_RES_SPIN_LOCK: 12812 case KF_ARG_PTR_TO_ARENA: 12813 break; 12814 case KF_ARG_PTR_TO_DYNPTR: 12815 arg_type = ARG_PTR_TO_DYNPTR; 12816 break; 12817 case KF_ARG_PTR_TO_CTX: 12818 arg_type = ARG_PTR_TO_CTX; 12819 break; 12820 case KF_ARG_PTR_TO_REFCOUNTED_KPTR: 12821 arg_type = ARG_PTR_TO_BTF_ID; 12822 btf_id_fixed_off_ok = false; 12823 break; 12824 default: 12825 verifier_bug(env, "unknown kfunc arg type %d", kf_arg_type); 12826 return -EFAULT; 12827 } 12828 12829 if (regno == meta->release_regno) 12830 arg_type |= OBJ_RELEASE; 12831 ret = __check_func_arg_reg_off(env, reg, argno, arg_type, 12832 btf_id_fixed_off_ok); 12833 if (ret < 0) 12834 return ret; 12835 12836 switch (base_type(kf_arg_type)) { 12837 case KF_ARG_CONST: 12838 if (reg->type != SCALAR_VALUE) { 12839 verbose(env, "%s is not a scalar\n", reg_arg_name(env, argno)); 12840 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 12841 "Pass an integer scalar value for this argument, not a pointer or resource object.", 12842 "the kfunc expects an integer scalar, but %s is %s", 12843 reg_arg_name(env, argno), 12844 bpf_diag_reg_type_plain(env, reg->type)); 12845 return -EINVAL; 12846 } 12847 12848 ret = process_const_arg(env, reg, argno, meta); 12849 if (ret < 0) { 12850 if (ret == -EINVAL) 12851 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 12852 "Pass a compile-time constant or a value the verifier can prove is constant at this call.", 12853 "the kfunc requires this scalar argument to be a verifier-known constant, but %s is variable on this path", 12854 reg_arg_name(env, argno)); 12855 return ret; 12856 } 12857 break; 12858 case KF_ARG_ANYTHING: 12859 if (reg->type != SCALAR_VALUE) { 12860 verbose(env, "%s is not a scalar\n", reg_arg_name(env, argno)); 12861 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 12862 "Pass an integer scalar value for this argument, not a pointer or resource object.", 12863 "the kfunc expects an integer scalar, but %s is %s", 12864 reg_arg_name(env, argno), 12865 bpf_diag_reg_type_plain(env, reg->type)); 12866 return -EINVAL; 12867 } 12868 break; 12869 case KF_ARG_CONST_ALLOC_SIZE_OR_ZERO: 12870 if (reg->type != SCALAR_VALUE) { 12871 verbose(env, "%s is not a scalar\n", reg_arg_name(env, argno)); 12872 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 12873 "Pass an integer scalar value for this argument, not a pointer or resource object.", 12874 "the kfunc expects an integer scalar, but %s is %s", 12875 reg_arg_name(env, argno), 12876 bpf_diag_reg_type_plain(env, reg->type)); 12877 return -EINVAL; 12878 } 12879 12880 if (is_kfunc_arg_scalar_with_name(btf, &args[i], "rdonly_buf_size")) 12881 meta->r0_rdonly = true; 12882 ret = process_const_alloc_mem_size(env, reg, argno, &meta->ret_mem); 12883 if (ret < 0) { 12884 if (ret == -EINVAL) 12885 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 12886 "Pass a verifier-known constant size for this kfunc buffer argument.", 12887 "the kfunc uses this argument as a return-buffer size, but %s is invalid or variable on this path", 12888 reg_arg_name(env, argno)); 12889 return ret; 12890 } 12891 break; 12892 case KF_ARG_PTR_TO_CTX: 12893 if (reg->type != PTR_TO_CTX) { 12894 verbose(env, "%s expected pointer to ctx, but got %s\n", 12895 reg_arg_name(env, argno), reg_type_str(env, reg->type)); 12896 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 12897 "Pass the original program context pointer or preserve it before modifying registers.", 12898 "the kfunc expects a context pointer, but %s is %s", 12899 reg_arg_name(env, argno), 12900 bpf_diag_reg_type_plain(env, reg->type)); 12901 return -EINVAL; 12902 } 12903 12904 if (meta->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx]) { 12905 ret = get_kern_ctx_btf_id(&env->log, resolve_prog_type(env->prog)); 12906 if (ret < 0) 12907 return -EINVAL; 12908 meta->ret_btf_id = ret; 12909 } 12910 break; 12911 case KF_ARG_PTR_TO_ARENA: 12912 if (reg->type != PTR_TO_ARENA && reg->type != SCALAR_VALUE) { 12913 verbose(env, "%s is not a pointer to arena or scalar\n", 12914 reg_arg_name(env, argno)); 12915 return -EINVAL; 12916 } 12917 break; 12918 case KF_ARG_PTR_TO_ALLOC_BTF_ID: 12919 if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC)) { 12920 if (!is_bpf_obj_drop_kfunc(meta->func_id)) { 12921 verbose(env, "%s expected for bpf_obj_drop()\n", 12922 reg_arg_name(env, argno)); 12923 return -EINVAL; 12924 } 12925 } else if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC | MEM_PERCPU)) { 12926 if (!is_bpf_percpu_obj_drop_kfunc(meta->func_id)) { 12927 verbose(env, "%s expected for bpf_percpu_obj_drop()\n", 12928 reg_arg_name(env, argno)); 12929 return -EINVAL; 12930 } 12931 } else { 12932 verbose(env, "%s expected pointer to allocated object\n", 12933 reg_arg_name(env, argno)); 12934 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 12935 "Pass a pointer returned by the matching BPF object allocation path.", 12936 "the kfunc expects an allocated object pointer, but %s is %s", 12937 reg_arg_name(env, argno), 12938 bpf_diag_reg_type_plain(env, reg->type)); 12939 return -EINVAL; 12940 } 12941 if (!reg_is_referenced(env, reg)) { 12942 verbose(env, "allocated object must be referenced\n"); 12943 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 12944 "Pass the owned object pointer before it is released or transferred.", 12945 "the allocated object pointer in %s must still carry verifier-tracked ownership, but this pointer no longer owns a live resource", 12946 reg_arg_name(env, argno)); 12947 return -EINVAL; 12948 } 12949 if (meta->btf == btf_vmlinux) { 12950 meta->arg_btf = reg->btf; 12951 meta->arg_btf_id = reg->btf_id; 12952 } 12953 break; 12954 case KF_ARG_PTR_TO_DYNPTR: 12955 { 12956 enum bpf_arg_type dynptr_arg_type = ARG_PTR_TO_DYNPTR; 12957 12958 if (is_kfunc_arg_uninit(btf, &args[i])) 12959 dynptr_arg_type |= MEM_UNINIT; 12960 12961 if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_from_skb]) { 12962 dynptr_arg_type |= DYNPTR_TYPE_SKB; 12963 } else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_from_xdp]) { 12964 dynptr_arg_type |= DYNPTR_TYPE_XDP; 12965 } else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_from_skb_meta]) { 12966 dynptr_arg_type |= DYNPTR_TYPE_SKB_META; 12967 } else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_from_file]) { 12968 dynptr_arg_type |= DYNPTR_TYPE_FILE; 12969 } else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_file_discard]) { 12970 dynptr_arg_type |= DYNPTR_TYPE_FILE | OBJ_RELEASE; 12971 } else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_clone] && 12972 (dynptr_arg_type & MEM_UNINIT)) { 12973 enum bpf_dynptr_type parent_type = meta->dynptr.type; 12974 12975 if (parent_type == BPF_DYNPTR_TYPE_INVALID) { 12976 verifier_bug(env, "no dynptr type for parent of clone"); 12977 return -EFAULT; 12978 } 12979 12980 dynptr_arg_type |= (unsigned int)get_dynptr_type_flag(parent_type); 12981 } 12982 12983 ret = process_dynptr_func(env, reg, argno, insn_idx, func_name, 12984 dynptr_arg_type, &meta->ref_obj, &meta->dynptr); 12985 if (ret < 0) 12986 return ret; 12987 break; 12988 } 12989 case KF_ARG_PTR_TO_ITER: 12990 if (meta->func_id == special_kfunc_list[KF_bpf_iter_css_task_new]) { 12991 if (!check_css_task_iter_allowlist(env)) { 12992 verbose(env, "css_task_iter is only allowed in bpf_lsm, bpf_iter and sleepable progs\n"); 12993 return -EINVAL; 12994 } 12995 } 12996 ret = process_iter_arg(env, reg, argno, insn_idx, meta); 12997 if (ret < 0) 12998 return ret; 12999 break; 13000 case KF_ARG_PTR_TO_LIST_HEAD: 13001 if (reg->type != PTR_TO_MAP_VALUE && 13002 reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) { 13003 verbose(env, "%s expected pointer to map value or allocated object\n", 13004 reg_arg_name(env, argno)); 13005 return -EINVAL; 13006 } 13007 if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC) && 13008 !reg_is_referenced(env, reg)) { 13009 verbose(env, "allocated object must be referenced\n"); 13010 return -EINVAL; 13011 } 13012 ret = process_kf_arg_ptr_to_list_head(env, reg, argno, meta); 13013 if (ret < 0) 13014 return ret; 13015 break; 13016 case KF_ARG_PTR_TO_RB_ROOT: 13017 if (reg->type != PTR_TO_MAP_VALUE && 13018 reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) { 13019 verbose(env, "%s expected pointer to map value or allocated object\n", 13020 reg_arg_name(env, argno)); 13021 return -EINVAL; 13022 } 13023 if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC) && 13024 !reg_is_referenced(env, reg)) { 13025 verbose(env, "allocated object must be referenced\n"); 13026 return -EINVAL; 13027 } 13028 ret = process_kf_arg_ptr_to_rbtree_root(env, reg, argno, meta); 13029 if (ret < 0) 13030 return ret; 13031 break; 13032 case KF_ARG_PTR_TO_LIST_NODE: 13033 if (is_kfunc_arg_nonown_allowed(btf, &args[i]) && 13034 type_is_non_owning_ref(reg->type) && !reg_is_referenced(env, reg)) { 13035 /* Allow bpf_list_front/back return value for 13036 * __nonown_allowed list-node arguments. 13037 */ 13038 goto check_ok; 13039 } 13040 if (reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) { 13041 verbose(env, "%s expected pointer to allocated object\n", 13042 reg_arg_name(env, argno)); 13043 return -EINVAL; 13044 } 13045 if (!reg_is_referenced(env, reg)) { 13046 verbose(env, "allocated object must be referenced\n"); 13047 return -EINVAL; 13048 } 13049 check_ok: 13050 ret = process_kf_arg_ptr_to_list_node(env, reg, argno, meta); 13051 if (ret < 0) 13052 return ret; 13053 break; 13054 case KF_ARG_PTR_TO_RB_NODE: 13055 if (is_bpf_rbtree_add_kfunc(meta->func_id)) { 13056 if (reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) { 13057 verbose(env, "%s expected pointer to allocated object\n", 13058 reg_arg_name(env, argno)); 13059 return -EINVAL; 13060 } 13061 if (!reg_is_referenced(env, reg)) { 13062 verbose(env, "allocated object must be referenced\n"); 13063 return -EINVAL; 13064 } 13065 } else { 13066 if (!type_is_non_owning_ref(reg->type) && 13067 !reg_is_referenced(env, reg)) { 13068 verbose(env, "%s can only take non-owning or refcounted bpf_rb_node pointer\n", func_name); 13069 return -EINVAL; 13070 } 13071 if (in_rbtree_lock_required_cb(env)) { 13072 verbose(env, "%s not allowed in rbtree cb\n", func_name); 13073 return -EINVAL; 13074 } 13075 } 13076 13077 ret = process_kf_arg_ptr_to_rbtree_node(env, reg, argno, meta); 13078 if (ret < 0) 13079 return ret; 13080 break; 13081 case KF_ARG_CONST_MAP_PTR: 13082 if (base_type(reg->type) != CONST_PTR_TO_MAP || 13083 type_may_be_null(reg->type)) { 13084 verbose(env, "pointer in %s isn't map pointer\n", 13085 reg_arg_name(env, argno)); 13086 return -EINVAL; 13087 } 13088 ret = process_map_ptr_arg(env, reg, argno, meta); 13089 if (ret < 0) 13090 return ret; 13091 break; 13092 case KF_ARG_PTR_TO_BTF_ID: 13093 /* Only base_type is checked, further checks are done here */ 13094 if (base_type(reg->type) == PTR_TO_BTF_ID || 13095 reg2btf_ids[base_type(reg->type)]) { 13096 if (!is_trusted_reg(env, reg) || 13097 bpf_type_has_unsafe_modifiers(reg->type)) { 13098 if (!is_kfunc_rcu(meta)) { 13099 const char *expected_type; 13100 13101 expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id); 13102 verbose(env, "%s must be referenced or trusted\n", 13103 reg_arg_name(env, argno)); 13104 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 13105 "Pass a pointer acquired from a verifier-tracked source, or call this kfunc only inside the required protection if it accepts RCU pointers.", 13106 "the kfunc requires a trusted or resource-owning pointer to %s, but %s is %s", 13107 expected_type, 13108 reg_arg_name(env, argno), 13109 bpf_diag_reg_type_plain(env, reg->type)); 13110 return -EINVAL; 13111 } 13112 if (!is_rcu_reg(reg)) { 13113 const char *expected_type; 13114 13115 expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id); 13116 verbose(env, "%s must be a rcu pointer\n", 13117 reg_arg_name(env, argno)); 13118 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 13119 "Use this kfunc with a pointer that is valid in an RCU read lock region.", 13120 "the kfunc requires an RCU-protected pointer to %s, but %s is %s", 13121 expected_type, 13122 reg_arg_name(env, argno), 13123 bpf_diag_reg_type_plain(env, reg->type)); 13124 return -EINVAL; 13125 } 13126 } 13127 13128 ret = process_kf_arg_ptr_to_btf_id(env, reg, ref_t, ref_tname, ref_id, meta, i, argno); 13129 if (ret < 0) 13130 return ret; 13131 break; 13132 } 13133 13134 if (!__btf_type_is_scalar_struct(env, meta->btf, ref_t, 0)) { 13135 enum bpf_reg_type reg2btf_type = lookup_reg2btf_ids(ref_id); 13136 const char *expected_type; 13137 13138 verbose(env, "%s is %s expected %s %s", 13139 reg_arg_name(env, argno), reg_type_str(env, reg->type), 13140 btf_type_str(ref_t), ref_tname); 13141 if (reg2btf_type != NOT_INIT) 13142 verbose(env, " or %s", reg_type_str(env, reg2btf_type)); 13143 verbose(env, "\n"); 13144 expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id); 13145 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 13146 "Pass a verifier-tracked pointer to the expected kernel object type, not a pointer to stack storage or another memory buffer.", 13147 "the kfunc expects a pointer to %s, but this argument is %s and cannot be used as that kernel object pointer", 13148 expected_type, 13149 bpf_diag_reg_type_plain(env, reg->type)); 13150 return -EINVAL; 13151 } 13152 13153 /* 13154 * If the register does not contain btf id but the argument type is a pointer to 13155 * scalar-only struct, allow verifying it as a fixed size memory. 13156 */ 13157 kf_arg_type = KF_ARG_PTR_TO_MEM | MEM_FIXED_SIZE; 13158 fallthrough; 13159 case KF_ARG_PTR_TO_MEM: 13160 if (kf_arg_type & MEM_FIXED_SIZE) { 13161 bool known_memory; 13162 13163 resolve_ret = btf_resolve_size(btf, ref_t, &type_size); 13164 if (IS_ERR(resolve_ret)) { 13165 verbose(env, "%s reference type('%s %s') size cannot be determined: %ld\n", 13166 reg_arg_name(env, argno), btf_type_str(ref_t), 13167 ref_tname, PTR_ERR(resolve_ret)); 13168 return -EINVAL; 13169 } 13170 ret = check_mem_reg(env, reg, argno, type_size, BPF_READ | BPF_WRITE, 13171 meta, &known_memory); 13172 if (ret < 0) { 13173 const char *expected_type; 13174 13175 expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id); 13176 if (known_memory) 13177 bpf_diag_call_arg_fmt( 13178 env, insn_idx, argno, func_name, 13179 "Pass memory with at least the required number of accessible bytes and suitable read and write access.", 13180 "the kfunc expects %u bytes of memory for %s, but the verifier cannot prove that %s provides a readable and writable range of that size", 13181 type_size, expected_type, 13182 bpf_diag_reg_type_plain(env, reg->type)); 13183 else 13184 bpf_diag_call_arg_fmt( 13185 env, insn_idx, argno, func_name, 13186 "Pass stack, map, context, or other verifier-known memory of the expected type and size, not an integer cast to a pointer.", 13187 "the kfunc expects %u bytes of memory for %s, but it is %s and not verifier-known memory", 13188 type_size, expected_type, 13189 bpf_diag_reg_type_plain(env, reg->type)); 13190 return ret; 13191 } 13192 } 13193 break; 13194 case KF_ARG_CONST_MEM_SIZE: 13195 ret = process_const_arg(env, reg, argno, meta); 13196 if (ret < 0) { 13197 if (ret == -EINVAL) 13198 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 13199 "Pass a compile-time constant or a value the verifier can prove is constant at this call.", 13200 "the kfunc requires this memory size to be a verifier-known constant, but %s is variable on this path", 13201 reg_arg_name(env, argno)); 13202 return ret; 13203 } 13204 fallthrough; 13205 case KF_ARG_MEM_SIZE: 13206 { 13207 struct bpf_reg_state *buff_reg = get_func_arg_reg(caller, regs, i - 1); 13208 struct bpf_reg_state *size_reg = reg; 13209 argno_t buff_argno = argno_from_arg(i); 13210 enum bpf_mem_size_failure failure; 13211 13212 if (reg->type != SCALAR_VALUE) { 13213 verbose(env, "%s is not a scalar\n", reg_arg_name(env, argno)); 13214 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 13215 "Pass an integer scalar length for this memory argument.", 13216 "the kfunc expects a scalar memory size, but %s is %s", 13217 reg_arg_name(env, argno), 13218 bpf_diag_reg_type_plain(env, reg->type)); 13219 return -EINVAL; 13220 } 13221 13222 if (bpf_register_is_null(buff_reg)) 13223 break; 13224 13225 ret = check_mem_size_reg(env, buff_reg, size_reg, buff_argno, argno, 13226 BPF_READ | BPF_WRITE, true, meta, &failure); 13227 if (ret < 0) { 13228 const char *buff_arg, *size_arg; 13229 13230 buff_arg = bpf_diag_arg_name(env, buff_argno); 13231 size_arg = bpf_diag_arg_name(env, argno); 13232 verbose(env, "%s and ", reg_arg_name(env, buff_argno)); 13233 verbose(env, "%s memory, len pair leads to invalid memory access\n", 13234 reg_arg_name(env, argno)); 13235 if (failure == BPF_MEM_SIZE_FAIL_MEMORY) { 13236 bpf_diag_call_arg_fmt(env, insn_idx, buff_argno, func_name, 13237 "Pass a stack, map, context, or other verifier-known memory pointer, and keep the paired length within that object.", 13238 "it is the memory pointer in a memory/length pair with %s, but %s does not describe verifier-readable memory for the requested length", 13239 size_arg, buff_arg); 13240 } else if (failure == BPF_MEM_SIZE_FAIL_SIZE) { 13241 if (reg_smin(size_reg) < 0) 13242 bpf_diag_call_arg_fmt( 13243 env, insn_idx, argno, func_name, 13244 "Constrain the memory size to a non-negative value smaller than BPF_MAX_VAR_SIZ before this call.", 13245 "the memory size in %s may be negative because its signed minimum is %lld", 13246 size_arg, reg_smin(size_reg)); 13247 else 13248 bpf_diag_call_arg_fmt( 13249 env, insn_idx, argno, func_name, 13250 "Constrain the memory size to a non-negative value smaller than BPF_MAX_VAR_SIZ before this call.", 13251 "the memory size in %s may reach %llu bytes, but variable memory accesses must stay below %u bytes", 13252 size_arg, reg_umax(size_reg), BPF_MAX_VAR_SIZ); 13253 } 13254 return ret; 13255 } 13256 break; 13257 } 13258 case KF_ARG_PTR_TO_CALLBACK: 13259 if (reg->type != PTR_TO_FUNC) { 13260 verbose(env, "%s expected pointer to func\n", reg_arg_name(env, argno)); 13261 return -EINVAL; 13262 } 13263 meta->subprogno = reg->subprogno; 13264 break; 13265 case KF_ARG_PTR_TO_REFCOUNTED_KPTR: 13266 if (!type_is_ptr_alloc_obj(reg->type)) { 13267 verbose(env, "%s is neither owning or non-owning ref\n", 13268 reg_arg_name(env, argno)); 13269 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 13270 "Pass an owning or non-owning pointer to a BPF-managed object containing a bpf_refcount field.", 13271 "the kfunc expects a pointer to a BPF-managed refcounted object, but %s is %s", 13272 reg_arg_name(env, argno), 13273 bpf_diag_reg_type_plain(env, reg->type)); 13274 return -EINVAL; 13275 } 13276 if (!type_is_non_owning_ref(reg->type) && reg_is_referenced(env, reg)) 13277 meta->arg_owning_ref = true; 13278 13279 rec = reg_btf_record(reg); 13280 if (!rec) { 13281 verifier_bug(env, "Couldn't find btf_record"); 13282 return -EFAULT; 13283 } 13284 13285 if (rec->refcount_off < 0) { 13286 verbose(env, "%s doesn't point to a type with bpf_refcount field\n", 13287 reg_arg_name(env, argno)); 13288 return -EINVAL; 13289 } 13290 13291 meta->arg_btf = reg->btf; 13292 meta->arg_btf_id = reg->btf_id; 13293 break; 13294 case KF_ARG_PTR_TO_CONST_STR: 13295 if (reg->type != PTR_TO_MAP_VALUE) { 13296 verbose(env, "%s doesn't point to a const string\n", 13297 reg_arg_name(env, argno)); 13298 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 13299 "Pass a constant string pointer that the verifier recognizes, such as a string stored in a read-only map value.", 13300 "the kfunc expects a pointer to a constant string stored in verifier-known memory, but %s is %s", 13301 reg_arg_name(env, argno), 13302 bpf_diag_reg_type_plain(env, reg->type)); 13303 return -EINVAL; 13304 } 13305 ret = check_arg_const_str(env, reg, argno); 13306 if (ret) 13307 return ret; 13308 break; 13309 case KF_ARG_PTR_TO_WORKQUEUE: 13310 if (reg->type != PTR_TO_MAP_VALUE) { 13311 verbose(env, "%s doesn't point to a map value\n", 13312 reg_arg_name(env, argno)); 13313 return -EINVAL; 13314 } 13315 ret = check_map_field_pointer(env, reg, argno, BPF_WORKQUEUE, &meta->map); 13316 if (ret < 0) 13317 return ret; 13318 break; 13319 case KF_ARG_PTR_TO_TIMER: 13320 if (reg->type != PTR_TO_MAP_VALUE) { 13321 verbose(env, "%s doesn't point to a map value\n", 13322 reg_arg_name(env, argno)); 13323 return -EINVAL; 13324 } 13325 ret = process_timer_func(env, reg, argno, &meta->map); 13326 if (ret < 0) 13327 return ret; 13328 break; 13329 case KF_ARG_PTR_TO_TASK_WORK: 13330 if (reg->type != PTR_TO_MAP_VALUE) { 13331 verbose(env, "%s doesn't point to a map value\n", 13332 reg_arg_name(env, argno)); 13333 return -EINVAL; 13334 } 13335 ret = check_map_field_pointer(env, reg, argno, BPF_TASK_WORK, &meta->map); 13336 if (ret < 0) 13337 return ret; 13338 break; 13339 case KF_ARG_PTR_TO_IRQ_FLAG: 13340 if (reg->type != PTR_TO_STACK) { 13341 verbose(env, "%s doesn't point to an irq flag on stack\n", 13342 reg_arg_name(env, argno)); 13343 bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name, 13344 "Pass the same stack slot used by bpf_local_irq_save() or bpf_res_spin_lock_irqsave().", 13345 "the kfunc expects a stack pointer to an IRQ flag slot, but %s is %s", 13346 reg_arg_name(env, argno), 13347 bpf_diag_reg_type_plain(env, reg->type)); 13348 return -EINVAL; 13349 } 13350 ret = process_irq_flag(env, reg, argno, meta); 13351 if (ret < 0) 13352 return ret; 13353 break; 13354 case KF_ARG_PTR_TO_RES_SPIN_LOCK: 13355 { 13356 int flags = PROCESS_RES_LOCK; 13357 13358 if (in_rbtree_lock_required_cb(env)) { 13359 verbose(env, "can't res_spin_{lock,unlock} in rbtree cb\n"); 13360 return -EACCES; 13361 } 13362 13363 if (reg->type != PTR_TO_MAP_VALUE && reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) { 13364 verbose(env, "%s doesn't point to map value or allocated object\n", 13365 reg_arg_name(env, argno)); 13366 return -EINVAL; 13367 } 13368 13369 if (!is_bpf_res_spin_lock_kfunc(meta->func_id)) 13370 return -EFAULT; 13371 if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock] || 13372 meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave]) 13373 flags |= PROCESS_SPIN_LOCK; 13374 if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave] || 13375 meta->func_id == special_kfunc_list[KF_bpf_res_spin_unlock_irqrestore]) 13376 flags |= PROCESS_LOCK_IRQ; 13377 ret = process_spin_lock(env, reg, argno, flags); 13378 if (ret < 0) 13379 return ret; 13380 break; 13381 } 13382 } 13383 } 13384 13385 return 0; 13386 } 13387 13388 int bpf_fetch_kfunc_arg_meta(struct bpf_verifier_env *env, 13389 s32 func_id, 13390 s16 offset, 13391 struct bpf_call_arg_meta *meta) 13392 { 13393 struct bpf_kfunc_meta kfunc; 13394 int err; 13395 13396 memset(meta, 0, sizeof(*meta)); 13397 13398 err = fetch_kfunc_meta(env, func_id, offset, &kfunc); 13399 if (err) 13400 return err; 13401 13402 meta->btf = kfunc.btf; 13403 meta->func_id = kfunc.id; 13404 meta->func_proto = kfunc.proto; 13405 meta->func_name = kfunc.name; 13406 13407 if (!kfunc.flags || !btf_kfunc_is_allowed(kfunc.btf, kfunc.id, env->prog)) 13408 return -EACCES; 13409 13410 meta->kfunc_flags = *kfunc.flags; 13411 13412 /* Only support release referenced argument passed by register */ 13413 if (is_kfunc_release(meta)) 13414 meta->release_regno = BPF_REG_1; 13415 13416 return 0; 13417 } 13418 13419 /* 13420 * Determine how many bytes a helper accesses through a stack pointer at 13421 * argument position @arg (0-based, corresponding to R1-R5). 13422 * 13423 * Returns: 13424 * > 0 known read access size in bytes 13425 * 0 doesn't read anything directly 13426 * S64_MIN unknown 13427 * < 0 known write access of (-return) bytes 13428 */ 13429 s64 bpf_helper_stack_access_bytes(struct bpf_verifier_env *env, struct bpf_insn *insn, 13430 int arg, int insn_idx) 13431 { 13432 struct bpf_insn_aux_data *aux = &env->insn_aux_data[insn_idx]; 13433 const struct bpf_func_proto *fn; 13434 enum bpf_arg_type at; 13435 s64 size; 13436 13437 if (bpf_get_helper_proto(env, insn->imm, &fn) < 0) 13438 return S64_MIN; 13439 13440 at = fn->arg_type[arg]; 13441 13442 switch (base_type(at)) { 13443 case ARG_PTR_TO_MAP_KEY: 13444 case ARG_PTR_TO_MAP_VALUE: { 13445 bool is_key = base_type(at) == ARG_PTR_TO_MAP_KEY; 13446 u64 val; 13447 int i, map_reg; 13448 13449 for (i = 0; i < arg; i++) { 13450 if (base_type(fn->arg_type[i]) == ARG_CONST_MAP_PTR) 13451 break; 13452 } 13453 if (i >= arg) 13454 goto scan_all_maps; 13455 13456 map_reg = BPF_REG_1 + i; 13457 13458 if (!(aux->const_reg_map_mask & BIT(map_reg))) 13459 goto scan_all_maps; 13460 13461 i = aux->const_reg_vals[map_reg]; 13462 if (i < env->used_map_cnt) { 13463 size = is_key ? env->used_maps[i]->key_size 13464 : env->used_maps[i]->value_size; 13465 goto out; 13466 } 13467 scan_all_maps: 13468 /* 13469 * Map pointer is not known at this call site (e.g. different 13470 * maps on merged paths). Conservatively return the largest 13471 * key_size or value_size across all maps used by the program. 13472 */ 13473 val = 0; 13474 for (i = 0; i < env->used_map_cnt; i++) { 13475 struct bpf_map *map = env->used_maps[i]; 13476 u32 sz = is_key ? map->key_size : map->value_size; 13477 13478 if (sz > val) 13479 val = sz; 13480 if (map->inner_map_meta) { 13481 sz = is_key ? map->inner_map_meta->key_size 13482 : map->inner_map_meta->value_size; 13483 if (sz > val) 13484 val = sz; 13485 } 13486 } 13487 if (!val) 13488 return S64_MIN; 13489 size = val; 13490 goto out; 13491 } 13492 case ARG_PTR_TO_MEM: 13493 if (at & MEM_FIXED_SIZE) { 13494 size = fn->arg_size[arg]; 13495 goto out; 13496 } 13497 if (arg + 1 < ARRAY_SIZE(fn->arg_type) && 13498 arg_type_is_mem_size(fn->arg_type[arg + 1])) { 13499 int size_reg = BPF_REG_1 + arg + 1; 13500 13501 if (aux->const_reg_mask & BIT(size_reg)) { 13502 size = (s64)aux->const_reg_vals[size_reg]; 13503 goto out; 13504 } 13505 /* 13506 * Size arg is const on each path but differs across merged 13507 * paths. MAX_BPF_STACK is a safe upper bound for reads. 13508 */ 13509 if (at & MEM_UNINIT) 13510 return 0; 13511 return MAX_BPF_STACK; 13512 } 13513 return S64_MIN; 13514 case ARG_PTR_TO_DYNPTR: 13515 size = BPF_DYNPTR_SIZE; 13516 break; 13517 case ARG_PTR_TO_STACK: 13518 /* 13519 * Only used by bpf_calls_callback() helpers. The helper itself 13520 * doesn't access stack. The callback subprog does and it's 13521 * analyzed separately. 13522 */ 13523 return 0; 13524 default: 13525 return S64_MIN; 13526 } 13527 out: 13528 /* 13529 * MEM_UNINIT args are write-only: the helper initializes the 13530 * buffer without reading it. 13531 */ 13532 if (at & MEM_UNINIT) 13533 return -size; 13534 return size; 13535 } 13536 13537 /* 13538 * Determine how many bytes a kfunc accesses through a stack pointer at 13539 * argument position @arg (0-based, corresponding to R1-R5). 13540 * 13541 * Returns: 13542 * > 0 known read access size in bytes 13543 * 0 doesn't access memory through that argument (ex: not a pointer) 13544 * S64_MIN unknown 13545 * < 0 known write access of (-return) bytes 13546 */ 13547 s64 bpf_kfunc_stack_access_bytes(struct bpf_verifier_env *env, struct bpf_insn *insn, 13548 int arg, int insn_idx) 13549 { 13550 struct bpf_insn_aux_data *aux = &env->insn_aux_data[insn_idx]; 13551 struct bpf_call_arg_meta meta; 13552 const struct btf_param *args; 13553 const struct btf_type *t, *ref_t; 13554 const struct btf *btf; 13555 u32 nargs, type_size; 13556 s64 size; 13557 13558 if (bpf_fetch_kfunc_arg_meta(env, insn->imm, insn->off, &meta) < 0) 13559 return S64_MIN; 13560 13561 btf = meta.btf; 13562 args = btf_params(meta.func_proto); 13563 nargs = btf_type_vlen(meta.func_proto); 13564 if (arg >= nargs) 13565 return 0; 13566 13567 t = btf_type_skip_modifiers(btf, args[arg].type, NULL); 13568 if (!btf_type_is_ptr(t)) 13569 return 0; 13570 13571 /* dynptr: fixed 16-byte on-stack representation */ 13572 if (is_kfunc_arg_dynptr(btf, &args[arg])) { 13573 size = BPF_DYNPTR_SIZE; 13574 goto out; 13575 } 13576 13577 /* ptr + __sz/__szk pair: size is in the next register */ 13578 if (arg + 1 < nargs && 13579 (btf_param_match_suffix(btf, &args[arg + 1], "__sz") || 13580 btf_param_match_suffix(btf, &args[arg + 1], "__szk"))) { 13581 int size_reg = BPF_REG_1 + arg + 1; 13582 13583 if (aux->const_reg_mask & BIT(size_reg)) { 13584 size = (s64)aux->const_reg_vals[size_reg]; 13585 goto out; 13586 } 13587 return MAX_BPF_STACK; 13588 } 13589 13590 /* fixed-size pointed-to type: resolve via BTF */ 13591 ref_t = btf_type_skip_modifiers(btf, t->type, NULL); 13592 if (!IS_ERR(btf_resolve_size(btf, ref_t, &type_size))) { 13593 size = type_size; 13594 goto out; 13595 } 13596 13597 return S64_MIN; 13598 out: 13599 /* KF_ITER_NEW kfuncs initialize the iterator state at arg 0 */ 13600 if (arg == 0 && meta.kfunc_flags & KF_ITER_NEW) 13601 return -size; 13602 if (is_kfunc_arg_uninit(btf, &args[arg])) 13603 return -size; 13604 return size; 13605 } 13606 13607 /* check special kfuncs and return: 13608 * 1 - not fall-through to 'else' branch, continue verification 13609 * 0 - fall-through to 'else' branch 13610 * < 0 - not fall-through to 'else' branch, return error 13611 */ 13612 static int check_special_kfunc(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta, 13613 struct bpf_reg_state *regs, struct bpf_insn_aux_data *insn_aux, 13614 const struct btf_type *ptr_type, struct btf *desc_btf) 13615 { 13616 const struct btf_type *ret_t; 13617 int err = 0; 13618 13619 if (meta->btf != btf_vmlinux) 13620 return 0; 13621 13622 if (is_bpf_obj_new_kfunc(meta->func_id) || is_bpf_percpu_obj_new_kfunc(meta->func_id)) { 13623 struct btf_struct_meta *struct_meta; 13624 struct btf *ret_btf; 13625 u32 ret_btf_id; 13626 13627 if (is_bpf_obj_new_kfunc(meta->func_id) && !bpf_global_ma_set) 13628 return -ENOMEM; 13629 13630 if (((u64)(u32)meta->arg_constant.value) != meta->arg_constant.value) { 13631 verbose(env, "local type ID argument must be in range [0, U32_MAX]\n"); 13632 return -EINVAL; 13633 } 13634 13635 ret_btf = env->prog->aux->btf; 13636 ret_btf_id = meta->arg_constant.value; 13637 13638 /* This may be NULL due to user not supplying a BTF */ 13639 if (!ret_btf) { 13640 verbose(env, "bpf_obj_new/bpf_percpu_obj_new requires prog BTF\n"); 13641 return -EINVAL; 13642 } 13643 13644 ret_t = btf_type_by_id(ret_btf, ret_btf_id); 13645 if (!ret_t || !__btf_type_is_struct(ret_t)) { 13646 verbose(env, "bpf_obj_new/bpf_percpu_obj_new type ID argument must be of a struct\n"); 13647 return -EINVAL; 13648 } 13649 13650 if (is_bpf_percpu_obj_new_kfunc(meta->func_id)) { 13651 if (ret_t->size > BPF_GLOBAL_PERCPU_MA_MAX_SIZE) { 13652 verbose(env, "bpf_percpu_obj_new type size (%d) is greater than %d\n", 13653 ret_t->size, BPF_GLOBAL_PERCPU_MA_MAX_SIZE); 13654 return -EINVAL; 13655 } 13656 13657 if (!bpf_global_percpu_ma_set) { 13658 mutex_lock(&bpf_percpu_ma_lock); 13659 if (!bpf_global_percpu_ma_set) { 13660 /* Charge memory allocated with bpf_global_percpu_ma to 13661 * root memcg. The obj_cgroup for root memcg is NULL. 13662 */ 13663 err = bpf_mem_alloc_percpu_init(&bpf_global_percpu_ma, NULL); 13664 if (!err) 13665 bpf_global_percpu_ma_set = true; 13666 } 13667 mutex_unlock(&bpf_percpu_ma_lock); 13668 if (err) 13669 return err; 13670 } 13671 13672 mutex_lock(&bpf_percpu_ma_lock); 13673 err = bpf_mem_alloc_percpu_unit_init(&bpf_global_percpu_ma, ret_t->size); 13674 mutex_unlock(&bpf_percpu_ma_lock); 13675 if (err) 13676 return err; 13677 } 13678 13679 struct_meta = btf_find_struct_meta(ret_btf, ret_btf_id); 13680 if (is_bpf_percpu_obj_new_kfunc(meta->func_id)) { 13681 if (!__btf_type_is_scalar_struct(env, ret_btf, ret_t, 0)) { 13682 verbose(env, "bpf_percpu_obj_new type ID argument must be of a struct of scalars\n"); 13683 return -EINVAL; 13684 } 13685 13686 if (struct_meta) { 13687 verbose(env, "bpf_percpu_obj_new type ID argument must not contain special fields\n"); 13688 return -EINVAL; 13689 } 13690 } 13691 13692 mark_reg_known_zero(env, regs, BPF_REG_0); 13693 regs[BPF_REG_0].type = PTR_TO_BTF_ID | MEM_ALLOC; 13694 regs[BPF_REG_0].btf = ret_btf; 13695 regs[BPF_REG_0].btf_id = ret_btf_id; 13696 if (is_bpf_percpu_obj_new_kfunc(meta->func_id)) 13697 regs[BPF_REG_0].type |= MEM_PERCPU; 13698 13699 insn_aux->obj_new_size = ret_t->size; 13700 insn_aux->kptr_struct_meta = struct_meta; 13701 } else if (is_bpf_refcount_acquire_kfunc(meta->func_id)) { 13702 mark_reg_known_zero(env, regs, BPF_REG_0); 13703 regs[BPF_REG_0].type = PTR_TO_BTF_ID | MEM_ALLOC; 13704 regs[BPF_REG_0].btf = meta->arg_btf; 13705 regs[BPF_REG_0].btf_id = meta->arg_btf_id; 13706 13707 insn_aux->kptr_struct_meta = 13708 btf_find_struct_meta(meta->arg_btf, 13709 meta->arg_btf_id); 13710 } else if (is_list_node_type(ptr_type)) { 13711 struct btf_field *field = meta->arg_list_head.field; 13712 13713 mark_reg_graph_node(regs, BPF_REG_0, &field->graph_root); 13714 } else if (is_rbtree_node_type(ptr_type)) { 13715 struct btf_field *field = meta->arg_rbtree_root.field; 13716 13717 mark_reg_graph_node(regs, BPF_REG_0, &field->graph_root); 13718 } else if (meta->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx]) { 13719 mark_reg_known_zero(env, regs, BPF_REG_0); 13720 regs[BPF_REG_0].type = PTR_TO_BTF_ID | PTR_TRUSTED; 13721 regs[BPF_REG_0].btf = desc_btf; 13722 regs[BPF_REG_0].btf_id = meta->ret_btf_id; 13723 } else if (meta->func_id == special_kfunc_list[KF_bpf_rdonly_cast]) { 13724 ret_t = btf_type_by_id(desc_btf, meta->arg_constant.value); 13725 if (!ret_t) { 13726 verbose(env, "Unknown type ID %lld passed to kfunc bpf_rdonly_cast\n", 13727 meta->arg_constant.value); 13728 return -EINVAL; 13729 } else if (btf_type_is_struct(ret_t)) { 13730 mark_reg_known_zero(env, regs, BPF_REG_0); 13731 regs[BPF_REG_0].type = PTR_TO_BTF_ID | PTR_UNTRUSTED; 13732 regs[BPF_REG_0].btf = desc_btf; 13733 regs[BPF_REG_0].btf_id = meta->arg_constant.value; 13734 } else if (btf_type_is_void(ret_t)) { 13735 mark_reg_known_zero(env, regs, BPF_REG_0); 13736 regs[BPF_REG_0].type = PTR_TO_MEM | MEM_RDONLY | PTR_UNTRUSTED; 13737 regs[BPF_REG_0].mem_size = 0; 13738 } else { 13739 verbose(env, 13740 "kfunc bpf_rdonly_cast type ID argument must be of a struct or void\n"); 13741 return -EINVAL; 13742 } 13743 } else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_slice] || 13744 meta->func_id == special_kfunc_list[KF_bpf_dynptr_slice_rdwr]) { 13745 enum bpf_type_flag type_flag = get_dynptr_type_flag(meta->dynptr.type); 13746 13747 mark_reg_known_zero(env, regs, BPF_REG_0); 13748 13749 if (!meta->arg_constant.found) { 13750 verifier_bug(env, "bpf_dynptr_slice(_rdwr) no constant size"); 13751 return -EFAULT; 13752 } 13753 13754 regs[BPF_REG_0].mem_size = meta->arg_constant.value; 13755 13756 /* PTR_MAYBE_NULL will be added when is_kfunc_ret_null is checked */ 13757 regs[BPF_REG_0].type = PTR_TO_MEM | type_flag; 13758 13759 if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_slice]) { 13760 regs[BPF_REG_0].type |= MEM_RDONLY; 13761 } else { 13762 /* this will set env->seen_direct_write to true */ 13763 if (!may_access_direct_pkt_data(env, NULL, BPF_WRITE)) { 13764 verbose(env, "the prog does not allow writes to packet data\n"); 13765 return -EINVAL; 13766 } 13767 } 13768 13769 if (!meta->dynptr.id) { 13770 verifier_bug(env, "no dynptr id"); 13771 return -EFAULT; 13772 } 13773 regs[BPF_REG_0].parent_id = meta->dynptr.id; 13774 } else { 13775 return 0; 13776 } 13777 13778 return 1; 13779 } 13780 13781 static int check_return_code(struct bpf_verifier_env *env, int regno, const char *reg_name); 13782 13783 static int check_kfunc_call(struct bpf_verifier_env *env, struct bpf_insn *insn, 13784 int *insn_idx_p) 13785 { 13786 bool sleepable, rcu_lock, rcu_unlock, preempt_disable, preempt_enable; 13787 enum bpf_prog_type prog_type = resolve_prog_type(env->prog); 13788 struct bpf_reg_state *regs = cur_regs(env); 13789 const char *func_name, *ptr_type_name; 13790 const struct btf_type *t, *ptr_type; 13791 struct bpf_call_arg_meta meta; 13792 struct bpf_insn_aux_data *insn_aux; 13793 const char *operation; 13794 int err, insn_idx = *insn_idx_p; 13795 u32 i, nargs, ptr_type_id; 13796 struct bpf_kfunc_desc *desc; 13797 struct btf *desc_btf; 13798 int id; 13799 13800 /* skip for now, but return error when we find this in fixup_kfunc_call */ 13801 if (!insn->imm) 13802 return 0; 13803 13804 err = bpf_fetch_kfunc_arg_meta(env, insn->imm, insn->off, &meta); 13805 if (err == -EACCES && meta.func_name) { 13806 verbose(env, "calling kernel function %s is not allowed\n", meta.func_name); 13807 operation = bpf_diag_fmt(env, "kfunc %s", meta.func_name); 13808 bpf_diag_policy( 13809 env, insn_idx, operation, "this program cannot call the kfunc", 13810 "Use a kfunc allowed for this program type and attach point, or change the program context."); 13811 } 13812 if (err) 13813 return err; 13814 desc_btf = meta.btf; 13815 func_name = meta.func_name; 13816 insn_aux = &env->insn_aux_data[insn_idx]; 13817 13818 desc = find_kfunc_desc(env->prog, insn->imm, insn->off); 13819 if (!desc) { 13820 verifier_bug(env, "kfunc descriptor not found for func_id %u", insn->imm); 13821 return -EFAULT; 13822 } 13823 meta.fn = &desc->proto; 13824 13825 insn_aux->is_iter_next = bpf_is_iter_next_kfunc(&meta); 13826 13827 if (!insn->off && 13828 (insn->imm == special_kfunc_list[KF_bpf_res_spin_lock] || 13829 insn->imm == special_kfunc_list[KF_bpf_res_spin_lock_irqsave])) { 13830 struct bpf_verifier_state *branch; 13831 struct bpf_reg_state *regs; 13832 13833 branch = push_stack(env, env->insn_idx + 1, env->insn_idx, false); 13834 if (IS_ERR(branch)) { 13835 verbose(env, "failed to push state for failed lock acquisition\n"); 13836 return PTR_ERR(branch); 13837 } 13838 13839 regs = branch->frame[branch->curframe]->regs; 13840 13841 /* Clear r0-r5 registers in forked state */ 13842 for (i = 0; i < CALLER_SAVED_REGS; i++) 13843 bpf_mark_reg_not_init(env, ®s[caller_saved[i]]); 13844 13845 mark_reg_unknown(env, regs, BPF_REG_0); 13846 err = __mark_reg_s32_range(env, regs, BPF_REG_0, -MAX_ERRNO, -1); 13847 if (err) { 13848 verbose(env, "failed to mark s32 range for retval in forked state for lock\n"); 13849 return err; 13850 } 13851 } else if (!insn->off && insn->imm == special_kfunc_list[KF___bpf_trap]) { 13852 verbose(env, "unexpected __bpf_trap() due to uninitialized variable?\n"); 13853 return -EFAULT; 13854 } 13855 13856 if (is_kfunc_destructive(&meta) && !capable(CAP_SYS_BOOT)) { 13857 verbose(env, "destructive kfunc calls require CAP_SYS_BOOT capability\n"); 13858 operation = bpf_diag_fmt(env, "destructive kfunc %s", meta.func_name); 13859 bpf_diag_policy( 13860 env, insn_idx, operation, "destructive kfuncs require CAP_SYS_BOOT", 13861 "Load the program with CAP_SYS_BOOT, or avoid destructive kfuncs."); 13862 return -EACCES; 13863 } 13864 13865 if (is_kfunc_perfmon(&meta) && !env->allow_ptr_leaks) { 13866 verbose(env, "%s is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n", 13867 func_name); 13868 operation = bpf_diag_fmt(env, "kfunc %s", func_name); 13869 bpf_diag_policy(env, insn_idx, operation, "the kfunc requires CAP_PERFMON", 13870 "Load the program with CAP_PERFMON, or avoid the kfunc."); 13871 return -EPERM; 13872 } 13873 13874 sleepable = bpf_is_kfunc_sleepable(&meta); 13875 if (sleepable && !in_sleepable(env)) { 13876 verbose(env, "program must be sleepable to call sleepable kfunc %s\n", func_name); 13877 operation = bpf_diag_fmt(env, "sleepable kfunc %s", func_name); 13878 bpf_diag_ctx_forbidden(env, insn_idx, operation, 13879 "Mark the program sleepable if the program type allows it, or use a non-sleepable kfunc."); 13880 return -EACCES; 13881 } 13882 13883 /* Track non-sleepable context for kfuncs, same as for helpers. */ 13884 if (!in_sleepable_context(env)) 13885 insn_aux->non_sleepable = true; 13886 13887 /* Check the arguments */ 13888 err = check_kfunc_args(env, &meta, insn_idx); 13889 if (err < 0) 13890 return err; 13891 13892 if ((is_bpf_obj_drop_kfunc(meta.func_id) || 13893 is_bpf_percpu_obj_drop_kfunc(meta.func_id)) && (is_tracing_prog_type(prog_type) || 13894 /* is_tracing_prog_type() for now doesn't cover non-iterator tracing progs. */ 13895 (prog_type == BPF_PROG_TYPE_TRACING && env->prog->expected_attach_type != BPF_TRACE_ITER 13896 && !env->prog->sleepable))) { 13897 struct btf_struct_meta *struct_meta; 13898 13899 struct_meta = btf_find_struct_meta(meta.arg_btf, meta.arg_btf_id); 13900 if (struct_meta && btf_record_has_nmi_unsafe_fields(struct_meta->record)) { 13901 verbose(env, "%s cannot be used in tracing programs on types with NMI unsafe fields\n", 13902 func_name); 13903 return -EINVAL; 13904 } 13905 } 13906 13907 if (is_bpf_rbtree_add_kfunc(meta.func_id)) { 13908 err = push_callback_call(env, insn, insn_idx, meta.subprogno, 13909 set_rbtree_add_callback_state); 13910 if (err) { 13911 verbose(env, "kfunc %s#%d failed callback verification\n", 13912 func_name, meta.func_id); 13913 return err; 13914 } 13915 } 13916 13917 if (is_bpf_wq_set_callback_kfunc(meta.func_id)) { 13918 err = push_callback_call(env, insn, insn_idx, meta.subprogno, 13919 set_timer_callback_state); 13920 if (err) { 13921 verbose(env, "kfunc %s#%d failed callback verification\n", 13922 func_name, meta.func_id); 13923 return err; 13924 } 13925 } 13926 13927 if (is_task_work_add_kfunc(meta.func_id)) { 13928 err = push_callback_call(env, insn, insn_idx, meta.subprogno, 13929 set_task_work_schedule_callback_state); 13930 if (err) { 13931 verbose(env, "kfunc %s#%d failed callback verification\n", 13932 func_name, meta.func_id); 13933 return err; 13934 } 13935 } 13936 13937 rcu_lock = is_kfunc_bpf_rcu_read_lock(&meta); 13938 rcu_unlock = is_kfunc_bpf_rcu_read_unlock(&meta); 13939 13940 preempt_disable = is_kfunc_bpf_preempt_disable(&meta); 13941 preempt_enable = is_kfunc_bpf_preempt_enable(&meta); 13942 13943 if (rcu_lock) { 13944 env->cur_state->active_rcu_locks++; 13945 bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_RCU, true, 13946 env->cur_state->active_rcu_locks); 13947 } else if (rcu_unlock) { 13948 if (env->cur_state->active_rcu_locks == 0) { 13949 verbose(env, "unmatched rcu read unlock (kernel function %s)\n", func_name); 13950 bpf_diag_ctx_underflow( 13951 env, insn_idx, func_name, BPF_DIAG_CONTEXT_RCU, 13952 "Remove the extra bpf_rcu_read_unlock() call, or ensure this path first enters an RCU read lock region."); 13953 return -EINVAL; 13954 } 13955 env->cur_state->active_rcu_locks--; 13956 bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_RCU, false, 13957 env->cur_state->active_rcu_locks); 13958 if (!in_rcu_cs(env)) 13959 invalidate_rcu_protected_refs(env); 13960 } else if (preempt_disable) { 13961 env->cur_state->active_preempt_locks++; 13962 bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_PREEMPT, true, 13963 env->cur_state->active_preempt_locks); 13964 } else if (preempt_enable) { 13965 if (env->cur_state->active_preempt_locks == 0) { 13966 verbose(env, "unmatched attempt to enable preemption (kernel function %s)\n", func_name); 13967 bpf_diag_ctx_underflow( 13968 env, insn_idx, func_name, BPF_DIAG_CONTEXT_PREEMPT, 13969 "Remove the extra bpf_preempt_enable() call, or ensure this path first disables preemption."); 13970 return -EINVAL; 13971 } 13972 env->cur_state->active_preempt_locks--; 13973 bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_PREEMPT, false, 13974 env->cur_state->active_preempt_locks); 13975 if (!in_rcu_cs(env)) 13976 invalidate_rcu_protected_refs(env); 13977 } 13978 13979 if (sleepable && !in_sleepable_context(env)) { 13980 verbose(env, "kernel func %s is sleepable within %s\n", 13981 func_name, non_sleepable_context_description(env)); 13982 operation = bpf_diag_fmt(env, "sleepable kfunc %s", func_name); 13983 bpf_diag_ctx_forbidden(env, insn_idx, operation, 13984 "Move the kfunc call outside the critical section, or use a non-sleepable kfunc."); 13985 return -EACCES; 13986 } 13987 13988 if (in_rbtree_lock_required_cb(env) && (rcu_lock || rcu_unlock)) { 13989 verbose(env, "Calling bpf_rcu_read_{lock,unlock} in unnecessary rbtree callback\n"); 13990 return -EACCES; 13991 } 13992 13993 if (is_kfunc_rcu_protected(&meta) && !in_rcu_cs(env)) { 13994 verbose(env, "kernel func %s requires RCU critical section protection\n", func_name); 13995 bpf_diag_ctx_required( 13996 env, insn_idx, func_name, BPF_DIAG_CONTEXT_RCU, 13997 "Call this kfunc between bpf_rcu_read_lock() and bpf_rcu_read_unlock(), keeping all exit paths balanced."); 13998 return -EACCES; 13999 } 14000 14001 /* In case of release function, we get register number of refcounted 14002 * PTR_TO_BTF_ID in bpf_kfunc_arg_meta, do the release now. 14003 */ 14004 if (meta.release_regno) { 14005 err = release_reg(env, ®s[meta.release_regno], false, !!meta.dynptr.id); 14006 if (err) 14007 return err; 14008 } 14009 14010 if (is_bpf_list_push_kfunc(meta.func_id) || is_bpf_rbtree_add_kfunc(meta.func_id)) { 14011 id = regs[BPF_REG_2].id; 14012 insn_aux->insert_off = regs[BPF_REG_2].var_off.value; 14013 insn_aux->kptr_struct_meta = btf_find_struct_meta(meta.arg_btf, meta.arg_btf_id); 14014 ref_convert_owning_non_owning(env, id); 14015 } 14016 14017 if (meta.func_id == special_kfunc_list[KF_bpf_throw]) { 14018 if (!bpf_jit_supports_exceptions()) { 14019 verbose(env, "JIT does not support calling kfunc %s#%d\n", 14020 func_name, meta.func_id); 14021 return -ENOTSUPP; 14022 } 14023 env->seen_exception = true; 14024 14025 /* In the case of the default callback, the cookie value passed 14026 * to bpf_throw becomes the return value of the program. 14027 */ 14028 if (!env->exception_callback_subprog) { 14029 err = check_return_code(env, BPF_REG_1, "R1"); 14030 if (err < 0) 14031 return err; 14032 } 14033 } 14034 14035 bpf_diag_record_caller_saved(env, regs); 14036 bpf_diag_mod_begin(env, ®s[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE); 14037 for (i = 0; i < CALLER_SAVED_REGS; i++) { 14038 u32 regno = caller_saved[i]; 14039 14040 bpf_mark_reg_not_init(env, ®s[regno]); 14041 } 14042 invalidate_outgoing_stack_args(env, cur_func(env)); 14043 14044 /* Check return type */ 14045 t = btf_type_skip_modifiers(desc_btf, meta.func_proto->type, NULL); 14046 14047 if (is_kfunc_acquire(&meta) && !btf_type_is_struct_ptr(meta.btf, t)) { 14048 if (meta.btf != btf_vmlinux || 14049 (!is_bpf_obj_new_kfunc(meta.func_id) && 14050 !is_bpf_percpu_obj_new_kfunc(meta.func_id) && 14051 !is_bpf_refcount_acquire_kfunc(meta.func_id))) { 14052 verbose(env, "acquire kernel function does not return PTR_TO_BTF_ID\n"); 14053 return -EINVAL; 14054 } 14055 } 14056 14057 if (btf_type_is_scalar(t)) { 14058 mark_reg_unknown(env, regs, BPF_REG_0); 14059 if (meta.btf == btf_vmlinux && (meta.func_id == special_kfunc_list[KF_bpf_res_spin_lock] || 14060 meta.func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave])) 14061 __mark_reg_const_zero(env, ®s[BPF_REG_0]); 14062 } else if (btf_type_is_ptr(t)) { 14063 ptr_type = btf_type_skip_modifiers(desc_btf, t->type, &ptr_type_id); 14064 err = check_special_kfunc(env, &meta, regs, insn_aux, ptr_type, desc_btf); 14065 if (err) { 14066 if (err < 0) 14067 return err; 14068 } else if (btf_type_is_void(ptr_type)) { 14069 /* kfunc returning 'void *' is equivalent to returning scalar */ 14070 mark_reg_unknown(env, regs, BPF_REG_0); 14071 } else if (!__btf_type_is_struct(ptr_type)) { 14072 if (!meta.ret_mem.found) { 14073 __u32 sz; 14074 14075 if (!IS_ERR(btf_resolve_size(desc_btf, ptr_type, &sz))) { 14076 meta.ret_mem.found = true; 14077 meta.ret_mem.size = sz; 14078 meta.r0_rdonly = true; 14079 } 14080 14081 if (meta.func_id == special_kfunc_list[KF_bpf_session_cookie]) 14082 meta.r0_rdonly = false; 14083 } 14084 if (!meta.ret_mem.found) { 14085 ptr_type_name = btf_name_by_offset(desc_btf, 14086 ptr_type->name_off); 14087 verbose(env, 14088 "kernel function %s returns pointer type %s %s is not supported\n", 14089 func_name, 14090 btf_type_str(ptr_type), 14091 ptr_type_name); 14092 return -EINVAL; 14093 } 14094 14095 mark_reg_known_zero(env, regs, BPF_REG_0); 14096 regs[BPF_REG_0].type = PTR_TO_MEM; 14097 regs[BPF_REG_0].mem_size = meta.ret_mem.size; 14098 14099 if (meta.r0_rdonly) 14100 regs[BPF_REG_0].type |= MEM_RDONLY; 14101 14102 /* Ensures we don't access the memory after a release_reference() */ 14103 if (meta.ref_obj.id) { 14104 err = validate_ref_obj(env, &meta.ref_obj); 14105 if (err) 14106 return err; 14107 regs[BPF_REG_0].parent_id = meta.ref_obj.id; 14108 } 14109 14110 if (is_kfunc_rcu_protected(&meta)) 14111 regs[BPF_REG_0].type |= MEM_RCU; 14112 } else { 14113 enum bpf_reg_type type = PTR_TO_BTF_ID; 14114 14115 if (meta.func_id == special_kfunc_list[KF_bpf_get_kmem_cache]) 14116 type |= PTR_UNTRUSTED; 14117 else if (is_kfunc_rcu_protected(&meta) || 14118 (bpf_is_iter_next_kfunc(&meta) && 14119 (get_iter_from_state(env->cur_state, &meta) 14120 ->type & MEM_RCU))) { 14121 /* 14122 * If the iterator's constructor (the _new 14123 * function e.g., bpf_iter_task_new) has been 14124 * annotated with BPF kfunc flag 14125 * KF_RCU_PROTECTED and was called within a RCU 14126 * read-side critical section, also propagate 14127 * the MEM_RCU flag to the pointer returned from 14128 * the iterator's next function (e.g., 14129 * bpf_iter_task_next). 14130 */ 14131 type |= MEM_RCU; 14132 } else { 14133 /* 14134 * Any PTR_TO_BTF_ID that is returned from a BPF 14135 * kfunc should by default be treated as 14136 * implicitly trusted. 14137 */ 14138 type |= PTR_TRUSTED; 14139 } 14140 14141 mark_reg_known_zero(env, regs, BPF_REG_0); 14142 regs[BPF_REG_0].btf = desc_btf; 14143 regs[BPF_REG_0].type = type; 14144 regs[BPF_REG_0].btf_id = ptr_type_id; 14145 } 14146 14147 if (is_kfunc_ret_null(&meta)) { 14148 regs[BPF_REG_0].type |= PTR_MAYBE_NULL; 14149 /* For mark_ptr_or_null_reg, see 93c230e3f5bd6 */ 14150 regs[BPF_REG_0].id = ++env->id_gen; 14151 } 14152 if (is_kfunc_acquire(&meta)) { 14153 id = acquire_reference(env, insn_idx, 0); 14154 if (id < 0) 14155 return id; 14156 regs[BPF_REG_0].id = id; 14157 } else if (is_rbtree_node_type(ptr_type) || is_list_node_type(ptr_type)) { 14158 ref_set_non_owning(env, ®s[BPF_REG_0]); 14159 } 14160 14161 if (reg_may_point_to_spin_lock(®s[BPF_REG_0]) && !regs[BPF_REG_0].id) 14162 regs[BPF_REG_0].id = ++env->id_gen; 14163 } else if (btf_type_is_void(t)) { 14164 if (meta.btf == btf_vmlinux) { 14165 if (is_bpf_obj_drop_kfunc(meta.func_id) || 14166 is_bpf_percpu_obj_drop_kfunc(meta.func_id)) { 14167 insn_aux->kptr_struct_meta = 14168 btf_find_struct_meta(meta.arg_btf, 14169 meta.arg_btf_id); 14170 } 14171 } 14172 } 14173 14174 if (bpf_is_kfunc_pkt_changing(&meta)) 14175 clear_all_pkt_pointers(env); 14176 14177 nargs = btf_type_vlen(meta.func_proto); 14178 if (nargs > MAX_BPF_FUNC_REG_ARGS) { 14179 struct bpf_func_state *caller = cur_func(env); 14180 struct bpf_subprog_info *caller_info = &env->subprog_info[caller->subprogno]; 14181 u16 out_stack_arg_cnt = nargs - MAX_BPF_FUNC_REG_ARGS; 14182 u16 stack_arg_cnt = bpf_in_stack_arg_cnt(caller_info) + out_stack_arg_cnt; 14183 14184 if (stack_arg_cnt > caller_info->stack_arg_cnt) 14185 caller_info->stack_arg_cnt = stack_arg_cnt; 14186 } 14187 14188 /* 14189 * Record R0 before process_iter_next_call() snapshots the alternate 14190 * iterator path's diagnostic position. 14191 */ 14192 bpf_diag_mod_end(env); 14193 14194 if (bpf_is_iter_next_kfunc(&meta)) { 14195 err = process_iter_next_call(env, insn_idx, &meta); 14196 if (err) 14197 return err; 14198 } 14199 14200 if (meta.func_id == special_kfunc_list[KF_bpf_session_cookie]) 14201 env->prog->call_session_cookie = true; 14202 14203 if (bpf_is_throw_kfunc(insn)) 14204 return process_bpf_exit_full(env, NULL, true); 14205 14206 return 0; 14207 } 14208 14209 static bool check_reg_sane_offset_scalar(struct bpf_verifier_env *env, 14210 const struct bpf_reg_state *reg, 14211 enum bpf_reg_type type) 14212 { 14213 bool known = tnum_is_const(reg->var_off); 14214 s64 val = reg->var_off.value; 14215 s64 smin = reg_smin(reg); 14216 14217 if (known && (val >= BPF_MAX_VAR_OFF || val <= -BPF_MAX_VAR_OFF)) { 14218 verbose(env, "math between %s pointer and %lld is not allowed\n", 14219 reg_type_str(env, type), val); 14220 return false; 14221 } 14222 14223 if (smin == S64_MIN) { 14224 verbose(env, "math between %s pointer and register with unbounded min value is not allowed\n", 14225 reg_type_str(env, type)); 14226 return false; 14227 } 14228 14229 if (smin >= BPF_MAX_VAR_OFF || smin <= -BPF_MAX_VAR_OFF) { 14230 verbose(env, "value %lld makes %s pointer be out of bounds\n", 14231 smin, reg_type_str(env, type)); 14232 return false; 14233 } 14234 14235 return true; 14236 } 14237 14238 static bool check_reg_sane_offset_ptr(struct bpf_verifier_env *env, 14239 const struct bpf_reg_state *reg, 14240 enum bpf_reg_type type) 14241 { 14242 bool known = tnum_is_const(reg->var_off); 14243 s64 val = reg->var_off.value; 14244 s64 smin = reg_smin(reg); 14245 14246 if (known && (val >= BPF_MAX_VAR_OFF || val <= -BPF_MAX_VAR_OFF)) { 14247 verbose(env, "%s pointer offset %lld is not allowed\n", 14248 reg_type_str(env, type), val); 14249 return false; 14250 } 14251 14252 if (smin >= BPF_MAX_VAR_OFF || smin <= -BPF_MAX_VAR_OFF) { 14253 verbose(env, "%s pointer offset %lld is not allowed\n", 14254 reg_type_str(env, type), smin); 14255 return false; 14256 } 14257 14258 return true; 14259 } 14260 14261 enum { 14262 REASON_BOUNDS = -1, 14263 REASON_TYPE = -2, 14264 REASON_PATHS = -3, 14265 REASON_LIMIT = -4, 14266 REASON_STACK = -5, 14267 }; 14268 14269 static int retrieve_ptr_limit(const struct bpf_reg_state *ptr_reg, 14270 u32 *alu_limit, bool mask_to_left) 14271 { 14272 u32 max = 0, ptr_limit = 0; 14273 14274 switch (ptr_reg->type) { 14275 case PTR_TO_STACK: 14276 /* Offset 0 is out-of-bounds, but acceptable start for the 14277 * left direction, see BPF_REG_FP. Also, unknown scalar 14278 * offset where we would need to deal with min/max bounds is 14279 * currently prohibited for unprivileged. 14280 */ 14281 max = MAX_BPF_STACK + mask_to_left; 14282 ptr_limit = -ptr_reg->var_off.value; 14283 break; 14284 case PTR_TO_MAP_VALUE: 14285 max = ptr_reg->map_ptr->value_size; 14286 ptr_limit = mask_to_left ? reg_smin(ptr_reg) : reg_umax(ptr_reg); 14287 break; 14288 default: 14289 return REASON_TYPE; 14290 } 14291 14292 if (ptr_limit >= max) 14293 return REASON_LIMIT; 14294 *alu_limit = ptr_limit; 14295 return 0; 14296 } 14297 14298 static bool can_skip_alu_sanitation(const struct bpf_verifier_env *env, 14299 const struct bpf_insn *insn) 14300 { 14301 return env->bypass_spec_v1 || 14302 BPF_SRC(insn->code) == BPF_K || 14303 cur_aux(env)->nospec; 14304 } 14305 14306 static int update_alu_sanitation_state(struct bpf_insn_aux_data *aux, 14307 u32 alu_state, u32 alu_limit) 14308 { 14309 /* If we arrived here from different branches with different 14310 * state or limits to sanitize, then this won't work. 14311 */ 14312 if (aux->alu_state && 14313 (aux->alu_state != alu_state || 14314 aux->alu_limit != alu_limit)) 14315 return REASON_PATHS; 14316 14317 /* Corresponding fixup done in do_misc_fixups(). */ 14318 aux->alu_state = alu_state; 14319 aux->alu_limit = alu_limit; 14320 return 0; 14321 } 14322 14323 static int sanitize_val_alu(struct bpf_verifier_env *env, 14324 struct bpf_insn *insn) 14325 { 14326 struct bpf_insn_aux_data *aux = cur_aux(env); 14327 14328 if (can_skip_alu_sanitation(env, insn)) 14329 return 0; 14330 14331 return update_alu_sanitation_state(aux, BPF_ALU_NON_POINTER, 0); 14332 } 14333 14334 static bool sanitize_needed(u8 opcode) 14335 { 14336 return opcode == BPF_ADD || opcode == BPF_SUB; 14337 } 14338 14339 struct bpf_sanitize_info { 14340 struct bpf_insn_aux_data aux; 14341 bool mask_to_left; 14342 }; 14343 14344 static int sanitize_speculative_path(struct bpf_verifier_env *env, 14345 const struct bpf_insn *insn, 14346 u32 next_idx, u32 curr_idx) 14347 { 14348 struct bpf_verifier_state *branch; 14349 struct bpf_reg_state *regs; 14350 14351 branch = push_stack(env, next_idx, curr_idx, true); 14352 if (!IS_ERR(branch) && insn) { 14353 regs = branch->frame[branch->curframe]->regs; 14354 if (BPF_SRC(insn->code) == BPF_K) { 14355 mark_reg_unknown(env, regs, insn->dst_reg); 14356 } else if (BPF_SRC(insn->code) == BPF_X) { 14357 mark_reg_unknown(env, regs, insn->dst_reg); 14358 mark_reg_unknown(env, regs, insn->src_reg); 14359 } 14360 } 14361 return PTR_ERR_OR_ZERO(branch); 14362 } 14363 14364 static int sanitize_ptr_alu(struct bpf_verifier_env *env, 14365 struct bpf_insn *insn, 14366 const struct bpf_reg_state *ptr_reg, 14367 const struct bpf_reg_state *off_reg, 14368 struct bpf_reg_state *dst_reg, 14369 struct bpf_sanitize_info *info, 14370 const bool commit_window) 14371 { 14372 struct bpf_insn_aux_data *aux = commit_window ? cur_aux(env) : &info->aux; 14373 struct bpf_verifier_state *vstate = env->cur_state; 14374 bool off_is_imm = tnum_is_const(off_reg->var_off); 14375 bool off_is_neg = reg_smin(off_reg) < 0; 14376 bool ptr_is_dst_reg = ptr_reg == dst_reg; 14377 u8 opcode = BPF_OP(insn->code); 14378 u32 alu_state, alu_limit; 14379 struct bpf_reg_state tmp; 14380 int err; 14381 14382 if (can_skip_alu_sanitation(env, insn)) 14383 return 0; 14384 14385 /* We already marked aux for masking from non-speculative 14386 * paths, thus we got here in the first place. We only care 14387 * to explore bad access from here. 14388 */ 14389 if (vstate->speculative) 14390 goto do_sim; 14391 14392 if (!commit_window) { 14393 if (!tnum_is_const(off_reg->var_off) && 14394 (reg_smin(off_reg) < 0) != (reg_smax(off_reg) < 0)) 14395 return REASON_BOUNDS; 14396 14397 info->mask_to_left = (opcode == BPF_ADD && off_is_neg) || 14398 (opcode == BPF_SUB && !off_is_neg); 14399 } 14400 14401 err = retrieve_ptr_limit(ptr_reg, &alu_limit, info->mask_to_left); 14402 if (err < 0) 14403 return err; 14404 14405 if (commit_window) { 14406 /* In commit phase we narrow the masking window based on 14407 * the observed pointer move after the simulated operation. 14408 */ 14409 alu_state = info->aux.alu_state; 14410 alu_limit = abs(info->aux.alu_limit - alu_limit); 14411 } else { 14412 alu_state = off_is_neg ? BPF_ALU_NEG_VALUE : 0; 14413 alu_state |= off_is_imm ? BPF_ALU_IMMEDIATE : 0; 14414 alu_state |= ptr_is_dst_reg ? 14415 BPF_ALU_SANITIZE_SRC : BPF_ALU_SANITIZE_DST; 14416 14417 /* Limit pruning on unknown scalars to enable deep search for 14418 * potential masking differences from other program paths. 14419 */ 14420 if (!off_is_imm) 14421 env->explore_alu_limits = true; 14422 } 14423 14424 err = update_alu_sanitation_state(aux, alu_state, alu_limit); 14425 if (err < 0) 14426 return err; 14427 do_sim: 14428 /* If we're in commit phase, we're done here given we already 14429 * pushed the truncated dst_reg into the speculative verification 14430 * stack. 14431 * 14432 * Also, when register is a known constant, we rewrite register-based 14433 * operation to immediate-based, and thus do not need masking (and as 14434 * a consequence, do not need to simulate the zero-truncation either). 14435 */ 14436 if (commit_window || off_is_imm) 14437 return 0; 14438 14439 /* Simulate and find potential out-of-bounds access under 14440 * speculative execution from truncation as a result of 14441 * masking when off was not within expected range. If off 14442 * sits in dst, then we temporarily need to move ptr there 14443 * to simulate dst (== 0) +/-= ptr. Needed, for example, 14444 * for cases where we use K-based arithmetic in one direction 14445 * and truncated reg-based in the other in order to explore 14446 * bad access. 14447 */ 14448 if (!ptr_is_dst_reg) { 14449 tmp = *dst_reg; 14450 *dst_reg = *ptr_reg; 14451 } 14452 err = sanitize_speculative_path(env, NULL, env->insn_idx + 1, env->insn_idx); 14453 if (err < 0) 14454 return REASON_STACK; 14455 if (!ptr_is_dst_reg) 14456 *dst_reg = tmp; 14457 return 0; 14458 } 14459 14460 static void sanitize_mark_insn_seen(struct bpf_verifier_env *env) 14461 { 14462 struct bpf_verifier_state *vstate = env->cur_state; 14463 14464 /* If we simulate paths under speculation, we don't update the 14465 * insn as 'seen' such that when we verify unreachable paths in 14466 * the non-speculative domain, sanitize_dead_code() can still 14467 * rewrite/sanitize them. 14468 */ 14469 if (!vstate->speculative) 14470 env->insn_aux_data[env->insn_idx].seen = env->pass_cnt; 14471 } 14472 14473 static int sanitize_err(struct bpf_verifier_env *env, const struct bpf_insn *insn, int reason) 14474 { 14475 static const char *err = "pointer arithmetic with it prohibited for !root"; 14476 const char *op = BPF_OP(insn->code) == BPF_ADD ? "add" : "sub"; 14477 u32 dst = insn->dst_reg, src = insn->src_reg; 14478 struct bpf_reg_state *regs = cur_regs(env); 14479 14480 switch (reason) { 14481 case REASON_BOUNDS: 14482 verbose(env, "R%d has unknown scalar with mixed signed bounds, %s\n", 14483 regs[src].type == SCALAR_VALUE ? src : dst, err); 14484 break; 14485 case REASON_TYPE: 14486 verbose(env, "R%d has pointer with unsupported alu operation, %s\n", 14487 regs[src].type == SCALAR_VALUE ? dst : src, err); 14488 break; 14489 case REASON_PATHS: 14490 verbose(env, "R%d tried to %s from different maps, paths or scalars, %s\n", 14491 dst, op, err); 14492 break; 14493 case REASON_LIMIT: 14494 verbose(env, "R%d tried to %s beyond pointer bounds, %s\n", 14495 dst, op, err); 14496 break; 14497 case REASON_STACK: 14498 verbose(env, "R%d could not be pushed for speculative verification, %s\n", 14499 dst, err); 14500 return -ENOMEM; 14501 default: 14502 verifier_bug(env, "unknown reason (%d)", reason); 14503 break; 14504 } 14505 14506 return -EACCES; 14507 } 14508 14509 /* check that stack access falls within stack limits and that 'reg' doesn't 14510 * have a variable offset. 14511 * 14512 * Variable offset is prohibited for unprivileged mode for simplicity since it 14513 * requires corresponding support in Spectre masking for stack ALU. See also 14514 * retrieve_ptr_limit(). 14515 */ 14516 static int check_stack_access_for_ptr_arithmetic( 14517 struct bpf_verifier_env *env, 14518 int regno, 14519 const struct bpf_reg_state *reg, 14520 int off) 14521 { 14522 if (!tnum_is_const(reg->var_off)) { 14523 char tn_buf[48]; 14524 14525 tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off); 14526 verbose(env, "R%d variable stack access prohibited for !root, var_off=%s off=%d\n", 14527 regno, tn_buf, off); 14528 return -EACCES; 14529 } 14530 14531 if (off >= 0 || off < -MAX_BPF_STACK) { 14532 verbose(env, "R%d stack pointer arithmetic goes out of range, " 14533 "prohibited for !root; off=%d\n", regno, off); 14534 return -EACCES; 14535 } 14536 14537 return 0; 14538 } 14539 14540 static int sanitize_check_bounds(struct bpf_verifier_env *env, 14541 const struct bpf_insn *insn, 14542 struct bpf_reg_state *dst_reg) 14543 { 14544 u32 dst = insn->dst_reg; 14545 14546 /* For unprivileged we require that resulting offset must be in bounds 14547 * in order to be able to sanitize access later on. 14548 */ 14549 if (env->bypass_spec_v1) 14550 return 0; 14551 14552 switch (dst_reg->type) { 14553 case PTR_TO_STACK: 14554 if (check_stack_access_for_ptr_arithmetic(env, dst, dst_reg, 14555 dst_reg->var_off.value)) 14556 return -EACCES; 14557 break; 14558 case PTR_TO_MAP_VALUE: 14559 if (check_map_access(env, dst_reg, argno_from_reg(dst), 0, 1, false, ACCESS_HELPER)) { 14560 verbose(env, "R%d pointer arithmetic of map value goes out of range, " 14561 "prohibited for !root\n", dst); 14562 return -EACCES; 14563 } 14564 break; 14565 default: 14566 return -EOPNOTSUPP; 14567 } 14568 14569 return 0; 14570 } 14571 14572 /* Handles arithmetic on a pointer and a scalar: computes new min/max and var_off. 14573 * Caller should also handle BPF_MOV case separately. 14574 * If we return -EACCES, caller may want to try again treating pointer as a 14575 * scalar. So we only emit a diagnostic if !env->allow_ptr_leaks. 14576 */ 14577 static int adjust_ptr_min_max_vals(struct bpf_verifier_env *env, struct bpf_insn *insn, 14578 u32 ptr_regno, const struct bpf_reg_state *ptr_reg, 14579 const struct bpf_reg_state *off_reg) 14580 { 14581 struct bpf_verifier_state *vstate = env->cur_state; 14582 struct bpf_func_state *state = vstate->frame[vstate->curframe]; 14583 struct bpf_reg_state *regs = state->regs, *dst_reg; 14584 bool known = tnum_is_const(off_reg->var_off); 14585 s64 smin_val = reg_smin(off_reg), smax_val = reg_smax(off_reg); 14586 u64 umin_val = reg_umin(off_reg), umax_val = reg_umax(off_reg); 14587 struct bpf_sanitize_info info = {}; 14588 u8 opcode = BPF_OP(insn->code); 14589 u32 dst = insn->dst_reg; 14590 const char *reason; 14591 int ret, bounds_ret; 14592 14593 dst_reg = ®s[dst]; 14594 14595 if ((known && (smin_val != smax_val || umin_val != umax_val)) || 14596 smin_val > smax_val || umin_val > umax_val) { 14597 /* Taint dst register if offset had invalid bounds derived from 14598 * e.g. dead branches. 14599 */ 14600 __mark_reg_unknown(env, dst_reg); 14601 return 0; 14602 } 14603 14604 if (BPF_CLASS(insn->code) != BPF_ALU64) { 14605 /* 32-bit ALU ops on pointers produce (meaningless) scalars */ 14606 if (opcode == BPF_SUB && env->allow_ptr_leaks) { 14607 __mark_reg_unknown(env, dst_reg); 14608 return 0; 14609 } 14610 14611 verbose(env, 14612 "R%d 32-bit pointer arithmetic prohibited\n", 14613 dst); 14614 reason = bpf_diag_fmt( 14615 env, "R%d holds %s. 32-bit ALU operations on pointers discard pointer tracking, so the verifier cannot keep the result as a safe pointer.", 14616 ptr_regno, bpf_diag_reg_type_plain(env, ptr_reg->type)); 14617 bpf_diag_register_type( 14618 env, env->insn_idx, ptr_regno, "32-bit pointer arithmetic", reason, 14619 "Use a 64-bit ALU instruction with an allowed, bounded scalar offset."); 14620 return -EACCES; 14621 } 14622 14623 if (ptr_reg->type & PTR_MAYBE_NULL) { 14624 verbose(env, "R%d pointer arithmetic on %s prohibited, null-check it first\n", 14625 dst, reg_type_str(env, ptr_reg->type)); 14626 reason = bpf_diag_fmt( 14627 env, "R%d may be NULL (%s). Pointer arithmetic is allowed only after the program proves the pointer is non-NULL on this path.", 14628 ptr_regno, reg_type_str(env, ptr_reg->type)); 14629 bpf_diag_register_type( 14630 env, env->insn_idx, ptr_regno, "pointer arithmetic before NULL check", reason, 14631 "Make sure that a NULL check precedes any arithmetic performed on the pointer."); 14632 return -EACCES; 14633 } 14634 14635 switch (base_type(ptr_reg->type)) { 14636 case PTR_TO_CTX: 14637 case PTR_TO_MAP_VALUE: 14638 case PTR_TO_MAP_KEY: 14639 case PTR_TO_STACK: 14640 case PTR_TO_PACKET_META: 14641 case PTR_TO_PACKET: 14642 case PTR_TO_TP_BUFFER: 14643 case PTR_TO_BTF_ID: 14644 case PTR_TO_MEM: 14645 case PTR_TO_BUF: 14646 case PTR_TO_FUNC: 14647 case CONST_PTR_TO_DYNPTR: 14648 break; 14649 case PTR_TO_FLOW_KEYS: 14650 if (known) 14651 break; 14652 fallthrough; 14653 case CONST_PTR_TO_MAP: 14654 /* smin_val represents the known value */ 14655 if (known && smin_val == 0 && opcode == BPF_ADD) 14656 break; 14657 fallthrough; 14658 default: 14659 verbose(env, "R%d pointer arithmetic on %s prohibited\n", 14660 dst, reg_type_str(env, ptr_reg->type)); 14661 reason = bpf_diag_fmt( 14662 env, "R%d holds %s. This pointer kind does not allow offset arithmetic.", 14663 ptr_regno, bpf_diag_reg_type_plain(env, ptr_reg->type)); 14664 bpf_diag_register_type( 14665 env, env->insn_idx, ptr_regno, "pointer arithmetic is not allowed", reason, 14666 "Do not change this pointer's offset; use it only in operations accepted for its kind."); 14667 return -EACCES; 14668 } 14669 14670 /* For 'scalar += pointer', dst_reg inherits the complete pointer 14671 * register state. Individual fields may be adjusted later by pointer 14672 * arithmetic. Callers guarantee that below does not overwrite off_reg. 14673 */ 14674 if (dst_reg != ptr_reg) 14675 *dst_reg = *ptr_reg; 14676 14677 /* 14678 * Accesses to untrusted PTR_TO_MEM are done through probe 14679 * instructions, hence no need to track offsets. 14680 */ 14681 if (base_type(ptr_reg->type) == PTR_TO_MEM && (ptr_reg->type & PTR_UNTRUSTED)) 14682 return 0; 14683 14684 if (!check_reg_sane_offset_scalar(env, off_reg, ptr_reg->type)) { 14685 reason = bpf_diag_fmt( 14686 env, "The scalar offset used with R%d is unbounded or outside the verifier's safe pointer-offset range [-%u, %u].", 14687 ptr_regno, BPF_MAX_VAR_OFF, BPF_MAX_VAR_OFF); 14688 bpf_diag_register_type( 14689 env, env->insn_idx, ptr_regno, "pointer offset is not safe", reason, 14690 "Clamp or bounds-check the scalar offset before applying it to the pointer."); 14691 return -EINVAL; 14692 } 14693 if (!check_reg_sane_offset_ptr(env, ptr_reg, ptr_reg->type)) { 14694 reason = bpf_diag_fmt( 14695 env, "R%d already has an offset outside the verifier's safe range [-%u, %u] for %s.", 14696 ptr_regno, BPF_MAX_VAR_OFF, BPF_MAX_VAR_OFF, 14697 bpf_diag_reg_type_plain(env, ptr_reg->type)); 14698 bpf_diag_register_type( 14699 env, env->insn_idx, ptr_regno, "pointer offset is not safe", reason, 14700 "Keep the base pointer within the verifier's allowed offset range before applying more arithmetic."); 14701 return -EINVAL; 14702 } 14703 14704 if (sanitize_needed(opcode)) { 14705 ret = sanitize_ptr_alu(env, insn, ptr_reg, off_reg, dst_reg, 14706 &info, false); 14707 if (ret < 0) 14708 return sanitize_err(env, insn, ret); 14709 } 14710 14711 /* 14712 * Pointer types do not carry 32-bit bounds at the moment. Blank r32 14713 * only after sanitize_ptr_alu() may have snapshotted dst_reg into a 14714 * speculative path: otherwise reg_bounds_sanity_check() might hit some 14715 * constraints violations. 14716 */ 14717 __mark_reg32_unbounded(dst_reg); 14718 14719 switch (opcode) { 14720 case BPF_ADD: 14721 /* 14722 * dst_reg gets the pointer type and since some positive 14723 * integer value was added to the pointer, give it a new 'id' 14724 * if it's a PTR_TO_PACKET. 14725 * this creates a new 'base' pointer, off_reg (variable) gets 14726 * added into the variable offset, and we copy the fixed offset 14727 * from ptr_reg. 14728 */ 14729 dst_reg->r64 = cnum64_add(ptr_reg->r64, off_reg->r64); 14730 dst_reg->var_off = tnum_add(ptr_reg->var_off, off_reg->var_off); 14731 dst_reg->raw = ptr_reg->raw; 14732 if (reg_is_pkt_pointer(ptr_reg)) { 14733 if (!known) 14734 dst_reg->id = ++env->id_gen; 14735 /* 14736 * Clear range for unknown addends since we can't know 14737 * where the pkt pointer ended up. Also clear AT_PKT_END / 14738 * BEYOND_PKT_END from prior comparison as any pointer 14739 * arithmetic invalidates them. 14740 */ 14741 if (!known || dst_reg->range < 0) 14742 memset(&dst_reg->raw, 0, sizeof(dst_reg->raw)); 14743 } 14744 break; 14745 case BPF_SUB: 14746 if (dst_reg != ptr_reg) { 14747 /* scalar -= pointer. Creates an unknown scalar */ 14748 verbose(env, "R%d tried to subtract pointer from scalar\n", 14749 dst); 14750 reason = bpf_diag_fmt( 14751 env, "This operation subtracts pointer register R%d from scalar register R%d. " 14752 "The verifier only tracks pointer-minus-scalar arithmetic for allowed pointer types.", 14753 ptr_regno, dst); 14754 bpf_diag_register_type( 14755 env, env->insn_idx, ptr_regno, "pointer subtracted from scalar", reason, 14756 "Keep the pointer as the base; only add or subtract bounded scalars when permitted."); 14757 return -EACCES; 14758 } 14759 /* We don't allow subtraction from FP, because (according to 14760 * test_verifier.c test "invalid fp arithmetic", JITs might not 14761 * be able to deal with it. 14762 */ 14763 if (ptr_reg->type == PTR_TO_STACK) { 14764 verbose(env, "R%d subtraction from stack pointer prohibited\n", 14765 dst); 14766 reason = bpf_diag_fmt( 14767 env, "R%d is a stack pointer. The verifier does not allow BPF_SUB to move stack pointers.", 14768 ptr_regno); 14769 bpf_diag_register_type( 14770 env, env->insn_idx, ptr_regno, "subtraction from stack pointer", reason, 14771 "Use addition from R10 to form stack addresses within the tracked stack frame."); 14772 return -EACCES; 14773 } 14774 dst_reg->r64 = cnum64_add(ptr_reg->r64, cnum64_negate(off_reg->r64)); 14775 dst_reg->var_off = tnum_sub(ptr_reg->var_off, off_reg->var_off); 14776 dst_reg->raw = ptr_reg->raw; 14777 if (reg_is_pkt_pointer(ptr_reg)) { 14778 if (!known) 14779 dst_reg->id = ++env->id_gen; 14780 /* 14781 * Clear range if the subtrahend may be negative since 14782 * pkt pointer could move past its bounds. A positive 14783 * subtrahend moves it backwards keeping positive range 14784 * intact. Also clear AT_PKT_END / BEYOND_PKT_END from 14785 * prior comparison as arithmetic invalidates them. 14786 */ 14787 if ((!known && smin_val < 0) || dst_reg->range < 0) 14788 memset(&dst_reg->raw, 0, sizeof(dst_reg->raw)); 14789 } 14790 break; 14791 case BPF_AND: 14792 case BPF_OR: 14793 case BPF_XOR: 14794 /* bitwise ops on pointers are troublesome, prohibit. */ 14795 verbose(env, "R%d bitwise operator %s on pointer prohibited\n", 14796 dst, bpf_alu_string[opcode >> 4]); 14797 reason = bpf_diag_fmt( 14798 env, "R%d holds %s. Bitwise operator %s would destroy the pointer value the verifier is tracking.", 14799 ptr_regno, bpf_diag_reg_type_plain(env, ptr_reg->type), 14800 bpf_alu_string[opcode >> 4]); 14801 bpf_diag_register_type( 14802 env, env->insn_idx, ptr_regno, "bitwise operation on pointer", reason, 14803 "Do bitwise operations on scalar values, not on pointer-valued registers."); 14804 return -EACCES; 14805 default: 14806 /* other operators (e.g. MUL,LSH) produce non-pointer results */ 14807 verbose(env, "R%d pointer arithmetic with %s operator prohibited\n", 14808 dst, bpf_alu_string[opcode >> 4]); 14809 reason = bpf_diag_fmt( 14810 env, "R%d holds %s. Operator %s is not one of the limited pointer arithmetic operations the verifier can track.", 14811 ptr_regno, bpf_diag_reg_type_plain(env, ptr_reg->type), 14812 bpf_alu_string[opcode >> 4]); 14813 bpf_diag_register_type( 14814 env, env->insn_idx, ptr_regno, "invalid pointer arithmetic operator", reason, 14815 "Use only verifier-supported addition or subtraction with a bounded scalar offset, or perform this operation on a scalar value."); 14816 return -EACCES; 14817 } 14818 14819 if (!check_reg_sane_offset_ptr(env, dst_reg, ptr_reg->type)) { 14820 reason = bpf_diag_fmt( 14821 env, "After this arithmetic, R%d would be outside the verifier's safe offset range [-%u, %u] for %s.", 14822 dst, BPF_MAX_VAR_OFF, BPF_MAX_VAR_OFF, 14823 bpf_diag_reg_type_plain(env, ptr_reg->type)); 14824 bpf_diag_register_type( 14825 env, env->insn_idx, ptr_regno, "pointer offset is not safe", reason, 14826 "Tighten the scalar bounds before the arithmetic so the resulting pointer remains within the allowed range."); 14827 return -EINVAL; 14828 } 14829 /* 14830 * A packet pointer that keeps its id or range is checked against a 14831 * range set from the checked pointer's umax, so var_off must not tighten 14832 * its umax. r32 must still match var_off for reg_bounds_sanity_check(). 14833 */ 14834 if (reg_is_pkt_pointer(dst_reg) && (known || dst_reg->range > 0)) 14835 __update_reg32_bounds(dst_reg); 14836 else 14837 reg_bounds_sync(dst_reg); 14838 bounds_ret = sanitize_check_bounds(env, insn, dst_reg); 14839 if (bounds_ret == -EACCES) 14840 return bounds_ret; 14841 if (sanitize_needed(opcode)) { 14842 ret = sanitize_ptr_alu(env, insn, dst_reg, off_reg, dst_reg, 14843 &info, true); 14844 if (verifier_bug_if(!can_skip_alu_sanitation(env, insn) 14845 && !env->cur_state->speculative 14846 && bounds_ret 14847 && !ret, 14848 env, "Pointer type unsupported by sanitize_check_bounds() not rejected by retrieve_ptr_limit() as required")) { 14849 return -EFAULT; 14850 } 14851 if (ret < 0) 14852 return sanitize_err(env, insn, ret); 14853 } 14854 14855 return 0; 14856 } 14857 14858 static void scalar32_min_max_add(struct bpf_reg_state *dst_reg, 14859 struct bpf_reg_state *src_reg) 14860 { 14861 dst_reg->r32 = cnum32_add(dst_reg->r32, src_reg->r32); 14862 } 14863 14864 static void scalar_min_max_add(struct bpf_reg_state *dst_reg, 14865 struct bpf_reg_state *src_reg) 14866 { 14867 dst_reg->r64 = cnum64_add(dst_reg->r64, src_reg->r64); 14868 } 14869 14870 static void scalar32_min_max_sub(struct bpf_reg_state *dst_reg, 14871 struct bpf_reg_state *src_reg) 14872 { 14873 dst_reg->r32 = cnum32_add(dst_reg->r32, cnum32_negate(src_reg->r32)); 14874 } 14875 14876 static void scalar_min_max_sub(struct bpf_reg_state *dst_reg, 14877 struct bpf_reg_state *src_reg) 14878 { 14879 dst_reg->r64 = cnum64_add(dst_reg->r64, cnum64_negate(src_reg->r64)); 14880 } 14881 14882 static void scalar32_min_max_mul(struct bpf_reg_state *dst_reg, 14883 struct bpf_reg_state *src_reg) 14884 { 14885 s32 smin = reg_s32_min(dst_reg); 14886 s32 smax = reg_s32_max(dst_reg); 14887 u32 umin = reg_u32_min(dst_reg); 14888 u32 umax = reg_u32_max(dst_reg); 14889 s32 tmp_prod[4]; 14890 14891 if (check_mul_overflow(umax, reg_u32_max(src_reg), &umax) || 14892 check_mul_overflow(umin, reg_u32_min(src_reg), &umin)) { 14893 /* Overflow possible, we know nothing */ 14894 umin = 0; 14895 umax = U32_MAX; 14896 } 14897 if (check_mul_overflow(smin, reg_s32_min(src_reg), &tmp_prod[0]) || 14898 check_mul_overflow(smin, reg_s32_max(src_reg), &tmp_prod[1]) || 14899 check_mul_overflow(smax, reg_s32_min(src_reg), &tmp_prod[2]) || 14900 check_mul_overflow(smax, reg_s32_max(src_reg), &tmp_prod[3])) { 14901 /* Overflow possible, we know nothing */ 14902 smin = S32_MIN; 14903 smax = S32_MAX; 14904 } else { 14905 smin = min_array(tmp_prod, 4); 14906 smax = max_array(tmp_prod, 4); 14907 } 14908 14909 dst_reg->r32 = cnum32_intersect(cnum32_from_urange(umin, umax), 14910 cnum32_from_srange(smin, smax)); 14911 } 14912 14913 static void scalar_min_max_mul(struct bpf_reg_state *dst_reg, 14914 struct bpf_reg_state *src_reg) 14915 { 14916 s64 smin = reg_smin(dst_reg); 14917 s64 smax = reg_smax(dst_reg); 14918 u64 umin = reg_umin(dst_reg); 14919 u64 umax = reg_umax(dst_reg); 14920 s64 tmp_prod[4]; 14921 14922 if (check_mul_overflow(umax, reg_umax(src_reg), &umax) || 14923 check_mul_overflow(umin, reg_umin(src_reg), &umin)) { 14924 /* Overflow possible, we know nothing */ 14925 umin = 0; 14926 umax = U64_MAX; 14927 } 14928 if (check_mul_overflow(smin, reg_smin(src_reg), &tmp_prod[0]) || 14929 check_mul_overflow(smin, reg_smax(src_reg), &tmp_prod[1]) || 14930 check_mul_overflow(smax, reg_smin(src_reg), &tmp_prod[2]) || 14931 check_mul_overflow(smax, reg_smax(src_reg), &tmp_prod[3])) { 14932 /* Overflow possible, we know nothing */ 14933 smin = S64_MIN; 14934 smax = S64_MAX; 14935 } else { 14936 smin = min_array(tmp_prod, 4); 14937 smax = max_array(tmp_prod, 4); 14938 } 14939 14940 dst_reg->r64 = cnum64_intersect(cnum64_from_urange(umin, umax), 14941 cnum64_from_srange(smin, smax)); 14942 } 14943 14944 static void scalar32_min_max_udiv(struct bpf_reg_state *dst_reg, 14945 struct bpf_reg_state *src_reg) 14946 { 14947 u32 src_val = reg_u32_min(src_reg); /* non-zero, const divisor */ 14948 14949 reg_set_urange32(dst_reg, reg_u32_min(dst_reg) / src_val, 14950 reg_u32_max(dst_reg) / src_val); 14951 14952 /* Reset other ranges/tnum to unbounded/unknown. */ 14953 reset_reg64_and_tnum(dst_reg); 14954 } 14955 14956 static void scalar_min_max_udiv(struct bpf_reg_state *dst_reg, 14957 struct bpf_reg_state *src_reg) 14958 { 14959 u64 src_val = reg_umin(src_reg); /* non-zero, const divisor */ 14960 14961 reg_set_urange64(dst_reg, div64_u64(reg_umin(dst_reg), src_val), 14962 div64_u64(reg_umax(dst_reg), src_val)); 14963 14964 /* Reset other ranges/tnum to unbounded/unknown. */ 14965 reset_reg32_and_tnum(dst_reg); 14966 } 14967 14968 static void scalar32_min_max_sdiv(struct bpf_reg_state *dst_reg, 14969 struct bpf_reg_state *src_reg) 14970 { 14971 s32 smin = reg_s32_min(dst_reg); 14972 s32 smax = reg_s32_max(dst_reg); 14973 s32 src_val = reg_s32_min(src_reg); /* non-zero, const divisor */ 14974 s32 res1, res2; 14975 14976 /* BPF div specification: S32_MIN / -1 = S32_MIN */ 14977 if (smin == S32_MIN && src_val == -1) { 14978 /* 14979 * If the dividend range contains more than just S32_MIN, 14980 * we cannot precisely track the result, so it becomes unbounded. 14981 * e.g., [S32_MIN, S32_MIN+10]/(-1), 14982 * = {S32_MIN} U [-(S32_MIN+10), -(S32_MIN+1)] 14983 * = {S32_MIN} U [S32_MAX-9, S32_MAX] = [S32_MIN, S32_MAX] 14984 * Otherwise (if dividend is exactly S32_MIN), result remains S32_MIN. 14985 */ 14986 if (smax != S32_MIN) { 14987 smin = S32_MIN; 14988 smax = S32_MAX; 14989 } 14990 goto reset; 14991 } 14992 14993 res1 = smin / src_val; 14994 res2 = smax / src_val; 14995 smin = min(res1, res2); 14996 smax = max(res1, res2); 14997 14998 reset: 14999 reg_set_srange32(dst_reg, smin, smax); 15000 /* Reset other ranges/tnum to unbounded/unknown. */ 15001 reset_reg64_and_tnum(dst_reg); 15002 } 15003 15004 static void scalar_min_max_sdiv(struct bpf_reg_state *dst_reg, 15005 struct bpf_reg_state *src_reg) 15006 { 15007 s64 smin = reg_smin(dst_reg); 15008 s64 smax = reg_smax(dst_reg); 15009 s64 src_val = reg_smin(src_reg); /* non-zero, const divisor */ 15010 s64 res1, res2; 15011 15012 /* BPF div specification: S64_MIN / -1 = S64_MIN */ 15013 if (smin == S64_MIN && src_val == -1) { 15014 /* 15015 * If the dividend range contains more than just S64_MIN, 15016 * we cannot precisely track the result, so it becomes unbounded. 15017 * e.g., [S64_MIN, S64_MIN+10]/(-1), 15018 * = {S64_MIN} U [-(S64_MIN+10), -(S64_MIN+1)] 15019 * = {S64_MIN} U [S64_MAX-9, S64_MAX] = [S64_MIN, S64_MAX] 15020 * Otherwise (if dividend is exactly S64_MIN), result remains S64_MIN. 15021 */ 15022 if (smax != S64_MIN) { 15023 smin = S64_MIN; 15024 smax = S64_MAX; 15025 } 15026 goto reset; 15027 } 15028 15029 res1 = div64_s64(smin, src_val); 15030 res2 = div64_s64(smax, src_val); 15031 smin = min(res1, res2); 15032 smax = max(res1, res2); 15033 15034 reset: 15035 reg_set_srange64(dst_reg, smin, smax); 15036 /* Reset other ranges/tnum to unbounded/unknown. */ 15037 reset_reg32_and_tnum(dst_reg); 15038 } 15039 15040 static void scalar32_min_max_umod(struct bpf_reg_state *dst_reg, 15041 struct bpf_reg_state *src_reg) 15042 { 15043 u32 src_val = reg_u32_min(src_reg); /* non-zero, const divisor */ 15044 u32 res_max = src_val - 1; 15045 15046 /* 15047 * If dst_umax <= res_max, the result remains unchanged. 15048 * e.g., [2, 5] % 10 = [2, 5]. 15049 */ 15050 if (reg_u32_max(dst_reg) <= res_max) 15051 return; 15052 15053 reg_set_urange32(dst_reg, 0, min(reg_u32_max(dst_reg), res_max)); 15054 15055 /* Reset other ranges/tnum to unbounded/unknown. */ 15056 reset_reg64_and_tnum(dst_reg); 15057 } 15058 15059 static void scalar_min_max_umod(struct bpf_reg_state *dst_reg, 15060 struct bpf_reg_state *src_reg) 15061 { 15062 u64 src_val = reg_umin(src_reg); /* non-zero, const divisor */ 15063 u64 res_max = src_val - 1; 15064 15065 /* 15066 * If dst_umax <= res_max, the result remains unchanged. 15067 * e.g., [2, 5] % 10 = [2, 5]. 15068 */ 15069 if (reg_umax(dst_reg) <= res_max) 15070 return; 15071 15072 reg_set_urange64(dst_reg, 0, min(reg_umax(dst_reg), res_max)); 15073 15074 /* Reset other ranges/tnum to unbounded/unknown. */ 15075 reset_reg32_and_tnum(dst_reg); 15076 } 15077 15078 static void scalar32_min_max_smod(struct bpf_reg_state *dst_reg, 15079 struct bpf_reg_state *src_reg) 15080 { 15081 s32 src_val = reg_s32_min(src_reg); /* non-zero, const divisor */ 15082 15083 /* 15084 * Safe absolute value calculation: 15085 * If src_val == S32_MIN (-2147483648), src_abs becomes 2147483648. 15086 * Here use unsigned integer to avoid overflow. 15087 */ 15088 u32 src_abs = (src_val > 0) ? (u32)src_val : -(u32)src_val; 15089 15090 /* 15091 * Calculate the maximum possible absolute value of the result. 15092 * Even if src_abs is 2147483648 (S32_MIN), subtracting 1 gives 15093 * 2147483647 (S32_MAX), which fits perfectly in s32. 15094 */ 15095 s32 res_max_abs = src_abs - 1; 15096 15097 /* 15098 * If the dividend is already within the result range, 15099 * the result remains unchanged. e.g., [-2, 5] % 10 = [-2, 5]. 15100 */ 15101 if (reg_s32_min(dst_reg) >= -res_max_abs && reg_s32_max(dst_reg) <= res_max_abs) 15102 return; 15103 15104 /* General case: result has the same sign as the dividend. */ 15105 if (reg_s32_min(dst_reg) >= 0) { 15106 reg_set_srange32(dst_reg, 0, min(reg_s32_max(dst_reg), res_max_abs)); 15107 } else if (reg_s32_max(dst_reg) <= 0) { 15108 reg_set_srange32(dst_reg, max(reg_s32_min(dst_reg), -res_max_abs), 0); 15109 } else { 15110 reg_set_srange32(dst_reg, -res_max_abs, res_max_abs); 15111 } 15112 15113 /* Reset other ranges/tnum to unbounded/unknown. */ 15114 reset_reg64_and_tnum(dst_reg); 15115 } 15116 15117 static void scalar_min_max_smod(struct bpf_reg_state *dst_reg, 15118 struct bpf_reg_state *src_reg) 15119 { 15120 s64 src_val = reg_smin(src_reg); /* non-zero, const divisor */ 15121 15122 /* 15123 * Safe absolute value calculation: 15124 * If src_val == S64_MIN (-2^63), src_abs becomes 2^63. 15125 * Here use unsigned integer to avoid overflow. 15126 */ 15127 u64 src_abs = (src_val > 0) ? (u64)src_val : -(u64)src_val; 15128 15129 /* 15130 * Calculate the maximum possible absolute value of the result. 15131 * Even if src_abs is 2^63 (S64_MIN), subtracting 1 gives 15132 * 2^63 - 1 (S64_MAX), which fits perfectly in s64. 15133 */ 15134 s64 res_max_abs = src_abs - 1; 15135 15136 /* 15137 * If the dividend is already within the result range, 15138 * the result remains unchanged. e.g., [-2, 5] % 10 = [-2, 5]. 15139 */ 15140 if (reg_smin(dst_reg) >= -res_max_abs && reg_smax(dst_reg) <= res_max_abs) 15141 return; 15142 15143 /* General case: result has the same sign as the dividend. */ 15144 if (reg_smin(dst_reg) >= 0) { 15145 reg_set_srange64(dst_reg, 0, min(reg_smax(dst_reg), res_max_abs)); 15146 } else if (reg_smax(dst_reg) <= 0) { 15147 reg_set_srange64(dst_reg, max(reg_smin(dst_reg), -res_max_abs), 0); 15148 } else { 15149 reg_set_srange64(dst_reg, -res_max_abs, res_max_abs); 15150 } 15151 15152 /* Reset other ranges/tnum to unbounded/unknown. */ 15153 reset_reg32_and_tnum(dst_reg); 15154 } 15155 15156 static void scalar32_min_max_and(struct bpf_reg_state *dst_reg, 15157 struct bpf_reg_state *src_reg) 15158 { 15159 bool src_known = tnum_subreg_is_const(src_reg->var_off); 15160 bool dst_known = tnum_subreg_is_const(dst_reg->var_off); 15161 struct tnum var32_off = tnum_subreg(dst_reg->var_off); 15162 u32 umax_val = reg_u32_max(src_reg); 15163 15164 if (src_known && dst_known) { 15165 __mark_reg32_known(dst_reg, var32_off.value); 15166 return; 15167 } 15168 15169 /* We get our minimum from the var_off, since that's inherently 15170 * bitwise. Our maximum is the minimum of the operands' maxima. 15171 */ 15172 reg_set_urange32(dst_reg, 15173 var32_off.value, 15174 min(reg_u32_max(dst_reg), umax_val)); 15175 } 15176 15177 static void scalar_min_max_and(struct bpf_reg_state *dst_reg, 15178 struct bpf_reg_state *src_reg) 15179 { 15180 bool src_known = tnum_is_const(src_reg->var_off); 15181 bool dst_known = tnum_is_const(dst_reg->var_off); 15182 u64 umax_val = reg_umax(src_reg); 15183 15184 if (src_known && dst_known) { 15185 __mark_reg_known(dst_reg, dst_reg->var_off.value); 15186 return; 15187 } 15188 15189 /* We get our minimum from the var_off, since that's inherently 15190 * bitwise. Our maximum is the minimum of the operands' maxima. 15191 */ 15192 reg_set_urange64(dst_reg, 15193 dst_reg->var_off.value, 15194 min(reg_umax(dst_reg), umax_val)); 15195 15196 /* We may learn something more from the var_off */ 15197 __update_reg_bounds(dst_reg); 15198 } 15199 15200 static void scalar32_min_max_or(struct bpf_reg_state *dst_reg, 15201 struct bpf_reg_state *src_reg) 15202 { 15203 bool src_known = tnum_subreg_is_const(src_reg->var_off); 15204 bool dst_known = tnum_subreg_is_const(dst_reg->var_off); 15205 struct tnum var32_off = tnum_subreg(dst_reg->var_off); 15206 u32 umin_val = reg_u32_min(src_reg); 15207 15208 if (src_known && dst_known) { 15209 __mark_reg32_known(dst_reg, var32_off.value); 15210 return; 15211 } 15212 15213 /* We get our maximum from the var_off, and our minimum is the 15214 * maximum of the operands' minima 15215 */ 15216 reg_set_urange32(dst_reg, 15217 max(reg_u32_min(dst_reg), umin_val), 15218 var32_off.value | var32_off.mask); 15219 } 15220 15221 static void scalar_min_max_or(struct bpf_reg_state *dst_reg, 15222 struct bpf_reg_state *src_reg) 15223 { 15224 bool src_known = tnum_is_const(src_reg->var_off); 15225 bool dst_known = tnum_is_const(dst_reg->var_off); 15226 u64 umin_val = reg_umin(src_reg); 15227 15228 if (src_known && dst_known) { 15229 __mark_reg_known(dst_reg, dst_reg->var_off.value); 15230 return; 15231 } 15232 15233 /* We get our maximum from the var_off, and our minimum is the 15234 * maximum of the operands' minima 15235 */ 15236 reg_set_urange64(dst_reg, 15237 max(reg_umin(dst_reg), umin_val), 15238 dst_reg->var_off.value | dst_reg->var_off.mask); 15239 15240 /* We may learn something more from the var_off */ 15241 __update_reg_bounds(dst_reg); 15242 } 15243 15244 static void scalar32_min_max_xor(struct bpf_reg_state *dst_reg, 15245 struct bpf_reg_state *src_reg) 15246 { 15247 bool src_known = tnum_subreg_is_const(src_reg->var_off); 15248 bool dst_known = tnum_subreg_is_const(dst_reg->var_off); 15249 struct tnum var32_off = tnum_subreg(dst_reg->var_off); 15250 15251 if (src_known && dst_known) { 15252 __mark_reg32_known(dst_reg, var32_off.value); 15253 return; 15254 } 15255 15256 /* We get both minimum and maximum from the var32_off. */ 15257 reg_set_urange32(dst_reg, var32_off.value, var32_off.value | var32_off.mask); 15258 } 15259 15260 static void scalar_min_max_xor(struct bpf_reg_state *dst_reg, 15261 struct bpf_reg_state *src_reg) 15262 { 15263 bool src_known = tnum_is_const(src_reg->var_off); 15264 bool dst_known = tnum_is_const(dst_reg->var_off); 15265 15266 if (src_known && dst_known) { 15267 /* dst_reg->var_off.value has been updated earlier */ 15268 __mark_reg_known(dst_reg, dst_reg->var_off.value); 15269 return; 15270 } 15271 15272 /* We get both minimum and maximum from the var_off. */ 15273 reg_set_urange64(dst_reg, 15274 dst_reg->var_off.value, 15275 dst_reg->var_off.value | dst_reg->var_off.mask); 15276 } 15277 15278 static void __scalar32_min_max_lsh(struct bpf_reg_state *dst_reg, 15279 u64 umin_val, u64 umax_val) 15280 { 15281 /* If we might shift our top bit out, then we know nothing */ 15282 if (umax_val > 31 || reg_u32_max(dst_reg) > 1ULL << (31 - umax_val)) 15283 reg_set_urange32(dst_reg, 0, U32_MAX); 15284 else 15285 /* We lose all sign bit information (except what we can pick 15286 * up from var_off) 15287 */ 15288 reg_set_urange32(dst_reg, reg_u32_min(dst_reg) << umin_val, 15289 reg_u32_max(dst_reg) << umax_val); 15290 } 15291 15292 static void scalar32_min_max_lsh(struct bpf_reg_state *dst_reg, 15293 struct bpf_reg_state *src_reg) 15294 { 15295 u32 umax_val = reg_u32_max(src_reg); 15296 u32 umin_val = reg_u32_min(src_reg); 15297 /* u32 alu operation will zext upper bits */ 15298 struct tnum subreg = tnum_subreg(dst_reg->var_off); 15299 15300 __scalar32_min_max_lsh(dst_reg, umin_val, umax_val); 15301 dst_reg->var_off = tnum_subreg(tnum_lshift(subreg, umin_val)); 15302 /* Not required but being careful mark reg64 bounds as unknown so 15303 * that we are forced to pick them up from tnum and zext later and 15304 * if some path skips this step we are still safe. 15305 */ 15306 __mark_reg64_unbounded(dst_reg); 15307 __update_reg32_bounds(dst_reg); 15308 } 15309 15310 static void __scalar64_min_max_lsh(struct bpf_reg_state *dst_reg, 15311 u64 umin_val, u64 umax_val) 15312 { 15313 struct cnum64 u, s; 15314 15315 /* Special case <<32 because it is a common compiler pattern to sign 15316 * extend subreg by doing <<32 s>>32. smin/smax assignments are correct 15317 * because s32 bounds don't flip sign when shifting to the left by 15318 * 32bits. 15319 */ 15320 if (umin_val == 32 && umax_val == 32) 15321 s = cnum64_from_srange((s64)reg_s32_min(dst_reg) << 32, 15322 (s64)reg_s32_max(dst_reg) << 32); 15323 else 15324 s = CNUM64_UNBOUNDED; 15325 15326 /* If we might shift our top bit out, then we know nothing */ 15327 if (reg_umax(dst_reg) > 1ULL << (63 - umax_val)) 15328 u = CNUM64_UNBOUNDED; 15329 else 15330 u = cnum64_from_urange(reg_umin(dst_reg) << umin_val, 15331 reg_umax(dst_reg) << umax_val); 15332 15333 dst_reg->r64 = cnum64_intersect(u, s); 15334 } 15335 15336 static void scalar_min_max_lsh(struct bpf_reg_state *dst_reg, 15337 struct bpf_reg_state *src_reg) 15338 { 15339 u64 umax_val = reg_umax(src_reg); 15340 u64 umin_val = reg_umin(src_reg); 15341 15342 /* scalar64 calc uses 32bit unshifted bounds so must be called first */ 15343 __scalar64_min_max_lsh(dst_reg, umin_val, umax_val); 15344 __scalar32_min_max_lsh(dst_reg, umin_val, umax_val); 15345 15346 dst_reg->var_off = tnum_lshift(dst_reg->var_off, umin_val); 15347 /* We may learn something more from the var_off */ 15348 __update_reg_bounds(dst_reg); 15349 } 15350 15351 static void scalar32_min_max_rsh(struct bpf_reg_state *dst_reg, 15352 struct bpf_reg_state *src_reg) 15353 { 15354 struct tnum subreg = tnum_subreg(dst_reg->var_off); 15355 u32 umax_val = reg_u32_max(src_reg); 15356 u32 umin_val = reg_u32_min(src_reg); 15357 15358 /* BPF_RSH is an unsigned shift. If the value in dst_reg might 15359 * be negative, then either: 15360 * 1) src_reg might be zero, so the sign bit of the result is 15361 * unknown, so we lose our signed bounds 15362 * 2) it's known negative, thus the unsigned bounds capture the 15363 * signed bounds 15364 * 3) the signed bounds cross zero, so they tell us nothing 15365 * about the result 15366 * If the value in dst_reg is known nonnegative, then again the 15367 * unsigned bounds capture the signed bounds. 15368 * Thus, in all cases it suffices to blow away our signed bounds 15369 * and rely on inferring new ones from the unsigned bounds and 15370 * var_off of the result. 15371 */ 15372 15373 dst_reg->var_off = tnum_rshift(subreg, umin_val); 15374 reg_set_urange32(dst_reg, reg_u32_min(dst_reg) >> umax_val, 15375 reg_u32_max(dst_reg) >> umin_val); 15376 15377 __mark_reg64_unbounded(dst_reg); 15378 __update_reg32_bounds(dst_reg); 15379 } 15380 15381 static void scalar_min_max_rsh(struct bpf_reg_state *dst_reg, 15382 struct bpf_reg_state *src_reg) 15383 { 15384 u64 umax_val = reg_umax(src_reg); 15385 u64 umin_val = reg_umin(src_reg); 15386 15387 /* BPF_RSH is an unsigned shift. If the value in dst_reg might 15388 * be negative, then either: 15389 * 1) src_reg might be zero, so the sign bit of the result is 15390 * unknown, so we lose our signed bounds 15391 * 2) it's known negative, thus the unsigned bounds capture the 15392 * signed bounds 15393 * 3) the signed bounds cross zero, so they tell us nothing 15394 * about the result 15395 * If the value in dst_reg is known nonnegative, then again the 15396 * unsigned bounds capture the signed bounds. 15397 * Thus, in all cases it suffices to blow away our signed bounds 15398 * and rely on inferring new ones from the unsigned bounds and 15399 * var_off of the result. 15400 */ 15401 dst_reg->var_off = tnum_rshift(dst_reg->var_off, umin_val); 15402 reg_set_urange64(dst_reg, reg_umin(dst_reg) >> umax_val, 15403 reg_umax(dst_reg) >> umin_val); 15404 15405 /* Its not easy to operate on alu32 bounds here because it depends 15406 * on bits being shifted in. Take easy way out and mark unbounded 15407 * so we can recalculate later from tnum. 15408 */ 15409 __mark_reg32_unbounded(dst_reg); 15410 __update_reg_bounds(dst_reg); 15411 } 15412 15413 static void scalar32_min_max_arsh(struct bpf_reg_state *dst_reg, 15414 struct bpf_reg_state *src_reg) 15415 { 15416 u64 umin_val = reg_u32_min(src_reg); 15417 15418 /* Upon reaching here, src_known is true and 15419 * umax_val is equal to umin_val. 15420 * Blow away the dst_reg umin_value/umax_value and rely on 15421 * dst_reg var_off to refine the result. 15422 */ 15423 reg_set_srange32(dst_reg, 15424 (u32)(((s32)reg_s32_min(dst_reg)) >> umin_val), 15425 (u32)(((s32)reg_s32_max(dst_reg)) >> umin_val)); 15426 15427 dst_reg->var_off = tnum_arshift(tnum_subreg(dst_reg->var_off), umin_val, 32); 15428 15429 __mark_reg64_unbounded(dst_reg); 15430 __update_reg32_bounds(dst_reg); 15431 } 15432 15433 static void scalar_min_max_arsh(struct bpf_reg_state *dst_reg, 15434 struct bpf_reg_state *src_reg) 15435 { 15436 u64 umin_val = reg_umin(src_reg); 15437 15438 /* Upon reaching here, src_known is true and umax_val is equal 15439 * to umin_val. 15440 */ 15441 reg_set_srange64(dst_reg, reg_smin(dst_reg) >> umin_val, 15442 reg_smax(dst_reg) >> umin_val); 15443 15444 dst_reg->var_off = tnum_arshift(dst_reg->var_off, umin_val, 64); 15445 15446 /* Its not easy to operate on alu32 bounds here because it depends 15447 * on bits being shifted in from upper 32-bits. Take easy way out 15448 * and mark unbounded so we can recalculate later from tnum. 15449 */ 15450 __mark_reg32_unbounded(dst_reg); 15451 __update_reg_bounds(dst_reg); 15452 } 15453 15454 static void scalar_byte_swap(struct bpf_reg_state *dst_reg, struct bpf_insn *insn) 15455 { 15456 /* 15457 * Byte swap operation - update var_off using tnum_bswap. 15458 * Three cases: 15459 * 1. bswap(16|32|64): opcode=0xd7 (BPF_END | BPF_ALU64 | BPF_TO_LE) 15460 * unconditional swap 15461 * 2. to_le(16|32|64): opcode=0xd4 (BPF_END | BPF_ALU | BPF_TO_LE) 15462 * swap on big-endian, truncation or no-op on little-endian 15463 * 3. to_be(16|32|64): opcode=0xdc (BPF_END | BPF_ALU | BPF_TO_BE) 15464 * swap on little-endian, truncation or no-op on big-endian 15465 */ 15466 15467 bool alu64 = BPF_CLASS(insn->code) == BPF_ALU64; 15468 bool to_le = BPF_SRC(insn->code) == BPF_TO_LE; 15469 bool is_big_endian; 15470 #ifdef CONFIG_CPU_BIG_ENDIAN 15471 is_big_endian = true; 15472 #else 15473 is_big_endian = false; 15474 #endif 15475 /* Apply bswap if alu64 or switch between big-endian and little-endian machines */ 15476 bool need_bswap = alu64 || (to_le == is_big_endian); 15477 15478 /* 15479 * If the register is mutated, manually reset its scalar ID to break 15480 * any existing ties and avoid incorrect bounds propagation. 15481 */ 15482 if (need_bswap || insn->imm == 16 || insn->imm == 32) 15483 clear_scalar_id(dst_reg); 15484 15485 if (need_bswap) { 15486 if (insn->imm == 16) 15487 dst_reg->var_off = tnum_bswap16(dst_reg->var_off); 15488 else if (insn->imm == 32) 15489 dst_reg->var_off = tnum_bswap32(dst_reg->var_off); 15490 else if (insn->imm == 64) 15491 dst_reg->var_off = tnum_bswap64(dst_reg->var_off); 15492 /* 15493 * Byteswap scrambles the range, so we must reset bounds. 15494 * Bounds will be re-derived from the new tnum later. 15495 */ 15496 __mark_reg_unbounded(dst_reg); 15497 } 15498 /* For bswap16/32, truncate dst register to match the swapped size */ 15499 if (insn->imm == 16 || insn->imm == 32) 15500 coerce_reg_to_size(dst_reg, insn->imm / 8); 15501 } 15502 15503 static bool is_safe_to_compute_dst_reg_range(struct bpf_insn *insn, 15504 const struct bpf_reg_state *src_reg) 15505 { 15506 bool src_is_const = false; 15507 u64 insn_bitness = (BPF_CLASS(insn->code) == BPF_ALU64) ? 64 : 32; 15508 15509 if (insn_bitness == 32) { 15510 if (tnum_subreg_is_const(src_reg->var_off) 15511 && reg_s32_min(src_reg) == reg_s32_max(src_reg) 15512 && reg_u32_min(src_reg) == reg_u32_max(src_reg)) 15513 src_is_const = true; 15514 } else { 15515 if (tnum_is_const(src_reg->var_off) 15516 && reg_smin(src_reg) == reg_smax(src_reg) 15517 && reg_umin(src_reg) == reg_umax(src_reg)) 15518 src_is_const = true; 15519 } 15520 15521 switch (BPF_OP(insn->code)) { 15522 case BPF_ADD: 15523 case BPF_SUB: 15524 case BPF_NEG: 15525 case BPF_AND: 15526 case BPF_XOR: 15527 case BPF_OR: 15528 case BPF_MUL: 15529 case BPF_END: 15530 return true; 15531 15532 /* 15533 * Division and modulo operators range is only safe to compute when the 15534 * divisor is a constant. 15535 */ 15536 case BPF_DIV: 15537 case BPF_MOD: 15538 return src_is_const; 15539 15540 /* Shift operators range is only computable if shift dimension operand 15541 * is a constant. Shifts greater than 31 or 63 are undefined. This 15542 * includes shifts by a negative number. 15543 */ 15544 case BPF_LSH: 15545 case BPF_RSH: 15546 case BPF_ARSH: 15547 return (src_is_const && reg_umax(src_reg) < insn_bitness); 15548 default: 15549 return false; 15550 } 15551 } 15552 15553 static int maybe_fork_scalars(struct bpf_verifier_env *env, struct bpf_insn *insn, 15554 struct bpf_reg_state *dst_reg) 15555 { 15556 struct bpf_verifier_state *branch; 15557 struct bpf_reg_state *regs; 15558 bool alu32; 15559 15560 if (reg_smin(dst_reg) == -1 && reg_smax(dst_reg) == 0) 15561 alu32 = false; 15562 else if (reg_s32_min(dst_reg) == -1 && reg_s32_max(dst_reg) == 0) 15563 alu32 = true; 15564 else 15565 return 0; 15566 15567 branch = push_stack(env, env->insn_idx, env->insn_idx, false); 15568 if (IS_ERR(branch)) 15569 return PTR_ERR(branch); 15570 15571 regs = branch->frame[branch->curframe]->regs; 15572 if (alu32) { 15573 __mark_reg32_known(®s[insn->dst_reg], 0); 15574 __mark_reg32_known(dst_reg, -1ull); 15575 } else { 15576 __mark_reg_known(®s[insn->dst_reg], 0); 15577 __mark_reg_known(dst_reg, -1ull); 15578 } 15579 return 0; 15580 } 15581 15582 /* WARNING: This function does calculations on 64-bit values, but the actual 15583 * execution may occur on 32-bit values. Therefore, things like bitshifts 15584 * need extra checks in the 32-bit case. 15585 */ 15586 static int adjust_scalar_min_max_vals(struct bpf_verifier_env *env, 15587 struct bpf_insn *insn, 15588 struct bpf_reg_state *dst_reg, 15589 struct bpf_reg_state src_reg) 15590 { 15591 u8 opcode = BPF_OP(insn->code); 15592 s16 off = insn->off; 15593 bool alu32 = (BPF_CLASS(insn->code) != BPF_ALU64); 15594 int ret; 15595 15596 if (!is_safe_to_compute_dst_reg_range(insn, &src_reg)) { 15597 __mark_reg_unknown(env, dst_reg); 15598 return 0; 15599 } 15600 15601 if (sanitize_needed(opcode)) { 15602 ret = sanitize_val_alu(env, insn); 15603 if (ret < 0) 15604 return sanitize_err(env, insn, ret); 15605 } 15606 15607 /* Calculate sign/unsigned bounds and tnum for alu32 and alu64 bit ops. 15608 * There are two classes of instructions: The first class we track both 15609 * alu32 and alu64 sign/unsigned bounds independently this provides the 15610 * greatest amount of precision when alu operations are mixed with jmp32 15611 * operations. These operations are BPF_ADD, BPF_SUB, BPF_MUL, BPF_ADD, 15612 * and BPF_OR. This is possible because these ops have fairly easy to 15613 * understand and calculate behavior in both 32-bit and 64-bit alu ops. 15614 * See alu32 verifier tests for examples. The second class of 15615 * operations, BPF_LSH, BPF_RSH, and BPF_ARSH, however are not so easy 15616 * with regards to tracking sign/unsigned bounds because the bits may 15617 * cross subreg boundaries in the alu64 case. When this happens we mark 15618 * the reg unbounded in the subreg bound space and use the resulting 15619 * tnum to calculate an approximation of the sign/unsigned bounds. 15620 */ 15621 switch (opcode) { 15622 case BPF_ADD: 15623 scalar32_min_max_add(dst_reg, &src_reg); 15624 scalar_min_max_add(dst_reg, &src_reg); 15625 dst_reg->var_off = tnum_add(dst_reg->var_off, src_reg.var_off); 15626 break; 15627 case BPF_SUB: 15628 scalar32_min_max_sub(dst_reg, &src_reg); 15629 scalar_min_max_sub(dst_reg, &src_reg); 15630 dst_reg->var_off = tnum_sub(dst_reg->var_off, src_reg.var_off); 15631 break; 15632 case BPF_NEG: 15633 env->fake_reg[0] = *dst_reg; 15634 __mark_reg_known(dst_reg, 0); 15635 scalar32_min_max_sub(dst_reg, &env->fake_reg[0]); 15636 scalar_min_max_sub(dst_reg, &env->fake_reg[0]); 15637 dst_reg->var_off = tnum_neg(env->fake_reg[0].var_off); 15638 break; 15639 case BPF_MUL: 15640 dst_reg->var_off = tnum_mul(dst_reg->var_off, src_reg.var_off); 15641 scalar32_min_max_mul(dst_reg, &src_reg); 15642 scalar_min_max_mul(dst_reg, &src_reg); 15643 break; 15644 case BPF_DIV: 15645 /* BPF div specification: x / 0 = 0 */ 15646 if ((alu32 && reg_u32_min(&src_reg) == 0) || (!alu32 && reg_umin(&src_reg) == 0)) { 15647 ___mark_reg_known(dst_reg, 0); 15648 break; 15649 } 15650 if (alu32) 15651 if (off == 1) 15652 scalar32_min_max_sdiv(dst_reg, &src_reg); 15653 else 15654 scalar32_min_max_udiv(dst_reg, &src_reg); 15655 else 15656 if (off == 1) 15657 scalar_min_max_sdiv(dst_reg, &src_reg); 15658 else 15659 scalar_min_max_udiv(dst_reg, &src_reg); 15660 break; 15661 case BPF_MOD: 15662 /* BPF mod specification: x % 0 = x */ 15663 if ((alu32 && reg_u32_min(&src_reg) == 0) || (!alu32 && reg_umin(&src_reg) == 0)) 15664 break; 15665 if (alu32) 15666 if (off == 1) 15667 scalar32_min_max_smod(dst_reg, &src_reg); 15668 else 15669 scalar32_min_max_umod(dst_reg, &src_reg); 15670 else 15671 if (off == 1) 15672 scalar_min_max_smod(dst_reg, &src_reg); 15673 else 15674 scalar_min_max_umod(dst_reg, &src_reg); 15675 break; 15676 case BPF_AND: 15677 if (tnum_is_const(src_reg.var_off)) { 15678 ret = maybe_fork_scalars(env, insn, dst_reg); 15679 if (ret) 15680 return ret; 15681 } 15682 dst_reg->var_off = tnum_and(dst_reg->var_off, src_reg.var_off); 15683 scalar32_min_max_and(dst_reg, &src_reg); 15684 scalar_min_max_and(dst_reg, &src_reg); 15685 break; 15686 case BPF_OR: 15687 if (tnum_is_const(src_reg.var_off)) { 15688 ret = maybe_fork_scalars(env, insn, dst_reg); 15689 if (ret) 15690 return ret; 15691 } 15692 dst_reg->var_off = tnum_or(dst_reg->var_off, src_reg.var_off); 15693 scalar32_min_max_or(dst_reg, &src_reg); 15694 scalar_min_max_or(dst_reg, &src_reg); 15695 break; 15696 case BPF_XOR: 15697 dst_reg->var_off = tnum_xor(dst_reg->var_off, src_reg.var_off); 15698 scalar32_min_max_xor(dst_reg, &src_reg); 15699 scalar_min_max_xor(dst_reg, &src_reg); 15700 break; 15701 case BPF_LSH: 15702 if (alu32) 15703 scalar32_min_max_lsh(dst_reg, &src_reg); 15704 else 15705 scalar_min_max_lsh(dst_reg, &src_reg); 15706 break; 15707 case BPF_RSH: 15708 if (alu32) 15709 scalar32_min_max_rsh(dst_reg, &src_reg); 15710 else 15711 scalar_min_max_rsh(dst_reg, &src_reg); 15712 break; 15713 case BPF_ARSH: 15714 if (alu32) 15715 scalar32_min_max_arsh(dst_reg, &src_reg); 15716 else 15717 scalar_min_max_arsh(dst_reg, &src_reg); 15718 break; 15719 case BPF_END: 15720 scalar_byte_swap(dst_reg, insn); 15721 break; 15722 default: 15723 break; 15724 } 15725 15726 /* 15727 * ALU32 ops are zero extended into 64bit register. 15728 * 15729 * BPF_END is already handled inside the helper (truncation), 15730 * so skip zext here to avoid unexpected zero extension. 15731 * e.g., le64: opcode=(BPF_END|BPF_ALU|BPF_TO_LE), imm=0x40 15732 * This is a 64bit byte swap operation with alu32==true, 15733 * but we should not zero extend the result. 15734 */ 15735 if (alu32 && opcode != BPF_END) 15736 zext_32_to_64(dst_reg); 15737 reg_bounds_sync(dst_reg); 15738 return 0; 15739 } 15740 15741 /* Handles ALU ops other than BPF_END, BPF_NEG and BPF_MOV: computes new min/max 15742 * and var_off. 15743 */ 15744 static int adjust_reg_min_max_vals(struct bpf_verifier_env *env, 15745 struct bpf_insn *insn) 15746 { 15747 struct bpf_verifier_state *vstate = env->cur_state; 15748 struct bpf_func_state *state = vstate->frame[vstate->curframe]; 15749 struct bpf_reg_state *regs = state->regs, *dst_reg, *src_reg; 15750 struct bpf_reg_state *ptr_reg = NULL, off_reg = {0}; 15751 bool alu32 = (BPF_CLASS(insn->code) != BPF_ALU64); 15752 struct bpf_insn_aux_data *aux = cur_aux(env); 15753 u8 opcode = BPF_OP(insn->code); 15754 int err; 15755 15756 dst_reg = ®s[insn->dst_reg]; 15757 if (BPF_SRC(insn->code) == BPF_X) 15758 src_reg = ®s[insn->src_reg]; 15759 else 15760 src_reg = NULL; 15761 15762 /* Case where at least one operand is an arena. */ 15763 if (dst_reg->type == PTR_TO_ARENA || (src_reg && src_reg->type == PTR_TO_ARENA)) { 15764 15765 if (dst_reg->type != PTR_TO_ARENA) 15766 *dst_reg = *src_reg; 15767 15768 if (BPF_CLASS(insn->code) == BPF_ALU64) { 15769 /* 15770 * Only arena pointers set needs_zext, but doing so 15771 * modifies the instruction at fixup time to an ALU32 15772 * and makes it unsuitable for 64-bit scalar args. We 15773 * prevent zext from being set if the instruction has 15774 * been previously called with non-arena registers. 15775 */ 15776 if (aux->prevent_zext) { 15777 verbose(env, "same insn cannot be used with and without arena pointer\n"); 15778 return -EINVAL; 15779 } 15780 15781 /* 15782 * 32-bit operations zero upper bits automatically. 15783 * 64-bit operations need to be converted to 32. 15784 */ 15785 aux->needs_zext = true; 15786 aux->zext_dst = true; 15787 } 15788 15789 /* Any arithmetic operations are allowed on arena pointers */ 15790 return 0; 15791 } 15792 15793 /* Prevent the instruction from being used with arena pointers (see above). */ 15794 if (env->prog->aux->arena && BPF_CLASS(insn->code) == BPF_ALU64) { 15795 if (aux->needs_zext) { 15796 verbose(env, "same insn cannot be used with and without arena pointer\n"); 15797 return -EINVAL; 15798 } 15799 15800 aux->prevent_zext = true; 15801 } 15802 15803 if (dst_reg->type != SCALAR_VALUE) 15804 ptr_reg = dst_reg; 15805 15806 if (BPF_SRC(insn->code) == BPF_X) { 15807 if (src_reg->type != SCALAR_VALUE) { 15808 if (dst_reg->type != SCALAR_VALUE) { 15809 /* Combining two pointers by any ALU op yields 15810 * an arbitrary scalar. Disallow all math except 15811 * pointer subtraction 15812 */ 15813 if (opcode == BPF_SUB && env->allow_ptr_leaks) { 15814 mark_reg_unknown(env, regs, insn->dst_reg); 15815 return 0; 15816 } 15817 verbose(env, "R%d pointer %s pointer prohibited\n", 15818 insn->dst_reg, 15819 bpf_alu_string[opcode >> 4]); 15820 return -EACCES; 15821 } else { 15822 /* scalar += pointer 15823 * This is legal, but we have to reverse our 15824 * src/dest handling in computing the range 15825 */ 15826 err = mark_chain_precision(env, insn->dst_reg); 15827 if (err) 15828 return err; 15829 off_reg = *dst_reg; 15830 return adjust_ptr_min_max_vals(env, insn, insn->src_reg, src_reg, 15831 &off_reg); 15832 } 15833 } else if (ptr_reg) { 15834 /* pointer += scalar */ 15835 err = mark_chain_precision(env, insn->src_reg); 15836 if (err) 15837 return err; 15838 return adjust_ptr_min_max_vals(env, insn, insn->dst_reg, dst_reg, src_reg); 15839 } else if (dst_reg->precise) { 15840 /* if dst_reg is precise, src_reg should be precise as well */ 15841 err = mark_chain_precision(env, insn->src_reg); 15842 if (err) 15843 return err; 15844 } 15845 } else { 15846 /* Pretend the src is a reg with a known value, since we only 15847 * need to be able to read from this state. 15848 */ 15849 off_reg.type = SCALAR_VALUE; 15850 __mark_reg_known(&off_reg, insn->imm); 15851 src_reg = &off_reg; 15852 if (ptr_reg) /* pointer += K */ 15853 return adjust_ptr_min_max_vals(env, insn, insn->dst_reg, ptr_reg, src_reg); 15854 } 15855 15856 /* Got here implies adding two SCALAR_VALUEs */ 15857 if (WARN_ON_ONCE(ptr_reg)) { 15858 print_verifier_state(env, vstate, vstate->curframe, true); 15859 verbose(env, "verifier internal error: unexpected ptr_reg\n"); 15860 return -EFAULT; 15861 } 15862 if (WARN_ON(!src_reg)) { 15863 print_verifier_state(env, vstate, vstate->curframe, true); 15864 verbose(env, "verifier internal error: no src_reg\n"); 15865 return -EFAULT; 15866 } 15867 /* 15868 * For alu32 linked register tracking, we need to check dst_reg's 15869 * umax_value before the ALU operation. After adjust_scalar_min_max_vals(), 15870 * alu32 ops will have zero-extended the result, making umax_value <= U32_MAX. 15871 */ 15872 u64 dst_umax = reg_umax(dst_reg); 15873 15874 err = adjust_scalar_min_max_vals(env, insn, dst_reg, *src_reg); 15875 if (err) 15876 return err; 15877 /* 15878 * Compilers can generate the code 15879 * r1 = r2 15880 * r1 += 0x1 15881 * if r2 < 1000 goto ... 15882 * use r1 in memory access 15883 * So remember constant delta between r2 and r1 and update r1 after 15884 * 'if' condition. 15885 */ 15886 if (env->bpf_capable && 15887 (BPF_OP(insn->code) == BPF_ADD || BPF_OP(insn->code) == BPF_SUB) && 15888 dst_reg->id && is_reg_const(src_reg, alu32) && 15889 !(BPF_SRC(insn->code) == BPF_X && insn->src_reg == insn->dst_reg)) { 15890 u64 val = reg_const_value(src_reg, alu32); 15891 s32 off; 15892 15893 if (!alu32 && ((s64)val < S32_MIN || (s64)val > S32_MAX)) 15894 goto clear_id; 15895 15896 if (alu32 && (dst_umax > U32_MAX)) 15897 goto clear_id; 15898 15899 off = (s32)val; 15900 15901 if (BPF_OP(insn->code) == BPF_SUB) { 15902 /* Negating S32_MIN would overflow */ 15903 if (off == S32_MIN) 15904 goto clear_id; 15905 off = -off; 15906 } 15907 15908 if (dst_reg->id & BPF_ADD_CONST) { 15909 /* 15910 * If the register already went through rX += val 15911 * we cannot accumulate another val into rx->off. 15912 */ 15913 clear_id: 15914 clear_scalar_id(dst_reg); 15915 } else { 15916 if (alu32) 15917 dst_reg->id |= BPF_ADD_CONST32; 15918 else 15919 dst_reg->id |= BPF_ADD_CONST64; 15920 dst_reg->delta = off; 15921 } 15922 } else { 15923 /* 15924 * Make sure ID is cleared otherwise dst_reg min/max could be 15925 * incorrectly propagated into other registers by sync_linked_regs() 15926 */ 15927 clear_scalar_id(dst_reg); 15928 } 15929 return 0; 15930 } 15931 15932 /* check validity of 32-bit and 64-bit arithmetic operations */ 15933 static int check_alu_op(struct bpf_verifier_env *env, struct bpf_insn *insn) 15934 { 15935 struct bpf_reg_state *regs = cur_regs(env); 15936 u8 opcode = BPF_OP(insn->code); 15937 int err; 15938 15939 bpf_diag_mod_begin(env, ®s[insn->dst_reg], NULL, BPF_DIAG_MOD_WRITE); 15940 15941 if (opcode == BPF_END || opcode == BPF_NEG) { 15942 /* check src operand */ 15943 err = check_reg_arg(env, insn->dst_reg, SRC_OP); 15944 if (err) 15945 return err; 15946 15947 if (is_pointer_value(env, insn->dst_reg)) { 15948 verbose(env, "R%d pointer arithmetic prohibited\n", 15949 insn->dst_reg); 15950 return -EACCES; 15951 } 15952 15953 /* check dest operand */ 15954 if (regs[insn->dst_reg].type == SCALAR_VALUE) { 15955 err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK); 15956 err = err ?: adjust_scalar_min_max_vals(env, insn, 15957 ®s[insn->dst_reg], 15958 regs[insn->dst_reg]); 15959 } else { 15960 err = check_reg_arg(env, insn->dst_reg, DST_OP); 15961 } 15962 if (err) 15963 return err; 15964 15965 } else if (opcode == BPF_MOV) { 15966 15967 if (BPF_SRC(insn->code) == BPF_X) { 15968 if (insn->off == BPF_ADDR_SPACE_CAST) { 15969 if (!env->prog->aux->arena) { 15970 verbose(env, "addr_space_cast insn can only be used in a program that has an associated arena\n"); 15971 return -EINVAL; 15972 } 15973 } 15974 15975 /* check src operand */ 15976 err = check_reg_arg(env, insn->src_reg, SRC_OP); 15977 if (err) 15978 return err; 15979 } 15980 15981 /* check dest operand, mark as required later */ 15982 err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK); 15983 if (err) 15984 return err; 15985 15986 if (BPF_SRC(insn->code) == BPF_X) { 15987 struct bpf_reg_state *src_reg = regs + insn->src_reg; 15988 struct bpf_reg_state *dst_reg = regs + insn->dst_reg; 15989 15990 if (BPF_CLASS(insn->code) == BPF_ALU64) { 15991 if (insn->imm) { 15992 /* off == BPF_ADDR_SPACE_CAST */ 15993 mark_reg_unknown(env, regs, insn->dst_reg); 15994 if (insn->imm == 1) /* cast from as(1) to as(0) */ 15995 dst_reg->type = PTR_TO_ARENA; 15996 } else if (insn->off == 0) { 15997 /* case: R1 = R2 15998 * copy register state to dest reg 15999 */ 16000 assign_scalar_id_before_mov(env, src_reg); 16001 *dst_reg = *src_reg; 16002 } else { 16003 /* case: R1 = (s8, s16 s32)R2 */ 16004 if (is_pointer_value(env, insn->src_reg)) { 16005 verbose(env, 16006 "R%d sign-extension part of pointer\n", 16007 insn->src_reg); 16008 return -EACCES; 16009 } else if (src_reg->type == SCALAR_VALUE) { 16010 bool no_sext; 16011 16012 no_sext = reg_umax(src_reg) < (1ULL << (insn->off - 1)); 16013 if (no_sext) 16014 assign_scalar_id_before_mov(env, src_reg); 16015 *dst_reg = *src_reg; 16016 if (!no_sext) 16017 clear_scalar_id(dst_reg); 16018 coerce_reg_to_size_sx(dst_reg, insn->off >> 3); 16019 } else { 16020 mark_reg_unknown(env, regs, insn->dst_reg); 16021 } 16022 } 16023 } else { 16024 /* R1 = (u32) R2 */ 16025 if (is_pointer_value(env, insn->src_reg)) { 16026 verbose(env, 16027 "R%d partial copy of pointer\n", 16028 insn->src_reg); 16029 return -EACCES; 16030 } else if (src_reg->type == SCALAR_VALUE) { 16031 if (insn->off == 0) { 16032 bool is_src_reg_u32 = get_reg_width(src_reg) <= 32; 16033 16034 if (is_src_reg_u32) 16035 assign_scalar_id_before_mov(env, src_reg); 16036 *dst_reg = *src_reg; 16037 /* Make sure ID is cleared if src_reg is not in u32 16038 * range otherwise dst_reg min/max could be incorrectly 16039 * propagated into src_reg by sync_linked_regs() 16040 */ 16041 if (!is_src_reg_u32) 16042 clear_scalar_id(dst_reg); 16043 } else { 16044 /* case: W1 = (s8, s16)W2 */ 16045 bool no_sext = reg_umax(src_reg) < (1ULL << (insn->off - 1)); 16046 16047 if (no_sext) 16048 assign_scalar_id_before_mov(env, src_reg); 16049 *dst_reg = *src_reg; 16050 if (!no_sext) 16051 clear_scalar_id(dst_reg); 16052 coerce_subreg_to_size_sx(dst_reg, insn->off >> 3); 16053 } 16054 } else { 16055 mark_reg_unknown(env, regs, 16056 insn->dst_reg); 16057 } 16058 zext_32_to_64(dst_reg); 16059 reg_bounds_sync(dst_reg); 16060 } 16061 } else { 16062 /* case: R = imm 16063 * remember the value we stored into this reg 16064 */ 16065 /* clear any state __mark_reg_known doesn't set */ 16066 mark_reg_unknown(env, regs, insn->dst_reg); 16067 regs[insn->dst_reg].type = SCALAR_VALUE; 16068 if (BPF_CLASS(insn->code) == BPF_ALU64) { 16069 __mark_reg_known(regs + insn->dst_reg, 16070 insn->imm); 16071 } else { 16072 __mark_reg_known(regs + insn->dst_reg, 16073 (u32)insn->imm); 16074 } 16075 } 16076 16077 } else { /* all other ALU ops: and, sub, xor, add, ... */ 16078 16079 if (BPF_SRC(insn->code) == BPF_X) { 16080 /* check src1 operand */ 16081 err = check_reg_arg(env, insn->src_reg, SRC_OP); 16082 if (err) 16083 return err; 16084 } 16085 16086 /* check src2 operand */ 16087 err = check_reg_arg(env, insn->dst_reg, SRC_OP); 16088 if (err) 16089 return err; 16090 16091 if ((opcode == BPF_MOD || opcode == BPF_DIV) && 16092 BPF_SRC(insn->code) == BPF_K && insn->imm == 0) { 16093 verbose(env, "div by zero\n"); 16094 return -EINVAL; 16095 } 16096 16097 if ((opcode == BPF_LSH || opcode == BPF_RSH || 16098 opcode == BPF_ARSH) && BPF_SRC(insn->code) == BPF_K) { 16099 int size = BPF_CLASS(insn->code) == BPF_ALU64 ? 64 : 32; 16100 16101 if (insn->imm < 0 || insn->imm >= size) { 16102 verbose(env, "invalid shift %d\n", insn->imm); 16103 return -EINVAL; 16104 } 16105 } 16106 16107 /* check dest operand */ 16108 err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK); 16109 err = err ?: adjust_reg_min_max_vals(env, insn); 16110 if (err) 16111 return err; 16112 } 16113 16114 err = reg_bounds_sanity_check(env, ®s[insn->dst_reg], "alu"); 16115 if (err) 16116 return err; 16117 16118 bpf_diag_mod_end(env); 16119 return 0; 16120 } 16121 16122 static void find_good_pkt_pointers(struct bpf_verifier_state *vstate, 16123 struct bpf_reg_state *dst_reg, 16124 enum bpf_reg_type type, 16125 bool range_right_open) 16126 { 16127 struct bpf_func_state *state; 16128 struct bpf_reg_state *reg; 16129 int new_range; 16130 16131 if (reg_umax(dst_reg) == 0 && range_right_open) 16132 /* This doesn't give us any range */ 16133 return; 16134 16135 if (reg_umax(dst_reg) > MAX_PACKET_OFF) 16136 /* Risk of overflow. For instance, ptr + (1<<63) may be less 16137 * than pkt_end, but that's because it's also less than pkt. 16138 */ 16139 return; 16140 16141 new_range = reg_umax(dst_reg); 16142 if (range_right_open) 16143 new_range++; 16144 16145 /* Examples for register markings: 16146 * 16147 * pkt_data in dst register: 16148 * 16149 * r2 = r3; 16150 * r2 += 8; 16151 * if (r2 > pkt_end) goto <handle exception> 16152 * <access okay> 16153 * 16154 * r2 = r3; 16155 * r2 += 8; 16156 * if (r2 < pkt_end) goto <access okay> 16157 * <handle exception> 16158 * 16159 * Where: 16160 * r2 == dst_reg, pkt_end == src_reg 16161 * r2=pkt(id=n,off=8,r=0) 16162 * r3=pkt(id=n,off=0,r=0) 16163 * 16164 * pkt_data in src register: 16165 * 16166 * r2 = r3; 16167 * r2 += 8; 16168 * if (pkt_end >= r2) goto <access okay> 16169 * <handle exception> 16170 * 16171 * r2 = r3; 16172 * r2 += 8; 16173 * if (pkt_end <= r2) goto <handle exception> 16174 * <access okay> 16175 * 16176 * Where: 16177 * pkt_end == dst_reg, r2 == src_reg 16178 * r2=pkt(id=n,off=8,r=0) 16179 * r3=pkt(id=n,off=0,r=0) 16180 * 16181 * Find register r3 and mark its range as r3=pkt(id=n,off=0,r=8) 16182 * or r3=pkt(id=n,off=0,r=8-1), so that range of bytes [r3, r3 + 8) 16183 * and [r3, r3 + 8-1) respectively is safe to access depending on 16184 * the check. 16185 */ 16186 16187 /* If our ids match, then we must have the same max_value. And we 16188 * don't care about the other reg's fixed offset, since if it's too big 16189 * the range won't allow anything. 16190 * reg_umax(dst_reg) is known < MAX_PACKET_OFF, therefore it fits in a u16. 16191 */ 16192 bpf_for_each_reg_in_vstate(vstate, state, reg, ({ 16193 if (reg->type == type && reg->id == dst_reg->id) 16194 /* keep the maximum range already checked */ 16195 reg->range = max(reg->range, new_range); 16196 })); 16197 } 16198 16199 static void regs_refine_cond_op(struct bpf_reg_state *reg1, struct bpf_reg_state *reg2, 16200 u8 opcode, bool is_jmp32); 16201 static u8 rev_opcode(u8 opcode); 16202 16203 /* 16204 * Learn more information about live branches by simulating refinement on both branches. 16205 * regs_refine_cond_op() is sound, so producing ill-formed register bounds for the branch means 16206 * that branch is dead. 16207 */ 16208 static int simulate_both_branches_taken(struct bpf_verifier_env *env, u8 opcode, bool is_jmp32) 16209 { 16210 /* Fallthrough (FALSE) branch */ 16211 regs_refine_cond_op(&env->false_reg1, &env->false_reg2, rev_opcode(opcode), is_jmp32); 16212 reg_bounds_sync(&env->false_reg1); 16213 reg_bounds_sync(&env->false_reg2); 16214 /* 16215 * If there is a range bounds violation in *any* of the abstract values in either 16216 * reg_states in the FALSE branch (i.e. reg1, reg2), the FALSE branch must be dead. Only 16217 * TRUE branch will be taken. 16218 */ 16219 if (range_bounds_violation(&env->false_reg1) || range_bounds_violation(&env->false_reg2)) 16220 return 1; 16221 16222 /* Jump (TRUE) branch */ 16223 regs_refine_cond_op(&env->true_reg1, &env->true_reg2, opcode, is_jmp32); 16224 reg_bounds_sync(&env->true_reg1); 16225 reg_bounds_sync(&env->true_reg2); 16226 /* 16227 * If there is a range bounds violation in *any* of the abstract values in either 16228 * reg_states in the TRUE branch (i.e. true_reg1, true_reg2), the TRUE branch must be dead. 16229 * Only FALSE branch will be taken. 16230 */ 16231 if (range_bounds_violation(&env->true_reg1) || range_bounds_violation(&env->true_reg2)) 16232 return 0; 16233 16234 /* Both branches are possible, we can't determine which one will be taken. */ 16235 return -1; 16236 } 16237 16238 /* 16239 * <reg1> <op> <reg2>, currently assuming reg2 is a constant 16240 */ 16241 static int is_scalar_branch_taken(struct bpf_verifier_env *env, struct bpf_reg_state *reg1, 16242 struct bpf_reg_state *reg2, u8 opcode, bool is_jmp32) 16243 { 16244 struct tnum t1 = is_jmp32 ? tnum_subreg(reg1->var_off) : reg1->var_off; 16245 struct tnum t2 = is_jmp32 ? tnum_subreg(reg2->var_off) : reg2->var_off; 16246 u64 umin1 = is_jmp32 ? (u64)reg_u32_min(reg1) : reg_umin(reg1); 16247 u64 umax1 = is_jmp32 ? (u64)reg_u32_max(reg1) : reg_umax(reg1); 16248 s64 smin1 = is_jmp32 ? (s64)reg_s32_min(reg1) : reg_smin(reg1); 16249 s64 smax1 = is_jmp32 ? (s64)reg_s32_max(reg1) : reg_smax(reg1); 16250 u64 umin2 = is_jmp32 ? (u64)reg_u32_min(reg2) : reg_umin(reg2); 16251 u64 umax2 = is_jmp32 ? (u64)reg_u32_max(reg2) : reg_umax(reg2); 16252 s64 smin2 = is_jmp32 ? (s64)reg_s32_min(reg2) : reg_smin(reg2); 16253 s64 smax2 = is_jmp32 ? (s64)reg_s32_max(reg2) : reg_smax(reg2); 16254 16255 if (reg1 == reg2) { 16256 switch (opcode) { 16257 case BPF_JGE: 16258 case BPF_JLE: 16259 case BPF_JSGE: 16260 case BPF_JSLE: 16261 case BPF_JEQ: 16262 return 1; 16263 case BPF_JGT: 16264 case BPF_JLT: 16265 case BPF_JSGT: 16266 case BPF_JSLT: 16267 case BPF_JNE: 16268 return 0; 16269 case BPF_JSET: 16270 if (tnum_is_const(t1)) 16271 return t1.value != 0; 16272 else 16273 return (smin1 <= 0 && smax1 >= 0) ? -1 : 1; 16274 default: 16275 return -1; 16276 } 16277 } 16278 16279 switch (opcode) { 16280 case BPF_JEQ: 16281 /* constants, umin/umax and smin/smax checks would be 16282 * redundant in this case because they all should match 16283 */ 16284 if (tnum_is_const(t1) && tnum_is_const(t2)) 16285 return t1.value == t2.value; 16286 if (!tnum_overlap(t1, t2)) 16287 return 0; 16288 /* non-overlapping ranges */ 16289 if (umin1 > umax2 || umax1 < umin2) 16290 return 0; 16291 if (smin1 > smax2 || smax1 < smin2) 16292 return 0; 16293 if (!is_jmp32) { 16294 /* if 64-bit ranges are inconclusive, see if we can 16295 * utilize 32-bit subrange knowledge to eliminate 16296 * branches that can't be taken a priori 16297 */ 16298 if (reg_u32_min(reg1) > reg_u32_max(reg2) || 16299 reg_u32_max(reg1) < reg_u32_min(reg2)) 16300 return 0; 16301 if (reg_s32_min(reg1) > reg_s32_max(reg2) || 16302 reg_s32_max(reg1) < reg_s32_min(reg2)) 16303 return 0; 16304 } 16305 break; 16306 case BPF_JNE: 16307 /* constants, umin/umax and smin/smax checks would be 16308 * redundant in this case because they all should match 16309 */ 16310 if (tnum_is_const(t1) && tnum_is_const(t2)) 16311 return t1.value != t2.value; 16312 if (!tnum_overlap(t1, t2)) 16313 return 1; 16314 /* non-overlapping ranges */ 16315 if (umin1 > umax2 || umax1 < umin2) 16316 return 1; 16317 if (smin1 > smax2 || smax1 < smin2) 16318 return 1; 16319 if (!is_jmp32) { 16320 /* if 64-bit ranges are inconclusive, see if we can 16321 * utilize 32-bit subrange knowledge to eliminate 16322 * branches that can't be taken a priori 16323 */ 16324 if (reg_u32_min(reg1) > reg_u32_max(reg2) || 16325 reg_u32_max(reg1) < reg_u32_min(reg2)) 16326 return 1; 16327 if (reg_s32_min(reg1) > reg_s32_max(reg2) || 16328 reg_s32_max(reg1) < reg_s32_min(reg2)) 16329 return 1; 16330 } 16331 break; 16332 case BPF_JSET: 16333 if (!is_reg_const(reg2, is_jmp32)) { 16334 swap(reg1, reg2); 16335 swap(t1, t2); 16336 } 16337 if (!is_reg_const(reg2, is_jmp32)) 16338 return -1; 16339 if ((~t1.mask & t1.value) & t2.value) 16340 return 1; 16341 if (!((t1.mask | t1.value) & t2.value)) 16342 return 0; 16343 break; 16344 case BPF_JGT: 16345 if (umin1 > umax2) 16346 return 1; 16347 else if (umax1 <= umin2) 16348 return 0; 16349 break; 16350 case BPF_JSGT: 16351 if (smin1 > smax2) 16352 return 1; 16353 else if (smax1 <= smin2) 16354 return 0; 16355 break; 16356 case BPF_JLT: 16357 if (umax1 < umin2) 16358 return 1; 16359 else if (umin1 >= umax2) 16360 return 0; 16361 break; 16362 case BPF_JSLT: 16363 if (smax1 < smin2) 16364 return 1; 16365 else if (smin1 >= smax2) 16366 return 0; 16367 break; 16368 case BPF_JGE: 16369 if (umin1 >= umax2) 16370 return 1; 16371 else if (umax1 < umin2) 16372 return 0; 16373 break; 16374 case BPF_JSGE: 16375 if (smin1 >= smax2) 16376 return 1; 16377 else if (smax1 < smin2) 16378 return 0; 16379 break; 16380 case BPF_JLE: 16381 if (umax1 <= umin2) 16382 return 1; 16383 else if (umin1 > umax2) 16384 return 0; 16385 break; 16386 case BPF_JSLE: 16387 if (smax1 <= smin2) 16388 return 1; 16389 else if (smin1 > smax2) 16390 return 0; 16391 break; 16392 } 16393 16394 return simulate_both_branches_taken(env, opcode, is_jmp32); 16395 } 16396 16397 static int flip_opcode(u32 opcode) 16398 { 16399 /* How can we transform "a <op> b" into "b <op> a"? */ 16400 static const u8 opcode_flip[16] = { 16401 /* these stay the same */ 16402 [BPF_JEQ >> 4] = BPF_JEQ, 16403 [BPF_JNE >> 4] = BPF_JNE, 16404 [BPF_JSET >> 4] = BPF_JSET, 16405 /* these swap "lesser" and "greater" (L and G in the opcodes) */ 16406 [BPF_JGE >> 4] = BPF_JLE, 16407 [BPF_JGT >> 4] = BPF_JLT, 16408 [BPF_JLE >> 4] = BPF_JGE, 16409 [BPF_JLT >> 4] = BPF_JGT, 16410 [BPF_JSGE >> 4] = BPF_JSLE, 16411 [BPF_JSGT >> 4] = BPF_JSLT, 16412 [BPF_JSLE >> 4] = BPF_JSGE, 16413 [BPF_JSLT >> 4] = BPF_JSGT 16414 }; 16415 return opcode_flip[opcode >> 4]; 16416 } 16417 16418 static int is_pkt_ptr_branch_taken(struct bpf_reg_state *dst_reg, 16419 struct bpf_reg_state *src_reg, 16420 u8 opcode) 16421 { 16422 struct bpf_reg_state *pkt; 16423 16424 if (src_reg->type == PTR_TO_PACKET_END) { 16425 pkt = dst_reg; 16426 } else if (dst_reg->type == PTR_TO_PACKET_END) { 16427 pkt = src_reg; 16428 opcode = flip_opcode(opcode); 16429 } else { 16430 return -1; 16431 } 16432 16433 if (pkt->range >= 0) 16434 return -1; 16435 16436 switch (opcode) { 16437 case BPF_JLE: 16438 /* pkt <= pkt_end */ 16439 fallthrough; 16440 case BPF_JGT: 16441 /* pkt > pkt_end */ 16442 if (pkt->range == BEYOND_PKT_END) 16443 /* pkt has at last one extra byte beyond pkt_end */ 16444 return opcode == BPF_JGT; 16445 break; 16446 case BPF_JLT: 16447 /* pkt < pkt_end */ 16448 fallthrough; 16449 case BPF_JGE: 16450 /* pkt >= pkt_end */ 16451 if (pkt->range == BEYOND_PKT_END || pkt->range == AT_PKT_END) 16452 return opcode == BPF_JGE; 16453 break; 16454 } 16455 return -1; 16456 } 16457 16458 /* compute branch direction of the expression "if (<reg1> opcode <reg2>) goto target;" 16459 * and return: 16460 * 1 - branch will be taken and "goto target" will be executed 16461 * 0 - branch will not be taken and fall-through to next insn 16462 * -1 - unknown. Example: "if (reg1 < 5)" is unknown when register value 16463 * range [0,10] 16464 */ 16465 static int is_branch_taken(struct bpf_verifier_env *env, struct bpf_reg_state *reg1, 16466 struct bpf_reg_state *reg2, u8 opcode, bool is_jmp32) 16467 { 16468 if (reg_is_pkt_pointer_any(reg1) && reg_is_pkt_pointer_any(reg2) && !is_jmp32) 16469 return is_pkt_ptr_branch_taken(reg1, reg2, opcode); 16470 16471 if (__is_pointer_value(false, reg1) || __is_pointer_value(false, reg2)) { 16472 u64 val; 16473 16474 /* 16475 * The low 32 bits of a valid pointer may well be zero, hence 16476 * nothing below applies to a 32-bit comparison. 16477 */ 16478 if (is_jmp32) 16479 return -1; 16480 16481 /* arrange that reg2 is a scalar, and reg1 is a pointer */ 16482 if (!is_reg_const(reg2, is_jmp32)) { 16483 opcode = flip_opcode(opcode); 16484 swap(reg1, reg2); 16485 } 16486 /* and ensure that reg2 is a constant */ 16487 if (!is_reg_const(reg2, is_jmp32)) 16488 return -1; 16489 16490 if (!reg_not_null(env, reg1)) 16491 return -1; 16492 16493 /* If pointer is valid tests against zero will fail so we can 16494 * use this to direct branch taken. 16495 */ 16496 val = reg_const_value(reg2, is_jmp32); 16497 if (val != 0) 16498 return -1; 16499 16500 switch (opcode) { 16501 case BPF_JEQ: 16502 return 0; 16503 case BPF_JNE: 16504 return 1; 16505 default: 16506 return -1; 16507 } 16508 } 16509 16510 /* now deal with two scalars, but not necessarily constants */ 16511 return is_scalar_branch_taken(env, reg1, reg2, opcode, is_jmp32); 16512 } 16513 16514 /* Opcode that corresponds to a *false* branch condition. 16515 * E.g., if r1 < r2, then reverse (false) condition is r1 >= r2 16516 */ 16517 static u8 rev_opcode(u8 opcode) 16518 { 16519 switch (opcode) { 16520 case BPF_JEQ: return BPF_JNE; 16521 case BPF_JNE: return BPF_JEQ; 16522 /* JSET doesn't have it's reverse opcode in BPF, so add 16523 * BPF_X flag to denote the reverse of that operation 16524 */ 16525 case BPF_JSET: return BPF_JSET | BPF_X; 16526 case BPF_JSET | BPF_X: return BPF_JSET; 16527 case BPF_JGE: return BPF_JLT; 16528 case BPF_JGT: return BPF_JLE; 16529 case BPF_JLE: return BPF_JGT; 16530 case BPF_JLT: return BPF_JGE; 16531 case BPF_JSGE: return BPF_JSLT; 16532 case BPF_JSGT: return BPF_JSLE; 16533 case BPF_JSLE: return BPF_JSGT; 16534 case BPF_JSLT: return BPF_JSGE; 16535 default: return 0; 16536 } 16537 } 16538 16539 /* Refine range knowledge for <reg1> <op> <reg>2 conditional operation. */ 16540 static void regs_refine_cond_op(struct bpf_reg_state *reg1, struct bpf_reg_state *reg2, 16541 u8 opcode, bool is_jmp32) 16542 { 16543 struct tnum t; 16544 u64 val; 16545 16546 /* In case of GE/GT/SGE/JST, reuse LE/LT/SLE/SLT logic from below */ 16547 switch (opcode) { 16548 case BPF_JGE: 16549 case BPF_JGT: 16550 case BPF_JSGE: 16551 case BPF_JSGT: 16552 opcode = flip_opcode(opcode); 16553 swap(reg1, reg2); 16554 break; 16555 default: 16556 break; 16557 } 16558 16559 switch (opcode) { 16560 case BPF_JEQ: 16561 if (is_jmp32) { 16562 reg1->r32 = cnum32_intersect(reg1->r32, reg2->r32); 16563 reg2->r32 = reg1->r32; 16564 16565 t = tnum_intersect(tnum_subreg(reg1->var_off), tnum_subreg(reg2->var_off)); 16566 reg1->var_off = tnum_with_subreg(reg1->var_off, t); 16567 reg2->var_off = tnum_with_subreg(reg2->var_off, t); 16568 } else { 16569 reg1->r64 = cnum64_intersect(reg1->r64, reg2->r64); 16570 reg2->r64 = reg1->r64; 16571 16572 reg1->var_off = tnum_intersect(reg1->var_off, reg2->var_off); 16573 reg2->var_off = reg1->var_off; 16574 } 16575 break; 16576 case BPF_JNE: 16577 if (!is_reg_const(reg2, is_jmp32)) 16578 swap(reg1, reg2); 16579 if (!is_reg_const(reg2, is_jmp32)) 16580 break; 16581 16582 /* try to recompute the bound of reg1 if reg2 is a const and 16583 * is exactly the edge of reg1. 16584 */ 16585 val = reg_const_value(reg2, is_jmp32); 16586 if (is_jmp32) { 16587 /* Complement of the range [val, val] as cnum32. */ 16588 cnum32_intersect_with(®1->r32, (struct cnum32){ val + 1, U32_MAX - 1 }); 16589 } else { 16590 /* Complement of the range [val, val] as cnum64. */ 16591 cnum64_intersect_with(®1->r64, (struct cnum64){ val + 1, U64_MAX - 1 }); 16592 } 16593 break; 16594 case BPF_JSET: 16595 if (!is_reg_const(reg2, is_jmp32)) 16596 swap(reg1, reg2); 16597 if (!is_reg_const(reg2, is_jmp32)) 16598 break; 16599 val = reg_const_value(reg2, is_jmp32); 16600 /* BPF_JSET (i.e., TRUE branch, *not* BPF_JSET | BPF_X) 16601 * requires single bit to learn something useful. E.g., if we 16602 * know that `r1 & 0x3` is true, then which bits (0, 1, or both) 16603 * are actually set? We can learn something definite only if 16604 * it's a single-bit value to begin with. 16605 * 16606 * BPF_JSET | BPF_X (i.e., negation of BPF_JSET) doesn't have 16607 * this restriction. I.e., !(r1 & 0x3) means neither bit 0 nor 16608 * bit 1 is set, which we can readily use in adjustments. 16609 */ 16610 if (!is_power_of_2(val)) 16611 break; 16612 if (is_jmp32) { 16613 t = tnum_or(tnum_subreg(reg1->var_off), tnum_const(val)); 16614 reg1->var_off = tnum_with_subreg(reg1->var_off, t); 16615 } else { 16616 reg1->var_off = tnum_or(reg1->var_off, tnum_const(val)); 16617 } 16618 break; 16619 case BPF_JSET | BPF_X: /* reverse of BPF_JSET, see rev_opcode() */ 16620 if (!is_reg_const(reg2, is_jmp32)) 16621 swap(reg1, reg2); 16622 if (!is_reg_const(reg2, is_jmp32)) 16623 break; 16624 val = reg_const_value(reg2, is_jmp32); 16625 /* Forget the ranges before narrowing tnums, to avoid invariant 16626 * violations if we're on a dead branch. 16627 */ 16628 __mark_reg_unbounded(reg1); 16629 if (is_jmp32) { 16630 t = tnum_and(tnum_subreg(reg1->var_off), tnum_const(~val)); 16631 reg1->var_off = tnum_with_subreg(reg1->var_off, t); 16632 } else { 16633 reg1->var_off = tnum_and(reg1->var_off, tnum_const(~val)); 16634 } 16635 break; 16636 case BPF_JLE: 16637 if (is_jmp32) { 16638 cnum32_intersect_with_urange(®1->r32, 0, reg_u32_max(reg2)); 16639 cnum32_intersect_with_urange(®2->r32, reg_u32_min(reg1), U32_MAX); 16640 } else { 16641 cnum64_intersect_with_urange(®1->r64, 0, reg_umax(reg2)); 16642 cnum64_intersect_with_urange(®2->r64, reg_umin(reg1), U64_MAX); 16643 } 16644 break; 16645 case BPF_JLT: 16646 if (is_jmp32) { 16647 cnum32_intersect_with_urange(®1->r32, 0, reg_u32_max(reg2) - 1); 16648 cnum32_intersect_with_urange(®2->r32, reg_u32_min(reg1) + 1, U32_MAX); 16649 } else { 16650 cnum64_intersect_with_urange(®1->r64, 0, reg_umax(reg2) - 1); 16651 cnum64_intersect_with_urange(®2->r64, reg_umin(reg1) + 1, U64_MAX); 16652 } 16653 break; 16654 case BPF_JSLE: 16655 if (is_jmp32) { 16656 cnum32_intersect_with_srange(®1->r32, S32_MIN, reg_s32_max(reg2)); 16657 cnum32_intersect_with_srange(®2->r32, reg_s32_min(reg1), S32_MAX); 16658 } else { 16659 cnum64_intersect_with_srange(®1->r64, S64_MIN, reg_smax(reg2)); 16660 cnum64_intersect_with_srange(®2->r64, reg_smin(reg1), S64_MAX); 16661 } 16662 break; 16663 case BPF_JSLT: 16664 if (is_jmp32) { 16665 cnum32_intersect_with_srange(®1->r32, S32_MIN, reg_s32_max(reg2) - 1); 16666 cnum32_intersect_with_srange(®2->r32, reg_s32_min(reg1) + 1, S32_MAX); 16667 } else { 16668 cnum64_intersect_with_srange(®1->r64, S64_MIN, reg_smax(reg2) - 1); 16669 cnum64_intersect_with_srange(®2->r64, reg_smin(reg1) + 1, S64_MAX); 16670 } 16671 break; 16672 default: 16673 return; 16674 } 16675 } 16676 16677 /* Check for invariant violations on the registers for both branches of a condition */ 16678 static int regs_bounds_sanity_check_branches(struct bpf_verifier_env *env) 16679 { 16680 int err; 16681 16682 err = reg_bounds_sanity_check(env, &env->true_reg1, "true_reg1"); 16683 err = err ?: reg_bounds_sanity_check(env, &env->true_reg2, "true_reg2"); 16684 err = err ?: reg_bounds_sanity_check(env, &env->false_reg1, "false_reg1"); 16685 err = err ?: reg_bounds_sanity_check(env, &env->false_reg2, "false_reg2"); 16686 return err; 16687 } 16688 16689 static void mark_ptr_or_null_reg(struct bpf_func_state *state, 16690 struct bpf_reg_state *reg, u32 id, 16691 bool is_null) 16692 { 16693 if (type_may_be_null(reg->type) && reg->id == id && 16694 (is_rcu_reg(reg) || !WARN_ON_ONCE(!reg->id))) { 16695 /* Old offset should have been known-zero, because we don't 16696 * allow pointer arithmetic on pointers that might be NULL. 16697 * If we see this happening, don't convert the register. 16698 * 16699 * But in some cases, some helpers that return local kptrs 16700 * advance offset for the returned pointer. In those cases, 16701 * it is fine to expect to see reg->var_off. 16702 */ 16703 if (!(type_is_ptr_alloc_obj(reg->type) || type_is_non_owning_ref(reg->type)) && 16704 WARN_ON_ONCE(!tnum_equals_const(reg->var_off, 0))) 16705 return; 16706 if (is_null) { 16707 /* We don't need id from this point 16708 * onwards anymore, thus we should better reset it, 16709 * so that state pruning has chances to take effect. 16710 */ 16711 __mark_reg_known_zero(reg); 16712 reg->type = SCALAR_VALUE; 16713 16714 return; 16715 } 16716 16717 mark_ptr_not_null_reg(reg); 16718 16719 /* 16720 * reg->id is preserved for object relationship tracking 16721 * and spin_lock lock state tracking 16722 */ 16723 } 16724 } 16725 16726 /* The logic is similar to find_good_pkt_pointers(), both could eventually 16727 * be folded together at some point. 16728 */ 16729 static void mark_ptr_or_null_regs(struct bpf_verifier_state *vstate, u32 regno, 16730 bool is_null) 16731 { 16732 struct bpf_func_state *state = vstate->frame[vstate->curframe]; 16733 struct bpf_reg_state *regs = state->regs, *reg; 16734 u32 id = regs[regno].id; 16735 16736 if (is_null && find_reference_state(vstate, id)) 16737 /* regs[regno] is in the " == NULL" branch. 16738 * No one could have freed the reference state before 16739 * doing the NULL check. 16740 */ 16741 WARN_ON_ONCE(__release_reference_nomark(vstate, id)); 16742 16743 bpf_for_each_reg_in_vstate(vstate, state, reg, ({ 16744 mark_ptr_or_null_reg(state, reg, id, is_null); 16745 })); 16746 } 16747 16748 static bool try_match_pkt_pointers(const struct bpf_insn *insn, 16749 struct bpf_reg_state *dst_reg, 16750 struct bpf_reg_state *src_reg, 16751 struct bpf_verifier_state *this_branch, 16752 struct bpf_verifier_state *other_branch) 16753 { 16754 if (BPF_SRC(insn->code) != BPF_X) 16755 return false; 16756 16757 /* Pointers are always 64-bit. */ 16758 if (BPF_CLASS(insn->code) == BPF_JMP32) 16759 return false; 16760 16761 switch (BPF_OP(insn->code)) { 16762 case BPF_JGT: 16763 if ((dst_reg->type == PTR_TO_PACKET && 16764 src_reg->type == PTR_TO_PACKET_END) || 16765 (dst_reg->type == PTR_TO_PACKET_META && 16766 reg_is_init_pkt_pointer(src_reg, PTR_TO_PACKET))) { 16767 /* pkt_data' > pkt_end, pkt_meta' > pkt_data */ 16768 find_good_pkt_pointers(this_branch, dst_reg, 16769 dst_reg->type, false); 16770 mark_pkt_end(other_branch, insn->dst_reg, true); 16771 } else if ((dst_reg->type == PTR_TO_PACKET_END && 16772 src_reg->type == PTR_TO_PACKET) || 16773 (reg_is_init_pkt_pointer(dst_reg, PTR_TO_PACKET) && 16774 src_reg->type == PTR_TO_PACKET_META)) { 16775 /* pkt_end > pkt_data', pkt_data > pkt_meta' */ 16776 find_good_pkt_pointers(other_branch, src_reg, 16777 src_reg->type, true); 16778 mark_pkt_end(this_branch, insn->src_reg, false); 16779 } else { 16780 return false; 16781 } 16782 break; 16783 case BPF_JLT: 16784 if ((dst_reg->type == PTR_TO_PACKET && 16785 src_reg->type == PTR_TO_PACKET_END) || 16786 (dst_reg->type == PTR_TO_PACKET_META && 16787 reg_is_init_pkt_pointer(src_reg, PTR_TO_PACKET))) { 16788 /* pkt_data' < pkt_end, pkt_meta' < pkt_data */ 16789 find_good_pkt_pointers(other_branch, dst_reg, 16790 dst_reg->type, true); 16791 mark_pkt_end(this_branch, insn->dst_reg, false); 16792 } else if ((dst_reg->type == PTR_TO_PACKET_END && 16793 src_reg->type == PTR_TO_PACKET) || 16794 (reg_is_init_pkt_pointer(dst_reg, PTR_TO_PACKET) && 16795 src_reg->type == PTR_TO_PACKET_META)) { 16796 /* pkt_end < pkt_data', pkt_data > pkt_meta' */ 16797 find_good_pkt_pointers(this_branch, src_reg, 16798 src_reg->type, false); 16799 mark_pkt_end(other_branch, insn->src_reg, true); 16800 } else { 16801 return false; 16802 } 16803 break; 16804 case BPF_JGE: 16805 if ((dst_reg->type == PTR_TO_PACKET && 16806 src_reg->type == PTR_TO_PACKET_END) || 16807 (dst_reg->type == PTR_TO_PACKET_META && 16808 reg_is_init_pkt_pointer(src_reg, PTR_TO_PACKET))) { 16809 /* pkt_data' >= pkt_end, pkt_meta' >= pkt_data */ 16810 find_good_pkt_pointers(this_branch, dst_reg, 16811 dst_reg->type, true); 16812 mark_pkt_end(other_branch, insn->dst_reg, false); 16813 } else if ((dst_reg->type == PTR_TO_PACKET_END && 16814 src_reg->type == PTR_TO_PACKET) || 16815 (reg_is_init_pkt_pointer(dst_reg, PTR_TO_PACKET) && 16816 src_reg->type == PTR_TO_PACKET_META)) { 16817 /* pkt_end >= pkt_data', pkt_data >= pkt_meta' */ 16818 find_good_pkt_pointers(other_branch, src_reg, 16819 src_reg->type, false); 16820 mark_pkt_end(this_branch, insn->src_reg, true); 16821 } else { 16822 return false; 16823 } 16824 break; 16825 case BPF_JLE: 16826 if ((dst_reg->type == PTR_TO_PACKET && 16827 src_reg->type == PTR_TO_PACKET_END) || 16828 (dst_reg->type == PTR_TO_PACKET_META && 16829 reg_is_init_pkt_pointer(src_reg, PTR_TO_PACKET))) { 16830 /* pkt_data' <= pkt_end, pkt_meta' <= pkt_data */ 16831 find_good_pkt_pointers(other_branch, dst_reg, 16832 dst_reg->type, false); 16833 mark_pkt_end(this_branch, insn->dst_reg, true); 16834 } else if ((dst_reg->type == PTR_TO_PACKET_END && 16835 src_reg->type == PTR_TO_PACKET) || 16836 (reg_is_init_pkt_pointer(dst_reg, PTR_TO_PACKET) && 16837 src_reg->type == PTR_TO_PACKET_META)) { 16838 /* pkt_end <= pkt_data', pkt_data <= pkt_meta' */ 16839 find_good_pkt_pointers(this_branch, src_reg, 16840 src_reg->type, true); 16841 mark_pkt_end(other_branch, insn->src_reg, false); 16842 } else { 16843 return false; 16844 } 16845 break; 16846 default: 16847 return false; 16848 } 16849 16850 return true; 16851 } 16852 16853 static void __collect_linked_regs(struct linked_regs *reg_set, struct bpf_reg_state *reg, 16854 u32 id, u32 frameno, u32 spi_or_reg, bool is_reg) 16855 { 16856 struct linked_reg *e; 16857 16858 if (reg->type != SCALAR_VALUE || (reg->id & ~BPF_ADD_CONST) != id) 16859 return; 16860 16861 e = linked_regs_push(reg_set); 16862 if (e) { 16863 e->frameno = frameno; 16864 e->is_reg = is_reg; 16865 e->regno = spi_or_reg; 16866 } else { 16867 clear_scalar_id(reg); 16868 } 16869 } 16870 16871 /* For all R being scalar registers or spilled scalar registers 16872 * in verifier state, save R in linked_regs if R->id == id. 16873 * If there are too many Rs sharing same id, reset id for leftover Rs. 16874 */ 16875 static void collect_linked_regs(struct bpf_verifier_env *env, 16876 struct bpf_verifier_state *vstate, 16877 u32 id, 16878 struct linked_regs *linked_regs) 16879 { 16880 struct bpf_insn_aux_data *aux = env->insn_aux_data; 16881 struct bpf_func_state *func; 16882 struct bpf_reg_state *reg; 16883 u16 live_regs; 16884 int i, j; 16885 16886 id = id & ~BPF_ADD_CONST; 16887 for (i = vstate->curframe; i >= 0; i--) { 16888 live_regs = aux[bpf_frame_insn_idx(vstate, i)].live_regs_before; 16889 func = vstate->frame[i]; 16890 for (j = 0; j < BPF_REG_FP; j++) { 16891 if (!(live_regs & BIT(j))) 16892 continue; 16893 reg = &func->regs[j]; 16894 __collect_linked_regs(linked_regs, reg, id, i, j, true); 16895 } 16896 for (j = 0; j < func->allocated_stack / BPF_REG_SIZE; j++) { 16897 if (!bpf_is_spilled_reg(&func->stack[j])) 16898 continue; 16899 reg = &func->stack[j].spilled_ptr; 16900 __collect_linked_regs(linked_regs, reg, id, i, j, false); 16901 } 16902 } 16903 } 16904 16905 /* For all R in linked_regs, copy known_reg range into R 16906 * if R->id == known_reg->id. 16907 */ 16908 static void sync_linked_regs(struct bpf_verifier_env *env, struct bpf_verifier_state *vstate, 16909 struct bpf_reg_state *known_reg, struct linked_regs *linked_regs) 16910 { 16911 struct bpf_reg_state fake_reg; 16912 struct bpf_reg_state *reg; 16913 struct linked_reg *e; 16914 int i; 16915 16916 for (i = 0; i < linked_regs->cnt; ++i) { 16917 e = &linked_regs->entries[i]; 16918 reg = e->is_reg ? &vstate->frame[e->frameno]->regs[e->regno] 16919 : &vstate->frame[e->frameno]->stack[e->spi].spilled_ptr; 16920 if (reg->type != SCALAR_VALUE || reg == known_reg) 16921 continue; 16922 if ((reg->id & ~BPF_ADD_CONST) != (known_reg->id & ~BPF_ADD_CONST)) 16923 continue; 16924 /* 16925 * Skip mixed 32/64-bit links: the delta relationship doesn't 16926 * hold across different ALU widths. 16927 */ 16928 if (((reg->id ^ known_reg->id) & BPF_ADD_CONST) == BPF_ADD_CONST) 16929 continue; 16930 if ((!(reg->id & BPF_ADD_CONST) && !(known_reg->id & BPF_ADD_CONST)) || 16931 reg->delta == known_reg->delta) { 16932 *reg = *known_reg; 16933 } else { 16934 s32 saved_off = reg->delta; 16935 u32 saved_id = reg->id; 16936 16937 fake_reg.type = SCALAR_VALUE; 16938 __mark_reg_known(&fake_reg, (s64)reg->delta - (s64)known_reg->delta); 16939 16940 /* reg = known_reg; reg += delta */ 16941 *reg = *known_reg; 16942 /* 16943 * Must preserve off and id, otherwise another sync_linked_regs() 16944 * will be incorrect. 16945 */ 16946 reg->delta = saved_off; 16947 reg->id = saved_id; 16948 16949 scalar32_min_max_add(reg, &fake_reg); 16950 scalar_min_max_add(reg, &fake_reg); 16951 reg->var_off = tnum_add(reg->var_off, fake_reg.var_off); 16952 if ((reg->id | known_reg->id) & BPF_ADD_CONST32) 16953 zext_32_to_64(reg); 16954 reg_bounds_sync(reg); 16955 } 16956 if (e->is_reg) 16957 mark_reg_scratched(env, e->regno); 16958 else 16959 mark_stack_slot_scratched(env, e->spi); 16960 } 16961 } 16962 16963 static int check_cond_jmp_op(struct bpf_verifier_env *env, 16964 struct bpf_insn *insn, int *insn_idx) 16965 { 16966 struct bpf_verifier_state *this_branch = env->cur_state; 16967 struct bpf_verifier_state *other_branch; 16968 struct bpf_reg_state *regs = this_branch->frame[this_branch->curframe]->regs; 16969 struct bpf_reg_state *dst_reg, *other_branch_regs, *src_reg = NULL; 16970 struct bpf_reg_state *eq_branch_regs; 16971 struct linked_regs linked_regs = {}; 16972 u8 opcode = BPF_OP(insn->code); 16973 int insn_flags = 0; 16974 bool is_jmp32; 16975 int pred = -1; 16976 int err; 16977 16978 /* Only conditional jumps are expected to reach here. */ 16979 if (opcode == BPF_JA || opcode > BPF_JCOND) { 16980 verbose(env, "invalid BPF_JMP/JMP32 opcode %x\n", opcode); 16981 return -EINVAL; 16982 } 16983 16984 if (opcode == BPF_JCOND) { 16985 struct bpf_verifier_state *cur_st = env->cur_state, *queued_st, *prev_st; 16986 int idx = *insn_idx; 16987 16988 prev_st = find_prev_entry(env, cur_st->parent, idx); 16989 16990 /* branch out 'fallthrough' insn as a new state to explore */ 16991 queued_st = push_stack(env, idx + 1, idx, false); 16992 if (IS_ERR(queued_st)) 16993 return PTR_ERR(queued_st); 16994 16995 queued_st->may_goto_depth++; 16996 if (prev_st) 16997 widen_imprecise_scalars(env, prev_st, queued_st); 16998 *insn_idx += insn->off; 16999 return 0; 17000 } 17001 17002 /* check src2 operand */ 17003 err = check_reg_arg(env, insn->dst_reg, SRC_OP); 17004 if (err) 17005 return err; 17006 17007 dst_reg = ®s[insn->dst_reg]; 17008 if (BPF_SRC(insn->code) == BPF_X) { 17009 /* check src1 operand */ 17010 err = check_reg_arg(env, insn->src_reg, SRC_OP); 17011 if (err) 17012 return err; 17013 17014 src_reg = ®s[insn->src_reg]; 17015 if (!(reg_is_pkt_pointer_any(dst_reg) && reg_is_pkt_pointer_any(src_reg)) && 17016 is_pointer_value(env, insn->src_reg)) { 17017 verbose(env, "R%d pointer comparison prohibited\n", 17018 insn->src_reg); 17019 return -EACCES; 17020 } 17021 17022 if (src_reg->type == PTR_TO_STACK) 17023 insn_flags |= INSN_F_SRC_REG_STACK; 17024 if (dst_reg->type == PTR_TO_STACK) 17025 insn_flags |= INSN_F_DST_REG_STACK; 17026 } else { 17027 src_reg = &env->fake_reg[0]; 17028 memset(src_reg, 0, sizeof(*src_reg)); 17029 src_reg->type = SCALAR_VALUE; 17030 __mark_reg_known(src_reg, insn->imm); 17031 17032 if (dst_reg->type == PTR_TO_STACK) 17033 insn_flags |= INSN_F_DST_REG_STACK; 17034 } 17035 17036 if (insn_flags) { 17037 err = bpf_push_jmp_history(env, this_branch, insn_flags, 0, 0, 0); 17038 if (err) 17039 return err; 17040 } 17041 17042 /* 17043 * Collect the linked registers before env->{true,false}_reg{1,2} setup, 17044 * otherwise ids dropped by collect_linked_regs() would be resurrected 17045 * when env->{true,false}_reg{1,2} are copied back. 17046 */ 17047 if (BPF_SRC(insn->code) == BPF_X && src_reg->type == SCALAR_VALUE && src_reg->id) 17048 collect_linked_regs(env, this_branch, src_reg->id, &linked_regs); 17049 if (dst_reg->type == SCALAR_VALUE && dst_reg->id) 17050 collect_linked_regs(env, this_branch, dst_reg->id, &linked_regs); 17051 17052 is_jmp32 = BPF_CLASS(insn->code) == BPF_JMP32; 17053 env->false_reg1 = *dst_reg; 17054 env->false_reg2 = *src_reg; 17055 env->true_reg1 = *dst_reg; 17056 env->true_reg2 = *src_reg; 17057 pred = is_branch_taken(env, dst_reg, src_reg, opcode, is_jmp32); 17058 if (pred >= 0) { 17059 /* If we get here with a dst_reg pointer type it is because 17060 * above is_branch_taken() special cased the 0 comparison. 17061 */ 17062 if (!__is_pointer_value(false, dst_reg)) 17063 err = mark_chain_precision(env, insn->dst_reg); 17064 if (BPF_SRC(insn->code) == BPF_X && !err && 17065 !__is_pointer_value(false, src_reg)) 17066 err = mark_chain_precision(env, insn->src_reg); 17067 if (err) 17068 return err; 17069 } 17070 17071 if (pred == 1) { 17072 /* Only follow the goto, ignore fall-through. If needed, push 17073 * the fall-through branch for simulation under speculative 17074 * execution. 17075 */ 17076 if (!env->bypass_spec_v1) { 17077 err = sanitize_speculative_path(env, insn, *insn_idx + 1, *insn_idx); 17078 if (err < 0) 17079 return err; 17080 } 17081 if (env->log.level & BPF_LOG_LEVEL) 17082 print_insn_state(env, this_branch, this_branch->curframe); 17083 *insn_idx += insn->off; 17084 return 0; 17085 } else if (pred == 0) { 17086 /* Only follow the fall-through branch, since that's where the 17087 * program will go. If needed, push the goto branch for 17088 * simulation under speculative execution. 17089 */ 17090 if (!env->bypass_spec_v1) { 17091 err = sanitize_speculative_path(env, insn, *insn_idx + insn->off + 1, 17092 *insn_idx); 17093 if (err < 0) 17094 return err; 17095 } 17096 if (env->log.level & BPF_LOG_LEVEL) 17097 print_insn_state(env, this_branch, this_branch->curframe); 17098 return 0; 17099 } 17100 17101 /* Push scalar registers sharing same ID to jump history, 17102 * do this before creating 'other_branch', so that both 17103 * 'this_branch' and 'other_branch' share this history 17104 * if parent state is created. 17105 */ 17106 if (linked_regs.cnt > 1) { 17107 err = bpf_push_jmp_history(env, this_branch, 0, 0, 0, linked_regs_pack(&linked_regs)); 17108 if (err) 17109 return err; 17110 } 17111 17112 other_branch = push_stack(env, *insn_idx + insn->off + 1, *insn_idx, false); 17113 if (IS_ERR(other_branch)) 17114 return PTR_ERR(other_branch); 17115 other_branch_regs = other_branch->frame[other_branch->curframe]->regs; 17116 17117 err = regs_bounds_sanity_check_branches(env); 17118 if (err) 17119 return err; 17120 17121 *dst_reg = env->false_reg1; 17122 *src_reg = env->false_reg2; 17123 other_branch_regs[insn->dst_reg] = env->true_reg1; 17124 if (BPF_SRC(insn->code) == BPF_X) 17125 other_branch_regs[insn->src_reg] = env->true_reg2; 17126 17127 if (BPF_SRC(insn->code) == BPF_X && 17128 src_reg->type == SCALAR_VALUE && src_reg->id && 17129 !WARN_ON_ONCE(src_reg->id != other_branch_regs[insn->src_reg].id)) { 17130 sync_linked_regs(env, this_branch, src_reg, &linked_regs); 17131 sync_linked_regs(env, other_branch, &other_branch_regs[insn->src_reg], 17132 &linked_regs); 17133 } 17134 if (dst_reg->type == SCALAR_VALUE && dst_reg->id && 17135 !WARN_ON_ONCE(dst_reg->id != other_branch_regs[insn->dst_reg].id)) { 17136 sync_linked_regs(env, this_branch, dst_reg, &linked_regs); 17137 sync_linked_regs(env, other_branch, &other_branch_regs[insn->dst_reg], 17138 &linked_regs); 17139 } 17140 17141 /* if one pointer register is compared to another pointer 17142 * register check if PTR_MAYBE_NULL could be lifted. 17143 * E.g. register A - maybe null 17144 * register B - not null 17145 * for JNE A, B, ... - A is not null in the false branch; 17146 * for JEQ A, B, ... - A is not null in the true branch. 17147 * 17148 * Since PTR_TO_BTF_ID points to a kernel struct that does 17149 * not need to be null checked by the BPF program, i.e., 17150 * could be null even without PTR_MAYBE_NULL marking, so 17151 * only propagate nullness when neither reg is that type. 17152 */ 17153 if (!is_jmp32 && BPF_SRC(insn->code) == BPF_X && 17154 __is_pointer_value(false, src_reg) && __is_pointer_value(false, dst_reg) && 17155 base_type(src_reg->type) != PTR_TO_BTF_ID && 17156 base_type(dst_reg->type) != PTR_TO_BTF_ID) { 17157 eq_branch_regs = NULL; 17158 switch (opcode) { 17159 case BPF_JEQ: 17160 eq_branch_regs = other_branch_regs; 17161 break; 17162 case BPF_JNE: 17163 eq_branch_regs = regs; 17164 break; 17165 default: 17166 /* do nothing */ 17167 break; 17168 } 17169 if (eq_branch_regs) { 17170 /* src == dst && dst != NULL => src != NULL */ 17171 if (reg_not_null(env, dst_reg) && type_may_be_null(src_reg->type)) 17172 mark_ptr_not_null_reg(&eq_branch_regs[insn->src_reg]); 17173 /* src == dst && src != NULL => dst != NULL */ 17174 if (reg_not_null(env, src_reg) && type_may_be_null(dst_reg->type)) 17175 mark_ptr_not_null_reg(&eq_branch_regs[insn->dst_reg]); 17176 } 17177 } 17178 17179 /* detect if R == 0 where R is returned from bpf_map_lookup_elem(). 17180 * Also does the same detection for a register whose the value is 17181 * known to be 0. 17182 * NOTE: these optimizations below are related with pointer comparison 17183 * which will never be JMP32. 17184 */ 17185 if (!is_jmp32 && (opcode == BPF_JEQ || opcode == BPF_JNE) && 17186 type_may_be_null(dst_reg->type) && 17187 ((BPF_SRC(insn->code) == BPF_K && insn->imm == 0) || 17188 (BPF_SRC(insn->code) == BPF_X && bpf_register_is_null(src_reg)))) { 17189 /* 17190 * For BPF_X the zero is a property of this execution path, 17191 * hence src_reg has to be precise. 17192 */ 17193 if (BPF_SRC(insn->code) == BPF_X) { 17194 err = mark_chain_precision(env, insn->src_reg); 17195 if (err) 17196 return err; 17197 } 17198 /* Mark all identical registers in each branch as either 17199 * safe or unknown depending R == 0 or R != 0 conditional. 17200 */ 17201 mark_ptr_or_null_regs(this_branch, insn->dst_reg, 17202 opcode == BPF_JNE); 17203 mark_ptr_or_null_regs(other_branch, insn->dst_reg, 17204 opcode == BPF_JEQ); 17205 } else if (!try_match_pkt_pointers(insn, dst_reg, ®s[insn->src_reg], 17206 this_branch, other_branch) && 17207 is_pointer_value(env, insn->dst_reg)) { 17208 verbose(env, "R%d pointer comparison prohibited\n", 17209 insn->dst_reg); 17210 return -EACCES; 17211 } 17212 if (env->log.level & BPF_LOG_LEVEL) 17213 print_insn_state(env, this_branch, this_branch->curframe); 17214 return 0; 17215 } 17216 17217 /* verify BPF_LD_IMM64 instruction */ 17218 static int check_ld_imm(struct bpf_verifier_env *env, struct bpf_insn *insn) 17219 { 17220 struct bpf_insn_aux_data *aux = cur_aux(env); 17221 struct bpf_reg_state *regs = cur_regs(env); 17222 struct bpf_reg_state *dst_reg; 17223 struct bpf_map *map; 17224 int err; 17225 17226 if (BPF_SIZE(insn->code) != BPF_DW) { 17227 verbose(env, "invalid BPF_LD_IMM insn\n"); 17228 return -EINVAL; 17229 } 17230 17231 err = check_reg_arg(env, insn->dst_reg, DST_OP); 17232 if (err) 17233 return err; 17234 17235 dst_reg = ®s[insn->dst_reg]; 17236 bpf_diag_mod_begin(env, dst_reg, NULL, BPF_DIAG_MOD_WRITE); 17237 if (insn->src_reg == 0) { 17238 u64 imm = ((u64)(insn + 1)->imm << 32) | (u32)insn->imm; 17239 17240 dst_reg->type = SCALAR_VALUE; 17241 __mark_reg_known(®s[insn->dst_reg], imm); 17242 bpf_diag_mod_end(env); 17243 return 0; 17244 } 17245 17246 /* All special src_reg cases are listed below. From this point onwards 17247 * we either succeed and assign a corresponding dst_reg->type after 17248 * zeroing the offset, or fail and reject the program. 17249 */ 17250 mark_reg_known_zero(env, regs, insn->dst_reg); 17251 17252 if (insn->src_reg == BPF_PSEUDO_BTF_ID) { 17253 dst_reg->type = aux->btf_var.reg_type; 17254 switch (base_type(dst_reg->type)) { 17255 case PTR_TO_MEM: 17256 dst_reg->mem_size = aux->btf_var.mem_size; 17257 break; 17258 case PTR_TO_BTF_ID: 17259 dst_reg->btf = aux->btf_var.btf; 17260 dst_reg->btf_id = aux->btf_var.btf_id; 17261 break; 17262 default: 17263 verifier_bug(env, "pseudo btf id: unexpected dst reg type"); 17264 return -EFAULT; 17265 } 17266 bpf_diag_mod_end(env); 17267 return 0; 17268 } 17269 17270 if (insn->src_reg == BPF_PSEUDO_FUNC) { 17271 struct bpf_prog_aux *aux = env->prog->aux; 17272 u32 subprogno = bpf_find_subprog(env, 17273 env->insn_idx + insn->imm + 1); 17274 17275 if (!aux->func_info) { 17276 verbose(env, "missing btf func_info\n"); 17277 return -EINVAL; 17278 } 17279 if (aux->func_info_aux[subprogno].linkage != BTF_FUNC_STATIC) { 17280 verbose(env, "callback function not static\n"); 17281 return -EINVAL; 17282 } 17283 /* 17284 * When env->subprog_cnt == 1 this instruction won't be rewritten 17285 * to hold a real function address. Assume that no usable program 17286 * combines e.g. main and timer callback and just reject here. 17287 */ 17288 if (subprogno == 0) { 17289 verbose(env, "callback function cannot be the main program\n"); 17290 return -EINVAL; 17291 } 17292 17293 dst_reg->type = PTR_TO_FUNC; 17294 dst_reg->subprogno = subprogno; 17295 bpf_diag_mod_end(env); 17296 return 0; 17297 } 17298 17299 map = env->used_maps[aux->map_index]; 17300 17301 if (insn->src_reg == BPF_PSEUDO_MAP_VALUE || 17302 insn->src_reg == BPF_PSEUDO_MAP_IDX_VALUE) { 17303 if (map->map_type == BPF_MAP_TYPE_ARENA) { 17304 __mark_reg_unknown(env, dst_reg); 17305 dst_reg->map_ptr = map; 17306 bpf_diag_mod_end(env); 17307 return 0; 17308 } 17309 __mark_reg_known(dst_reg, aux->map_off); 17310 dst_reg->type = PTR_TO_MAP_VALUE; 17311 dst_reg->map_ptr = map; 17312 WARN_ON_ONCE(map->map_type != BPF_MAP_TYPE_INSN_ARRAY && 17313 map->max_entries != 1); 17314 /* We want reg->id to be same (0) as map_value is not distinct */ 17315 } else if (insn->src_reg == BPF_PSEUDO_MAP_FD || 17316 insn->src_reg == BPF_PSEUDO_MAP_IDX) { 17317 dst_reg->type = CONST_PTR_TO_MAP; 17318 dst_reg->map_ptr = map; 17319 } else { 17320 verifier_bug(env, "unexpected src reg value for ldimm64"); 17321 return -EFAULT; 17322 } 17323 17324 bpf_diag_mod_end(env); 17325 return 0; 17326 } 17327 17328 static bool may_access_skb(enum bpf_prog_type type) 17329 { 17330 switch (type) { 17331 case BPF_PROG_TYPE_SOCKET_FILTER: 17332 case BPF_PROG_TYPE_SCHED_CLS: 17333 case BPF_PROG_TYPE_SCHED_ACT: 17334 return true; 17335 default: 17336 return false; 17337 } 17338 } 17339 17340 /* verify safety of LD_ABS|LD_IND instructions: 17341 * - they can only appear in the programs where ctx == skb 17342 * - since they are wrappers of function calls, they scratch R1-R5 registers, 17343 * preserve R6-R9, and store return value into R0 17344 * 17345 * Implicit input: 17346 * ctx == skb == R6 == CTX 17347 * 17348 * Explicit input: 17349 * SRC == any register 17350 * IMM == 32-bit immediate 17351 * 17352 * Output: 17353 * R0 - 8/16/32-bit skb data converted to cpu endianness 17354 */ 17355 static int check_ld_abs(struct bpf_verifier_env *env, struct bpf_insn *insn) 17356 { 17357 struct bpf_verifier_state *state = env->cur_state; 17358 struct bpf_reg_state *regs = cur_regs(env); 17359 static const int ctx_reg = BPF_REG_6; 17360 u8 mode = BPF_MODE(insn->code); 17361 int i, err; 17362 17363 if (!may_access_skb(resolve_prog_type(env->prog))) { 17364 verbose(env, "BPF_LD_[ABS|IND] instructions not allowed for this program type\n"); 17365 return -EINVAL; 17366 } 17367 17368 for (i = state->curframe; i; i--) { 17369 if (state->frame[i]->in_callback_fn) { 17370 verbose(env, "cannot use BPF_LD_[ABS|IND] within callback\n"); 17371 return -EINVAL; 17372 } 17373 } 17374 17375 if (!env->ops->gen_ld_abs) { 17376 verifier_bug(env, "gen_ld_abs is null"); 17377 return -EFAULT; 17378 } 17379 17380 /* check whether implicit source operand (register R6) is readable */ 17381 err = check_reg_arg(env, ctx_reg, SRC_OP); 17382 if (err) 17383 return err; 17384 17385 /* Disallow usage of BPF_LD_[ABS|IND] with reference tracking, as 17386 * gen_ld_abs() may terminate the program at runtime, leading to 17387 * reference leak. 17388 */ 17389 err = check_resource_leak(env, false, true, "BPF_LD_[ABS|IND]"); 17390 if (err) 17391 return err; 17392 17393 if (regs[ctx_reg].type != PTR_TO_CTX) { 17394 verbose(env, 17395 "at the time of BPF_LD_ABS|IND R6 != pointer to skb\n"); 17396 return -EINVAL; 17397 } 17398 17399 if (mode == BPF_IND) { 17400 /* check explicit source operand */ 17401 err = check_reg_arg(env, insn->src_reg, SRC_OP); 17402 if (err) 17403 return err; 17404 } 17405 17406 err = check_ptr_off_reg(env, ®s[ctx_reg], ctx_reg); 17407 if (err < 0) 17408 return err; 17409 17410 /* reset caller saved regs to unreadable */ 17411 bpf_diag_record_caller_saved(env, regs); 17412 bpf_diag_mod_begin(env, ®s[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE); 17413 for (i = 0; i < CALLER_SAVED_REGS; i++) { 17414 bpf_mark_reg_not_init(env, ®s[caller_saved[i]]); 17415 check_reg_arg(env, caller_saved[i], DST_OP_NO_MARK); 17416 } 17417 17418 /* mark destination R0 register as readable, since it contains 17419 * the value fetched from the packet. 17420 * Already marked as written above. 17421 */ 17422 mark_reg_unknown(env, regs, BPF_REG_0); 17423 bpf_diag_mod_end(env); 17424 /* 17425 * See bpf_gen_ld_abs() which emits a hidden BPF_EXIT with r0=0 17426 * which must be explored by the verifier when in a subprog. 17427 */ 17428 if (env->cur_state->curframe) { 17429 struct bpf_verifier_state *branch; 17430 17431 mark_reg_scratched(env, BPF_REG_0); 17432 branch = push_stack(env, env->insn_idx + 1, env->insn_idx, false); 17433 if (IS_ERR(branch)) 17434 return PTR_ERR(branch); 17435 mark_reg_known_zero(env, regs, BPF_REG_0); 17436 err = prepare_func_exit(env, &env->insn_idx); 17437 if (err) 17438 return err; 17439 env->insn_idx--; 17440 } 17441 return 0; 17442 } 17443 17444 static bool return_retval_range(struct bpf_verifier_env *env, struct bpf_retval_range *range) 17445 { 17446 enum bpf_prog_type prog_type = resolve_prog_type(env->prog); 17447 17448 /* Default return value range. */ 17449 *range = retval_range(0, 1); 17450 17451 switch (prog_type) { 17452 case BPF_PROG_TYPE_CGROUP_SOCK_ADDR: 17453 switch (env->prog->expected_attach_type) { 17454 case BPF_CGROUP_UDP4_RECVMSG: 17455 case BPF_CGROUP_UDP6_RECVMSG: 17456 case BPF_CGROUP_UNIX_RECVMSG: 17457 case BPF_CGROUP_INET4_GETPEERNAME: 17458 case BPF_CGROUP_INET6_GETPEERNAME: 17459 case BPF_CGROUP_UNIX_GETPEERNAME: 17460 case BPF_CGROUP_INET4_GETSOCKNAME: 17461 case BPF_CGROUP_INET6_GETSOCKNAME: 17462 case BPF_CGROUP_UNIX_GETSOCKNAME: 17463 *range = retval_range(1, 1); 17464 break; 17465 case BPF_CGROUP_INET4_BIND: 17466 case BPF_CGROUP_INET6_BIND: 17467 *range = retval_range(0, 3); 17468 break; 17469 default: 17470 break; 17471 } 17472 break; 17473 case BPF_PROG_TYPE_CGROUP_SKB: 17474 if (env->prog->expected_attach_type == BPF_CGROUP_INET_EGRESS) 17475 *range = retval_range(0, 3); 17476 break; 17477 case BPF_PROG_TYPE_CGROUP_SOCK: 17478 case BPF_PROG_TYPE_SOCK_OPS: 17479 case BPF_PROG_TYPE_CGROUP_DEVICE: 17480 case BPF_PROG_TYPE_CGROUP_SYSCTL: 17481 case BPF_PROG_TYPE_CGROUP_SOCKOPT: 17482 break; 17483 case BPF_PROG_TYPE_RAW_TRACEPOINT: 17484 if (!env->prog->aux->attach_btf_id) 17485 return false; 17486 *range = retval_range(0, 0); 17487 break; 17488 case BPF_PROG_TYPE_TRACING: 17489 switch (env->prog->expected_attach_type) { 17490 case BPF_TRACE_FENTRY: 17491 case BPF_TRACE_FEXIT: 17492 case BPF_TRACE_FSESSION: 17493 case BPF_TRACE_FENTRY_MULTI: 17494 case BPF_TRACE_FEXIT_MULTI: 17495 case BPF_TRACE_FSESSION_MULTI: 17496 *range = retval_range(0, 0); 17497 break; 17498 case BPF_TRACE_RAW_TP: 17499 case BPF_MODIFY_RETURN: 17500 return false; 17501 case BPF_TRACE_ITER: 17502 default: 17503 break; 17504 } 17505 break; 17506 case BPF_PROG_TYPE_KPROBE: 17507 switch (env->prog->expected_attach_type) { 17508 case BPF_TRACE_KPROBE_SESSION: 17509 case BPF_TRACE_UPROBE_SESSION: 17510 break; 17511 default: 17512 return false; 17513 } 17514 break; 17515 case BPF_PROG_TYPE_SK_LOOKUP: 17516 *range = retval_range(SK_DROP, SK_PASS); 17517 break; 17518 17519 case BPF_PROG_TYPE_LSM: 17520 if (env->prog->expected_attach_type != BPF_LSM_CGROUP) { 17521 /* no range found, any return value is allowed */ 17522 if (!get_func_retval_range(env->prog, range)) 17523 return false; 17524 /* no restricted range, any return value is allowed */ 17525 if (range->minval == S32_MIN && range->maxval == S32_MAX) 17526 return false; 17527 range->return_32bit = true; 17528 } else if (!env->prog->aux->attach_func_proto->type) { 17529 /* Make sure programs that attach to void 17530 * hooks don't try to modify return value. 17531 */ 17532 *range = retval_range(1, 1); 17533 } 17534 break; 17535 17536 case BPF_PROG_TYPE_NETFILTER: 17537 *range = retval_range(NF_DROP, NF_ACCEPT); 17538 break; 17539 case BPF_PROG_TYPE_STRUCT_OPS: 17540 *range = retval_range(0, 0); 17541 break; 17542 case BPF_PROG_TYPE_EXT: 17543 /* freplace program can return anything as its return value 17544 * depends on the to-be-replaced kernel func or bpf program. 17545 */ 17546 default: 17547 return false; 17548 } 17549 17550 /* Continue calculating. */ 17551 17552 return true; 17553 } 17554 17555 static bool program_returns_void(struct bpf_verifier_env *env) 17556 { 17557 const struct bpf_prog *prog = env->prog; 17558 enum bpf_prog_type prog_type = prog->type; 17559 17560 switch (prog_type) { 17561 case BPF_PROG_TYPE_LSM: 17562 /* See return_retval_range, for BPF_LSM_CGROUP can be 0 or 0-1 depending on hook. */ 17563 if (prog->expected_attach_type != BPF_LSM_CGROUP && 17564 !prog->aux->attach_func_proto->type) 17565 return true; 17566 break; 17567 case BPF_PROG_TYPE_STRUCT_OPS: 17568 if (!prog->aux->attach_func_proto->type) 17569 return true; 17570 break; 17571 case BPF_PROG_TYPE_EXT: 17572 /* 17573 * If the actual program is an extension, let it 17574 * return void - attaching will succeed only if the 17575 * program being replaced also returns void, and since 17576 * it has passed verification its actual type doesn't matter. 17577 */ 17578 if (subprog_returns_void(env, 0)) 17579 return true; 17580 break; 17581 default: 17582 break; 17583 } 17584 return false; 17585 } 17586 17587 static int check_return_code(struct bpf_verifier_env *env, int regno, const char *reg_name) 17588 { 17589 const char *exit_ctx = "At program exit"; 17590 struct tnum enforce_attach_type_range = tnum_unknown; 17591 const struct bpf_prog *prog = env->prog; 17592 struct bpf_reg_state *reg = reg_state(env, regno); 17593 struct bpf_retval_range range = retval_range(0, 1); 17594 enum bpf_prog_type prog_type = resolve_prog_type(env->prog); 17595 struct bpf_func_state *frame = env->cur_state->frame[0]; 17596 const struct btf_type *reg_type, *ret_type = NULL; 17597 int err; 17598 17599 /* LSM and struct_ops func-ptr's return type could be "void" */ 17600 if (!frame->in_async_callback_fn && program_returns_void(env)) 17601 return 0; 17602 17603 if (prog_type == BPF_PROG_TYPE_STRUCT_OPS) { 17604 /* Allow a struct_ops program to return a referenced kptr if it 17605 * matches the operator's return type and is in its unmodified 17606 * form. A scalar zero (i.e., a null pointer) is also allowed. 17607 */ 17608 reg_type = reg->btf ? btf_type_by_id(reg->btf, reg->btf_id) : NULL; 17609 ret_type = btf_type_resolve_ptr(prog->aux->attach_btf, 17610 prog->aux->attach_func_proto->type, 17611 NULL); 17612 if (ret_type && ret_type == reg_type && reg_is_referenced(env, reg)) 17613 return __check_ptr_off_reg(env, reg, argno_from_reg(regno), false); 17614 } 17615 17616 /* eBPF calling convention is such that R0 is used 17617 * to return the value from eBPF program. 17618 * Make sure that it's readable at this time 17619 * of bpf_exit, which means that program wrote 17620 * something into it earlier 17621 */ 17622 err = check_reg_arg(env, regno, SRC_OP); 17623 if (err) 17624 return err; 17625 17626 if (is_pointer_value(env, regno)) { 17627 verbose(env, "R%d leaks addr as return value\n", regno); 17628 return -EACCES; 17629 } 17630 17631 if (frame->in_async_callback_fn) { 17632 exit_ctx = "At async callback return"; 17633 range = frame->callback_ret_range; 17634 goto enforce_retval; 17635 } 17636 17637 if (prog_type == BPF_PROG_TYPE_STRUCT_OPS && !ret_type) 17638 return 0; 17639 17640 if (prog_type == BPF_PROG_TYPE_CGROUP_SKB && (env->prog->expected_attach_type == BPF_CGROUP_INET_EGRESS)) 17641 enforce_attach_type_range = tnum_range(2, 3); 17642 17643 if (!return_retval_range(env, &range)) 17644 return 0; 17645 17646 enforce_retval: 17647 if (reg->type != SCALAR_VALUE) { 17648 verbose(env, "%s the register R%d is not a known value (%s)\n", 17649 exit_ctx, regno, reg_type_str(env, reg->type)); 17650 return -EINVAL; 17651 } 17652 17653 err = mark_chain_precision(env, regno); 17654 if (err) 17655 return err; 17656 17657 if (!retval_range_within(range, reg)) { 17658 verbose_invalid_scalar(env, reg, range, exit_ctx, reg_name); 17659 if (prog->expected_attach_type == BPF_LSM_CGROUP && 17660 prog_type == BPF_PROG_TYPE_LSM && 17661 !prog->aux->attach_func_proto->type) 17662 verbose(env, "Note, BPF_LSM_CGROUP that attach to void LSM hooks can't modify return value!\n"); 17663 return -EINVAL; 17664 } 17665 17666 if (!tnum_is_unknown(enforce_attach_type_range) && 17667 tnum_in(enforce_attach_type_range, reg->var_off)) 17668 env->prog->enforce_expected_attach_type = 1; 17669 return 0; 17670 } 17671 17672 static int check_global_subprog_return_code(struct bpf_verifier_env *env) 17673 { 17674 struct bpf_reg_state *reg = reg_state(env, BPF_REG_0); 17675 struct bpf_func_state *cur_frame = cur_func(env); 17676 int err; 17677 17678 if (subprog_returns_void(env, cur_frame->subprogno)) 17679 return 0; 17680 17681 err = check_reg_arg(env, BPF_REG_0, SRC_OP); 17682 if (err) 17683 return err; 17684 17685 /* Pointers to arena are safe to pass between subprograms. */ 17686 if (is_arena_reg(env, BPF_REG_0)) 17687 return 0; 17688 17689 if (is_pointer_value(env, BPF_REG_0)) { 17690 verbose(env, "R%d leaks addr as return value\n", BPF_REG_0); 17691 return -EACCES; 17692 } 17693 17694 if (reg->type != SCALAR_VALUE) { 17695 verbose(env, "At subprogram exit the register R0 is not a scalar value (%s)\n", 17696 reg_type_str(env, reg->type)); 17697 return -EINVAL; 17698 } 17699 17700 return 0; 17701 } 17702 17703 /* Bitmask with 1s for all caller saved registers */ 17704 #define ALL_CALLER_SAVED_REGS ((1u << CALLER_SAVED_REGS) - 1) 17705 17706 /* True if do_misc_fixups() replaces calls to helper number 'imm', 17707 * replacement patch is presumed to follow bpf_fastcall contract 17708 * (see mark_fastcall_pattern_for_call() below). 17709 */ 17710 bool bpf_verifier_inlines_helper_call(struct bpf_verifier_env *env, s32 imm) 17711 { 17712 switch (imm) { 17713 #ifdef CONFIG_X86_64 17714 case BPF_FUNC_get_smp_processor_id: 17715 #ifdef CONFIG_SMP 17716 case BPF_FUNC_get_current_task_btf: 17717 case BPF_FUNC_get_current_task: 17718 #endif 17719 return env->prog->jit_requested && bpf_jit_supports_percpu_insn(); 17720 #endif 17721 default: 17722 return false; 17723 } 17724 } 17725 17726 /* If @call is a kfunc or helper call, fills @cs and returns true, 17727 * otherwise returns false. 17728 */ 17729 bool bpf_get_call_summary(struct bpf_verifier_env *env, struct bpf_insn *call, 17730 struct bpf_call_summary *cs) 17731 { 17732 struct bpf_call_arg_meta meta; 17733 const struct bpf_func_proto *fn; 17734 int i; 17735 17736 if (bpf_helper_call(call)) { 17737 if (bpf_get_helper_proto(env, call->imm, &fn) < 0) 17738 /* error would be reported later */ 17739 return false; 17740 cs->fastcall = fn->allow_fastcall && 17741 (bpf_verifier_inlines_helper_call(env, call->imm) || 17742 bpf_jit_inlines_helper_call(call->imm)); 17743 cs->is_void = fn->ret_type == RET_VOID; 17744 cs->num_params = 0; 17745 for (i = 0; i < ARRAY_SIZE(fn->arg_type); ++i) { 17746 if (fn->arg_type[i] == ARG_DONTCARE) 17747 break; 17748 cs->num_params++; 17749 } 17750 return true; 17751 } 17752 17753 if (bpf_pseudo_kfunc_call(call)) { 17754 int err; 17755 17756 err = bpf_fetch_kfunc_arg_meta(env, call->imm, call->off, &meta); 17757 if (err < 0) 17758 /* error would be reported later */ 17759 return false; 17760 cs->num_params = btf_type_vlen(meta.func_proto); 17761 cs->fastcall = meta.kfunc_flags & KF_FASTCALL; 17762 cs->is_void = btf_type_is_void(btf_type_by_id(meta.btf, meta.func_proto->type)); 17763 return true; 17764 } 17765 17766 return false; 17767 } 17768 17769 /* LLVM define a bpf_fastcall function attribute. 17770 * This attribute means that function scratches only some of 17771 * the caller saved registers defined by ABI. 17772 * For BPF the set of such registers could be defined as follows: 17773 * - R0 is scratched only if function is non-void; 17774 * - R1-R5 are scratched only if corresponding parameter type is defined 17775 * in the function prototype. 17776 * 17777 * The contract between kernel and clang allows to simultaneously use 17778 * such functions and maintain backwards compatibility with old 17779 * kernels that don't understand bpf_fastcall calls: 17780 * 17781 * - for bpf_fastcall calls clang allocates registers as-if relevant r0-r5 17782 * registers are not scratched by the call; 17783 * 17784 * - as a post-processing step, clang visits each bpf_fastcall call and adds 17785 * spill/fill for every live r0-r5; 17786 * 17787 * - stack offsets used for the spill/fill are allocated as lowest 17788 * stack offsets in whole function and are not used for any other 17789 * purposes; 17790 * 17791 * - when kernel loads a program, it looks for such patterns 17792 * (bpf_fastcall function surrounded by spills/fills) and checks if 17793 * spill/fill stack offsets are used exclusively in fastcall patterns; 17794 * 17795 * - if so, and if verifier or current JIT inlines the call to the 17796 * bpf_fastcall function (e.g. a helper call), kernel removes unnecessary 17797 * spill/fill pairs; 17798 * 17799 * - when old kernel loads a program, presence of spill/fill pairs 17800 * keeps BPF program valid, albeit slightly less efficient. 17801 * 17802 * For example: 17803 * 17804 * r1 = 1; 17805 * r2 = 2; 17806 * *(u64 *)(r10 - 8) = r1; r1 = 1; 17807 * *(u64 *)(r10 - 16) = r2; r2 = 2; 17808 * call %[to_be_inlined] --> call %[to_be_inlined] 17809 * r2 = *(u64 *)(r10 - 16); r0 = r1; 17810 * r1 = *(u64 *)(r10 - 8); r0 += r2; 17811 * r0 = r1; exit; 17812 * r0 += r2; 17813 * exit; 17814 * 17815 * The purpose of mark_fastcall_pattern_for_call is to: 17816 * - look for such patterns; 17817 * - mark spill and fill instructions in env->insn_aux_data[*].fastcall_pattern; 17818 * - mark set env->insn_aux_data[*].fastcall_spills_num for call instruction; 17819 * - update env->subprog_info[*]->fastcall_stack_off to find an offset 17820 * at which bpf_fastcall spill/fill stack slots start; 17821 * - update env->subprog_info[*]->keep_fastcall_stack. 17822 * 17823 * The .fastcall_pattern and .fastcall_stack_off are used by 17824 * check_fastcall_stack_contract() to check if every stack access to 17825 * fastcall spill/fill stack slot originates from spill/fill 17826 * instructions, members of fastcall patterns. 17827 * 17828 * If such condition holds true for a subprogram, fastcall patterns could 17829 * be rewritten by remove_fastcall_spills_fills(). 17830 * Otherwise bpf_fastcall patterns are not changed in the subprogram 17831 * (code, presumably, generated by an older clang version). 17832 * 17833 * For example, it is *not* safe to remove spill/fill below: 17834 * 17835 * r1 = 1; 17836 * *(u64 *)(r10 - 8) = r1; r1 = 1; 17837 * call %[to_be_inlined] --> call %[to_be_inlined] 17838 * r1 = *(u64 *)(r10 - 8); r0 = *(u64 *)(r10 - 8); <---- wrong !!! 17839 * r0 = *(u64 *)(r10 - 8); r0 += r1; 17840 * r0 += r1; exit; 17841 * exit; 17842 * 17843 * Both uses of the marks assume that a pattern is entered at its first 17844 * spill and thus executes as a unit, hence a pattern is not grown past 17845 * an instruction targeted by a jump. 17846 */ 17847 static void mark_fastcall_pattern_for_call(struct bpf_verifier_env *env, 17848 struct bpf_subprog_info *subprog, 17849 int insn_idx, s16 lowest_off) 17850 { 17851 struct bpf_insn *insns = env->prog->insnsi, *stx, *ldx; 17852 struct bpf_insn *call = &env->prog->insnsi[insn_idx]; 17853 u32 clobbered_regs_mask; 17854 struct bpf_call_summary cs; 17855 u32 expected_regs_mask; 17856 s16 off; 17857 int i; 17858 17859 if (!bpf_get_call_summary(env, call, &cs)) 17860 return; 17861 17862 /* A bitmask specifying which caller saved registers are clobbered 17863 * by a call to a helper/kfunc *as if* this helper/kfunc follows 17864 * bpf_fastcall contract: 17865 * - includes R0 if function is non-void; 17866 * - includes R1-R5 if corresponding parameter has is described 17867 * in the function prototype. 17868 */ 17869 clobbered_regs_mask = GENMASK(cs.num_params, cs.is_void ? 1 : 0); 17870 /* e.g. if helper call clobbers r{0,1}, expect r{2,3,4,5} in the pattern */ 17871 expected_regs_mask = ~clobbered_regs_mask & ALL_CALLER_SAVED_REGS; 17872 17873 /* match pairs of form: 17874 * 17875 * *(u64 *)(r10 - Y) = rX (where Y % 8 == 0) 17876 * ... 17877 * call %[to_be_inlined] 17878 * ... 17879 * rX = *(u64 *)(r10 - Y) 17880 */ 17881 for (i = 1, off = lowest_off; i <= ARRAY_SIZE(caller_saved); ++i, off += BPF_REG_SIZE) { 17882 if (insn_idx - i < 0 || insn_idx + i >= env->prog->len) 17883 break; 17884 /* stx/ldx/call must not be a jump targets, a jump to the first stx is fine */ 17885 if (bpf_is_jump_target(env, insn_idx - i + 1) || 17886 bpf_is_jump_target(env, insn_idx + i)) 17887 break; 17888 stx = &insns[insn_idx - i]; 17889 ldx = &insns[insn_idx + i]; 17890 /* must be a stack spill/fill pair */ 17891 if (stx->code != (BPF_STX | BPF_MEM | BPF_DW) || 17892 ldx->code != (BPF_LDX | BPF_MEM | BPF_DW) || 17893 stx->dst_reg != BPF_REG_10 || 17894 ldx->src_reg != BPF_REG_10) 17895 break; 17896 /* must be a spill/fill for the same reg */ 17897 if (stx->src_reg != ldx->dst_reg) 17898 break; 17899 /* must be one of the previously unseen registers */ 17900 if ((BIT(stx->src_reg) & expected_regs_mask) == 0) 17901 break; 17902 /* must be a spill/fill for the same expected offset, 17903 * no need to check offset alignment, BPF_DW stack access 17904 * is always 8-byte aligned. 17905 */ 17906 if (stx->off != off || ldx->off != off) 17907 break; 17908 expected_regs_mask &= ~BIT(stx->src_reg); 17909 env->insn_aux_data[insn_idx - i].fastcall_pattern = 1; 17910 env->insn_aux_data[insn_idx + i].fastcall_pattern = 1; 17911 } 17912 if (i == 1) 17913 return; 17914 17915 /* Conditionally set 'fastcall_spills_num' to allow forward 17916 * compatibility when more helper functions are marked as 17917 * bpf_fastcall at compile time than current kernel supports, e.g: 17918 * 17919 * 1: *(u64 *)(r10 - 8) = r1 17920 * 2: call A ;; assume A is bpf_fastcall for current kernel 17921 * 3: r1 = *(u64 *)(r10 - 8) 17922 * 4: *(u64 *)(r10 - 8) = r1 17923 * 5: call B ;; assume B is not bpf_fastcall for current kernel 17924 * 6: r1 = *(u64 *)(r10 - 8) 17925 * 17926 * There is no need to block bpf_fastcall rewrite for such program. 17927 * Set 'fastcall_pattern' for both calls to keep check_fastcall_stack_contract() happy, 17928 * don't set 'fastcall_spills_num' for call B so that remove_fastcall_spills_fills() 17929 * does not remove spill/fill pair {4,6}. 17930 */ 17931 if (cs.fastcall) 17932 env->insn_aux_data[insn_idx].fastcall_spills_num = i - 1; 17933 else 17934 subprog->keep_fastcall_stack = 1; 17935 subprog->fastcall_stack_off = min(subprog->fastcall_stack_off, off); 17936 } 17937 17938 static int mark_fastcall_patterns(struct bpf_verifier_env *env) 17939 { 17940 struct bpf_subprog_info *subprog = env->subprog_info; 17941 struct bpf_insn *insn; 17942 s16 lowest_off; 17943 int s, i; 17944 17945 for (s = 0; s < env->subprog_cnt; ++s, ++subprog) { 17946 /* find lowest stack spill offset used in this subprog */ 17947 lowest_off = 0; 17948 for (i = subprog->start; i < (subprog + 1)->start; ++i) { 17949 insn = env->prog->insnsi + i; 17950 if (insn->code != (BPF_STX | BPF_MEM | BPF_DW) || 17951 insn->dst_reg != BPF_REG_10) 17952 continue; 17953 lowest_off = min(lowest_off, insn->off); 17954 } 17955 /* use this offset to find fastcall patterns */ 17956 for (i = subprog->start; i < (subprog + 1)->start; ++i) { 17957 insn = env->prog->insnsi + i; 17958 if (insn->code != (BPF_JMP | BPF_CALL)) 17959 continue; 17960 mark_fastcall_pattern_for_call(env, subprog, i, lowest_off); 17961 } 17962 } 17963 return 0; 17964 } 17965 17966 static void adjust_btf_func(struct bpf_verifier_env *env) 17967 { 17968 struct bpf_prog_aux *aux = env->prog->aux; 17969 int i; 17970 17971 if (!aux->func_info) 17972 return; 17973 17974 /* func_info is not available for hidden subprogs */ 17975 for (i = 0; i < env->subprog_cnt - env->hidden_subprog_cnt; i++) 17976 aux->func_info[i].insn_off = env->subprog_info[i].start; 17977 } 17978 17979 /* Find id in idset and increment its count, or add new entry */ 17980 static void idset_cnt_inc(struct bpf_idset *idset, u32 id) 17981 { 17982 u32 i; 17983 17984 for (i = 0; i < idset->num_ids; i++) { 17985 if (idset->entries[i].id == id) { 17986 idset->entries[i].cnt++; 17987 return; 17988 } 17989 } 17990 /* New id */ 17991 if (idset->num_ids < BPF_ID_MAP_SIZE) { 17992 idset->entries[idset->num_ids].id = id; 17993 idset->entries[idset->num_ids].cnt = 1; 17994 idset->num_ids++; 17995 } 17996 } 17997 17998 /* Find id in idset and return its count, or 0 if not found */ 17999 static u32 idset_cnt_get(struct bpf_idset *idset, u32 id) 18000 { 18001 u32 i; 18002 18003 for (i = 0; i < idset->num_ids; i++) { 18004 if (idset->entries[i].id == id) 18005 return idset->entries[i].cnt; 18006 } 18007 return 0; 18008 } 18009 18010 /* 18011 * Clear singular scalar ids in a state. 18012 * A register with a non-zero id is called singular if no other register shares 18013 * the same base id. Such registers can be treated as independent (id=0). 18014 */ 18015 void bpf_clear_singular_ids(struct bpf_verifier_env *env, 18016 struct bpf_verifier_state *st) 18017 { 18018 struct bpf_idset *idset = &env->idset_scratch; 18019 struct bpf_func_state *func; 18020 struct bpf_reg_state *reg; 18021 18022 idset->num_ids = 0; 18023 18024 bpf_for_each_reg_in_vstate(st, func, reg, ({ 18025 if (reg->type != SCALAR_VALUE) 18026 continue; 18027 if (!reg->id) 18028 continue; 18029 idset_cnt_inc(idset, reg->id & ~BPF_ADD_CONST); 18030 })); 18031 18032 bpf_for_each_reg_in_vstate(st, func, reg, ({ 18033 if (reg->type != SCALAR_VALUE) 18034 continue; 18035 if (!reg->id) 18036 continue; 18037 if (idset_cnt_get(idset, reg->id & ~BPF_ADD_CONST) == 1) 18038 clear_scalar_id(reg); 18039 })); 18040 } 18041 18042 /* Return true if it's OK to have the same insn return a different type. */ 18043 static bool reg_type_mismatch_ok(enum bpf_reg_type type) 18044 { 18045 switch (base_type(type)) { 18046 case PTR_TO_CTX: 18047 case PTR_TO_SOCKET: 18048 case PTR_TO_SOCK_COMMON: 18049 case PTR_TO_TCP_SOCK: 18050 case PTR_TO_XDP_SOCK: 18051 case PTR_TO_BTF_ID: 18052 case PTR_TO_ARENA: 18053 return false; 18054 case PTR_TO_MEM: 18055 return !bpf_may_fault_on_deref(type); 18056 default: 18057 return true; 18058 } 18059 } 18060 18061 /* If an instruction was previously used with particular pointer types, then we 18062 * need to be careful to avoid cases such as the below, where it may be ok 18063 * for one branch accessing the pointer, but not ok for the other branch: 18064 * 18065 * R1 = sock_ptr 18066 * goto X; 18067 * ... 18068 * R1 = some_other_valid_ptr; 18069 * goto X; 18070 * ... 18071 * R2 = *(u32 *)(R1 + 0); 18072 */ 18073 static bool reg_type_mismatch(enum bpf_reg_type src, enum bpf_reg_type prev) 18074 { 18075 return src != prev && (!reg_type_mismatch_ok(src) || 18076 !reg_type_mismatch_ok(prev)); 18077 } 18078 18079 static bool is_ptr_to_mem(enum bpf_reg_type type) 18080 { 18081 return base_type(type) == PTR_TO_MEM; 18082 } 18083 18084 static enum bpf_reg_type merge_ptr_types(enum bpf_reg_type type_a, 18085 enum bpf_reg_type type_b) 18086 { 18087 bool to_mem = is_ptr_to_mem(type_a) || is_ptr_to_mem(type_b); 18088 enum bpf_reg_type type_merged = to_mem ? PTR_TO_MEM : PTR_TO_BTF_ID; 18089 18090 if (bpf_may_fault_on_deref(type_a) || bpf_may_fault_on_deref(type_b)) 18091 type_merged |= to_mem ? MEM_RDONLY | PTR_UNTRUSTED : 18092 PTR_UNTRUSTED; 18093 else 18094 type_merged |= ((type_a | type_b) & MEM_RDONLY); 18095 return type_merged; 18096 } 18097 18098 static int save_aux_ptr_type(struct bpf_verifier_env *env, enum bpf_reg_type type, 18099 bool allow_trust_mismatch) 18100 { 18101 enum bpf_reg_type *prev_type = &env->insn_aux_data[env->insn_idx].ptr_type; 18102 18103 if (*prev_type == NOT_INIT) { 18104 /* Saw a valid insn 18105 * dst_reg = *(u32 *)(src_reg + off) 18106 * save type to validate intersecting paths 18107 */ 18108 *prev_type = type; 18109 } else if (reg_type_mismatch(type, *prev_type)) { 18110 /* Abuser program is trying to use the same insn 18111 * dst_reg = *(u32*) (src_reg + off) 18112 * with different pointer types: 18113 * src_reg == ctx in one branch and 18114 * src_reg == stack|map in some other branch. 18115 * Reject it. 18116 */ 18117 if (allow_trust_mismatch && 18118 bpf_is_ptr_to_mem_or_btf_id(type) && 18119 bpf_is_ptr_to_mem_or_btf_id(*prev_type)) { 18120 /* 18121 * Have to support a use case when one path through 18122 * the program yields a TRUSTED pointer while another 18123 * is UNTRUSTED. Merge them into a type which keeps 18124 * the BPF_PROBE_MEM/BPF_PROBE_MEMSX rewrite when 18125 * either side needs it. 18126 */ 18127 *prev_type = merge_ptr_types(type, *prev_type); 18128 } else { 18129 verbose(env, "same insn cannot be used with different pointers\n"); 18130 return -EINVAL; 18131 } 18132 } 18133 18134 return 0; 18135 } 18136 18137 enum { 18138 PROCESS_BPF_EXIT = 1, 18139 INSN_IDX_UPDATED = 2, 18140 }; 18141 18142 static int process_bpf_exit_full(struct bpf_verifier_env *env, 18143 bool *do_print_state, 18144 bool exception_exit) 18145 { 18146 struct bpf_func_state *cur_frame = cur_func(env); 18147 18148 /* We must do check_reference_leak here before 18149 * prepare_func_exit to handle the case when 18150 * state->curframe > 0, it may be a callback function, 18151 * for which reference_state must match caller reference 18152 * state when it exits. 18153 */ 18154 int err = check_resource_leak(env, exception_exit, 18155 exception_exit || !env->cur_state->curframe, 18156 exception_exit ? "bpf_throw" : 18157 "BPF_EXIT instruction in main prog"); 18158 if (err) 18159 return err; 18160 18161 /* The side effect of the prepare_func_exit which is 18162 * being skipped is that it frees bpf_func_state. 18163 * Typically, process_bpf_exit will only be hit with 18164 * outermost exit. copy_verifier_state in pop_stack will 18165 * handle freeing of any extra bpf_func_state left over 18166 * from not processing all nested function exits. We 18167 * also skip return code checks as they are not needed 18168 * for exceptional exits. 18169 */ 18170 if (exception_exit) 18171 return PROCESS_BPF_EXIT; 18172 18173 if (env->cur_state->curframe) { 18174 /* exit from nested function */ 18175 err = prepare_func_exit(env, &env->insn_idx); 18176 if (err) 18177 return err; 18178 *do_print_state = true; 18179 return INSN_IDX_UPDATED; 18180 } 18181 18182 /* 18183 * Return from a regular global subprogram differs from return 18184 * from the main program or async/exception callback. 18185 * Main program exit implies return code restrictions 18186 * that depend on program type. 18187 * Exit from exception callback is equivalent to main program exit. 18188 * Exit from async callback implies return code restrictions 18189 * that depend on async scheduling mechanism. 18190 */ 18191 if (cur_frame->subprogno && 18192 !cur_frame->in_async_callback_fn && 18193 !cur_frame->in_exception_callback_fn) 18194 err = check_global_subprog_return_code(env); 18195 else 18196 err = check_return_code(env, BPF_REG_0, "R0"); 18197 if (err) 18198 return err; 18199 return PROCESS_BPF_EXIT; 18200 } 18201 18202 static int indirect_jump_min_max_index(struct bpf_verifier_env *env, 18203 int regno, 18204 struct bpf_map *map, 18205 u32 *pmin_index, u32 *pmax_index) 18206 { 18207 struct bpf_reg_state *reg = reg_state(env, regno); 18208 u64 min_index = reg_umin(reg); 18209 u64 max_index = reg_umax(reg); 18210 const u32 size = 8; 18211 18212 if (min_index > (u64) U32_MAX * size) { 18213 verbose(env, "the sum of R%u umin_value %llu is too big\n", regno, reg_umin(reg)); 18214 return -ERANGE; 18215 } 18216 if (max_index > (u64) U32_MAX * size) { 18217 verbose(env, "the sum of R%u umax_value %llu is too big\n", regno, reg_umax(reg)); 18218 return -ERANGE; 18219 } 18220 18221 min_index /= size; 18222 max_index /= size; 18223 18224 if (max_index >= map->max_entries) { 18225 verbose(env, "R%u points to outside of jump table: [%llu,%llu] max_entries %u\n", 18226 regno, min_index, max_index, map->max_entries); 18227 return -EINVAL; 18228 } 18229 18230 *pmin_index = min_index; 18231 *pmax_index = max_index; 18232 return 0; 18233 } 18234 18235 /* gotox *dst_reg */ 18236 static int check_indirect_jump(struct bpf_verifier_env *env, struct bpf_insn *insn) 18237 { 18238 struct bpf_verifier_state *other_branch; 18239 struct bpf_reg_state *dst_reg; 18240 struct bpf_map *map; 18241 u32 min_index, max_index; 18242 int err = 0; 18243 int n; 18244 int i; 18245 18246 dst_reg = reg_state(env, insn->dst_reg); 18247 if (dst_reg->type != PTR_TO_INSN) { 18248 verbose(env, "R%d has type %s, expected PTR_TO_INSN\n", 18249 insn->dst_reg, reg_type_str(env, dst_reg->type)); 18250 return -EINVAL; 18251 } 18252 18253 map = dst_reg->map_ptr; 18254 if (verifier_bug_if(!map, env, "R%d has an empty map pointer", insn->dst_reg)) 18255 return -EFAULT; 18256 18257 if (verifier_bug_if(map->map_type != BPF_MAP_TYPE_INSN_ARRAY, env, 18258 "R%d has incorrect map type %d", insn->dst_reg, map->map_type)) 18259 return -EFAULT; 18260 18261 err = indirect_jump_min_max_index(env, insn->dst_reg, map, &min_index, &max_index); 18262 if (err) 18263 return err; 18264 18265 /* Ensure that the buffer is large enough */ 18266 if (!env->gotox_tmp_buf || env->gotox_tmp_buf->cnt < max_index - min_index + 1) { 18267 env->gotox_tmp_buf = bpf_iarray_realloc(env->gotox_tmp_buf, 18268 max_index - min_index + 1); 18269 if (!env->gotox_tmp_buf) 18270 return -ENOMEM; 18271 } 18272 18273 n = bpf_copy_insn_array_uniq(map, min_index, max_index, env->gotox_tmp_buf->items); 18274 if (n < 0) 18275 return n; 18276 if (n == 0) { 18277 verbose(env, "register R%d doesn't point to any offset in map id=%d\n", 18278 insn->dst_reg, map->id); 18279 return -EINVAL; 18280 } 18281 18282 for (i = 0; i < n - 1; i++) { 18283 mark_indirect_target(env, env->gotox_tmp_buf->items[i]); 18284 other_branch = push_stack(env, env->gotox_tmp_buf->items[i], 18285 env->insn_idx, env->cur_state->speculative); 18286 if (IS_ERR(other_branch)) 18287 return PTR_ERR(other_branch); 18288 } 18289 env->insn_idx = env->gotox_tmp_buf->items[n-1]; 18290 mark_indirect_target(env, env->insn_idx); 18291 return INSN_IDX_UPDATED; 18292 } 18293 18294 static int do_check_insn(struct bpf_verifier_env *env, bool *do_print_state) 18295 { 18296 int err; 18297 struct bpf_insn *insn = &env->prog->insnsi[env->insn_idx]; 18298 u8 class = BPF_CLASS(insn->code); 18299 18300 switch (class) { 18301 case BPF_ALU: 18302 case BPF_ALU64: 18303 return check_alu_op(env, insn); 18304 18305 case BPF_LDX: 18306 return check_load_mem(env, insn, false, 18307 BPF_MODE(insn->code) == BPF_MEMSX, 18308 true, "ldx"); 18309 18310 case BPF_STX: 18311 if (BPF_MODE(insn->code) == BPF_ATOMIC) 18312 return check_atomic(env, insn); 18313 return check_store_reg(env, insn, false); 18314 18315 case BPF_ST: { 18316 /* Handle stack arg write (store immediate) */ 18317 if (is_stack_arg_st(insn)) { 18318 struct bpf_verifier_state *vstate = env->cur_state; 18319 struct bpf_func_state *state = vstate->frame[vstate->curframe]; 18320 18321 return check_stack_arg_write(env, state, insn->off, NULL); 18322 } 18323 18324 enum bpf_reg_type dst_reg_type; 18325 18326 err = check_reg_arg(env, insn->dst_reg, SRC_OP); 18327 if (err) 18328 return err; 18329 18330 dst_reg_type = cur_regs(env)[insn->dst_reg].type; 18331 18332 err = check_mem_access(env, env->insn_idx, cur_regs(env) + insn->dst_reg, argno_from_reg(insn->dst_reg), 18333 insn->off, BPF_SIZE(insn->code), 18334 BPF_WRITE, -1, false, false); 18335 if (err) 18336 return err; 18337 18338 return save_aux_ptr_type(env, dst_reg_type, false); 18339 } 18340 case BPF_JMP: 18341 case BPF_JMP32: { 18342 u8 opcode = BPF_OP(insn->code); 18343 18344 env->jmps_processed++; 18345 if (opcode == BPF_CALL) { 18346 if (env->cur_state->active_locks) { 18347 if ((insn->src_reg == BPF_REG_0 && 18348 insn->imm != BPF_FUNC_spin_unlock && 18349 insn->imm != BPF_FUNC_kptr_xchg) || 18350 (insn->src_reg == BPF_PSEUDO_KFUNC_CALL && 18351 !kfunc_spin_allowed(env, insn->imm, insn->off))) { 18352 verbose(env, 18353 "function calls are not allowed while holding a lock\n"); 18354 bpf_diag_ctx_active( 18355 env, env->insn_idx, 18356 "function call", BPF_DIAG_CONTEXT_LOCK, 18357 "Release the BPF spin lock before making this call, or move the call outside the locked region."); 18358 return -EINVAL; 18359 } 18360 } 18361 mark_reg_scratched(env, BPF_REG_0); 18362 if (bpf_in_stack_arg_cnt(&env->subprog_info[cur_func(env)->subprogno])) 18363 cur_func(env)->no_stack_arg_load = true; 18364 if (insn->src_reg == BPF_PSEUDO_CALL) 18365 return check_func_call(env, insn, &env->insn_idx); 18366 if (insn->src_reg == BPF_PSEUDO_KFUNC_CALL) 18367 return check_kfunc_call(env, insn, &env->insn_idx); 18368 return check_helper_call(env, insn, &env->insn_idx); 18369 } else if (opcode == BPF_JA) { 18370 if (BPF_SRC(insn->code) == BPF_X) 18371 return check_indirect_jump(env, insn); 18372 18373 if (class == BPF_JMP) 18374 env->insn_idx += insn->off + 1; 18375 else 18376 env->insn_idx += insn->imm + 1; 18377 return INSN_IDX_UPDATED; 18378 } else if (opcode == BPF_EXIT) { 18379 return process_bpf_exit_full(env, do_print_state, false); 18380 } 18381 return check_cond_jmp_op(env, insn, &env->insn_idx); 18382 } 18383 case BPF_LD: { 18384 u8 mode = BPF_MODE(insn->code); 18385 18386 if (mode == BPF_ABS || mode == BPF_IND) 18387 return check_ld_abs(env, insn); 18388 18389 if (mode == BPF_IMM) { 18390 err = check_ld_imm(env, insn); 18391 if (err) 18392 return err; 18393 18394 env->insn_idx++; 18395 sanitize_mark_insn_seen(env); 18396 } 18397 return 0; 18398 } 18399 } 18400 /* all class values are handled above. silence compiler warning */ 18401 return -EFAULT; 18402 } 18403 18404 static int do_check(struct bpf_verifier_env *env) 18405 { 18406 bool pop_log = !(env->log.level & BPF_LOG_LEVEL2); 18407 struct bpf_verifier_state *state = env->cur_state; 18408 struct bpf_insn *insns = env->prog->insnsi; 18409 int insn_cnt = env->prog->len; 18410 bool do_print_state = false; 18411 int prev_insn_idx = -1; 18412 18413 for (;;) { 18414 struct bpf_insn *insn; 18415 struct bpf_insn_aux_data *insn_aux; 18416 int err; 18417 18418 /* reset current history entry on each new instruction */ 18419 env->cur_hist_ent = NULL; 18420 18421 env->prev_insn_idx = prev_insn_idx; 18422 if (env->insn_idx >= insn_cnt) { 18423 verbose(env, "invalid insn idx %d insn_cnt %d\n", 18424 env->insn_idx, insn_cnt); 18425 return -EFAULT; 18426 } 18427 18428 insn = &insns[env->insn_idx]; 18429 insn_aux = &env->insn_aux_data[env->insn_idx]; 18430 18431 account_processed_insn(env); 18432 18433 if (env->insn_processed > BPF_COMPLEXITY_LIMIT_INSNS) { 18434 verbose(env, 18435 "BPF program is too large. Processed %d insn\n", 18436 env->insn_processed); 18437 return -E2BIG; 18438 } 18439 18440 state->last_insn_idx = env->prev_insn_idx; 18441 state->insn_idx = env->insn_idx; 18442 /* 18443 * Record the incoming edge so active and queued paths use the same 18444 * branch-recording path. A zero-offset conditional has identical 18445 * successors, so its outcome cannot be reconstructed from the edge. 18446 */ 18447 if (!state->speculative && prev_insn_idx >= 0 && prev_insn_idx < insn_cnt) { 18448 struct bpf_insn *prev_insn = &insns[prev_insn_idx]; 18449 int fallthrough_idx = prev_insn_idx + 1; 18450 int branch_idx = prev_insn_idx + bpf_jmp_offset(prev_insn) + 1; 18451 u8 class = BPF_CLASS(prev_insn->code); 18452 u8 opcode = BPF_OP(prev_insn->code); 18453 18454 if ((class == BPF_JMP || class == BPF_JMP32) && 18455 opcode != BPF_JA && opcode != BPF_CALL && opcode != BPF_EXIT && 18456 opcode <= BPF_JCOND && branch_idx != fallthrough_idx) { 18457 if (env->insn_idx == branch_idx) 18458 bpf_diag_record_branch(env, prev_insn_idx, true); 18459 else if (env->insn_idx == fallthrough_idx) 18460 bpf_diag_record_branch(env, prev_insn_idx, false); 18461 } 18462 } 18463 18464 if (bpf_is_prune_point(env, env->insn_idx)) { 18465 err = bpf_is_state_visited(env, env->insn_idx); 18466 if (err < 0) 18467 return err; 18468 if (err == 1) { 18469 /* found equivalent state, can prune the search */ 18470 if (env->log.level & BPF_LOG_LEVEL) { 18471 if (do_print_state) 18472 verbose(env, "\nfrom %d to %d%s: safe\n", 18473 env->prev_insn_idx, env->insn_idx, 18474 env->cur_state->speculative ? 18475 " (speculative execution)" : ""); 18476 else 18477 verbose(env, "%d: safe\n", env->insn_idx); 18478 } 18479 goto process_bpf_exit; 18480 } 18481 } 18482 18483 if (bpf_is_jmp_point(env, env->insn_idx)) { 18484 err = bpf_push_jmp_history(env, state, 0, 0, 0, 0); 18485 if (err) 18486 return err; 18487 } 18488 18489 if (signal_pending(current)) 18490 return -EAGAIN; 18491 18492 if (need_resched()) 18493 cond_resched(); 18494 18495 if (env->log.level & BPF_LOG_LEVEL2 && do_print_state) { 18496 verbose(env, "\nfrom %d to %d%s:", 18497 env->prev_insn_idx, env->insn_idx, 18498 env->cur_state->speculative ? 18499 " (speculative execution)" : ""); 18500 print_verifier_state(env, state, state->curframe, true); 18501 do_print_state = false; 18502 } 18503 18504 if (env->log.level & BPF_LOG_LEVEL) { 18505 if (verifier_state_scratched(env)) 18506 print_insn_state(env, state, state->curframe); 18507 18508 verbose_linfo(env, env->insn_idx, "; "); 18509 env->prev_log_pos = env->log.end_pos; 18510 verbose(env, "%d: ", env->insn_idx); 18511 bpf_verbose_insn(env, insn); 18512 verbose(env, "\n"); 18513 env->prev_insn_print_pos = env->log.end_pos - env->prev_log_pos; 18514 env->prev_log_pos = env->log.end_pos; 18515 } 18516 18517 if (bpf_prog_is_offloaded(env->prog->aux)) { 18518 err = bpf_prog_offload_verify_insn(env, env->insn_idx, 18519 env->prev_insn_idx); 18520 if (err) 18521 return err; 18522 } 18523 18524 sanitize_mark_insn_seen(env); 18525 prev_insn_idx = env->insn_idx; 18526 18527 /* Sanity check: precomputed constants must match verifier state */ 18528 if (!state->speculative && insn_aux->const_reg_mask) { 18529 struct bpf_reg_state *regs = cur_regs(env); 18530 u16 mask = insn_aux->const_reg_mask; 18531 18532 for (int r = 0; r < ARRAY_SIZE(insn_aux->const_reg_vals); r++) { 18533 u32 cval = insn_aux->const_reg_vals[r]; 18534 18535 if (!(mask & BIT(r))) 18536 continue; 18537 if (regs[r].type != SCALAR_VALUE) 18538 continue; 18539 if (!tnum_is_const(regs[r].var_off)) 18540 continue; 18541 if (verifier_bug_if((u32)regs[r].var_off.value != cval, 18542 env, "const R%d: %u != %llu", 18543 r, cval, regs[r].var_off.value)) 18544 return -EFAULT; 18545 } 18546 } 18547 18548 /* Reduce verification complexity by stopping speculative path 18549 * verification when a nospec is encountered. 18550 */ 18551 if (state->speculative && insn_aux->nospec) 18552 goto process_bpf_exit; 18553 18554 err = do_check_insn(env, &do_print_state); 18555 if (error_recoverable_with_nospec(err) && state->speculative) { 18556 /* Prevent this speculative path from ever reaching the 18557 * insn that would have been unsafe to execute. 18558 */ 18559 insn_aux->nospec = true; 18560 /* If it was an ADD/SUB insn, potentially remove any 18561 * markings for alu sanitization. 18562 */ 18563 insn_aux->alu_state = 0; 18564 goto process_bpf_exit; 18565 } else if (err < 0) { 18566 return err; 18567 } else if (err == PROCESS_BPF_EXIT) { 18568 goto process_bpf_exit; 18569 } else if (err == INSN_IDX_UPDATED) { 18570 } else if (err == 0) { 18571 env->insn_idx++; 18572 } 18573 18574 if (state->speculative && insn_aux->nospec_result) { 18575 /* If we are on a path that performed a jump-op, this 18576 * may skip a nospec patched-in after the jump. This can 18577 * currently never happen because nospec_result is only 18578 * used for the write-ops 18579 * `*(size*)(dst_reg+off)=src_reg|imm32` and helper 18580 * calls. These must never skip the following insn 18581 * (i.e., bpf_insn_successors()'s opcode_info.can_jump 18582 * is false). Still, add a warning to document this in 18583 * case nospec_result is used elsewhere in the future. 18584 * 18585 * All non-branch instructions have a single 18586 * fall-through edge. For these, nospec_result should 18587 * already work. 18588 */ 18589 if (verifier_bug_if((BPF_CLASS(insn->code) == BPF_JMP || 18590 BPF_CLASS(insn->code) == BPF_JMP32) && 18591 BPF_OP(insn->code) != BPF_CALL, env, 18592 "speculation barrier after jump instruction may not have the desired effect")) 18593 return -EFAULT; 18594 process_bpf_exit: 18595 account_current_path(env); 18596 mark_verifier_state_scratched(env); 18597 err = bpf_update_branch_counts(env, env->cur_state); 18598 if (err) 18599 return err; 18600 err = pop_stack(env, &prev_insn_idx, &env->insn_idx, 18601 pop_log); 18602 if (err < 0) { 18603 if (err != -ENOENT) 18604 return err; 18605 break; 18606 } else { 18607 do_print_state = true; 18608 continue; 18609 } 18610 } 18611 } 18612 18613 return 0; 18614 } 18615 18616 static int find_btf_percpu_datasec(struct btf *btf) 18617 { 18618 const struct btf_type *t; 18619 const char *tname; 18620 int i, n; 18621 18622 /* 18623 * Both vmlinux and module each have their own ".data..percpu" 18624 * DATASECs in BTF. So for module's case, we need to skip vmlinux BTF 18625 * types to look at only module's own BTF types. 18626 */ 18627 n = btf_nr_types(btf); 18628 for (i = btf_named_start_id(btf, true); i < n; i++) { 18629 t = btf_type_by_id(btf, i); 18630 if (BTF_INFO_KIND(t->info) != BTF_KIND_DATASEC) 18631 continue; 18632 18633 tname = btf_name_by_offset(btf, t->name_off); 18634 if (!strcmp(tname, ".data..percpu")) 18635 return i; 18636 } 18637 18638 return -ENOENT; 18639 } 18640 18641 /* 18642 * Add btf to the env->used_btfs array. If needed, refcount the 18643 * corresponding kernel module. To simplify caller's logic 18644 * in case of error or if btf was added before the function 18645 * decreases the btf refcount. 18646 */ 18647 static int __add_used_btf(struct bpf_verifier_env *env, struct btf *btf) 18648 { 18649 struct btf_mod_pair *btf_mod; 18650 int ret = 0; 18651 int i; 18652 18653 /* check whether we recorded this BTF (and maybe module) already */ 18654 for (i = 0; i < env->used_btf_cnt; i++) 18655 if (env->used_btfs[i].btf == btf) 18656 goto ret_put; 18657 18658 if (env->signature) { 18659 verbose(env, "signed program cannot bind any BTF\n"); 18660 ret = -EACCES; 18661 goto ret_put; 18662 } 18663 if (env->used_btf_cnt >= MAX_USED_BTFS) { 18664 verbose(env, "The total number of btfs per program has reached the limit of %u\n", 18665 MAX_USED_BTFS); 18666 ret = -E2BIG; 18667 goto ret_put; 18668 } 18669 18670 btf_mod = &env->used_btfs[env->used_btf_cnt]; 18671 btf_mod->btf = btf; 18672 btf_mod->module = NULL; 18673 18674 /* if we reference variables from kernel module, bump its refcount */ 18675 if (btf_is_module(btf)) { 18676 btf_mod->module = btf_try_get_module(btf); 18677 if (!btf_mod->module) { 18678 ret = -ENXIO; 18679 goto ret_put; 18680 } 18681 } 18682 18683 env->used_btf_cnt++; 18684 return 0; 18685 18686 ret_put: 18687 /* Either error or this BTF was already added */ 18688 btf_put(btf); 18689 return ret; 18690 } 18691 18692 /* replace pseudo btf_id with kernel symbol address */ 18693 static int __check_pseudo_btf_id(struct bpf_verifier_env *env, 18694 struct bpf_insn *insn, 18695 struct bpf_insn_aux_data *aux, 18696 struct btf *btf) 18697 { 18698 const struct btf_var_secinfo *vsi; 18699 const struct btf_type *datasec; 18700 const struct btf_type *t; 18701 const char *sym_name; 18702 bool percpu = false; 18703 u32 type, id = insn->imm; 18704 s32 datasec_id; 18705 u64 addr; 18706 int i; 18707 18708 t = btf_type_by_id(btf, id); 18709 if (!t) { 18710 verbose(env, "ldimm64 insn specifies invalid btf_id %d.\n", id); 18711 return -ENOENT; 18712 } 18713 18714 if (!btf_type_is_var(t) && !btf_type_is_func(t)) { 18715 verbose(env, "pseudo btf_id %d in ldimm64 isn't KIND_VAR or KIND_FUNC\n", id); 18716 return -EINVAL; 18717 } 18718 18719 sym_name = btf_name_by_offset(btf, t->name_off); 18720 addr = kallsyms_lookup_name(sym_name); 18721 if (!addr) { 18722 verbose(env, "ldimm64 failed to find the address for kernel symbol '%s'.\n", 18723 sym_name); 18724 return -ENOENT; 18725 } 18726 insn[0].imm = (u32)addr; 18727 insn[1].imm = addr >> 32; 18728 18729 if (btf_type_is_func(t)) { 18730 aux->btf_var.reg_type = PTR_TO_MEM | MEM_RDONLY; 18731 aux->btf_var.mem_size = 0; 18732 return 0; 18733 } 18734 18735 datasec_id = find_btf_percpu_datasec(btf); 18736 if (datasec_id > 0) { 18737 datasec = btf_type_by_id(btf, datasec_id); 18738 for_each_vsi(i, datasec, vsi) { 18739 if (vsi->type == id) { 18740 percpu = true; 18741 break; 18742 } 18743 } 18744 } 18745 18746 type = t->type; 18747 t = btf_type_skip_modifiers(btf, type, NULL); 18748 if (percpu) { 18749 aux->btf_var.reg_type = PTR_TO_BTF_ID | MEM_PERCPU; 18750 aux->btf_var.btf = btf; 18751 aux->btf_var.btf_id = type; 18752 } else if (!btf_type_is_struct(t)) { 18753 const struct btf_type *ret; 18754 const char *tname; 18755 u32 tsize; 18756 18757 /* resolve the type size of ksym. */ 18758 ret = btf_resolve_size(btf, t, &tsize); 18759 if (IS_ERR(ret)) { 18760 tname = btf_name_by_offset(btf, t->name_off); 18761 verbose(env, "ldimm64 unable to resolve the size of type '%s': %ld\n", 18762 tname, PTR_ERR(ret)); 18763 return -EINVAL; 18764 } 18765 aux->btf_var.reg_type = PTR_TO_MEM | MEM_RDONLY; 18766 aux->btf_var.mem_size = tsize; 18767 } else { 18768 aux->btf_var.reg_type = PTR_TO_BTF_ID; 18769 aux->btf_var.btf = btf; 18770 aux->btf_var.btf_id = type; 18771 } 18772 18773 return 0; 18774 } 18775 18776 static int check_pseudo_btf_id(struct bpf_verifier_env *env, 18777 struct bpf_insn *insn, 18778 struct bpf_insn_aux_data *aux) 18779 { 18780 struct btf *btf; 18781 int btf_fd; 18782 int err; 18783 18784 btf_fd = insn[1].imm; 18785 if (btf_fd) { 18786 btf = btf_get_by_fd(btf_fd); 18787 if (IS_ERR(btf)) { 18788 verbose(env, "invalid module BTF object FD specified.\n"); 18789 return -EINVAL; 18790 } 18791 } else { 18792 if (!btf_vmlinux) { 18793 verbose(env, "kernel is missing BTF, make sure CONFIG_DEBUG_INFO_BTF=y is specified in Kconfig.\n"); 18794 return -EINVAL; 18795 } 18796 btf_get(btf_vmlinux); 18797 btf = btf_vmlinux; 18798 } 18799 18800 err = __check_pseudo_btf_id(env, insn, aux, btf); 18801 if (err) { 18802 btf_put(btf); 18803 return err; 18804 } 18805 18806 return __add_used_btf(env, btf); 18807 } 18808 18809 static bool is_tracing_prog_type(enum bpf_prog_type type) 18810 { 18811 switch (type) { 18812 case BPF_PROG_TYPE_KPROBE: 18813 case BPF_PROG_TYPE_TRACEPOINT: 18814 case BPF_PROG_TYPE_PERF_EVENT: 18815 case BPF_PROG_TYPE_RAW_TRACEPOINT: 18816 case BPF_PROG_TYPE_RAW_TRACEPOINT_WRITABLE: 18817 return true; 18818 default: 18819 return false; 18820 } 18821 } 18822 18823 static bool bpf_map_is_cgroup_storage(struct bpf_map *map) 18824 { 18825 return (map->map_type == BPF_MAP_TYPE_CGROUP_STORAGE || 18826 map->map_type == BPF_MAP_TYPE_PERCPU_CGROUP_STORAGE); 18827 } 18828 18829 static int check_map_prog_compatibility(struct bpf_verifier_env *env, 18830 struct bpf_map *map, 18831 struct bpf_prog *prog) 18832 18833 { 18834 enum bpf_prog_type prog_type = resolve_prog_type(prog); 18835 18836 if (map->excl_prog_sha && 18837 memcmp(map->excl_prog_sha, prog->digest, SHA256_DIGEST_SIZE)) { 18838 verbose(env, "program's hash doesn't match map's excl_prog_hash\n"); 18839 return -EACCES; 18840 } 18841 18842 if (btf_record_has_field(map->record, BPF_LIST_HEAD) || 18843 btf_record_has_field(map->record, BPF_RB_ROOT)) { 18844 if (is_tracing_prog_type(prog_type)) { 18845 verbose(env, "tracing progs cannot use bpf_{list_head,rb_root} yet\n"); 18846 return -EINVAL; 18847 } 18848 } 18849 18850 if (btf_record_has_field(map->record, BPF_SPIN_LOCK | BPF_RES_SPIN_LOCK)) { 18851 if (prog_type == BPF_PROG_TYPE_SOCKET_FILTER) { 18852 verbose(env, "socket filter progs cannot use bpf_spin_lock yet\n"); 18853 return -EINVAL; 18854 } 18855 } 18856 18857 if (btf_record_has_field(map->record, BPF_SPIN_LOCK)) { 18858 if (is_tracing_prog_type(prog_type)) { 18859 verbose(env, "tracing progs cannot use bpf_spin_lock yet\n"); 18860 return -EINVAL; 18861 } 18862 } 18863 18864 if ((bpf_prog_is_offloaded(prog->aux) || bpf_map_is_offloaded(map)) && 18865 !bpf_offload_prog_map_match(prog, map)) { 18866 verbose(env, "offload device mismatch between prog and map\n"); 18867 return -EINVAL; 18868 } 18869 18870 if (map->map_type == BPF_MAP_TYPE_STRUCT_OPS) { 18871 verbose(env, "bpf_struct_ops map cannot be used in prog\n"); 18872 return -EINVAL; 18873 } 18874 18875 if (prog->sleepable) 18876 switch (map->map_type) { 18877 case BPF_MAP_TYPE_HASH: 18878 case BPF_MAP_TYPE_RHASH: 18879 case BPF_MAP_TYPE_LRU_HASH: 18880 case BPF_MAP_TYPE_ARRAY: 18881 case BPF_MAP_TYPE_PERCPU_HASH: 18882 case BPF_MAP_TYPE_PERCPU_ARRAY: 18883 case BPF_MAP_TYPE_LRU_PERCPU_HASH: 18884 case BPF_MAP_TYPE_LPM_TRIE: 18885 case BPF_MAP_TYPE_ARRAY_OF_MAPS: 18886 case BPF_MAP_TYPE_HASH_OF_MAPS: 18887 case BPF_MAP_TYPE_RINGBUF: 18888 case BPF_MAP_TYPE_USER_RINGBUF: 18889 case BPF_MAP_TYPE_INODE_STORAGE: 18890 case BPF_MAP_TYPE_SK_STORAGE: 18891 case BPF_MAP_TYPE_TASK_STORAGE: 18892 case BPF_MAP_TYPE_CGRP_STORAGE: 18893 case BPF_MAP_TYPE_QUEUE: 18894 case BPF_MAP_TYPE_STACK: 18895 case BPF_MAP_TYPE_ARENA: 18896 case BPF_MAP_TYPE_INSN_ARRAY: 18897 case BPF_MAP_TYPE_PROG_ARRAY: 18898 break; 18899 default: 18900 verbose(env, 18901 "Sleepable programs can only use array, hash, ringbuf and local storage maps\n"); 18902 return -EINVAL; 18903 } 18904 18905 if (bpf_map_is_cgroup_storage(map) && 18906 bpf_cgroup_storage_assign(env->prog->aux, map)) { 18907 verbose(env, "only one cgroup storage of each type is allowed\n"); 18908 return -EBUSY; 18909 } 18910 18911 if (map->map_type == BPF_MAP_TYPE_ARENA) { 18912 if (env->prog->aux->arena) { 18913 verbose(env, "Only one arena per program\n"); 18914 return -EBUSY; 18915 } 18916 if (!env->allow_ptr_leaks || !env->bpf_capable) { 18917 verbose(env, "CAP_BPF and CAP_PERFMON are required to use arena\n"); 18918 return -EPERM; 18919 } 18920 if (!env->prog->jit_requested) { 18921 verbose(env, "JIT is required to use arena\n"); 18922 return -EOPNOTSUPP; 18923 } 18924 if (!bpf_jit_supports_arena()) { 18925 verbose(env, "JIT doesn't support arena\n"); 18926 return -EOPNOTSUPP; 18927 } 18928 env->prog->aux->arena = (void *)map; 18929 env->prog->jit_required = true; 18930 if (!bpf_arena_get_user_vm_start(env->prog->aux->arena)) { 18931 verbose(env, "arena's user address must be set via map_extra or mmap()\n"); 18932 return -EINVAL; 18933 } 18934 } 18935 18936 return 0; 18937 } 18938 18939 static int __add_used_map(struct bpf_verifier_env *env, struct bpf_map *map) 18940 { 18941 int i, err; 18942 18943 /* check whether we recorded this map already */ 18944 for (i = 0; i < env->used_map_cnt; i++) 18945 if (env->used_maps[i] == map) 18946 return i; 18947 18948 if (env->signature && 18949 env->prog->aux->sig.verdict == BPF_SIG_VERIFIED) { 18950 verbose(env, "signed program cannot bind map '%s' not covered by the signature\n", 18951 map->name); 18952 return -EACCES; 18953 } 18954 if (env->used_map_cnt >= MAX_USED_MAPS) { 18955 verbose(env, "The total number of maps per program has reached the limit of %u\n", 18956 MAX_USED_MAPS); 18957 return -E2BIG; 18958 } 18959 18960 err = check_map_prog_compatibility(env, map, env->prog); 18961 if (err) 18962 return err; 18963 18964 if (env->prog->sleepable) 18965 atomic64_inc(&map->sleepable_refcnt); 18966 18967 /* hold the map. If the program is rejected by verifier, 18968 * the map will be released by release_maps() or it 18969 * will be used by the valid program until it's unloaded 18970 * and all maps are released in bpf_free_used_maps() 18971 */ 18972 bpf_map_inc(map); 18973 18974 env->used_maps[env->used_map_cnt++] = map; 18975 18976 if (map->map_type == BPF_MAP_TYPE_INSN_ARRAY) { 18977 err = bpf_insn_array_init(map, env->prog); 18978 if (err) { 18979 verbose(env, "Failed to properly initialize insn array\n"); 18980 return err; 18981 } 18982 env->insn_array_maps[env->insn_array_map_cnt++] = map; 18983 env->prog->jit_required = true; 18984 } 18985 18986 return env->used_map_cnt - 1; 18987 } 18988 18989 /* Add map behind fd to used maps list, if it's not already there, and return 18990 * its index. 18991 * Returns <0 on error, or >= 0 index, on success. 18992 */ 18993 static int add_used_map(struct bpf_verifier_env *env, int fd) 18994 { 18995 struct bpf_map *map; 18996 CLASS(fd, f)(fd); 18997 18998 map = __bpf_map_get(f); 18999 if (IS_ERR(map)) { 19000 verbose(env, "fd %d is not pointing to valid bpf_map\n", fd); 19001 return PTR_ERR(map); 19002 } 19003 19004 return __add_used_map(env, map); 19005 } 19006 19007 static int fd_array_get_map_idx_continuous(struct bpf_verifier_env *env, u32 idx) 19008 { 19009 struct bpf_map *map; 19010 19011 if (idx >= env->fd_array_cnt) { 19012 verbose(env, "fd_idx %u out of bounds, fd_array_cnt %u\n", 19013 idx, env->fd_array_cnt); 19014 return -EINVAL; 19015 } 19016 map = fd_slot_map(env->fd_array[idx]); 19017 if (!map) { 19018 verbose(env, "fd_idx %u is not a map\n", idx); 19019 return -EINVAL; 19020 } 19021 return __add_used_map(env, map); 19022 } 19023 19024 static int fd_array_get_map_idx_sparse(struct bpf_verifier_env *env, u32 idx) 19025 { 19026 int fd; 19027 19028 if (copy_from_bpfptr_offset(&fd, env->fd_array_raw, 19029 (size_t)idx * sizeof(fd), sizeof(fd))) 19030 return -EFAULT; 19031 return add_used_map(env, fd); 19032 } 19033 19034 static int fd_array_get_map_idx(struct bpf_verifier_env *env, u32 idx) 19035 { 19036 if (env->fd_array) 19037 return fd_array_get_map_idx_continuous(env, idx); 19038 if (env->signature) { 19039 verbose(env, "signed program must bind maps via a continuous fd_array (fd_array_cnt)\n"); 19040 return -EACCES; 19041 } 19042 if (!bpfptr_is_null(env->fd_array_raw)) 19043 return fd_array_get_map_idx_sparse(env, idx); 19044 19045 verbose(env, "fd_idx without fd_array is invalid\n"); 19046 return -EPROTO; 19047 } 19048 19049 static int check_alu_fields(struct bpf_verifier_env *env, struct bpf_insn *insn) 19050 { 19051 u8 class = BPF_CLASS(insn->code); 19052 u8 opcode = BPF_OP(insn->code); 19053 19054 switch (opcode) { 19055 case BPF_NEG: 19056 if (BPF_SRC(insn->code) != BPF_K || insn->src_reg != BPF_REG_0 || 19057 insn->off != 0 || insn->imm != 0) { 19058 verbose(env, "BPF_NEG uses reserved fields\n"); 19059 return -EINVAL; 19060 } 19061 return 0; 19062 case BPF_END: 19063 if (insn->src_reg != BPF_REG_0 || insn->off != 0 || 19064 (insn->imm != 16 && insn->imm != 32 && insn->imm != 64) || 19065 (class == BPF_ALU64 && BPF_SRC(insn->code) != BPF_TO_LE)) { 19066 verbose(env, "BPF_END uses reserved fields\n"); 19067 return -EINVAL; 19068 } 19069 return 0; 19070 case BPF_MOV: 19071 if (BPF_SRC(insn->code) == BPF_X) { 19072 if (class == BPF_ALU) { 19073 if ((insn->off != 0 && insn->off != 8 && insn->off != 16) || 19074 insn->imm) { 19075 verbose(env, "BPF_MOV uses reserved fields\n"); 19076 return -EINVAL; 19077 } 19078 } else if (insn->off == BPF_ADDR_SPACE_CAST) { 19079 if (insn->imm != 1 && insn->imm != 1u << 16) { 19080 verbose(env, "addr_space_cast insn can only convert between address space 1 and 0\n"); 19081 return -EINVAL; 19082 } 19083 } else if ((insn->off != 0 && insn->off != 8 && 19084 insn->off != 16 && insn->off != 32) || insn->imm) { 19085 verbose(env, "BPF_MOV uses reserved fields\n"); 19086 return -EINVAL; 19087 } 19088 } else if (insn->src_reg != BPF_REG_0 || insn->off != 0) { 19089 verbose(env, "BPF_MOV uses reserved fields\n"); 19090 return -EINVAL; 19091 } 19092 return 0; 19093 case BPF_ADD: 19094 case BPF_SUB: 19095 case BPF_AND: 19096 case BPF_OR: 19097 case BPF_XOR: 19098 case BPF_LSH: 19099 case BPF_RSH: 19100 case BPF_ARSH: 19101 case BPF_MUL: 19102 case BPF_DIV: 19103 case BPF_MOD: 19104 if (BPF_SRC(insn->code) == BPF_X) { 19105 if (insn->imm != 0 || (insn->off != 0 && insn->off != 1) || 19106 (insn->off == 1 && opcode != BPF_MOD && opcode != BPF_DIV)) { 19107 verbose(env, "BPF_ALU uses reserved fields\n"); 19108 return -EINVAL; 19109 } 19110 } else if (insn->src_reg != BPF_REG_0 || 19111 (insn->off != 0 && insn->off != 1) || 19112 (insn->off == 1 && opcode != BPF_MOD && opcode != BPF_DIV)) { 19113 verbose(env, "BPF_ALU uses reserved fields\n"); 19114 return -EINVAL; 19115 } 19116 return 0; 19117 default: 19118 verbose(env, "invalid BPF_ALU opcode %x\n", opcode); 19119 return -EINVAL; 19120 } 19121 } 19122 19123 static int check_jmp_fields(struct bpf_verifier_env *env, struct bpf_insn *insn) 19124 { 19125 u8 class = BPF_CLASS(insn->code); 19126 u8 opcode = BPF_OP(insn->code); 19127 19128 switch (opcode) { 19129 case BPF_CALL: 19130 if (BPF_SRC(insn->code) != BPF_K || 19131 (insn->src_reg != BPF_PSEUDO_KFUNC_CALL && insn->off != 0) || 19132 (insn->src_reg != BPF_REG_0 && insn->src_reg != BPF_PSEUDO_CALL && 19133 insn->src_reg != BPF_PSEUDO_KFUNC_CALL) || 19134 insn->dst_reg != BPF_REG_0 || class == BPF_JMP32) { 19135 verbose(env, "BPF_CALL uses reserved fields\n"); 19136 return -EINVAL; 19137 } 19138 return 0; 19139 case BPF_JA: 19140 if (BPF_SRC(insn->code) == BPF_X) { 19141 if (insn->src_reg != BPF_REG_0 || insn->imm != 0 || insn->off != 0) { 19142 verbose(env, "BPF_JA|BPF_X uses reserved fields\n"); 19143 return -EINVAL; 19144 } 19145 } else if (insn->src_reg != BPF_REG_0 || insn->dst_reg != BPF_REG_0 || 19146 (class == BPF_JMP && insn->imm != 0) || 19147 (class == BPF_JMP32 && insn->off != 0)) { 19148 verbose(env, "BPF_JA uses reserved fields\n"); 19149 return -EINVAL; 19150 } 19151 return 0; 19152 case BPF_EXIT: 19153 if (BPF_SRC(insn->code) != BPF_K || insn->imm != 0 || 19154 insn->src_reg != BPF_REG_0 || insn->dst_reg != BPF_REG_0 || 19155 class == BPF_JMP32) { 19156 verbose(env, "BPF_EXIT uses reserved fields\n"); 19157 return -EINVAL; 19158 } 19159 return 0; 19160 case BPF_JCOND: 19161 if (insn->code != (BPF_JMP | BPF_JCOND) || insn->src_reg != BPF_MAY_GOTO || 19162 insn->dst_reg || insn->imm) { 19163 verbose(env, "invalid may_goto imm %d\n", insn->imm); 19164 return -EINVAL; 19165 } 19166 return 0; 19167 default: 19168 if (BPF_SRC(insn->code) == BPF_X) { 19169 if (insn->imm != 0) { 19170 verbose(env, "BPF_JMP/JMP32 uses reserved fields\n"); 19171 return -EINVAL; 19172 } 19173 } else if (insn->src_reg != BPF_REG_0) { 19174 verbose(env, "BPF_JMP/JMP32 uses reserved fields\n"); 19175 return -EINVAL; 19176 } 19177 return 0; 19178 } 19179 } 19180 19181 static int check_insn_fields(struct bpf_verifier_env *env, struct bpf_insn *insn) 19182 { 19183 switch (BPF_CLASS(insn->code)) { 19184 case BPF_ALU: 19185 case BPF_ALU64: 19186 return check_alu_fields(env, insn); 19187 case BPF_LDX: 19188 if ((BPF_MODE(insn->code) != BPF_MEM && BPF_MODE(insn->code) != BPF_MEMSX) || 19189 insn->imm != 0) { 19190 verbose(env, "BPF_LDX uses reserved fields\n"); 19191 return -EINVAL; 19192 } 19193 return 0; 19194 case BPF_STX: 19195 if (BPF_MODE(insn->code) == BPF_ATOMIC) 19196 return 0; 19197 if (BPF_MODE(insn->code) != BPF_MEM || insn->imm != 0) { 19198 verbose(env, "BPF_STX uses reserved fields\n"); 19199 return -EINVAL; 19200 } 19201 return 0; 19202 case BPF_ST: 19203 if (BPF_MODE(insn->code) != BPF_MEM || insn->src_reg != BPF_REG_0) { 19204 verbose(env, "BPF_ST uses reserved fields\n"); 19205 return -EINVAL; 19206 } 19207 return 0; 19208 case BPF_JMP: 19209 case BPF_JMP32: 19210 return check_jmp_fields(env, insn); 19211 case BPF_LD: { 19212 u8 mode = BPF_MODE(insn->code); 19213 19214 if (mode == BPF_ABS || mode == BPF_IND) { 19215 if (insn->dst_reg != BPF_REG_0 || insn->off != 0 || 19216 BPF_SIZE(insn->code) == BPF_DW || 19217 (mode == BPF_ABS && insn->src_reg != BPF_REG_0)) { 19218 verbose(env, "BPF_LD_[ABS|IND] uses reserved fields\n"); 19219 return -EINVAL; 19220 } 19221 } else if (mode != BPF_IMM) { 19222 verbose(env, "invalid BPF_LD mode\n"); 19223 return -EINVAL; 19224 } 19225 return 0; 19226 } 19227 default: 19228 verbose(env, "unknown insn class %d\n", BPF_CLASS(insn->code)); 19229 return -EINVAL; 19230 } 19231 } 19232 19233 /* 19234 * Check that insns are sane and rewrite pseudo imm in ld_imm64 instructions: 19235 * 19236 * 1. if it accesses map FD, replace it with actual map pointer. 19237 * 2. if it accesses btf_id of a VAR, replace it with pointer to the var. 19238 * 19239 * NOTE: btf_vmlinux is required for converting pseudo btf_id. 19240 */ 19241 static int check_and_resolve_insns(struct bpf_verifier_env *env) 19242 { 19243 struct bpf_insn *insn = env->prog->insnsi; 19244 int insn_cnt = env->prog->len; 19245 int i, err; 19246 19247 err = bpf_prog_calc_tag(env->prog); 19248 if (err) 19249 return err; 19250 19251 for (i = 0; i < insn_cnt; i++, insn++) { 19252 if (insn->dst_reg >= MAX_BPF_REG && 19253 !is_stack_arg_st(insn) && !is_stack_arg_stx(insn)) { 19254 verbose(env, "R%d is invalid\n", insn->dst_reg); 19255 return -EINVAL; 19256 } 19257 if (insn->src_reg >= MAX_BPF_REG && !is_stack_arg_ldx(insn)) { 19258 verbose(env, "R%d is invalid\n", insn->src_reg); 19259 return -EINVAL; 19260 } 19261 if (insn[0].code == (BPF_LD | BPF_IMM | BPF_DW)) { 19262 struct bpf_insn_aux_data *aux; 19263 struct bpf_map *map; 19264 int map_idx; 19265 u64 addr; 19266 19267 if (i == insn_cnt - 1 || insn[1].code != 0 || 19268 insn[1].dst_reg != 0 || insn[1].src_reg != 0 || 19269 insn[1].off != 0) { 19270 verbose(env, "invalid bpf_ld_imm64 insn\n"); 19271 return -EINVAL; 19272 } 19273 19274 if (insn[0].off != 0) { 19275 verbose(env, "BPF_LD_IMM64 uses reserved fields\n"); 19276 return -EINVAL; 19277 } 19278 19279 if (insn[0].src_reg == 0) 19280 /* valid generic load 64-bit imm */ 19281 goto next_insn; 19282 19283 if (insn[0].src_reg == BPF_PSEUDO_BTF_ID) { 19284 aux = &env->insn_aux_data[i]; 19285 err = check_pseudo_btf_id(env, insn, aux); 19286 if (err) 19287 return err; 19288 goto next_insn; 19289 } 19290 19291 if (insn[0].src_reg == BPF_PSEUDO_FUNC) { 19292 aux = &env->insn_aux_data[i]; 19293 aux->ptr_type = PTR_TO_FUNC; 19294 goto next_insn; 19295 } 19296 19297 /* In final convert_pseudo_ld_imm64() step, this is 19298 * converted into regular 64-bit imm load insn. 19299 */ 19300 switch (insn[0].src_reg) { 19301 case BPF_PSEUDO_MAP_VALUE: 19302 case BPF_PSEUDO_MAP_IDX_VALUE: 19303 break; 19304 case BPF_PSEUDO_MAP_FD: 19305 case BPF_PSEUDO_MAP_IDX: 19306 if (insn[1].imm == 0) 19307 break; 19308 fallthrough; 19309 default: 19310 verbose(env, "unrecognized bpf_ld_imm64 insn\n"); 19311 return -EINVAL; 19312 } 19313 19314 switch (insn[0].src_reg) { 19315 case BPF_PSEUDO_MAP_IDX_VALUE: 19316 case BPF_PSEUDO_MAP_IDX: 19317 map_idx = fd_array_get_map_idx(env, insn[0].imm); 19318 break; 19319 default: 19320 if (env->signature) { 19321 verbose(env, "signed program cannot reference a map by fd, only via fd_array index\n"); 19322 return -EINVAL; 19323 } 19324 map_idx = add_used_map(env, insn[0].imm); 19325 break; 19326 } 19327 19328 if (map_idx < 0) 19329 return map_idx; 19330 map = env->used_maps[map_idx]; 19331 19332 aux = &env->insn_aux_data[i]; 19333 aux->map_index = map_idx; 19334 19335 if (insn[0].src_reg == BPF_PSEUDO_MAP_FD || 19336 insn[0].src_reg == BPF_PSEUDO_MAP_IDX) { 19337 addr = (unsigned long)map; 19338 } else { 19339 u32 off = insn[1].imm; 19340 19341 if (!map->ops->map_direct_value_addr) { 19342 verbose(env, "no direct value access support for this map type\n"); 19343 return -EINVAL; 19344 } 19345 19346 err = map->ops->map_direct_value_addr(map, &addr, off); 19347 if (err) { 19348 verbose(env, "invalid access to map value pointer, value_size=%u off=%u\n", 19349 map->value_size, off); 19350 return err; 19351 } 19352 19353 aux->map_off = off; 19354 addr += off; 19355 } 19356 19357 insn[0].imm = (u32)addr; 19358 insn[1].imm = addr >> 32; 19359 19360 next_insn: 19361 insn++; 19362 i++; 19363 continue; 19364 } 19365 19366 /* Basic sanity check before we invest more work here. */ 19367 if (!bpf_opcode_in_insntable(insn->code)) { 19368 verbose(env, "unknown opcode %02x\n", insn->code); 19369 return -EINVAL; 19370 } 19371 19372 err = check_insn_fields(env, insn); 19373 if (err) 19374 return err; 19375 } 19376 19377 /* now all pseudo BPF_LD_IMM64 instructions load valid 19378 * 'struct bpf_map *' into a register instead of user map_fd. 19379 * These pointers will be used later by verifier to validate map access. 19380 */ 19381 return 0; 19382 } 19383 19384 /* drop refcnt of maps used by the rejected program */ 19385 static void release_maps(struct bpf_verifier_env *env) 19386 { 19387 __bpf_free_used_maps(env->prog->aux, env->used_maps, 19388 env->used_map_cnt); 19389 } 19390 19391 /* drop refcnt of maps used by the rejected program */ 19392 static void release_btfs(struct bpf_verifier_env *env) 19393 { 19394 __bpf_free_used_btfs(env->used_btfs, env->used_btf_cnt); 19395 } 19396 19397 /* convert pseudo BPF_LD_IMM64 into generic BPF_LD_IMM64 */ 19398 static void convert_pseudo_ld_imm64(struct bpf_verifier_env *env) 19399 { 19400 struct bpf_insn *insn = env->prog->insnsi; 19401 int insn_cnt = env->prog->len; 19402 int i; 19403 19404 for (i = 0; i < insn_cnt; i++, insn++) { 19405 if (insn->code != (BPF_LD | BPF_IMM | BPF_DW)) 19406 continue; 19407 if (insn->src_reg == BPF_PSEUDO_FUNC) 19408 continue; 19409 insn->src_reg = 0; 19410 } 19411 } 19412 19413 static void release_insn_arrays(struct bpf_verifier_env *env) 19414 { 19415 int i; 19416 19417 for (i = 0; i < env->insn_array_map_cnt; i++) 19418 bpf_insn_array_release(env->insn_array_maps[i]); 19419 } 19420 19421 /* The verifier does more data flow analysis than llvm and will not 19422 * explore branches that are dead at run time. Malicious programs can 19423 * have dead code too. Therefore replace all dead at-run-time code 19424 * with 'ja -1'. 19425 * 19426 * Just nops are not optimal, e.g. if they would sit at the end of the 19427 * program and through another bug we would manage to jump there, then 19428 * we'd execute beyond program memory otherwise. Returning exception 19429 * code also wouldn't work since we can have subprogs where the dead 19430 * code could be located. 19431 */ 19432 static void sanitize_dead_code(struct bpf_verifier_env *env) 19433 { 19434 struct bpf_insn_aux_data *aux_data = env->insn_aux_data; 19435 struct bpf_insn trap = BPF_JMP_IMM(BPF_JA, 0, 0, -1); 19436 struct bpf_insn *insn = env->prog->insnsi; 19437 const int insn_cnt = env->prog->len; 19438 int i; 19439 19440 for (i = 0; i < insn_cnt; i++) { 19441 if (aux_data[i].seen) 19442 continue; 19443 memcpy(insn + i, &trap, sizeof(trap)); 19444 aux_data[i].zext_dst = false; 19445 } 19446 } 19447 19448 static void free_states(struct bpf_verifier_env *env) 19449 { 19450 struct bpf_verifier_state_list *sl; 19451 struct list_head *head, *pos, *tmp; 19452 struct bpf_scc_info *info; 19453 int i, j; 19454 19455 bpf_free_verifier_state(env->cur_state, true); 19456 env->cur_state = NULL; 19457 while (!pop_stack(env, NULL, NULL, false)); 19458 19459 list_for_each_safe(pos, tmp, &env->free_list) { 19460 sl = container_of(pos, struct bpf_verifier_state_list, node); 19461 bpf_free_verifier_state(&sl->state, false); 19462 kfree(sl); 19463 } 19464 INIT_LIST_HEAD(&env->free_list); 19465 19466 for (i = 0; i < env->scc_cnt; ++i) { 19467 info = env->scc_info[i]; 19468 if (!info) 19469 continue; 19470 for (j = 0; j < info->num_visits; j++) 19471 bpf_free_backedges(&info->visits[j]); 19472 kvfree(info); 19473 env->scc_info[i] = NULL; 19474 } 19475 19476 if (!env->explored_states) 19477 return; 19478 19479 for (i = 0; i < state_htab_size(env); i++) { 19480 head = &env->explored_states[i]; 19481 19482 list_for_each_safe(pos, tmp, head) { 19483 sl = container_of(pos, struct bpf_verifier_state_list, node); 19484 bpf_free_verifier_state(&sl->state, false); 19485 kfree(sl); 19486 } 19487 INIT_LIST_HEAD(&env->explored_states[i]); 19488 } 19489 } 19490 19491 static int do_check_common(struct bpf_verifier_env *env, int subprog, bool is_sleepable) 19492 { 19493 bool pop_log = !(env->log.level & BPF_LOG_LEVEL2); 19494 struct bpf_subprog_info *sub = subprog_info(env, subprog); 19495 struct bpf_prog_aux *aux = env->prog->aux; 19496 struct bpf_verifier_state *state; 19497 struct bpf_reg_state *regs; 19498 u32 old_insns_total = sub->insns_total; 19499 u32 insn_processed = env->insn_processed; 19500 int ret, i; 19501 19502 env->prev_linfo = NULL; 19503 env->pass_cnt++; 19504 19505 state = kzalloc_obj(struct bpf_verifier_state, GFP_KERNEL_ACCOUNT); 19506 if (!state) 19507 return -ENOMEM; 19508 state->curframe = 0; 19509 state->speculative = false; 19510 state->branches = 1; 19511 state->in_sleepable = is_sleepable; 19512 state->frame[0] = kzalloc_obj(struct bpf_func_state, GFP_KERNEL_ACCOUNT); 19513 if (!state->frame[0]) { 19514 kfree(state); 19515 return -ENOMEM; 19516 } 19517 env->cur_state = state; 19518 init_func_state(env, state->frame[0], 19519 BPF_MAIN_FUNC /* callsite */, 19520 0 /* frameno */, 19521 subprog); 19522 state->first_insn_idx = env->subprog_info[subprog].start; 19523 state->last_insn_idx = -1; 19524 19525 regs = state->frame[state->curframe]->regs; 19526 if (subprog || env->prog->type == BPF_PROG_TYPE_EXT) { 19527 const char *sub_name = bpf_subprog_name(env, subprog); 19528 struct bpf_subprog_arg_info *arg; 19529 struct bpf_reg_state *reg; 19530 19531 if (env->log.level & BPF_LOG_LEVEL) 19532 verbose(env, "Validating %s() func#%d...\n", sub_name, subprog); 19533 ret = btf_prepare_func_args(env, subprog); 19534 if (ret) 19535 goto out; 19536 19537 if (subprog_is_exc_cb(env, subprog)) { 19538 state->frame[0]->in_exception_callback_fn = true; 19539 19540 /* 19541 * Global functions are scalar or void, make sure 19542 * we return a scalar. 19543 */ 19544 if (subprog_returns_void(env, subprog)) { 19545 verbose(env, "exception cb cannot return void\n"); 19546 ret = -EINVAL; 19547 goto out; 19548 } 19549 19550 /* Also ensure the callback only has a single scalar argument. */ 19551 if (sub->arg_cnt != 1 || sub->args[0].arg_type != ARG_ANYTHING) { 19552 verbose(env, "exception cb only supports single integer argument\n"); 19553 ret = -EINVAL; 19554 goto out; 19555 } 19556 } 19557 for (i = BPF_REG_1; i <= min_t(u32, sub->arg_cnt, MAX_BPF_FUNC_REG_ARGS); i++) { 19558 arg = &sub->args[i - BPF_REG_1]; 19559 reg = ®s[i]; 19560 19561 if (arg->arg_type == ARG_PTR_TO_CTX) { 19562 reg->type = PTR_TO_CTX; 19563 mark_reg_known_zero(env, regs, i); 19564 } else if (arg->arg_type == ARG_ANYTHING) { 19565 reg->type = SCALAR_VALUE; 19566 mark_reg_unknown(env, regs, i); 19567 } else if (arg->arg_type == ARG_PTR_TO_DYNPTR) { 19568 /* assume unspecial LOCAL dynptr type */ 19569 __mark_dynptr_reg(reg, BPF_DYNPTR_TYPE_LOCAL, true, ++env->id_gen, 0); 19570 } else if (base_type(arg->arg_type) == ARG_PTR_TO_MEM) { 19571 reg->type = PTR_TO_MEM; 19572 reg->type |= arg->arg_type & 19573 (PTR_MAYBE_NULL | PTR_UNTRUSTED | MEM_RDONLY); 19574 mark_reg_known_zero(env, regs, i); 19575 reg->mem_size = arg->mem_size; 19576 if (arg->arg_type & PTR_MAYBE_NULL) 19577 reg->id = ++env->id_gen; 19578 } else if (base_type(arg->arg_type) == ARG_PTR_TO_BTF_ID) { 19579 reg->type = PTR_TO_BTF_ID; 19580 if (arg->arg_type & PTR_MAYBE_NULL) 19581 reg->type |= PTR_MAYBE_NULL; 19582 if (arg->arg_type & PTR_UNTRUSTED) 19583 reg->type |= PTR_UNTRUSTED; 19584 if (arg->arg_type & PTR_TRUSTED) 19585 reg->type |= PTR_TRUSTED; 19586 mark_reg_known_zero(env, regs, i); 19587 reg->btf = bpf_get_btf_vmlinux(); /* can't fail at this point */ 19588 reg->btf_id = arg->btf_id; 19589 reg->id = ++env->id_gen; 19590 } else if (base_type(arg->arg_type) == ARG_PTR_TO_ARENA) { 19591 /* caller can pass either PTR_TO_ARENA or SCALAR */ 19592 mark_reg_unknown(env, regs, i); 19593 } else { 19594 verifier_bug(env, "unhandled arg#%d type %d", 19595 i - BPF_REG_1 + 1, arg->arg_type); 19596 ret = -EFAULT; 19597 goto out; 19598 } 19599 } 19600 if (env->prog->type == BPF_PROG_TYPE_EXT && sub->arg_cnt > MAX_BPF_FUNC_REG_ARGS) { 19601 verbose(env, "freplace programs with >%d args not supported yet\n", 19602 MAX_BPF_FUNC_REG_ARGS); 19603 ret = -EINVAL; 19604 goto out; 19605 } 19606 } else { 19607 /* if main BPF program has associated BTF info, validate that 19608 * it's matching expected signature, and otherwise mark BTF 19609 * info for main program as unreliable 19610 */ 19611 if (env->prog->aux->func_info_aux) { 19612 ret = btf_prepare_func_args(env, 0); 19613 if (ret || sub->arg_cnt != 1 || sub->args[0].arg_type != ARG_PTR_TO_CTX) { 19614 env->prog->aux->func_info_aux[0].unreliable = true; 19615 sub->arg_cnt = 1; 19616 sub->stack_arg_cnt = 0; 19617 } 19618 } 19619 19620 /* 1st arg to a function */ 19621 regs[BPF_REG_1].type = PTR_TO_CTX; 19622 mark_reg_known_zero(env, regs, BPF_REG_1); 19623 } 19624 19625 /* Acquire references for struct_ops program arguments tagged with "__ref" */ 19626 if (!subprog && env->prog->type == BPF_PROG_TYPE_STRUCT_OPS) { 19627 for (i = 0; i < aux->ctx_arg_info_size; i++) { 19628 ret = aux->ctx_arg_info[i].refcounted ? acquire_reference(env, 0, 0) : 0; 19629 if (ret < 0) 19630 goto out; 19631 19632 aux->ctx_arg_info[i].ref_id = ret; 19633 } 19634 } 19635 19636 ret = do_check(env); 19637 out: 19638 account_current_path(env); 19639 if (!ret) { 19640 if (pop_log) 19641 bpf_vlog_reset(&env->log, 0); 19642 bpf_diag_event_log_restore(env, 0); 19643 } 19644 free_states(env); 19645 19646 /* 19647 * The override is needed to account for async subprograms, which 19648 * are verified with their own set of stack frames and thus are 19649 * not accounted as callees by account_current_path(). 19650 * Accumulate their total counts as total counts of the main or 19651 * global subprog hosting the async call. 19652 * Start from the saved total of earlier contexts: adding to the current 19653 * total would count this pass's synchronous paths twice. 19654 */ 19655 sub->insns_total = old_insns_total + (env->insn_processed - insn_processed); 19656 return ret; 19657 } 19658 19659 /* Lazily verify all global functions based on their BTF, if they are called 19660 * from main BPF program or any of subprograms transitively. 19661 * BPF global subprogs called from dead code are not validated. 19662 * All callable global functions must pass verification. 19663 * Otherwise the whole program is rejected. 19664 * Consider: 19665 * int bar(int); 19666 * int foo(int f) 19667 * { 19668 * return bar(f); 19669 * } 19670 * int bar(int b) 19671 * { 19672 * ... 19673 * } 19674 * foo() will be verified first for R1=any_scalar_value. During verification it 19675 * will be assumed that bar() already verified successfully and call to bar() 19676 * from foo() will be checked for type match only. Later bar() will be verified 19677 * independently to check that it's safe for R1=any_scalar_value. 19678 */ 19679 static int do_check_subprogs(struct bpf_verifier_env *env) 19680 { 19681 struct bpf_prog_aux *aux = env->prog->aux; 19682 struct bpf_func_info_aux *sub_aux; 19683 int context, i, ret, new_cnt; 19684 19685 if (!aux->func_info) 19686 return 0; 19687 19688 /* 19689 * Callbacks cannot throw, so the exception callback always runs in the 19690 * main program's context. It is presumed to be always called. 19691 */ 19692 if (env->exception_callback_subprog) { 19693 sub_aux = subprog_aux(env, env->exception_callback_subprog); 19694 sub_aux->called[env->prog->sleepable] = true; 19695 } 19696 19697 again: 19698 new_cnt = 0; 19699 for (i = 1; i < env->subprog_cnt; i++) { 19700 if (!bpf_subprog_is_global(env, i)) 19701 continue; 19702 19703 sub_aux = subprog_aux(env, i); 19704 for (context = 0; context < ARRAY_SIZE(sub_aux->called); context++) { 19705 if (!sub_aux->called[context] || sub_aux->verified[context]) 19706 continue; 19707 19708 env->insn_idx = env->subprog_info[i].start; 19709 WARN_ON_ONCE(env->insn_idx == 0); 19710 ret = do_check_common(env, i, context); 19711 if (ret) 19712 return ret; 19713 if (env->log.level & BPF_LOG_LEVEL) 19714 verbose(env, "Func#%d ('%s') is safe for any args " 19715 "that match its prototype\n", 19716 i, bpf_subprog_name(env, i)); 19717 19718 sub_aux->verified[context] = true; 19719 new_cnt++; 19720 } 19721 } 19722 19723 /* 19724 * We can't loop forever as each pass verifies at least one new context, 19725 * and there are only two contexts per global subprog. 19726 */ 19727 if (new_cnt) 19728 goto again; 19729 19730 return 0; 19731 } 19732 19733 static int do_check_main(struct bpf_verifier_env *env) 19734 { 19735 int ret; 19736 19737 env->insn_idx = 0; 19738 ret = do_check_common(env, 0, env->prog->sleepable); 19739 if (!ret) 19740 env->prog->aux->stack_depth = env->subprog_info[0].stack_depth; 19741 return ret; 19742 } 19743 19744 static void print_verification_stats(struct bpf_verifier_env *env) 19745 { 19746 /* Skip over hidden subprogs which are not verified. */ 19747 int i, subprog_cnt = env->subprog_cnt - env->hidden_subprog_cnt; 19748 19749 if (env->log.level & BPF_LOG_STATS) { 19750 verbose(env, "verification time %lld usec\n", 19751 div_u64(env->verification_time, 1000)); 19752 verbose(env, "stack depth max %d\n", env->max_stack_depth); 19753 for (i = 0; i < subprog_cnt; i++) { 19754 const char *name = env->subprog_info[i].name; 19755 const char *kind; 19756 19757 if (!name || !name[0]) 19758 name = "<unknown>"; 19759 kind = i == 0 ? "main" : 19760 bpf_subprog_is_global(env, i) ? "global" : "static"; 19761 verbose(env, "subprog %d (%s) %s insns_self %d insns_total %d stack %d\n", 19762 i, name, kind, env->subprog_info[i].insns_self, 19763 env->subprog_info[i].insns_total, 19764 env->subprog_info[i].stack_depth); 19765 } 19766 } 19767 verbose(env, "processed %d insns (limit %d) max_states_per_insn %d " 19768 "total_states %d peak_states %d mark_read %d\n", 19769 env->insn_processed, BPF_COMPLEXITY_LIMIT_INSNS, 19770 env->max_states_per_insn, env->total_states, 19771 env->peak_states, env->longest_mark_read_walk); 19772 } 19773 19774 int bpf_prog_ctx_arg_info_init(struct bpf_prog *prog, 19775 const struct bpf_ctx_arg_aux *info, u32 cnt) 19776 { 19777 prog->aux->ctx_arg_info = kmemdup_array(info, cnt, sizeof(*info), GFP_KERNEL_ACCOUNT); 19778 prog->aux->ctx_arg_info_size = cnt; 19779 19780 return prog->aux->ctx_arg_info ? 0 : -ENOMEM; 19781 } 19782 19783 static int check_struct_ops_btf_id(struct bpf_verifier_env *env) 19784 { 19785 const struct btf_type *t, *func_proto; 19786 const struct bpf_struct_ops_desc *st_ops_desc; 19787 const struct bpf_struct_ops_arg_info *arg_info; 19788 const struct bpf_struct_ops *st_ops; 19789 const struct btf_member *member; 19790 struct bpf_prog *prog = env->prog; 19791 bool has_refcounted_arg = false; 19792 u32 btf_id, member_idx, member_off; 19793 struct btf *btf; 19794 const char *mname; 19795 int i, err; 19796 19797 if (!prog->gpl_compatible) { 19798 verbose(env, "struct ops programs must have a GPL compatible license\n"); 19799 return -EINVAL; 19800 } 19801 19802 if (!prog->aux->attach_btf_id) 19803 return -ENOTSUPP; 19804 19805 btf = prog->aux->attach_btf; 19806 if (btf_is_module(btf)) { 19807 /* Make sure st_ops is valid through the lifetime of env */ 19808 env->attach_btf_mod = btf_try_get_module(btf); 19809 if (!env->attach_btf_mod) { 19810 verbose(env, "struct_ops module %s is not found\n", 19811 btf_get_name(btf)); 19812 return -ENOTSUPP; 19813 } 19814 } 19815 19816 btf_id = prog->aux->attach_btf_id; 19817 st_ops_desc = bpf_struct_ops_find(btf, btf_id); 19818 if (!st_ops_desc) { 19819 verbose(env, "attach_btf_id %u is not a supported struct\n", 19820 btf_id); 19821 return -ENOTSUPP; 19822 } 19823 st_ops = st_ops_desc->st_ops; 19824 19825 t = st_ops_desc->type; 19826 member_idx = prog->expected_attach_type; 19827 if (member_idx >= btf_type_vlen(t)) { 19828 verbose(env, "attach to invalid member idx %u of struct %s\n", 19829 member_idx, st_ops->name); 19830 return -EINVAL; 19831 } 19832 19833 member = &btf_type_member(t)[member_idx]; 19834 mname = btf_name_by_offset(btf, member->name_off); 19835 func_proto = btf_type_resolve_func_ptr(btf, member->type, 19836 NULL); 19837 if (!func_proto) { 19838 verbose(env, "attach to invalid member %s(@idx %u) of struct %s\n", 19839 mname, member_idx, st_ops->name); 19840 return -EINVAL; 19841 } 19842 19843 member_off = __btf_member_bit_offset(t, member) / 8; 19844 err = bpf_struct_ops_supported(st_ops, member_off); 19845 if (err) { 19846 verbose(env, "attach to unsupported member %s of struct %s\n", 19847 mname, st_ops->name); 19848 return err; 19849 } 19850 19851 if (st_ops->check_member) { 19852 err = st_ops->check_member(t, member, prog); 19853 19854 if (err) { 19855 verbose(env, "attach to unsupported member %s of struct %s\n", 19856 mname, st_ops->name); 19857 return err; 19858 } 19859 } 19860 19861 if (prog->aux->priv_stack_requested && !bpf_jit_supports_private_stack()) { 19862 verbose(env, "Private stack not supported by jit\n"); 19863 return -EACCES; 19864 } 19865 19866 arg_info = &st_ops_desc->arg_info[member_idx]; 19867 for (i = 0; i < arg_info->cnt; i++) { 19868 const struct bpf_ctx_arg_aux *info = &arg_info->info[i]; 19869 19870 if (info->refcounted) 19871 has_refcounted_arg = true; 19872 if (base_type(info->reg_type) == PTR_TO_ARENA) { 19873 if (!bpf_jit_supports_arena_args()) { 19874 verbose(env, "JIT does not support arena arguments\n"); 19875 return -ENOTSUPP; 19876 } 19877 if (!prog->aux->arena) { 19878 verbose(env, 19879 "arena argument of %s requires a program with an associated arena\n", 19880 mname); 19881 return -EINVAL; 19882 } 19883 } 19884 } 19885 19886 /* Tail call is not allowed for programs with refcounted arguments since we 19887 * cannot guarantee that valid refcounted kptrs will be passed to the callee. 19888 */ 19889 for (i = 0; i < env->subprog_cnt; i++) { 19890 if (has_refcounted_arg && env->subprog_info[i].has_tail_call) { 19891 verbose(env, "program with __ref argument cannot tail call\n"); 19892 return -EINVAL; 19893 } 19894 } 19895 19896 prog->aux->st_ops = st_ops; 19897 prog->aux->attach_st_ops_member_off = member_off; 19898 19899 prog->aux->attach_func_proto = func_proto; 19900 prog->aux->attach_func_name = mname; 19901 env->ops = st_ops->verifier_ops; 19902 19903 return bpf_prog_ctx_arg_info_init(prog, arg_info->info, arg_info->cnt); 19904 } 19905 #define SECURITY_PREFIX "security_" 19906 19907 #ifdef CONFIG_FUNCTION_ERROR_INJECTION 19908 19909 /* list of non-sleepable functions that are otherwise on 19910 * ALLOW_ERROR_INJECTION list 19911 */ 19912 BTF_SET_START(btf_non_sleepable_error_inject) 19913 /* Three functions below can be called from sleepable and non-sleepable context. 19914 * Assume non-sleepable from bpf safety point of view. 19915 */ 19916 BTF_ID(func, __filemap_add_folio) 19917 #ifdef CONFIG_FAIL_PAGE_ALLOC 19918 BTF_ID(func, should_fail_alloc_page) 19919 #endif 19920 #ifdef CONFIG_FAILSLAB 19921 BTF_ID(func, should_failslab) 19922 #endif 19923 BTF_SET_END(btf_non_sleepable_error_inject) 19924 19925 static int check_non_sleepable_error_inject(u32 btf_id) 19926 { 19927 return btf_id_set_contains(&btf_non_sleepable_error_inject, btf_id); 19928 } 19929 19930 static int check_attach_sleepable(u32 btf_id, unsigned long addr, const char *func_name) 19931 { 19932 /* fentry/fexit/fmod_ret progs can be sleepable if they are 19933 * attached to ALLOW_ERROR_INJECTION and are not in denylist. 19934 */ 19935 if (!check_non_sleepable_error_inject(btf_id) && 19936 within_error_injection_list(addr)) 19937 return 0; 19938 19939 return -EINVAL; 19940 } 19941 19942 static int check_attach_modify_return(unsigned long addr, const char *func_name) 19943 { 19944 if (within_error_injection_list(addr) || 19945 !strncmp(SECURITY_PREFIX, func_name, sizeof(SECURITY_PREFIX) - 1)) 19946 return 0; 19947 19948 return -EINVAL; 19949 } 19950 19951 #else 19952 19953 /* Unfortunately, the arch-specific prefixes are hard-coded in arch syscall code 19954 * so we need to hard-code them, too. Ftrace has arch_syscall_match_sym_name() 19955 * but that just compares two concrete function names. 19956 */ 19957 static bool has_arch_syscall_prefix(const char *func_name) 19958 { 19959 #if defined(__x86_64__) 19960 return !strncmp(func_name, "__x64_", 6); 19961 #elif defined(__i386__) 19962 return !strncmp(func_name, "__ia32_", 7); 19963 #elif defined(__s390x__) 19964 return !strncmp(func_name, "__s390x_", 8); 19965 #elif defined(__aarch64__) 19966 return !strncmp(func_name, "__arm64_", 8); 19967 #elif defined(__riscv) 19968 return !strncmp(func_name, "__riscv_", 8); 19969 #elif defined(__powerpc__) || defined(__powerpc64__) 19970 return !strncmp(func_name, "sys_", 4); 19971 #elif defined(__loongarch__) 19972 return !strncmp(func_name, "sys_", 4); 19973 #else 19974 return false; 19975 #endif 19976 } 19977 19978 /* Without error injection, allow sleepable and fmod_ret progs on syscalls. */ 19979 19980 static int check_attach_sleepable(u32 btf_id, unsigned long addr, const char *func_name) 19981 { 19982 if (has_arch_syscall_prefix(func_name)) 19983 return 0; 19984 19985 return -EINVAL; 19986 } 19987 19988 static int check_attach_modify_return(unsigned long addr, const char *func_name) 19989 { 19990 if (has_arch_syscall_prefix(func_name) || 19991 !strncmp(SECURITY_PREFIX, func_name, sizeof(SECURITY_PREFIX) - 1)) 19992 return 0; 19993 19994 return -EINVAL; 19995 } 19996 19997 #endif /* CONFIG_FUNCTION_ERROR_INJECTION */ 19998 19999 static bool is_tracing_multi_id(const struct bpf_prog *prog, u32 btf_id) 20000 { 20001 return is_tracing_multi(prog->expected_attach_type) && bpf_multi_func_btf_id[0] == btf_id; 20002 } 20003 20004 static int btf_id_allow_sleepable(u32 btf_id, unsigned long addr, const struct bpf_prog *prog, 20005 const struct btf *btf) 20006 { 20007 const struct btf_type *t; 20008 const char *tname; 20009 20010 if (!btf_is_kernel(btf)) 20011 return -EINVAL; 20012 20013 switch (prog->type) { 20014 case BPF_PROG_TYPE_TRACING: 20015 t = btf_type_by_id(btf, btf_id); 20016 if (!t) 20017 return -EINVAL; 20018 tname = btf_name_by_offset(btf, t->name_off); 20019 if (!tname) 20020 return -EINVAL; 20021 20022 /* 20023 * *.multi sleepable programs will pass initial sleepable check, 20024 * the actual attached btf ids are checked later during the link 20025 * attachment. 20026 */ 20027 if (is_tracing_multi_id(prog, btf_id)) 20028 return 0; 20029 if (!check_attach_sleepable(btf_id, addr, tname)) 20030 return 0; 20031 /* 20032 * fentry/fexit/fmod_ret progs can also be sleepable if they are 20033 * in the fmodret id set with the KF_SLEEPABLE flag. 20034 */ 20035 else { 20036 u32 *flags = btf_kfunc_is_modify_return(btf, btf_id, prog); 20037 20038 if (flags && (*flags & KF_SLEEPABLE)) 20039 return 0; 20040 } 20041 break; 20042 case BPF_PROG_TYPE_LSM: 20043 /* 20044 * LSM progs check that they are attached to bpf_lsm_*() funcs. 20045 * Only some of them are sleepable. 20046 */ 20047 if (bpf_lsm_is_sleepable_hook(btf_id)) 20048 return 0; 20049 break; 20050 default: 20051 break; 20052 } 20053 return -EINVAL; 20054 } 20055 20056 /* 20057 * Resolve the prototype describing a trace target's real ABI. A 20058 * KF_IMPLICIT_ARGS kfunc has its injected args stripped from the public 20059 * prototype, so use the _impl prototype; other targets use their own. 20060 */ 20061 static const struct btf_type * 20062 btf_attach_func_proto(struct bpf_verifier_log *log, struct btf *btf, u32 func_id) 20063 { 20064 const struct btf_type *func; 20065 struct module *mod = NULL; 20066 const char *name; 20067 int implicit; 20068 20069 func = btf_type_by_id(btf, func_id); 20070 if (!func || !btf_type_is_func(func)) 20071 return NULL; 20072 name = btf_name_by_offset(btf, func->name_off); 20073 20074 /* 20075 * btf_kfunc_check_flag() reads kfunc_set_tab, which for a module is 20076 * stable only once it is live; hold a module ref across the read to 20077 * exclude a concurrent module load. 20078 */ 20079 if (btf_is_module(btf)) { 20080 mod = btf_try_get_module(btf); 20081 if (!mod) 20082 return NULL; 20083 } 20084 implicit = btf_kfunc_check_flag(btf, func_id, KF_IMPLICIT_ARGS); 20085 module_put(mod); 20086 20087 if (implicit == -EINVAL) { 20088 bpf_log(log, "kfunc %s has inconsistent KF_IMPLICIT_ARGS\n", name); 20089 return NULL; 20090 } 20091 if (implicit > 0) 20092 return find_kfunc_impl_proto(log, btf, name); 20093 20094 return btf_type_by_id(btf, func->type); 20095 } 20096 20097 static bool attach_uses_trampoline_retval(enum bpf_attach_type type) 20098 { 20099 switch (type) { 20100 case BPF_MODIFY_RETURN: 20101 case BPF_TRACE_FEXIT: 20102 case BPF_TRACE_FEXIT_MULTI: 20103 case BPF_TRACE_FSESSION: 20104 case BPF_TRACE_FSESSION_MULTI: 20105 return true; 20106 default: 20107 return false; 20108 } 20109 } 20110 20111 int bpf_check_attach_target(struct bpf_verifier_log *log, 20112 const struct bpf_prog *prog, 20113 const struct bpf_prog *tgt_prog, 20114 u32 btf_id, 20115 struct bpf_attach_target_info *tgt_info) 20116 { 20117 bool prog_extension = prog->type == BPF_PROG_TYPE_EXT; 20118 bool prog_tracing = prog->type == BPF_PROG_TYPE_TRACING; 20119 char trace_symbol[KSYM_SYMBOL_LEN]; 20120 const char prefix[] = "btf_trace_"; 20121 struct bpf_raw_event_map *btp; 20122 int ret = 0, subprog = -1, i; 20123 const struct btf_type *t; 20124 bool conservative = true; 20125 const char *tname, *fname; 20126 struct btf *btf; 20127 long addr = 0; 20128 struct module *mod = NULL; 20129 20130 if (!btf_id) { 20131 bpf_log(log, "Tracing programs must provide btf_id\n"); 20132 return -EINVAL; 20133 } 20134 btf = tgt_prog ? tgt_prog->aux->btf : prog->aux->attach_btf; 20135 if (!btf) { 20136 bpf_log(log, 20137 "Tracing program can only be attached to another program annotated with BTF\n"); 20138 return -EINVAL; 20139 } 20140 t = btf_type_by_id(btf, btf_id); 20141 if (!t) { 20142 bpf_log(log, "attach_btf_id %u is invalid\n", btf_id); 20143 return -EINVAL; 20144 } 20145 tname = btf_name_by_offset(btf, t->name_off); 20146 if (!tname) { 20147 bpf_log(log, "attach_btf_id %u doesn't have a name\n", btf_id); 20148 return -EINVAL; 20149 } 20150 if (tgt_prog) { 20151 struct bpf_prog_aux *aux = tgt_prog->aux; 20152 bool tgt_changes_pkt_data; 20153 bool tgt_might_sleep; 20154 20155 if (bpf_prog_is_dev_bound(prog->aux) && 20156 !bpf_prog_dev_bound_match(prog, tgt_prog)) { 20157 bpf_log(log, "Target program bound device mismatch"); 20158 return -EINVAL; 20159 } 20160 20161 for (i = 0; i < aux->func_info_cnt; i++) 20162 if (aux->func_info[i].type_id == btf_id) { 20163 subprog = i; 20164 break; 20165 } 20166 if (subprog == -1) { 20167 bpf_log(log, "Subprog %s doesn't exist\n", tname); 20168 return -EINVAL; 20169 } 20170 /* 20171 * A struct_ops indirect trampoline converts arena arguments 20172 * before invoking its program. A tracing or extension program 20173 * attached to the main program would see the converted offset as a 20174 * regular BTF pointer. 20175 */ 20176 if (subprog == 0 && bpf_prog_has_arena_ctx_arg(tgt_prog)) { 20177 bpf_log(log, "Cannot attach to a target with arena context arguments\n"); 20178 return -EOPNOTSUPP; 20179 } 20180 if (aux->func && aux->func[subprog]->aux->exception_cb) { 20181 bpf_log(log, 20182 "%s programs cannot attach to exception callback\n", 20183 prog_extension ? "Extension" : "Tracing"); 20184 return -EINVAL; 20185 } 20186 conservative = aux->func_info_aux[subprog].unreliable; 20187 if (prog_extension) { 20188 if (conservative) { 20189 bpf_log(log, 20190 "Cannot replace static functions\n"); 20191 return -EINVAL; 20192 } 20193 if (!prog->jit_requested) { 20194 bpf_log(log, 20195 "Extension programs should be JITed\n"); 20196 return -EINVAL; 20197 } 20198 tgt_changes_pkt_data = aux->func 20199 ? aux->func[subprog]->aux->changes_pkt_data 20200 : aux->changes_pkt_data; 20201 if (prog->aux->changes_pkt_data && !tgt_changes_pkt_data) { 20202 bpf_log(log, 20203 "Extension program changes packet data, while original does not\n"); 20204 return -EINVAL; 20205 } 20206 20207 tgt_might_sleep = aux->func 20208 ? aux->func[subprog]->aux->might_sleep 20209 : aux->might_sleep; 20210 if (prog->aux->might_sleep && !tgt_might_sleep) { 20211 bpf_log(log, 20212 "Extension program may sleep, while original does not\n"); 20213 return -EINVAL; 20214 } 20215 } 20216 if (!tgt_prog->jited) { 20217 bpf_log(log, "Can attach to only JITed progs\n"); 20218 return -EINVAL; 20219 } 20220 if (prog_tracing) { 20221 if (aux->attach_tracing_prog) { 20222 /* 20223 * Target program is an fentry/fexit which is already attached 20224 * to another tracing program. More levels of nesting 20225 * attachment are not allowed. 20226 */ 20227 bpf_log(log, "Cannot nest tracing program attach more than once\n"); 20228 return -EINVAL; 20229 } 20230 } else if (tgt_prog->type == prog->type) { 20231 /* 20232 * To avoid potential call chain cycles, prevent attaching of a 20233 * program extension to another extension. It's ok to attach 20234 * fentry/fexit to extension program. 20235 */ 20236 bpf_log(log, "Cannot recursively attach\n"); 20237 return -EINVAL; 20238 } 20239 if (tgt_prog->type == BPF_PROG_TYPE_TRACING && 20240 prog_extension && 20241 (tgt_prog->expected_attach_type == BPF_TRACE_FENTRY || 20242 tgt_prog->expected_attach_type == BPF_TRACE_FEXIT || 20243 tgt_prog->expected_attach_type == BPF_TRACE_FENTRY_MULTI || 20244 tgt_prog->expected_attach_type == BPF_TRACE_FEXIT_MULTI || 20245 tgt_prog->expected_attach_type == BPF_TRACE_FSESSION || 20246 tgt_prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI)) { 20247 /* Program extensions can extend all program types 20248 * except fentry/fexit. The reason is the following. 20249 * The fentry/fexit programs are used for performance 20250 * analysis, stats and can be attached to any program 20251 * type. When extension program is replacing XDP function 20252 * it is necessary to allow performance analysis of all 20253 * functions. Both original XDP program and its program 20254 * extension. Hence attaching fentry/fexit to 20255 * BPF_PROG_TYPE_EXT is allowed. If extending of 20256 * fentry/fexit was allowed it would be possible to create 20257 * long call chain fentry->extension->fentry->extension 20258 * beyond reasonable stack size. Hence extending fentry 20259 * is not allowed. 20260 */ 20261 bpf_log(log, "Cannot extend fentry/fexit/fsession\n"); 20262 return -EINVAL; 20263 } 20264 } else { 20265 if (prog_extension) { 20266 bpf_log(log, "Cannot replace kernel functions\n"); 20267 return -EINVAL; 20268 } 20269 } 20270 20271 switch (prog->expected_attach_type) { 20272 case BPF_TRACE_RAW_TP: 20273 if (tgt_prog) { 20274 bpf_log(log, 20275 "Only FENTRY/FEXIT/FSESSION progs are attachable to another BPF prog\n"); 20276 return -EINVAL; 20277 } 20278 if (!btf_type_is_typedef(t)) { 20279 bpf_log(log, "attach_btf_id %u is not a typedef\n", 20280 btf_id); 20281 return -EINVAL; 20282 } 20283 if (strncmp(prefix, tname, sizeof(prefix) - 1)) { 20284 bpf_log(log, "attach_btf_id %u points to wrong type name %s\n", 20285 btf_id, tname); 20286 return -EINVAL; 20287 } 20288 tname += sizeof(prefix) - 1; 20289 20290 /* The func_proto of "btf_trace_##tname" is generated from typedef without argument 20291 * names. Thus using bpf_raw_event_map to get argument names. 20292 */ 20293 btp = bpf_get_raw_tracepoint(tname); 20294 if (!btp) 20295 return -EINVAL; 20296 if (prog->sleepable && !tracepoint_is_faultable(btp->tp)) { 20297 bpf_log(log, "Sleepable program cannot attach to non-faultable tracepoint %s\n", 20298 tname); 20299 bpf_put_raw_tracepoint(btp); 20300 return -EINVAL; 20301 } 20302 fname = kallsyms_lookup((unsigned long)btp->bpf_func, NULL, NULL, NULL, 20303 trace_symbol); 20304 bpf_put_raw_tracepoint(btp); 20305 20306 if (fname) 20307 ret = btf_find_by_name_kind(btf, fname, BTF_KIND_FUNC); 20308 20309 if (!fname || ret < 0) { 20310 bpf_log(log, "Cannot find btf of tracepoint template, fall back to %s%s.\n", 20311 prefix, tname); 20312 t = btf_type_by_id(btf, t->type); 20313 if (!btf_type_is_ptr(t)) 20314 /* should never happen in valid vmlinux build */ 20315 return -EINVAL; 20316 } else { 20317 t = btf_type_by_id(btf, ret); 20318 if (!btf_type_is_func(t)) 20319 /* should never happen in valid vmlinux build */ 20320 return -EINVAL; 20321 } 20322 20323 t = btf_type_by_id(btf, t->type); 20324 if (!btf_type_is_func_proto(t)) 20325 /* should never happen in valid vmlinux build */ 20326 return -EINVAL; 20327 20328 break; 20329 case BPF_TRACE_ITER: 20330 if (!btf_type_is_func(t)) { 20331 bpf_log(log, "attach_btf_id %u is not a function\n", 20332 btf_id); 20333 return -EINVAL; 20334 } 20335 t = btf_type_by_id(btf, t->type); 20336 if (!btf_type_is_func_proto(t)) 20337 return -EINVAL; 20338 ret = btf_distill_func_proto(log, btf, t, tname, &tgt_info->fmodel); 20339 if (ret) 20340 return ret; 20341 break; 20342 default: 20343 if (!prog_extension) 20344 return -EINVAL; 20345 fallthrough; 20346 case BPF_MODIFY_RETURN: 20347 case BPF_LSM_MAC: 20348 case BPF_LSM_CGROUP: 20349 case BPF_TRACE_FENTRY: 20350 case BPF_TRACE_FEXIT: 20351 case BPF_TRACE_FSESSION: 20352 case BPF_TRACE_FSESSION_MULTI: 20353 case BPF_TRACE_FENTRY_MULTI: 20354 case BPF_TRACE_FEXIT_MULTI: 20355 if ((prog->expected_attach_type == BPF_TRACE_FSESSION || 20356 prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI) && 20357 !bpf_jit_supports_fsession()) { 20358 bpf_log(log, "JIT does not support fsession\n"); 20359 return -EOPNOTSUPP; 20360 } 20361 if (!btf_type_is_func(t)) { 20362 bpf_log(log, "attach_btf_id %u is not a function\n", 20363 btf_id); 20364 return -EINVAL; 20365 } 20366 if (prog_extension && 20367 btf_check_type_match(log, prog, btf, t)) 20368 return -EINVAL; 20369 t = btf_attach_func_proto(log, btf, btf_id); 20370 if (!t || !btf_type_is_func_proto(t)) 20371 return -EINVAL; 20372 20373 if ((prog->aux->saved_dst_prog_type || prog->aux->saved_dst_attach_type) && 20374 (!tgt_prog || prog->aux->saved_dst_prog_type != tgt_prog->type || 20375 prog->aux->saved_dst_attach_type != tgt_prog->expected_attach_type)) 20376 return -EINVAL; 20377 20378 if (tgt_prog && conservative) 20379 t = NULL; 20380 20381 ret = btf_distill_func_proto(log, btf, t, tname, &tgt_info->fmodel); 20382 if (ret < 0) 20383 return ret; 20384 20385 if (tgt_info->fmodel.ret_size > 8 && 20386 attach_uses_trampoline_retval(prog->expected_attach_type)) { 20387 bpf_log(log, 20388 "Attach to function %s with a >8 byte return value is not supported for this attach type\n", 20389 tname); 20390 return -EOPNOTSUPP; 20391 } 20392 20393 /* 20394 * *.multi programs don't need an address during program 20395 * verification, we just take the module ref if needed. 20396 */ 20397 if (is_tracing_multi_id(prog, btf_id)) { 20398 if (btf_is_module(btf)) { 20399 mod = btf_try_get_module(btf); 20400 if (!mod) 20401 return -ENOENT; 20402 } 20403 addr = 0; 20404 } else if (tgt_prog) { 20405 if (subprog == 0) 20406 addr = (long) tgt_prog->bpf_func; 20407 else 20408 addr = (long) tgt_prog->aux->func[subprog]->bpf_func; 20409 } else { 20410 if (btf_is_module(btf)) { 20411 mod = btf_try_get_module(btf); 20412 if (mod) 20413 addr = find_kallsyms_symbol_value(mod, tname); 20414 else 20415 addr = 0; 20416 } else { 20417 addr = kallsyms_lookup_name(tname); 20418 } 20419 if (!addr) { 20420 module_put(mod); 20421 bpf_log(log, 20422 "The address of function %s cannot be found\n", 20423 tname); 20424 return -ENOENT; 20425 } 20426 } 20427 20428 if (prog->sleepable) { 20429 ret = btf_id_allow_sleepable(btf_id, addr, prog, btf); 20430 if (ret) { 20431 module_put(mod); 20432 bpf_log(log, "%s is not sleepable\n", tname); 20433 return ret; 20434 } 20435 } else if (prog->expected_attach_type == BPF_MODIFY_RETURN) { 20436 if (tgt_prog) { 20437 module_put(mod); 20438 bpf_log(log, "can't modify return codes of BPF programs\n"); 20439 return -EINVAL; 20440 } 20441 ret = -EINVAL; 20442 if (btf_kfunc_is_modify_return(btf, btf_id, prog) || 20443 !check_attach_modify_return(addr, tname)) 20444 ret = 0; 20445 if (ret) { 20446 module_put(mod); 20447 bpf_log(log, "%s() is not modifiable\n", tname); 20448 return ret; 20449 } 20450 } 20451 20452 break; 20453 } 20454 tgt_info->tgt_addr = addr; 20455 tgt_info->tgt_name = tname; 20456 tgt_info->tgt_type = t; 20457 tgt_info->tgt_mod = mod; 20458 return 0; 20459 } 20460 20461 BTF_SET_START(btf_id_deny) 20462 BTF_ID_UNUSED 20463 #ifdef CONFIG_SMP 20464 BTF_ID(func, ___migrate_enable) 20465 BTF_ID(func, migrate_disable) 20466 BTF_ID(func, migrate_enable) 20467 #endif 20468 #if !defined CONFIG_PREEMPT_RCU && !defined CONFIG_TINY_RCU 20469 BTF_ID(func, rcu_read_unlock_strict) 20470 #endif 20471 #if defined(CONFIG_DEBUG_PREEMPT) || defined(CONFIG_TRACE_PREEMPT_TOGGLE) 20472 BTF_ID(func, preempt_count_add) 20473 BTF_ID(func, preempt_count_sub) 20474 #endif 20475 #ifdef CONFIG_PREEMPT_RCU 20476 BTF_ID(func, __rcu_read_lock) 20477 BTF_ID(func, __rcu_read_unlock) 20478 #endif 20479 BTF_SET_END(btf_id_deny) 20480 20481 /* fexit and fmod_ret can't be used to attach to __noreturn functions. 20482 * Currently, we must manually list all __noreturn functions here. Once a more 20483 * robust solution is implemented, this workaround can be removed. 20484 */ 20485 BTF_SET_START(noreturn_deny) 20486 #ifdef CONFIG_IA32_EMULATION 20487 BTF_ID(func, __ia32_sys_exit) 20488 BTF_ID(func, __ia32_sys_exit_group) 20489 #endif 20490 #ifdef CONFIG_KUNIT 20491 BTF_ID(func, __kunit_abort) 20492 BTF_ID(func, kunit_try_catch_throw) 20493 #endif 20494 #ifdef CONFIG_MODULES 20495 BTF_ID(func, __module_put_and_kthread_exit) 20496 #endif 20497 #ifdef CONFIG_X86_64 20498 BTF_ID(func, __x64_sys_exit) 20499 BTF_ID(func, __x64_sys_exit_group) 20500 #endif 20501 BTF_ID(func, do_exit) 20502 BTF_ID(func, do_group_exit) 20503 BTF_ID(func, kthread_complete_and_exit) 20504 BTF_ID(func, make_task_dead) 20505 BTF_SET_END(noreturn_deny) 20506 20507 static bool can_be_sleepable(struct bpf_prog *prog) 20508 { 20509 if (prog->type == BPF_PROG_TYPE_TRACING) { 20510 switch (prog->expected_attach_type) { 20511 case BPF_TRACE_FENTRY: 20512 case BPF_TRACE_FEXIT: 20513 case BPF_MODIFY_RETURN: 20514 case BPF_TRACE_ITER: 20515 case BPF_TRACE_FSESSION: 20516 case BPF_TRACE_RAW_TP: 20517 case BPF_TRACE_FENTRY_MULTI: 20518 case BPF_TRACE_FEXIT_MULTI: 20519 case BPF_TRACE_FSESSION_MULTI: 20520 return true; 20521 default: 20522 return false; 20523 } 20524 } 20525 if (prog->type == BPF_PROG_TYPE_LSM) 20526 return prog->expected_attach_type != BPF_LSM_CGROUP; 20527 20528 return prog->type == BPF_PROG_TYPE_KPROBE /* only for uprobes */ || 20529 prog->type == BPF_PROG_TYPE_STRUCT_OPS || 20530 prog->type == BPF_PROG_TYPE_RAW_TRACEPOINT || 20531 prog->type == BPF_PROG_TYPE_TRACEPOINT; 20532 } 20533 20534 static int check_attach_btf_id(struct bpf_verifier_env *env) 20535 { 20536 struct bpf_prog *prog = env->prog; 20537 struct bpf_prog *tgt_prog = prog->aux->dst_prog; 20538 struct bpf_attach_target_info tgt_info = {}; 20539 u32 btf_id = prog->aux->attach_btf_id; 20540 struct bpf_trampoline *tr; 20541 int ret; 20542 u64 key; 20543 20544 if (prog->type == BPF_PROG_TYPE_SYSCALL) { 20545 if (prog->sleepable) 20546 /* attach_btf_id checked to be zero already */ 20547 return 0; 20548 verbose(env, "Syscall programs can only be sleepable\n"); 20549 return -EINVAL; 20550 } 20551 20552 if (prog->sleepable && !can_be_sleepable(prog)) { 20553 verbose(env, "Program of this type cannot be sleepable\n"); 20554 return -EINVAL; 20555 } 20556 20557 if (prog->type == BPF_PROG_TYPE_STRUCT_OPS) 20558 return check_struct_ops_btf_id(env); 20559 20560 if (prog->type != BPF_PROG_TYPE_TRACING && 20561 prog->type != BPF_PROG_TYPE_LSM && 20562 prog->type != BPF_PROG_TYPE_EXT) 20563 return 0; 20564 20565 ret = bpf_check_attach_target(&env->log, prog, tgt_prog, btf_id, &tgt_info); 20566 if (ret) 20567 return ret; 20568 20569 if (tgt_prog && prog->type == BPF_PROG_TYPE_EXT) { 20570 /* to make freplace equivalent to their targets, they need to 20571 * inherit env->ops and expected_attach_type for the rest of the 20572 * verification 20573 */ 20574 env->ops = bpf_verifier_ops[tgt_prog->type]; 20575 prog->expected_attach_type = tgt_prog->expected_attach_type; 20576 } 20577 20578 /* store info about the attachment target that will be used later */ 20579 prog->aux->attach_func_proto = tgt_info.tgt_type; 20580 prog->aux->attach_func_name = tgt_info.tgt_name; 20581 prog->aux->mod = tgt_info.tgt_mod; 20582 20583 if (tgt_prog) { 20584 prog->aux->saved_dst_prog_type = tgt_prog->type; 20585 prog->aux->saved_dst_attach_type = tgt_prog->expected_attach_type; 20586 } 20587 20588 if (prog->expected_attach_type == BPF_TRACE_RAW_TP) { 20589 prog->aux->attach_btf_trace = true; 20590 return 0; 20591 } else if (prog->expected_attach_type == BPF_TRACE_ITER) { 20592 return bpf_iter_prog_supported(prog); 20593 } 20594 20595 if (prog->type == BPF_PROG_TYPE_LSM) { 20596 ret = bpf_lsm_verify_prog(&env->log, prog); 20597 if (ret < 0) 20598 return ret; 20599 } else if (prog->type == BPF_PROG_TYPE_TRACING && 20600 btf_id_set_contains(&btf_id_deny, btf_id)) { 20601 verbose(env, "Attaching tracing programs to function '%s' is rejected.\n", 20602 tgt_info.tgt_name); 20603 return -EINVAL; 20604 } else if ((prog->expected_attach_type == BPF_TRACE_FEXIT || 20605 prog->expected_attach_type == BPF_TRACE_FSESSION || 20606 prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI || 20607 prog->expected_attach_type == BPF_MODIFY_RETURN) && 20608 btf_id_set_contains(&noreturn_deny, btf_id)) { 20609 verbose(env, "Attaching fexit/fsession/fmod_ret to __noreturn function '%s' is rejected.\n", 20610 tgt_info.tgt_name); 20611 return -EINVAL; 20612 } 20613 20614 /* 20615 * We don't get trampoline for tracing_multi programs at this point, 20616 * it's done when tracing_multi link is created. 20617 */ 20618 if (prog->type == BPF_PROG_TYPE_TRACING && 20619 is_tracing_multi(prog->expected_attach_type)) 20620 return 0; 20621 20622 key = bpf_trampoline_compute_key(tgt_prog, prog->aux->attach_btf, btf_id); 20623 tr = bpf_trampoline_get(key, &tgt_info); 20624 if (!tr) 20625 return -ENOMEM; 20626 20627 if (tgt_prog && tgt_prog->aux->tail_call_reachable) 20628 bpf_trampoline_set_flags(tr, BPF_TRAMP_F_TAIL_CALL_CTX); 20629 20630 prog->aux->dst_trampoline = tr; 20631 return 0; 20632 } 20633 20634 int bpf_check_attach_btf_id_multi(struct btf *btf, struct bpf_prog *prog, u32 btf_id, 20635 struct bpf_attach_target_info *tgt_info) 20636 { 20637 const struct btf_type *t; 20638 unsigned long addr; 20639 const char *tname; 20640 int err; 20641 20642 if (!btf_id || !btf) 20643 return -EINVAL; 20644 20645 /* Check noreturn attachment. */ 20646 if ((prog->expected_attach_type == BPF_TRACE_FEXIT_MULTI || 20647 prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI) && 20648 btf_id_set_contains(&noreturn_deny, btf_id)) 20649 return -EINVAL; 20650 /* Check denied attachment. */ 20651 if (btf_id_set_contains(&btf_id_deny, btf_id)) 20652 return -EINVAL; 20653 20654 /* Check and get function target data. */ 20655 t = btf_type_by_id(btf, btf_id); 20656 if (!t) 20657 return -EINVAL; 20658 tname = btf_name_by_offset(btf, t->name_off); 20659 if (!tname) 20660 return -EINVAL; 20661 t = btf_attach_func_proto(NULL, btf, btf_id); 20662 if (!t || !btf_type_is_func_proto(t)) 20663 return -EINVAL; 20664 err = btf_distill_func_proto(NULL, btf, t, tname, &tgt_info->fmodel); 20665 if (err < 0) 20666 return err; 20667 if (tgt_info->fmodel.ret_size > 8 && 20668 attach_uses_trampoline_retval(prog->expected_attach_type)) 20669 return -EOPNOTSUPP; 20670 if (btf_is_module(btf)) { 20671 /* The bpf program already holds reference to module. */ 20672 if (WARN_ON_ONCE(!prog->aux->mod)) 20673 return -EINVAL; 20674 addr = find_kallsyms_symbol_value(prog->aux->mod, tname); 20675 } else { 20676 addr = kallsyms_lookup_name(tname); 20677 } 20678 if (!addr || !ftrace_location(addr)) 20679 return -ENOENT; 20680 20681 /* Check sleepable program attachment. */ 20682 if (prog->sleepable) { 20683 err = btf_id_allow_sleepable(btf_id, addr, prog, btf); 20684 if (err) 20685 return err; 20686 } 20687 tgt_info->tgt_addr = addr; 20688 return 0; 20689 } 20690 20691 struct btf *bpf_get_btf_vmlinux(void) 20692 { 20693 /* Pairs with the smp_store_release() on the parse path below. */ 20694 struct btf *btf = smp_load_acquire(&btf_vmlinux); 20695 20696 if (!btf && IS_ENABLED(CONFIG_DEBUG_INFO_BTF)) { 20697 mutex_lock(&btf_vmlinux_lock); 20698 btf = btf_vmlinux; 20699 if (!btf) { 20700 btf = btf_parse_vmlinux(); 20701 /* 20702 * Order the parsed BTF contents and the globals the 20703 * parse populated (e.g. bpf_ctx_convert.t) before 20704 * the pointer publication. Pairs with the acquire 20705 * on the lockless fast path above. 20706 */ 20707 smp_store_release(&btf_vmlinux, btf); 20708 } 20709 mutex_unlock(&btf_vmlinux_lock); 20710 } 20711 return btf; 20712 } 20713 20714 /* 20715 * The add_fd_from_fd_array() is executed only if fd_array_cnt is non-zero. In 20716 * this case expect that every file descriptor in the array is either a map or 20717 * a BTF. Everything else is considered to be trash. 20718 */ 20719 static int add_fd_from_fd_array(struct bpf_verifier_env *env, u32 idx, int fd) 20720 { 20721 struct bpf_map *map; 20722 struct btf *btf; 20723 CLASS(fd, f)(fd); 20724 int err; 20725 20726 map = __bpf_map_get(f); 20727 if (!IS_ERR(map)) { 20728 err = __add_used_map(env, map); 20729 if (err < 0) 20730 return err; 20731 fd_slot_set_map(&env->fd_array[idx], map); 20732 return 0; 20733 } 20734 20735 btf = __btf_get_by_fd(f); 20736 if (!IS_ERR(btf)) { 20737 btf_get(btf); 20738 err = __add_used_btf(env, btf); 20739 if (err < 0) 20740 return err; 20741 fd_slot_set_btf(&env->fd_array[idx], btf); 20742 return 0; 20743 } 20744 20745 verbose(env, "fd %d is not pointing to valid bpf_map or btf\n", fd); 20746 return PTR_ERR(map); 20747 } 20748 20749 /* 20750 * A continuous fd_array is resolved into an in-memory cache with one slot 20751 * per entry. The bound here is deliberately generous and not derived from 20752 * the per-program object limits: Duplicate entries /are/ permitted, and 20753 * the number of distinct maps and BTFs a program can bind is enforced when 20754 * each entry is resolved by __add_used_map() and __add_used_btf(). 20755 */ 20756 #define MAX_FD_ARRAY_CNT 4096 20757 20758 static int process_fd_array_continuous(struct bpf_verifier_env *env, 20759 bpfptr_t fd_array, u32 cnt) 20760 { 20761 int fd, ret; 20762 u32 i; 20763 20764 if (cnt > MAX_FD_ARRAY_CNT) { 20765 verbose(env, "fd_array has too many entries (%u, max %u)\n", 20766 cnt, MAX_FD_ARRAY_CNT); 20767 return -E2BIG; 20768 } 20769 20770 env->fd_array = kvzalloc_objs(*env->fd_array, cnt, GFP_KERNEL_ACCOUNT); 20771 if (!env->fd_array) 20772 return -ENOMEM; 20773 env->fd_array_cnt = cnt; 20774 for (i = 0; i < cnt; i++) { 20775 if (copy_from_bpfptr_offset(&fd, fd_array, 20776 (size_t)i * sizeof(fd), sizeof(fd))) 20777 return -EFAULT; 20778 ret = add_fd_from_fd_array(env, i, fd); 20779 if (ret) 20780 return ret; 20781 } 20782 return 0; 20783 } 20784 20785 static int process_fd_array(struct bpf_verifier_env *env, 20786 union bpf_attr *attr, bpfptr_t uattr) 20787 { 20788 bpfptr_t fd_array = make_bpfptr(attr->fd_array, uattr.is_kernel); 20789 20790 if (bpfptr_is_null(fd_array)) { 20791 if (attr->fd_array_cnt) { 20792 verbose(env, "fd_array_cnt %u without fd_array is invalid\n", 20793 attr->fd_array_cnt); 20794 return -EINVAL; 20795 } 20796 return 0; 20797 } 20798 /* 20799 * New API: the caller passes fd_array_cnt and a continuous array that 20800 * is resolved and bound up front. Legacy API (no fd_array_cnt): keep 20801 * the caller's array and resolve entries on the spot at each reference. 20802 */ 20803 if (attr->fd_array_cnt) 20804 return process_fd_array_continuous(env, fd_array, 20805 attr->fd_array_cnt); 20806 env->fd_array_raw = fd_array; 20807 return 0; 20808 } 20809 20810 /* replace a generic kfunc with a specialized version if necessary */ 20811 static int specialize_kfunc(struct bpf_verifier_env *env, struct bpf_kfunc_desc *desc, int insn_idx) 20812 { 20813 struct bpf_prog *prog = env->prog; 20814 bool seen_direct_write; 20815 void *xdp_kfunc; 20816 bool is_rdonly; 20817 u32 func_id = desc->func_id; 20818 u16 offset = desc->offset; 20819 unsigned long addr = desc->addr; 20820 20821 if (offset) /* return if module BTF is used */ 20822 return 0; 20823 20824 if (bpf_dev_bound_kfunc_id(func_id)) { 20825 xdp_kfunc = bpf_dev_bound_resolve_kfunc(prog, func_id); 20826 if (xdp_kfunc) 20827 addr = (unsigned long)xdp_kfunc; 20828 /* fallback to default kfunc when not supported by netdev */ 20829 } else if (func_id == special_kfunc_list[KF_bpf_dynptr_from_skb]) { 20830 seen_direct_write = env->seen_direct_write; 20831 is_rdonly = !may_access_direct_pkt_data(env, NULL, BPF_WRITE); 20832 20833 if (is_rdonly) 20834 addr = (unsigned long)bpf_dynptr_from_skb_rdonly; 20835 20836 /* restore env->seen_direct_write to its original value, since 20837 * may_access_direct_pkt_data mutates it 20838 */ 20839 env->seen_direct_write = seen_direct_write; 20840 } else if (func_id == special_kfunc_list[KF_bpf_set_dentry_xattr]) { 20841 if (bpf_lsm_has_d_inode_locked(prog)) 20842 addr = (unsigned long)bpf_set_dentry_xattr_locked; 20843 } else if (func_id == special_kfunc_list[KF_bpf_remove_dentry_xattr]) { 20844 if (bpf_lsm_has_d_inode_locked(prog)) 20845 addr = (unsigned long)bpf_remove_dentry_xattr_locked; 20846 } else if (func_id == special_kfunc_list[KF_bpf_dynptr_from_file]) { 20847 if (!env->insn_aux_data[insn_idx].non_sleepable) 20848 addr = (unsigned long)bpf_dynptr_from_file_sleepable; 20849 } else if (func_id == special_kfunc_list[KF_bpf_arena_alloc_pages]) { 20850 if (env->insn_aux_data[insn_idx].non_sleepable) 20851 addr = (unsigned long)bpf_arena_alloc_pages_non_sleepable; 20852 } else if (func_id == special_kfunc_list[KF_bpf_arena_free_pages]) { 20853 if (env->insn_aux_data[insn_idx].non_sleepable) 20854 addr = (unsigned long)bpf_arena_free_pages_non_sleepable; 20855 } 20856 desc->addr = addr; 20857 return 0; 20858 } 20859 20860 static void __fixup_collection_insert_kfunc(struct bpf_insn_aux_data *insn_aux, 20861 u16 struct_meta_reg, 20862 u16 node_offset_reg, 20863 struct bpf_insn *insn, 20864 struct bpf_insn *insn_buf, 20865 int *cnt) 20866 { 20867 struct btf_struct_meta *kptr_struct_meta = insn_aux->kptr_struct_meta; 20868 struct bpf_insn addr[2] = { BPF_LD_IMM64(struct_meta_reg, (long)kptr_struct_meta) }; 20869 20870 insn_buf[0] = addr[0]; 20871 insn_buf[1] = addr[1]; 20872 insn_buf[2] = BPF_MOV64_IMM(node_offset_reg, insn_aux->insert_off); 20873 insn_buf[3] = *insn; 20874 *cnt = 4; 20875 } 20876 20877 int bpf_fixup_kfunc_call(struct bpf_verifier_env *env, struct bpf_insn *insn, 20878 struct bpf_insn *insn_buf, int insn_idx, int *cnt) 20879 { 20880 struct bpf_kfunc_desc *desc; 20881 int err; 20882 20883 if (!insn->imm) { 20884 verbose(env, "invalid kernel function call not eliminated in verifier pass\n"); 20885 return -EINVAL; 20886 } 20887 20888 *cnt = 0; 20889 20890 /* insn->imm has the btf func_id. Replace it with an offset relative to 20891 * __bpf_call_base, unless the JIT needs to call functions that are 20892 * further than 32 bits away (bpf_jit_supports_far_kfunc_call()). 20893 */ 20894 desc = find_kfunc_desc(env->prog, insn->imm, insn->off); 20895 if (!desc) { 20896 verifier_bug(env, "kernel function descriptor not found for func_id %u", 20897 insn->imm); 20898 return -EFAULT; 20899 } 20900 20901 err = specialize_kfunc(env, desc, insn_idx); 20902 if (err) 20903 return err; 20904 20905 if (!bpf_jit_supports_far_kfunc_call()) 20906 insn->imm = BPF_CALL_IMM(desc->addr); 20907 20908 if (is_bpf_obj_new_kfunc(desc->func_id) || is_bpf_percpu_obj_new_kfunc(desc->func_id)) { 20909 struct btf_struct_meta *kptr_struct_meta = env->insn_aux_data[insn_idx].kptr_struct_meta; 20910 struct bpf_insn addr[2] = { BPF_LD_IMM64(BPF_REG_2, (long)kptr_struct_meta) }; 20911 u64 obj_new_size = env->insn_aux_data[insn_idx].obj_new_size; 20912 20913 if (is_bpf_percpu_obj_new_kfunc(desc->func_id) && kptr_struct_meta) { 20914 verifier_bug(env, "NULL kptr_struct_meta expected at insn_idx %d", 20915 insn_idx); 20916 return -EFAULT; 20917 } 20918 20919 insn_buf[0] = BPF_MOV64_IMM(BPF_REG_1, obj_new_size); 20920 insn_buf[1] = addr[0]; 20921 insn_buf[2] = addr[1]; 20922 insn_buf[3] = *insn; 20923 *cnt = 4; 20924 } else if (is_bpf_obj_drop_kfunc(desc->func_id) || 20925 is_bpf_percpu_obj_drop_kfunc(desc->func_id) || 20926 is_bpf_refcount_acquire_kfunc(desc->func_id)) { 20927 struct btf_struct_meta *kptr_struct_meta = env->insn_aux_data[insn_idx].kptr_struct_meta; 20928 struct bpf_insn addr[2] = { BPF_LD_IMM64(BPF_REG_2, (long)kptr_struct_meta) }; 20929 20930 if (is_bpf_percpu_obj_drop_kfunc(desc->func_id) && kptr_struct_meta) { 20931 verifier_bug(env, "NULL kptr_struct_meta expected at insn_idx %d", 20932 insn_idx); 20933 return -EFAULT; 20934 } 20935 20936 if (is_bpf_refcount_acquire_kfunc(desc->func_id) && !kptr_struct_meta) { 20937 verifier_bug(env, "kptr_struct_meta expected at insn_idx %d", 20938 insn_idx); 20939 return -EFAULT; 20940 } 20941 20942 insn_buf[0] = addr[0]; 20943 insn_buf[1] = addr[1]; 20944 insn_buf[2] = *insn; 20945 *cnt = 3; 20946 } else if (is_bpf_list_push_kfunc(desc->func_id) || 20947 is_bpf_rbtree_add_kfunc(desc->func_id)) { 20948 struct btf_struct_meta *kptr_struct_meta = env->insn_aux_data[insn_idx].kptr_struct_meta; 20949 int struct_meta_reg = BPF_REG_3; 20950 int node_offset_reg = BPF_REG_4; 20951 20952 /* list_add/rbtree_add have an extra arg (prev/less), 20953 * so args-to-fixup are in diff regs. 20954 */ 20955 if (desc->func_id == special_kfunc_list[KF_bpf_list_add] || 20956 is_bpf_rbtree_add_kfunc(desc->func_id)) { 20957 struct_meta_reg = BPF_REG_4; 20958 node_offset_reg = BPF_REG_5; 20959 } 20960 20961 if (!kptr_struct_meta) { 20962 verifier_bug(env, "kptr_struct_meta expected at insn_idx %d", 20963 insn_idx); 20964 return -EFAULT; 20965 } 20966 20967 __fixup_collection_insert_kfunc(&env->insn_aux_data[insn_idx], struct_meta_reg, 20968 node_offset_reg, insn, insn_buf, cnt); 20969 } else if (desc->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx] || 20970 desc->func_id == special_kfunc_list[KF_bpf_rdonly_cast]) { 20971 insn_buf[0] = BPF_MOV64_REG(BPF_REG_0, BPF_REG_1); 20972 *cnt = 1; 20973 } else if (desc->func_id == special_kfunc_list[KF_bpf_session_is_return] && 20974 (env->prog->expected_attach_type == BPF_TRACE_FSESSION || 20975 env->prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI)) { 20976 20977 /* 20978 * inline the bpf_session_is_return() for fsession: 20979 * bool bpf_session_is_return(void *ctx) 20980 * { 20981 * return (((u64 *)ctx)[-1] >> BPF_TRAMP_IS_RETURN_SHIFT) & 1; 20982 * } 20983 */ 20984 insn_buf[0] = BPF_LDX_MEM(BPF_DW, BPF_REG_0, BPF_REG_1, -8); 20985 insn_buf[1] = BPF_ALU64_IMM(BPF_RSH, BPF_REG_0, BPF_TRAMP_IS_RETURN_SHIFT); 20986 insn_buf[2] = BPF_ALU64_IMM(BPF_AND, BPF_REG_0, 1); 20987 *cnt = 3; 20988 } else if (desc->func_id == special_kfunc_list[KF_bpf_session_cookie] && 20989 (env->prog->expected_attach_type == BPF_TRACE_FSESSION || 20990 env->prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI)) { 20991 /* 20992 * inline bpf_session_cookie() for fsession: 20993 * __u64 *bpf_session_cookie(void *ctx) 20994 * { 20995 * u64 off = (((u64 *)ctx)[-1] >> BPF_TRAMP_COOKIE_INDEX_SHIFT) & 0xFF; 20996 * return &((u64 *)ctx)[-off]; 20997 * } 20998 */ 20999 insn_buf[0] = BPF_LDX_MEM(BPF_DW, BPF_REG_0, BPF_REG_1, -8); 21000 insn_buf[1] = BPF_ALU64_IMM(BPF_RSH, BPF_REG_0, BPF_TRAMP_COOKIE_INDEX_SHIFT); 21001 insn_buf[2] = BPF_ALU64_IMM(BPF_AND, BPF_REG_0, 0xFF); 21002 insn_buf[3] = BPF_ALU64_IMM(BPF_LSH, BPF_REG_0, 3); 21003 insn_buf[4] = BPF_ALU64_REG(BPF_SUB, BPF_REG_0, BPF_REG_1); 21004 insn_buf[5] = BPF_ALU64_IMM(BPF_NEG, BPF_REG_0, 0); 21005 *cnt = 6; 21006 } else if (desc->func_id == special_kfunc_list[KF_bpf_iter_num_new]) { 21007 /* inline bpf_iter_num_new(&it, start, end); R1=&it, R2=start, R3=end */ 21008 int i = 0; 21009 21010 /* if (start > end) goto einval; */ 21011 insn_buf[i++] = BPF_JMP32_REG(BPF_JSGT, BPF_REG_2, BPF_REG_3, 8); 21012 /* r0 = (u32)end - (u32)start; if (r0 > BPF_MAX_LOOPS) goto e2big; */ 21013 insn_buf[i++] = BPF_MOV32_REG(BPF_REG_0, BPF_REG_3); 21014 insn_buf[i++] = BPF_ALU32_REG(BPF_SUB, BPF_REG_0, BPF_REG_2); 21015 insn_buf[i++] = BPF_JMP_IMM(BPF_JGT, BPF_REG_0, BPF_MAX_LOOPS, 8); 21016 /* s->cur = start - 1; s->end = end; return 0; */ 21017 insn_buf[i++] = BPF_ALU32_IMM(BPF_ADD, BPF_REG_2, -1); 21018 insn_buf[i++] = BPF_STX_MEM(BPF_W, BPF_REG_1, BPF_REG_2, 0); 21019 insn_buf[i++] = BPF_STX_MEM(BPF_W, BPF_REG_1, BPF_REG_3, 4); 21020 insn_buf[i++] = BPF_MOV64_IMM(BPF_REG_0, 0); 21021 insn_buf[i++] = BPF_JMP_A(5); 21022 /* einval: s->cur = s->end = 0; return -EINVAL; */ 21023 insn_buf[i++] = BPF_ST_MEM(BPF_DW, BPF_REG_1, 0, 0); 21024 insn_buf[i++] = BPF_MOV64_IMM(BPF_REG_0, -EINVAL); 21025 insn_buf[i++] = BPF_JMP_A(2); 21026 /* e2big: s->cur = s->end = 0; return -E2BIG; */ 21027 insn_buf[i++] = BPF_ST_MEM(BPF_DW, BPF_REG_1, 0, 0); 21028 insn_buf[i++] = BPF_MOV64_IMM(BPF_REG_0, -E2BIG); 21029 *cnt = i; 21030 } else if (desc->func_id == special_kfunc_list[KF_bpf_iter_num_next]) { 21031 /* inline bpf_iter_num_next(&it); R1=&it, returns &s->cur or NULL */ 21032 int i = 0; 21033 21034 /* r0 = s->cur + 1; if ((s32)r0 >= s->end) goto done; */ 21035 insn_buf[i++] = BPF_LDX_MEM(BPF_W, BPF_REG_0, BPF_REG_1, 0); 21036 insn_buf[i++] = BPF_ALU32_IMM(BPF_ADD, BPF_REG_0, 1); 21037 insn_buf[i++] = BPF_LDX_MEM(BPF_W, BPF_REG_2, BPF_REG_1, 4); 21038 insn_buf[i++] = BPF_JMP32_REG(BPF_JSGE, BPF_REG_0, BPF_REG_2, 3); 21039 /* s->cur = r0; return &s->cur; */ 21040 insn_buf[i++] = BPF_STX_MEM(BPF_W, BPF_REG_1, BPF_REG_0, 0); 21041 insn_buf[i++] = BPF_MOV64_REG(BPF_REG_0, BPF_REG_1); 21042 insn_buf[i++] = BPF_JMP_A(2); 21043 /* done: s->cur = s->end = 0; return NULL; */ 21044 insn_buf[i++] = BPF_ST_MEM(BPF_DW, BPF_REG_1, 0, 0); 21045 insn_buf[i++] = BPF_MOV64_IMM(BPF_REG_0, 0); 21046 *cnt = i; 21047 } else if (desc->func_id == special_kfunc_list[KF_bpf_iter_num_destroy]) { 21048 /* bpf_iter_num_destroy() is a no-op; emit a nop to drop the call */ 21049 insn_buf[0] = BPF_JMP_A(0); 21050 *cnt = 1; 21051 } 21052 21053 if (env->insn_aux_data[insn_idx].arg_prog) { 21054 u32 regno = env->insn_aux_data[insn_idx].arg_prog; 21055 struct bpf_insn ld_addrs[2] = { BPF_LD_IMM64(regno, (long)env->prog->aux) }; 21056 int idx = *cnt; 21057 21058 insn_buf[idx++] = ld_addrs[0]; 21059 insn_buf[idx++] = ld_addrs[1]; 21060 insn_buf[idx++] = *insn; 21061 *cnt = idx; 21062 } 21063 return 0; 21064 } 21065 21066 static enum bpf_sig_keyring bpf_classify_keyring(s32 keyring_id) 21067 { 21068 switch (keyring_id) { 21069 case 0: 21070 return BPF_SIG_KEYRING_BUILTIN; 21071 case (s32)(unsigned long)VERIFY_USE_SECONDARY_KEYRING: 21072 return BPF_SIG_KEYRING_SECONDARY; 21073 case (s32)(unsigned long)VERIFY_USE_PLATFORM_KEYRING: 21074 return BPF_SIG_KEYRING_PLATFORM; 21075 default: 21076 return BPF_SIG_KEYRING_USER; 21077 } 21078 } 21079 21080 /* 21081 * Verify the PKCS#7 signature of a loaded program. Called from bpf_check() 21082 * once the program's metadata maps have been resolved into used_maps, so 21083 * the exact maps folded into the signature are the ones the program binds. 21084 * 21085 * The signature covers the instructions followed by the frozen contents of 21086 * each map, in @maps order: insns || map_0 || map_1 || [...]. On success the 21087 * verdict and keyring info are recorded on prog->aux. 21088 */ 21089 static int bpf_prog_verify_signature(struct bpf_verifier_env *env, 21090 union bpf_attr *attr, bool is_kernel) 21091 { 21092 bpfptr_t usig = make_bpfptr(attr->signature, is_kernel); 21093 struct bpf_dynptr_kern sig_ptr, data_ptr; 21094 struct bpf_prog *prog = env->prog; 21095 struct bpf_map **maps = env->used_maps; 21096 struct bpf_key *key = NULL; 21097 void *sig, *data = NULL; 21098 u32 map_cnt = env->used_map_cnt; 21099 u32 i, off, insns_sz; 21100 u64 data_sz; 21101 int err = 0; 21102 21103 /* 21104 * Don't attempt to use kmalloc_large or vmalloc for signatures. 21105 * Practical signature for BPF program should be below this limit. 21106 */ 21107 if (!attr->signature_size || 21108 attr->signature_size > KMALLOC_MAX_CACHE_SIZE) 21109 return -EINVAL; 21110 if (system_keyring_id_check(attr->keyring_id) == 0) 21111 key = bpf_lookup_system_key(attr->keyring_id); 21112 else 21113 key = bpf_lookup_user_key(attr->keyring_id, 0); 21114 if (!key) { 21115 verbose(env, "cannot resolve signing keyring with keyring_id %d\n", 21116 attr->keyring_id); 21117 return -EINVAL; 21118 } 21119 21120 sig = kvmemdup_bpfptr(usig, attr->signature_size); 21121 if (IS_ERR(sig)) { 21122 bpf_key_put(key); 21123 return PTR_ERR(sig); 21124 } 21125 21126 insns_sz = prog->len * sizeof(struct bpf_insn); 21127 data_sz = insns_sz; 21128 for (i = 0; i < map_cnt; i++) { 21129 struct bpf_map *map = maps[i]; 21130 21131 if (map->map_type != BPF_MAP_TYPE_ARRAY || 21132 !map->ops->map_direct_value_addr) { 21133 verbose(env, "signed program metadata map '%s' must be an array\n", 21134 map->name); 21135 err = -EINVAL; 21136 goto out; 21137 } 21138 if (!READ_ONCE(map->frozen)) { 21139 verbose(env, "signed program metadata map '%s' must be frozen\n", 21140 map->name); 21141 err = -EPERM; 21142 goto out; 21143 } 21144 if (bpf_map_write_active(map)) { 21145 verbose(env, "signed program metadata map '%s' has active writers\n", 21146 map->name); 21147 err = -EBUSY; 21148 goto out; 21149 } 21150 if (!map->excl_prog_sha) { 21151 verbose(env, "signed program metadata map '%s' must be exclusive\n", 21152 map->name); 21153 err = -EPERM; 21154 goto out; 21155 } 21156 data_sz += map->value_size; 21157 } 21158 if (bpf_dynptr_check_size(data_sz)) { 21159 verbose(env, "signed payload too large: %llu bytes\n", data_sz); 21160 err = -E2BIG; 21161 goto out; 21162 } 21163 data = kvmalloc(data_sz, GFP_KERNEL_ACCOUNT | __GFP_ZERO); 21164 if (!data) { 21165 err = -ENOMEM; 21166 goto out; 21167 } 21168 memcpy(data, prog->insnsi, insns_sz); 21169 off = insns_sz; 21170 for (i = 0; i < map_cnt; i++) { 21171 struct bpf_map *map = maps[i]; 21172 u64 addr; 21173 21174 err = map->ops->map_direct_value_addr(map, &addr, 0); 21175 if (err) { 21176 verbose(env, "failed to read signed metadata map '%s': %d\n", 21177 map->name, err); 21178 goto out; 21179 } 21180 memcpy(data + off, (void *)(unsigned long)addr, 21181 map->value_size); 21182 off += map->value_size; 21183 } 21184 21185 bpf_dynptr_init(&data_ptr, data, BPF_DYNPTR_TYPE_LOCAL, 0, data_sz); 21186 bpf_dynptr_init(&sig_ptr, sig, BPF_DYNPTR_TYPE_LOCAL, 0, 21187 attr->signature_size); 21188 21189 err = bpf_verify_pkcs7_signature((struct bpf_dynptr *)&data_ptr, 21190 (struct bpf_dynptr *)&sig_ptr, key); 21191 if (err) { 21192 verbose(env, "signature verification failed: %d\n", err); 21193 } else { 21194 verbose(env, "signature verification passed\n"); 21195 prog->aux->sig.keyring_serial = bpf_key_serial(key); 21196 prog->aux->sig.keyring_type = bpf_classify_keyring(attr->keyring_id); 21197 prog->aux->sig.verdict = BPF_SIG_VERIFIED; 21198 } 21199 out: 21200 kvfree(data); 21201 bpf_key_put(key); 21202 kvfree(sig); 21203 return err; 21204 } 21205 21206 int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr, 21207 struct bpf_log_attr *attr_log) 21208 { 21209 u64 start_time = ktime_get_ns(); 21210 struct bpf_verifier_env *env; 21211 int i, len, ret = -EINVAL, err; 21212 bool is_priv; 21213 21214 BTF_TYPE_EMIT(enum bpf_features); 21215 21216 /* no program is valid */ 21217 if (ARRAY_SIZE(bpf_verifier_ops) == 0) 21218 return -EINVAL; 21219 21220 /* 'struct bpf_verifier_env' can be global, but since it's not small, 21221 * allocate/free it every time bpf_check() is called 21222 */ 21223 env = kvzalloc_obj(struct bpf_verifier_env, GFP_KERNEL_ACCOUNT); 21224 if (!env) 21225 return -ENOMEM; 21226 21227 env->bt.env = env; 21228 env->prog = *prog; 21229 env->ops = bpf_verifier_ops[env->prog->type]; 21230 21231 env->allow_ptr_leaks = bpf_allow_ptr_leaks(env->prog->aux->token); 21232 env->allow_uninit_stack = bpf_allow_uninit_stack(env->prog->aux->token); 21233 env->bypass_spec_v1 = bpf_bypass_spec_v1(env->prog->aux->token); 21234 env->bypass_spec_v4 = bpf_bypass_spec_v4(env->prog->aux->token); 21235 env->bpf_capable = is_priv = bpf_token_capable(env->prog->aux->token, CAP_BPF); 21236 env->signature = attr->signature; 21237 21238 /* user could have requested verbose verifier output 21239 * and supplied buffer to store the verification trace 21240 */ 21241 ret = bpf_vlog_init(&env->log, attr_log->level, attr_log->ubuf, attr_log->size); 21242 if (ret) 21243 goto err_free_env; 21244 ret = bpf_diag_init(env); 21245 if (ret) 21246 goto err_prep; 21247 if (env->prog->insnsi[env->prog->len - 1].code == (BPF_LD | BPF_IMM | BPF_DW)) { 21248 verbose(env, "invalid bpf_ld_imm64 insn\n"); 21249 ret = -EINVAL; 21250 goto err_prep; 21251 } 21252 if (env->signature) { 21253 ret = bpf_prog_calc_tag(env->prog); 21254 if (ret < 0) 21255 goto err_prep; 21256 } 21257 21258 ret = process_fd_array(env, attr, uattr); 21259 if (ret) 21260 goto err_prep; 21261 21262 if (env->signature) { 21263 ret = bpf_prog_verify_signature(env, attr, uattr.is_kernel); 21264 if (ret) 21265 goto err_prep; 21266 } 21267 21268 ret = security_bpf_prog_load(env->prog, attr, env->prog->aux->token, 21269 uattr.is_kernel); 21270 if (ret) 21271 goto err_prep; 21272 21273 bpf_get_btf_vmlinux(); 21274 21275 /* Serialize verification of unprivileged programs. */ 21276 if (!is_priv) 21277 mutex_lock(&bpf_verifier_lock); 21278 21279 len = env->insn_aux_data_len = env->prog->len; 21280 env->insn_aux_data = 21281 __vmalloc(array_size(sizeof(struct bpf_insn_aux_data), len), 21282 GFP_KERNEL_ACCOUNT | __GFP_ZERO); 21283 ret = -ENOMEM; 21284 if (!env->insn_aux_data) 21285 goto skip_full_check; 21286 for (i = 0; i < len; i++) 21287 env->insn_aux_data[i].orig_idx = i; 21288 env->succ = bpf_iarray_realloc(NULL, 2); 21289 if (!env->succ) 21290 goto skip_full_check; 21291 21292 mark_verifier_state_clean(env); 21293 21294 if (IS_ERR(btf_vmlinux)) { 21295 /* Either gcc or pahole or kernel are broken. */ 21296 verbose(env, "in-kernel BTF is malformed\n"); 21297 ret = PTR_ERR(btf_vmlinux); 21298 goto skip_full_check; 21299 } 21300 21301 env->strict_alignment = !!(attr->prog_flags & BPF_F_STRICT_ALIGNMENT); 21302 if (!IS_ENABLED(CONFIG_HAVE_EFFICIENT_UNALIGNED_ACCESS)) 21303 env->strict_alignment = true; 21304 if (attr->prog_flags & BPF_F_ANY_ALIGNMENT) 21305 env->strict_alignment = false; 21306 21307 if (is_priv) 21308 env->test_state_freq = attr->prog_flags & BPF_F_TEST_STATE_FREQ; 21309 env->test_reg_invariants = attr->prog_flags & BPF_F_TEST_REG_INVARIANTS; 21310 21311 env->explored_states = kvzalloc_objs(struct list_head, 21312 state_htab_size(env), 21313 GFP_KERNEL_ACCOUNT); 21314 ret = -ENOMEM; 21315 if (!env->explored_states) 21316 goto skip_full_check; 21317 21318 for (i = 0; i < state_htab_size(env); i++) 21319 INIT_LIST_HEAD(&env->explored_states[i]); 21320 INIT_LIST_HEAD(&env->free_list); 21321 21322 /* Prepare BTF and func_info needed to discover all subprograms. */ 21323 ret = bpf_prepare_btf_info(env, attr, uattr); 21324 if (ret < 0) 21325 goto skip_full_check; 21326 21327 /* Apply CO-RE before validating the program's instruction layout. */ 21328 ret = bpf_check_core_relo(env, attr, uattr); 21329 if (ret < 0) 21330 goto skip_full_check; 21331 21332 /* Discover all subprograms before validating their layout and BTF. */ 21333 ret = add_subprogs(env); 21334 if (ret < 0) 21335 goto skip_full_check; 21336 21337 ret = check_subprogs(env); 21338 if (ret < 0) 21339 goto skip_full_check; 21340 21341 /* Validate BTF against the complete subprogram layout. */ 21342 ret = bpf_check_btf_info(env, attr, uattr); 21343 if (ret < 0) 21344 goto skip_full_check; 21345 21346 /* Validate instructions and resolve the program's referenced resources. */ 21347 ret = check_and_resolve_insns(env); 21348 if (ret < 0) 21349 goto skip_full_check; 21350 21351 /* Build kfunc prototypes after resolving program resources. */ 21352 ret = add_kfuncs(env); 21353 if (ret < 0) 21354 goto skip_full_check; 21355 21356 if (bpf_prog_is_offloaded(env->prog->aux)) { 21357 ret = bpf_prog_offload_verifier_prep(env->prog); 21358 if (ret) 21359 goto skip_full_check; 21360 } 21361 21362 ret = bpf_check_cfg(env); 21363 if (ret < 0) 21364 goto skip_full_check; 21365 21366 ret = bpf_compute_postorder(env); 21367 if (ret < 0) 21368 goto skip_full_check; 21369 21370 ret = bpf_stack_liveness_init(env); 21371 if (ret) 21372 goto skip_full_check; 21373 21374 ret = check_attach_btf_id(env); 21375 if (ret) 21376 goto skip_full_check; 21377 21378 ret = bpf_compute_const_regs(env); 21379 if (ret < 0) 21380 goto skip_full_check; 21381 21382 ret = bpf_prune_dead_branches(env); 21383 if (ret < 0) 21384 goto skip_full_check; 21385 21386 ret = sort_subprogs_topo(env); 21387 if (ret < 0) 21388 goto skip_full_check; 21389 21390 ret = bpf_compute_scc(env); 21391 if (ret < 0) 21392 goto skip_full_check; 21393 21394 ret = bpf_compute_live_registers(env); 21395 if (ret < 0) 21396 goto skip_full_check; 21397 21398 ret = mark_fastcall_patterns(env); 21399 if (ret < 0) 21400 goto skip_full_check; 21401 21402 ret = do_check_main(env); 21403 ret = ret ?: do_check_subprogs(env); 21404 21405 if (ret == 0 && bpf_prog_is_offloaded(env->prog->aux)) 21406 ret = bpf_prog_offload_finalize(env); 21407 21408 skip_full_check: 21409 kvfree(env->explored_states); 21410 21411 /* might decrease stack depth, keep it before passes that 21412 * allocate additional slots. 21413 */ 21414 if (ret == 0) 21415 ret = bpf_remove_fastcall_spills_fills(env); 21416 21417 if (ret == 0) 21418 ret = check_max_stack_depth(env); 21419 21420 /* instruction rewrites happen after this point */ 21421 if (ret == 0) 21422 ret = bpf_optimize_bpf_loop(env); 21423 21424 if (is_priv) { 21425 if (ret == 0) 21426 bpf_opt_hard_wire_dead_code_branches(env); 21427 if (ret == 0) 21428 ret = bpf_opt_remove_dead_code(env); 21429 if (ret == 0) 21430 ret = bpf_opt_remove_nops(env); 21431 } else { 21432 if (ret == 0) 21433 sanitize_dead_code(env); 21434 } 21435 21436 if (ret == 0) 21437 /* program is valid, convert *(u32*)(ctx + off) accesses */ 21438 ret = bpf_convert_ctx_accesses(env); 21439 21440 if (ret == 0) 21441 ret = bpf_do_misc_fixups(env); 21442 21443 /* do 32-bit optimization after insn patching has done so those patched 21444 * insns could be handled correctly. 21445 */ 21446 if (ret == 0 && !bpf_prog_is_offloaded(env->prog->aux)) { 21447 ret = bpf_opt_subreg_zext_lo32_rnd_hi32(env, attr); 21448 env->prog->aux->verifier_zext = bpf_jit_needs_zext() ? !ret 21449 : false; 21450 } 21451 21452 if (ret == 0) 21453 ret = bpf_fixup_call_args(env); 21454 21455 env->verification_time = ktime_get_ns() - start_time; 21456 print_verification_stats(env); 21457 env->prog->aux->verified_insns = env->insn_processed; 21458 21459 /* preserve original error even if log finalization is successful */ 21460 err = bpf_log_attr_finalize(attr_log, &env->log); 21461 if (err) 21462 ret = err; 21463 21464 if (ret) 21465 goto err_release_maps; 21466 21467 if (env->used_map_cnt) { 21468 /* if program passed verifier, update used_maps in bpf_prog_info */ 21469 env->prog->aux->used_maps = kmalloc_objs(env->used_maps[0], 21470 env->used_map_cnt, 21471 GFP_KERNEL_ACCOUNT); 21472 21473 if (!env->prog->aux->used_maps) { 21474 ret = -ENOMEM; 21475 goto err_release_maps; 21476 } 21477 21478 memcpy(env->prog->aux->used_maps, env->used_maps, 21479 sizeof(env->used_maps[0]) * env->used_map_cnt); 21480 env->prog->aux->used_map_cnt = env->used_map_cnt; 21481 } 21482 if (env->used_btf_cnt) { 21483 /* if program passed verifier, update used_btfs in bpf_prog_aux */ 21484 env->prog->aux->used_btfs = kmalloc_objs(env->used_btfs[0], 21485 env->used_btf_cnt, 21486 GFP_KERNEL_ACCOUNT); 21487 if (!env->prog->aux->used_btfs) { 21488 ret = -ENOMEM; 21489 goto err_release_maps; 21490 } 21491 21492 memcpy(env->prog->aux->used_btfs, env->used_btfs, 21493 sizeof(env->used_btfs[0]) * env->used_btf_cnt); 21494 env->prog->aux->used_btf_cnt = env->used_btf_cnt; 21495 } 21496 if (env->used_map_cnt || env->used_btf_cnt) { 21497 /* program is valid. Convert pseudo bpf_ld_imm64 into generic 21498 * bpf_ld_imm64 instructions 21499 */ 21500 convert_pseudo_ld_imm64(env); 21501 } 21502 21503 adjust_btf_func(env); 21504 21505 /* extension progs temporarily inherit the attach_type of their targets 21506 for verification purposes, so set it back to zero before returning 21507 */ 21508 if (env->prog->type == BPF_PROG_TYPE_EXT) 21509 env->prog->expected_attach_type = 0; 21510 21511 env->prog = __bpf_prog_select_runtime(env, env->prog, &ret); 21512 21513 err_release_maps: 21514 if (ret) 21515 release_insn_arrays(env); 21516 if (!env->prog->aux->used_maps) 21517 /* if we didn't copy map pointers into bpf_prog_info, release 21518 * them now. Otherwise free_used_maps() will release them. 21519 */ 21520 release_maps(env); 21521 if (!env->prog->aux->used_btfs) 21522 release_btfs(env); 21523 21524 *prog = env->prog; 21525 21526 module_put(env->attach_btf_mod); 21527 if (!is_priv) 21528 mutex_unlock(&bpf_verifier_lock); 21529 goto err_free_env; 21530 err_prep: 21531 err = bpf_log_attr_finalize(attr_log, &env->log); 21532 if (err) 21533 ret = err; 21534 release_insn_arrays(env); 21535 release_maps(env); 21536 release_btfs(env); 21537 err_free_env: 21538 if (env->insn_aux_data) 21539 bpf_clear_insn_aux_data(env, 0, env->insn_aux_data_len); 21540 vfree(env->insn_aux_data); 21541 kvfree(env->fd_array); 21542 bpf_stack_liveness_free(env); 21543 kvfree(env->cfg.insn_postorder); 21544 kvfree(env->scc_info); 21545 kvfree(env->succ); 21546 kvfree(env->gotox_tmp_buf); 21547 bpf_diag_free(env); 21548 kvfree(env); 21549 return ret; 21550 } 21551