xref: /linux/kernel/kprobes.c (revision 5bfa9f1a9dcb6ecb607adbc1c0226605c972935b)
1 // SPDX-License-Identifier: GPL-2.0-or-later
2 /*
3  *  Kernel Probes (KProbes)
4  *
5  * Copyright (C) IBM Corporation, 2002, 2004
6  *
7  * 2002-Oct	Created by Vamsi Krishna S <vamsi_krishna@in.ibm.com> Kernel
8  *		Probes initial implementation (includes suggestions from
9  *		Rusty Russell).
10  * 2004-Aug	Updated by Prasanna S Panchamukhi <prasanna@in.ibm.com> with
11  *		hlists and exceptions notifier as suggested by Andi Kleen.
12  * 2004-July	Suparna Bhattacharya <suparna@in.ibm.com> added jumper probes
13  *		interface to access function arguments.
14  * 2004-Sep	Prasanna S Panchamukhi <prasanna@in.ibm.com> Changed Kprobes
15  *		exceptions notifier to be first on the priority list.
16  * 2005-May	Hien Nguyen <hien@us.ibm.com>, Jim Keniston
17  *		<jkenisto@us.ibm.com> and Prasanna S Panchamukhi
18  *		<prasanna@in.ibm.com> added function-return probes.
19  */
20 
21 #define pr_fmt(fmt) "kprobes: " fmt
22 
23 #include <linux/kprobes.h>
24 #include <linux/hash.h>
25 #include <linux/init.h>
26 #include <linux/slab.h>
27 #include <linux/stddef.h>
28 #include <linux/export.h>
29 #include <linux/kallsyms.h>
30 #include <linux/freezer.h>
31 #include <linux/seq_file.h>
32 #include <linux/debugfs.h>
33 #include <linux/sysctl.h>
34 #include <linux/kdebug.h>
35 #include <linux/kthread.h>
36 #include <linux/memory.h>
37 #include <linux/ftrace.h>
38 #include <linux/cpu.h>
39 #include <linux/jump_label.h>
40 #include <linux/static_call.h>
41 #include <linux/perf_event.h>
42 #include <linux/execmem.h>
43 #include <linux/cleanup.h>
44 #include <linux/wait.h>
45 #include <linux/wait_bit.h>
46 
47 #include <asm/sections.h>
48 #include <asm/cacheflush.h>
49 #include <asm/errno.h>
50 #include <linux/uaccess.h>
51 
52 #define KPROBE_HASH_BITS 6
53 #define KPROBE_TABLE_SIZE (1 << KPROBE_HASH_BITS)
54 
55 #if !defined(CONFIG_OPTPROBES) || !defined(CONFIG_SYSCTL)
56 #define kprobe_sysctls_init() do { } while (0)
57 #endif
58 
59 static int kprobes_initialized;
60 /* kprobe_table can be accessed by
61  * - Normal hlist traversal and RCU add/del under 'kprobe_mutex' is held.
62  * Or
63  * - RCU hlist traversal under disabling preempt (breakpoint handlers)
64  */
65 static struct hlist_head kprobe_table[KPROBE_TABLE_SIZE];
66 
67 /* NOTE: change this value only with 'kprobe_mutex' held */
68 static bool kprobes_all_disarmed;
69 
70 /* This protects 'kprobe_table' and 'optimizing_list' */
71 static DEFINE_MUTEX(kprobe_mutex);
72 static DEFINE_PER_CPU(struct kprobe *, kprobe_instance);
73 
74 kprobe_opcode_t * __weak kprobe_lookup_name(const char *name,
75 					unsigned int __unused)
76 {
77 	return ((kprobe_opcode_t *)(kallsyms_lookup_name(name)));
78 }
79 
80 /*
81  * Blacklist -- list of 'struct kprobe_blacklist_entry' to store info where
82  * kprobes can not probe.
83  */
84 static LIST_HEAD(kprobe_blacklist);
85 
86 #ifdef __ARCH_WANT_KPROBES_INSN_SLOT
87 /*
88  * 'kprobe::ainsn.insn' points to the copy of the instruction to be
89  * single-stepped. x86_64, POWER4 and above have no-exec support and
90  * stepping on the instruction on a vmalloced/kmalloced/data page
91  * is a recipe for disaster
92  */
93 struct kprobe_insn_page {
94 	struct list_head list;
95 	kprobe_opcode_t *insns;		/* Page of instruction slots */
96 	struct kprobe_insn_cache *cache;
97 	int nused;
98 	int ngarbage;
99 	char slot_used[];
100 };
101 
102 static int slots_per_page(struct kprobe_insn_cache *c)
103 {
104 	return PAGE_SIZE/(c->insn_size * sizeof(kprobe_opcode_t));
105 }
106 
107 enum kprobe_slot_state {
108 	SLOT_CLEAN = 0,
109 	SLOT_DIRTY = 1,
110 	SLOT_USED = 2,
111 };
112 
113 void __weak *alloc_insn_page(void)
114 {
115 	/*
116 	 * Use execmem_alloc() so this page is within +/- 2GB of where the
117 	 * kernel image and loaded module images reside. This is required
118 	 * for most of the architectures.
119 	 * (e.g. x86-64 needs this to handle the %rip-relative fixups.)
120 	 */
121 	return execmem_alloc(EXECMEM_KPROBES, PAGE_SIZE);
122 }
123 
124 static void free_insn_page(void *page)
125 {
126 	execmem_free(page);
127 }
128 
129 struct kprobe_insn_cache kprobe_insn_slots = {
130 	.mutex = __MUTEX_INITIALIZER(kprobe_insn_slots.mutex),
131 	.alloc = alloc_insn_page,
132 	.free = free_insn_page,
133 	.sym = KPROBE_INSN_PAGE_SYM,
134 	.pages = LIST_HEAD_INIT(kprobe_insn_slots.pages),
135 	.insn_size = MAX_INSN_SIZE,
136 	.nr_garbage = 0,
137 };
138 static int collect_garbage_slots(struct kprobe_insn_cache *c);
139 
140 /**
141  * __get_insn_slot - Find a slot on an executable page for an instruction.
142  * @c: Pointer to kprobe instruction cache
143  *
144  * Description: Locates available slot on existing executable pages,
145  *              allocates an executable page if there's no room on existing ones.
146  * Return: Pointer to instruction slot on success, NULL on failure.
147  */
148 kprobe_opcode_t *__get_insn_slot(struct kprobe_insn_cache *c)
149 {
150 	struct kprobe_insn_page *kip;
151 
152 	/* Since the slot array is not protected by rcu, we need a mutex */
153 	guard(mutex)(&c->mutex);
154 	do {
155 		guard(rcu)();
156 		list_for_each_entry_rcu(kip, &c->pages, list) {
157 			if (kip->nused < slots_per_page(c)) {
158 				int i;
159 
160 				for (i = 0; i < slots_per_page(c); i++) {
161 					if (kip->slot_used[i] == SLOT_CLEAN) {
162 						kip->slot_used[i] = SLOT_USED;
163 						kip->nused++;
164 						return kip->insns + (i * c->insn_size);
165 					}
166 				}
167 				/* kip->nused is broken. Fix it. */
168 				kip->nused = slots_per_page(c);
169 				WARN_ON(1);
170 			}
171 		}
172 	/* If there are any garbage slots, collect it and try again. */
173 	} while (c->nr_garbage && collect_garbage_slots(c) == 0);
174 
175 	/* All out of space.  Need to allocate a new page. */
176 	kip = kmalloc_flex(*kip, slot_used, slots_per_page(c));
177 	if (!kip)
178 		return NULL;
179 
180 	kip->insns = c->alloc();
181 	if (!kip->insns) {
182 		kfree(kip);
183 		return NULL;
184 	}
185 	INIT_LIST_HEAD(&kip->list);
186 	memset(kip->slot_used, SLOT_CLEAN, slots_per_page(c));
187 	kip->slot_used[0] = SLOT_USED;
188 	kip->nused = 1;
189 	kip->ngarbage = 0;
190 	kip->cache = c;
191 	list_add_rcu(&kip->list, &c->pages);
192 
193 	/* Record the perf ksymbol register event after adding the page */
194 	perf_event_ksymbol(PERF_RECORD_KSYMBOL_TYPE_OOL, (unsigned long)kip->insns,
195 			   PAGE_SIZE, false, c->sym);
196 
197 	return kip->insns;
198 }
199 
200 /* Return true if all garbages are collected, otherwise false. */
201 static bool collect_one_slot(struct kprobe_insn_page *kip, int idx)
202 {
203 	kip->slot_used[idx] = SLOT_CLEAN;
204 	kip->nused--;
205 	if (kip->nused != 0)
206 		return false;
207 
208 	/*
209 	 * Page is no longer in use.  Free it unless
210 	 * it's the last one.  We keep the last one
211 	 * so as not to have to set it up again the
212 	 * next time somebody inserts a probe.
213 	 */
214 	if (!list_is_singular(&kip->list)) {
215 		/*
216 		 * Record perf ksymbol unregister event before removing
217 		 * the page.
218 		 */
219 		perf_event_ksymbol(PERF_RECORD_KSYMBOL_TYPE_OOL,
220 				   (unsigned long)kip->insns, PAGE_SIZE, true,
221 				   kip->cache->sym);
222 		list_del_rcu(&kip->list);
223 		synchronize_rcu();
224 		kip->cache->free(kip->insns);
225 		kfree(kip);
226 	}
227 	return true;
228 }
229 
230 static int collect_garbage_slots(struct kprobe_insn_cache *c)
231 {
232 	struct kprobe_insn_page *kip, *next;
233 
234 	/* Ensure no-one is interrupted on the garbages */
235 	synchronize_rcu();
236 
237 	list_for_each_entry_safe(kip, next, &c->pages, list) {
238 		int i;
239 
240 		if (kip->ngarbage == 0)
241 			continue;
242 		kip->ngarbage = 0;	/* we will collect all garbages */
243 		for (i = 0; i < slots_per_page(c); i++) {
244 			if (kip->slot_used[i] == SLOT_DIRTY && collect_one_slot(kip, i))
245 				break;
246 		}
247 	}
248 	c->nr_garbage = 0;
249 	return 0;
250 }
251 
252 static long __find_insn_page(struct kprobe_insn_cache *c,
253 	kprobe_opcode_t *slot, struct kprobe_insn_page **pkip)
254 {
255 	struct kprobe_insn_page *kip = NULL;
256 	long idx;
257 
258 	guard(rcu)();
259 	list_for_each_entry_rcu(kip, &c->pages, list) {
260 		idx = ((long)slot - (long)kip->insns) /
261 			(c->insn_size * sizeof(kprobe_opcode_t));
262 		if (idx >= 0 && idx < slots_per_page(c)) {
263 			*pkip = kip;
264 			return idx;
265 		}
266 	}
267 	/* Could not find this slot. */
268 	WARN_ON(1);
269 	*pkip = NULL;
270 	return -1;
271 }
272 
273 void __free_insn_slot(struct kprobe_insn_cache *c,
274 		      kprobe_opcode_t *slot, int dirty)
275 {
276 	struct kprobe_insn_page *kip = NULL;
277 	long idx;
278 
279 	guard(mutex)(&c->mutex);
280 	idx = __find_insn_page(c, slot, &kip);
281 	/* Mark and sweep: this may sleep */
282 	if (kip) {
283 		/* Check double free */
284 		WARN_ON(kip->slot_used[idx] != SLOT_USED);
285 		if (dirty) {
286 			kip->slot_used[idx] = SLOT_DIRTY;
287 			kip->ngarbage++;
288 			if (++c->nr_garbage > slots_per_page(c))
289 				collect_garbage_slots(c);
290 		} else {
291 			collect_one_slot(kip, idx);
292 		}
293 	}
294 }
295 
296 /*
297  * Check given address is on the page of kprobe instruction slots.
298  * This will be used for checking whether the address on a stack
299  * is on a text area or not.
300  */
301 bool __is_insn_slot_addr(struct kprobe_insn_cache *c, unsigned long addr)
302 {
303 	struct kprobe_insn_page *kip;
304 	bool ret = false;
305 
306 	rcu_read_lock();
307 	list_for_each_entry_rcu(kip, &c->pages, list) {
308 		if (addr >= (unsigned long)kip->insns &&
309 		    addr < (unsigned long)kip->insns + PAGE_SIZE) {
310 			ret = true;
311 			break;
312 		}
313 	}
314 	rcu_read_unlock();
315 
316 	return ret;
317 }
318 
319 int kprobe_cache_get_kallsym(struct kprobe_insn_cache *c, unsigned int *symnum,
320 			     unsigned long *value, char *type, char *sym)
321 {
322 	struct kprobe_insn_page *kip;
323 	int ret = -ERANGE;
324 
325 	rcu_read_lock();
326 	list_for_each_entry_rcu(kip, &c->pages, list) {
327 		if ((*symnum)--)
328 			continue;
329 		strscpy(sym, c->sym, KSYM_NAME_LEN);
330 		*type = 't';
331 		*value = (unsigned long)kip->insns;
332 		ret = 0;
333 		break;
334 	}
335 	rcu_read_unlock();
336 
337 	return ret;
338 }
339 
340 #ifdef CONFIG_OPTPROBES
341 void __weak *alloc_optinsn_page(void)
342 {
343 	return alloc_insn_page();
344 }
345 
346 void __weak free_optinsn_page(void *page)
347 {
348 	free_insn_page(page);
349 }
350 
351 /* For optimized_kprobe buffer */
352 struct kprobe_insn_cache kprobe_optinsn_slots = {
353 	.mutex = __MUTEX_INITIALIZER(kprobe_optinsn_slots.mutex),
354 	.alloc = alloc_optinsn_page,
355 	.free = free_optinsn_page,
356 	.sym = KPROBE_OPTINSN_PAGE_SYM,
357 	.pages = LIST_HEAD_INIT(kprobe_optinsn_slots.pages),
358 	/* .insn_size is initialized later */
359 	.nr_garbage = 0,
360 };
361 #endif /* CONFIG_OPTPROBES */
362 #endif /* __ARCH_WANT_KPROBES_INSN_SLOT */
363 
364 /* We have preemption disabled.. so it is safe to use __ versions */
365 static inline void set_kprobe_instance(struct kprobe *kp)
366 {
367 	__this_cpu_write(kprobe_instance, kp);
368 }
369 
370 static inline void reset_kprobe_instance(void)
371 {
372 	__this_cpu_write(kprobe_instance, NULL);
373 }
374 
375 /*
376  * This routine is called either:
377  *	- under the 'kprobe_mutex' - during kprobe_[un]register().
378  *				OR
379  *	- with preemption disabled - from architecture specific code.
380  */
381 struct kprobe *get_kprobe(void *addr)
382 {
383 	struct hlist_head *head;
384 	struct kprobe *p;
385 
386 	head = &kprobe_table[hash_ptr(addr, KPROBE_HASH_BITS)];
387 	hlist_for_each_entry_rcu(p, head, hlist,
388 				 lockdep_is_held(&kprobe_mutex)) {
389 		if (p->addr == addr)
390 			return p;
391 	}
392 
393 	return NULL;
394 }
395 NOKPROBE_SYMBOL(get_kprobe);
396 
397 static int aggr_pre_handler(struct kprobe *p, struct pt_regs *regs);
398 
399 /* Return true if 'p' is an aggregator */
400 static inline bool kprobe_aggrprobe(struct kprobe *p)
401 {
402 	return p->pre_handler == aggr_pre_handler;
403 }
404 
405 /* Return true if 'p' is unused */
406 static inline bool kprobe_unused(struct kprobe *p)
407 {
408 	return kprobe_aggrprobe(p) && kprobe_disabled(p) &&
409 	       list_empty(&p->list);
410 }
411 
412 /* Keep all fields in the kprobe consistent. */
413 static inline void copy_kprobe(struct kprobe *ap, struct kprobe *p)
414 {
415 	memcpy(&p->opcode, &ap->opcode, sizeof(kprobe_opcode_t));
416 	memcpy(&p->ainsn, &ap->ainsn, sizeof(struct arch_specific_insn));
417 }
418 
419 #ifdef CONFIG_OPTPROBES
420 /* NOTE: This is protected by 'kprobe_mutex'. */
421 static bool kprobes_allow_optimization;
422 
423 /*
424  * Call all 'kprobe::pre_handler' on the list, but ignores its return value.
425  * This must be called from arch-dep optimized caller.
426  */
427 void opt_pre_handler(struct kprobe *p, struct pt_regs *regs)
428 {
429 	struct kprobe *kp;
430 
431 	list_for_each_entry_rcu(kp, &p->list, list) {
432 		if (kp->pre_handler && likely(!kprobe_disabled(kp))) {
433 			set_kprobe_instance(kp);
434 			kp->pre_handler(kp, regs);
435 		}
436 		reset_kprobe_instance();
437 	}
438 }
439 NOKPROBE_SYMBOL(opt_pre_handler);
440 
441 /* Free optimized instructions and optimized_kprobe */
442 static void free_aggr_kprobe(struct kprobe *p)
443 {
444 	struct optimized_kprobe *op;
445 
446 	op = container_of(p, struct optimized_kprobe, kp);
447 	arch_remove_optimized_kprobe(op);
448 	arch_remove_kprobe(p);
449 	kfree(op);
450 }
451 
452 /* Return true if the kprobe is ready for optimization. */
453 static inline int kprobe_optready(struct kprobe *p)
454 {
455 	struct optimized_kprobe *op;
456 
457 	if (kprobe_aggrprobe(p)) {
458 		op = container_of(p, struct optimized_kprobe, kp);
459 		return arch_prepared_optinsn(&op->optinsn);
460 	}
461 
462 	return 0;
463 }
464 
465 /* Return true if the kprobe is disarmed. Note: p must be on hash list */
466 bool kprobe_disarmed(struct kprobe *p)
467 {
468 	struct optimized_kprobe *op;
469 
470 	/* If kprobe is not aggr/opt probe, just return kprobe is disabled */
471 	if (!kprobe_aggrprobe(p))
472 		return kprobe_disabled(p);
473 
474 	op = container_of(p, struct optimized_kprobe, kp);
475 
476 	return kprobe_disabled(p) && list_empty(&op->list);
477 }
478 
479 /* Return true if the probe is queued on (un)optimizing lists */
480 static bool kprobe_queued(struct kprobe *p)
481 {
482 	struct optimized_kprobe *op;
483 
484 	if (kprobe_aggrprobe(p)) {
485 		op = container_of(p, struct optimized_kprobe, kp);
486 		if (!list_empty(&op->list))
487 			return true;
488 	}
489 	return false;
490 }
491 
492 /*
493  * Return an optimized kprobe whose optimizing code replaces
494  * instructions including 'addr' (exclude breakpoint).
495  */
496 static struct kprobe *get_optimized_kprobe(kprobe_opcode_t *addr)
497 {
498 	int i;
499 	struct kprobe *p = NULL;
500 	struct optimized_kprobe *op;
501 
502 	/* Don't check i == 0, since that is a breakpoint case. */
503 	for (i = 1; !p && i < MAX_OPTIMIZED_LENGTH / sizeof(kprobe_opcode_t); i++)
504 		p = get_kprobe(addr - i);
505 
506 	if (p && kprobe_optready(p)) {
507 		op = container_of(p, struct optimized_kprobe, kp);
508 		if (arch_within_optimized_kprobe(op, addr))
509 			return p;
510 	}
511 
512 	return NULL;
513 }
514 
515 /* Optimization staging list, protected by 'kprobe_mutex' */
516 static LIST_HEAD(optimizing_list);
517 static LIST_HEAD(unoptimizing_list);
518 static LIST_HEAD(freeing_list);
519 
520 static void optimize_kprobe(struct kprobe *p);
521 static struct task_struct *kprobe_optimizer_task;
522 static wait_queue_head_t kprobe_optimizer_wait;
523 static atomic_t optimizer_state;
524 enum {
525 	OPTIMIZER_ST_IDLE = 0,
526 	OPTIMIZER_ST_KICKED = 1,
527 	OPTIMIZER_ST_FLUSHING = 2,
528 };
529 
530 /* Bumped at the end of each kprobe_optimizer() pass, under 'kprobe_mutex' */
531 static unsigned long optimizer_passes;
532 
533 #define OPTIMIZE_DELAY 5
534 
535 /*
536  * Optimize (replace a breakpoint with a jump) kprobes listed on
537  * 'optimizing_list'.
538  */
539 static void do_optimize_kprobes(void)
540 {
541 	lockdep_assert_held(&text_mutex);
542 	/*
543 	 * The optimization/unoptimization refers 'online_cpus' via
544 	 * stop_machine() and cpu-hotplug modifies the 'online_cpus'.
545 	 * And same time, 'text_mutex' will be held in cpu-hotplug and here.
546 	 * This combination can cause a deadlock (cpu-hotplug tries to lock
547 	 * 'text_mutex' but stop_machine() can not be done because
548 	 * the 'online_cpus' has been changed)
549 	 * To avoid this deadlock, caller must have locked cpu-hotplug
550 	 * for preventing cpu-hotplug outside of 'text_mutex' locking.
551 	 */
552 	lockdep_assert_cpus_held();
553 
554 	/* Optimization never be done when disarmed */
555 	if (kprobes_all_disarmed || !kprobes_allow_optimization ||
556 	    list_empty(&optimizing_list))
557 		return;
558 
559 	arch_optimize_kprobes(&optimizing_list);
560 }
561 
562 /*
563  * Unoptimize (replace a jump with a breakpoint and remove the breakpoint
564  * if need) kprobes listed on 'unoptimizing_list'.
565  */
566 static void do_unoptimize_kprobes(void)
567 {
568 	struct optimized_kprobe *op, *tmp;
569 
570 	lockdep_assert_held(&text_mutex);
571 	/* See comment in do_optimize_kprobes() */
572 	lockdep_assert_cpus_held();
573 
574 	if (!list_empty(&unoptimizing_list))
575 		arch_unoptimize_kprobes(&unoptimizing_list, &freeing_list);
576 
577 	/* Loop on 'freeing_list' for disarming and removing from kprobe hash list */
578 	list_for_each_entry_safe(op, tmp, &freeing_list, list) {
579 		/* Switching from detour code to origin */
580 		op->kp.flags &= ~KPROBE_FLAG_OPTIMIZED;
581 		/* Disarm probes if marked disabled and not gone */
582 		if (kprobe_disabled(&op->kp) && !kprobe_gone(&op->kp))
583 			arch_disarm_kprobe(&op->kp);
584 		if (kprobe_unused(&op->kp)) {
585 			/*
586 			 * Remove unused probes from hash list. After waiting
587 			 * for synchronization, these probes are reclaimed.
588 			 * (reclaiming is done by do_free_cleaned_kprobes().)
589 			 */
590 			hlist_del_rcu(&op->kp.hlist);
591 		} else
592 			list_del_init(&op->list);
593 	}
594 }
595 
596 /* Reclaim all kprobes on the 'freeing_list' */
597 static void do_free_cleaned_kprobes(void)
598 {
599 	struct optimized_kprobe *op, *tmp;
600 
601 	list_for_each_entry_safe(op, tmp, &freeing_list, list) {
602 		list_del_init(&op->list);
603 		if (WARN_ON_ONCE(!kprobe_unused(&op->kp))) {
604 			/*
605 			 * This must not happen, but if there is a kprobe
606 			 * still in use, keep it on kprobes hash list.
607 			 */
608 			continue;
609 		}
610 
611 		/*
612 		 * The aggregator was holding back another probe while it sat on the
613 		 * unoptimizing/freeing lists.  Now that the aggregator has been fully
614 		 * reverted we can safely retry the optimization of that sibling.
615 		 */
616 
617 		struct kprobe *_p = get_optimized_kprobe(op->kp.addr);
618 		if (unlikely(_p))
619 			optimize_kprobe(_p);
620 
621 		free_aggr_kprobe(&op->kp);
622 	}
623 }
624 
625 static void kick_kprobe_optimizer(void);
626 
627 /* Kprobe jump optimizer */
628 static void kprobe_optimizer(void)
629 {
630 	guard(mutex)(&kprobe_mutex);
631 
632 	scoped_guard(cpus_read_lock) {
633 		guard(mutex)(&text_mutex);
634 
635 		/*
636 		 * Step 1: Unoptimize kprobes and collect cleaned (unused and disarmed)
637 		 * kprobes before waiting for quiesence period.
638 		 */
639 		do_unoptimize_kprobes();
640 
641 		/*
642 		 * Step 2: Wait for quiesence period to ensure all potentially
643 		 * preempted tasks to have normally scheduled. Because optprobe
644 		 * may modify multiple instructions, there is a chance that Nth
645 		 * instruction is preempted. In that case, such tasks can return
646 		 * to 2nd-Nth byte of jump instruction. This wait is for avoiding it.
647 		 * Note that on non-preemptive kernel, this is transparently converted
648 		 * to synchronoze_sched() to wait for all interrupts to have completed.
649 		 */
650 		synchronize_rcu_tasks();
651 
652 		/* Step 3: Optimize kprobes after quiesence period */
653 		do_optimize_kprobes();
654 
655 		/* Step 4: Free cleaned kprobes after quiesence period */
656 		do_free_cleaned_kprobes();
657 	}
658 
659 	/* Step 5: Wake up flushers, and kick optimizer again if needed. */
660 	optimizer_passes++;
661 	wake_up_var_locked(&optimizer_passes, &kprobe_mutex);
662 
663 	if (!list_empty(&optimizing_list) || !list_empty(&unoptimizing_list))
664 		kick_kprobe_optimizer();	/*normal kick*/
665 }
666 
667 static int kprobe_optimizer_thread(void *data)
668 {
669 	while (!kthread_should_stop()) {
670 		/* To avoid hung_task, wait in interruptible state. */
671 		wait_event_interruptible(kprobe_optimizer_wait,
672 			   atomic_read(&optimizer_state) != OPTIMIZER_ST_IDLE ||
673 			   kthread_should_stop());
674 
675 		if (kthread_should_stop())
676 			break;
677 
678 		/*
679 		 * If it was a normal kick, wait for OPTIMIZE_DELAY.
680 		 * This wait can be interrupted by a flush request.
681 		 */
682 		if (atomic_read(&optimizer_state) == 1)
683 			wait_event_interruptible_timeout(
684 				kprobe_optimizer_wait,
685 				atomic_read(&optimizer_state) == OPTIMIZER_ST_FLUSHING ||
686 				kthread_should_stop(),
687 				OPTIMIZE_DELAY);
688 
689 		if (kthread_should_stop())
690 			break;
691 
692 		atomic_set(&optimizer_state, OPTIMIZER_ST_IDLE);
693 
694 		kprobe_optimizer();
695 	}
696 	return 0;
697 }
698 
699 /* Start optimizer after OPTIMIZE_DELAY passed */
700 static void kick_kprobe_optimizer(void)
701 {
702 	lockdep_assert_held(&kprobe_mutex);
703 	if (atomic_cmpxchg(&optimizer_state,
704 		OPTIMIZER_ST_IDLE, OPTIMIZER_ST_KICKED) == OPTIMIZER_ST_IDLE)
705 		wake_up(&kprobe_optimizer_wait);
706 }
707 
708 static void wait_for_kprobe_optimizer_locked(void)
709 {
710 	lockdep_assert_held(&kprobe_mutex);
711 
712 	while (!list_empty(&optimizing_list) || !list_empty(&unoptimizing_list)) {
713 		unsigned long passes = optimizer_passes;
714 
715 		/*
716 		 * Set state to OPTIMIZER_ST_FLUSHING and wake up the thread if it's
717 		 * idle. If it's already kicked, it will see the state change.
718 		 */
719 		if (atomic_xchg_acquire(&optimizer_state,
720 			OPTIMIZER_ST_FLUSHING) != OPTIMIZER_ST_FLUSHING)
721 			wake_up(&kprobe_optimizer_wait);
722 
723 		/*
724 		 * kprobe_optimizer() holds 'kprobe_mutex' for a whole pass, which
725 		 * this drops while sleeping, so a new count means a full pass ran.
726 		 */
727 		wait_var_event_mutex(&optimizer_passes,
728 				     optimizer_passes != passes, &kprobe_mutex);
729 	}
730 }
731 
732 /* Wait for completing optimization and unoptimization */
733 void wait_for_kprobe_optimizer(void)
734 {
735 	guard(mutex)(&kprobe_mutex);
736 
737 	wait_for_kprobe_optimizer_locked();
738 }
739 
740 bool optprobe_queued_unopt(struct optimized_kprobe *op)
741 {
742 	struct optimized_kprobe *_op;
743 
744 	list_for_each_entry(_op, &unoptimizing_list, list) {
745 		if (op == _op)
746 			return true;
747 	}
748 
749 	return false;
750 }
751 
752 /* Optimize kprobe if p is ready to be optimized */
753 static void optimize_kprobe(struct kprobe *p)
754 {
755 	struct optimized_kprobe *op;
756 
757 	/* Check if the kprobe is disabled or not ready for optimization. */
758 	if (!kprobe_optready(p) || !kprobes_allow_optimization ||
759 	    (kprobe_disabled(p) || kprobes_all_disarmed))
760 		return;
761 
762 	/* kprobes with 'post_handler' can not be optimized */
763 	if (p->post_handler)
764 		return;
765 
766 	op = container_of(p, struct optimized_kprobe, kp);
767 
768 	/* Check there is no other kprobes at the optimized instructions */
769 	if (arch_check_optimized_kprobe(op) < 0)
770 		return;
771 
772 	/* Check if it is already optimized. */
773 	if (op->kp.flags & KPROBE_FLAG_OPTIMIZED) {
774 		if (optprobe_queued_unopt(op)) {
775 			/* This is under unoptimizing. Just dequeue the probe */
776 			list_del_init(&op->list);
777 		}
778 		return;
779 	}
780 	op->kp.flags |= KPROBE_FLAG_OPTIMIZED;
781 
782 	/*
783 	 * On the 'unoptimizing_list' and 'optimizing_list',
784 	 * 'op' must have OPTIMIZED flag
785 	 */
786 	if (WARN_ON_ONCE(!list_empty(&op->list)))
787 		return;
788 
789 	list_add(&op->list, &optimizing_list);
790 	kick_kprobe_optimizer();
791 }
792 
793 /* Short cut to direct unoptimizing */
794 static void force_unoptimize_kprobe(struct optimized_kprobe *op)
795 {
796 	lockdep_assert_cpus_held();
797 	arch_unoptimize_kprobe(op);
798 	op->kp.flags &= ~KPROBE_FLAG_OPTIMIZED;
799 }
800 
801 /* Unoptimize a kprobe if p is optimized */
802 static void unoptimize_kprobe(struct kprobe *p, bool force)
803 {
804 	struct optimized_kprobe *op;
805 
806 	if (!kprobe_aggrprobe(p) || kprobe_disarmed(p))
807 		return; /* This is not an optprobe nor optimized */
808 
809 	op = container_of(p, struct optimized_kprobe, kp);
810 	if (!kprobe_optimized(p))
811 		return;
812 
813 	if (!list_empty(&op->list)) {
814 		if (optprobe_queued_unopt(op)) {
815 			/* Queued in unoptimizing queue */
816 			if (force) {
817 				/*
818 				 * Forcibly unoptimize the kprobe here, and queue it
819 				 * in the freeing list for release afterwards.
820 				 */
821 				force_unoptimize_kprobe(op);
822 				list_move(&op->list, &freeing_list);
823 			}
824 		} else {
825 			/* Dequeue from the optimizing queue */
826 			list_del_init(&op->list);
827 			op->kp.flags &= ~KPROBE_FLAG_OPTIMIZED;
828 		}
829 		return;
830 	}
831 
832 	/* Optimized kprobe case */
833 	if (force) {
834 		/* Forcibly update the code: this is a special case */
835 		force_unoptimize_kprobe(op);
836 	} else {
837 		list_add(&op->list, &unoptimizing_list);
838 		kick_kprobe_optimizer();
839 	}
840 }
841 
842 /* Cancel unoptimizing for reusing */
843 static int reuse_unused_kprobe(struct kprobe *ap)
844 {
845 	struct optimized_kprobe *op;
846 
847 	/*
848 	 * Unused kprobe MUST be on the way of delayed unoptimizing (means
849 	 * there is still a relative jump) and disabled.
850 	 */
851 	op = container_of(ap, struct optimized_kprobe, kp);
852 	WARN_ON_ONCE(list_empty(&op->list));
853 	/* Enable the probe again */
854 	ap->flags &= ~KPROBE_FLAG_DISABLED;
855 	/* Optimize it again. (remove from 'op->list') */
856 	if (!kprobe_optready(ap))
857 		return -EINVAL;
858 
859 	optimize_kprobe(ap);
860 	return 0;
861 }
862 
863 /* Remove optimized instructions */
864 static void kill_optimized_kprobe(struct kprobe *p)
865 {
866 	struct optimized_kprobe *op;
867 
868 	op = container_of(p, struct optimized_kprobe, kp);
869 	if (!list_empty(&op->list))
870 		/* Dequeue from the (un)optimization queue */
871 		list_del_init(&op->list);
872 	op->kp.flags &= ~KPROBE_FLAG_OPTIMIZED;
873 
874 	if (kprobe_unused(p)) {
875 		/*
876 		 * Unused kprobe is on unoptimizing or freeing list. We move it
877 		 * to freeing_list and let the kprobe_optimizer() remove it from
878 		 * the kprobe hash list and free it.
879 		 */
880 		if (optprobe_queued_unopt(op))
881 			list_move(&op->list, &freeing_list);
882 	}
883 
884 	/* Don't touch the code, because it is already freed. */
885 	arch_remove_optimized_kprobe(op);
886 }
887 
888 static inline
889 void __prepare_optimized_kprobe(struct optimized_kprobe *op, struct kprobe *p)
890 {
891 	if (!kprobe_ftrace(p))
892 		arch_prepare_optimized_kprobe(op, p);
893 }
894 
895 /* Try to prepare optimized instructions */
896 static void prepare_optimized_kprobe(struct kprobe *p)
897 {
898 	struct optimized_kprobe *op;
899 
900 	op = container_of(p, struct optimized_kprobe, kp);
901 	__prepare_optimized_kprobe(op, p);
902 }
903 
904 /* Allocate new optimized_kprobe and try to prepare optimized instructions. */
905 static struct kprobe *alloc_aggr_kprobe(struct kprobe *p)
906 {
907 	struct optimized_kprobe *op;
908 
909 	op = kzalloc_obj(struct optimized_kprobe);
910 	if (!op)
911 		return NULL;
912 
913 	INIT_LIST_HEAD(&op->list);
914 	op->kp.addr = p->addr;
915 	__prepare_optimized_kprobe(op, p);
916 
917 	return &op->kp;
918 }
919 
920 static void init_aggr_kprobe(struct kprobe *ap, struct kprobe *p);
921 
922 /*
923  * Prepare an optimized_kprobe and optimize it.
924  * NOTE: 'p' must be a normal registered kprobe.
925  */
926 static void try_to_optimize_kprobe(struct kprobe *p)
927 {
928 	struct kprobe *ap;
929 	struct optimized_kprobe *op;
930 
931 	/* Impossible to optimize ftrace-based kprobe. */
932 	if (kprobe_ftrace(p))
933 		return;
934 
935 	/* For preparing optimization, jump_label_text_reserved() is called. */
936 	guard(cpus_read_lock)();
937 	guard(jump_label_lock)();
938 	guard(mutex)(&text_mutex);
939 
940 	ap = alloc_aggr_kprobe(p);
941 	if (!ap)
942 		return;
943 
944 	op = container_of(ap, struct optimized_kprobe, kp);
945 	if (!arch_prepared_optinsn(&op->optinsn)) {
946 		/* If failed to setup optimizing, fallback to kprobe. */
947 		arch_remove_optimized_kprobe(op);
948 		kfree(op);
949 		return;
950 	}
951 
952 	init_aggr_kprobe(ap, p);
953 	optimize_kprobe(ap);	/* This just kicks optimizer thread. */
954 }
955 
956 static void optimize_all_kprobes(void)
957 {
958 	struct hlist_head *head;
959 	struct kprobe *p;
960 	unsigned int i;
961 
962 	guard(mutex)(&kprobe_mutex);
963 	/* If optimization is already allowed, just return. */
964 	if (kprobes_allow_optimization)
965 		return;
966 
967 	cpus_read_lock();
968 	kprobes_allow_optimization = true;
969 	for (i = 0; i < KPROBE_TABLE_SIZE; i++) {
970 		head = &kprobe_table[i];
971 		hlist_for_each_entry(p, head, hlist)
972 			if (!kprobe_disabled(p))
973 				optimize_kprobe(p);
974 	}
975 	cpus_read_unlock();
976 	pr_info("kprobe jump-optimization is enabled. All kprobes are optimized if possible.\n");
977 }
978 
979 #ifdef CONFIG_SYSCTL
980 static void unoptimize_all_kprobes(void)
981 {
982 	struct hlist_head *head;
983 	struct kprobe *p;
984 	unsigned int i;
985 
986 	guard(mutex)(&kprobe_mutex);
987 	/* If optimization is already prohibited, just return. */
988 	if (!kprobes_allow_optimization)
989 		return;
990 
991 	cpus_read_lock();
992 	kprobes_allow_optimization = false;
993 	for (i = 0; i < KPROBE_TABLE_SIZE; i++) {
994 		head = &kprobe_table[i];
995 		hlist_for_each_entry(p, head, hlist) {
996 			if (!kprobe_disabled(p))
997 				unoptimize_kprobe(p, false);
998 		}
999 	}
1000 	cpus_read_unlock();
1001 	/* Wait for unoptimizing completion. */
1002 	wait_for_kprobe_optimizer_locked();
1003 	pr_info("kprobe jump-optimization is disabled. All kprobes are based on software breakpoint.\n");
1004 }
1005 
1006 static DEFINE_MUTEX(kprobe_sysctl_mutex);
1007 static int sysctl_kprobes_optimization;
1008 static int proc_kprobes_optimization_handler(const struct ctl_table *table,
1009 					     int write, void *buffer,
1010 					     size_t *length, loff_t *ppos)
1011 {
1012 	int ret;
1013 
1014 	guard(mutex)(&kprobe_sysctl_mutex);
1015 	sysctl_kprobes_optimization = kprobes_allow_optimization ? 1 : 0;
1016 	ret = proc_dointvec_minmax(table, write, buffer, length, ppos);
1017 
1018 	if (sysctl_kprobes_optimization)
1019 		optimize_all_kprobes();
1020 	else
1021 		unoptimize_all_kprobes();
1022 
1023 	return ret;
1024 }
1025 
1026 static const struct ctl_table kprobe_sysctls[] = {
1027 	{
1028 		.procname	= "kprobes-optimization",
1029 		.data		= &sysctl_kprobes_optimization,
1030 		.maxlen		= sizeof(int),
1031 		.mode		= 0644,
1032 		.proc_handler	= proc_kprobes_optimization_handler,
1033 		.extra1		= SYSCTL_ZERO,
1034 		.extra2		= SYSCTL_ONE,
1035 	},
1036 };
1037 
1038 static void __init kprobe_sysctls_init(void)
1039 {
1040 	register_sysctl_init("debug", kprobe_sysctls);
1041 }
1042 #endif /* CONFIG_SYSCTL */
1043 
1044 /* Put a breakpoint for a probe. */
1045 static void __arm_kprobe(struct kprobe *p)
1046 {
1047 	struct kprobe *_p;
1048 
1049 	lockdep_assert_held(&text_mutex);
1050 
1051 	/* Find the overlapping optimized kprobes. */
1052 	_p = get_optimized_kprobe(p->addr);
1053 	if (unlikely(_p))
1054 		/* Fallback to unoptimized kprobe */
1055 		unoptimize_kprobe(_p, true);
1056 
1057 	arch_arm_kprobe(p);
1058 	optimize_kprobe(p);	/* Try to optimize (add kprobe to a list) */
1059 }
1060 
1061 /* Remove the breakpoint of a probe. */
1062 static void __disarm_kprobe(struct kprobe *p, bool reopt)
1063 {
1064 	struct kprobe *_p;
1065 
1066 	lockdep_assert_held(&text_mutex);
1067 
1068 	/* Try to unoptimize */
1069 	unoptimize_kprobe(p, kprobes_all_disarmed);
1070 
1071 	if (!kprobe_queued(p)) {
1072 		arch_disarm_kprobe(p);
1073 		/* If another kprobe was blocked, re-optimize it. */
1074 		_p = get_optimized_kprobe(p->addr);
1075 		if (unlikely(_p) && reopt)
1076 			optimize_kprobe(_p);
1077 	}
1078 }
1079 
1080 static void __init init_optprobe(void)
1081 {
1082 #ifdef __ARCH_WANT_KPROBES_INSN_SLOT
1083 	/* Init 'kprobe_optinsn_slots' for allocation */
1084 	kprobe_optinsn_slots.insn_size = MAX_OPTINSN_SIZE;
1085 #endif
1086 
1087 	init_waitqueue_head(&kprobe_optimizer_wait);
1088 	atomic_set(&optimizer_state, OPTIMIZER_ST_IDLE);
1089 	kprobe_optimizer_task = kthread_run(kprobe_optimizer_thread, NULL,
1090 					    "kprobe-optimizer");
1091 }
1092 #else /* !CONFIG_OPTPROBES */
1093 
1094 #define init_optprobe()				do {} while (0)
1095 #define optimize_kprobe(p)			do {} while (0)
1096 #define unoptimize_kprobe(p, f)			do {} while (0)
1097 #define kill_optimized_kprobe(p)		do {} while (0)
1098 #define prepare_optimized_kprobe(p)		do {} while (0)
1099 #define try_to_optimize_kprobe(p)		do {} while (0)
1100 #define __arm_kprobe(p)				arch_arm_kprobe(p)
1101 #define __disarm_kprobe(p, o)			arch_disarm_kprobe(p)
1102 #define kprobe_disarmed(p)			kprobe_disabled(p)
1103 #define wait_for_kprobe_optimizer_locked()			\
1104 	lockdep_assert_held(&kprobe_mutex)
1105 
1106 static int reuse_unused_kprobe(struct kprobe *ap)
1107 {
1108 	/*
1109 	 * If the optimized kprobe is NOT supported, the aggr kprobe is
1110 	 * released at the same time that the last aggregated kprobe is
1111 	 * unregistered.
1112 	 * Thus there should be no chance to reuse unused kprobe.
1113 	 */
1114 	WARN_ON_ONCE(1);
1115 	return -EINVAL;
1116 }
1117 
1118 static void free_aggr_kprobe(struct kprobe *p)
1119 {
1120 	arch_remove_kprobe(p);
1121 	kfree(p);
1122 }
1123 
1124 static struct kprobe *alloc_aggr_kprobe(struct kprobe *p)
1125 {
1126 	return kzalloc_obj(struct kprobe);
1127 }
1128 #endif /* CONFIG_OPTPROBES */
1129 
1130 #ifdef CONFIG_KPROBES_ON_FTRACE
1131 static struct ftrace_ops kprobe_ftrace_ops __read_mostly = {
1132 	.func = kprobe_ftrace_handler,
1133 	.flags = FTRACE_OPS_FL_SAVE_REGS,
1134 };
1135 
1136 static struct ftrace_ops kprobe_ipmodify_ops __read_mostly = {
1137 	.func = kprobe_ftrace_handler,
1138 	.flags = FTRACE_OPS_FL_SAVE_REGS | FTRACE_OPS_FL_IPMODIFY,
1139 };
1140 
1141 static int kprobe_ipmodify_enabled;
1142 static int kprobe_ftrace_enabled;
1143 bool kprobe_ftrace_disabled;
1144 
1145 static int __arm_kprobe_ftrace(struct kprobe *p, struct ftrace_ops *ops,
1146 			       int *cnt)
1147 {
1148 	int ret;
1149 
1150 	lockdep_assert_held(&kprobe_mutex);
1151 
1152 	ret = ftrace_set_filter_ip(ops, (unsigned long)p->addr, 0, 0);
1153 	if (ret < 0)
1154 		return ret;
1155 
1156 	if (*cnt == 0) {
1157 		ret = register_ftrace_function(ops);
1158 		if (ret < 0) {
1159 			/*
1160 			 * At this point, sinec ops is not registered, we should be sefe from
1161 			 * registering empty filter.
1162 			 */
1163 			ftrace_set_filter_ip(ops, (unsigned long)p->addr, 1, 0);
1164 			return ret;
1165 		}
1166 	}
1167 
1168 	(*cnt)++;
1169 	return ret;
1170 }
1171 
1172 static int arm_kprobe_ftrace(struct kprobe *p)
1173 {
1174 	bool ipmodify = (p->post_handler != NULL);
1175 
1176 	return __arm_kprobe_ftrace(p,
1177 		ipmodify ? &kprobe_ipmodify_ops : &kprobe_ftrace_ops,
1178 		ipmodify ? &kprobe_ipmodify_enabled : &kprobe_ftrace_enabled);
1179 }
1180 
1181 static int __disarm_kprobe_ftrace(struct kprobe *p, struct ftrace_ops *ops,
1182 				  int *cnt)
1183 {
1184 	int ret;
1185 
1186 	lockdep_assert_held(&kprobe_mutex);
1187 	if (unlikely(kprobe_ftrace_disabled)) {
1188 		/* Now ftrace is disabled forever, disarm is already done. */
1189 		return 0;
1190 	}
1191 
1192 	if (*cnt == 1) {
1193 		ret = unregister_ftrace_function(ops);
1194 		if (WARN(ret < 0, "Failed to unregister kprobe-ftrace (error %d)\n", ret))
1195 			return ret;
1196 	}
1197 
1198 	(*cnt)--;
1199 
1200 	ret = ftrace_set_filter_ip(ops, (unsigned long)p->addr, 1, 0);
1201 	WARN_ONCE(ret < 0, "Failed to disarm kprobe-ftrace at %pS (error %d)\n",
1202 		  p->addr, ret);
1203 	return ret;
1204 }
1205 
1206 static int disarm_kprobe_ftrace(struct kprobe *p)
1207 {
1208 	bool ipmodify = (p->post_handler != NULL);
1209 
1210 	return __disarm_kprobe_ftrace(p,
1211 		ipmodify ? &kprobe_ipmodify_ops : &kprobe_ftrace_ops,
1212 		ipmodify ? &kprobe_ipmodify_enabled : &kprobe_ftrace_enabled);
1213 }
1214 
1215 void kprobe_ftrace_kill(void)
1216 {
1217 	kprobe_ftrace_disabled = true;
1218 }
1219 #else	/* !CONFIG_KPROBES_ON_FTRACE */
1220 static inline int arm_kprobe_ftrace(struct kprobe *p)
1221 {
1222 	return -ENODEV;
1223 }
1224 
1225 static inline int disarm_kprobe_ftrace(struct kprobe *p)
1226 {
1227 	return -ENODEV;
1228 }
1229 #endif
1230 
1231 static int prepare_kprobe(struct kprobe *p)
1232 {
1233 	/* Must ensure p->addr is really on ftrace */
1234 	if (kprobe_ftrace(p))
1235 		return arch_prepare_kprobe_ftrace(p);
1236 
1237 	return arch_prepare_kprobe(p);
1238 }
1239 
1240 static int arm_kprobe(struct kprobe *kp)
1241 {
1242 	if (unlikely(kprobe_ftrace(kp)))
1243 		return arm_kprobe_ftrace(kp);
1244 
1245 	guard(cpus_read_lock)();
1246 	guard(mutex)(&text_mutex);
1247 	__arm_kprobe(kp);
1248 	return 0;
1249 }
1250 
1251 static int disarm_kprobe(struct kprobe *kp, bool reopt)
1252 {
1253 	if (unlikely(kprobe_ftrace(kp)))
1254 		return disarm_kprobe_ftrace(kp);
1255 
1256 	guard(cpus_read_lock)();
1257 	guard(mutex)(&text_mutex);
1258 	__disarm_kprobe(kp, reopt);
1259 	return 0;
1260 }
1261 
1262 /*
1263  * Aggregate handlers for multiple kprobes support - these handlers
1264  * take care of invoking the individual kprobe handlers on p->list
1265  */
1266 static int aggr_pre_handler(struct kprobe *p, struct pt_regs *regs)
1267 {
1268 	struct kprobe *kp;
1269 
1270 	list_for_each_entry_rcu(kp, &p->list, list) {
1271 		if (kp->pre_handler && likely(!kprobe_disabled(kp))) {
1272 			set_kprobe_instance(kp);
1273 			if (kp->pre_handler(kp, regs))
1274 				return 1;
1275 		}
1276 		reset_kprobe_instance();
1277 	}
1278 	return 0;
1279 }
1280 NOKPROBE_SYMBOL(aggr_pre_handler);
1281 
1282 static void aggr_post_handler(struct kprobe *p, struct pt_regs *regs,
1283 			      unsigned long flags)
1284 {
1285 	struct kprobe *kp;
1286 
1287 	list_for_each_entry_rcu(kp, &p->list, list) {
1288 		if (kp->post_handler && likely(!kprobe_disabled(kp))) {
1289 			set_kprobe_instance(kp);
1290 			kp->post_handler(kp, regs, flags);
1291 			reset_kprobe_instance();
1292 		}
1293 	}
1294 }
1295 NOKPROBE_SYMBOL(aggr_post_handler);
1296 
1297 /* Walks the list and increments 'nmissed' if 'p' has child probes. */
1298 void kprobes_inc_nmissed_count(struct kprobe *p)
1299 {
1300 	struct kprobe *kp;
1301 
1302 	if (!kprobe_aggrprobe(p)) {
1303 		p->nmissed++;
1304 	} else {
1305 		list_for_each_entry_rcu(kp, &p->list, list)
1306 			kp->nmissed++;
1307 	}
1308 }
1309 NOKPROBE_SYMBOL(kprobes_inc_nmissed_count);
1310 
1311 static struct kprobe kprobe_busy = {
1312 	.addr = (void *) get_kprobe,
1313 };
1314 
1315 void kprobe_busy_begin(void)
1316 {
1317 	struct kprobe_ctlblk *kcb;
1318 
1319 	preempt_disable();
1320 	__this_cpu_write(current_kprobe, &kprobe_busy);
1321 	kcb = get_kprobe_ctlblk();
1322 	kcb->kprobe_status = KPROBE_HIT_ACTIVE;
1323 }
1324 
1325 void kprobe_busy_end(void)
1326 {
1327 	__this_cpu_write(current_kprobe, NULL);
1328 	preempt_enable();
1329 }
1330 
1331 /* Add the new probe to 'ap->list'. */
1332 static int add_new_kprobe(struct kprobe *ap, struct kprobe *p)
1333 {
1334 	if (p->post_handler)
1335 		unoptimize_kprobe(ap, true);	/* Fall back to normal kprobe */
1336 
1337 	list_add_rcu(&p->list, &ap->list);
1338 	if (p->post_handler && !ap->post_handler)
1339 		ap->post_handler = aggr_post_handler;
1340 
1341 	return 0;
1342 }
1343 
1344 /*
1345  * Fill in the required fields of the aggregator kprobe. Replace the
1346  * earlier kprobe in the hlist with the aggregator kprobe.
1347  */
1348 static void init_aggr_kprobe(struct kprobe *ap, struct kprobe *p)
1349 {
1350 	/* Copy the insn slot of 'p' to 'ap'. */
1351 	copy_kprobe(p, ap);
1352 	flush_insn_slot(ap);
1353 	ap->addr = p->addr;
1354 	ap->flags = p->flags & ~KPROBE_FLAG_OPTIMIZED;
1355 	ap->pre_handler = aggr_pre_handler;
1356 	/* We don't care the kprobe which has gone. */
1357 	if (p->post_handler && !kprobe_gone(p))
1358 		ap->post_handler = aggr_post_handler;
1359 
1360 	INIT_LIST_HEAD(&ap->list);
1361 	INIT_HLIST_NODE(&ap->hlist);
1362 
1363 	list_add_rcu(&p->list, &ap->list);
1364 	hlist_replace_rcu(&p->hlist, &ap->hlist);
1365 }
1366 
1367 /*
1368  * This registers the second or subsequent kprobe at the same address.
1369  */
1370 static int register_aggr_kprobe(struct kprobe *orig_p, struct kprobe *p)
1371 {
1372 	int ret = 0;
1373 	struct kprobe *ap = orig_p;
1374 
1375 	scoped_guard(cpus_read_lock) {
1376 		/* For preparing optimization, jump_label_text_reserved() is called */
1377 		guard(jump_label_lock)();
1378 		guard(mutex)(&text_mutex);
1379 
1380 		if (!kprobe_aggrprobe(orig_p)) {
1381 			/* If 'orig_p' is not an 'aggr_kprobe', create new one. */
1382 			ap = alloc_aggr_kprobe(orig_p);
1383 			if (!ap)
1384 				return -ENOMEM;
1385 			init_aggr_kprobe(ap, orig_p);
1386 		} else if (kprobe_unused(ap)) {
1387 			/* This probe is going to die. Rescue it */
1388 			ret = reuse_unused_kprobe(ap);
1389 			if (ret)
1390 				return ret;
1391 		}
1392 
1393 		if (kprobe_gone(ap)) {
1394 			/*
1395 			 * Attempting to insert new probe at the same location that
1396 			 * had a probe in the module vaddr area which already
1397 			 * freed. So, the instruction slot has already been
1398 			 * released. We need a new slot for the new probe.
1399 			 */
1400 			ret = arch_prepare_kprobe(ap);
1401 			if (ret)
1402 				/*
1403 				 * Even if fail to allocate new slot, don't need to
1404 				 * free the 'ap'. It will be used next time, or
1405 				 * freed by unregister_kprobe().
1406 				 */
1407 				return ret;
1408 
1409 			/* Prepare optimized instructions if possible. */
1410 			prepare_optimized_kprobe(ap);
1411 
1412 			/*
1413 			 * Clear gone flag to prevent allocating new slot again, and
1414 			 * set disabled flag because it is not armed yet.
1415 			 */
1416 			ap->flags = (ap->flags & ~KPROBE_FLAG_GONE)
1417 					| KPROBE_FLAG_DISABLED;
1418 		}
1419 
1420 		/* Copy the insn slot of 'p' to 'ap'. */
1421 		copy_kprobe(ap, p);
1422 		ret = add_new_kprobe(ap, p);
1423 	}
1424 
1425 	if (ret == 0 && kprobe_disabled(ap) && !kprobe_disabled(p)) {
1426 		ap->flags &= ~KPROBE_FLAG_DISABLED;
1427 		if (!kprobes_all_disarmed) {
1428 			/* Arm the breakpoint again. */
1429 			ret = arm_kprobe(ap);
1430 			if (ret) {
1431 				ap->flags |= KPROBE_FLAG_DISABLED;
1432 				list_del_rcu(&p->list);
1433 				synchronize_rcu();
1434 			}
1435 		}
1436 	}
1437 	return ret;
1438 }
1439 
1440 bool __weak arch_within_kprobe_blacklist(unsigned long addr)
1441 {
1442 	/* The '__kprobes' functions and entry code must not be probed. */
1443 	return addr >= (unsigned long)__kprobes_text_start &&
1444 	       addr < (unsigned long)__kprobes_text_end;
1445 }
1446 
1447 static bool __within_kprobe_blacklist(unsigned long addr)
1448 {
1449 	struct kprobe_blacklist_entry *ent;
1450 
1451 	if (arch_within_kprobe_blacklist(addr))
1452 		return true;
1453 	/*
1454 	 * If 'kprobe_blacklist' is defined, check the address and
1455 	 * reject any probe registration in the prohibited area.
1456 	 * Note: this can return true during transition period where
1457 	 * (start_addr, end_addr) in the black list is shrinking
1458 	 * but old entry has not been removed yet. This is acceptable
1459 	 * because the worst case is that we reject more probes than
1460 	 * we should.
1461 	 */
1462 	guard(rcu)();
1463 	list_for_each_entry_rcu(ent, &kprobe_blacklist, list) {
1464 		if (addr >= ent->start_addr && addr < ent->end_addr)
1465 			return true;
1466 	}
1467 	return false;
1468 }
1469 
1470 bool within_kprobe_blacklist(unsigned long addr)
1471 {
1472 	char symname[KSYM_NAME_LEN], *p;
1473 
1474 	if (__within_kprobe_blacklist(addr))
1475 		return true;
1476 
1477 	/* Check if the address is on a suffixed-symbol */
1478 	if (!lookup_symbol_name(addr, symname)) {
1479 		p = strchr(symname, '.');
1480 		if (!p)
1481 			return false;
1482 		*p = '\0';
1483 		addr = (unsigned long)kprobe_lookup_name(symname, 0);
1484 		if (addr)
1485 			return __within_kprobe_blacklist(addr);
1486 	}
1487 	return false;
1488 }
1489 
1490 /*
1491  * arch_adjust_kprobe_addr - adjust the address
1492  * @addr: symbol base address
1493  * @offset: offset within the symbol
1494  * @on_func_entry: was this @addr+@offset on the function entry
1495  *
1496  * Typically returns @addr + @offset, except for special cases where the
1497  * function might be prefixed by a CFI landing pad, in that case any offset
1498  * inside the landing pad is mapped to the first 'real' instruction of the
1499  * symbol.
1500  *
1501  * Specifically, for things like IBT/BTI, skip the resp. ENDBR/BTI.C
1502  * instruction at +0.
1503  */
1504 kprobe_opcode_t *__weak arch_adjust_kprobe_addr(unsigned long addr,
1505 						unsigned long offset,
1506 						bool *on_func_entry)
1507 {
1508 	*on_func_entry = !offset;
1509 	return (kprobe_opcode_t *)(addr + offset);
1510 }
1511 
1512 /*
1513  * If 'symbol_name' is specified, look it up and add the 'offset'
1514  * to it. This way, we can specify a relative address to a symbol.
1515  * This returns encoded errors if it fails to look up symbol or invalid
1516  * combination of parameters.
1517  */
1518 static kprobe_opcode_t *
1519 _kprobe_addr(kprobe_opcode_t *addr, const char *symbol_name,
1520 	     unsigned long offset, bool *on_func_entry)
1521 {
1522 	if ((symbol_name && addr) || (!symbol_name && !addr))
1523 		return ERR_PTR(-EINVAL);
1524 
1525 	if (symbol_name) {
1526 		/*
1527 		 * Input: @sym + @offset
1528 		 * Output: @addr + @offset
1529 		 *
1530 		 * NOTE: kprobe_lookup_name() does *NOT* fold the offset
1531 		 *       argument into it's output!
1532 		 */
1533 		addr = kprobe_lookup_name(symbol_name, offset);
1534 		if (!addr)
1535 			return ERR_PTR(-ENOENT);
1536 	}
1537 
1538 	/*
1539 	 * So here we have @addr + @offset, displace it into a new
1540 	 * @addr' + @offset' where @addr' is the symbol start address.
1541 	 */
1542 	addr = (void *)addr + offset;
1543 	if (!kallsyms_lookup_size_offset((unsigned long)addr, NULL, &offset))
1544 		return ERR_PTR(-ENOENT);
1545 	addr = (void *)addr - offset;
1546 
1547 	/*
1548 	 * Then ask the architecture to re-combine them, taking care of
1549 	 * magical function entry details while telling us if this was indeed
1550 	 * at the start of the function.
1551 	 */
1552 	addr = arch_adjust_kprobe_addr((unsigned long)addr, offset, on_func_entry);
1553 	if (!addr)
1554 		return ERR_PTR(-EINVAL);
1555 
1556 	return addr;
1557 }
1558 
1559 static kprobe_opcode_t *kprobe_addr(struct kprobe *p)
1560 {
1561 	bool on_func_entry;
1562 
1563 	return _kprobe_addr(p->addr, p->symbol_name, p->offset, &on_func_entry);
1564 }
1565 
1566 /*
1567  * Check the 'p' is valid and return the aggregator kprobe
1568  * at the same address.
1569  */
1570 static struct kprobe *__get_valid_kprobe(struct kprobe *p)
1571 {
1572 	struct kprobe *ap, *list_p;
1573 
1574 	lockdep_assert_held(&kprobe_mutex);
1575 
1576 	ap = get_kprobe(p->addr);
1577 	if (unlikely(!ap))
1578 		return NULL;
1579 
1580 	if (p == ap)
1581 		return ap;
1582 
1583 	list_for_each_entry(list_p, &ap->list, list)
1584 		if (list_p == p)
1585 		/* kprobe p is a valid probe */
1586 			return ap;
1587 
1588 	return NULL;
1589 }
1590 
1591 /*
1592  * Warn and return error if the kprobe is being re-registered since
1593  * there must be a software bug.
1594  */
1595 static inline int warn_kprobe_rereg(struct kprobe *p)
1596 {
1597 	guard(mutex)(&kprobe_mutex);
1598 
1599 	if (WARN_ON_ONCE(__get_valid_kprobe(p)))
1600 		return -EINVAL;
1601 
1602 	return 0;
1603 }
1604 
1605 static int check_ftrace_location(struct kprobe *p)
1606 {
1607 	unsigned long addr = (unsigned long)p->addr;
1608 
1609 	if (ftrace_location(addr) == addr) {
1610 #ifdef CONFIG_KPROBES_ON_FTRACE
1611 		p->flags |= KPROBE_FLAG_FTRACE;
1612 #else
1613 		return -EINVAL;
1614 #endif
1615 	}
1616 	return 0;
1617 }
1618 
1619 static bool is_cfi_preamble_symbol(unsigned long addr)
1620 {
1621 	char symbuf[KSYM_NAME_LEN];
1622 
1623 	if (lookup_symbol_name(addr, symbuf))
1624 		return false;
1625 
1626 	return str_has_prefix(symbuf, "__cfi_") ||
1627 		str_has_prefix(symbuf, "__pfx_");
1628 }
1629 
1630 static int check_kprobe_address_safe(struct kprobe *p,
1631 				     struct module **probed_mod)
1632 {
1633 	int ret;
1634 
1635 	ret = check_ftrace_location(p);
1636 	if (ret)
1637 		return ret;
1638 
1639 	guard(jump_label_lock)();
1640 
1641 	/* Ensure the address is in a text area, and find a module if exists. */
1642 	*probed_mod = NULL;
1643 	if (!core_kernel_text((unsigned long) p->addr)) {
1644 		guard(rcu)();
1645 		*probed_mod = __module_text_address((unsigned long) p->addr);
1646 		if (!(*probed_mod))
1647 			return -EINVAL;
1648 
1649 		/*
1650 		 * We must hold a refcount of the probed module while updating
1651 		 * its code to prohibit unexpected unloading.
1652 		 */
1653 		if (unlikely(!try_module_get(*probed_mod)))
1654 			return -ENOENT;
1655 	}
1656 	/* Ensure it is not in reserved area. */
1657 	if (in_gate_area_no_mm((unsigned long) p->addr) ||
1658 	    within_kprobe_blacklist((unsigned long) p->addr) ||
1659 	    jump_label_text_reserved(p->addr, p->addr) ||
1660 	    static_call_text_reserved(p->addr, p->addr) ||
1661 	    find_bug((unsigned long)p->addr) ||
1662 	    is_cfi_preamble_symbol((unsigned long)p->addr)) {
1663 		module_put(*probed_mod);
1664 		return -EINVAL;
1665 	}
1666 
1667 	/* Get module refcount and reject __init functions for loaded modules. */
1668 	if (IS_ENABLED(CONFIG_MODULES) && *probed_mod) {
1669 		/*
1670 		 * If the module freed '.init.text', we couldn't insert
1671 		 * kprobes in there.
1672 		 */
1673 		if (within_module_init((unsigned long)p->addr, *probed_mod) &&
1674 		    !module_is_coming(*probed_mod)) {
1675 			module_put(*probed_mod);
1676 			return -ENOENT;
1677 		}
1678 	}
1679 
1680 	return 0;
1681 }
1682 
1683 static int __register_kprobe(struct kprobe *p)
1684 {
1685 	int ret;
1686 	struct kprobe *old_p;
1687 
1688 	guard(mutex)(&kprobe_mutex);
1689 
1690 	old_p = get_kprobe(p->addr);
1691 	if (old_p)
1692 		/* Since this may unoptimize 'old_p', locking 'text_mutex'. */
1693 		return register_aggr_kprobe(old_p, p);
1694 
1695 	scoped_guard(cpus_read_lock) {
1696 		/* Prevent text modification */
1697 		guard(mutex)(&text_mutex);
1698 		ret = prepare_kprobe(p);
1699 		if (ret)
1700 			return ret;
1701 	}
1702 
1703 	INIT_HLIST_NODE(&p->hlist);
1704 	hlist_add_head_rcu(&p->hlist,
1705 		       &kprobe_table[hash_ptr(p->addr, KPROBE_HASH_BITS)]);
1706 
1707 	if (!kprobes_all_disarmed && !kprobe_disabled(p)) {
1708 		ret = arm_kprobe(p);
1709 		if (ret) {
1710 			hlist_del_rcu(&p->hlist);
1711 			synchronize_rcu();
1712 		}
1713 	}
1714 
1715 	/* Try to optimize kprobe */
1716 	try_to_optimize_kprobe(p);
1717 	return 0;
1718 }
1719 
1720 int register_kprobe(struct kprobe *p)
1721 {
1722 	int ret;
1723 	struct module *probed_mod;
1724 	kprobe_opcode_t *addr;
1725 	bool on_func_entry;
1726 
1727 	/* Canonicalize probe address from symbol */
1728 	addr = _kprobe_addr(p->addr, p->symbol_name, p->offset, &on_func_entry);
1729 	if (IS_ERR(addr))
1730 		return PTR_ERR(addr);
1731 	p->addr = addr;
1732 
1733 	ret = warn_kprobe_rereg(p);
1734 	if (ret)
1735 		return ret;
1736 
1737 	/* User can pass only KPROBE_FLAG_DISABLED to register_kprobe */
1738 	p->flags &= KPROBE_FLAG_DISABLED;
1739 	if (on_func_entry)
1740 		p->flags |= KPROBE_FLAG_ON_FUNC_ENTRY;
1741 	p->nmissed = 0;
1742 	INIT_LIST_HEAD(&p->list);
1743 
1744 	ret = check_kprobe_address_safe(p, &probed_mod);
1745 	if (ret)
1746 		return ret;
1747 
1748 	ret = __register_kprobe(p);
1749 
1750 	if (probed_mod)
1751 		module_put(probed_mod);
1752 
1753 	return ret;
1754 }
1755 EXPORT_SYMBOL_GPL(register_kprobe);
1756 
1757 /* Check if all probes on the 'ap' are disabled. */
1758 static bool aggr_kprobe_disabled(struct kprobe *ap)
1759 {
1760 	struct kprobe *kp;
1761 
1762 	lockdep_assert_held(&kprobe_mutex);
1763 
1764 	list_for_each_entry(kp, &ap->list, list)
1765 		if (!kprobe_disabled(kp))
1766 			/*
1767 			 * Since there is an active probe on the list,
1768 			 * we can't disable this 'ap'.
1769 			 */
1770 			return false;
1771 
1772 	return true;
1773 }
1774 
1775 static struct kprobe *__disable_kprobe(struct kprobe *p)
1776 {
1777 	struct kprobe *orig_p;
1778 	int ret;
1779 
1780 	lockdep_assert_held(&kprobe_mutex);
1781 
1782 	/* Get an original kprobe for return */
1783 	orig_p = __get_valid_kprobe(p);
1784 	if (unlikely(orig_p == NULL))
1785 		return ERR_PTR(-EINVAL);
1786 
1787 	if (kprobe_disabled(p))
1788 		return orig_p;
1789 
1790 	/* Disable probe if it is a child probe */
1791 	if (p != orig_p)
1792 		p->flags |= KPROBE_FLAG_DISABLED;
1793 
1794 	/* Try to disarm and disable this/parent probe */
1795 	if (p == orig_p || aggr_kprobe_disabled(orig_p)) {
1796 		/*
1797 		 * Don't be lazy here.  Even if 'kprobes_all_disarmed'
1798 		 * is false, 'orig_p' might not have been armed yet.
1799 		 * Note arm_all_kprobes() __tries__ to arm all kprobes
1800 		 * on the best effort basis.
1801 		 */
1802 		if (!kprobes_all_disarmed && !kprobe_disabled(orig_p)) {
1803 			ret = disarm_kprobe(orig_p, true);
1804 			if (ret) {
1805 				p->flags &= ~KPROBE_FLAG_DISABLED;
1806 				return ERR_PTR(ret);
1807 			}
1808 		}
1809 		orig_p->flags |= KPROBE_FLAG_DISABLED;
1810 	}
1811 
1812 	return orig_p;
1813 }
1814 
1815 /*
1816  * Unregister a kprobe without a scheduler synchronization.
1817  */
1818 static int __unregister_kprobe_top(struct kprobe *p)
1819 {
1820 	struct kprobe *ap, *list_p;
1821 
1822 	/* Disable kprobe. This will disarm it if needed. */
1823 	ap = __disable_kprobe(p);
1824 	if (IS_ERR(ap))
1825 		return PTR_ERR(ap);
1826 
1827 	WARN_ON(ap != p && !kprobe_aggrprobe(ap));
1828 
1829 	/*
1830 	 * If the probe is an independent(and non-optimized) kprobe
1831 	 * (not an aggrprobe), the last kprobe on the aggrprobe, or
1832 	 * kprobe is already disarmed, just remove from the hash list.
1833 	 */
1834 	if (ap == p ||
1835 		(list_is_singular(&ap->list) && kprobe_disarmed(ap))) {
1836 		/*
1837 		 * !disarmed could be happen if the probe is under delayed
1838 		 * unoptimizing.
1839 		 */
1840 		hlist_del_rcu(&ap->hlist);
1841 		return 0;
1842 	}
1843 
1844 	/* If disabling probe has special handlers, update aggrprobe */
1845 	if (p->post_handler && !kprobe_gone(p)) {
1846 		list_for_each_entry(list_p, &ap->list, list) {
1847 			if ((list_p != p) && (list_p->post_handler))
1848 				break;
1849 		}
1850 		/* No other probe has post_handler */
1851 		if (list_entry_is_head(list_p, &ap->list, list)) {
1852 			/*
1853 			 * For the kprobe-on-ftrace case, we keep the
1854 			 * post_handler setting to identify this aggrprobe
1855 			 * armed with kprobe_ipmodify_ops.
1856 			 */
1857 			if (!kprobe_ftrace(ap))
1858 				ap->post_handler = NULL;
1859 		}
1860 	}
1861 
1862 	/*
1863 	 * Remove from the aggrprobe: this path will do nothing in
1864 	 * __unregister_kprobe_bottom().
1865 	 */
1866 	list_del_rcu(&p->list);
1867 	if (!kprobe_disabled(ap) && !kprobes_all_disarmed)
1868 		/*
1869 		 * Try to optimize this probe again, because post
1870 		 * handler may have been changed.
1871 		 */
1872 		optimize_kprobe(ap);
1873 	return 0;
1874 
1875 }
1876 
1877 static void __unregister_kprobe_bottom(struct kprobe *p)
1878 {
1879 	struct kprobe *ap;
1880 
1881 	if (list_empty(&p->list))
1882 		/* This is an independent kprobe */
1883 		arch_remove_kprobe(p);
1884 	else if (list_is_singular(&p->list)) {
1885 		/* This is the last child of an aggrprobe */
1886 		ap = list_entry(p->list.next, struct kprobe, list);
1887 		list_del(&p->list);
1888 		free_aggr_kprobe(ap);
1889 	}
1890 	/* Otherwise, do nothing. */
1891 }
1892 
1893 int register_kprobes(struct kprobe **kps, int num)
1894 {
1895 	int i, ret = 0;
1896 
1897 	if (num <= 0)
1898 		return -EINVAL;
1899 	for (i = 0; i < num; i++) {
1900 		ret = register_kprobe(kps[i]);
1901 		if (ret < 0) {
1902 			if (i > 0)
1903 				unregister_kprobes(kps, i);
1904 			break;
1905 		}
1906 	}
1907 	return ret;
1908 }
1909 EXPORT_SYMBOL_GPL(register_kprobes);
1910 
1911 void unregister_kprobe(struct kprobe *p)
1912 {
1913 	unregister_kprobes(&p, 1);
1914 }
1915 EXPORT_SYMBOL_GPL(unregister_kprobe);
1916 
1917 void unregister_kprobes(struct kprobe **kps, int num)
1918 {
1919 	int i;
1920 
1921 	if (num <= 0)
1922 		return;
1923 	scoped_guard(mutex, &kprobe_mutex) {
1924 		for (i = 0; i < num; i++)
1925 			if (__unregister_kprobe_top(kps[i]) < 0)
1926 				kps[i]->addr = NULL;
1927 	}
1928 	synchronize_rcu();
1929 	for (i = 0; i < num; i++)
1930 		if (kps[i]->addr)
1931 			__unregister_kprobe_bottom(kps[i]);
1932 }
1933 EXPORT_SYMBOL_GPL(unregister_kprobes);
1934 
1935 int __weak kprobe_exceptions_notify(struct notifier_block *self,
1936 					unsigned long val, void *data)
1937 {
1938 	return NOTIFY_DONE;
1939 }
1940 NOKPROBE_SYMBOL(kprobe_exceptions_notify);
1941 
1942 static struct notifier_block kprobe_exceptions_nb = {
1943 	.notifier_call = kprobe_exceptions_notify,
1944 	.priority = 0x7fffffff /* we need to be notified first */
1945 };
1946 
1947 #ifdef CONFIG_KRETPROBES
1948 
1949 #if !defined(CONFIG_KRETPROBE_ON_RETHOOK)
1950 
1951 /* callbacks for objpool of kretprobe instances */
1952 static int kretprobe_init_inst(void *nod, void *context)
1953 {
1954 	struct kretprobe_instance *ri = nod;
1955 
1956 	ri->rph = context;
1957 	return 0;
1958 }
1959 static int kretprobe_fini_pool(struct objpool_head *head, void *context)
1960 {
1961 	kfree(context);
1962 	return 0;
1963 }
1964 
1965 static void free_rp_inst_rcu(struct rcu_head *head)
1966 {
1967 	struct kretprobe_instance *ri = container_of(head, struct kretprobe_instance, rcu);
1968 	struct kretprobe_holder *rph = ri->rph;
1969 
1970 	objpool_drop(ri, &rph->pool);
1971 }
1972 NOKPROBE_SYMBOL(free_rp_inst_rcu);
1973 
1974 static void recycle_rp_inst(struct kretprobe_instance *ri)
1975 {
1976 	struct kretprobe *rp = get_kretprobe(ri);
1977 
1978 	if (likely(rp))
1979 		objpool_push(ri, &rp->rph->pool);
1980 	else
1981 		call_rcu(&ri->rcu, free_rp_inst_rcu);
1982 }
1983 NOKPROBE_SYMBOL(recycle_rp_inst);
1984 
1985 /*
1986  * This function is called from delayed_put_task_struct() when a task is
1987  * dead and cleaned up to recycle any kretprobe instances associated with
1988  * this task. These left over instances represent probed functions that
1989  * have been called but will never return.
1990  */
1991 void kprobe_flush_task(struct task_struct *tk)
1992 {
1993 	struct kretprobe_instance *ri;
1994 	struct llist_node *node;
1995 
1996 	/* Early boot, not yet initialized. */
1997 	if (unlikely(!kprobes_initialized))
1998 		return;
1999 
2000 	kprobe_busy_begin();
2001 
2002 	node = __llist_del_all(&tk->kretprobe_instances);
2003 	while (node) {
2004 		ri = container_of(node, struct kretprobe_instance, llist);
2005 		node = node->next;
2006 
2007 		recycle_rp_inst(ri);
2008 	}
2009 
2010 	kprobe_busy_end();
2011 }
2012 NOKPROBE_SYMBOL(kprobe_flush_task);
2013 
2014 static inline void free_rp_inst(struct kretprobe *rp)
2015 {
2016 	struct kretprobe_holder *rph = rp->rph;
2017 
2018 	if (!rph)
2019 		return;
2020 	rp->rph = NULL;
2021 	objpool_fini(&rph->pool);
2022 }
2023 
2024 /* This assumes the 'tsk' is the current task or the is not running. */
2025 static kprobe_opcode_t *__kretprobe_find_ret_addr(struct task_struct *tsk,
2026 						  struct llist_node **cur)
2027 {
2028 	struct kretprobe_instance *ri = NULL;
2029 	struct llist_node *node = *cur;
2030 
2031 	if (!node)
2032 		node = tsk->kretprobe_instances.first;
2033 	else
2034 		node = node->next;
2035 
2036 	while (node) {
2037 		ri = container_of(node, struct kretprobe_instance, llist);
2038 		if (ri->ret_addr != kretprobe_trampoline_addr()) {
2039 			*cur = node;
2040 			return ri->ret_addr;
2041 		}
2042 		node = node->next;
2043 	}
2044 	return NULL;
2045 }
2046 NOKPROBE_SYMBOL(__kretprobe_find_ret_addr);
2047 
2048 /**
2049  * kretprobe_find_ret_addr -- Find correct return address modified by kretprobe
2050  * @tsk: Target task
2051  * @fp: A frame pointer
2052  * @cur: a storage of the loop cursor llist_node pointer for next call
2053  *
2054  * Find the correct return address modified by a kretprobe on @tsk in unsigned
2055  * long type. If it finds the return address, this returns that address value,
2056  * or this returns 0.
2057  * The @tsk must be 'current' or a task which is not running. @fp is a hint
2058  * to get the currect return address - which is compared with the
2059  * kretprobe_instance::fp field. The @cur is a loop cursor for searching the
2060  * kretprobe return addresses on the @tsk. The '*@cur' should be NULL at the
2061  * first call, but '@cur' itself must NOT NULL.
2062  */
2063 unsigned long kretprobe_find_ret_addr(struct task_struct *tsk, void *fp,
2064 				      struct llist_node **cur)
2065 {
2066 	struct kretprobe_instance *ri;
2067 	kprobe_opcode_t *ret;
2068 
2069 	if (WARN_ON_ONCE(!cur))
2070 		return 0;
2071 
2072 	do {
2073 		ret = __kretprobe_find_ret_addr(tsk, cur);
2074 		if (!ret)
2075 			break;
2076 		ri = container_of(*cur, struct kretprobe_instance, llist);
2077 	} while (ri->fp != fp);
2078 
2079 	return (unsigned long)ret;
2080 }
2081 NOKPROBE_SYMBOL(kretprobe_find_ret_addr);
2082 
2083 void __weak arch_kretprobe_fixup_return(struct pt_regs *regs,
2084 					kprobe_opcode_t *correct_ret_addr)
2085 {
2086 	/*
2087 	 * Do nothing by default. Please fill this to update the fake return
2088 	 * address on the stack with the correct one on each arch if possible.
2089 	 */
2090 }
2091 
2092 unsigned long __kretprobe_trampoline_handler(struct pt_regs *regs,
2093 					     void *frame_pointer)
2094 {
2095 	struct kretprobe_instance *ri = NULL;
2096 	struct llist_node *first, *node = NULL;
2097 	kprobe_opcode_t *correct_ret_addr;
2098 	struct kretprobe *rp;
2099 
2100 	/* Find correct address and all nodes for this frame. */
2101 	correct_ret_addr = __kretprobe_find_ret_addr(current, &node);
2102 	if (!correct_ret_addr) {
2103 		pr_err("kretprobe: Return address not found, not execute handler. Maybe there is a bug in the kernel.\n");
2104 		BUG_ON(1);
2105 	}
2106 
2107 	/*
2108 	 * Set the return address as the instruction pointer, because if the
2109 	 * user handler calls stack_trace_save_regs() with this 'regs',
2110 	 * the stack trace will start from the instruction pointer.
2111 	 */
2112 	instruction_pointer_set(regs, (unsigned long)correct_ret_addr);
2113 
2114 	/* Run the user handler of the nodes. */
2115 	first = current->kretprobe_instances.first;
2116 	while (first) {
2117 		ri = container_of(first, struct kretprobe_instance, llist);
2118 
2119 		if (WARN_ON_ONCE(ri->fp != frame_pointer))
2120 			break;
2121 
2122 		rp = get_kretprobe(ri);
2123 		if (rp && rp->handler) {
2124 			struct kprobe *prev = kprobe_running();
2125 
2126 			__this_cpu_write(current_kprobe, &rp->kp);
2127 			ri->ret_addr = correct_ret_addr;
2128 			rp->handler(ri, regs);
2129 			__this_cpu_write(current_kprobe, prev);
2130 		}
2131 		if (first == node)
2132 			break;
2133 
2134 		first = first->next;
2135 	}
2136 
2137 	arch_kretprobe_fixup_return(regs, correct_ret_addr);
2138 
2139 	/* Unlink all nodes for this frame. */
2140 	first = current->kretprobe_instances.first;
2141 	current->kretprobe_instances.first = node->next;
2142 	node->next = NULL;
2143 
2144 	/* Recycle free instances. */
2145 	while (first) {
2146 		ri = container_of(first, struct kretprobe_instance, llist);
2147 		first = first->next;
2148 
2149 		recycle_rp_inst(ri);
2150 	}
2151 
2152 	return (unsigned long)correct_ret_addr;
2153 }
2154 NOKPROBE_SYMBOL(__kretprobe_trampoline_handler)
2155 
2156 /*
2157  * This kprobe pre_handler is registered with every kretprobe. When probe
2158  * hits it will set up the return probe.
2159  */
2160 static int pre_handler_kretprobe(struct kprobe *p, struct pt_regs *regs)
2161 {
2162 	struct kretprobe *rp = container_of(p, struct kretprobe, kp);
2163 	struct kretprobe_holder *rph = rp->rph;
2164 	struct kretprobe_instance *ri;
2165 
2166 	ri = objpool_pop(&rph->pool);
2167 	if (!ri) {
2168 		rp->nmissed++;
2169 		return 0;
2170 	}
2171 
2172 	if (rp->entry_handler && rp->entry_handler(ri, regs)) {
2173 		objpool_push(ri, &rph->pool);
2174 		return 0;
2175 	}
2176 
2177 	arch_prepare_kretprobe(ri, regs);
2178 
2179 	__llist_add(&ri->llist, &current->kretprobe_instances);
2180 
2181 	return 0;
2182 }
2183 NOKPROBE_SYMBOL(pre_handler_kretprobe);
2184 #else /* CONFIG_KRETPROBE_ON_RETHOOK */
2185 /*
2186  * This kprobe pre_handler is registered with every kretprobe. When probe
2187  * hits it will set up the return probe.
2188  */
2189 static int pre_handler_kretprobe(struct kprobe *p, struct pt_regs *regs)
2190 {
2191 	struct kretprobe *rp = container_of(p, struct kretprobe, kp);
2192 	struct kretprobe_instance *ri;
2193 	struct rethook_node *rhn;
2194 
2195 	rhn = rethook_try_get(rp->rh);
2196 	if (!rhn) {
2197 		rp->nmissed++;
2198 		return 0;
2199 	}
2200 
2201 	ri = container_of(rhn, struct kretprobe_instance, node);
2202 
2203 	if (rp->entry_handler && rp->entry_handler(ri, regs))
2204 		rethook_recycle(rhn);
2205 	else
2206 		rethook_hook(rhn, regs, kprobe_ftrace(p));
2207 
2208 	return 0;
2209 }
2210 NOKPROBE_SYMBOL(pre_handler_kretprobe);
2211 
2212 static void kretprobe_rethook_handler(struct rethook_node *rh, void *data,
2213 				      unsigned long ret_addr,
2214 				      struct pt_regs *regs)
2215 {
2216 	struct kretprobe *rp = (struct kretprobe *)data;
2217 	struct kretprobe_instance *ri;
2218 	struct kprobe_ctlblk *kcb;
2219 
2220 	/* The data must NOT be null. This means rethook data structure is broken. */
2221 	if (WARN_ON_ONCE(!data) || !rp->handler)
2222 		return;
2223 
2224 	__this_cpu_write(current_kprobe, &rp->kp);
2225 	kcb = get_kprobe_ctlblk();
2226 	kcb->kprobe_status = KPROBE_HIT_ACTIVE;
2227 
2228 	ri = container_of(rh, struct kretprobe_instance, node);
2229 	rp->handler(ri, regs);
2230 
2231 	__this_cpu_write(current_kprobe, NULL);
2232 }
2233 NOKPROBE_SYMBOL(kretprobe_rethook_handler);
2234 
2235 #endif /* !CONFIG_KRETPROBE_ON_RETHOOK */
2236 
2237 /**
2238  * kprobe_on_func_entry() -- check whether given address is function entry
2239  * @addr: Target address
2240  * @sym:  Target symbol name
2241  * @offset: The offset from the symbol or the address
2242  *
2243  * This checks whether the given @addr+@offset or @sym+@offset is on the
2244  * function entry address or not.
2245  * This returns 0 if it is the function entry, or -EINVAL if it is not.
2246  * And also it returns -ENOENT if it fails the symbol or address lookup.
2247  * Caller must pass @addr or @sym (either one must be NULL), or this
2248  * returns -EINVAL.
2249  */
2250 int kprobe_on_func_entry(kprobe_opcode_t *addr, const char *sym, unsigned long offset)
2251 {
2252 	bool on_func_entry;
2253 	kprobe_opcode_t *kp_addr = _kprobe_addr(addr, sym, offset, &on_func_entry);
2254 
2255 	if (IS_ERR(kp_addr))
2256 		return PTR_ERR(kp_addr);
2257 
2258 	if (!on_func_entry)
2259 		return -EINVAL;
2260 
2261 	return 0;
2262 }
2263 
2264 int register_kretprobe(struct kretprobe *rp)
2265 {
2266 	int ret;
2267 	int i;
2268 	void *addr;
2269 
2270 	ret = kprobe_on_func_entry(rp->kp.addr, rp->kp.symbol_name, rp->kp.offset);
2271 	if (ret)
2272 		return ret;
2273 
2274 	/* If only 'rp->kp.addr' is specified, check reregistering kprobes */
2275 	if (rp->kp.addr && warn_kprobe_rereg(&rp->kp))
2276 		return -EINVAL;
2277 
2278 	if (kretprobe_blacklist_size) {
2279 		addr = kprobe_addr(&rp->kp);
2280 		if (IS_ERR(addr))
2281 			return PTR_ERR(addr);
2282 
2283 		for (i = 0; kretprobe_blacklist[i].name != NULL; i++) {
2284 			if (kretprobe_blacklist[i].addr == addr)
2285 				return -EINVAL;
2286 		}
2287 	}
2288 
2289 	if (rp->data_size > KRETPROBE_MAX_DATA_SIZE)
2290 		return -E2BIG;
2291 
2292 	rp->kp.pre_handler = pre_handler_kretprobe;
2293 	rp->kp.post_handler = NULL;
2294 
2295 	/* Pre-allocate memory for max kretprobe instances */
2296 	if (rp->maxactive <= 0)
2297 		rp->maxactive = max_t(unsigned int, 10, 2*num_possible_cpus());
2298 
2299 #ifdef CONFIG_KRETPROBE_ON_RETHOOK
2300 	rp->rh = rethook_alloc((void *)rp, kretprobe_rethook_handler,
2301 				sizeof(struct kretprobe_instance) +
2302 				rp->data_size, rp->maxactive);
2303 	if (IS_ERR(rp->rh))
2304 		return PTR_ERR(rp->rh);
2305 
2306 	rp->nmissed = 0;
2307 	/* Establish function entry probe point */
2308 	ret = register_kprobe(&rp->kp);
2309 	if (ret != 0) {
2310 		rethook_free(rp->rh);
2311 		rp->rh = NULL;
2312 	}
2313 #else	/* !CONFIG_KRETPROBE_ON_RETHOOK */
2314 	rp->rph = kzalloc_obj(struct kretprobe_holder);
2315 	if (!rp->rph)
2316 		return -ENOMEM;
2317 
2318 	if (objpool_init(&rp->rph->pool, rp->maxactive, rp->data_size +
2319 			sizeof(struct kretprobe_instance), GFP_KERNEL,
2320 			rp->rph, kretprobe_init_inst, kretprobe_fini_pool)) {
2321 		kfree(rp->rph);
2322 		rp->rph = NULL;
2323 		return -ENOMEM;
2324 	}
2325 	rcu_assign_pointer(rp->rph->rp, rp);
2326 	rp->nmissed = 0;
2327 	/* Establish function entry probe point */
2328 	ret = register_kprobe(&rp->kp);
2329 	if (ret != 0)
2330 		free_rp_inst(rp);
2331 #endif
2332 	return ret;
2333 }
2334 EXPORT_SYMBOL_GPL(register_kretprobe);
2335 
2336 int register_kretprobes(struct kretprobe **rps, int num)
2337 {
2338 	int ret = 0, i;
2339 
2340 	if (num <= 0)
2341 		return -EINVAL;
2342 	for (i = 0; i < num; i++) {
2343 		ret = register_kretprobe(rps[i]);
2344 		if (ret < 0) {
2345 			if (i > 0)
2346 				unregister_kretprobes(rps, i);
2347 			break;
2348 		}
2349 	}
2350 	return ret;
2351 }
2352 EXPORT_SYMBOL_GPL(register_kretprobes);
2353 
2354 void unregister_kretprobe(struct kretprobe *rp)
2355 {
2356 	unregister_kretprobes(&rp, 1);
2357 }
2358 EXPORT_SYMBOL_GPL(unregister_kretprobe);
2359 
2360 void unregister_kretprobes(struct kretprobe **rps, int num)
2361 {
2362 	int i;
2363 
2364 	if (num <= 0)
2365 		return;
2366 	for (i = 0; i < num; i++) {
2367 		guard(mutex)(&kprobe_mutex);
2368 
2369 		if (__unregister_kprobe_top(&rps[i]->kp) < 0)
2370 			rps[i]->kp.addr = NULL;
2371 #ifdef CONFIG_KRETPROBE_ON_RETHOOK
2372 		rethook_free(rps[i]->rh);
2373 #else
2374 		rcu_assign_pointer(rps[i]->rph->rp, NULL);
2375 #endif
2376 	}
2377 
2378 	synchronize_rcu();
2379 	for (i = 0; i < num; i++) {
2380 		if (rps[i]->kp.addr) {
2381 			__unregister_kprobe_bottom(&rps[i]->kp);
2382 #ifndef CONFIG_KRETPROBE_ON_RETHOOK
2383 			free_rp_inst(rps[i]);
2384 #endif
2385 		}
2386 	}
2387 }
2388 EXPORT_SYMBOL_GPL(unregister_kretprobes);
2389 
2390 #else /* CONFIG_KRETPROBES */
2391 int register_kretprobe(struct kretprobe *rp)
2392 {
2393 	return -EOPNOTSUPP;
2394 }
2395 EXPORT_SYMBOL_GPL(register_kretprobe);
2396 
2397 int register_kretprobes(struct kretprobe **rps, int num)
2398 {
2399 	return -EOPNOTSUPP;
2400 }
2401 EXPORT_SYMBOL_GPL(register_kretprobes);
2402 
2403 void unregister_kretprobe(struct kretprobe *rp)
2404 {
2405 }
2406 EXPORT_SYMBOL_GPL(unregister_kretprobe);
2407 
2408 void unregister_kretprobes(struct kretprobe **rps, int num)
2409 {
2410 }
2411 EXPORT_SYMBOL_GPL(unregister_kretprobes);
2412 
2413 static int pre_handler_kretprobe(struct kprobe *p, struct pt_regs *regs)
2414 {
2415 	return 0;
2416 }
2417 NOKPROBE_SYMBOL(pre_handler_kretprobe);
2418 
2419 #endif /* CONFIG_KRETPROBES */
2420 
2421 /* Set the kprobe gone and remove its instruction buffer. */
2422 static void kill_kprobe(struct kprobe *p)
2423 {
2424 	struct kprobe *kp;
2425 
2426 	lockdep_assert_held(&kprobe_mutex);
2427 
2428 	/*
2429 	 * The module is going away. We should disarm the kprobe which
2430 	 * is using ftrace, because ftrace framework is still available at
2431 	 * 'MODULE_STATE_GOING' notification.
2432 	 */
2433 	if (kprobe_ftrace(p) && !kprobe_disabled(p) && !kprobes_all_disarmed)
2434 		disarm_kprobe_ftrace(p);
2435 
2436 	p->flags |= KPROBE_FLAG_GONE;
2437 	if (kprobe_aggrprobe(p)) {
2438 		/*
2439 		 * If this is an aggr_kprobe, we have to list all the
2440 		 * chained probes and mark them GONE.
2441 		 */
2442 		list_for_each_entry(kp, &p->list, list)
2443 			kp->flags |= KPROBE_FLAG_GONE;
2444 		p->post_handler = NULL;
2445 		kill_optimized_kprobe(p);
2446 	}
2447 	/*
2448 	 * Here, we can remove insn_slot safely, because no thread calls
2449 	 * the original probed function (which will be freed soon) any more.
2450 	 */
2451 	arch_remove_kprobe(p);
2452 }
2453 
2454 /* Disable one kprobe */
2455 int disable_kprobe(struct kprobe *kp)
2456 {
2457 	struct kprobe *p;
2458 
2459 	guard(mutex)(&kprobe_mutex);
2460 
2461 	/* Disable this kprobe */
2462 	p = __disable_kprobe(kp);
2463 
2464 	return IS_ERR(p) ? PTR_ERR(p) : 0;
2465 }
2466 EXPORT_SYMBOL_GPL(disable_kprobe);
2467 
2468 /* Enable one kprobe */
2469 int enable_kprobe(struct kprobe *kp)
2470 {
2471 	int ret = 0;
2472 	struct kprobe *p;
2473 
2474 	guard(mutex)(&kprobe_mutex);
2475 
2476 	/* Check whether specified probe is valid. */
2477 	p = __get_valid_kprobe(kp);
2478 	if (unlikely(p == NULL))
2479 		return -EINVAL;
2480 
2481 	if (kprobe_gone(kp))
2482 		/* This kprobe has gone, we couldn't enable it. */
2483 		return -EINVAL;
2484 
2485 	if (p != kp)
2486 		kp->flags &= ~KPROBE_FLAG_DISABLED;
2487 
2488 	if (!kprobes_all_disarmed && kprobe_disabled(p)) {
2489 		p->flags &= ~KPROBE_FLAG_DISABLED;
2490 		ret = arm_kprobe(p);
2491 		if (ret) {
2492 			p->flags |= KPROBE_FLAG_DISABLED;
2493 			if (p != kp)
2494 				kp->flags |= KPROBE_FLAG_DISABLED;
2495 		}
2496 	}
2497 	return ret;
2498 }
2499 EXPORT_SYMBOL_GPL(enable_kprobe);
2500 
2501 /* Caller must NOT call this in usual path. This is only for critical case */
2502 void dump_kprobe(struct kprobe *kp)
2503 {
2504 	pr_err("Dump kprobe:\n.symbol_name = %s, .offset = %x, .addr = %pS\n",
2505 	       kp->symbol_name, kp->offset, kp->addr);
2506 }
2507 NOKPROBE_SYMBOL(dump_kprobe);
2508 
2509 int kprobe_add_ksym_blacklist(unsigned long entry)
2510 {
2511 	struct kprobe_blacklist_entry *ent;
2512 	unsigned long offset = 0, size = 0;
2513 
2514 	if (!kernel_text_address(entry) ||
2515 	    !kallsyms_lookup_size_offset(entry, &size, &offset))
2516 		return -EINVAL;
2517 
2518 	ent = kmalloc_obj(*ent);
2519 	if (!ent)
2520 		return -ENOMEM;
2521 	ent->start_addr = entry;
2522 	ent->end_addr = entry + size;
2523 	INIT_LIST_HEAD(&ent->list);
2524 	list_add_tail_rcu(&ent->list, &kprobe_blacklist);
2525 
2526 	return (int)size;
2527 }
2528 
2529 /* Add all symbols in given area into kprobe blacklist */
2530 int kprobe_add_area_blacklist(unsigned long start, unsigned long end)
2531 {
2532 	unsigned long entry;
2533 	int ret = 0;
2534 
2535 	for (entry = start; entry < end; entry += ret) {
2536 		ret = kprobe_add_ksym_blacklist(entry);
2537 		if (ret < 0)
2538 			return ret;
2539 		if (ret == 0)	/* In case of alias symbol */
2540 			ret = 1;
2541 	}
2542 	return 0;
2543 }
2544 
2545 int __weak arch_kprobe_get_kallsym(unsigned int *symnum, unsigned long *value,
2546 				   char *type, char *sym)
2547 {
2548 	return -ERANGE;
2549 }
2550 
2551 int kprobe_get_kallsym(unsigned int symnum, unsigned long *value, char *type,
2552 		       char *sym)
2553 {
2554 #ifdef __ARCH_WANT_KPROBES_INSN_SLOT
2555 	if (!kprobe_cache_get_kallsym(&kprobe_insn_slots, &symnum, value, type, sym))
2556 		return 0;
2557 #ifdef CONFIG_OPTPROBES
2558 	if (!kprobe_cache_get_kallsym(&kprobe_optinsn_slots, &symnum, value, type, sym))
2559 		return 0;
2560 #endif
2561 #endif
2562 	if (!arch_kprobe_get_kallsym(&symnum, value, type, sym))
2563 		return 0;
2564 	return -ERANGE;
2565 }
2566 
2567 int __init __weak arch_populate_kprobe_blacklist(void)
2568 {
2569 	return 0;
2570 }
2571 
2572 /*
2573  * Lookup and populate the kprobe_blacklist.
2574  *
2575  * Unlike the kretprobe blacklist, we'll need to determine
2576  * the range of addresses that belong to the said functions,
2577  * since a kprobe need not necessarily be at the beginning
2578  * of a function.
2579  */
2580 static int __init populate_kprobe_blacklist(unsigned long *start,
2581 					     unsigned long *end)
2582 {
2583 	unsigned long entry;
2584 	unsigned long *iter;
2585 	int ret;
2586 
2587 	for (iter = start; iter < end; iter++) {
2588 		entry = (unsigned long)dereference_symbol_descriptor((void *)*iter);
2589 		ret = kprobe_add_ksym_blacklist(entry);
2590 		if (ret == -EINVAL)
2591 			continue;
2592 		if (ret < 0)
2593 			return ret;
2594 	}
2595 
2596 	/* Symbols in '__kprobes_text' are blacklisted */
2597 	ret = kprobe_add_area_blacklist((unsigned long)__kprobes_text_start,
2598 					(unsigned long)__kprobes_text_end);
2599 	if (ret)
2600 		return ret;
2601 
2602 	/* Symbols in 'noinstr' section are blacklisted */
2603 	ret = kprobe_add_area_blacklist((unsigned long)__noinstr_text_start,
2604 					(unsigned long)__noinstr_text_end);
2605 
2606 	return ret ? : arch_populate_kprobe_blacklist();
2607 }
2608 
2609 #ifdef CONFIG_MODULES
2610 /* Remove all symbols in given area from kprobe blacklist */
2611 static void kprobe_remove_area_blacklist(unsigned long start, unsigned long end)
2612 {
2613 	struct kprobe_blacklist_entry *ent, *n;
2614 
2615 	list_for_each_entry_safe(ent, n, &kprobe_blacklist, list) {
2616 		if (ent->start_addr < start || ent->start_addr >= end)
2617 			continue;
2618 		list_del_rcu(&ent->list);
2619 		kfree_rcu(ent, rcu);
2620 	}
2621 }
2622 
2623 static void kprobe_remove_ksym_blacklist(unsigned long entry)
2624 {
2625 	kprobe_remove_area_blacklist(entry, entry + 1);
2626 }
2627 
2628 static void add_module_kprobe_blacklist(struct module *mod)
2629 {
2630 	unsigned long start, end;
2631 	int i;
2632 
2633 	if (mod->kprobe_blacklist) {
2634 		for (i = 0; i < mod->num_kprobe_blacklist; i++)
2635 			kprobe_add_ksym_blacklist(mod->kprobe_blacklist[i]);
2636 	}
2637 
2638 	start = (unsigned long)mod->kprobes_text_start;
2639 	if (start) {
2640 		end = start + mod->kprobes_text_size;
2641 		kprobe_add_area_blacklist(start, end);
2642 	}
2643 
2644 	start = (unsigned long)mod->noinstr_text_start;
2645 	if (start) {
2646 		end = start + mod->noinstr_text_size;
2647 		kprobe_add_area_blacklist(start, end);
2648 	}
2649 }
2650 
2651 static void remove_module_kprobe_blacklist(struct module *mod)
2652 {
2653 	unsigned long start, end;
2654 	int i;
2655 
2656 	if (mod->kprobe_blacklist) {
2657 		for (i = 0; i < mod->num_kprobe_blacklist; i++)
2658 			kprobe_remove_ksym_blacklist(mod->kprobe_blacklist[i]);
2659 	}
2660 
2661 	start = (unsigned long)mod->kprobes_text_start;
2662 	if (start) {
2663 		end = start + mod->kprobes_text_size;
2664 		kprobe_remove_area_blacklist(start, end);
2665 	}
2666 
2667 	start = (unsigned long)mod->noinstr_text_start;
2668 	if (start) {
2669 		end = start + mod->noinstr_text_size;
2670 		kprobe_remove_area_blacklist(start, end);
2671 	}
2672 }
2673 
2674 /* Module notifier call back, checking kprobes on the module */
2675 static int kprobes_module_callback(struct notifier_block *nb,
2676 				   unsigned long val, void *data)
2677 {
2678 	struct module *mod = data;
2679 	struct hlist_head *head;
2680 	struct kprobe *p;
2681 	unsigned int i;
2682 	int checkcore = (val == MODULE_STATE_GOING);
2683 
2684 	guard(mutex)(&kprobe_mutex);
2685 
2686 	if (val == MODULE_STATE_COMING)
2687 		add_module_kprobe_blacklist(mod);
2688 
2689 	if (val != MODULE_STATE_GOING && val != MODULE_STATE_LIVE)
2690 		return NOTIFY_DONE;
2691 
2692 	/*
2693 	 * When 'MODULE_STATE_GOING' was notified, both of module '.text' and
2694 	 * '.init.text' sections would be freed. When 'MODULE_STATE_LIVE' was
2695 	 * notified, only '.init.text' section would be freed. We need to
2696 	 * disable kprobes which have been inserted in the sections.
2697 	 */
2698 	for (i = 0; i < KPROBE_TABLE_SIZE; i++) {
2699 		head = &kprobe_table[i];
2700 		hlist_for_each_entry(p, head, hlist)
2701 			if (within_module_init((unsigned long)p->addr, mod) ||
2702 			    (checkcore &&
2703 			     within_module_core((unsigned long)p->addr, mod))) {
2704 				/*
2705 				 * The vaddr this probe is installed will soon
2706 				 * be vfreed buy not synced to disk. Hence,
2707 				 * disarming the breakpoint isn't needed.
2708 				 *
2709 				 * Note, this will also move any optimized probes
2710 				 * that are pending to be removed from their
2711 				 * corresponding lists to the 'freeing_list' and
2712 				 * will not be touched by the delayed
2713 				 * kprobe_optimizer() work handler.
2714 				 */
2715 				kill_kprobe(p);
2716 			}
2717 	}
2718 	if (val == MODULE_STATE_GOING)
2719 		remove_module_kprobe_blacklist(mod);
2720 	return NOTIFY_DONE;
2721 }
2722 
2723 static struct notifier_block kprobe_module_nb = {
2724 	.notifier_call = kprobes_module_callback,
2725 	.priority = 0
2726 };
2727 
2728 static int kprobe_register_module_notifier(void)
2729 {
2730 	return register_module_notifier(&kprobe_module_nb);
2731 }
2732 #else
2733 static int kprobe_register_module_notifier(void)
2734 {
2735 	return 0;
2736 }
2737 #endif /* CONFIG_MODULES */
2738 
2739 void kprobe_free_init_mem(void)
2740 {
2741 	void *start = (void *)(&__init_begin);
2742 	void *end = (void *)(&__init_end);
2743 	struct hlist_head *head;
2744 	struct kprobe *p;
2745 	int i;
2746 
2747 	guard(mutex)(&kprobe_mutex);
2748 
2749 	/* Kill all kprobes on initmem because the target code has been freed. */
2750 	for (i = 0; i < KPROBE_TABLE_SIZE; i++) {
2751 		head = &kprobe_table[i];
2752 		hlist_for_each_entry(p, head, hlist) {
2753 			if (start <= (void *)p->addr && (void *)p->addr < end)
2754 				kill_kprobe(p);
2755 		}
2756 	}
2757 }
2758 
2759 static int __init init_kprobes(void)
2760 {
2761 	int i, err;
2762 
2763 	/* FIXME allocate the probe table, currently defined statically */
2764 	/* initialize all list heads */
2765 	for (i = 0; i < KPROBE_TABLE_SIZE; i++)
2766 		INIT_HLIST_HEAD(&kprobe_table[i]);
2767 
2768 	err = populate_kprobe_blacklist(__start_kprobe_blacklist,
2769 					__stop_kprobe_blacklist);
2770 	if (err)
2771 		pr_err("Failed to populate blacklist (error %d), kprobes not restricted, be careful using them!\n", err);
2772 
2773 	if (kretprobe_blacklist_size) {
2774 		/* lookup the function address from its name */
2775 		for (i = 0; kretprobe_blacklist[i].name != NULL; i++) {
2776 			kretprobe_blacklist[i].addr =
2777 				kprobe_lookup_name(kretprobe_blacklist[i].name, 0);
2778 			if (!kretprobe_blacklist[i].addr)
2779 				pr_err("Failed to lookup symbol '%s' for kretprobe blacklist. Maybe the target function is removed or renamed.\n",
2780 				       kretprobe_blacklist[i].name);
2781 		}
2782 	}
2783 
2784 	/* By default, kprobes are armed */
2785 	kprobes_all_disarmed = false;
2786 
2787 	/* Initialize the optimization infrastructure */
2788 	init_optprobe();
2789 
2790 	err = arch_init_kprobes();
2791 	if (!err)
2792 		err = register_die_notifier(&kprobe_exceptions_nb);
2793 	if (!err)
2794 		err = kprobe_register_module_notifier();
2795 
2796 	kprobes_initialized = (err == 0);
2797 	kprobe_sysctls_init();
2798 	return err;
2799 }
2800 early_initcall(init_kprobes);
2801 
2802 #if defined(CONFIG_OPTPROBES)
2803 static int __init init_optprobes(void)
2804 {
2805 	/*
2806 	 * Enable kprobe optimization - this kicks the optimizer which
2807 	 * depends on synchronize_rcu_tasks() and ksoftirqd, that is
2808 	 * not spawned in early initcall. So delay the optimization.
2809 	 */
2810 	optimize_all_kprobes();
2811 
2812 	return 0;
2813 }
2814 subsys_initcall(init_optprobes);
2815 #endif
2816 
2817 #ifdef CONFIG_DEBUG_FS
2818 static void report_probe(struct seq_file *pi, struct kprobe *p,
2819 		const char *sym, int offset, char *modname, struct kprobe *pp)
2820 {
2821 	char *kprobe_type;
2822 	void *addr = p->addr;
2823 
2824 	if (p->pre_handler == pre_handler_kretprobe)
2825 		kprobe_type = "r";
2826 	else
2827 		kprobe_type = "k";
2828 
2829 	if (!kallsyms_show_value(pi->file->f_cred))
2830 		addr = NULL;
2831 
2832 	if (sym)
2833 		seq_printf(pi, "%px  %s  %s+0x%x  %s ",
2834 			addr, kprobe_type, sym, offset,
2835 			(modname ? modname : " "));
2836 	else	/* try to use %pS */
2837 		seq_printf(pi, "%px  %s  %pS ",
2838 			addr, kprobe_type, p->addr);
2839 
2840 	if (!pp)
2841 		pp = p;
2842 	seq_printf(pi, "%s%s%s%s\n",
2843 		(kprobe_gone(p) ? "[GONE]" : ""),
2844 		((kprobe_disabled(p) && !kprobe_gone(p)) ?  "[DISABLED]" : ""),
2845 		(kprobe_optimized(pp) ? "[OPTIMIZED]" : ""),
2846 		(kprobe_ftrace(pp) ? "[FTRACE]" : ""));
2847 }
2848 
2849 static void *kprobe_seq_start(struct seq_file *f, loff_t *pos)
2850 {
2851 	return (*pos < KPROBE_TABLE_SIZE) ? pos : NULL;
2852 }
2853 
2854 static void *kprobe_seq_next(struct seq_file *f, void *v, loff_t *pos)
2855 {
2856 	(*pos)++;
2857 	if (*pos >= KPROBE_TABLE_SIZE)
2858 		return NULL;
2859 	return pos;
2860 }
2861 
2862 static void kprobe_seq_stop(struct seq_file *f, void *v)
2863 {
2864 	/* Nothing to do */
2865 }
2866 
2867 static int show_kprobe_addr(struct seq_file *pi, void *v)
2868 {
2869 	struct hlist_head *head;
2870 	struct kprobe *p, *kp;
2871 	const char *sym;
2872 	unsigned int i = *(loff_t *) v;
2873 	unsigned long offset = 0;
2874 	char *modname, namebuf[KSYM_NAME_LEN];
2875 
2876 	head = &kprobe_table[i];
2877 	preempt_disable();
2878 	hlist_for_each_entry_rcu(p, head, hlist) {
2879 		sym = kallsyms_lookup((unsigned long)p->addr, NULL,
2880 					&offset, &modname, namebuf);
2881 		if (kprobe_aggrprobe(p)) {
2882 			list_for_each_entry_rcu(kp, &p->list, list)
2883 				report_probe(pi, kp, sym, offset, modname, p);
2884 		} else
2885 			report_probe(pi, p, sym, offset, modname, NULL);
2886 	}
2887 	preempt_enable();
2888 	return 0;
2889 }
2890 
2891 static const struct seq_operations kprobes_sops = {
2892 	.start = kprobe_seq_start,
2893 	.next  = kprobe_seq_next,
2894 	.stop  = kprobe_seq_stop,
2895 	.show  = show_kprobe_addr
2896 };
2897 
2898 DEFINE_SEQ_ATTRIBUTE(kprobes);
2899 
2900 /* kprobes/blacklist -- shows which functions can not be probed */
2901 static void *kprobe_blacklist_seq_start(struct seq_file *m, loff_t *pos)
2902 {
2903 	mutex_lock(&kprobe_mutex);
2904 	return seq_list_start(&kprobe_blacklist, *pos);
2905 }
2906 
2907 static void *kprobe_blacklist_seq_next(struct seq_file *m, void *v, loff_t *pos)
2908 {
2909 	return seq_list_next(v, &kprobe_blacklist, pos);
2910 }
2911 
2912 static int kprobe_blacklist_seq_show(struct seq_file *m, void *v)
2913 {
2914 	struct kprobe_blacklist_entry *ent =
2915 		list_entry(v, struct kprobe_blacklist_entry, list);
2916 
2917 	/*
2918 	 * If '/proc/kallsyms' is not showing kernel address, we won't
2919 	 * show them here either.
2920 	 */
2921 	if (!kallsyms_show_value(m->file->f_cred))
2922 		seq_printf(m, "0x%px-0x%px\t%ps\n", NULL, NULL,
2923 			   (void *)ent->start_addr);
2924 	else
2925 		seq_printf(m, "0x%px-0x%px\t%ps\n", (void *)ent->start_addr,
2926 			   (void *)ent->end_addr, (void *)ent->start_addr);
2927 	return 0;
2928 }
2929 
2930 static void kprobe_blacklist_seq_stop(struct seq_file *f, void *v)
2931 {
2932 	mutex_unlock(&kprobe_mutex);
2933 }
2934 
2935 static const struct seq_operations kprobe_blacklist_sops = {
2936 	.start = kprobe_blacklist_seq_start,
2937 	.next  = kprobe_blacklist_seq_next,
2938 	.stop  = kprobe_blacklist_seq_stop,
2939 	.show  = kprobe_blacklist_seq_show,
2940 };
2941 DEFINE_SEQ_ATTRIBUTE(kprobe_blacklist);
2942 
2943 static int arm_all_kprobes(void)
2944 {
2945 	struct hlist_head *head;
2946 	struct kprobe *p;
2947 	unsigned int i, total = 0, errors = 0;
2948 	int err, ret = 0;
2949 
2950 	guard(mutex)(&kprobe_mutex);
2951 
2952 	/* If kprobes are armed, just return */
2953 	if (!kprobes_all_disarmed)
2954 		return 0;
2955 
2956 	/*
2957 	 * optimize_kprobe() called by arm_kprobe() checks
2958 	 * kprobes_all_disarmed, so set kprobes_all_disarmed before
2959 	 * arm_kprobe.
2960 	 */
2961 	kprobes_all_disarmed = false;
2962 	/* Arming kprobes doesn't optimize kprobe itself */
2963 	for (i = 0; i < KPROBE_TABLE_SIZE; i++) {
2964 		head = &kprobe_table[i];
2965 		/* Arm all kprobes on a best-effort basis */
2966 		hlist_for_each_entry(p, head, hlist) {
2967 			if (!kprobe_disabled(p)) {
2968 				err = arm_kprobe(p);
2969 				if (err)  {
2970 					errors++;
2971 					ret = err;
2972 				}
2973 				total++;
2974 			}
2975 		}
2976 	}
2977 
2978 	if (errors)
2979 		pr_warn("Kprobes globally enabled, but failed to enable %d out of %d probes. Please check which kprobes are kept disabled via debugfs.\n",
2980 			errors, total);
2981 	else
2982 		pr_info("Kprobes globally enabled\n");
2983 
2984 	return ret;
2985 }
2986 
2987 static int disarm_all_kprobes(void)
2988 {
2989 	struct hlist_head *head;
2990 	struct kprobe *p;
2991 	unsigned int i, total = 0, errors = 0;
2992 	int err, ret = 0;
2993 
2994 	guard(mutex)(&kprobe_mutex);
2995 
2996 	/* If kprobes are already disarmed, just return */
2997 	if (kprobes_all_disarmed)
2998 		return 0;
2999 
3000 	kprobes_all_disarmed = true;
3001 
3002 	for (i = 0; i < KPROBE_TABLE_SIZE; i++) {
3003 		head = &kprobe_table[i];
3004 		/* Disarm all kprobes on a best-effort basis */
3005 		hlist_for_each_entry(p, head, hlist) {
3006 			if (!arch_trampoline_kprobe(p) && !kprobe_disabled(p)) {
3007 				err = disarm_kprobe(p, false);
3008 				if (err) {
3009 					errors++;
3010 					ret = err;
3011 				}
3012 				total++;
3013 			}
3014 		}
3015 	}
3016 
3017 	if (errors)
3018 		pr_warn("Kprobes globally disabled, but failed to disable %d out of %d probes. Please check which kprobes are kept enabled via debugfs.\n",
3019 			errors, total);
3020 	else
3021 		pr_info("Kprobes globally disabled\n");
3022 
3023 	/* Wait for disarming all kprobes by optimizer */
3024 	wait_for_kprobe_optimizer_locked();
3025 	return ret;
3026 }
3027 
3028 /*
3029  * XXX: The debugfs bool file interface doesn't allow for callbacks
3030  * when the bool state is switched. We can reuse that facility when
3031  * available
3032  */
3033 static ssize_t read_enabled_file_bool(struct file *file,
3034 	       char __user *user_buf, size_t count, loff_t *ppos)
3035 {
3036 	char buf[3];
3037 
3038 	if (!kprobes_all_disarmed)
3039 		buf[0] = '1';
3040 	else
3041 		buf[0] = '0';
3042 	buf[1] = '\n';
3043 	buf[2] = 0x00;
3044 	return simple_read_from_buffer(user_buf, count, ppos, buf, 2);
3045 }
3046 
3047 static ssize_t write_enabled_file_bool(struct file *file,
3048 	       const char __user *user_buf, size_t count, loff_t *ppos)
3049 {
3050 	bool enable;
3051 	int ret;
3052 
3053 	ret = kstrtobool_from_user(user_buf, count, &enable);
3054 	if (ret)
3055 		return ret;
3056 
3057 	ret = enable ? arm_all_kprobes() : disarm_all_kprobes();
3058 	if (ret)
3059 		return ret;
3060 
3061 	return count;
3062 }
3063 
3064 static const struct file_operations fops_kp = {
3065 	.read =         read_enabled_file_bool,
3066 	.write =        write_enabled_file_bool,
3067 	.llseek =	default_llseek,
3068 };
3069 
3070 static int __init debugfs_kprobe_init(void)
3071 {
3072 	struct dentry *dir;
3073 
3074 	dir = debugfs_create_dir("kprobes", NULL);
3075 
3076 	debugfs_create_file("list", 0400, dir, NULL, &kprobes_fops);
3077 
3078 	debugfs_create_file("enabled", 0600, dir, NULL, &fops_kp);
3079 
3080 	debugfs_create_file("blacklist", 0400, dir, NULL,
3081 			    &kprobe_blacklist_fops);
3082 
3083 	return 0;
3084 }
3085 
3086 late_initcall(debugfs_kprobe_init);
3087 #endif /* CONFIG_DEBUG_FS */
3088