xref: /linux/kernel/trace/fprobe.c (revision 1d653a183973f5283a3db5a38cd5e195eb152244)
1 // SPDX-License-Identifier: GPL-2.0
2 /*
3  * fprobe - Simple ftrace probe wrapper for function entry.
4  */
5 #define pr_fmt(fmt) "fprobe: " fmt
6 
7 #include <linux/cleanup.h>
8 #include <linux/err.h>
9 #include <linux/fprobe.h>
10 #include <linux/kallsyms.h>
11 #include <linux/kprobes.h>
12 #include <linux/list.h>
13 #include <linux/mutex.h>
14 #include <linux/rhashtable.h>
15 #include <linux/slab.h>
16 #include <linux/sort.h>
17 
18 #include <asm/fprobe.h>
19 
20 #include "trace.h"
21 
22 #define FPROBE_IP_HASH_BITS 8
23 #define FPROBE_IP_TABLE_SIZE (1 << FPROBE_IP_HASH_BITS)
24 
25 #define FPROBE_HASH_BITS 6
26 #define FPROBE_TABLE_SIZE (1 << FPROBE_HASH_BITS)
27 
28 #define SIZE_IN_LONG(x) ((x + sizeof(long) - 1) >> (sizeof(long) == 8 ? 3 : 2))
29 
30 /*
31  * fprobe_table: hold 'fprobe_hlist::hlist' for checking the fprobe still
32  *   exists. The key is the address of fprobe instance.
33  * fprobe_ip_table: hold 'fprobe_hlist::array[*]' for searching the fprobe
34  *   instance related to the function address. The key is the ftrace IP
35  *   address.
36  *
37  * When unregistering the fprobe, fprobe_hlist::fp and fprobe_hlist::array[*].fp
38  * are set NULL and delete those from both hash tables (by hlist_del_rcu).
39  * After an RCU grace period, the fprobe_hlist itself will be released.
40  *
41  * fprobe_table and fprobe_ip_table can be accessed from either
42  *  - Normal hlist traversal and RCU add/del under 'fprobe_mutex' is held.
43  *  - RCU hlist traversal under disabling preempt
44  */
45 static struct hlist_head fprobe_table[FPROBE_TABLE_SIZE];
46 static struct rhltable fprobe_ip_table;
47 static DEFINE_MUTEX(fprobe_mutex);
48 static struct fgraph_ops fprobe_graph_ops;
49 
50 static u32 fprobe_node_hashfn(const void *data, u32 len, u32 seed)
51 {
52 	return hash_ptr(*(unsigned long **)data, 32);
53 }
54 
55 static int fprobe_node_cmp(struct rhashtable_compare_arg *arg,
56 			   const void *ptr)
57 {
58 	unsigned long key = *(unsigned long *)arg->key;
59 	const struct fprobe_hlist_node *n = ptr;
60 
61 	return n->addr != key;
62 }
63 
64 static u32 fprobe_node_obj_hashfn(const void *data, u32 len, u32 seed)
65 {
66 	const struct fprobe_hlist_node *n = data;
67 
68 	return hash_ptr((void *)n->addr, 32);
69 }
70 
71 static const struct rhashtable_params fprobe_rht_params = {
72 	.head_offset		= offsetof(struct fprobe_hlist_node, hlist),
73 	.key_offset		= offsetof(struct fprobe_hlist_node, addr),
74 	.key_len		= sizeof_field(struct fprobe_hlist_node, addr),
75 	.hashfn			= fprobe_node_hashfn,
76 	.obj_hashfn		= fprobe_node_obj_hashfn,
77 	.obj_cmpfn		= fprobe_node_cmp,
78 	.automatic_shrinking	= true,
79 };
80 
81 /* Node insertion and deletion requires the fprobe_mutex */
82 static int __insert_fprobe_node(struct fprobe_hlist_node *node, struct fprobe *fp)
83 {
84 	int ret;
85 
86 	lockdep_assert_held(&fprobe_mutex);
87 
88 	ret = rhltable_insert(&fprobe_ip_table, &node->hlist, fprobe_rht_params);
89 	/* Set the fprobe pointer if insertion was successful. */
90 	if (!ret)
91 		WRITE_ONCE(node->fp, fp);
92 	return ret;
93 }
94 
95 static void __delete_fprobe_node(struct fprobe_hlist_node *node)
96 {
97 	lockdep_assert_held(&fprobe_mutex);
98 
99 	/* Avoid double deleting and non-inserted nodes */
100 	if (READ_ONCE(node->fp) != NULL) {
101 		WRITE_ONCE(node->fp, NULL);
102 		rhltable_remove(&fprobe_ip_table, &node->hlist,
103 				fprobe_rht_params);
104 	}
105 }
106 
107 /* Check existence of the fprobe */
108 static bool fprobe_registered(struct fprobe *fp)
109 {
110 	struct hlist_head *head;
111 	struct fprobe_hlist *fph;
112 
113 	head = &fprobe_table[hash_ptr(fp, FPROBE_HASH_BITS)];
114 	hlist_for_each_entry_rcu(fph, head, hlist,
115 				 lockdep_is_held(&fprobe_mutex)) {
116 		if (fph->fp == fp)
117 			return true;
118 	}
119 	return false;
120 }
121 NOKPROBE_SYMBOL(fprobe_registered);
122 
123 static int add_fprobe_hash(struct fprobe *fp)
124 {
125 	struct fprobe_hlist *fph = fp->hlist_array;
126 	struct hlist_head *head;
127 
128 	lockdep_assert_held(&fprobe_mutex);
129 
130 	if (WARN_ON_ONCE(!fph))
131 		return -EINVAL;
132 
133 	head = &fprobe_table[hash_ptr(fp, FPROBE_HASH_BITS)];
134 	hlist_add_head_rcu(&fp->hlist_array->hlist, head);
135 	return 0;
136 }
137 
138 static int del_fprobe_hash(struct fprobe *fp)
139 {
140 	struct fprobe_hlist *fph = fp->hlist_array;
141 
142 	lockdep_assert_held(&fprobe_mutex);
143 
144 	if (WARN_ON_ONCE(!fph))
145 		return -EINVAL;
146 
147 	if (!fprobe_registered(fp))
148 		return -ENOENT;
149 
150 	fph->fp = NULL;
151 	hlist_del_rcu(&fph->hlist);
152 	return 0;
153 }
154 
155 #ifdef ARCH_DEFINE_ENCODE_FPROBE_HEADER
156 
157 /* The arch should encode fprobe_header info into one unsigned long */
158 #define FPROBE_HEADER_SIZE_IN_LONG	1
159 
160 static inline bool write_fprobe_header(unsigned long *stack,
161 					struct fprobe *fp, unsigned int size_words)
162 {
163 	if (WARN_ON_ONCE(size_words > MAX_FPROBE_DATA_SIZE_WORD ||
164 			 !arch_fprobe_header_encodable(fp)))
165 		return false;
166 
167 	*stack = arch_encode_fprobe_header(fp, size_words);
168 	return true;
169 }
170 
171 static inline void read_fprobe_header(unsigned long *stack,
172 					struct fprobe **fp, unsigned int *size_words)
173 {
174 	if (!*stack) {
175 		*fp = NULL;
176 		*size_words = 0;
177 		return;
178 	}
179 	*fp = arch_decode_fprobe_header_fp(*stack);
180 	*size_words = arch_decode_fprobe_header_size(*stack);
181 }
182 
183 #else
184 
185 /* Generic fprobe_header */
186 struct __fprobe_header {
187 	struct fprobe *fp;
188 	unsigned long size_words;
189 };
190 
191 #define FPROBE_HEADER_SIZE_IN_LONG	SIZE_IN_LONG(sizeof(struct __fprobe_header))
192 
193 static inline bool write_fprobe_header(unsigned long *stack,
194 					struct fprobe *fp, unsigned int size_words)
195 {
196 	struct __fprobe_header *fph = (struct __fprobe_header *)stack;
197 
198 	if (WARN_ON_ONCE(size_words > MAX_FPROBE_DATA_SIZE_WORD))
199 		return false;
200 
201 	fph->fp = fp;
202 	fph->size_words = size_words;
203 	return true;
204 }
205 
206 static inline void read_fprobe_header(unsigned long *stack,
207 					struct fprobe **fp, unsigned int *size_words)
208 {
209 	struct __fprobe_header *fph = (struct __fprobe_header *)stack;
210 
211 	if (!*stack) {
212 		*fp = NULL;
213 		*size_words = 0;
214 		return;
215 	}
216 
217 	*fp = fph->fp;
218 	*size_words = fph->size_words;
219 }
220 
221 #endif
222 
223 /*
224  * fprobe shadow stack management:
225  * Since fprobe shares a single fgraph_ops, it needs to share the stack entry
226  * among the probes on the same function exit. Note that a new probe can be
227  * registered before a target function is returning, we can not use the hash
228  * table to find the corresponding probes. Thus the probe address is stored on
229  * the shadow stack with its entry data size.
230  *
231  */
232 static inline int __fprobe_handler(unsigned long ip, unsigned long parent_ip,
233 				   struct fprobe *fp, struct ftrace_regs *fregs,
234 				   void *data)
235 {
236 	if (!fp->entry_handler)
237 		return 0;
238 
239 	return fp->entry_handler(fp, ip, parent_ip, fregs, data);
240 }
241 
242 static inline int __fprobe_kprobe_handler(unsigned long ip, unsigned long parent_ip,
243 					  struct fprobe *fp, struct ftrace_regs *fregs,
244 					  void *data)
245 {
246 	int ret;
247 	/*
248 	 * This user handler is shared with other kprobes and is not expected to be
249 	 * called recursively. So if any other kprobe handler is running, this will
250 	 * exit as kprobe does. See the section 'Share the callbacks with kprobes'
251 	 * in Documentation/trace/fprobe.rst for more information.
252 	 */
253 	if (unlikely(kprobe_running())) {
254 		fp->nmissed++;
255 		return 0;
256 	}
257 
258 	kprobe_busy_begin();
259 	ret = __fprobe_handler(ip, parent_ip, fp, fregs, data);
260 	kprobe_busy_end();
261 	return ret;
262 }
263 
264 static int fprobe_fgraph_entry(struct ftrace_graph_ent *trace, struct fgraph_ops *gops,
265 			       struct ftrace_regs *fregs);
266 static void fprobe_return(struct ftrace_graph_ret *trace,
267 			  struct fgraph_ops *gops,
268 			  struct ftrace_regs *fregs);
269 
270 static struct fgraph_ops fprobe_graph_ops = {
271 	.entryfunc	= fprobe_fgraph_entry,
272 	.retfunc	= fprobe_return,
273 };
274 /* Number of fgraph fprobe nodes */
275 static int nr_fgraph_fprobes;
276 /* Is fprobe_graph_ops registered? */
277 static bool fprobe_graph_registered;
278 
279 /* Add @addrs to the ftrace filter and register fgraph if needed. */
280 static int fprobe_graph_add_ips(unsigned long *addrs, int num)
281 {
282 	int ret;
283 
284 	lockdep_assert_held(&fprobe_mutex);
285 
286 	ret = ftrace_set_filter_ips(&fprobe_graph_ops.ops, addrs, num, 0, 0);
287 	if (ret)
288 		return ret;
289 
290 	if (!fprobe_graph_registered) {
291 		ret = register_ftrace_graph(&fprobe_graph_ops);
292 		if (WARN_ON_ONCE(ret)) {
293 			ftrace_free_filter(&fprobe_graph_ops.ops);
294 			return ret;
295 		}
296 		fprobe_graph_registered = true;
297 	}
298 	return 0;
299 }
300 
301 static void __fprobe_graph_unregister(void)
302 {
303 	if (fprobe_graph_registered) {
304 		unregister_ftrace_graph(&fprobe_graph_ops);
305 		ftrace_free_filter(&fprobe_graph_ops.ops);
306 		fprobe_graph_registered = false;
307 	}
308 }
309 
310 /* Remove @addrs from the ftrace filter and unregister fgraph if possible. */
311 static void fprobe_graph_remove_ips(unsigned long *addrs, int num)
312 {
313 	lockdep_assert_held(&fprobe_mutex);
314 
315 	if (!nr_fgraph_fprobes)
316 		__fprobe_graph_unregister();
317 	else if (num)
318 		ftrace_set_filter_ips(&fprobe_graph_ops.ops, addrs, num, 1, 0);
319 }
320 
321 #if defined(CONFIG_DYNAMIC_FTRACE_WITH_ARGS) || defined(CONFIG_DYNAMIC_FTRACE_WITH_REGS)
322 
323 /* ftrace_ops callback, this processes fprobes which have only entry_handler. */
324 static void fprobe_ftrace_entry(unsigned long ip, unsigned long parent_ip,
325 	struct ftrace_ops *ops, struct ftrace_regs *fregs)
326 {
327 	struct fprobe_hlist_node *node;
328 	struct rhlist_head *head, *pos;
329 	struct fprobe *fp;
330 	int bit;
331 
332 	bit = ftrace_test_recursion_trylock(ip, parent_ip);
333 	if (bit < 0)
334 		return;
335 
336 	/*
337 	 * ftrace_test_recursion_trylock() disables preemption, but
338 	 * rhltable_lookup() checks whether rcu_read_lcok is held.
339 	 * So we take rcu_read_lock() here.
340 	 */
341 	rcu_read_lock();
342 	head = rhltable_lookup(&fprobe_ip_table, &ip, fprobe_rht_params);
343 
344 	rhl_for_each_entry_rcu(node, pos, head, hlist) {
345 		if (node->addr != ip)
346 			break;
347 		fp = READ_ONCE(node->fp);
348 		if (unlikely(!fp || fprobe_disabled(fp) || fp->exit_handler))
349 			continue;
350 
351 		if (fprobe_shared_with_kprobes(fp))
352 			__fprobe_kprobe_handler(ip, parent_ip, fp, fregs, NULL);
353 		else
354 			__fprobe_handler(ip, parent_ip, fp, fregs, NULL);
355 	}
356 	rcu_read_unlock();
357 	ftrace_test_recursion_unlock(bit);
358 }
359 NOKPROBE_SYMBOL(fprobe_ftrace_entry);
360 
361 static struct ftrace_ops fprobe_ftrace_ops = {
362 	.func	= fprobe_ftrace_entry,
363 	.flags	= FTRACE_OPS_FL_SAVE_ARGS,
364 };
365 /* Number of ftrace fprobe nodes */
366 static int nr_ftrace_fprobes;
367 /* Is fprobe_ftrace_ops registered? */
368 static bool fprobe_ftrace_registered;
369 
370 static int fprobe_ftrace_add_ips(unsigned long *addrs, int num)
371 {
372 	int ret;
373 
374 	lockdep_assert_held(&fprobe_mutex);
375 
376 	ret = ftrace_set_filter_ips(&fprobe_ftrace_ops, addrs, num, 0, 0);
377 	if (ret)
378 		return ret;
379 
380 	if (!fprobe_ftrace_registered) {
381 		ret = register_ftrace_function(&fprobe_ftrace_ops);
382 		if (ret) {
383 			ftrace_free_filter(&fprobe_ftrace_ops);
384 			return ret;
385 		}
386 		fprobe_ftrace_registered = true;
387 	}
388 	return 0;
389 }
390 
391 static void __fprobe_ftrace_unregister(void)
392 {
393 	if (fprobe_ftrace_registered) {
394 		unregister_ftrace_function(&fprobe_ftrace_ops);
395 		ftrace_free_filter(&fprobe_ftrace_ops);
396 		fprobe_ftrace_registered = false;
397 	}
398 }
399 
400 static void fprobe_ftrace_remove_ips(unsigned long *addrs, int num)
401 {
402 	lockdep_assert_held(&fprobe_mutex);
403 
404 	if (!nr_ftrace_fprobes)
405 		__fprobe_ftrace_unregister();
406 	else if (num)
407 		ftrace_set_filter_ips(&fprobe_ftrace_ops, addrs, num, 1, 0);
408 }
409 
410 static bool fprobe_is_ftrace(struct fprobe *fp)
411 {
412 	return !fp->exit_handler;
413 }
414 
415 /* Node insertion and deletion requires the fprobe_mutex */
416 static int insert_fprobe_node(struct fprobe_hlist_node *node, struct fprobe *fp)
417 {
418 	int ret;
419 
420 	lockdep_assert_held(&fprobe_mutex);
421 
422 	ret = __insert_fprobe_node(node, fp);
423 	if (!ret) {
424 		if (fprobe_is_ftrace(fp))
425 			nr_ftrace_fprobes++;
426 		else
427 			nr_fgraph_fprobes++;
428 	}
429 
430 	return ret;
431 }
432 
433 static void delete_fprobe_node(struct fprobe_hlist_node *node)
434 {
435 	struct fprobe *fp;
436 
437 	lockdep_assert_held(&fprobe_mutex);
438 
439 	fp = READ_ONCE(node->fp);
440 	if (fp) {
441 		if (fprobe_is_ftrace(fp))
442 			nr_ftrace_fprobes--;
443 		else
444 			nr_fgraph_fprobes--;
445 	}
446 	__delete_fprobe_node(node);
447 }
448 
449 static bool fprobe_exists_on_hash(unsigned long ip, bool ftrace)
450 {
451 	struct rhlist_head *head, *pos;
452 	struct fprobe_hlist_node *node;
453 	struct fprobe *fp;
454 
455 	guard(rcu)();
456 	head = rhltable_lookup(&fprobe_ip_table, &ip,
457 				fprobe_rht_params);
458 	if (!head)
459 		return false;
460 	/* We have to check the same type on the list. */
461 	rhl_for_each_entry_rcu(node, pos, head, hlist) {
462 		if (node->addr != ip)
463 			break;
464 		fp = READ_ONCE(node->fp);
465 		if (likely(fp)) {
466 			if ((!ftrace && fp->exit_handler) ||
467 			    (ftrace && !fp->exit_handler))
468 				return true;
469 		}
470 	}
471 
472 	return false;
473 }
474 
475 #ifdef CONFIG_MODULES
476 static void fprobe_remove_ips(unsigned long *ips, unsigned int cnt)
477 {
478 	fprobe_graph_remove_ips(ips, cnt);
479 	fprobe_ftrace_remove_ips(ips, cnt);
480 }
481 #endif
482 #else
483 static int fprobe_ftrace_add_ips(unsigned long *addrs, int num)
484 {
485 	return -ENOENT;
486 }
487 
488 static void fprobe_ftrace_remove_ips(unsigned long *addrs, int num)
489 {
490 }
491 
492 static bool fprobe_is_ftrace(struct fprobe *fp)
493 {
494 	return false;
495 }
496 
497 /* Node insertion and deletion requires the fprobe_mutex */
498 static int insert_fprobe_node(struct fprobe_hlist_node *node, struct fprobe *fp)
499 {
500 	int ret;
501 
502 	lockdep_assert_held(&fprobe_mutex);
503 
504 	ret = __insert_fprobe_node(node, fp);
505 	if (!ret)
506 		nr_fgraph_fprobes++;
507 
508 	return ret;
509 }
510 
511 static void delete_fprobe_node(struct fprobe_hlist_node *node)
512 {
513 	struct fprobe *fp;
514 
515 	lockdep_assert_held(&fprobe_mutex);
516 
517 	fp = READ_ONCE(node->fp);
518 	if (fp)
519 		nr_fgraph_fprobes--;
520 	__delete_fprobe_node(node);
521 }
522 
523 static bool fprobe_exists_on_hash(unsigned long ip, bool ftrace __maybe_unused)
524 {
525 	struct rhlist_head *head, *pos;
526 	struct fprobe_hlist_node *node;
527 	struct fprobe *fp;
528 
529 	guard(rcu)();
530 	head = rhltable_lookup(&fprobe_ip_table, &ip,
531 				fprobe_rht_params);
532 	if (!head)
533 		return false;
534 	/* We only need to check fp is there. */
535 	rhl_for_each_entry_rcu(node, pos, head, hlist) {
536 		if (node->addr != ip)
537 			break;
538 		fp = READ_ONCE(node->fp);
539 		if (likely(fp))
540 			return true;
541 	}
542 
543 	return false;
544 }
545 
546 #ifdef CONFIG_MODULES
547 static void fprobe_remove_ips(unsigned long *ips, unsigned int cnt)
548 {
549 	if (!nr_fgraph_fprobes)
550 		__fprobe_graph_unregister();
551 	else if (cnt)
552 		ftrace_set_filter_ips(&fprobe_graph_ops.ops, ips, cnt, 1, 0);
553 }
554 #endif
555 #endif /* !CONFIG_DYNAMIC_FTRACE_WITH_ARGS && !CONFIG_DYNAMIC_FTRACE_WITH_REGS */
556 
557 /* fgraph_ops callback, this processes fprobes which have exit_handler. */
558 static int fprobe_fgraph_entry(struct ftrace_graph_ent *trace, struct fgraph_ops *gops,
559 			       struct ftrace_regs *fregs)
560 {
561 	unsigned long *fgraph_data = NULL;
562 	unsigned long func = trace->func;
563 	struct fprobe_hlist_node *node;
564 	struct rhlist_head *head, *pos;
565 	unsigned long ret_ip;
566 	int reserved_words;
567 	struct fprobe *fp;
568 	int used, ret;
569 
570 	if (WARN_ON_ONCE(!fregs))
571 		return 0;
572 
573 	guard(rcu)();
574 	head = rhltable_lookup(&fprobe_ip_table, &func, fprobe_rht_params);
575 	reserved_words = 0;
576 	rhl_for_each_entry_rcu(node, pos, head, hlist) {
577 		if (node->addr != func)
578 			continue;
579 		fp = READ_ONCE(node->fp);
580 		if (!fp || !fp->exit_handler)
581 			continue;
582 		/*
583 		 * Since fprobe can be enabled until the next loop, we ignore the
584 		 * fprobe's disabled flag in this loop.
585 		 */
586 		reserved_words +=
587 			FPROBE_HEADER_SIZE_IN_LONG + SIZE_IN_LONG(fp->entry_data_size);
588 	}
589 	if (reserved_words) {
590 		fgraph_data = fgraph_reserve_data(gops->idx, reserved_words * sizeof(long));
591 		if (unlikely(!fgraph_data)) {
592 			rhl_for_each_entry_rcu(node, pos, head, hlist) {
593 				if (node->addr != func)
594 					continue;
595 				fp = READ_ONCE(node->fp);
596 				if (fp && !fprobe_disabled(fp) && !fprobe_is_ftrace(fp))
597 					fp->nmissed++;
598 			}
599 			return 0;
600 		}
601 	}
602 
603 	/*
604 	 * TODO: recursion detection has been done in the fgraph. Thus we need
605 	 * to add a callback to increment missed counter.
606 	 */
607 	ret_ip = ftrace_regs_get_return_address(fregs);
608 	used = 0;
609 	rhl_for_each_entry_rcu(node, pos, head, hlist) {
610 		int data_size;
611 		void *data;
612 
613 		if (node->addr != func)
614 			continue;
615 		fp = READ_ONCE(node->fp);
616 		if (unlikely(!fp || fprobe_disabled(fp) || fprobe_is_ftrace(fp)))
617 			continue;
618 
619 		data_size = fp->entry_data_size;
620 		/*
621 		 * The list may have grown since it was sized, so this node
622 		 * may not fit. Skip it as missed rather than overrun the
623 		 * reservation.
624 		 */
625 		if (fp->exit_handler &&
626 		    used + FPROBE_HEADER_SIZE_IN_LONG + SIZE_IN_LONG(data_size) > reserved_words) {
627 			fp->nmissed++;
628 			continue;
629 		}
630 		if (data_size && fp->exit_handler)
631 			data = fgraph_data + used + FPROBE_HEADER_SIZE_IN_LONG;
632 		else
633 			data = NULL;
634 
635 		if (fprobe_shared_with_kprobes(fp))
636 			ret = __fprobe_kprobe_handler(func, ret_ip, fp, fregs, data);
637 		else
638 			ret = __fprobe_handler(func, ret_ip, fp, fregs, data);
639 
640 		/* If entry_handler returns !0, nmissed is not counted but skips exit_handler. */
641 		if (!ret && fp->exit_handler) {
642 			int size_words = SIZE_IN_LONG(data_size);
643 
644 			if (write_fprobe_header(&fgraph_data[used], fp, size_words))
645 				used += FPROBE_HEADER_SIZE_IN_LONG + size_words;
646 		}
647 	}
648 
649 	/* Terminate the list, fgraph_reserve_data() does not clear it. */
650 	if (used && used < reserved_words)
651 		fgraph_data[used] = 0;
652 
653 	/* If any exit_handler is set, data must be used. */
654 	return used != 0;
655 }
656 NOKPROBE_SYMBOL(fprobe_fgraph_entry);
657 
658 static void fprobe_return(struct ftrace_graph_ret *trace,
659 			  struct fgraph_ops *gops,
660 			  struct ftrace_regs *fregs)
661 {
662 	unsigned long *fgraph_data = NULL;
663 	unsigned long ret_ip;
664 	struct fprobe *fp;
665 	int size, curr;
666 	int size_words;
667 
668 	fgraph_data = (unsigned long *)fgraph_retrieve_data(gops->idx, &size);
669 	if (WARN_ON_ONCE(!fgraph_data))
670 		return;
671 	size_words = SIZE_IN_LONG(size);
672 	ret_ip = ftrace_regs_get_instruction_pointer(fregs);
673 
674 	preempt_disable_notrace();
675 
676 	curr = 0;
677 	while (size_words > curr) {
678 		read_fprobe_header(&fgraph_data[curr], &fp, &size);
679 		if (!fp)
680 			break;
681 		curr += FPROBE_HEADER_SIZE_IN_LONG;
682 		if (fprobe_registered(fp) && !fprobe_disabled(fp)) {
683 			if (WARN_ON_ONCE(curr + size > size_words))
684 				break;
685 			fp->exit_handler(fp, trace->func, ret_ip, fregs,
686 					 size ? fgraph_data + curr : NULL);
687 		}
688 		curr += size;
689 	}
690 	preempt_enable_notrace();
691 }
692 NOKPROBE_SYMBOL(fprobe_return);
693 
694 #ifdef CONFIG_MODULES
695 
696 #define FPROBE_IPS_BATCH_INIT 128
697 /* instruction pointer address list */
698 struct fprobe_addr_list {
699 	int index;
700 	int size;
701 	unsigned long *addrs;
702 };
703 
704 static int fprobe_remove_node_in_module(struct module *mod, struct fprobe_hlist_node *node,
705 					 struct fprobe_addr_list *alist)
706 {
707 	lockdep_assert_in_rcu_read_lock();
708 
709 	if (!within_module(node->addr, mod))
710 		return 0;
711 
712 	delete_fprobe_node(node);
713 	/* If no address list is available, we can't track this address. */
714 	if (!alist->addrs)
715 		return 0;
716 	/*
717 	 * Don't care the type here, because all fprobes on the same
718 	 * address must be removed eventually.
719 	 */
720 	if (!rhltable_lookup(&fprobe_ip_table, &node->addr, fprobe_rht_params)) {
721 		alist->addrs[alist->index++] = node->addr;
722 		if (alist->index == alist->size)
723 			return -ENOSPC;
724 	}
725 
726 	return 0;
727 }
728 
729 /* Handle module unloading to manage fprobe_ip_table. */
730 static int fprobe_module_callback(struct notifier_block *nb,
731 				  unsigned long val, void *data)
732 {
733 	struct fprobe_addr_list alist = {.size = FPROBE_IPS_BATCH_INIT};
734 	struct fprobe_hlist_node *node;
735 	struct rhashtable_iter iter;
736 	struct module *mod = data;
737 	bool retry;
738 
739 	if (val != MODULE_STATE_GOING)
740 		return NOTIFY_DONE;
741 
742 	alist.addrs = kcalloc(alist.size, sizeof(*alist.addrs), GFP_KERNEL);
743 	/*
744 	 * If failed to alloc memory, ftrace_ops will not be able to remove ips from
745 	 * hash, but we can still remove nodes from fprobe_ip_table, so we can avoid
746 	 * the potential wrong callback. So just print a warning here and try to
747 	 * continue without address list.
748 	 */
749 	WARN_ONCE(!alist.addrs,
750 		"Failed to allocate memory for fprobe_addr_list, ftrace_ops will not be updated");
751 
752 	mutex_lock(&fprobe_mutex);
753 again:
754 	retry = false;
755 	alist.index = 0;
756 	rhltable_walk_enter(&fprobe_ip_table, &iter);
757 	do {
758 		rhashtable_walk_start(&iter);
759 
760 		while ((node = rhashtable_walk_next(&iter)) && !IS_ERR(node))
761 			if (fprobe_remove_node_in_module(mod, node, &alist) < 0) {
762 				retry = true;
763 				break;
764 			}
765 
766 		rhashtable_walk_stop(&iter);
767 	} while (node == ERR_PTR(-EAGAIN) && !retry);
768 	rhashtable_walk_exit(&iter);
769 	/* Remove any ips from hash table(s) */
770 	fprobe_remove_ips(alist.addrs, alist.index);
771 	/*
772 	 * If we break rhashtable walk loop except for -EAGAIN, we need
773 	 * to restart looping from start for safety. Anyway, this is
774 	 * not a hotpath.
775 	 */
776 	if (retry)
777 		goto again;
778 
779 	mutex_unlock(&fprobe_mutex);
780 
781 	kfree(alist.addrs);
782 
783 	return NOTIFY_DONE;
784 }
785 
786 static struct notifier_block fprobe_module_nb = {
787 	.notifier_call = fprobe_module_callback,
788 	.priority = 0,
789 };
790 
791 static int __init init_fprobe_module(void)
792 {
793 	return register_module_notifier(&fprobe_module_nb);
794 }
795 early_initcall(init_fprobe_module);
796 #endif
797 
798 static int symbols_cmp(const void *a, const void *b)
799 {
800 	const char **str_a = (const char **) a;
801 	const char **str_b = (const char **) b;
802 
803 	return strcmp(*str_a, *str_b);
804 }
805 
806 /* Convert ftrace location address from symbols */
807 static unsigned long *get_ftrace_locations(const char **syms, int num)
808 {
809 	unsigned long *addrs;
810 
811 	/* Convert symbols to symbol address */
812 	addrs = kcalloc(num, sizeof(*addrs), GFP_KERNEL);
813 	if (!addrs)
814 		return ERR_PTR(-ENOMEM);
815 
816 	/* ftrace_lookup_symbols expects sorted symbols */
817 	sort(syms, num, sizeof(*syms), symbols_cmp, NULL);
818 
819 	if (!ftrace_lookup_symbols(syms, num, addrs))
820 		return addrs;
821 
822 	kfree(addrs);
823 	return ERR_PTR(-ENOENT);
824 }
825 
826 struct filter_match_data {
827 	const char *filter;
828 	const char *notfilter;
829 	size_t index;
830 	size_t size;
831 	unsigned long *addrs;
832 	struct module **mods;
833 };
834 
835 static int filter_match_callback(void *data, const char *name, unsigned long addr)
836 {
837 	struct filter_match_data *match = data;
838 
839 	if (!glob_match(match->filter, name) ||
840 	    (match->notfilter && glob_match(match->notfilter, name)))
841 		return 0;
842 
843 	if (!ftrace_location(addr))
844 		return 0;
845 
846 	if (match->addrs) {
847 		struct module *mod = __module_text_address(addr);
848 
849 		if (mod && !try_module_get(mod))
850 			return 0;
851 
852 		match->mods[match->index] = mod;
853 		match->addrs[match->index] = addr;
854 	}
855 	match->index++;
856 	return match->index == match->size;
857 }
858 
859 /*
860  * Make IP list from the filter/no-filter glob patterns.
861  * Return the number of matched symbols, or errno.
862  * If @addrs == NULL, this just counts the number of matched symbols. If @addrs
863  * is passed with an array, we need to pass the an @mods array of the same size
864  * to increment the module refcount for each symbol.
865  * This means we also need to call `module_put` for each element of @mods after
866  * using the @addrs.
867  */
868 static int get_ips_from_filter(const char *filter, const char *notfilter,
869 			       unsigned long *addrs, struct module **mods,
870 			       size_t size)
871 {
872 	struct filter_match_data match = { .filter = filter, .notfilter = notfilter,
873 		.index = 0, .size = size, .addrs = addrs, .mods = mods};
874 	int ret;
875 
876 	if (addrs && !mods)
877 		return -EINVAL;
878 
879 	ret = kallsyms_on_each_symbol(filter_match_callback, &match);
880 	if (ret < 0)
881 		return ret;
882 	if (IS_ENABLED(CONFIG_MODULES)) {
883 		ret = module_kallsyms_on_each_symbol(NULL, filter_match_callback, &match);
884 		if (ret < 0)
885 			return ret;
886 	}
887 
888 	return match.index ?: -ENOENT;
889 }
890 
891 static void fprobe_fail_cleanup(struct fprobe *fp)
892 {
893 	kfree(fp->hlist_array);
894 	fp->hlist_array = NULL;
895 }
896 
897 /* Initialize the fprobe data structure. */
898 static int fprobe_init(struct fprobe *fp, unsigned long *addrs, int num)
899 {
900 	struct fprobe_hlist *hlist_array;
901 	unsigned long addr;
902 	int size, i;
903 
904 	if (!fp || !addrs || num <= 0)
905 		return -EINVAL;
906 
907 	size = ALIGN(fp->entry_data_size, sizeof(long));
908 	if (size > MAX_FPROBE_DATA_SIZE)
909 		return -E2BIG;
910 	fp->entry_data_size = size;
911 
912 	hlist_array = kzalloc_flex(*hlist_array, array, num);
913 	if (!hlist_array)
914 		return -ENOMEM;
915 
916 	fp->nmissed = 0;
917 
918 	hlist_array->size = num;
919 	fp->hlist_array = hlist_array;
920 	hlist_array->fp = fp;
921 	for (i = 0; i < num; i++) {
922 		addr = ftrace_location(addrs[i]);
923 		if (!addr) {
924 			fprobe_fail_cleanup(fp);
925 			return -ENOENT;
926 		}
927 		hlist_array->array[i].addr = addr;
928 	}
929 	return 0;
930 }
931 
932 #define FPROBE_IPS_MAX	INT_MAX
933 
934 int fprobe_count_ips_from_filter(const char *filter, const char *notfilter)
935 {
936 	return get_ips_from_filter(filter, notfilter, NULL, NULL, FPROBE_IPS_MAX);
937 }
938 
939 /**
940  * register_fprobe() - Register fprobe to ftrace by pattern.
941  * @fp: A fprobe data structure to be registered.
942  * @filter: A wildcard pattern of probed symbols.
943  * @notfilter: A wildcard pattern of NOT probed symbols.
944  *
945  * Register @fp to ftrace for enabling the probe on the symbols matched to @filter.
946  * If @notfilter is not NULL, the symbols matched the @notfilter are not probed.
947  *
948  * Return 0 if @fp is registered successfully, -errno if not.
949  */
950 int register_fprobe(struct fprobe *fp, const char *filter, const char *notfilter)
951 {
952 	unsigned long *addrs __free(kfree) = NULL;
953 	struct module **mods __free(kfree) = NULL;
954 	int ret, num;
955 
956 	if (!fp || !filter)
957 		return -EINVAL;
958 
959 	num = get_ips_from_filter(filter, notfilter, NULL, NULL, FPROBE_IPS_MAX);
960 	if (num < 0)
961 		return num;
962 
963 	addrs = kzalloc_objs(*addrs, num);
964 	if (!addrs)
965 		return -ENOMEM;
966 
967 	mods = kzalloc_objs(*mods, num);
968 	if (!mods)
969 		return -ENOMEM;
970 
971 	ret = get_ips_from_filter(filter, notfilter, addrs, mods, num);
972 	if (ret >= 0)
973 		ret = register_fprobe_ips(fp, addrs, ret);
974 
975 	for (int i = 0; i < num; i++) {
976 		if (mods[i])
977 			module_put(mods[i]);
978 	}
979 	return ret;
980 }
981 EXPORT_SYMBOL_GPL(register_fprobe);
982 
983 static int unregister_fprobe_nolock(struct fprobe *fp);
984 
985 /**
986  * register_fprobe_ips() - Register fprobe to ftrace by address.
987  * @fp: A fprobe data structure to be registered.
988  * @addrs: An array of target function address.
989  * @num: The number of entries of @addrs.
990  *
991  * Register @fp to ftrace for enabling the probe on the address given by @addrs.
992  * The @addrs must be the addresses of ftrace location address, which may be
993  * the symbol address + arch-dependent offset.
994  * If you unsure what this mean, please use other registration functions.
995  *
996  * Return 0 if @fp is registered successfully, -errno if not.
997  */
998 int register_fprobe_ips(struct fprobe *fp, unsigned long *addrs, int num)
999 {
1000 	struct fprobe_hlist *hlist_array;
1001 	int ret, i;
1002 
1003 	guard(mutex)(&fprobe_mutex);
1004 	if (fprobe_registered(fp))
1005 		return -EEXIST;
1006 
1007 	ret = fprobe_init(fp, addrs, num);
1008 	if (ret)
1009 		return ret;
1010 
1011 	if (fprobe_is_ftrace(fp))
1012 		ret = fprobe_ftrace_add_ips(addrs, num);
1013 	else
1014 		ret = fprobe_graph_add_ips(addrs, num);
1015 	if (ret) {
1016 		fprobe_fail_cleanup(fp);
1017 		return ret;
1018 	}
1019 
1020 	hlist_array = fp->hlist_array;
1021 	ret = add_fprobe_hash(fp);
1022 	for (i = 0; i < hlist_array->size && !ret; i++)
1023 		ret = insert_fprobe_node(&hlist_array->array[i], fp);
1024 
1025 	if (ret) {
1026 		unregister_fprobe_nolock(fp);
1027 		/* In error case, wait for clean up safely. */
1028 		synchronize_rcu();
1029 	}
1030 
1031 	return ret;
1032 }
1033 EXPORT_SYMBOL_GPL(register_fprobe_ips);
1034 
1035 /**
1036  * register_fprobe_syms() - Register fprobe to ftrace by symbols.
1037  * @fp: A fprobe data structure to be registered.
1038  * @syms: An array of target symbols.
1039  * @num: The number of entries of @syms.
1040  *
1041  * Register @fp to the symbols given by @syms array. This will be useful if
1042  * you are sure the symbols exist in the kernel.
1043  *
1044  * Return 0 if @fp is registered successfully, -errno if not.
1045  */
1046 int register_fprobe_syms(struct fprobe *fp, const char **syms, int num)
1047 {
1048 	unsigned long *addrs;
1049 	int ret;
1050 
1051 	if (!fp || !syms || num <= 0)
1052 		return -EINVAL;
1053 
1054 	addrs = get_ftrace_locations(syms, num);
1055 	if (IS_ERR(addrs))
1056 		return PTR_ERR(addrs);
1057 
1058 	ret = register_fprobe_ips(fp, addrs, num);
1059 
1060 	kfree(addrs);
1061 
1062 	return ret;
1063 }
1064 EXPORT_SYMBOL_GPL(register_fprobe_syms);
1065 
1066 bool fprobe_is_registered(struct fprobe *fp)
1067 {
1068 	if (!fp || !fp->hlist_array)
1069 		return false;
1070 	return true;
1071 }
1072 
1073 static int unregister_fprobe_nolock(struct fprobe *fp)
1074 {
1075 	struct fprobe_hlist *hlist_array = fp->hlist_array;
1076 	unsigned long *addrs = NULL;
1077 	int i, count;
1078 
1079 	addrs = kcalloc(hlist_array->size, sizeof(unsigned long), GFP_KERNEL);
1080 	/*
1081 	 * This will remove fprobe_hash_node from the hash table even if
1082 	 * memory allocation fails. However, ftrace_ops will not be updated.
1083 	 * Anyway, when the last fprobe is unregistered, ftrace_ops is also
1084 	 * unregistered.
1085 	 */
1086 	if (!addrs)
1087 		pr_warn("Failed to allocate working array. ftrace_ops may not sync.\n");
1088 
1089 	/* Remove non-synonim ips from table and hash */
1090 	count = 0;
1091 	for (i = 0; i < hlist_array->size; i++) {
1092 		delete_fprobe_node(&hlist_array->array[i]);
1093 		if (addrs && !fprobe_exists_on_hash(hlist_array->array[i].addr,
1094 						    fprobe_is_ftrace(fp)))
1095 			addrs[count++] = hlist_array->array[i].addr;
1096 	}
1097 	del_fprobe_hash(fp);
1098 
1099 	if (fprobe_is_ftrace(fp))
1100 		fprobe_ftrace_remove_ips(addrs, count);
1101 	else
1102 		fprobe_graph_remove_ips(addrs, count);
1103 
1104 	kfree_rcu(hlist_array, rcu);
1105 	fp->hlist_array = NULL;
1106 	kfree(addrs);
1107 
1108 	return 0;
1109 }
1110 
1111 /**
1112  * unregister_fprobe_async() - Unregister fprobe without RCU GP wait
1113  * @fp: A fprobe data structure to be unregistered.
1114  *
1115  * Unregister fprobe (and remove ftrace hooks from the function entries).
1116  * This function will NOT wait until the fprobe is no longer used.
1117  *
1118  * Return 0 if @fp is unregistered successfully, -errno if not.
1119  */
1120 int unregister_fprobe_async(struct fprobe *fp)
1121 {
1122 	guard(mutex)(&fprobe_mutex);
1123 	if (!fp || !fprobe_registered(fp))
1124 		return -EINVAL;
1125 
1126 	return unregister_fprobe_nolock(fp);
1127 }
1128 
1129 /**
1130  * unregister_fprobe() - Unregister fprobe with RCU GP wait
1131  * @fp: A fprobe data structure to be unregistered.
1132  *
1133  * Unregister fprobe (and remove ftrace hooks from the function entries).
1134  * This function will block until the fprobe is no longer used.
1135  *
1136  * Return 0 if @fp is unregistered successfully, -errno if not.
1137  */
1138 int unregister_fprobe(struct fprobe *fp)
1139 {
1140 	int ret = unregister_fprobe_async(fp);
1141 
1142 	if (!ret)
1143 		synchronize_rcu();
1144 	return ret;
1145 }
1146 EXPORT_SYMBOL_GPL(unregister_fprobe);
1147 
1148 static int __init fprobe_initcall(void)
1149 {
1150 	rhltable_init(&fprobe_ip_table, &fprobe_rht_params);
1151 	return 0;
1152 }
1153 core_initcall(fprobe_initcall);
1154