1 // SPDX-License-Identifier: GPL-2.0
2 /*
3 * event tracer
4 *
5 * Copyright (C) 2008 Red Hat Inc, Steven Rostedt <srostedt@redhat.com>
6 *
7 * - Added format output of fields of the trace point.
8 * This was based off of work by Tom Zanussi <tzanussi@gmail.com>.
9 *
10 */
11
12 #define pr_fmt(fmt) fmt
13
14 #include <linux/workqueue.h>
15 #include <linux/security.h>
16 #include <linux/spinlock.h>
17 #include <linux/seq_buf.h>
18 #include <linux/kthread.h>
19 #include <linux/tracefs.h>
20 #include <linux/uaccess.h>
21 #include <linux/module.h>
22 #include <linux/ctype.h>
23 #include <linux/sort.h>
24 #include <linux/slab.h>
25 #include <linux/delay.h>
26 #include <linux/btf.h>
27
28 #include <trace/events/sched.h>
29 #include <trace/syscall.h>
30
31 #include <asm/setup.h>
32
33 #include "trace_output.h"
34
35 #undef TRACE_SYSTEM
36 #define TRACE_SYSTEM "TRACE_SYSTEM"
37
38 DEFINE_MUTEX(event_mutex);
39
40 LIST_HEAD(ftrace_events);
41 static LIST_HEAD(ftrace_generic_fields);
42 static LIST_HEAD(ftrace_common_fields);
43 static bool eventdir_initialized;
44
45 static LIST_HEAD(module_strings);
46
47 struct module_string {
48 struct list_head next;
49 struct module *module;
50 char *str;
51 };
52
53 #define GFP_TRACE (GFP_KERNEL | __GFP_ZERO)
54
55 static struct kmem_cache *field_cachep;
56 static struct kmem_cache *file_cachep;
57
system_refcount(struct event_subsystem * system)58 static inline int system_refcount(struct event_subsystem *system)
59 {
60 return system->ref_count;
61 }
62
system_refcount_inc(struct event_subsystem * system)63 static int system_refcount_inc(struct event_subsystem *system)
64 {
65 return system->ref_count++;
66 }
67
system_refcount_dec(struct event_subsystem * system)68 static int system_refcount_dec(struct event_subsystem *system)
69 {
70 return --system->ref_count;
71 }
72
73 /* Double loops, do not use break, only goto's work */
74 #define do_for_each_event_file(tr, file) \
75 list_for_each_entry(tr, &ftrace_trace_arrays, list) { \
76 list_for_each_entry(file, &tr->events, list)
77
78 #define do_for_each_event_file_safe(tr, file) \
79 list_for_each_entry(tr, &ftrace_trace_arrays, list) { \
80 struct trace_event_file *___n; \
81 list_for_each_entry_safe(file, ___n, &tr->events, list)
82
83 #define while_for_each_event_file() \
84 }
85
86 static struct ftrace_event_field *
__find_event_field(struct list_head * head,const char * name)87 __find_event_field(struct list_head *head, const char *name)
88 {
89 struct ftrace_event_field *field;
90
91 list_for_each_entry(field, head, link) {
92 if (!strcmp(field->name, name))
93 return field;
94 }
95
96 return NULL;
97 }
98
99 struct ftrace_event_field *
trace_find_event_field(struct trace_event_call * call,char * name)100 trace_find_event_field(struct trace_event_call *call, char *name)
101 {
102 struct ftrace_event_field *field;
103 struct list_head *head;
104
105 head = trace_get_fields(call);
106 field = __find_event_field(head, name);
107 if (field)
108 return field;
109
110 field = __find_event_field(&ftrace_generic_fields, name);
111 if (field)
112 return field;
113
114 return __find_event_field(&ftrace_common_fields, name);
115 }
116
__trace_define_field(struct list_head * head,const char * type,const char * name,int offset,int size,int is_signed,int filter_type,int len,int need_test)117 static int __trace_define_field(struct list_head *head, const char *type,
118 const char *name, int offset, int size,
119 int is_signed, int filter_type, int len,
120 int need_test)
121 {
122 struct ftrace_event_field *field;
123
124 field = kmem_cache_alloc(field_cachep, GFP_TRACE);
125 if (!field)
126 return -ENOMEM;
127
128 field->name = name;
129 field->type = type;
130
131 if (filter_type == FILTER_OTHER)
132 field->filter_type = filter_assign_type(type);
133 else
134 field->filter_type = filter_type;
135
136 field->offset = offset;
137 field->size = size;
138 field->is_signed = is_signed;
139 field->needs_test = need_test;
140 field->len = len;
141
142 list_add(&field->link, head);
143
144 return 0;
145 }
146
trace_define_field(struct trace_event_call * call,const char * type,const char * name,int offset,int size,int is_signed,int filter_type)147 int trace_define_field(struct trace_event_call *call, const char *type,
148 const char *name, int offset, int size, int is_signed,
149 int filter_type)
150 {
151 struct list_head *head;
152
153 if (WARN_ON(!call->class))
154 return 0;
155
156 head = trace_get_fields(call);
157 return __trace_define_field(head, type, name, offset, size,
158 is_signed, filter_type, 0, 0);
159 }
160 EXPORT_SYMBOL_GPL(trace_define_field);
161
trace_define_field_ext(struct trace_event_call * call,const char * type,const char * name,int offset,int size,int is_signed,int filter_type,int len,int need_test)162 static int trace_define_field_ext(struct trace_event_call *call, const char *type,
163 const char *name, int offset, int size, int is_signed,
164 int filter_type, int len, int need_test)
165 {
166 struct list_head *head;
167
168 if (WARN_ON(!call->class))
169 return 0;
170
171 head = trace_get_fields(call);
172 return __trace_define_field(head, type, name, offset, size,
173 is_signed, filter_type, len, need_test);
174 }
175
176 #define __generic_field(type, item, filter_type) \
177 ret = __trace_define_field(&ftrace_generic_fields, #type, \
178 #item, 0, 0, is_signed_type(type), \
179 filter_type, 0, 0); \
180 if (ret) \
181 return ret;
182
183 #define __common_field(type, item) \
184 ret = __trace_define_field(&ftrace_common_fields, #type, \
185 "common_" #item, \
186 offsetof(typeof(ent), item), \
187 sizeof(ent.item), \
188 is_signed_type(type), FILTER_OTHER, \
189 0, 0); \
190 if (ret) \
191 return ret;
192
trace_define_generic_fields(void)193 static int trace_define_generic_fields(void)
194 {
195 int ret;
196
197 __generic_field(int, CPU, FILTER_CPU);
198 __generic_field(int, cpu, FILTER_CPU);
199 __generic_field(int, common_cpu, FILTER_CPU);
200 __generic_field(char *, COMM, FILTER_COMM);
201 __generic_field(char *, comm, FILTER_COMM);
202 __generic_field(char *, stacktrace, FILTER_STACKTRACE);
203 __generic_field(char *, STACKTRACE, FILTER_STACKTRACE);
204
205 return ret;
206 }
207
trace_define_common_fields(void)208 static int trace_define_common_fields(void)
209 {
210 int ret;
211 struct trace_entry ent;
212
213 __common_field(unsigned short, type);
214 __common_field(unsigned char, flags);
215 /* Holds both preempt_count and migrate_disable */
216 __common_field(unsigned char, preempt_count);
217 __common_field(int, pid);
218
219 return ret;
220 }
221
trace_destroy_fields(struct trace_event_call * call)222 static void trace_destroy_fields(struct trace_event_call *call)
223 {
224 struct ftrace_event_field *field, *next;
225 struct list_head *head;
226
227 head = trace_get_fields(call);
228 list_for_each_entry_safe(field, next, head, link) {
229 list_del(&field->link);
230 kmem_cache_free(field_cachep, field);
231 }
232 }
233
234 /*
235 * run-time version of trace_event_get_offsets_<call>() that returns the last
236 * accessible offset of trace fields excluding __dynamic_array bytes
237 */
trace_event_get_offsets(struct trace_event_call * call)238 int trace_event_get_offsets(struct trace_event_call *call)
239 {
240 struct ftrace_event_field *tail;
241 struct list_head *head;
242
243 head = trace_get_fields(call);
244 /*
245 * head->next points to the last field with the largest offset,
246 * since it was added last by trace_define_field()
247 */
248 tail = list_first_entry(head, struct ftrace_event_field, link);
249 return tail->offset + tail->size;
250 }
251
252
find_event_field(const char * fmt,struct trace_event_call * call)253 static struct trace_event_fields *find_event_field(const char *fmt,
254 struct trace_event_call *call)
255 {
256 struct trace_event_fields *field = call->class->fields_array;
257 const char *p = fmt;
258 int len;
259
260 if (!(len = str_has_prefix(fmt, "REC->")))
261 return NULL;
262 fmt += len;
263 for (p = fmt; *p; p++) {
264 if (!isalnum(*p) && *p != '_')
265 break;
266 }
267 len = p - fmt;
268
269 for (; field->type; field++) {
270 if (strncmp(field->name, fmt, len) || field->name[len])
271 continue;
272
273 return field;
274 }
275 return NULL;
276 }
277
278 /*
279 * Check if the referenced field is an array and return true,
280 * as arrays are OK to dereference.
281 */
test_field(const char * fmt,struct trace_event_call * call)282 static bool test_field(const char *fmt, struct trace_event_call *call)
283 {
284 struct trace_event_fields *field;
285
286 field = find_event_field(fmt, call);
287 if (!field)
288 return false;
289
290 /* This is an array and is OK to dereference. */
291 return strchr(field->type, '[') != NULL;
292 }
293
294 /* Look for a string within an argument */
find_print_string(const char * arg,const char * str,const char * end)295 static bool find_print_string(const char *arg, const char *str, const char *end)
296 {
297 const char *r;
298
299 r = strstr(arg, str);
300 return r && r < end;
301 }
302
303 /* Return true if the argument pointer is safe */
process_pointer(const char * fmt,int len,struct trace_event_call * call)304 static bool process_pointer(const char *fmt, int len, struct trace_event_call *call)
305 {
306 const char *r, *e, *a;
307
308 e = fmt + len;
309
310 /* Find the REC-> in the argument */
311 r = strstr(fmt, "REC->");
312 if (r && r < e) {
313 /*
314 * Addresses of events on the buffer, or an array on the buffer is
315 * OK to dereference. There's ways to fool this, but
316 * this is to catch common mistakes, not malicious code.
317 */
318 a = strchr(fmt, '&');
319 if ((a && (a < r)) || test_field(r, call))
320 return true;
321 } else if (find_print_string(fmt, "__get_dynamic_array(", e)) {
322 return true;
323 } else if (find_print_string(fmt, "__get_rel_dynamic_array(", e)) {
324 return true;
325 } else if (find_print_string(fmt, "__get_dynamic_array_len(", e)) {
326 return true;
327 } else if (find_print_string(fmt, "__get_rel_dynamic_array_len(", e)) {
328 return true;
329 } else if (find_print_string(fmt, "__get_sockaddr(", e)) {
330 return true;
331 } else if (find_print_string(fmt, "__get_rel_sockaddr(", e)) {
332 return true;
333 }
334 return false;
335 }
336
337 /* Return true if the string is safe */
process_string(const char * fmt,int len,struct trace_event_call * call)338 static bool process_string(const char *fmt, int len, struct trace_event_call *call)
339 {
340 struct trace_event_fields *field;
341 const char *r, *e, *s;
342
343 e = fmt + len;
344
345 /*
346 * There are several helper functions that return strings.
347 * If the argument contains a function, then assume its field is valid.
348 * It is considered that the argument has a function if it has:
349 * alphanumeric or '_' before a parenthesis.
350 */
351 s = fmt;
352 do {
353 r = strstr(s, "(");
354 if (!r || r >= e)
355 break;
356 for (int i = 1; r - i >= s; i++) {
357 char ch = *(r - i);
358 if (isspace(ch))
359 continue;
360 if (isalnum(ch) || ch == '_')
361 return true;
362 /* Anything else, this isn't a function */
363 break;
364 }
365 /* A function could be wrapped in parenthesis, try the next one */
366 s = r + 1;
367 } while (s < e);
368
369 /*
370 * Check for arrays. If the argument has: foo[REC->val]
371 * then it is very likely that foo is an array of strings
372 * that are safe to use.
373 */
374 r = strstr(s, "[");
375 if (r && r < e) {
376 r = strstr(r, "REC->");
377 if (r && r < e)
378 return true;
379 }
380
381 /*
382 * If there's any strings in the argument consider this arg OK as it
383 * could be: REC->field ? "foo" : "bar" and we don't want to get into
384 * verifying that logic here.
385 */
386 if (find_print_string(fmt, "\"", e))
387 return true;
388
389 /* Dereferenced strings are also valid like any other pointer */
390 if (process_pointer(fmt, len, call))
391 return true;
392
393 /* Make sure the field is found */
394 field = find_event_field(fmt, call);
395 if (!field)
396 return false;
397
398 /* Test this field's string before printing the event */
399 call->flags |= TRACE_EVENT_FL_TEST_STR;
400 field->needs_test = 1;
401
402 return true;
403 }
404
test_double_dereference(const char * str,int len,struct trace_event_call * call)405 static void test_double_dereference(const char *str, int len,
406 struct trace_event_call *call)
407 {
408 const char *ptr;
409 const char *end = str + len;
410
411 ptr = strstr(str, "REC->");
412
413 while (ptr && ptr < end) {
414
415 ptr += 5;
416 for (; ptr < end; ptr++) {
417 if (ptr[0] == '-' && ptr[1] == '>') {
418 pr_warn("TRACE EVENT ERROR: Event %s has double dereference in TP_printk: %.*s\n",
419 trace_event_name(call), len, str);
420 WARN_ONCE(1, "Event %s has double dereference in TP_printk: %.*s\n",
421 trace_event_name(call), len, str);
422 return;
423 }
424 if (!isalnum(*ptr) && *ptr != '_')
425 break;
426 }
427
428 ptr = strstr(ptr, "REC->");
429 }
430 }
431
handle_dereference_arg(const char * arg_str,u64 string_flags,int len,u64 * dereference_flags,int arg,struct trace_event_call * call)432 static void handle_dereference_arg(const char *arg_str, u64 string_flags, int len,
433 u64 *dereference_flags, int arg,
434 struct trace_event_call *call)
435 {
436 if (string_flags & (1ULL << arg)) {
437 if (process_string(arg_str, len, call))
438 *dereference_flags &= ~(1ULL << arg);
439 } else if (process_pointer(arg_str, len, call))
440 *dereference_flags &= ~(1ULL << arg);
441 else
442 pr_warn("TRACE EVENT ERROR: Bad dereference argument: '%.*s'\n",
443 len, arg_str);
444 }
445
446 /*
447 * Examine the print fmt of the event looking for unsafe dereference
448 * pointers using %p* that could be recorded in the trace event and
449 * much later referenced after the pointer was freed. Dereferencing
450 * pointers are OK, if it is dereferenced into the event itself.
451 */
test_event_printk(struct trace_event_call * call)452 static void test_event_printk(struct trace_event_call *call)
453 {
454 u64 dereference_flags = 0;
455 u64 string_flags = 0;
456 bool first = true;
457 const char *fmt;
458 int parens = 0;
459 char in_quote = 0;
460 int start_arg = 0;
461 int arg = 0;
462 int i, e;
463
464 fmt = call->print_fmt;
465
466 if (!fmt)
467 return;
468
469 for (i = 0; fmt[i]; i++) {
470 switch (fmt[i]) {
471 case '\\':
472 i++;
473 if (!fmt[i])
474 return;
475 continue;
476 case '"':
477 case '\'':
478 /*
479 * The print fmt starts with a string that
480 * is processed first to find %p* usage,
481 * then after the first string, the print fmt
482 * contains arguments that are used to check
483 * if the dereferenced %p* usage is safe.
484 */
485 if (first) {
486 if (fmt[i] == '\'')
487 continue;
488 if (in_quote) {
489 arg = 0;
490 first = false;
491 }
492 }
493 if (in_quote) {
494 if (in_quote == fmt[i])
495 in_quote = 0;
496 } else {
497 in_quote = fmt[i];
498 }
499 continue;
500 case '%':
501 if (!first || !in_quote)
502 continue;
503 i++;
504 if (!fmt[i])
505 return;
506 switch (fmt[i]) {
507 case '%':
508 continue;
509 case 'p':
510 do_pointer:
511 /* Find dereferencing fields */
512 switch (fmt[i + 1]) {
513 case 'B': case 'R': case 'r':
514 case 'b': case 'M': case 'm':
515 case 'I': case 'i': case 'E':
516 case 'U': case 'V': case 'N':
517 case 'a': case 'd': case 'D':
518 case 'g': case 't': case 'C':
519 case 'O': case 'f':
520 if (WARN_ONCE(arg == 63,
521 "Too many args for event: %s",
522 trace_event_name(call)))
523 return;
524 dereference_flags |= 1ULL << arg;
525 }
526 break;
527 default:
528 {
529 bool star = false;
530 int j;
531
532 /* Increment arg if %*s exists. */
533 for (j = 0; fmt[i + j]; j++) {
534 if (isdigit(fmt[i + j]) ||
535 fmt[i + j] == '.')
536 continue;
537 if (fmt[i + j] == '*') {
538 star = true;
539 /* Handle %*pbl case */
540 if (!j && fmt[i + 1] == 'p') {
541 arg++;
542 i++;
543 goto do_pointer;
544 }
545 continue;
546 }
547 if ((fmt[i + j] == 's')) {
548 if (star)
549 arg++;
550 if (WARN_ONCE(arg == 63,
551 "Too many args for event: %s",
552 trace_event_name(call)))
553 return;
554 dereference_flags |= 1ULL << arg;
555 string_flags |= 1ULL << arg;
556 }
557 break;
558 }
559 break;
560 } /* default */
561
562 } /* switch */
563 arg++;
564 continue;
565 case '(':
566 if (in_quote)
567 continue;
568 parens++;
569 continue;
570 case ')':
571 if (in_quote)
572 continue;
573 parens--;
574 if (WARN_ONCE(parens < 0,
575 "Paren mismatch for event: %s\narg='%s'\n%*s",
576 trace_event_name(call),
577 fmt + start_arg,
578 (i - start_arg) + 5, "^"))
579 return;
580 continue;
581 case ',':
582 if (in_quote || parens)
583 continue;
584 e = i;
585 i++;
586 while (isspace(fmt[i]))
587 i++;
588
589 /*
590 * If start_arg is zero, then this is the start of the
591 * first argument. The processing of the argument happens
592 * when the end of the argument is found, as it needs to
593 * handle parenthesis and such.
594 */
595 if (!start_arg) {
596 start_arg = i;
597 /* Balance out the i++ in the for loop */
598 i--;
599 continue;
600 }
601
602 test_double_dereference(fmt + start_arg, e - start_arg, call);
603
604 if (dereference_flags & (1ULL << arg)) {
605 handle_dereference_arg(fmt + start_arg, string_flags,
606 e - start_arg,
607 &dereference_flags, arg, call);
608 }
609
610 start_arg = i;
611 arg++;
612 /* Balance out the i++ in the for loop */
613 i--;
614 }
615 }
616
617 test_double_dereference(fmt + start_arg, i - start_arg, call);
618
619 if (dereference_flags & (1ULL << arg)) {
620 handle_dereference_arg(fmt + start_arg, string_flags,
621 i - start_arg,
622 &dereference_flags, arg, call);
623 }
624
625 /*
626 * If you triggered the below warning, the trace event reported
627 * uses an unsafe dereference pointer %p*. As the data stored
628 * at the trace event time may no longer exist when the trace
629 * event is printed, dereferencing to the original source is
630 * unsafe. The source of the dereference must be copied into the
631 * event itself, and the dereference must access the copy instead.
632 */
633 if (WARN_ON_ONCE(dereference_flags)) {
634 arg = 1;
635 while (!(dereference_flags & 1)) {
636 dereference_flags >>= 1;
637 arg++;
638 }
639 pr_warn("event %s has unsafe dereference of argument %d\n",
640 trace_event_name(call), arg);
641 pr_warn("print_fmt: %s\n", fmt);
642 }
643 }
644
trace_event_raw_init(struct trace_event_call * call)645 int trace_event_raw_init(struct trace_event_call *call)
646 {
647 int id;
648
649 id = register_trace_event(&call->event);
650 if (!id)
651 return -ENODEV;
652
653 test_event_printk(call);
654
655 return 0;
656 }
657 EXPORT_SYMBOL_GPL(trace_event_raw_init);
658
trace_event_ignore_this_pid(struct trace_event_file * trace_file)659 bool trace_event_ignore_this_pid(struct trace_event_file *trace_file)
660 {
661 struct trace_array *tr = trace_file->tr;
662 struct trace_pid_list *no_pid_list;
663 struct trace_pid_list *pid_list;
664
665 pid_list = rcu_dereference_raw(tr->filtered_pids);
666 no_pid_list = rcu_dereference_raw(tr->filtered_no_pids);
667
668 if (!pid_list && !no_pid_list)
669 return false;
670
671 /*
672 * This is recorded at every sched_switch for this task.
673 * Thus, even if the task migrates the ignore value will be the same.
674 */
675 return this_cpu_read(tr->array_buffer.data->ignore_pid) != 0;
676 }
677 EXPORT_SYMBOL_GPL(trace_event_ignore_this_pid);
678
679 /**
680 * trace_event_buffer_reserve - reserve space on the ring buffer for an event
681 * @fbuffer: information about how to save the event
682 * @trace_file: the instance file descriptor for the event
683 * @len: The length of the event
684 *
685 * The @fbuffer has information about the ring buffer and data will
686 * be added to it to be used by the call to trace_event_buffer_commit().
687 * The @trace_file is the desrciptor with information about the status
688 * of the given event for a specific trace_array instance.
689 * The @len is the length of data to save for the event.
690 *
691 * Returns a pointer to the data on the ring buffer or NULL if the
692 * event was not reserved (event was filtered, too big, or the buffer
693 * simply was disabled for write).
694 */
trace_event_buffer_reserve(struct trace_event_buffer * fbuffer,struct trace_event_file * trace_file,unsigned long len)695 void *trace_event_buffer_reserve(struct trace_event_buffer *fbuffer,
696 struct trace_event_file *trace_file,
697 unsigned long len)
698 {
699 struct trace_event_call *event_call = trace_file->event_call;
700
701 if ((trace_file->flags & EVENT_FILE_FL_PID_FILTER) &&
702 trace_event_ignore_this_pid(trace_file))
703 return NULL;
704
705 /*
706 * If CONFIG_PREEMPTION is enabled, then the tracepoint itself disables
707 * preemption (adding one to the preempt_count). Since we are
708 * interested in the preempt_count at the time the tracepoint was
709 * hit, we need to subtract one to offset the increment.
710 */
711 fbuffer->trace_ctx = tracing_gen_ctx_dec();
712 fbuffer->trace_file = trace_file;
713
714 fbuffer->event =
715 trace_event_buffer_lock_reserve(&fbuffer->buffer, trace_file,
716 event_call->event.type, len,
717 fbuffer->trace_ctx);
718 if (!fbuffer->event)
719 return NULL;
720
721 fbuffer->regs = NULL;
722 fbuffer->entry = ring_buffer_event_data(fbuffer->event);
723 return fbuffer->entry;
724 }
725 EXPORT_SYMBOL_GPL(trace_event_buffer_reserve);
726
trace_event_reg(struct trace_event_call * call,enum trace_reg type,void * data)727 int trace_event_reg(struct trace_event_call *call,
728 enum trace_reg type, void *data)
729 {
730 struct trace_event_file *file = data;
731
732 WARN_ON(!(call->flags & TRACE_EVENT_FL_TRACEPOINT));
733 switch (type) {
734 case TRACE_REG_REGISTER:
735 return tracepoint_probe_register(call->tp,
736 call->class->probe,
737 file);
738 case TRACE_REG_UNREGISTER:
739 tracepoint_probe_unregister(call->tp,
740 call->class->probe,
741 file);
742 return 0;
743
744 #ifdef CONFIG_PERF_EVENTS
745 case TRACE_REG_PERF_REGISTER:
746 if (!call->class->perf_probe)
747 return -ENODEV;
748 return tracepoint_probe_register(call->tp,
749 call->class->perf_probe,
750 call);
751 case TRACE_REG_PERF_UNREGISTER:
752 tracepoint_probe_unregister(call->tp,
753 call->class->perf_probe,
754 call);
755 return 0;
756 case TRACE_REG_PERF_OPEN:
757 case TRACE_REG_PERF_CLOSE:
758 case TRACE_REG_PERF_ADD:
759 case TRACE_REG_PERF_DEL:
760 return 0;
761 #endif
762 }
763 return 0;
764 }
765 EXPORT_SYMBOL_GPL(trace_event_reg);
766
trace_event_enable_cmd_record(bool enable)767 void trace_event_enable_cmd_record(bool enable)
768 {
769 struct trace_event_file *file;
770 struct trace_array *tr;
771
772 lockdep_assert_held(&event_mutex);
773
774 do_for_each_event_file(tr, file) {
775
776 if (!(file->flags & EVENT_FILE_FL_ENABLED))
777 continue;
778
779 if (enable) {
780 tracing_start_cmdline_record();
781 set_bit(EVENT_FILE_FL_RECORDED_CMD_BIT, &file->flags);
782 } else {
783 tracing_stop_cmdline_record();
784 clear_bit(EVENT_FILE_FL_RECORDED_CMD_BIT, &file->flags);
785 }
786 } while_for_each_event_file();
787 }
788
trace_event_enable_tgid_record(bool enable)789 void trace_event_enable_tgid_record(bool enable)
790 {
791 struct trace_event_file *file;
792 struct trace_array *tr;
793
794 lockdep_assert_held(&event_mutex);
795
796 do_for_each_event_file(tr, file) {
797 if (!(file->flags & EVENT_FILE_FL_ENABLED))
798 continue;
799
800 if (enable) {
801 tracing_start_tgid_record();
802 set_bit(EVENT_FILE_FL_RECORDED_TGID_BIT, &file->flags);
803 } else {
804 tracing_stop_tgid_record();
805 clear_bit(EVENT_FILE_FL_RECORDED_TGID_BIT,
806 &file->flags);
807 }
808 } while_for_each_event_file();
809 }
810
__ftrace_event_enable_disable(struct trace_event_file * file,int enable,int soft_disable)811 static int __ftrace_event_enable_disable(struct trace_event_file *file,
812 int enable, int soft_disable)
813 {
814 struct trace_event_call *call = file->event_call;
815 struct trace_array *tr = file->tr;
816 bool soft_mode = atomic_read(&file->sm_ref) != 0;
817 int ret = 0;
818 int disable;
819
820 switch (enable) {
821 case 0:
822 /*
823 * When soft_disable is set and enable is cleared, the sm_ref
824 * reference counter is decremented. If it reaches 0, we want
825 * to clear the SOFT_DISABLED flag but leave the event in the
826 * state that it was. That is, if the event was enabled and
827 * SOFT_DISABLED isn't set, then do nothing. But if SOFT_DISABLED
828 * is set we do not want the event to be enabled before we
829 * clear the bit.
830 *
831 * When soft_disable is not set but the soft_mode is,
832 * we do nothing. Do not disable the tracepoint, otherwise
833 * "soft enable"s (clearing the SOFT_DISABLED bit) won't work.
834 */
835 if (soft_disable) {
836 if (atomic_dec_return(&file->sm_ref) > 0)
837 break;
838 disable = file->flags & EVENT_FILE_FL_SOFT_DISABLED;
839 soft_mode = false;
840 /* Disable use of trace_buffered_event */
841 trace_buffered_event_disable();
842 } else
843 disable = !soft_mode;
844
845 if (disable && (file->flags & EVENT_FILE_FL_ENABLED)) {
846 clear_bit(EVENT_FILE_FL_ENABLED_BIT, &file->flags);
847 if (file->flags & EVENT_FILE_FL_RECORDED_CMD) {
848 tracing_stop_cmdline_record();
849 clear_bit(EVENT_FILE_FL_RECORDED_CMD_BIT, &file->flags);
850 }
851
852 if (file->flags & EVENT_FILE_FL_RECORDED_TGID) {
853 tracing_stop_tgid_record();
854 clear_bit(EVENT_FILE_FL_RECORDED_TGID_BIT, &file->flags);
855 }
856
857 ret = call->class->reg(call, TRACE_REG_UNREGISTER, file);
858
859 WARN_ON_ONCE(ret);
860 }
861 /* If in soft mode, just set the SOFT_DISABLE_BIT, else clear it */
862 if (soft_mode)
863 set_bit(EVENT_FILE_FL_SOFT_DISABLED_BIT, &file->flags);
864 else
865 clear_bit(EVENT_FILE_FL_SOFT_DISABLED_BIT, &file->flags);
866 break;
867 case 1:
868 /*
869 * When soft_disable is set and enable is set, we want to
870 * register the tracepoint for the event, but leave the event
871 * as is. That means, if the event was already enabled, we do
872 * nothing. If the event is disabled, we set SOFT_DISABLED
873 * before enabling the event tracepoint, so it still seems
874 * to be disabled.
875 */
876 if (!soft_disable)
877 clear_bit(EVENT_FILE_FL_SOFT_DISABLED_BIT, &file->flags);
878 else {
879 if (atomic_inc_return(&file->sm_ref) > 1)
880 break;
881 /* Enable use of trace_buffered_event */
882 trace_buffered_event_enable();
883 }
884
885 if (!(file->flags & EVENT_FILE_FL_ENABLED)) {
886 bool cmd = false, tgid = false;
887
888 /* Keep the event disabled, when going to soft mode. */
889 if (soft_disable)
890 set_bit(EVENT_FILE_FL_SOFT_DISABLED_BIT, &file->flags);
891
892 if (tr->trace_flags & TRACE_ITER(RECORD_CMD)) {
893 cmd = true;
894 tracing_start_cmdline_record();
895 set_bit(EVENT_FILE_FL_RECORDED_CMD_BIT, &file->flags);
896 }
897
898 if (tr->trace_flags & TRACE_ITER(RECORD_TGID)) {
899 tgid = true;
900 tracing_start_tgid_record();
901 set_bit(EVENT_FILE_FL_RECORDED_TGID_BIT, &file->flags);
902 }
903
904 ret = call->class->reg(call, TRACE_REG_REGISTER, file);
905 if (ret) {
906 if (cmd)
907 tracing_stop_cmdline_record();
908 if (tgid)
909 tracing_stop_tgid_record();
910 pr_info("event trace: Could not enable event "
911 "%s\n", trace_event_name(call));
912 break;
913 }
914 set_bit(EVENT_FILE_FL_ENABLED_BIT, &file->flags);
915
916 /* WAS_ENABLED gets set but never cleared. */
917 set_bit(EVENT_FILE_FL_WAS_ENABLED_BIT, &file->flags);
918 }
919 break;
920 }
921
922 return ret;
923 }
924
trace_event_enable_disable(struct trace_event_file * file,int enable,int soft_disable)925 int trace_event_enable_disable(struct trace_event_file *file,
926 int enable, int soft_disable)
927 {
928 return __ftrace_event_enable_disable(file, enable, soft_disable);
929 }
930
ftrace_event_enable_disable(struct trace_event_file * file,int enable)931 static int ftrace_event_enable_disable(struct trace_event_file *file,
932 int enable)
933 {
934 return __ftrace_event_enable_disable(file, enable, 0);
935 }
936
937 #ifdef CONFIG_MODULES
938 struct event_mod_load {
939 struct list_head list;
940 char *module;
941 char *match;
942 char *system;
943 char *event;
944 };
945
free_event_mod(struct event_mod_load * event_mod)946 static void free_event_mod(struct event_mod_load *event_mod)
947 {
948 list_del(&event_mod->list);
949 kfree(event_mod->module);
950 kfree(event_mod->match);
951 kfree(event_mod->system);
952 kfree(event_mod->event);
953 kfree(event_mod);
954 }
955
clear_mod_events(struct trace_array * tr)956 static void clear_mod_events(struct trace_array *tr)
957 {
958 struct event_mod_load *event_mod, *n;
959
960 list_for_each_entry_safe(event_mod, n, &tr->mod_events, list) {
961 free_event_mod(event_mod);
962 }
963 }
964
remove_cache_mod(struct trace_array * tr,const char * mod,const char * match,const char * system,const char * event)965 static int remove_cache_mod(struct trace_array *tr, const char *mod,
966 const char *match, const char *system, const char *event)
967 {
968 struct event_mod_load *event_mod, *n;
969 int ret = -EINVAL;
970
971 list_for_each_entry_safe(event_mod, n, &tr->mod_events, list) {
972 if (strcmp(event_mod->module, mod) != 0)
973 continue;
974
975 if (match && (!event_mod->match || strcmp(event_mod->match, match) != 0))
976 continue;
977
978 if (system &&
979 (!event_mod->system || strcmp(event_mod->system, system) != 0))
980 continue;
981
982 if (event &&
983 (!event_mod->event || strcmp(event_mod->event, event) != 0))
984 continue;
985
986 free_event_mod(event_mod);
987 ret = 0;
988 }
989
990 return ret;
991 }
992
cache_mod(struct trace_array * tr,const char * mod,int set,const char * match,const char * system,const char * event)993 static int cache_mod(struct trace_array *tr, const char *mod, int set,
994 const char *match, const char *system, const char *event)
995 {
996 struct event_mod_load *event_mod;
997
998 /* If the module exists, then this just failed to find an event */
999 if (module_exists(mod))
1000 return -EINVAL;
1001
1002 /* See if this is to remove a cached filter */
1003 if (!set)
1004 return remove_cache_mod(tr, mod, match, system, event);
1005
1006 event_mod = kzalloc_obj(*event_mod);
1007 if (!event_mod)
1008 return -ENOMEM;
1009
1010 INIT_LIST_HEAD(&event_mod->list);
1011 event_mod->module = kstrdup(mod, GFP_KERNEL);
1012 if (!event_mod->module)
1013 goto out_free;
1014
1015 if (match) {
1016 event_mod->match = kstrdup(match, GFP_KERNEL);
1017 if (!event_mod->match)
1018 goto out_free;
1019 }
1020
1021 if (system) {
1022 event_mod->system = kstrdup(system, GFP_KERNEL);
1023 if (!event_mod->system)
1024 goto out_free;
1025 }
1026
1027 if (event) {
1028 event_mod->event = kstrdup(event, GFP_KERNEL);
1029 if (!event_mod->event)
1030 goto out_free;
1031 }
1032
1033 list_add(&event_mod->list, &tr->mod_events);
1034
1035 return 0;
1036
1037 out_free:
1038 free_event_mod(event_mod);
1039
1040 return -ENOMEM;
1041 }
1042 #else /* CONFIG_MODULES */
clear_mod_events(struct trace_array * tr)1043 static inline void clear_mod_events(struct trace_array *tr) { }
cache_mod(struct trace_array * tr,const char * mod,int set,const char * match,const char * system,const char * event)1044 static int cache_mod(struct trace_array *tr, const char *mod, int set,
1045 const char *match, const char *system, const char *event)
1046 {
1047 return -EINVAL;
1048 }
1049 #endif
1050
ftrace_clear_events(struct trace_array * tr)1051 static void ftrace_clear_events(struct trace_array *tr)
1052 {
1053 struct trace_event_file *file;
1054
1055 mutex_lock(&event_mutex);
1056 list_for_each_entry(file, &tr->events, list) {
1057 ftrace_event_enable_disable(file, 0);
1058 }
1059 clear_mod_events(tr);
1060 mutex_unlock(&event_mutex);
1061 }
1062
1063 static void
event_filter_pid_sched_process_exit(void * data,struct task_struct * task)1064 event_filter_pid_sched_process_exit(void *data, struct task_struct *task)
1065 {
1066 struct trace_pid_list *pid_list;
1067 struct trace_array *tr = data;
1068
1069 guard(preempt)();
1070 pid_list = rcu_dereference_raw(tr->filtered_pids);
1071 trace_filter_add_remove_task(pid_list, NULL, task);
1072
1073 pid_list = rcu_dereference_raw(tr->filtered_no_pids);
1074 trace_filter_add_remove_task(pid_list, NULL, task);
1075 }
1076
1077 static void
event_filter_pid_sched_process_fork(void * data,struct task_struct * self,struct task_struct * task)1078 event_filter_pid_sched_process_fork(void *data,
1079 struct task_struct *self,
1080 struct task_struct *task)
1081 {
1082 struct trace_pid_list *pid_list;
1083 struct trace_array *tr = data;
1084
1085 guard(preempt)();
1086 pid_list = rcu_dereference_sched(tr->filtered_pids);
1087 trace_filter_add_remove_task(pid_list, self, task);
1088
1089 pid_list = rcu_dereference_sched(tr->filtered_no_pids);
1090 trace_filter_add_remove_task(pid_list, self, task);
1091 }
1092
trace_event_follow_fork(struct trace_array * tr,bool enable)1093 void trace_event_follow_fork(struct trace_array *tr, bool enable)
1094 {
1095 if (enable) {
1096 register_trace_prio_sched_process_fork(event_filter_pid_sched_process_fork,
1097 tr, INT_MIN);
1098 register_trace_prio_sched_process_free(event_filter_pid_sched_process_exit,
1099 tr, INT_MAX);
1100 } else {
1101 unregister_trace_sched_process_fork(event_filter_pid_sched_process_fork,
1102 tr);
1103 unregister_trace_sched_process_free(event_filter_pid_sched_process_exit,
1104 tr);
1105 }
1106 }
1107
1108 static void
event_filter_pid_sched_switch_probe_pre(void * data,bool preempt,struct task_struct * prev,struct task_struct * next,unsigned int prev_state)1109 event_filter_pid_sched_switch_probe_pre(void *data, bool preempt,
1110 struct task_struct *prev,
1111 struct task_struct *next,
1112 unsigned int prev_state)
1113 {
1114 struct trace_array *tr = data;
1115 struct trace_pid_list *no_pid_list;
1116 struct trace_pid_list *pid_list;
1117 bool ret;
1118
1119 pid_list = rcu_dereference_sched(tr->filtered_pids);
1120 no_pid_list = rcu_dereference_sched(tr->filtered_no_pids);
1121
1122 /*
1123 * Sched switch is funny, as we only want to ignore it
1124 * in the notrace case if both prev and next should be ignored.
1125 */
1126 ret = trace_ignore_this_task(NULL, no_pid_list, prev) &&
1127 trace_ignore_this_task(NULL, no_pid_list, next);
1128
1129 this_cpu_write(tr->array_buffer.data->ignore_pid, ret ||
1130 (trace_ignore_this_task(pid_list, NULL, prev) &&
1131 trace_ignore_this_task(pid_list, NULL, next)));
1132 }
1133
1134 static void
event_filter_pid_sched_switch_probe_post(void * data,bool preempt,struct task_struct * prev,struct task_struct * next,unsigned int prev_state)1135 event_filter_pid_sched_switch_probe_post(void *data, bool preempt,
1136 struct task_struct *prev,
1137 struct task_struct *next,
1138 unsigned int prev_state)
1139 {
1140 struct trace_array *tr = data;
1141 struct trace_pid_list *no_pid_list;
1142 struct trace_pid_list *pid_list;
1143
1144 pid_list = rcu_dereference_sched(tr->filtered_pids);
1145 no_pid_list = rcu_dereference_sched(tr->filtered_no_pids);
1146
1147 this_cpu_write(tr->array_buffer.data->ignore_pid,
1148 trace_ignore_this_task(pid_list, no_pid_list, next));
1149 }
1150
1151 static void
event_filter_pid_sched_wakeup_probe_pre(void * data,struct task_struct * task)1152 event_filter_pid_sched_wakeup_probe_pre(void *data, struct task_struct *task)
1153 {
1154 struct trace_array *tr = data;
1155 struct trace_pid_list *no_pid_list;
1156 struct trace_pid_list *pid_list;
1157
1158 /* Nothing to do if we are already tracing */
1159 if (!this_cpu_read(tr->array_buffer.data->ignore_pid))
1160 return;
1161
1162 pid_list = rcu_dereference_sched(tr->filtered_pids);
1163 no_pid_list = rcu_dereference_sched(tr->filtered_no_pids);
1164
1165 this_cpu_write(tr->array_buffer.data->ignore_pid,
1166 trace_ignore_this_task(pid_list, no_pid_list, task));
1167 }
1168
1169 static void
event_filter_pid_sched_wakeup_probe_post(void * data,struct task_struct * task)1170 event_filter_pid_sched_wakeup_probe_post(void *data, struct task_struct *task)
1171 {
1172 struct trace_array *tr = data;
1173 struct trace_pid_list *no_pid_list;
1174 struct trace_pid_list *pid_list;
1175
1176 /* Nothing to do if we are not tracing */
1177 if (this_cpu_read(tr->array_buffer.data->ignore_pid))
1178 return;
1179
1180 pid_list = rcu_dereference_sched(tr->filtered_pids);
1181 no_pid_list = rcu_dereference_sched(tr->filtered_no_pids);
1182
1183 /* Set tracing if current is enabled */
1184 this_cpu_write(tr->array_buffer.data->ignore_pid,
1185 trace_ignore_this_task(pid_list, no_pid_list, current));
1186 }
1187
unregister_pid_events(struct trace_array * tr)1188 static void unregister_pid_events(struct trace_array *tr)
1189 {
1190 unregister_trace_sched_switch(event_filter_pid_sched_switch_probe_pre, tr);
1191 unregister_trace_sched_switch(event_filter_pid_sched_switch_probe_post, tr);
1192
1193 unregister_trace_sched_wakeup(event_filter_pid_sched_wakeup_probe_pre, tr);
1194 unregister_trace_sched_wakeup(event_filter_pid_sched_wakeup_probe_post, tr);
1195
1196 unregister_trace_sched_wakeup_new(event_filter_pid_sched_wakeup_probe_pre, tr);
1197 unregister_trace_sched_wakeup_new(event_filter_pid_sched_wakeup_probe_post, tr);
1198
1199 unregister_trace_sched_waking(event_filter_pid_sched_wakeup_probe_pre, tr);
1200 unregister_trace_sched_waking(event_filter_pid_sched_wakeup_probe_post, tr);
1201 }
1202
__ftrace_clear_event_pids(struct trace_array * tr,int type)1203 static void __ftrace_clear_event_pids(struct trace_array *tr, int type)
1204 {
1205 struct trace_pid_list *pid_list;
1206 struct trace_pid_list *no_pid_list;
1207 struct trace_event_file *file;
1208 int cpu;
1209
1210 pid_list = rcu_dereference_protected(tr->filtered_pids,
1211 lockdep_is_held(&event_mutex));
1212 no_pid_list = rcu_dereference_protected(tr->filtered_no_pids,
1213 lockdep_is_held(&event_mutex));
1214
1215 /* Make sure there's something to do */
1216 if (!pid_type_enabled(type, pid_list, no_pid_list))
1217 return;
1218
1219 if (!still_need_pid_events(type, pid_list, no_pid_list)) {
1220 unregister_pid_events(tr);
1221
1222 list_for_each_entry(file, &tr->events, list) {
1223 clear_bit(EVENT_FILE_FL_PID_FILTER_BIT, &file->flags);
1224 }
1225
1226 for_each_possible_cpu(cpu)
1227 per_cpu_ptr(tr->array_buffer.data, cpu)->ignore_pid = false;
1228 }
1229
1230 if (type & TRACE_PIDS)
1231 rcu_assign_pointer(tr->filtered_pids, NULL);
1232
1233 if (type & TRACE_NO_PIDS)
1234 rcu_assign_pointer(tr->filtered_no_pids, NULL);
1235
1236 /* Wait till all users are no longer using pid filtering */
1237 tracepoint_synchronize_unregister();
1238
1239 if ((type & TRACE_PIDS) && pid_list)
1240 trace_pid_list_free(pid_list);
1241
1242 if ((type & TRACE_NO_PIDS) && no_pid_list)
1243 trace_pid_list_free(no_pid_list);
1244 }
1245
ftrace_clear_event_pids(struct trace_array * tr,int type)1246 static void ftrace_clear_event_pids(struct trace_array *tr, int type)
1247 {
1248 mutex_lock(&event_mutex);
1249 __ftrace_clear_event_pids(tr, type);
1250 mutex_unlock(&event_mutex);
1251 }
1252
__put_system(struct event_subsystem * system)1253 static void __put_system(struct event_subsystem *system)
1254 {
1255 struct event_filter *filter = system->filter;
1256
1257 WARN_ON_ONCE(system_refcount(system) == 0);
1258 if (system_refcount_dec(system))
1259 return;
1260
1261 list_del(&system->list);
1262
1263 if (filter) {
1264 kfree(filter->filter_string);
1265 kfree(filter);
1266 }
1267 kfree_const(system->name);
1268 kfree(system);
1269 }
1270
__get_system(struct event_subsystem * system)1271 static void __get_system(struct event_subsystem *system)
1272 {
1273 WARN_ON_ONCE(system_refcount(system) == 0);
1274 system_refcount_inc(system);
1275 }
1276
__get_system_dir(struct trace_subsystem_dir * dir)1277 static void __get_system_dir(struct trace_subsystem_dir *dir)
1278 {
1279 WARN_ON_ONCE(dir->ref_count == 0);
1280 dir->ref_count++;
1281 __get_system(dir->subsystem);
1282 }
1283
__put_system_dir(struct trace_subsystem_dir * dir)1284 static void __put_system_dir(struct trace_subsystem_dir *dir)
1285 {
1286 WARN_ON_ONCE(dir->ref_count == 0);
1287 /* If the subsystem is about to be freed, the dir must be too */
1288 WARN_ON_ONCE(system_refcount(dir->subsystem) == 1 && dir->ref_count != 1);
1289
1290 __put_system(dir->subsystem);
1291 if (!--dir->ref_count)
1292 kfree(dir);
1293 }
1294
put_system(struct trace_subsystem_dir * dir)1295 static void put_system(struct trace_subsystem_dir *dir)
1296 {
1297 mutex_lock(&event_mutex);
1298 __put_system_dir(dir);
1299 mutex_unlock(&event_mutex);
1300 }
1301
remove_subsystem(struct trace_subsystem_dir * dir)1302 static void remove_subsystem(struct trace_subsystem_dir *dir)
1303 {
1304 if (!dir)
1305 return;
1306
1307 if (!--dir->nr_events) {
1308 eventfs_remove_dir(dir->ei);
1309 list_del(&dir->list);
1310 __put_system_dir(dir);
1311 }
1312 }
1313
event_file_get(struct trace_event_file * file)1314 void event_file_get(struct trace_event_file *file)
1315 {
1316 refcount_inc(&file->ref);
1317 }
1318
event_file_put(struct trace_event_file * file)1319 void event_file_put(struct trace_event_file *file)
1320 {
1321 if (WARN_ON_ONCE(!refcount_read(&file->ref))) {
1322 if (file->flags & EVENT_FILE_FL_FREED)
1323 kmem_cache_free(file_cachep, file);
1324 return;
1325 }
1326
1327 if (refcount_dec_and_test(&file->ref)) {
1328 /* Count should only go to zero when it is freed */
1329 if (WARN_ON_ONCE(!(file->flags & EVENT_FILE_FL_FREED)))
1330 return;
1331 kmem_cache_free(file_cachep, file);
1332 }
1333 }
1334
remove_event_file_dir(struct trace_event_file * file)1335 static void remove_event_file_dir(struct trace_event_file *file)
1336 {
1337 eventfs_remove_dir(file->ei);
1338 list_del(&file->list);
1339 remove_subsystem(file->system);
1340 free_event_filter(file->filter);
1341 file->flags |= EVENT_FILE_FL_FREED;
1342 event_file_put(file);
1343
1344 /* Wake up hist poll waiters to notice the EVENT_FILE_FL_FREED flag. */
1345 hist_poll_wakeup();
1346 }
1347
1348 /*
1349 * __ftrace_set_clr_event(NULL, NULL, NULL, set) will set/unset all events.
1350 */
1351 static int
__ftrace_set_clr_event_nolock(struct trace_array * tr,const char * match,const char * sub,const char * event,int set,const char * mod)1352 __ftrace_set_clr_event_nolock(struct trace_array *tr, const char *match,
1353 const char *sub, const char *event, int set,
1354 const char *mod)
1355 {
1356 struct trace_event_file *file;
1357 struct trace_event_call *call;
1358 char *module __free(kfree) = NULL;
1359 const char *name;
1360 int ret = -EINVAL;
1361 int eret = 0;
1362
1363 if (mod) {
1364 char *p;
1365
1366 module = kstrdup(mod, GFP_KERNEL);
1367 if (!module)
1368 return -ENOMEM;
1369
1370 /* Replace all '-' with '_' as that's what modules do */
1371 for (p = strchr(module, '-'); p; p = strchr(p + 1, '-'))
1372 *p = '_';
1373 }
1374
1375 list_for_each_entry(file, &tr->events, list) {
1376
1377 call = file->event_call;
1378
1379 /* If a module is specified, skip events that are not that module */
1380 if (module &&
1381 ((call->flags & TRACE_EVENT_FL_DYNAMIC) ||
1382 !call->module || strcmp(module_name(call->module), module)))
1383 continue;
1384
1385 name = trace_event_name(call);
1386
1387 if (!name || !call->class || !call->class->reg)
1388 continue;
1389
1390 if (call->flags & TRACE_EVENT_FL_IGNORE_ENABLE)
1391 continue;
1392
1393 if (match &&
1394 strcmp(match, name) != 0 &&
1395 strcmp(match, call->class->system) != 0)
1396 continue;
1397
1398 if (sub && strcmp(sub, call->class->system) != 0)
1399 continue;
1400
1401 if (event && strcmp(event, name) != 0)
1402 continue;
1403
1404 ret = ftrace_event_enable_disable(file, set);
1405
1406 /*
1407 * Save the first error and return that. Some events
1408 * may still have been enabled, but let the user
1409 * know that something went wrong.
1410 */
1411 if (ret && !eret)
1412 eret = ret;
1413
1414 ret = eret;
1415 }
1416
1417 /*
1418 * If this is a module setting and nothing was found,
1419 * check if the module was loaded. If it wasn't cache it.
1420 */
1421 if (module && ret == -EINVAL && !eret)
1422 ret = cache_mod(tr, module, set, match, sub, event);
1423
1424 return ret;
1425 }
1426
__ftrace_set_clr_event(struct trace_array * tr,const char * match,const char * sub,const char * event,int set,const char * mod)1427 static int __ftrace_set_clr_event(struct trace_array *tr, const char *match,
1428 const char *sub, const char *event, int set,
1429 const char *mod)
1430 {
1431 int ret;
1432
1433 if (trace_array_is_readonly(tr))
1434 return -EACCES;
1435
1436 mutex_lock(&event_mutex);
1437 ret = __ftrace_set_clr_event_nolock(tr, match, sub, event, set, mod);
1438 mutex_unlock(&event_mutex);
1439
1440 return ret;
1441 }
1442
ftrace_set_clr_event(struct trace_array * tr,char * buf,int set)1443 int ftrace_set_clr_event(struct trace_array *tr, char *buf, int set)
1444 {
1445 char *event = NULL, *sub = NULL, *match, *mod;
1446 int ret;
1447
1448 if (!tr)
1449 return -ENOENT;
1450
1451 /* Modules events can be appended with :mod:<module> */
1452 mod = strstr(buf, ":mod:");
1453 if (mod) {
1454 *mod = '\0';
1455 /* move to the module name */
1456 mod += 5;
1457 }
1458
1459 /*
1460 * The buf format can be <subsystem>:<event-name>
1461 * *:<event-name> means any event by that name.
1462 * :<event-name> is the same.
1463 *
1464 * <subsystem>:* means all events in that subsystem
1465 * <subsystem>: means the same.
1466 *
1467 * <name> (no ':') means all events in a subsystem with
1468 * the name <name> or any event that matches <name>
1469 */
1470
1471 match = strsep(&buf, ":");
1472 if (buf) {
1473 sub = match;
1474 event = buf;
1475 match = NULL;
1476
1477 if (!strlen(sub) || strcmp(sub, "*") == 0)
1478 sub = NULL;
1479 if (!strlen(event) || strcmp(event, "*") == 0)
1480 event = NULL;
1481 } else if (mod) {
1482 /* Allow wildcard for no length or star */
1483 if (!strlen(match) || strcmp(match, "*") == 0)
1484 match = NULL;
1485 }
1486
1487 ret = __ftrace_set_clr_event(tr, match, sub, event, set, mod);
1488
1489 /* Put back the colon to allow this to be called again */
1490 if (buf)
1491 *(buf - 1) = ':';
1492
1493 return ret;
1494 }
1495
1496 /**
1497 * trace_set_clr_event - enable or disable an event
1498 * @system: system name to match (NULL for any system)
1499 * @event: event name to match (NULL for all events, within system)
1500 * @set: 1 to enable, 0 to disable
1501 *
1502 * This is a way for other parts of the kernel to enable or disable
1503 * event recording.
1504 *
1505 * Returns 0 on success, -EINVAL if the parameters do not match any
1506 * registered events.
1507 */
trace_set_clr_event(const char * system,const char * event,int set)1508 int trace_set_clr_event(const char *system, const char *event, int set)
1509 {
1510 struct trace_array *tr = top_trace_array();
1511
1512 if (!tr)
1513 return -ENODEV;
1514
1515 return __ftrace_set_clr_event(tr, NULL, system, event, set, NULL);
1516 }
1517 EXPORT_SYMBOL_GPL(trace_set_clr_event);
1518
1519 /**
1520 * trace_array_set_clr_event - enable or disable an event for a trace array.
1521 * @tr: concerned trace array.
1522 * @system: system name to match (NULL for any system)
1523 * @event: event name to match (NULL for all events, within system)
1524 * @enable: true to enable, false to disable
1525 *
1526 * This is a way for other parts of the kernel to enable or disable
1527 * event recording.
1528 *
1529 * Returns 0 on success, -EINVAL if the parameters do not match any
1530 * registered events.
1531 */
trace_array_set_clr_event(struct trace_array * tr,const char * system,const char * event,bool enable)1532 int trace_array_set_clr_event(struct trace_array *tr, const char *system,
1533 const char *event, bool enable)
1534 {
1535 int set;
1536
1537 if (!tr)
1538 return -ENOENT;
1539
1540 set = (enable == true) ? 1 : 0;
1541 return __ftrace_set_clr_event(tr, NULL, system, event, set, NULL);
1542 }
1543 EXPORT_SYMBOL_GPL(trace_array_set_clr_event);
1544
1545 /* 128 should be much more than enough */
1546 #define EVENT_BUF_SIZE 127
1547
1548 static ssize_t
ftrace_event_write(struct file * file,const char __user * ubuf,size_t cnt,loff_t * ppos)1549 ftrace_event_write(struct file *file, const char __user *ubuf,
1550 size_t cnt, loff_t *ppos)
1551 {
1552 struct trace_parser parser;
1553 struct seq_file *m = file->private_data;
1554 struct trace_array *tr = m->private;
1555 ssize_t read, ret;
1556
1557 if (!cnt)
1558 return 0;
1559
1560 ret = tracing_update_buffers(tr);
1561 if (ret < 0)
1562 return ret;
1563
1564 if (trace_parser_get_init(&parser, EVENT_BUF_SIZE + 1))
1565 return -ENOMEM;
1566
1567 read = trace_get_user(&parser, ubuf, cnt, ppos);
1568
1569 if (read >= 0 && trace_parser_loaded((&parser))) {
1570 int set = 1;
1571
1572 if (*parser.buffer == '!')
1573 set = 0;
1574
1575 ret = ftrace_set_clr_event(tr, parser.buffer + !set, set);
1576 if (ret)
1577 goto out_put;
1578 }
1579
1580 ret = read;
1581
1582 out_put:
1583 trace_parser_put(&parser);
1584
1585 return ret;
1586 }
1587
1588 static void *
t_next(struct seq_file * m,void * v,loff_t * pos)1589 t_next(struct seq_file *m, void *v, loff_t *pos)
1590 {
1591 struct trace_event_file *file = v;
1592 struct trace_event_call *call;
1593 struct trace_array *tr = m->private;
1594
1595 (*pos)++;
1596
1597 list_for_each_entry_continue(file, &tr->events, list) {
1598 call = file->event_call;
1599 /*
1600 * The ftrace subsystem is for showing formats only.
1601 * They can not be enabled or disabled via the event files.
1602 */
1603 if (call->class && call->class->reg &&
1604 !(call->flags & TRACE_EVENT_FL_IGNORE_ENABLE))
1605 return file;
1606 }
1607
1608 return NULL;
1609 }
1610
t_start(struct seq_file * m,loff_t * pos)1611 static void *t_start(struct seq_file *m, loff_t *pos)
1612 {
1613 struct trace_event_file *file;
1614 struct trace_array *tr = m->private;
1615 loff_t l;
1616
1617 mutex_lock(&event_mutex);
1618
1619 file = list_entry(&tr->events, struct trace_event_file, list);
1620 for (l = 0; l <= *pos; ) {
1621 file = t_next(m, file, &l);
1622 if (!file)
1623 break;
1624 }
1625 return file;
1626 }
1627
1628 enum set_event_iter_type {
1629 SET_EVENT_FILE,
1630 SET_EVENT_MOD,
1631 };
1632
1633 struct set_event_iter {
1634 enum set_event_iter_type type;
1635 union {
1636 struct trace_event_file *file;
1637 struct event_mod_load *event_mod;
1638 };
1639 };
1640
1641 static void *
s_next(struct seq_file * m,void * v,loff_t * pos)1642 s_next(struct seq_file *m, void *v, loff_t *pos)
1643 {
1644 struct set_event_iter *iter = v;
1645 struct trace_event_file *file;
1646 struct trace_array *tr = m->private;
1647
1648 (*pos)++;
1649
1650 if (iter->type == SET_EVENT_FILE) {
1651 file = iter->file;
1652 list_for_each_entry_continue(file, &tr->events, list) {
1653 if (file->flags & EVENT_FILE_FL_ENABLED) {
1654 iter->file = file;
1655 return iter;
1656 }
1657 }
1658 #ifdef CONFIG_MODULES
1659 iter->type = SET_EVENT_MOD;
1660 iter->event_mod = list_entry(&tr->mod_events, struct event_mod_load, list);
1661 #endif
1662 }
1663
1664 #ifdef CONFIG_MODULES
1665 list_for_each_entry_continue(iter->event_mod, &tr->mod_events, list)
1666 return iter;
1667 #endif
1668
1669 /*
1670 * The iter is allocated in s_start() and passed via the 'v'
1671 * parameter. To stop the iterator, NULL must be returned. But
1672 * the return value is what the 'v' parameter in s_stop() receives
1673 * and frees. Free iter here as it will no longer be used.
1674 */
1675 kfree(iter);
1676 return NULL;
1677 }
1678
s_start(struct seq_file * m,loff_t * pos)1679 static void *s_start(struct seq_file *m, loff_t *pos)
1680 {
1681 struct trace_array *tr = m->private;
1682 struct set_event_iter *iter;
1683 loff_t l;
1684
1685 iter = kzalloc_obj(*iter);
1686 mutex_lock(&event_mutex);
1687 if (!iter)
1688 return NULL;
1689
1690 iter->type = SET_EVENT_FILE;
1691 iter->file = list_entry(&tr->events, struct trace_event_file, list);
1692
1693 for (l = 0; l <= *pos; ) {
1694 iter = s_next(m, iter, &l);
1695 if (!iter)
1696 break;
1697 }
1698 return iter;
1699 }
1700
t_show(struct seq_file * m,void * v)1701 static int t_show(struct seq_file *m, void *v)
1702 {
1703 struct trace_event_file *file = v;
1704 struct trace_event_call *call = file->event_call;
1705
1706 if (strcmp(call->class->system, TRACE_SYSTEM) != 0)
1707 seq_printf(m, "%s:", call->class->system);
1708 seq_printf(m, "%s\n", trace_event_name(call));
1709
1710 return 0;
1711 }
1712
t_stop(struct seq_file * m,void * p)1713 static void t_stop(struct seq_file *m, void *p)
1714 {
1715 mutex_unlock(&event_mutex);
1716 }
1717
get_call_len(struct trace_event_call * call)1718 static int get_call_len(struct trace_event_call *call)
1719 {
1720 int len;
1721
1722 /* Get the length of "<system>:<event>" */
1723 len = strlen(call->class->system) + 1;
1724 len += strlen(trace_event_name(call));
1725
1726 /* Set the index to 32 bytes to separate event from data */
1727 return len >= 32 ? 1 : 32 - len;
1728 }
1729
1730 /**
1731 * t_show_filters - seq_file callback to display active event filters
1732 * @m: The seq_file interface for formatted output
1733 * @v: The current trace_event_file being iterated
1734 *
1735 * Identifies and prints active filters for the current event file in the
1736 * iteration. If a filter is applied to the current event and, if so,
1737 * prints the system name, event name, and the filter string.
1738 */
t_show_filters(struct seq_file * m,void * v)1739 static int t_show_filters(struct seq_file *m, void *v)
1740 {
1741 struct trace_event_file *file = v;
1742 struct trace_event_call *call = file->event_call;
1743 struct event_filter *filter;
1744 int len;
1745
1746 guard(rcu)();
1747 filter = rcu_dereference(file->filter);
1748 if (!filter || !filter->filter_string)
1749 return 0;
1750
1751 len = get_call_len(call);
1752
1753 seq_printf(m, "%s:%s%*s%s\n", call->class->system,
1754 trace_event_name(call), len, "", filter->filter_string);
1755
1756 return 0;
1757 }
1758
1759 /**
1760 * t_show_triggers - seq_file callback to display active event triggers
1761 * @m: The seq_file interface for formatted output
1762 * @v: The current trace_event_file being iterated
1763 *
1764 * Iterates through the trigger list of the current event file and prints
1765 * each active trigger's configuration using its associated print
1766 * operation.
1767 */
t_show_triggers(struct seq_file * m,void * v)1768 static int t_show_triggers(struct seq_file *m, void *v)
1769 {
1770 struct trace_event_file *file = v;
1771 struct trace_event_call *call = file->event_call;
1772 struct event_trigger_data *data;
1773 int len;
1774
1775 /*
1776 * The event_mutex is held by t_start(), protecting the
1777 * file->triggers list traversal.
1778 */
1779 if (list_empty(&file->triggers))
1780 return 0;
1781
1782 len = get_call_len(call);
1783
1784 list_for_each_entry_rcu(data, &file->triggers, list) {
1785 seq_printf(m, "%s:%s%*s", call->class->system,
1786 trace_event_name(call), len, "");
1787
1788 data->cmd_ops->print(m, data);
1789 }
1790
1791 return 0;
1792 }
1793
1794 #ifdef CONFIG_MODULES
s_show(struct seq_file * m,void * v)1795 static int s_show(struct seq_file *m, void *v)
1796 {
1797 struct set_event_iter *iter = v;
1798 const char *system;
1799 const char *event;
1800
1801 if (iter->type == SET_EVENT_FILE)
1802 return t_show(m, iter->file);
1803
1804 /* When match is set, system and event are not */
1805 if (iter->event_mod->match) {
1806 seq_printf(m, "%s:mod:%s\n", iter->event_mod->match,
1807 iter->event_mod->module);
1808 return 0;
1809 }
1810
1811 system = iter->event_mod->system ? : "*";
1812 event = iter->event_mod->event ? : "*";
1813
1814 seq_printf(m, "%s:%s:mod:%s\n", system, event, iter->event_mod->module);
1815
1816 return 0;
1817 }
1818 #else /* CONFIG_MODULES */
s_show(struct seq_file * m,void * v)1819 static int s_show(struct seq_file *m, void *v)
1820 {
1821 struct set_event_iter *iter = v;
1822
1823 return t_show(m, iter->file);
1824 }
1825 #endif
1826
s_stop(struct seq_file * m,void * v)1827 static void s_stop(struct seq_file *m, void *v)
1828 {
1829 kfree(v);
1830 t_stop(m, NULL);
1831 }
1832
1833 static void *
__next(struct seq_file * m,void * v,loff_t * pos,int type)1834 __next(struct seq_file *m, void *v, loff_t *pos, int type)
1835 {
1836 struct trace_array *tr = m->private;
1837 struct trace_pid_list *pid_list;
1838
1839 if (type == TRACE_PIDS)
1840 pid_list = rcu_dereference_sched(tr->filtered_pids);
1841 else
1842 pid_list = rcu_dereference_sched(tr->filtered_no_pids);
1843
1844 return trace_pid_next(pid_list, v, pos);
1845 }
1846
1847 static void *
p_next(struct seq_file * m,void * v,loff_t * pos)1848 p_next(struct seq_file *m, void *v, loff_t *pos)
1849 {
1850 return __next(m, v, pos, TRACE_PIDS);
1851 }
1852
1853 static void *
np_next(struct seq_file * m,void * v,loff_t * pos)1854 np_next(struct seq_file *m, void *v, loff_t *pos)
1855 {
1856 return __next(m, v, pos, TRACE_NO_PIDS);
1857 }
1858
__start(struct seq_file * m,loff_t * pos,int type)1859 static void *__start(struct seq_file *m, loff_t *pos, int type)
1860 __acquires(RCU)
1861 {
1862 struct trace_pid_list *pid_list;
1863 struct trace_array *tr = m->private;
1864
1865 /*
1866 * Grab the mutex, to keep calls to p_next() having the same
1867 * tr->filtered_pids as p_start() has.
1868 * If we just passed the tr->filtered_pids around, then RCU would
1869 * have been enough, but doing that makes things more complex.
1870 */
1871 mutex_lock(&event_mutex);
1872 rcu_read_lock_sched();
1873
1874 if (type == TRACE_PIDS)
1875 pid_list = rcu_dereference_sched(tr->filtered_pids);
1876 else
1877 pid_list = rcu_dereference_sched(tr->filtered_no_pids);
1878
1879 if (!pid_list)
1880 return NULL;
1881
1882 return trace_pid_start(pid_list, pos);
1883 }
1884
p_start(struct seq_file * m,loff_t * pos)1885 static void *p_start(struct seq_file *m, loff_t *pos)
1886 __acquires(RCU)
1887 {
1888 return __start(m, pos, TRACE_PIDS);
1889 }
1890
np_start(struct seq_file * m,loff_t * pos)1891 static void *np_start(struct seq_file *m, loff_t *pos)
1892 __acquires(RCU)
1893 {
1894 return __start(m, pos, TRACE_NO_PIDS);
1895 }
1896
p_stop(struct seq_file * m,void * p)1897 static void p_stop(struct seq_file *m, void *p)
1898 __releases(RCU)
1899 {
1900 rcu_read_unlock_sched();
1901 mutex_unlock(&event_mutex);
1902 }
1903
1904 static ssize_t
event_enable_read(struct file * filp,char __user * ubuf,size_t cnt,loff_t * ppos)1905 event_enable_read(struct file *filp, char __user *ubuf, size_t cnt,
1906 loff_t *ppos)
1907 {
1908 struct trace_event_file *file;
1909 unsigned long flags;
1910 char buf[4] = "0";
1911
1912 mutex_lock(&event_mutex);
1913 file = event_file_file(filp);
1914 if (likely(file))
1915 flags = file->flags;
1916 mutex_unlock(&event_mutex);
1917
1918 if (!file)
1919 return -ENODEV;
1920
1921 if (flags & EVENT_FILE_FL_ENABLED &&
1922 !(flags & EVENT_FILE_FL_SOFT_DISABLED))
1923 strcpy(buf, "1");
1924
1925 if (atomic_read(&file->sm_ref) != 0)
1926 strcat(buf, "*");
1927
1928 strcat(buf, "\n");
1929
1930 return simple_read_from_buffer(ubuf, cnt, ppos, buf, strlen(buf));
1931 }
1932
1933 static ssize_t
event_enable_write(struct file * filp,const char __user * ubuf,size_t cnt,loff_t * ppos)1934 event_enable_write(struct file *filp, const char __user *ubuf, size_t cnt,
1935 loff_t *ppos)
1936 {
1937 struct trace_event_file *file;
1938 unsigned long val;
1939 int ret;
1940
1941 ret = kstrtoul_from_user(ubuf, cnt, 10, &val);
1942 if (ret)
1943 return ret;
1944
1945 guard(mutex)(&event_mutex);
1946
1947 switch (val) {
1948 case 0:
1949 case 1:
1950 file = event_file_file(filp);
1951 if (!file)
1952 return -ENODEV;
1953 ret = tracing_update_buffers(file->tr);
1954 if (ret < 0)
1955 return ret;
1956 ret = ftrace_event_enable_disable(file, val);
1957 if (ret < 0)
1958 return ret;
1959 break;
1960
1961 default:
1962 return -EINVAL;
1963 }
1964
1965 *ppos += cnt;
1966
1967 return cnt;
1968 }
1969
1970 /*
1971 * Returns:
1972 * 0 : no events exist?
1973 * 1 : all events are disabled
1974 * 2 : all events are enabled
1975 * 3 : some events are enabled and some are enabled
1976 */
trace_events_enabled(struct trace_array * tr,const char * system)1977 int trace_events_enabled(struct trace_array *tr, const char *system)
1978 {
1979 struct trace_event_call *call;
1980 struct trace_event_file *file;
1981 int set = 0;
1982
1983 guard(mutex)(&event_mutex);
1984
1985 list_for_each_entry(file, &tr->events, list) {
1986 call = file->event_call;
1987 if ((call->flags & TRACE_EVENT_FL_IGNORE_ENABLE) ||
1988 !trace_event_name(call) || !call->class || !call->class->reg)
1989 continue;
1990
1991 if (system && strcmp(call->class->system, system) != 0)
1992 continue;
1993
1994 /*
1995 * We need to find out if all the events are set
1996 * or if all events or cleared, or if we have
1997 * a mixture.
1998 */
1999 set |= (1 << !!(file->flags & EVENT_FILE_FL_ENABLED));
2000
2001 /*
2002 * If we have a mixture, no need to look further.
2003 */
2004 if (set == 3)
2005 break;
2006 }
2007
2008 return set;
2009 }
2010
2011 static ssize_t
system_enable_read(struct file * filp,char __user * ubuf,size_t cnt,loff_t * ppos)2012 system_enable_read(struct file *filp, char __user *ubuf, size_t cnt,
2013 loff_t *ppos)
2014 {
2015 const char set_to_char[4] = { '?', '0', '1', 'X' };
2016 struct trace_subsystem_dir *dir = filp->private_data;
2017 struct event_subsystem *system = dir->subsystem;
2018 struct trace_array *tr = dir->tr;
2019 char buf[2];
2020 int set;
2021 int ret;
2022
2023 set = trace_events_enabled(tr, system ? system->name : NULL);
2024
2025 buf[0] = set_to_char[set];
2026 buf[1] = '\n';
2027
2028 ret = simple_read_from_buffer(ubuf, cnt, ppos, buf, 2);
2029
2030 return ret;
2031 }
2032
2033 static ssize_t
system_enable_write(struct file * filp,const char __user * ubuf,size_t cnt,loff_t * ppos)2034 system_enable_write(struct file *filp, const char __user *ubuf, size_t cnt,
2035 loff_t *ppos)
2036 {
2037 struct trace_subsystem_dir *dir = filp->private_data;
2038 struct event_subsystem *system = dir->subsystem;
2039 const char *name = NULL;
2040 unsigned long val;
2041 ssize_t ret;
2042
2043 ret = kstrtoul_from_user(ubuf, cnt, 10, &val);
2044 if (ret)
2045 return ret;
2046
2047 ret = tracing_update_buffers(dir->tr);
2048 if (ret < 0)
2049 return ret;
2050
2051 if (val != 0 && val != 1)
2052 return -EINVAL;
2053
2054 /*
2055 * Opening of "enable" adds a ref count to system,
2056 * so the name is safe to use.
2057 */
2058 if (system)
2059 name = system->name;
2060
2061 ret = __ftrace_set_clr_event(dir->tr, NULL, name, NULL, val, NULL);
2062 if (ret)
2063 goto out;
2064
2065 ret = cnt;
2066
2067 out:
2068 *ppos += cnt;
2069
2070 return ret;
2071 }
2072
2073 enum {
2074 FORMAT_HEADER = 1,
2075 FORMAT_FIELD_SEPERATOR = 2,
2076 FORMAT_PRINTFMT = 3,
2077 };
2078
f_next(struct seq_file * m,void * v,loff_t * pos)2079 static void *f_next(struct seq_file *m, void *v, loff_t *pos)
2080 {
2081 struct trace_event_file *file = event_file_data(m->private);
2082 struct trace_event_call *call = file->event_call;
2083 struct list_head *common_head = &ftrace_common_fields;
2084 struct list_head *head = trace_get_fields(call);
2085 struct list_head *node = v;
2086
2087 (*pos)++;
2088
2089 switch ((unsigned long)v) {
2090 case FORMAT_HEADER:
2091 node = common_head;
2092 break;
2093
2094 case FORMAT_FIELD_SEPERATOR:
2095 node = head;
2096 break;
2097
2098 case FORMAT_PRINTFMT:
2099 /* all done */
2100 return NULL;
2101 }
2102
2103 node = node->prev;
2104 if (node == common_head)
2105 return (void *)FORMAT_FIELD_SEPERATOR;
2106 else if (node == head)
2107 return (void *)FORMAT_PRINTFMT;
2108 else
2109 return node;
2110 }
2111
f_show(struct seq_file * m,void * v)2112 static int f_show(struct seq_file *m, void *v)
2113 {
2114 struct trace_event_file *file = event_file_data(m->private);
2115 struct trace_event_call *call = file->event_call;
2116 struct ftrace_event_field *field;
2117 const char *array_descriptor;
2118
2119 switch ((unsigned long)v) {
2120 case FORMAT_HEADER:
2121 seq_printf(m, "name: %s\n", trace_event_name(call));
2122 seq_printf(m, "ID: %d\n", call->event.type);
2123 seq_puts(m, "format:\n");
2124 return 0;
2125
2126 case FORMAT_FIELD_SEPERATOR:
2127 seq_putc(m, '\n');
2128 return 0;
2129
2130 case FORMAT_PRINTFMT:
2131 seq_printf(m, "\nprint fmt: %s\n",
2132 call->print_fmt);
2133 return 0;
2134 }
2135
2136 field = list_entry(v, struct ftrace_event_field, link);
2137 /*
2138 * Smartly shows the array type(except dynamic array).
2139 * Normal:
2140 * field:TYPE VAR
2141 * If TYPE := TYPE[LEN], it is shown:
2142 * field:TYPE VAR[LEN]
2143 */
2144 array_descriptor = strchr(field->type, '[');
2145
2146 if (str_has_prefix(field->type, "__data_loc"))
2147 array_descriptor = NULL;
2148
2149 if (!array_descriptor)
2150 seq_printf(m, "\tfield:%s %s;\toffset:%u;\tsize:%u;\tsigned:%d;\n",
2151 field->type, field->name, field->offset,
2152 field->size, !!field->is_signed);
2153 else if (field->len)
2154 seq_printf(m, "\tfield:%.*s %s[%d];\toffset:%u;\tsize:%u;\tsigned:%d;\n",
2155 (int)(array_descriptor - field->type),
2156 field->type, field->name,
2157 field->len, field->offset,
2158 field->size, !!field->is_signed);
2159 else
2160 seq_printf(m, "\tfield:%.*s %s[];\toffset:%u;\tsize:%u;\tsigned:%d;\n",
2161 (int)(array_descriptor - field->type),
2162 field->type, field->name,
2163 field->offset, field->size, !!field->is_signed);
2164
2165 return 0;
2166 }
2167
f_start(struct seq_file * m,loff_t * pos)2168 static void *f_start(struct seq_file *m, loff_t *pos)
2169 {
2170 struct trace_event_file *file;
2171 void *p = (void *)FORMAT_HEADER;
2172 loff_t l = 0;
2173
2174 /* ->stop() is called even if ->start() fails */
2175 mutex_lock(&event_mutex);
2176 file = event_file_file(m->private);
2177 if (!file)
2178 return ERR_PTR(-ENODEV);
2179
2180 while (l < *pos && p)
2181 p = f_next(m, p, &l);
2182
2183 return p;
2184 }
2185
f_stop(struct seq_file * m,void * p)2186 static void f_stop(struct seq_file *m, void *p)
2187 {
2188 mutex_unlock(&event_mutex);
2189 }
2190
2191 static const struct seq_operations trace_format_seq_ops = {
2192 .start = f_start,
2193 .next = f_next,
2194 .stop = f_stop,
2195 .show = f_show,
2196 };
2197
trace_format_open(struct inode * inode,struct file * file)2198 static int trace_format_open(struct inode *inode, struct file *file)
2199 {
2200 struct seq_file *m;
2201 int ret;
2202
2203 /* Do we want to hide event format files on tracefs lockdown? */
2204
2205 ret = seq_open(file, &trace_format_seq_ops);
2206 if (ret < 0)
2207 return ret;
2208
2209 m = file->private_data;
2210 m->private = file;
2211
2212 return 0;
2213 }
2214
2215 #ifdef CONFIG_PERF_EVENTS
2216 static ssize_t
event_id_read(struct file * filp,char __user * ubuf,size_t cnt,loff_t * ppos)2217 event_id_read(struct file *filp, char __user *ubuf, size_t cnt, loff_t *ppos)
2218 {
2219 /* id is directly in i_private and available for inode's lifetime. */
2220 int id = (long)file_inode(filp)->i_private;
2221 char buf[32];
2222 int len;
2223
2224 WARN_ON(!id);
2225
2226 len = sprintf(buf, "%d\n", id);
2227
2228 return simple_read_from_buffer(ubuf, cnt, ppos, buf, len);
2229 }
2230 #endif
2231
2232 #ifdef CONFIG_BPF_EVENTS
2233 static ssize_t
event_btf_ids_read(struct file * filp,char __user * ubuf,size_t cnt,loff_t * ppos)2234 event_btf_ids_read(struct file *filp, char __user *ubuf, size_t cnt, loff_t *ppos)
2235 {
2236 struct trace_event_file *file;
2237 struct trace_event_call *call;
2238 const struct btf_type *t;
2239 struct module *mod = NULL;
2240 u32 raw_id = 0, tp_id = 0, obj_id = 0;
2241 const u32 *ids;
2242 struct btf *btf;
2243 char buf[128];
2244 int len;
2245
2246 /* Module unload could free call->class and ids[] mid-read. */
2247 scoped_guard(mutex, &event_mutex) {
2248 file = event_file_file(filp);
2249 if (!file)
2250 return -ENODEV;
2251
2252 call = file->event_call;
2253 ids = call->class->btf_ids;
2254 if (!ids)
2255 return -ENOENT;
2256 if (!(call->flags & TRACE_EVENT_FL_DYNAMIC))
2257 mod = (struct module *)call->module;
2258
2259 btf = btf_get_module_btf(mod);
2260 if (IS_ERR_OR_NULL(btf))
2261 return -ENOENT;
2262
2263 /* Module-local ids in ids[] need base+local relocation. */
2264 tp_id = btf_relocate_id(btf, ids[1]);
2265
2266 /*
2267 * Without FL_TRACEPOINT the dispatcher is shared (e.g. all
2268 * per-syscall events fan out from __bpf_trace_sys_enter), so
2269 * raw_btf_id has no per-event attach point — report 0.
2270 */
2271 if (call->flags & TRACE_EVENT_FL_TRACEPOINT) {
2272 t = btf_type_by_id(btf, btf_relocate_id(btf, ids[0]));
2273 raw_id = t ? t->type : 0;
2274 }
2275 obj_id = btf_obj_id(btf);
2276 btf_put(btf);
2277 }
2278
2279 len = scnprintf(buf, sizeof(buf),
2280 "btf_obj_id: %u\nraw_btf_id: %u\ntp_btf_id: %u\n",
2281 obj_id, raw_id, tp_id);
2282
2283 return simple_read_from_buffer(ubuf, cnt, ppos, buf, len);
2284 }
2285 #endif
2286
2287 static ssize_t
event_filter_read(struct file * filp,char __user * ubuf,size_t cnt,loff_t * ppos)2288 event_filter_read(struct file *filp, char __user *ubuf, size_t cnt,
2289 loff_t *ppos)
2290 {
2291 struct trace_event_file *file;
2292 struct trace_seq *s;
2293 int r = -ENODEV;
2294
2295 if (*ppos)
2296 return 0;
2297
2298 s = kmalloc_obj(*s);
2299
2300 if (!s)
2301 return -ENOMEM;
2302
2303 trace_seq_init(s);
2304
2305 mutex_lock(&event_mutex);
2306 file = event_file_file(filp);
2307 if (file)
2308 print_event_filter(file, s);
2309 mutex_unlock(&event_mutex);
2310
2311 if (file)
2312 r = simple_read_from_buffer(ubuf, cnt, ppos,
2313 s->buffer, trace_seq_used(s));
2314
2315 kfree(s);
2316
2317 return r;
2318 }
2319
2320 static ssize_t
event_filter_write(struct file * filp,const char __user * ubuf,size_t cnt,loff_t * ppos)2321 event_filter_write(struct file *filp, const char __user *ubuf, size_t cnt,
2322 loff_t *ppos)
2323 {
2324 struct trace_event_file *file;
2325 char *buf;
2326 int err = -ENODEV;
2327
2328 if (cnt >= PAGE_SIZE)
2329 return -EINVAL;
2330
2331 buf = memdup_user_nul(ubuf, cnt);
2332 if (IS_ERR(buf))
2333 return PTR_ERR(buf);
2334
2335 mutex_lock(&event_mutex);
2336 file = event_file_file(filp);
2337 if (file)
2338 err = apply_event_filter(file, buf);
2339 mutex_unlock(&event_mutex);
2340
2341 kfree(buf);
2342 if (err < 0)
2343 return err;
2344
2345 *ppos += cnt;
2346
2347 return cnt;
2348 }
2349
2350 static LIST_HEAD(event_subsystems);
2351
subsystem_open(struct inode * inode,struct file * filp)2352 static int subsystem_open(struct inode *inode, struct file *filp)
2353 {
2354 struct trace_subsystem_dir *dir = NULL, *iter_dir;
2355 struct trace_array *tr = NULL, *iter_tr;
2356 struct event_subsystem *system = NULL;
2357 int ret;
2358
2359 if (unlikely(tracing_disabled))
2360 return -ENODEV;
2361
2362 /* Make sure the system still exists */
2363 mutex_lock(&event_mutex);
2364 mutex_lock(&trace_types_lock);
2365 list_for_each_entry(iter_tr, &ftrace_trace_arrays, list) {
2366 list_for_each_entry(iter_dir, &iter_tr->systems, list) {
2367 if (iter_dir == inode->i_private) {
2368 /* Don't open systems with no events */
2369 tr = iter_tr;
2370 dir = iter_dir;
2371 if (dir->nr_events) {
2372 __get_system_dir(dir);
2373 system = dir->subsystem;
2374 }
2375 goto exit_loop;
2376 }
2377 }
2378 }
2379 exit_loop:
2380 mutex_unlock(&trace_types_lock);
2381 mutex_unlock(&event_mutex);
2382
2383 if (!system)
2384 return -ENODEV;
2385
2386 /* Still need to increment the ref count of the system */
2387 if (trace_array_get(tr) < 0) {
2388 put_system(dir);
2389 return -ENODEV;
2390 }
2391
2392 ret = tracing_open_generic(inode, filp);
2393 if (ret < 0) {
2394 trace_array_put(tr);
2395 put_system(dir);
2396 }
2397
2398 return ret;
2399 }
2400
system_tr_open(struct inode * inode,struct file * filp)2401 static int system_tr_open(struct inode *inode, struct file *filp)
2402 {
2403 struct trace_subsystem_dir *dir;
2404 struct trace_array *tr = inode->i_private;
2405 int ret;
2406
2407 /* Make a temporary dir that has no system but points to tr */
2408 dir = kzalloc_obj(*dir);
2409 if (!dir)
2410 return -ENOMEM;
2411
2412 ret = tracing_open_generic_tr(inode, filp);
2413 if (ret < 0) {
2414 kfree(dir);
2415 return ret;
2416 }
2417 dir->tr = tr;
2418 filp->private_data = dir;
2419
2420 return 0;
2421 }
2422
subsystem_release(struct inode * inode,struct file * file)2423 static int subsystem_release(struct inode *inode, struct file *file)
2424 {
2425 struct trace_subsystem_dir *dir = file->private_data;
2426
2427 trace_array_put(dir->tr);
2428
2429 /*
2430 * If dir->subsystem is NULL, then this is a temporary
2431 * descriptor that was made for a trace_array to enable
2432 * all subsystems.
2433 */
2434 if (dir->subsystem)
2435 put_system(dir);
2436 else
2437 kfree(dir);
2438
2439 return 0;
2440 }
2441
2442 static ssize_t
subsystem_filter_read(struct file * filp,char __user * ubuf,size_t cnt,loff_t * ppos)2443 subsystem_filter_read(struct file *filp, char __user *ubuf, size_t cnt,
2444 loff_t *ppos)
2445 {
2446 struct trace_subsystem_dir *dir = filp->private_data;
2447 struct event_subsystem *system = dir->subsystem;
2448 struct trace_seq *s;
2449 int r;
2450
2451 if (*ppos)
2452 return 0;
2453
2454 s = kmalloc_obj(*s);
2455 if (!s)
2456 return -ENOMEM;
2457
2458 trace_seq_init(s);
2459
2460 print_subsystem_event_filter(system, s);
2461 r = simple_read_from_buffer(ubuf, cnt, ppos,
2462 s->buffer, trace_seq_used(s));
2463
2464 kfree(s);
2465
2466 return r;
2467 }
2468
2469 static ssize_t
subsystem_filter_write(struct file * filp,const char __user * ubuf,size_t cnt,loff_t * ppos)2470 subsystem_filter_write(struct file *filp, const char __user *ubuf, size_t cnt,
2471 loff_t *ppos)
2472 {
2473 struct trace_subsystem_dir *dir = filp->private_data;
2474 char *buf;
2475 int err;
2476
2477 if (cnt >= PAGE_SIZE)
2478 return -EINVAL;
2479
2480 buf = memdup_user_nul(ubuf, cnt);
2481 if (IS_ERR(buf))
2482 return PTR_ERR(buf);
2483
2484 err = apply_subsystem_event_filter(dir, buf);
2485 kfree(buf);
2486 if (err < 0)
2487 return err;
2488
2489 *ppos += cnt;
2490
2491 return cnt;
2492 }
2493
2494 static ssize_t
show_header_page_file(struct file * filp,char __user * ubuf,size_t cnt,loff_t * ppos)2495 show_header_page_file(struct file *filp, char __user *ubuf, size_t cnt, loff_t *ppos)
2496 {
2497 struct trace_array *tr = filp->private_data;
2498 struct trace_seq *s;
2499 int r;
2500
2501 if (*ppos)
2502 return 0;
2503
2504 s = kmalloc_obj(*s);
2505 if (!s)
2506 return -ENOMEM;
2507
2508 trace_seq_init(s);
2509
2510 ring_buffer_print_page_header(tr->array_buffer.buffer, s);
2511 r = simple_read_from_buffer(ubuf, cnt, ppos,
2512 s->buffer, trace_seq_used(s));
2513
2514 kfree(s);
2515
2516 return r;
2517 }
2518
2519 static ssize_t
show_header_event_file(struct file * filp,char __user * ubuf,size_t cnt,loff_t * ppos)2520 show_header_event_file(struct file *filp, char __user *ubuf, size_t cnt, loff_t *ppos)
2521 {
2522 struct trace_seq *s;
2523 int r;
2524
2525 if (*ppos)
2526 return 0;
2527
2528 s = kmalloc_obj(*s);
2529 if (!s)
2530 return -ENOMEM;
2531
2532 trace_seq_init(s);
2533
2534 ring_buffer_print_entry_header(s);
2535 r = simple_read_from_buffer(ubuf, cnt, ppos,
2536 s->buffer, trace_seq_used(s));
2537
2538 kfree(s);
2539
2540 return r;
2541 }
2542
ignore_task_cpu(void * data)2543 static void ignore_task_cpu(void *data)
2544 {
2545 struct trace_array *tr = data;
2546 struct trace_pid_list *pid_list;
2547 struct trace_pid_list *no_pid_list;
2548
2549 /*
2550 * This function is called by on_each_cpu() while the
2551 * event_mutex is held.
2552 */
2553 pid_list = rcu_dereference_protected(tr->filtered_pids,
2554 mutex_is_locked(&event_mutex));
2555 no_pid_list = rcu_dereference_protected(tr->filtered_no_pids,
2556 mutex_is_locked(&event_mutex));
2557
2558 this_cpu_write(tr->array_buffer.data->ignore_pid,
2559 trace_ignore_this_task(pid_list, no_pid_list, current));
2560 }
2561
register_pid_events(struct trace_array * tr)2562 static void register_pid_events(struct trace_array *tr)
2563 {
2564 /*
2565 * Register a probe that is called before all other probes
2566 * to set ignore_pid if next or prev do not match.
2567 * Register a probe this is called after all other probes
2568 * to only keep ignore_pid set if next pid matches.
2569 */
2570 register_trace_prio_sched_switch(event_filter_pid_sched_switch_probe_pre,
2571 tr, INT_MAX);
2572 register_trace_prio_sched_switch(event_filter_pid_sched_switch_probe_post,
2573 tr, 0);
2574
2575 register_trace_prio_sched_wakeup(event_filter_pid_sched_wakeup_probe_pre,
2576 tr, INT_MAX);
2577 register_trace_prio_sched_wakeup(event_filter_pid_sched_wakeup_probe_post,
2578 tr, 0);
2579
2580 register_trace_prio_sched_wakeup_new(event_filter_pid_sched_wakeup_probe_pre,
2581 tr, INT_MAX);
2582 register_trace_prio_sched_wakeup_new(event_filter_pid_sched_wakeup_probe_post,
2583 tr, 0);
2584
2585 register_trace_prio_sched_waking(event_filter_pid_sched_wakeup_probe_pre,
2586 tr, INT_MAX);
2587 register_trace_prio_sched_waking(event_filter_pid_sched_wakeup_probe_post,
2588 tr, 0);
2589 }
2590
2591 static ssize_t
event_pid_write(struct file * filp,const char __user * ubuf,size_t cnt,loff_t * ppos,int type)2592 event_pid_write(struct file *filp, const char __user *ubuf,
2593 size_t cnt, loff_t *ppos, int type)
2594 {
2595 struct seq_file *m = filp->private_data;
2596 struct trace_array *tr = m->private;
2597 struct trace_pid_list *filtered_pids = NULL;
2598 struct trace_pid_list *other_pids = NULL;
2599 struct trace_pid_list *pid_list;
2600 struct trace_event_file *file;
2601 ssize_t ret;
2602
2603 if (!cnt)
2604 return 0;
2605
2606 ret = tracing_update_buffers(tr);
2607 if (ret < 0)
2608 return ret;
2609
2610 guard(mutex)(&event_mutex);
2611
2612 if (type == TRACE_PIDS) {
2613 filtered_pids = rcu_dereference_protected(tr->filtered_pids,
2614 lockdep_is_held(&event_mutex));
2615 other_pids = rcu_dereference_protected(tr->filtered_no_pids,
2616 lockdep_is_held(&event_mutex));
2617 } else {
2618 filtered_pids = rcu_dereference_protected(tr->filtered_no_pids,
2619 lockdep_is_held(&event_mutex));
2620 other_pids = rcu_dereference_protected(tr->filtered_pids,
2621 lockdep_is_held(&event_mutex));
2622 }
2623
2624 ret = trace_pid_write(filtered_pids, &pid_list, ubuf, cnt);
2625 if (ret < 0)
2626 return ret;
2627
2628 if (type == TRACE_PIDS)
2629 rcu_assign_pointer(tr->filtered_pids, pid_list);
2630 else
2631 rcu_assign_pointer(tr->filtered_no_pids, pid_list);
2632
2633 list_for_each_entry(file, &tr->events, list) {
2634 set_bit(EVENT_FILE_FL_PID_FILTER_BIT, &file->flags);
2635 }
2636
2637 if (filtered_pids) {
2638 tracepoint_synchronize_unregister();
2639 trace_pid_list_free(filtered_pids);
2640 } else if (pid_list && !other_pids) {
2641 register_pid_events(tr);
2642 }
2643
2644 /*
2645 * Ignoring of pids is done at task switch. But we have to
2646 * check for those tasks that are currently running.
2647 * Always do this in case a pid was appended or removed.
2648 */
2649 on_each_cpu(ignore_task_cpu, tr, 1);
2650
2651 *ppos += ret;
2652
2653 return ret;
2654 }
2655
2656 static ssize_t
ftrace_event_pid_write(struct file * filp,const char __user * ubuf,size_t cnt,loff_t * ppos)2657 ftrace_event_pid_write(struct file *filp, const char __user *ubuf,
2658 size_t cnt, loff_t *ppos)
2659 {
2660 return event_pid_write(filp, ubuf, cnt, ppos, TRACE_PIDS);
2661 }
2662
2663 static ssize_t
ftrace_event_npid_write(struct file * filp,const char __user * ubuf,size_t cnt,loff_t * ppos)2664 ftrace_event_npid_write(struct file *filp, const char __user *ubuf,
2665 size_t cnt, loff_t *ppos)
2666 {
2667 return event_pid_write(filp, ubuf, cnt, ppos, TRACE_NO_PIDS);
2668 }
2669
2670 static int ftrace_event_avail_open(struct inode *inode, struct file *file);
2671 static int ftrace_event_set_open(struct inode *inode, struct file *file);
2672 static int ftrace_event_show_filters_open(struct inode *inode, struct file *file);
2673 static int ftrace_event_show_triggers_open(struct inode *inode, struct file *file);
2674 static int ftrace_event_set_pid_open(struct inode *inode, struct file *file);
2675 static int ftrace_event_set_npid_open(struct inode *inode, struct file *file);
2676 static int ftrace_event_release(struct inode *inode, struct file *file);
2677
2678 static const struct seq_operations show_event_seq_ops = {
2679 .start = t_start,
2680 .next = t_next,
2681 .show = t_show,
2682 .stop = t_stop,
2683 };
2684
2685 static const struct seq_operations show_set_event_seq_ops = {
2686 .start = s_start,
2687 .next = s_next,
2688 .show = s_show,
2689 .stop = s_stop,
2690 };
2691
2692 static const struct seq_operations show_show_event_filters_seq_ops = {
2693 .start = t_start,
2694 .next = t_next,
2695 .show = t_show_filters,
2696 .stop = t_stop,
2697 };
2698
2699 static const struct seq_operations show_show_event_triggers_seq_ops = {
2700 .start = t_start,
2701 .next = t_next,
2702 .show = t_show_triggers,
2703 .stop = t_stop,
2704 };
2705
2706 static const struct seq_operations show_set_pid_seq_ops = {
2707 .start = p_start,
2708 .next = p_next,
2709 .show = trace_pid_show,
2710 .stop = p_stop,
2711 };
2712
2713 static const struct seq_operations show_set_no_pid_seq_ops = {
2714 .start = np_start,
2715 .next = np_next,
2716 .show = trace_pid_show,
2717 .stop = p_stop,
2718 };
2719
2720 static const struct file_operations ftrace_avail_fops = {
2721 .open = ftrace_event_avail_open,
2722 .read = seq_read,
2723 .llseek = seq_lseek,
2724 .release = seq_release,
2725 };
2726
2727 static const struct file_operations ftrace_set_event_fops = {
2728 .open = ftrace_event_set_open,
2729 .read = seq_read,
2730 .write = ftrace_event_write,
2731 .llseek = seq_lseek,
2732 .release = ftrace_event_release,
2733 };
2734
2735 static const struct file_operations ftrace_show_event_filters_fops = {
2736 .open = ftrace_event_show_filters_open,
2737 .read = seq_read,
2738 .llseek = seq_lseek,
2739 .release = seq_release,
2740 };
2741
2742 static const struct file_operations ftrace_show_event_triggers_fops = {
2743 .open = ftrace_event_show_triggers_open,
2744 .read = seq_read,
2745 .llseek = seq_lseek,
2746 .release = seq_release,
2747 };
2748
2749 static const struct file_operations ftrace_set_event_pid_fops = {
2750 .open = ftrace_event_set_pid_open,
2751 .read = seq_read,
2752 .write = ftrace_event_pid_write,
2753 .llseek = seq_lseek,
2754 .release = ftrace_event_release,
2755 };
2756
2757 static const struct file_operations ftrace_set_event_notrace_pid_fops = {
2758 .open = ftrace_event_set_npid_open,
2759 .read = seq_read,
2760 .write = ftrace_event_npid_write,
2761 .llseek = seq_lseek,
2762 .release = ftrace_event_release,
2763 };
2764
2765 static const struct file_operations ftrace_enable_fops = {
2766 .open = tracing_open_file_tr,
2767 .read = event_enable_read,
2768 .write = event_enable_write,
2769 .release = tracing_release_file_tr,
2770 .llseek = default_llseek,
2771 };
2772
2773 static const struct file_operations ftrace_event_format_fops = {
2774 .open = trace_format_open,
2775 .read = seq_read,
2776 .llseek = seq_lseek,
2777 .release = seq_release,
2778 };
2779
2780 #ifdef CONFIG_PERF_EVENTS
2781 static const struct file_operations ftrace_event_id_fops = {
2782 .read = event_id_read,
2783 .llseek = default_llseek,
2784 };
2785 #endif
2786
2787 #ifdef CONFIG_BPF_EVENTS
2788 static const struct file_operations ftrace_event_btf_ids_fops = {
2789 .read = event_btf_ids_read,
2790 .llseek = default_llseek,
2791 };
2792 #endif
2793
2794 static const struct file_operations ftrace_event_filter_fops = {
2795 .open = tracing_open_file_tr,
2796 .read = event_filter_read,
2797 .write = event_filter_write,
2798 .release = tracing_release_file_tr,
2799 .llseek = default_llseek,
2800 };
2801
2802 static const struct file_operations ftrace_subsystem_filter_fops = {
2803 .open = subsystem_open,
2804 .read = subsystem_filter_read,
2805 .write = subsystem_filter_write,
2806 .llseek = default_llseek,
2807 .release = subsystem_release,
2808 };
2809
2810 static const struct file_operations ftrace_system_enable_fops = {
2811 .open = subsystem_open,
2812 .read = system_enable_read,
2813 .write = system_enable_write,
2814 .llseek = default_llseek,
2815 .release = subsystem_release,
2816 };
2817
2818 static const struct file_operations ftrace_tr_enable_fops = {
2819 .open = system_tr_open,
2820 .read = system_enable_read,
2821 .write = system_enable_write,
2822 .llseek = default_llseek,
2823 .release = subsystem_release,
2824 };
2825
2826 static const struct file_operations ftrace_show_header_page_fops = {
2827 .open = tracing_open_generic_tr,
2828 .read = show_header_page_file,
2829 .llseek = default_llseek,
2830 .release = tracing_release_generic_tr,
2831 };
2832
2833 static const struct file_operations ftrace_show_header_event_fops = {
2834 .open = tracing_open_generic_tr,
2835 .read = show_header_event_file,
2836 .llseek = default_llseek,
2837 .release = tracing_release_generic_tr,
2838 };
2839
2840 static int
ftrace_event_open(struct inode * inode,struct file * file,const struct seq_operations * seq_ops)2841 ftrace_event_open(struct inode *inode, struct file *file,
2842 const struct seq_operations *seq_ops)
2843 {
2844 struct seq_file *m;
2845 int ret;
2846
2847 ret = security_locked_down(LOCKDOWN_TRACEFS);
2848 if (ret)
2849 return ret;
2850
2851 ret = seq_open(file, seq_ops);
2852 if (ret < 0)
2853 return ret;
2854 m = file->private_data;
2855 /* copy tr over to seq ops */
2856 m->private = inode->i_private;
2857
2858 return ret;
2859 }
2860
ftrace_event_release(struct inode * inode,struct file * file)2861 static int ftrace_event_release(struct inode *inode, struct file *file)
2862 {
2863 struct trace_array *tr = inode->i_private;
2864
2865 trace_array_put(tr);
2866
2867 return seq_release(inode, file);
2868 }
2869
2870 static int
ftrace_event_avail_open(struct inode * inode,struct file * file)2871 ftrace_event_avail_open(struct inode *inode, struct file *file)
2872 {
2873 const struct seq_operations *seq_ops = &show_event_seq_ops;
2874
2875 /* Checks for tracefs lockdown */
2876 return ftrace_event_open(inode, file, seq_ops);
2877 }
2878
2879 static int
ftrace_event_set_open(struct inode * inode,struct file * file)2880 ftrace_event_set_open(struct inode *inode, struct file *file)
2881 {
2882 const struct seq_operations *seq_ops = &show_set_event_seq_ops;
2883 struct trace_array *tr = inode->i_private;
2884 int ret;
2885
2886 ret = tracing_check_open_get_tr(tr);
2887 if (ret)
2888 return ret;
2889
2890 if ((file->f_mode & FMODE_WRITE) &&
2891 (file->f_flags & O_TRUNC))
2892 ftrace_clear_events(tr);
2893
2894 ret = ftrace_event_open(inode, file, seq_ops);
2895 if (ret < 0)
2896 trace_array_put(tr);
2897 return ret;
2898 }
2899
2900 /**
2901 * ftrace_event_show_filters_open - open interface for set_event_filters
2902 * @inode: The inode of the file
2903 * @file: The file being opened
2904 *
2905 * Connects the set_event_filters file to the sequence operations
2906 * required to iterate over and display active event filters.
2907 */
2908 static int
ftrace_event_show_filters_open(struct inode * inode,struct file * file)2909 ftrace_event_show_filters_open(struct inode *inode, struct file *file)
2910 {
2911 return ftrace_event_open(inode, file, &show_show_event_filters_seq_ops);
2912 }
2913
2914 /**
2915 * ftrace_event_show_triggers_open - open interface for show_event_triggers
2916 * @inode: The inode of the file
2917 * @file: The file being opened
2918 *
2919 * Connects the show_event_triggers file to the sequence operations
2920 * required to iterate over and display active event triggers.
2921 */
2922 static int
ftrace_event_show_triggers_open(struct inode * inode,struct file * file)2923 ftrace_event_show_triggers_open(struct inode *inode, struct file *file)
2924 {
2925 return ftrace_event_open(inode, file, &show_show_event_triggers_seq_ops);
2926 }
2927
2928 static int
ftrace_event_set_pid_open(struct inode * inode,struct file * file)2929 ftrace_event_set_pid_open(struct inode *inode, struct file *file)
2930 {
2931 const struct seq_operations *seq_ops = &show_set_pid_seq_ops;
2932 struct trace_array *tr = inode->i_private;
2933 int ret;
2934
2935 ret = tracing_check_open_get_tr(tr);
2936 if (ret)
2937 return ret;
2938
2939 if ((file->f_mode & FMODE_WRITE) &&
2940 (file->f_flags & O_TRUNC))
2941 ftrace_clear_event_pids(tr, TRACE_PIDS);
2942
2943 ret = ftrace_event_open(inode, file, seq_ops);
2944 if (ret < 0)
2945 trace_array_put(tr);
2946 return ret;
2947 }
2948
2949 static int
ftrace_event_set_npid_open(struct inode * inode,struct file * file)2950 ftrace_event_set_npid_open(struct inode *inode, struct file *file)
2951 {
2952 const struct seq_operations *seq_ops = &show_set_no_pid_seq_ops;
2953 struct trace_array *tr = inode->i_private;
2954 int ret;
2955
2956 ret = tracing_check_open_get_tr(tr);
2957 if (ret)
2958 return ret;
2959
2960 if ((file->f_mode & FMODE_WRITE) &&
2961 (file->f_flags & O_TRUNC))
2962 ftrace_clear_event_pids(tr, TRACE_NO_PIDS);
2963
2964 ret = ftrace_event_open(inode, file, seq_ops);
2965 if (ret < 0)
2966 trace_array_put(tr);
2967 return ret;
2968 }
2969
2970 static struct event_subsystem *
create_new_subsystem(const char * name)2971 create_new_subsystem(const char *name)
2972 {
2973 struct event_subsystem *system;
2974
2975 /* need to create new entry */
2976 system = kmalloc_obj(*system);
2977 if (!system)
2978 return NULL;
2979
2980 system->ref_count = 1;
2981
2982 /* Only allocate if dynamic (kprobes and modules) */
2983 system->name = kstrdup_const(name, GFP_KERNEL);
2984 if (!system->name)
2985 goto out_free;
2986
2987 system->filter = kzalloc_obj(struct event_filter);
2988 if (!system->filter)
2989 goto out_free;
2990
2991 list_add(&system->list, &event_subsystems);
2992
2993 return system;
2994
2995 out_free:
2996 kfree_const(system->name);
2997 kfree(system);
2998 return NULL;
2999 }
3000
system_callback(const char * name,umode_t * mode,void ** data,const struct file_operations ** fops)3001 static int system_callback(const char *name, umode_t *mode, void **data,
3002 const struct file_operations **fops)
3003 {
3004 if (strcmp(name, "filter") == 0)
3005 *fops = &ftrace_subsystem_filter_fops;
3006
3007 else if (strcmp(name, "enable") == 0)
3008 *fops = &ftrace_system_enable_fops;
3009
3010 else
3011 return 0;
3012
3013 *mode = TRACE_MODE_WRITE;
3014 return 1;
3015 }
3016
3017 static struct eventfs_inode *
event_subsystem_dir(struct trace_array * tr,const char * name,struct trace_event_file * file,struct eventfs_inode * parent)3018 event_subsystem_dir(struct trace_array *tr, const char *name,
3019 struct trace_event_file *file, struct eventfs_inode *parent)
3020 {
3021 struct event_subsystem *system, *iter;
3022 struct trace_subsystem_dir *dir;
3023 struct eventfs_inode *ei;
3024 int nr_entries;
3025 static struct eventfs_entry system_entries[] = {
3026 {
3027 .name = "filter",
3028 .callback = system_callback,
3029 },
3030 {
3031 .name = "enable",
3032 .callback = system_callback,
3033 }
3034 };
3035
3036 /* First see if we did not already create this dir */
3037 list_for_each_entry(dir, &tr->systems, list) {
3038 system = dir->subsystem;
3039 if (strcmp(system->name, name) == 0) {
3040 dir->nr_events++;
3041 file->system = dir;
3042 return dir->ei;
3043 }
3044 }
3045
3046 /* Now see if the system itself exists. */
3047 system = NULL;
3048 list_for_each_entry(iter, &event_subsystems, list) {
3049 if (strcmp(iter->name, name) == 0) {
3050 system = iter;
3051 break;
3052 }
3053 }
3054
3055 dir = kmalloc_obj(*dir);
3056 if (!dir)
3057 goto out_fail;
3058
3059 if (!system) {
3060 system = create_new_subsystem(name);
3061 if (!system)
3062 goto out_free;
3063 } else
3064 __get_system(system);
3065
3066 /* ftrace only has directories no files, readonly instance too. */
3067 if (strcmp(name, "ftrace") == 0 || trace_array_is_readonly(tr))
3068 nr_entries = 0;
3069 else
3070 nr_entries = ARRAY_SIZE(system_entries);
3071
3072 ei = eventfs_create_dir(name, parent, system_entries, nr_entries, dir);
3073 if (IS_ERR(ei)) {
3074 pr_warn("Failed to create system directory %s\n", name);
3075 __put_system(system);
3076 goto out_free;
3077 }
3078
3079 dir->ei = ei;
3080 dir->tr = tr;
3081 dir->ref_count = 1;
3082 dir->nr_events = 1;
3083 dir->subsystem = system;
3084 file->system = dir;
3085
3086 list_add(&dir->list, &tr->systems);
3087
3088 return dir->ei;
3089
3090 out_free:
3091 kfree(dir);
3092 out_fail:
3093 /* Only print this message if failed on memory allocation */
3094 if (!dir || !system)
3095 pr_warn("No memory to create event subsystem %s\n", name);
3096 return NULL;
3097 }
3098
3099 static int
event_define_fields(struct trace_event_call * call)3100 event_define_fields(struct trace_event_call *call)
3101 {
3102 struct list_head *head;
3103 int ret = 0;
3104
3105 /*
3106 * Other events may have the same class. Only update
3107 * the fields if they are not already defined.
3108 */
3109 head = trace_get_fields(call);
3110 if (list_empty(head)) {
3111 struct trace_event_fields *field = call->class->fields_array;
3112 unsigned int offset = sizeof(struct trace_entry);
3113
3114 for (; field->type; field++) {
3115 if (field->type == TRACE_FUNCTION_TYPE) {
3116 field->define_fields(call);
3117 break;
3118 }
3119
3120 offset = ALIGN(offset, field->align);
3121 ret = trace_define_field_ext(call, field->type, field->name,
3122 offset, field->size,
3123 field->is_signed, field->filter_type,
3124 field->len, field->needs_test);
3125 if (WARN_ON_ONCE(ret)) {
3126 pr_err("error code is %d\n", ret);
3127 break;
3128 }
3129
3130 offset += field->size;
3131 }
3132 }
3133
3134 return ret;
3135 }
3136
event_callback(const char * name,umode_t * mode,void ** data,const struct file_operations ** fops)3137 static int event_callback(const char *name, umode_t *mode, void **data,
3138 const struct file_operations **fops)
3139 {
3140 struct trace_event_file *file = *data;
3141 struct trace_event_call *call = file->event_call;
3142
3143 if (strcmp(name, "format") == 0) {
3144 *mode = TRACE_MODE_READ;
3145 *fops = &ftrace_event_format_fops;
3146 return 1;
3147 }
3148
3149 /*
3150 * Only event directories that can be enabled should have
3151 * triggers or filters, with the exception of the "print"
3152 * event that can have a "trigger" file.
3153 */
3154 if (!(call->flags & TRACE_EVENT_FL_IGNORE_ENABLE)) {
3155 if (call->class->reg && strcmp(name, "enable") == 0) {
3156 *mode = TRACE_MODE_WRITE;
3157 *fops = &ftrace_enable_fops;
3158 return 1;
3159 }
3160
3161 if (strcmp(name, "filter") == 0) {
3162 *mode = TRACE_MODE_WRITE;
3163 *fops = &ftrace_event_filter_fops;
3164 return 1;
3165 }
3166 }
3167
3168 if (!(call->flags & TRACE_EVENT_FL_IGNORE_ENABLE) ||
3169 strcmp(trace_event_name(call), "print") == 0) {
3170 if (strcmp(name, "trigger") == 0) {
3171 *mode = TRACE_MODE_WRITE;
3172 *fops = &event_trigger_fops;
3173 return 1;
3174 }
3175 }
3176
3177 #ifdef CONFIG_PERF_EVENTS
3178 if (call->event.type && call->class->reg &&
3179 strcmp(name, "id") == 0) {
3180 *mode = TRACE_MODE_READ;
3181 *data = (void *)(long)call->event.type;
3182 *fops = &ftrace_event_id_fops;
3183 return 1;
3184 }
3185 #endif
3186
3187 #ifdef CONFIG_BPF_EVENTS
3188 if (call->class->btf_ids && strcmp(name, "btf_ids") == 0) {
3189 *mode = TRACE_MODE_READ;
3190 *fops = &ftrace_event_btf_ids_fops;
3191 return 1;
3192 }
3193 #endif
3194
3195 #ifdef CONFIG_HIST_TRIGGERS
3196 if (strcmp(name, "hist") == 0) {
3197 *mode = TRACE_MODE_READ;
3198 *fops = &event_hist_fops;
3199 return 1;
3200 }
3201 #endif
3202 #ifdef CONFIG_HIST_TRIGGERS_DEBUG
3203 if (strcmp(name, "hist_debug") == 0) {
3204 *mode = TRACE_MODE_READ;
3205 *fops = &event_hist_debug_fops;
3206 return 1;
3207 }
3208 #endif
3209 #ifdef CONFIG_TRACE_EVENT_INJECT
3210 if (call->event.type && call->class->reg &&
3211 strcmp(name, "inject") == 0) {
3212 *mode = 0200;
3213 *fops = &event_inject_fops;
3214 return 1;
3215 }
3216 #endif
3217 return 0;
3218 }
3219
3220 /* The file is incremented on creation and freeing the enable file decrements it */
event_release(const char * name,void * data)3221 static void event_release(const char *name, void *data)
3222 {
3223 struct trace_event_file *file = data;
3224
3225 event_file_put(file);
3226 }
3227
3228 static int
event_create_dir(struct eventfs_inode * parent,struct trace_event_file * file)3229 event_create_dir(struct eventfs_inode *parent, struct trace_event_file *file)
3230 {
3231 struct trace_event_call *call = file->event_call;
3232 struct trace_array *tr = file->tr;
3233 struct eventfs_inode *e_events;
3234 struct eventfs_inode *ei;
3235 const char *name;
3236 int nr_entries;
3237 int ret;
3238 static struct eventfs_entry event_entries[] = {
3239 {
3240 .name = "format",
3241 .callback = event_callback,
3242 },
3243 #ifdef CONFIG_PERF_EVENTS
3244 {
3245 .name = "id",
3246 .callback = event_callback,
3247 },
3248 #endif
3249 #ifdef CONFIG_BPF_EVENTS
3250 {
3251 .name = "btf_ids",
3252 .callback = event_callback,
3253 },
3254 #endif
3255 #define NR_RO_EVENT_ENTRIES (1 + IS_ENABLED(CONFIG_PERF_EVENTS) + \
3256 IS_ENABLED(CONFIG_BPF_EVENTS))
3257 /* Readonly files must be above this line and counted by NR_RO_EVENT_ENTRIES. */
3258 {
3259 .name = "enable",
3260 .callback = event_callback,
3261 .release = event_release,
3262 },
3263 {
3264 .name = "filter",
3265 .callback = event_callback,
3266 },
3267 {
3268 .name = "trigger",
3269 .callback = event_callback,
3270 },
3271 #ifdef CONFIG_HIST_TRIGGERS
3272 {
3273 .name = "hist",
3274 .callback = event_callback,
3275 },
3276 #endif
3277 #ifdef CONFIG_HIST_TRIGGERS_DEBUG
3278 {
3279 .name = "hist_debug",
3280 .callback = event_callback,
3281 },
3282 #endif
3283 #ifdef CONFIG_TRACE_EVENT_INJECT
3284 {
3285 .name = "inject",
3286 .callback = event_callback,
3287 },
3288 #endif
3289 };
3290
3291 /*
3292 * If the trace point header did not define TRACE_SYSTEM
3293 * then the system would be called "TRACE_SYSTEM". This should
3294 * never happen.
3295 */
3296 if (WARN_ON_ONCE(strcmp(call->class->system, TRACE_SYSTEM) == 0))
3297 return -ENODEV;
3298
3299 ret = event_define_fields(call);
3300 if (ret < 0) {
3301 pr_warn("Could not initialize trace point events/%s\n",
3302 trace_event_name(call));
3303 return ret;
3304 }
3305
3306 e_events = event_subsystem_dir(tr, call->class->system, file, parent);
3307 if (!e_events)
3308 return -ENOMEM;
3309
3310 if (trace_array_is_readonly(tr))
3311 nr_entries = NR_RO_EVENT_ENTRIES;
3312 else
3313 nr_entries = ARRAY_SIZE(event_entries);
3314
3315 name = trace_event_name(call);
3316 ei = eventfs_create_dir(name, e_events, event_entries, nr_entries, file);
3317 if (IS_ERR(ei)) {
3318 pr_warn("Could not create tracefs '%s' directory\n", name);
3319 return -1;
3320 }
3321
3322 file->ei = ei;
3323
3324 /* Gets decremented on freeing of the "enable" file */
3325 event_file_get(file);
3326
3327 return 0;
3328 }
3329
remove_event_from_tracers(struct trace_event_call * call)3330 static void remove_event_from_tracers(struct trace_event_call *call)
3331 {
3332 struct trace_event_file *file;
3333 struct trace_array *tr;
3334
3335 do_for_each_event_file_safe(tr, file) {
3336 if (file->event_call != call)
3337 continue;
3338
3339 remove_event_file_dir(file);
3340 /*
3341 * The do_for_each_event_file_safe() is
3342 * a double loop. After finding the call for this
3343 * trace_array, we use break to jump to the next
3344 * trace_array.
3345 */
3346 break;
3347 } while_for_each_event_file();
3348 }
3349
event_remove(struct trace_event_call * call)3350 static void event_remove(struct trace_event_call *call)
3351 {
3352 struct trace_array *tr;
3353 struct trace_event_file *file;
3354
3355 do_for_each_event_file(tr, file) {
3356 if (file->event_call != call)
3357 continue;
3358
3359 if (file->flags & EVENT_FILE_FL_WAS_ENABLED)
3360 tr->clear_trace = true;
3361
3362 ftrace_event_enable_disable(file, 0);
3363 /*
3364 * The do_for_each_event_file() is
3365 * a double loop. After finding the call for this
3366 * trace_array, we use break to jump to the next
3367 * trace_array.
3368 */
3369 break;
3370 } while_for_each_event_file();
3371
3372 if (call->event.funcs)
3373 __unregister_trace_event(&call->event);
3374 remove_event_from_tracers(call);
3375 list_del(&call->list);
3376 }
3377
event_init(struct trace_event_call * call)3378 static int event_init(struct trace_event_call *call)
3379 {
3380 int ret = 0;
3381 const char *name;
3382
3383 name = trace_event_name(call);
3384 if (WARN_ON(!name))
3385 return -EINVAL;
3386
3387 if (call->class->raw_init) {
3388 ret = call->class->raw_init(call);
3389 if (ret < 0 && ret != -ENOSYS)
3390 pr_warn("Could not initialize trace events/%s\n", name);
3391 }
3392
3393 return ret;
3394 }
3395
3396 static int
__register_event(struct trace_event_call * call,struct module * mod)3397 __register_event(struct trace_event_call *call, struct module *mod)
3398 {
3399 int ret;
3400
3401 ret = event_init(call);
3402 if (ret < 0)
3403 return ret;
3404
3405 down_write(&trace_event_sem);
3406 list_add(&call->list, &ftrace_events);
3407 up_write(&trace_event_sem);
3408
3409 if (call->flags & TRACE_EVENT_FL_DYNAMIC)
3410 atomic_set(&call->refcnt, 0);
3411 else
3412 call->module = mod;
3413
3414 return 0;
3415 }
3416
eval_replace(char * ptr,struct trace_eval_map * map,int len)3417 static char *eval_replace(char *ptr, struct trace_eval_map *map, int len)
3418 {
3419 int rlen;
3420 int elen;
3421
3422 /* Find the length of the eval value as a string */
3423 elen = snprintf(ptr, 0, "%ld", map->eval_value);
3424 /* Make sure there's enough room to replace the string with the value */
3425 if (len < elen)
3426 return NULL;
3427
3428 snprintf(ptr, elen + 1, "%ld", map->eval_value);
3429
3430 /* Get the rest of the string of ptr */
3431 rlen = strlen(ptr + len);
3432 memmove(ptr + elen, ptr + len, rlen);
3433 /* Make sure we end the new string */
3434 ptr[elen + rlen] = 0;
3435
3436 return ptr + elen;
3437 }
3438
update_event_printk(struct trace_event_call * call,struct trace_eval_map * map)3439 static void update_event_printk(struct trace_event_call *call,
3440 struct trace_eval_map *map)
3441 {
3442 char *ptr;
3443 int quote = 0;
3444 int len = strlen(map->eval_string);
3445
3446 for (ptr = call->print_fmt; *ptr; ptr++) {
3447 if (*ptr == '\\') {
3448 ptr++;
3449 /* paranoid */
3450 if (!*ptr)
3451 break;
3452 continue;
3453 }
3454 if (*ptr == '"') {
3455 quote ^= 1;
3456 continue;
3457 }
3458 if (quote)
3459 continue;
3460 if (isdigit(*ptr)) {
3461 /* skip numbers */
3462 do {
3463 ptr++;
3464 /* Check for alpha chars like ULL */
3465 } while (isalnum(*ptr));
3466 if (!*ptr)
3467 break;
3468 /*
3469 * A number must have some kind of delimiter after
3470 * it, and we can ignore that too.
3471 */
3472 continue;
3473 }
3474 if (isalpha(*ptr) || *ptr == '_') {
3475 if (strncmp(map->eval_string, ptr, len) == 0 &&
3476 !isalnum(ptr[len]) && ptr[len] != '_') {
3477 ptr = eval_replace(ptr, map, len);
3478 /* enum/sizeof string smaller than value */
3479 if (WARN_ON_ONCE(!ptr))
3480 return;
3481 /*
3482 * No need to decrement here, as eval_replace()
3483 * returns the pointer to the character passed
3484 * the eval, and two evals can not be placed
3485 * back to back without something in between.
3486 * We can skip that something in between.
3487 */
3488 continue;
3489 }
3490 skip_more:
3491 do {
3492 ptr++;
3493 } while (isalnum(*ptr) || *ptr == '_');
3494 if (!*ptr)
3495 break;
3496 /*
3497 * If what comes after this variable is a '.' or
3498 * '->' then we can continue to ignore that string.
3499 */
3500 if (*ptr == '.' || (ptr[0] == '-' && ptr[1] == '>')) {
3501 ptr += *ptr == '.' ? 1 : 2;
3502 if (!*ptr)
3503 break;
3504 goto skip_more;
3505 }
3506 /*
3507 * Once again, we can skip the delimiter that came
3508 * after the string.
3509 */
3510 continue;
3511 }
3512 }
3513 }
3514
add_str_to_module(struct module * module,char * str)3515 static void add_str_to_module(struct module *module, char *str)
3516 {
3517 struct module_string *modstr;
3518
3519 modstr = kmalloc_obj(*modstr);
3520
3521 /*
3522 * If we failed to allocate memory here, then we'll just
3523 * let the str memory leak when the module is removed.
3524 * If this fails to allocate, there's worse problems than
3525 * a leaked string on module removal.
3526 */
3527 if (WARN_ON_ONCE(!modstr))
3528 return;
3529
3530 modstr->module = module;
3531 modstr->str = str;
3532
3533 list_add(&modstr->next, &module_strings);
3534 }
3535
3536 #define ATTRIBUTE_STR "__attribute__("
3537 #define ATTRIBUTE_STR_LEN (sizeof(ATTRIBUTE_STR) - 1)
3538
3539 /* Remove all __attribute__() from @type. Return allocated string or @type. */
sanitize_field_type(const char * type)3540 static char *sanitize_field_type(const char *type)
3541 {
3542 char *attr, *tmp, *next, *ret = (char *)type;
3543 int depth;
3544
3545 next = (char *)type;
3546 while ((attr = strstr(next, ATTRIBUTE_STR))) {
3547 /* Retry if "__attribute__(" is a part of another word. */
3548 if (attr != next && !isspace(attr[-1])) {
3549 next = attr + ATTRIBUTE_STR_LEN;
3550 continue;
3551 }
3552
3553 if (ret == type) {
3554 ret = kstrdup(type, GFP_KERNEL);
3555 if (WARN_ON_ONCE(!ret))
3556 return NULL;
3557 attr = ret + (attr - type);
3558 }
3559
3560 /* the ATTRIBUTE_STR already has the first '(' */
3561 depth = 1;
3562 next = attr + ATTRIBUTE_STR_LEN;
3563 do {
3564 tmp = strpbrk(next, "()");
3565 /* There is unbalanced parentheses */
3566 if (WARN_ON_ONCE(!tmp)) {
3567 kfree(ret);
3568 return (char *)type;
3569 }
3570
3571 if (*tmp == '(')
3572 depth++;
3573 else
3574 depth--;
3575 next = tmp + 1;
3576 } while (depth > 0);
3577 next = skip_spaces(next);
3578 strcpy(attr, next);
3579 next = attr;
3580 }
3581 return ret;
3582 }
3583
find_replacable_eval(const char * type,const char * eval_string,int len)3584 static char *find_replacable_eval(const char *type, const char *eval_string,
3585 int len)
3586 {
3587 char *ptr;
3588
3589 if (!eval_string)
3590 return NULL;
3591
3592 ptr = strchr(type, '[');
3593 if (!ptr)
3594 return NULL;
3595 ptr++;
3596
3597 if (!isalpha(*ptr) && *ptr != '_')
3598 return NULL;
3599
3600 if (strncmp(eval_string, ptr, len) != 0)
3601 return NULL;
3602
3603 return ptr;
3604 }
3605
update_event_fields(struct trace_event_call * call,struct trace_eval_map * map)3606 static void update_event_fields(struct trace_event_call *call,
3607 struct trace_eval_map *map)
3608 {
3609 struct ftrace_event_field *field;
3610 const char *eval_string = NULL;
3611 struct list_head *head;
3612 int len = 0;
3613 char *ptr;
3614 char *str;
3615
3616 /* Dynamic events should never have field maps */
3617 if (call->flags & TRACE_EVENT_FL_DYNAMIC)
3618 return;
3619
3620 if (map) {
3621 eval_string = map->eval_string;
3622 len = strlen(map->eval_string);
3623 }
3624
3625 head = trace_get_fields(call);
3626 list_for_each_entry(field, head, link) {
3627 str = sanitize_field_type(field->type);
3628 if (!str)
3629 return;
3630
3631 ptr = find_replacable_eval(str, eval_string, len);
3632 if (ptr) {
3633 if (str == field->type) {
3634 str = kstrdup(field->type, GFP_KERNEL);
3635 if (WARN_ON_ONCE(!str))
3636 return;
3637 ptr = str + (ptr - field->type);
3638 }
3639
3640 ptr = eval_replace(ptr, map, len);
3641 /* enum/sizeof string smaller than value */
3642 if (WARN_ON_ONCE(!ptr)) {
3643 kfree(str);
3644 continue;
3645 }
3646 }
3647
3648 if (str == field->type)
3649 continue;
3650 /*
3651 * If the event is part of a module, then we need to free the string
3652 * when the module is removed. Otherwise, it will stay allocated
3653 * until a reboot.
3654 */
3655 if (call->module)
3656 add_str_to_module(call->module, str);
3657
3658 field->type = str;
3659 if (field->filter_type == FILTER_OTHER)
3660 field->filter_type = filter_assign_type(field->type);
3661 }
3662 }
3663
3664 /* Update all events for replacing eval and sanitizing */
trace_event_update_all(struct trace_eval_map ** map,int len,struct module * mod)3665 void trace_event_update_all(struct trace_eval_map **map, int len, struct module *mod)
3666 {
3667 struct trace_event_call *call, *p;
3668 const char *last_system = NULL;
3669 bool first = false;
3670 bool updated;
3671 int last_i;
3672 int i;
3673
3674 mutex_lock(&event_mutex);
3675 down_write(&trace_event_sem);
3676 list_for_each_entry_safe(call, p, &ftrace_events, list) {
3677
3678 if (mod && call->module != mod)
3679 continue;
3680
3681 /* events are usually grouped together with systems */
3682 if (!last_system || call->class->system != last_system) {
3683 first = true;
3684 last_i = 0;
3685 last_system = call->class->system;
3686 }
3687
3688 updated = false;
3689 /*
3690 * Since calls are grouped by systems, the likelihood that the
3691 * next call in the iteration belongs to the same system as the
3692 * previous call is high. As an optimization, we skip searching
3693 * for a map[] that matches the call's system if the last call
3694 * was from the same system. That's what last_i is for. If the
3695 * call has the same system as the previous call, then last_i
3696 * will be the index of the first map[] that has a matching
3697 * system.
3698 */
3699 for (i = last_i; i < len; i++) {
3700 if (call->class->system == map[i]->system) {
3701 /* Save the first system if need be */
3702 if (first) {
3703 last_i = i;
3704 first = false;
3705 }
3706 update_event_printk(call, map[i]);
3707 update_event_fields(call, map[i]);
3708 updated = true;
3709 }
3710 }
3711 /* If not updated yet, update field for sanitizing. */
3712 if (!updated)
3713 update_event_fields(call, NULL);
3714 cond_resched();
3715 }
3716 up_write(&trace_event_sem);
3717 mutex_unlock(&event_mutex);
3718 }
3719
event_in_systems(struct trace_event_call * call,const char * systems)3720 static bool event_in_systems(struct trace_event_call *call,
3721 const char *systems)
3722 {
3723 const char *system;
3724 const char *p;
3725
3726 if (!systems)
3727 return true;
3728
3729 system = call->class->system;
3730 p = strstr(systems, system);
3731 if (!p)
3732 return false;
3733
3734 if (p != systems && !isspace(*(p - 1)) && *(p - 1) != ',')
3735 return false;
3736
3737 p += strlen(system);
3738 return !*p || isspace(*p) || *p == ',';
3739 }
3740
3741 #ifdef CONFIG_HIST_TRIGGERS
3742 /*
3743 * Wake up waiter on the hist_poll_wq from irq_work because the hist trigger
3744 * may happen in any context.
3745 */
hist_poll_event_irq_work(struct irq_work * work)3746 static void hist_poll_event_irq_work(struct irq_work *work)
3747 {
3748 wake_up_all(&hist_poll_wq);
3749 }
3750
3751 DEFINE_IRQ_WORK(hist_poll_work, hist_poll_event_irq_work);
3752 DECLARE_WAIT_QUEUE_HEAD(hist_poll_wq);
3753 #endif
3754
3755 static struct trace_event_file *
trace_create_new_event(struct trace_event_call * call,struct trace_array * tr)3756 trace_create_new_event(struct trace_event_call *call,
3757 struct trace_array *tr)
3758 {
3759 struct trace_pid_list *no_pid_list;
3760 struct trace_pid_list *pid_list;
3761 struct trace_event_file *file;
3762 unsigned int first;
3763
3764 if (!event_in_systems(call, tr->system_names))
3765 return NULL;
3766
3767 file = kmem_cache_alloc(file_cachep, GFP_TRACE);
3768 if (!file)
3769 return ERR_PTR(-ENOMEM);
3770
3771 pid_list = rcu_dereference_protected(tr->filtered_pids,
3772 lockdep_is_held(&event_mutex));
3773 no_pid_list = rcu_dereference_protected(tr->filtered_no_pids,
3774 lockdep_is_held(&event_mutex));
3775
3776 if (!trace_pid_list_first(pid_list, &first) ||
3777 !trace_pid_list_first(no_pid_list, &first))
3778 file->flags |= EVENT_FILE_FL_PID_FILTER;
3779
3780 file->event_call = call;
3781 file->tr = tr;
3782 atomic_set(&file->sm_ref, 0);
3783 atomic_set(&file->tm_ref, 0);
3784 INIT_LIST_HEAD(&file->triggers);
3785 list_add(&file->list, &tr->events);
3786 refcount_set(&file->ref, 1);
3787
3788 return file;
3789 }
3790
3791 #define MAX_BOOT_TRIGGERS 32
3792
3793 static struct boot_triggers {
3794 const char *event;
3795 char *trigger;
3796 } bootup_triggers[MAX_BOOT_TRIGGERS];
3797
3798 static char bootup_trigger_buf[COMMAND_LINE_SIZE];
3799 static int boot_trigger_buf_len;
3800 static int nr_boot_triggers;
3801
setup_trace_triggers(char * str)3802 static __init int setup_trace_triggers(char *str)
3803 {
3804 char *trigger;
3805 char *buf;
3806 int len = boot_trigger_buf_len;
3807 int i;
3808
3809 if (len >= COMMAND_LINE_SIZE)
3810 return 1;
3811
3812 strscpy(bootup_trigger_buf + len, str, COMMAND_LINE_SIZE - len);
3813 trace_set_ring_buffer_expanded(NULL);
3814 disable_tracing_selftest("running event triggers");
3815
3816 buf = bootup_trigger_buf + len;
3817 boot_trigger_buf_len += strlen(buf) + 1;
3818
3819 for (i = nr_boot_triggers; i < MAX_BOOT_TRIGGERS; i++) {
3820 trigger = strsep(&buf, ",");
3821 if (!trigger)
3822 break;
3823 bootup_triggers[i].event = strsep(&trigger, ".");
3824 bootup_triggers[i].trigger = trigger;
3825 if (!bootup_triggers[i].trigger)
3826 break;
3827 }
3828
3829 nr_boot_triggers = i;
3830 return 1;
3831 }
3832 __setup("trace_trigger=", setup_trace_triggers);
3833
3834 /* Add an event to a trace directory */
3835 static int
__trace_add_new_event(struct trace_event_call * call,struct trace_array * tr)3836 __trace_add_new_event(struct trace_event_call *call, struct trace_array *tr)
3837 {
3838 struct trace_event_file *file;
3839
3840 file = trace_create_new_event(call, tr);
3841 /*
3842 * trace_create_new_event() returns ERR_PTR(-ENOMEM) if failed
3843 * allocation, or NULL if the event is not part of the tr->system_names.
3844 * When the event is not part of the tr->system_names, return zero, not
3845 * an error.
3846 */
3847 if (!file)
3848 return 0;
3849
3850 if (IS_ERR(file))
3851 return PTR_ERR(file);
3852
3853 if (eventdir_initialized)
3854 return event_create_dir(tr->event_dir, file);
3855 else
3856 return event_define_fields(call);
3857 }
3858
trace_early_triggers(struct trace_event_file * file,const char * name)3859 static void trace_early_triggers(struct trace_event_file *file, const char *name)
3860 {
3861 int ret;
3862 int i;
3863
3864 for (i = 0; i < nr_boot_triggers; i++) {
3865 if (strcmp(name, bootup_triggers[i].event))
3866 continue;
3867 mutex_lock(&event_mutex);
3868 ret = trigger_process_regex(file, bootup_triggers[i].trigger);
3869 mutex_unlock(&event_mutex);
3870 if (ret)
3871 pr_err("Failed to register trigger '%s' on event %s\n",
3872 bootup_triggers[i].trigger,
3873 bootup_triggers[i].event);
3874 }
3875 }
3876
3877 /*
3878 * Just create a descriptor for early init. A descriptor is required
3879 * for enabling events at boot. We want to enable events before
3880 * the filesystem is initialized.
3881 */
3882 static int
__trace_early_add_new_event(struct trace_event_call * call,struct trace_array * tr)3883 __trace_early_add_new_event(struct trace_event_call *call,
3884 struct trace_array *tr)
3885 {
3886 struct trace_event_file *file;
3887 int ret;
3888
3889 file = trace_create_new_event(call, tr);
3890 /*
3891 * trace_create_new_event() returns ERR_PTR(-ENOMEM) if failed
3892 * allocation, or NULL if the event is not part of the tr->system_names.
3893 * When the event is not part of the tr->system_names, return zero, not
3894 * an error.
3895 */
3896 if (!file)
3897 return 0;
3898
3899 if (IS_ERR(file))
3900 return PTR_ERR(file);
3901
3902 ret = event_define_fields(call);
3903 if (ret)
3904 return ret;
3905
3906 trace_early_triggers(file, trace_event_name(call));
3907
3908 return 0;
3909 }
3910
3911 struct ftrace_module_file_ops;
3912 static void __add_event_to_tracers(struct trace_event_call *call);
3913
3914 /* Add an additional event_call dynamically */
trace_add_event_call(struct trace_event_call * call)3915 int trace_add_event_call(struct trace_event_call *call)
3916 {
3917 int ret;
3918 lockdep_assert_held(&event_mutex);
3919
3920 guard(mutex)(&trace_types_lock);
3921
3922 ret = __register_event(call, NULL);
3923 if (ret < 0)
3924 return ret;
3925
3926 __add_event_to_tracers(call);
3927 return ret;
3928 }
3929 EXPORT_SYMBOL_GPL(trace_add_event_call);
3930
3931 /*
3932 * Must be called under locking of trace_types_lock, event_mutex and
3933 * trace_event_sem.
3934 */
__trace_remove_event_call(struct trace_event_call * call)3935 static void __trace_remove_event_call(struct trace_event_call *call)
3936 {
3937 event_remove(call);
3938 trace_destroy_fields(call);
3939 }
3940
probe_remove_event_call(struct trace_event_call * call)3941 static int probe_remove_event_call(struct trace_event_call *call)
3942 {
3943 struct trace_array *tr;
3944 struct trace_event_file *file;
3945
3946 #ifdef CONFIG_PERF_EVENTS
3947 if (call->perf_refcount)
3948 return -EBUSY;
3949 #endif
3950 do_for_each_event_file(tr, file) {
3951 if (file->event_call != call)
3952 continue;
3953 /*
3954 * We can't rely on ftrace_event_enable_disable(enable => 0)
3955 * we are going to do, soft mode can suppress
3956 * TRACE_REG_UNREGISTER.
3957 */
3958 if (file->flags & EVENT_FILE_FL_ENABLED)
3959 goto busy;
3960
3961 if (file->flags & EVENT_FILE_FL_WAS_ENABLED)
3962 tr->clear_trace = true;
3963 /*
3964 * The do_for_each_event_file_safe() is
3965 * a double loop. After finding the call for this
3966 * trace_array, we use break to jump to the next
3967 * trace_array.
3968 */
3969 break;
3970 } while_for_each_event_file();
3971
3972 __trace_remove_event_call(call);
3973
3974 return 0;
3975 busy:
3976 /* No need to clear the trace now */
3977 list_for_each_entry(tr, &ftrace_trace_arrays, list) {
3978 tr->clear_trace = false;
3979 }
3980 return -EBUSY;
3981 }
3982
3983 /* Remove an event_call */
trace_remove_event_call(struct trace_event_call * call)3984 int trace_remove_event_call(struct trace_event_call *call)
3985 {
3986 int ret;
3987
3988 lockdep_assert_held(&event_mutex);
3989
3990 mutex_lock(&trace_types_lock);
3991 down_write(&trace_event_sem);
3992 ret = probe_remove_event_call(call);
3993 up_write(&trace_event_sem);
3994 mutex_unlock(&trace_types_lock);
3995
3996 return ret;
3997 }
3998 EXPORT_SYMBOL_GPL(trace_remove_event_call);
3999
4000 #define for_each_event(event, start, end) \
4001 for (event = start; \
4002 (unsigned long)event < (unsigned long)end; \
4003 event++)
4004
4005 #ifdef CONFIG_MODULES
update_mod_cache(struct trace_array * tr,struct module * mod)4006 static void update_mod_cache(struct trace_array *tr, struct module *mod)
4007 {
4008 struct event_mod_load *event_mod, *n;
4009
4010 list_for_each_entry_safe(event_mod, n, &tr->mod_events, list) {
4011 if (strcmp(event_mod->module, mod->name) != 0)
4012 continue;
4013
4014 __ftrace_set_clr_event_nolock(tr, event_mod->match,
4015 event_mod->system,
4016 event_mod->event, 1, mod->name);
4017 free_event_mod(event_mod);
4018 }
4019 }
4020
update_cache_events(struct module * mod)4021 static void update_cache_events(struct module *mod)
4022 {
4023 struct trace_array *tr;
4024
4025 list_for_each_entry(tr, &ftrace_trace_arrays, list)
4026 update_mod_cache(tr, mod);
4027 }
4028
trace_module_add_events(struct module * mod)4029 static void trace_module_add_events(struct module *mod)
4030 {
4031 struct trace_event_call **call, **start, **end;
4032
4033 if (!mod->num_trace_events)
4034 return;
4035
4036 /* Don't add infrastructure for mods without tracepoints */
4037 if (trace_module_has_bad_taint(mod)) {
4038 pr_err("%s: module has bad taint, not creating trace events\n",
4039 mod->name);
4040 return;
4041 }
4042
4043 start = mod->trace_events;
4044 end = mod->trace_events + mod->num_trace_events;
4045
4046 for_each_event(call, start, end) {
4047 if (!__register_event(*call, mod))
4048 __add_event_to_tracers(*call);
4049 }
4050
4051 update_cache_events(mod);
4052 }
4053
trace_module_remove_events(struct module * mod)4054 static void trace_module_remove_events(struct module *mod)
4055 {
4056 struct trace_event_call *call, *p;
4057 struct module_string *modstr, *m;
4058
4059 down_write(&trace_event_sem);
4060 list_for_each_entry_safe(call, p, &ftrace_events, list) {
4061 if ((call->flags & TRACE_EVENT_FL_DYNAMIC) || !call->module)
4062 continue;
4063 if (call->module == mod)
4064 __trace_remove_event_call(call);
4065 }
4066 /* Check for any strings allocated for this module */
4067 list_for_each_entry_safe(modstr, m, &module_strings, next) {
4068 if (modstr->module != mod)
4069 continue;
4070 list_del(&modstr->next);
4071 kfree(modstr->str);
4072 kfree(modstr);
4073 }
4074 up_write(&trace_event_sem);
4075
4076 /*
4077 * It is safest to reset the ring buffer if the module being unloaded
4078 * registered any events that were used. The only worry is if
4079 * a new module gets loaded, and takes on the same id as the events
4080 * of this module. When printing out the buffer, traced events left
4081 * over from this module may be passed to the new module events and
4082 * unexpected results may occur.
4083 */
4084 tracing_reset_all_online_cpus_unlocked();
4085 }
4086
trace_module_notify(struct notifier_block * self,unsigned long val,void * data)4087 static int trace_module_notify(struct notifier_block *self,
4088 unsigned long val, void *data)
4089 {
4090 struct module *mod = data;
4091
4092 mutex_lock(&event_mutex);
4093 mutex_lock(&trace_types_lock);
4094 switch (val) {
4095 case MODULE_STATE_COMING:
4096 trace_module_add_events(mod);
4097 break;
4098 case MODULE_STATE_GOING:
4099 trace_module_remove_events(mod);
4100 break;
4101 }
4102 mutex_unlock(&trace_types_lock);
4103 mutex_unlock(&event_mutex);
4104
4105 return NOTIFY_OK;
4106 }
4107
4108 static struct notifier_block trace_module_nb = {
4109 .notifier_call = trace_module_notify,
4110 .priority = 1, /* higher than trace.c module notify */
4111 };
4112 #endif /* CONFIG_MODULES */
4113
4114 /* Create a new event directory structure for a trace directory. */
4115 static void
__trace_add_event_dirs(struct trace_array * tr)4116 __trace_add_event_dirs(struct trace_array *tr)
4117 {
4118 struct trace_event_call *call;
4119 int ret;
4120
4121 lockdep_assert_held(&trace_event_sem);
4122
4123 list_for_each_entry(call, &ftrace_events, list) {
4124 ret = __trace_add_new_event(call, tr);
4125 if (ret < 0)
4126 pr_warn("Could not create directory for event %s\n",
4127 trace_event_name(call));
4128 }
4129 }
4130
4131 /* Returns any file that matches the system and event */
4132 struct trace_event_file *
__find_event_file(struct trace_array * tr,const char * system,const char * event)4133 __find_event_file(struct trace_array *tr, const char *system, const char *event)
4134 {
4135 struct trace_event_file *file;
4136 struct trace_event_call *call;
4137 const char *name;
4138
4139 list_for_each_entry(file, &tr->events, list) {
4140
4141 call = file->event_call;
4142 name = trace_event_name(call);
4143
4144 if (!name || !call->class)
4145 continue;
4146
4147 if (strcmp(event, name) == 0 &&
4148 strcmp(system, call->class->system) == 0)
4149 return file;
4150 }
4151 return NULL;
4152 }
4153
4154 /* Returns valid trace event files that match system and event */
4155 struct trace_event_file *
find_event_file(struct trace_array * tr,const char * system,const char * event)4156 find_event_file(struct trace_array *tr, const char *system, const char *event)
4157 {
4158 struct trace_event_file *file;
4159
4160 file = __find_event_file(tr, system, event);
4161 if (!file || !file->event_call->class->reg ||
4162 file->event_call->flags & TRACE_EVENT_FL_IGNORE_ENABLE)
4163 return NULL;
4164
4165 return file;
4166 }
4167
4168 /**
4169 * trace_get_event_file - Find and return a trace event file
4170 * @instance: The name of the trace instance containing the event
4171 * @system: The name of the system containing the event
4172 * @event: The name of the event
4173 *
4174 * Return a trace event file given the trace instance name, trace
4175 * system, and trace event name. If the instance name is NULL, it
4176 * refers to the top-level trace array.
4177 *
4178 * This function will look it up and return it if found, after calling
4179 * trace_array_get() to prevent the instance from going away, and
4180 * increment the event's module refcount to prevent it from being
4181 * removed.
4182 *
4183 * To release the file, call trace_put_event_file(), which will call
4184 * trace_array_put() and decrement the event's module refcount.
4185 *
4186 * Return: The trace event on success, ERR_PTR otherwise.
4187 */
trace_get_event_file(const char * instance,const char * system,const char * event)4188 struct trace_event_file *trace_get_event_file(const char *instance,
4189 const char *system,
4190 const char *event)
4191 {
4192 struct trace_array *tr = top_trace_array();
4193 struct trace_event_file *file = NULL;
4194 int ret = -EINVAL;
4195
4196 if (instance) {
4197 tr = trace_array_find_get(instance);
4198 if (!tr)
4199 return ERR_PTR(-ENOENT);
4200 } else {
4201 ret = trace_array_get(tr);
4202 if (ret)
4203 return ERR_PTR(ret);
4204 }
4205
4206 guard(mutex)(&event_mutex);
4207
4208 file = find_event_file(tr, system, event);
4209 if (!file) {
4210 trace_array_put(tr);
4211 return ERR_PTR(-EINVAL);
4212 }
4213
4214 /* Don't let event modules unload while in use */
4215 ret = trace_event_try_get_ref(file->event_call);
4216 if (!ret) {
4217 trace_array_put(tr);
4218 return ERR_PTR(-EBUSY);
4219 }
4220
4221 return file;
4222 }
4223 EXPORT_SYMBOL_GPL(trace_get_event_file);
4224
4225 /**
4226 * trace_put_event_file - Release a file from trace_get_event_file()
4227 * @file: The trace event file
4228 *
4229 * If a file was retrieved using trace_get_event_file(), this should
4230 * be called when it's no longer needed. It will cancel the previous
4231 * trace_array_get() called by that function, and decrement the
4232 * event's module refcount.
4233 */
trace_put_event_file(struct trace_event_file * file)4234 void trace_put_event_file(struct trace_event_file *file)
4235 {
4236 mutex_lock(&event_mutex);
4237 trace_event_put_ref(file->event_call);
4238 mutex_unlock(&event_mutex);
4239
4240 trace_array_put(file->tr);
4241 }
4242 EXPORT_SYMBOL_GPL(trace_put_event_file);
4243
4244 #ifdef CONFIG_DYNAMIC_FTRACE
4245 struct event_probe_data {
4246 struct trace_event_file *file;
4247 unsigned long count;
4248 int ref;
4249 bool enable;
4250 };
4251
update_event_probe(struct event_probe_data * data)4252 static void update_event_probe(struct event_probe_data *data)
4253 {
4254 if (data->enable)
4255 clear_bit(EVENT_FILE_FL_SOFT_DISABLED_BIT, &data->file->flags);
4256 else
4257 set_bit(EVENT_FILE_FL_SOFT_DISABLED_BIT, &data->file->flags);
4258 }
4259
4260 static void
event_enable_probe(unsigned long ip,unsigned long parent_ip,struct trace_array * tr,struct ftrace_probe_ops * ops,void * data)4261 event_enable_probe(unsigned long ip, unsigned long parent_ip,
4262 struct trace_array *tr, struct ftrace_probe_ops *ops,
4263 void *data)
4264 {
4265 struct ftrace_func_mapper *mapper = data;
4266 struct event_probe_data *edata;
4267 void **pdata;
4268
4269 pdata = ftrace_func_mapper_find_ip(mapper, ip);
4270 if (!pdata || !*pdata)
4271 return;
4272
4273 edata = *pdata;
4274 update_event_probe(edata);
4275 }
4276
4277 static void
event_enable_count_probe(unsigned long ip,unsigned long parent_ip,struct trace_array * tr,struct ftrace_probe_ops * ops,void * data)4278 event_enable_count_probe(unsigned long ip, unsigned long parent_ip,
4279 struct trace_array *tr, struct ftrace_probe_ops *ops,
4280 void *data)
4281 {
4282 struct ftrace_func_mapper *mapper = data;
4283 struct event_probe_data *edata;
4284 void **pdata;
4285
4286 pdata = ftrace_func_mapper_find_ip(mapper, ip);
4287 if (!pdata || !*pdata)
4288 return;
4289
4290 edata = *pdata;
4291
4292 if (!edata->count)
4293 return;
4294
4295 /* Skip if the event is in a state we want to switch to */
4296 if (edata->enable == !(edata->file->flags & EVENT_FILE_FL_SOFT_DISABLED))
4297 return;
4298
4299 if (edata->count != -1)
4300 (edata->count)--;
4301
4302 update_event_probe(edata);
4303 }
4304
4305 static int
event_enable_print(struct seq_file * m,unsigned long ip,struct ftrace_probe_ops * ops,void * data)4306 event_enable_print(struct seq_file *m, unsigned long ip,
4307 struct ftrace_probe_ops *ops, void *data)
4308 {
4309 struct ftrace_func_mapper *mapper = data;
4310 struct event_probe_data *edata;
4311 void **pdata;
4312
4313 pdata = ftrace_func_mapper_find_ip(mapper, ip);
4314
4315 if (WARN_ON_ONCE(!pdata || !*pdata))
4316 return 0;
4317
4318 edata = *pdata;
4319
4320 seq_printf(m, "%ps:", (void *)ip);
4321
4322 seq_printf(m, "%s:%s:%s",
4323 edata->enable ? ENABLE_EVENT_STR : DISABLE_EVENT_STR,
4324 edata->file->event_call->class->system,
4325 trace_event_name(edata->file->event_call));
4326
4327 if (edata->count == -1)
4328 seq_puts(m, ":unlimited\n");
4329 else
4330 seq_printf(m, ":count=%ld\n", edata->count);
4331
4332 return 0;
4333 }
4334
4335 static int
event_enable_init(struct ftrace_probe_ops * ops,struct trace_array * tr,unsigned long ip,void * init_data,void ** data)4336 event_enable_init(struct ftrace_probe_ops *ops, struct trace_array *tr,
4337 unsigned long ip, void *init_data, void **data)
4338 {
4339 struct ftrace_func_mapper *mapper = *data;
4340 struct event_probe_data *edata = init_data;
4341 int ret;
4342
4343 if (!mapper) {
4344 mapper = allocate_ftrace_func_mapper();
4345 if (!mapper)
4346 return -ENODEV;
4347 *data = mapper;
4348 }
4349
4350 ret = ftrace_func_mapper_add_ip(mapper, ip, edata);
4351 if (ret < 0)
4352 return ret;
4353
4354 edata->ref++;
4355
4356 return 0;
4357 }
4358
free_probe_data(void * data)4359 static int free_probe_data(void *data)
4360 {
4361 struct event_probe_data *edata = data;
4362
4363 edata->ref--;
4364 if (!edata->ref) {
4365 /* Remove soft mode */
4366 __ftrace_event_enable_disable(edata->file, 0, 1);
4367 trace_event_put_ref(edata->file->event_call);
4368 kfree(edata);
4369 }
4370 return 0;
4371 }
4372
4373 static void
event_enable_free(struct ftrace_probe_ops * ops,struct trace_array * tr,unsigned long ip,void * data)4374 event_enable_free(struct ftrace_probe_ops *ops, struct trace_array *tr,
4375 unsigned long ip, void *data)
4376 {
4377 struct ftrace_func_mapper *mapper = data;
4378 struct event_probe_data *edata;
4379
4380 if (!ip) {
4381 if (!mapper)
4382 return;
4383 free_ftrace_func_mapper(mapper, free_probe_data);
4384 return;
4385 }
4386
4387 edata = ftrace_func_mapper_remove_ip(mapper, ip);
4388
4389 if (WARN_ON_ONCE(!edata))
4390 return;
4391
4392 if (WARN_ON_ONCE(edata->ref <= 0))
4393 return;
4394
4395 free_probe_data(edata);
4396 }
4397
4398 static struct ftrace_probe_ops event_enable_probe_ops = {
4399 .func = event_enable_probe,
4400 .print = event_enable_print,
4401 .init = event_enable_init,
4402 .free = event_enable_free,
4403 };
4404
4405 static struct ftrace_probe_ops event_enable_count_probe_ops = {
4406 .func = event_enable_count_probe,
4407 .print = event_enable_print,
4408 .init = event_enable_init,
4409 .free = event_enable_free,
4410 };
4411
4412 static struct ftrace_probe_ops event_disable_probe_ops = {
4413 .func = event_enable_probe,
4414 .print = event_enable_print,
4415 .init = event_enable_init,
4416 .free = event_enable_free,
4417 };
4418
4419 static struct ftrace_probe_ops event_disable_count_probe_ops = {
4420 .func = event_enable_count_probe,
4421 .print = event_enable_print,
4422 .init = event_enable_init,
4423 .free = event_enable_free,
4424 };
4425
4426 static int
event_enable_func(struct trace_array * tr,struct ftrace_hash * hash,char * glob,char * cmd,char * param,int enabled)4427 event_enable_func(struct trace_array *tr, struct ftrace_hash *hash,
4428 char *glob, char *cmd, char *param, int enabled)
4429 {
4430 struct trace_event_file *file;
4431 struct ftrace_probe_ops *ops;
4432 struct event_probe_data *data;
4433 unsigned long count = -1;
4434 const char *system;
4435 const char *event;
4436 char *number;
4437 bool enable;
4438 int ret;
4439
4440 if (!tr)
4441 return -ENODEV;
4442
4443 /* hash funcs only work with set_ftrace_filter */
4444 if (!enabled || !param)
4445 return -EINVAL;
4446
4447 system = strsep(¶m, ":");
4448 if (!param)
4449 return -EINVAL;
4450
4451 event = strsep(¶m, ":");
4452
4453 guard(mutex)(&event_mutex);
4454
4455 file = find_event_file(tr, system, event);
4456 if (!file)
4457 return -EINVAL;
4458
4459 enable = strcmp(cmd, ENABLE_EVENT_STR) == 0;
4460
4461 if (enable)
4462 ops = param ? &event_enable_count_probe_ops : &event_enable_probe_ops;
4463 else
4464 ops = param ? &event_disable_count_probe_ops : &event_disable_probe_ops;
4465
4466 if (glob[0] == '!')
4467 return unregister_ftrace_function_probe_func(glob+1, tr, ops);
4468
4469 if (param) {
4470 number = strsep(¶m, ":");
4471
4472 if (!strlen(number))
4473 return -EINVAL;
4474
4475 /*
4476 * We use the callback data field (which is a pointer)
4477 * as our counter.
4478 */
4479 ret = kstrtoul(number, 0, &count);
4480 if (ret)
4481 return ret;
4482 }
4483
4484 /* Don't let event modules unload while probe registered */
4485 ret = trace_event_try_get_ref(file->event_call);
4486 if (!ret)
4487 return -EBUSY;
4488
4489 ret = __ftrace_event_enable_disable(file, 1, 1);
4490 if (ret < 0)
4491 goto out_put;
4492
4493 ret = -ENOMEM;
4494 data = kzalloc_obj(*data);
4495 if (!data)
4496 goto out_put;
4497
4498 data->enable = enable;
4499 data->count = count;
4500 data->file = file;
4501
4502 ret = register_ftrace_function_probe(glob, tr, ops, data);
4503 /*
4504 * The above returns on success the # of functions enabled,
4505 * but if it didn't find any functions it returns zero.
4506 * Consider no functions a failure too.
4507 */
4508
4509 /* Just return zero, not the number of enabled functions */
4510 if (ret > 0)
4511 return 0;
4512
4513 kfree(data);
4514
4515 if (!ret)
4516 ret = -ENOENT;
4517
4518 __ftrace_event_enable_disable(file, 0, 1);
4519 out_put:
4520 trace_event_put_ref(file->event_call);
4521 return ret;
4522 }
4523
4524 static struct ftrace_func_command event_enable_cmd = {
4525 .name = ENABLE_EVENT_STR,
4526 .func = event_enable_func,
4527 };
4528
4529 static struct ftrace_func_command event_disable_cmd = {
4530 .name = DISABLE_EVENT_STR,
4531 .func = event_enable_func,
4532 };
4533
register_event_cmds(void)4534 static __init int register_event_cmds(void)
4535 {
4536 int ret;
4537
4538 ret = register_ftrace_command(&event_enable_cmd);
4539 if (WARN_ON(ret < 0))
4540 return ret;
4541 ret = register_ftrace_command(&event_disable_cmd);
4542 if (WARN_ON(ret < 0))
4543 unregister_ftrace_command(&event_enable_cmd);
4544 return ret;
4545 }
4546 #else
register_event_cmds(void)4547 static inline int register_event_cmds(void) { return 0; }
4548 #endif /* CONFIG_DYNAMIC_FTRACE */
4549
4550 /*
4551 * The top level array and trace arrays created by boot-time tracing
4552 * have already had its trace_event_file descriptors created in order
4553 * to allow for early events to be recorded.
4554 * This function is called after the tracefs has been initialized,
4555 * and we now have to create the files associated to the events.
4556 */
__trace_early_add_event_dirs(struct trace_array * tr)4557 static void __trace_early_add_event_dirs(struct trace_array *tr)
4558 {
4559 struct trace_event_file *file;
4560 int ret;
4561
4562
4563 list_for_each_entry(file, &tr->events, list) {
4564 ret = event_create_dir(tr->event_dir, file);
4565 if (ret < 0)
4566 pr_warn("Could not create directory for event %s\n",
4567 trace_event_name(file->event_call));
4568 }
4569 }
4570
4571 /*
4572 * For early boot up, the top trace array and the trace arrays created
4573 * by boot-time tracing require to have a list of events that can be
4574 * enabled. This must be done before the filesystem is set up in order
4575 * to allow events to be traced early.
4576 */
__trace_early_add_events(struct trace_array * tr)4577 void __trace_early_add_events(struct trace_array *tr)
4578 {
4579 struct trace_event_call *call;
4580 int ret;
4581
4582 list_for_each_entry(call, &ftrace_events, list) {
4583 /* Early boot up should not have any modules loaded */
4584 if (!(call->flags & TRACE_EVENT_FL_DYNAMIC) &&
4585 WARN_ON_ONCE(call->module))
4586 continue;
4587
4588 ret = __trace_early_add_new_event(call, tr);
4589 if (ret < 0)
4590 pr_warn("Could not create early event %s\n",
4591 trace_event_name(call));
4592 }
4593 }
4594
4595 /* Remove the event directory structure for a trace directory. */
4596 static void
__trace_remove_event_dirs(struct trace_array * tr)4597 __trace_remove_event_dirs(struct trace_array *tr)
4598 {
4599 struct trace_event_file *file, *next;
4600
4601 list_for_each_entry_safe(file, next, &tr->events, list)
4602 remove_event_file_dir(file);
4603 }
4604
__add_event_to_tracers(struct trace_event_call * call)4605 static void __add_event_to_tracers(struct trace_event_call *call)
4606 {
4607 struct trace_array *tr;
4608
4609 list_for_each_entry(tr, &ftrace_trace_arrays, list)
4610 __trace_add_new_event(call, tr);
4611 }
4612
4613 extern struct trace_event_call *__start_ftrace_events[];
4614 extern struct trace_event_call *__stop_ftrace_events[];
4615
4616 static char bootup_event_buf[COMMAND_LINE_SIZE] __initdata;
4617 static struct seq_buf bootup_event_seq __initdata = {
4618 .buffer = bootup_event_buf,
4619 .size = sizeof(bootup_event_buf),
4620 };
4621
setup_trace_event(char * str)4622 static __init int setup_trace_event(char *str)
4623 {
4624 if (seq_buf_used(&bootup_event_seq) > 0)
4625 seq_buf_puts(&bootup_event_seq, ",");
4626
4627 seq_buf_puts(&bootup_event_seq, str);
4628
4629 if (seq_buf_has_overflowed(&bootup_event_seq))
4630 return -ENOMEM;
4631
4632 trace_set_ring_buffer_expanded(NULL);
4633 disable_tracing_selftest("running event tracing");
4634
4635 return 1;
4636 }
4637 __setup("trace_event=", setup_trace_event);
4638
events_callback(const char * name,umode_t * mode,void ** data,const struct file_operations ** fops)4639 static int events_callback(const char *name, umode_t *mode, void **data,
4640 const struct file_operations **fops)
4641 {
4642 if (strcmp(name, "enable") == 0) {
4643 *mode = TRACE_MODE_WRITE;
4644 *fops = &ftrace_tr_enable_fops;
4645 return 1;
4646 }
4647
4648 if (strcmp(name, "header_page") == 0) {
4649 *mode = TRACE_MODE_READ;
4650 *fops = &ftrace_show_header_page_fops;
4651
4652 } else if (strcmp(name, "header_event") == 0) {
4653 *mode = TRACE_MODE_READ;
4654 *fops = &ftrace_show_header_event_fops;
4655 } else
4656 return 0;
4657
4658 return 1;
4659 }
4660
4661 /* Expects to have event_mutex held when called */
4662 static int
create_event_toplevel_files(struct dentry * parent,struct trace_array * tr)4663 create_event_toplevel_files(struct dentry *parent, struct trace_array *tr)
4664 {
4665 struct eventfs_inode *e_events;
4666 struct dentry *entry;
4667 int nr_entries;
4668 static struct eventfs_entry events_entries[] = {
4669 {
4670 .name = "header_page",
4671 .callback = events_callback,
4672 },
4673 {
4674 .name = "header_event",
4675 .callback = events_callback,
4676 },
4677 #define NR_RO_TOP_ENTRIES 2
4678 /* Readonly files must be above this line and counted by NR_RO_TOP_ENTRIES. */
4679 {
4680 .name = "enable",
4681 .callback = events_callback,
4682 },
4683 };
4684
4685 if (!trace_array_is_readonly(tr)) {
4686 entry = trace_create_file("set_event", TRACE_MODE_WRITE, parent,
4687 tr, &ftrace_set_event_fops);
4688 if (!entry)
4689 return -ENOMEM;
4690
4691 /* There are not as crucial, just warn if they are not created */
4692 trace_create_file("show_event_filters", TRACE_MODE_READ, parent, tr,
4693 &ftrace_show_event_filters_fops);
4694
4695 trace_create_file("show_event_triggers", TRACE_MODE_READ, parent, tr,
4696 &ftrace_show_event_triggers_fops);
4697
4698 trace_create_file("set_event_pid", TRACE_MODE_WRITE, parent,
4699 tr, &ftrace_set_event_pid_fops);
4700
4701 trace_create_file("set_event_notrace_pid",
4702 TRACE_MODE_WRITE, parent, tr,
4703 &ftrace_set_event_notrace_pid_fops);
4704 nr_entries = ARRAY_SIZE(events_entries);
4705 } else {
4706 nr_entries = NR_RO_TOP_ENTRIES;
4707 }
4708
4709 e_events = eventfs_create_events_dir("events", parent, events_entries,
4710 nr_entries, tr);
4711 if (IS_ERR(e_events)) {
4712 pr_warn("Could not create tracefs 'events' directory\n");
4713 return -ENOMEM;
4714 }
4715
4716 tr->event_dir = e_events;
4717
4718 return 0;
4719 }
4720
4721 /**
4722 * event_trace_add_tracer - add a instance of a trace_array to events
4723 * @parent: The parent dentry to place the files/directories for events in
4724 * @tr: The trace array associated with these events
4725 *
4726 * When a new instance is created, it needs to set up its events
4727 * directory, as well as other files associated with events. It also
4728 * creates the event hierarchy in the @parent/events directory.
4729 *
4730 * Returns 0 on success.
4731 *
4732 * Must be called with event_mutex held.
4733 */
event_trace_add_tracer(struct dentry * parent,struct trace_array * tr)4734 int event_trace_add_tracer(struct dentry *parent, struct trace_array *tr)
4735 {
4736 int ret;
4737
4738 lockdep_assert_held(&event_mutex);
4739
4740 ret = create_event_toplevel_files(parent, tr);
4741 if (ret)
4742 goto out;
4743
4744 down_write(&trace_event_sem);
4745 /* If tr already has the event list, it is initialized in early boot. */
4746 if (unlikely(!list_empty(&tr->events)))
4747 __trace_early_add_event_dirs(tr);
4748 else
4749 __trace_add_event_dirs(tr);
4750 up_write(&trace_event_sem);
4751
4752 out:
4753 return ret;
4754 }
4755
4756 /*
4757 * The top trace array already had its file descriptors created.
4758 * Now the files themselves need to be created.
4759 */
4760 static __init int
early_event_add_tracer(struct dentry * parent,struct trace_array * tr)4761 early_event_add_tracer(struct dentry *parent, struct trace_array *tr)
4762 {
4763 int ret;
4764
4765 guard(mutex)(&event_mutex);
4766
4767 ret = create_event_toplevel_files(parent, tr);
4768 if (ret)
4769 return ret;
4770
4771 down_write(&trace_event_sem);
4772 __trace_early_add_event_dirs(tr);
4773 up_write(&trace_event_sem);
4774
4775 return 0;
4776 }
4777
4778 /* Must be called with event_mutex held */
event_trace_del_tracer(struct trace_array * tr)4779 int event_trace_del_tracer(struct trace_array *tr)
4780 {
4781 lockdep_assert_held(&event_mutex);
4782
4783 /* Disable any event triggers and associated soft-disabled events */
4784 clear_event_triggers(tr);
4785
4786 /* Clear the pid list */
4787 __ftrace_clear_event_pids(tr, TRACE_PIDS | TRACE_NO_PIDS);
4788
4789 /* Disable any running events */
4790 __ftrace_set_clr_event_nolock(tr, NULL, NULL, NULL, 0, NULL);
4791
4792 /* Make sure no more events are being executed */
4793 tracepoint_synchronize_unregister();
4794
4795 down_write(&trace_event_sem);
4796 __trace_remove_event_dirs(tr);
4797 eventfs_remove_events_dir(tr->event_dir);
4798 up_write(&trace_event_sem);
4799
4800 tr->event_dir = NULL;
4801
4802 return 0;
4803 }
4804
event_trace_memsetup(void)4805 static __init int event_trace_memsetup(void)
4806 {
4807 field_cachep = KMEM_CACHE(ftrace_event_field, SLAB_PANIC);
4808 file_cachep = KMEM_CACHE(trace_event_file, SLAB_PANIC);
4809 return 0;
4810 }
4811
4812 /*
4813 * Helper function to enable or disable a comma-separated list of events
4814 * from the bootup buffer.
4815 */
__early_set_events(struct trace_array * tr,char * buf,bool enable)4816 static __init void __early_set_events(struct trace_array *tr, char *buf, bool enable)
4817 {
4818 char *token;
4819
4820 while ((token = strsep(&buf, ","))) {
4821 if (*token) {
4822 if (enable) {
4823 if (ftrace_set_clr_event(tr, token, 1))
4824 pr_warn("Failed to enable trace event: %s\n", token);
4825 } else {
4826 ftrace_set_clr_event(tr, token, 0);
4827 }
4828 }
4829
4830 /* Put back the comma to allow this to be called again */
4831 if (buf)
4832 *(buf - 1) = ',';
4833 }
4834 }
4835
4836 /**
4837 * early_enable_events - enable events from the bootup buffer
4838 * @tr: The trace array to enable the events in
4839 * @buf: The buffer containing the comma separated list of events
4840 * @disable_first: If true, disable all events in @buf before enabling them
4841 *
4842 * This function enables events from the bootup buffer. If @disable_first
4843 * is true, it will first disable all events in the buffer before enabling
4844 * them.
4845 *
4846 * For syscall events, which rely on a global refcount to register the
4847 * SYSCALL_WORK_SYSCALL_TRACEPOINT flag (especially for pid 1), we must
4848 * ensure the refcount hits zero before re-enabling them. A simple
4849 * "disable then enable" per-event is not enough if multiple syscalls are
4850 * used, as the refcount will stay above zero. Thus, we need a two-phase
4851 * approach: disable all, then enable all.
4852 */
4853 __init void
early_enable_events(struct trace_array * tr,char * buf,bool disable_first)4854 early_enable_events(struct trace_array *tr, char *buf, bool disable_first)
4855 {
4856 if (disable_first)
4857 __early_set_events(tr, buf, false);
4858
4859 __early_set_events(tr, buf, true);
4860 }
4861
event_trace_enable(void)4862 static __init int event_trace_enable(void)
4863 {
4864 struct trace_array *tr = top_trace_array();
4865 struct trace_event_call **iter, *call;
4866 int ret;
4867
4868 if (!tr)
4869 return -ENODEV;
4870
4871 for_each_event(iter, __start_ftrace_events, __stop_ftrace_events) {
4872
4873 call = *iter;
4874 ret = event_init(call);
4875 if (!ret)
4876 list_add(&call->list, &ftrace_events);
4877 }
4878
4879 register_trigger_cmds();
4880
4881 /*
4882 * We need the top trace array to have a working set of trace
4883 * points at early init, before the debug files and directories
4884 * are created. Create the file entries now, and attach them
4885 * to the actual file dentries later.
4886 */
4887 __trace_early_add_events(tr);
4888
4889 seq_buf_str(&bootup_event_seq);
4890 early_enable_events(tr, bootup_event_buf, false);
4891
4892 trace_printk_start_comm();
4893
4894 register_event_cmds();
4895
4896
4897 return 0;
4898 }
4899
4900 /*
4901 * event_trace_enable() is called from trace_event_init() first to
4902 * initialize events and perhaps start any events that are on the
4903 * command line. Unfortunately, there are some events that will not
4904 * start this early, like the system call tracepoints that need
4905 * to set the %SYSCALL_WORK_SYSCALL_TRACEPOINT flag of pid 1. But
4906 * event_trace_enable() is called before pid 1 starts, and this flag
4907 * is never set, making the syscall tracepoint never get reached, but
4908 * the event is enabled regardless (and not doing anything).
4909 */
event_trace_enable_again(void)4910 static __init int event_trace_enable_again(void)
4911 {
4912 struct trace_array *tr;
4913
4914 tr = top_trace_array();
4915 if (!tr)
4916 return -ENODEV;
4917
4918 seq_buf_str(&bootup_event_seq);
4919 early_enable_events(tr, bootup_event_buf, true);
4920
4921 return 0;
4922 }
4923
4924 early_initcall(event_trace_enable_again);
4925
4926 /* Init fields which doesn't related to the tracefs */
event_trace_init_fields(void)4927 static __init int event_trace_init_fields(void)
4928 {
4929 if (trace_define_generic_fields())
4930 pr_warn("tracing: Failed to allocated generic fields");
4931
4932 if (trace_define_common_fields())
4933 pr_warn("tracing: Failed to allocate common fields");
4934
4935 return 0;
4936 }
4937
event_trace_init(void)4938 __init int event_trace_init(void)
4939 {
4940 struct trace_array *tr;
4941 int ret;
4942
4943 tr = top_trace_array();
4944 if (!tr)
4945 return -ENODEV;
4946
4947 trace_create_file("available_events", TRACE_MODE_READ,
4948 NULL, tr, &ftrace_avail_fops);
4949
4950 ret = early_event_add_tracer(NULL, tr);
4951 if (ret)
4952 return ret;
4953
4954 #ifdef CONFIG_MODULES
4955 ret = register_module_notifier(&trace_module_nb);
4956 if (ret)
4957 pr_warn("Failed to register trace events module notifier\n");
4958 #endif
4959
4960 eventdir_initialized = true;
4961
4962 return 0;
4963 }
4964
trace_event_init(void)4965 void __init trace_event_init(void)
4966 {
4967 event_trace_memsetup();
4968 init_ftrace_syscalls();
4969 event_trace_enable();
4970 event_trace_init_fields();
4971 }
4972
4973 #ifdef CONFIG_EVENT_TRACE_STARTUP_TEST
4974
4975 static DEFINE_SPINLOCK(test_spinlock);
4976 static DEFINE_SPINLOCK(test_spinlock_irq);
4977 static DEFINE_MUTEX(test_mutex);
4978
test_work(struct work_struct * dummy)4979 static __init void test_work(struct work_struct *dummy)
4980 {
4981 spin_lock(&test_spinlock);
4982 spin_lock_irq(&test_spinlock_irq);
4983 udelay(1);
4984 spin_unlock_irq(&test_spinlock_irq);
4985 spin_unlock(&test_spinlock);
4986
4987 mutex_lock(&test_mutex);
4988 msleep(1);
4989 mutex_unlock(&test_mutex);
4990 }
4991
event_test_thread(void * unused)4992 static __init int event_test_thread(void *unused)
4993 {
4994 void *test_malloc;
4995
4996 test_malloc = kmalloc(1234, GFP_KERNEL);
4997 if (!test_malloc)
4998 pr_info("failed to kmalloc\n");
4999
5000 schedule_on_each_cpu(test_work);
5001
5002 kfree(test_malloc);
5003
5004 set_current_state(TASK_INTERRUPTIBLE);
5005 while (!kthread_should_stop()) {
5006 schedule();
5007 set_current_state(TASK_INTERRUPTIBLE);
5008 }
5009 __set_current_state(TASK_RUNNING);
5010
5011 return 0;
5012 }
5013
5014 /*
5015 * Do various things that may trigger events.
5016 */
event_test_stuff(void)5017 static __init void event_test_stuff(void)
5018 {
5019 struct task_struct *test_thread;
5020
5021 test_thread = kthread_run(event_test_thread, NULL, "test-events");
5022 msleep(1);
5023 kthread_stop(test_thread);
5024 }
5025
5026 /*
5027 * For every trace event defined, we will test each trace point separately,
5028 * and then by groups, and finally all trace points.
5029 */
event_trace_self_tests(void)5030 static __init void event_trace_self_tests(void)
5031 {
5032 struct trace_subsystem_dir *dir;
5033 struct trace_event_file *file;
5034 struct trace_event_call *call;
5035 struct event_subsystem *system;
5036 struct trace_array *tr;
5037 int ret;
5038
5039 tr = top_trace_array();
5040 if (!tr)
5041 return;
5042
5043 pr_info("Running tests on trace events:\n");
5044
5045 list_for_each_entry(file, &tr->events, list) {
5046
5047 call = file->event_call;
5048
5049 /* Only test those that have a probe */
5050 if (!call->class || !call->class->probe)
5051 continue;
5052
5053 /*
5054 * Testing syscall events here is pretty useless, but
5055 * we still do it if configured. But this is time consuming.
5056 * What we really need is a user thread to perform the
5057 * syscalls as we test.
5058 */
5059 #ifndef CONFIG_EVENT_TRACE_TEST_SYSCALLS
5060 if (call->class->system &&
5061 strcmp(call->class->system, "syscalls") == 0)
5062 continue;
5063 #endif
5064
5065 pr_info("Testing event %s: ", trace_event_name(call));
5066
5067 /*
5068 * If an event is already enabled, someone is using
5069 * it and the self test should not be on.
5070 */
5071 if (file->flags & EVENT_FILE_FL_ENABLED) {
5072 pr_warn("Enabled event during self test!\n");
5073 WARN_ON_ONCE(1);
5074 continue;
5075 }
5076
5077 ftrace_event_enable_disable(file, 1);
5078 event_test_stuff();
5079 ftrace_event_enable_disable(file, 0);
5080
5081 pr_cont("OK\n");
5082 }
5083
5084 /* Now test at the sub system level */
5085
5086 pr_info("Running tests on trace event systems:\n");
5087
5088 list_for_each_entry(dir, &tr->systems, list) {
5089
5090 system = dir->subsystem;
5091
5092 /* the ftrace system is special, skip it */
5093 if (strcmp(system->name, "ftrace") == 0)
5094 continue;
5095
5096 pr_info("Testing event system %s: ", system->name);
5097
5098 ret = __ftrace_set_clr_event(tr, NULL, system->name, NULL, 1, NULL);
5099 if (WARN_ON_ONCE(ret)) {
5100 pr_warn("error enabling system %s\n",
5101 system->name);
5102 continue;
5103 }
5104
5105 event_test_stuff();
5106
5107 ret = __ftrace_set_clr_event(tr, NULL, system->name, NULL, 0, NULL);
5108 if (WARN_ON_ONCE(ret)) {
5109 pr_warn("error disabling system %s\n",
5110 system->name);
5111 continue;
5112 }
5113
5114 pr_cont("OK\n");
5115 }
5116
5117 /* Test with all events enabled */
5118
5119 pr_info("Running tests on all trace events:\n");
5120 pr_info("Testing all events: ");
5121
5122 ret = __ftrace_set_clr_event(tr, NULL, NULL, NULL, 1, NULL);
5123 if (WARN_ON_ONCE(ret)) {
5124 pr_warn("error enabling all events\n");
5125 return;
5126 }
5127
5128 event_test_stuff();
5129
5130 /* reset sysname */
5131 ret = __ftrace_set_clr_event(tr, NULL, NULL, NULL, 0, NULL);
5132 if (WARN_ON_ONCE(ret)) {
5133 pr_warn("error disabling all events\n");
5134 return;
5135 }
5136
5137 pr_cont("OK\n");
5138 }
5139
5140 #ifdef CONFIG_FUNCTION_TRACER
5141
5142 static DEFINE_PER_CPU(atomic_t, ftrace_test_event_disable);
5143
5144 static struct trace_event_file event_trace_file __initdata;
5145
5146 static void __init
function_test_events_call(unsigned long ip,unsigned long parent_ip,struct ftrace_ops * op,struct ftrace_regs * regs)5147 function_test_events_call(unsigned long ip, unsigned long parent_ip,
5148 struct ftrace_ops *op, struct ftrace_regs *regs)
5149 {
5150 struct trace_buffer *buffer;
5151 struct ring_buffer_event *event;
5152 struct ftrace_entry *entry;
5153 unsigned int trace_ctx;
5154 long disabled;
5155 int cpu;
5156
5157 trace_ctx = tracing_gen_ctx();
5158 preempt_disable_notrace();
5159 cpu = raw_smp_processor_id();
5160 disabled = atomic_inc_return(&per_cpu(ftrace_test_event_disable, cpu));
5161
5162 if (disabled != 1)
5163 goto out;
5164
5165 event = trace_event_buffer_lock_reserve(&buffer, &event_trace_file,
5166 TRACE_FN, sizeof(*entry),
5167 trace_ctx);
5168 if (!event)
5169 goto out;
5170 entry = ring_buffer_event_data(event);
5171 entry->ip = ip;
5172 entry->parent_ip = parent_ip;
5173
5174 event_trigger_unlock_commit(&event_trace_file, buffer, event,
5175 entry, trace_ctx);
5176 out:
5177 atomic_dec(&per_cpu(ftrace_test_event_disable, cpu));
5178 preempt_enable_notrace();
5179 }
5180
5181 static struct ftrace_ops trace_ops __initdata =
5182 {
5183 .func = function_test_events_call,
5184 };
5185
event_trace_self_test_with_function(void)5186 static __init void event_trace_self_test_with_function(void)
5187 {
5188 int ret;
5189
5190 event_trace_file.tr = top_trace_array();
5191 if (WARN_ON(!event_trace_file.tr))
5192 return;
5193
5194 ret = register_ftrace_function(&trace_ops);
5195 if (WARN_ON(ret < 0)) {
5196 pr_info("Failed to enable function tracer for event tests\n");
5197 return;
5198 }
5199 pr_info("Running tests again, along with the function tracer\n");
5200 event_trace_self_tests();
5201 unregister_ftrace_function(&trace_ops);
5202 }
5203 #else
event_trace_self_test_with_function(void)5204 static __init void event_trace_self_test_with_function(void)
5205 {
5206 }
5207 #endif
5208
event_trace_self_tests_init(void)5209 static __init int event_trace_self_tests_init(void)
5210 {
5211 if (!tracing_selftest_disabled) {
5212 event_trace_self_tests();
5213 event_trace_self_test_with_function();
5214 }
5215
5216 return 0;
5217 }
5218
5219 late_initcall(event_trace_self_tests_init);
5220
5221 #endif
5222