diff options
| author | Ingo Molnar <mingo@kernel.org> | 2016-04-27 11:02:24 -0400 |
|---|---|---|
| committer | Ingo Molnar <mingo@kernel.org> | 2016-04-27 11:02:24 -0400 |
| commit | a8944c5bf86dc6c153a71f2a386738c0d3f5ff9c (patch) | |
| tree | a251b1d510831dc071eadbbbe3e38a85fe643365 /kernel/bpf/stackmap.c | |
| parent | 67d61296ffcc850bffdd4466430cb91e5328f39a (diff) | |
| parent | 4cb93446c587d56e2a54f4f83113daba2c0b6dee (diff) | |
Merge tag 'perf-core-for-mingo-20160427' of git://git.kernel.org/pub/scm/linux/kernel/git/acme/linux into perf/core
Pull perf/core improvements and fixes from Arnaldo Carvalho de Melo:
User visible changes:
- perf trace --pf maj/min/all works with --call-graph: (Arnaldo Carvalho de Melo)
Tracing write syscalls and major page faults with callchains while starting
firefox, limiting the stack to 5 frames:
# perf trace -e write --pf maj --max-stack 5 firefox
589.549 ( 0.014 ms): firefox/15377 write(fd: 4, buf: 0x7fff80acc898, count: 151) = 151
[0xfaed] (/usr/lib64/libpthread-2.22.so)
fire_glxtest_process+0x5c (/usr/lib64/firefox/libxul.so)
InstallGdkErrorHandler+0x41 (/usr/lib64/firefox/libxul.so)
XREMain::XRE_mainInit+0x12c (/usr/lib64/firefox/libxul.so)
XREMain::XRE_main+0x1e4 (/usr/lib64/firefox/libxul.so)
760.704 ( 0.000 ms): firefox/15332 majfault [gtk_tree_view_accessible_get_type+0x0] => /usr/lib64/libgtk-3.so.0.1800.9@0xa0850 (x.)
gtk_tree_view_accessible_get_type+0x0 (/usr/lib64/libgtk-3.so.0.1800.9)
gtk_tree_view_class_intern_init+0x1a54 (/usr/lib64/libgtk-3.so.0.1800.9)
g_type_class_ref+0x6dd (/usr/lib64/libgobject-2.0.so.0.4600.2)
[0x115378] (/usr/lib64/libgnutls.so.30.6.3)
This automagically selects "--call-graph dwarf", use "--call-graph fp" on systems
where -fno-omit-frame-pointer was used to built the components of interest, to
incur in less overhead, or tune "--call-graph dwarf" appropriately, see 'perf record --help'.
- Allow /proc/sys/kernel/perf_event_max_stack, that defaults to the old hard coded value
of PERF_MAX_STACK_DEPTH (127), useful for huge callstacks for things like Groovy, Ruby, etc,
and also to reduce overhead by limiting it to a smaller value, upcoming work will allow
this to be done per-event (Arnaldo Carvalho de Melo)
- Make 'perf trace --min-stack' be honoured by --pf and --event (Arnaldo Carvalho de Melo)
- Make 'perf evlist -v' decode perf_event_attr->branch_sample_type (Arnaldo Carvalho de Melo)
# perf record --call lbr usleep 1
# perf evlist -v
cycles:ppp: ... sample_type: IP|TID|TIME|CALLCHAIN|PERIOD|BRANCH_STACK, ...
branch_sample_type: USER|CALL_STACK|NO_FLAGS|NO_CYCLES
#
- Clear dummy entry accumulated period, fixing such 'perf top/report' output
as: (Kan Liang)
4769.98% 0.01% 0.00% 0.01% tchain_edit [kernel] [k] update_fast_timekeeper
- System calls with pid_t arguments gets them augmented with the COMM event
more thoroughly:
# trace -e perf_event_open perf stat -e cycles -p 15608
6.876 ( 0.014 ms): perf_event_open(attr_uptr: 0x2ae20d8, pid: 15608 (hexchat), cpu: -1, group_fd: -1, flags: FD_CLOEXEC) = 3
6.882 ( 0.005 ms): perf_event_open(attr_uptr: 0x2ae20d8, pid: 15639 (gmain), cpu: -1, group_fd: -1, flags: FD_CLOEXEC) = 4
6.889 ( 0.005 ms): perf_event_open(attr_uptr: 0x2ae20d8, pid: 15640 (gdbus), cpu: -1, group_fd: -1, flags: FD_CLOEXEC) = 5
^^^^^^^^^^^^^^^^^^
^C
- Fix offline module name mismatch issue in 'perf probe' (Ravi Bangoria)
- Fix module probe issue if no dwarf support in (Ravi Bangoria)
Assorted fixes:
- Fix off-by-one in write_buildid() (Andrey Ryabinin)
- Fix segfault when printing callchains in 'perf script' (Chris Phlipot)
- Replace assignment with comparison on assert check in 'perf test' entry (Colin Ian King)
- Fix off-by-one comparison in intel-pt code (Colin Ian King)
- Close target file on error path in 'perf probe' (Masami Hiramatsu)
- Set default kprobe group name if not given in 'perf probe' (Masami Hiramatsu)
- Avoid partial perf_event_header reads (Wang Nan)
Infrastructure changes:
- Update x86's syscall_64.tbl copy, adding preadv2 & pwritev2 (Arnaldo Carvalho de Melo)
- Make the x86 clean quiet wrt syscall table removal (Jiri Olsa)
Cleanups:
- Simplify wrapper for LOCK_PI in 'perf bench futex' (Davidlohr Bueso)
- Remove duplicate const qualifier (Eric Engestrom)
Signed-off-by: Arnaldo Carvalho de Melo <acme@redhat.com>
Signed-off-by: Ingo Molnar <mingo@kernel.org>
Diffstat (limited to 'kernel/bpf/stackmap.c')
| -rw-r--r-- | kernel/bpf/stackmap.c | 8 |
1 files changed, 4 insertions, 4 deletions
diff --git a/kernel/bpf/stackmap.c b/kernel/bpf/stackmap.c index 499d9e933f8e..f5a19548be12 100644 --- a/kernel/bpf/stackmap.c +++ b/kernel/bpf/stackmap.c | |||
| @@ -66,7 +66,7 @@ static struct bpf_map *stack_map_alloc(union bpf_attr *attr) | |||
| 66 | /* check sanity of attributes */ | 66 | /* check sanity of attributes */ |
| 67 | if (attr->max_entries == 0 || attr->key_size != 4 || | 67 | if (attr->max_entries == 0 || attr->key_size != 4 || |
| 68 | value_size < 8 || value_size % 8 || | 68 | value_size < 8 || value_size % 8 || |
| 69 | value_size / 8 > PERF_MAX_STACK_DEPTH) | 69 | value_size / 8 > sysctl_perf_event_max_stack) |
| 70 | return ERR_PTR(-EINVAL); | 70 | return ERR_PTR(-EINVAL); |
| 71 | 71 | ||
| 72 | /* hash table size must be power of 2 */ | 72 | /* hash table size must be power of 2 */ |
| @@ -124,8 +124,8 @@ static u64 bpf_get_stackid(u64 r1, u64 r2, u64 flags, u64 r4, u64 r5) | |||
| 124 | struct perf_callchain_entry *trace; | 124 | struct perf_callchain_entry *trace; |
| 125 | struct stack_map_bucket *bucket, *new_bucket, *old_bucket; | 125 | struct stack_map_bucket *bucket, *new_bucket, *old_bucket; |
| 126 | u32 max_depth = map->value_size / 8; | 126 | u32 max_depth = map->value_size / 8; |
| 127 | /* stack_map_alloc() checks that max_depth <= PERF_MAX_STACK_DEPTH */ | 127 | /* stack_map_alloc() checks that max_depth <= sysctl_perf_event_max_stack */ |
| 128 | u32 init_nr = PERF_MAX_STACK_DEPTH - max_depth; | 128 | u32 init_nr = sysctl_perf_event_max_stack - max_depth; |
| 129 | u32 skip = flags & BPF_F_SKIP_FIELD_MASK; | 129 | u32 skip = flags & BPF_F_SKIP_FIELD_MASK; |
| 130 | u32 hash, id, trace_nr, trace_len; | 130 | u32 hash, id, trace_nr, trace_len; |
| 131 | bool user = flags & BPF_F_USER_STACK; | 131 | bool user = flags & BPF_F_USER_STACK; |
| @@ -143,7 +143,7 @@ static u64 bpf_get_stackid(u64 r1, u64 r2, u64 flags, u64 r4, u64 r5) | |||
| 143 | return -EFAULT; | 143 | return -EFAULT; |
| 144 | 144 | ||
| 145 | /* get_perf_callchain() guarantees that trace->nr >= init_nr | 145 | /* get_perf_callchain() guarantees that trace->nr >= init_nr |
| 146 | * and trace-nr <= PERF_MAX_STACK_DEPTH, so trace_nr <= max_depth | 146 | * and trace-nr <= sysctl_perf_event_max_stack, so trace_nr <= max_depth |
| 147 | */ | 147 | */ |
| 148 | trace_nr = trace->nr - init_nr; | 148 | trace_nr = trace->nr - init_nr; |
| 149 | 149 | ||
