aboutsummaryrefslogtreecommitdiffstats
path: root/include/gk20a/fecs_trace_gk20a.c
diff options
context:
space:
mode:
Diffstat (limited to 'include/gk20a/fecs_trace_gk20a.c')
-rw-r--r--include/gk20a/fecs_trace_gk20a.c744
1 files changed, 744 insertions, 0 deletions
diff --git a/include/gk20a/fecs_trace_gk20a.c b/include/gk20a/fecs_trace_gk20a.c
new file mode 100644
index 0000000..5c1c5e0
--- /dev/null
+++ b/include/gk20a/fecs_trace_gk20a.c
@@ -0,0 +1,744 @@
1/*
2 * Copyright (c) 2016-2019, NVIDIA CORPORATION. All rights reserved.
3 *
4 * Permission is hereby granted, free of charge, to any person obtaining a
5 * copy of this software and associated documentation files (the "Software"),
6 * to deal in the Software without restriction, including without limitation
7 * the rights to use, copy, modify, merge, publish, distribute, sublicense,
8 * and/or sell copies of the Software, and to permit persons to whom the
9 * Software is furnished to do so, subject to the following conditions:
10 *
11 * The above copyright notice and this permission notice shall be included in
12 * all copies or substantial portions of the Software.
13 *
14 * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
15 * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
16 * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
17 * THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
18 * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
19 * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
20 * DEALINGS IN THE SOFTWARE.
21 */
22
23#include <nvgpu/kmem.h>
24#include <nvgpu/dma.h>
25#include <nvgpu/enabled.h>
26#include <nvgpu/bug.h>
27#include <nvgpu/hashtable.h>
28#include <nvgpu/circ_buf.h>
29#include <nvgpu/thread.h>
30#include <nvgpu/barrier.h>
31#include <nvgpu/mm.h>
32#include <nvgpu/enabled.h>
33#include <nvgpu/ctxsw_trace.h>
34#include <nvgpu/io.h>
35#include <nvgpu/utils.h>
36#include <nvgpu/timers.h>
37#include <nvgpu/channel.h>
38
39#include "fecs_trace_gk20a.h"
40#include "gk20a.h"
41#include "gr_gk20a.h"
42
43#include <nvgpu/log.h>
44#include <nvgpu/fecs_trace.h>
45
46#include <nvgpu/hw/gk20a/hw_ctxsw_prog_gk20a.h>
47#include <nvgpu/hw/gk20a/hw_gr_gk20a.h>
48
49struct gk20a_fecs_trace_hash_ent {
50 u32 context_ptr;
51 pid_t pid;
52 struct hlist_node node;
53};
54
55struct gk20a_fecs_trace {
56
57 DECLARE_HASHTABLE(pid_hash_table, GK20A_FECS_TRACE_HASH_BITS);
58 struct nvgpu_mutex hash_lock;
59 struct nvgpu_mutex poll_lock;
60 struct nvgpu_thread poll_task;
61 bool init;
62 struct nvgpu_mutex enable_lock;
63 u32 enable_count;
64};
65
66#ifdef CONFIG_GK20A_CTXSW_TRACE
67u32 gk20a_fecs_trace_record_ts_tag_invalid_ts_v(void)
68{
69 return ctxsw_prog_record_timestamp_timestamp_hi_tag_invalid_timestamp_v();
70}
71
72u32 gk20a_fecs_trace_record_ts_tag_v(u64 ts)
73{
74 return ctxsw_prog_record_timestamp_timestamp_hi_tag_v((u32) (ts >> 32));
75}
76
77u64 gk20a_fecs_trace_record_ts_timestamp_v(u64 ts)
78{
79 return ts & ~(((u64)ctxsw_prog_record_timestamp_timestamp_hi_tag_m()) << 32);
80}
81
82static u32 gk20a_fecs_trace_fecs_context_ptr(struct gk20a *g, struct channel_gk20a *ch)
83{
84 return (u32) (nvgpu_inst_block_addr(g, &ch->inst_block) >> 12LL);
85}
86
87int gk20a_fecs_trace_num_ts(void)
88{
89 return (ctxsw_prog_record_timestamp_record_size_in_bytes_v()
90 - sizeof(struct gk20a_fecs_trace_record)) / sizeof(u64);
91}
92
93struct gk20a_fecs_trace_record *gk20a_fecs_trace_get_record(
94 struct gk20a *g, int idx)
95{
96 struct nvgpu_mem *mem = &g->gr.global_ctx_buffer[FECS_TRACE_BUFFER].mem;
97
98 return (struct gk20a_fecs_trace_record *)
99 ((u8 *) mem->cpu_va
100 + (idx * ctxsw_prog_record_timestamp_record_size_in_bytes_v()));
101}
102
103bool gk20a_fecs_trace_is_valid_record(struct gk20a_fecs_trace_record *r)
104{
105 /*
106 * testing magic_hi should suffice. magic_lo is sometimes used
107 * as a sequence number in experimental ucode.
108 */
109 return (r->magic_hi
110 == ctxsw_prog_record_timestamp_magic_value_hi_v_value_v());
111}
112
113int gk20a_fecs_trace_get_read_index(struct gk20a *g)
114{
115 return gr_gk20a_elpg_protected_call(g,
116 gk20a_readl(g, gr_fecs_mailbox1_r()));
117}
118
119int gk20a_fecs_trace_get_write_index(struct gk20a *g)
120{
121 return gr_gk20a_elpg_protected_call(g,
122 gk20a_readl(g, gr_fecs_mailbox0_r()));
123}
124
125static int gk20a_fecs_trace_set_read_index(struct gk20a *g, int index)
126{
127 nvgpu_log(g, gpu_dbg_ctxsw, "set read=%d", index);
128 return gr_gk20a_elpg_protected_call(g,
129 (gk20a_writel(g, gr_fecs_mailbox1_r(), index), 0));
130}
131
132void gk20a_fecs_trace_hash_dump(struct gk20a *g)
133{
134 u32 bkt;
135 struct gk20a_fecs_trace_hash_ent *ent;
136 struct gk20a_fecs_trace *trace = g->fecs_trace;
137
138 nvgpu_log(g, gpu_dbg_ctxsw, "dumping hash table");
139
140 nvgpu_mutex_acquire(&trace->hash_lock);
141 hash_for_each(trace->pid_hash_table, bkt, ent, node)
142 {
143 nvgpu_log(g, gpu_dbg_ctxsw, " ent=%p bkt=%x context_ptr=%x pid=%d",
144 ent, bkt, ent->context_ptr, ent->pid);
145
146 }
147 nvgpu_mutex_release(&trace->hash_lock);
148}
149
150static int gk20a_fecs_trace_hash_add(struct gk20a *g, u32 context_ptr, pid_t pid)
151{
152 struct gk20a_fecs_trace_hash_ent *he;
153 struct gk20a_fecs_trace *trace = g->fecs_trace;
154
155 nvgpu_log(g, gpu_dbg_fn | gpu_dbg_ctxsw,
156 "adding hash entry context_ptr=%x -> pid=%d", context_ptr, pid);
157
158 he = nvgpu_kzalloc(g, sizeof(*he));
159 if (unlikely(!he)) {
160 nvgpu_warn(g,
161 "can't alloc new hash entry for context_ptr=%x pid=%d",
162 context_ptr, pid);
163 return -ENOMEM;
164 }
165
166 he->context_ptr = context_ptr;
167 he->pid = pid;
168 nvgpu_mutex_acquire(&trace->hash_lock);
169 hash_add(trace->pid_hash_table, &he->node, context_ptr);
170 nvgpu_mutex_release(&trace->hash_lock);
171 return 0;
172}
173
174static void gk20a_fecs_trace_hash_del(struct gk20a *g, u32 context_ptr)
175{
176 struct hlist_node *tmp;
177 struct gk20a_fecs_trace_hash_ent *ent;
178 struct gk20a_fecs_trace *trace = g->fecs_trace;
179
180 nvgpu_log(g, gpu_dbg_fn | gpu_dbg_ctxsw,
181 "freeing hash entry context_ptr=%x", context_ptr);
182
183 nvgpu_mutex_acquire(&trace->hash_lock);
184 hash_for_each_possible_safe(trace->pid_hash_table, ent, tmp, node,
185 context_ptr) {
186 if (ent->context_ptr == context_ptr) {
187 hash_del(&ent->node);
188 nvgpu_log(g, gpu_dbg_ctxsw,
189 "freed hash entry=%p context_ptr=%x", ent,
190 ent->context_ptr);
191 nvgpu_kfree(g, ent);
192 break;
193 }
194 }
195 nvgpu_mutex_release(&trace->hash_lock);
196}
197
198static void gk20a_fecs_trace_free_hash_table(struct gk20a *g)
199{
200 u32 bkt;
201 struct hlist_node *tmp;
202 struct gk20a_fecs_trace_hash_ent *ent;
203 struct gk20a_fecs_trace *trace = g->fecs_trace;
204
205 nvgpu_log(g, gpu_dbg_fn | gpu_dbg_ctxsw, "trace=%p", trace);
206
207 nvgpu_mutex_acquire(&trace->hash_lock);
208 hash_for_each_safe(trace->pid_hash_table, bkt, tmp, ent, node) {
209 hash_del(&ent->node);
210 nvgpu_kfree(g, ent);
211 }
212 nvgpu_mutex_release(&trace->hash_lock);
213
214}
215
216static pid_t gk20a_fecs_trace_find_pid(struct gk20a *g, u32 context_ptr)
217{
218 struct gk20a_fecs_trace_hash_ent *ent;