diff options
Diffstat (limited to 'include/gk20a/regops_gk20a.c')
| -rw-r--r-- | include/gk20a/regops_gk20a.c | 472 |
1 files changed, 472 insertions, 0 deletions
diff --git a/include/gk20a/regops_gk20a.c b/include/gk20a/regops_gk20a.c new file mode 100644 index 0000000..0aec4f8 --- /dev/null +++ b/include/gk20a/regops_gk20a.c | |||
| @@ -0,0 +1,472 @@ | |||
| 1 | /* | ||
| 2 | * Tegra GK20A GPU Debugger Driver Register Ops | ||
| 3 | * | ||
| 4 | * Copyright (c) 2013-2018, NVIDIA CORPORATION. All rights reserved. | ||
| 5 | * | ||
| 6 | * Permission is hereby granted, free of charge, to any person obtaining a | ||
| 7 | * copy of this software and associated documentation files (the "Software"), | ||
| 8 | * to deal in the Software without restriction, including without limitation | ||
| 9 | * the rights to use, copy, modify, merge, publish, distribute, sublicense, | ||
| 10 | * and/or sell copies of the Software, and to permit persons to whom the | ||
| 11 | * Software is furnished to do so, subject to the following conditions: | ||
| 12 | * | ||
| 13 | * The above copyright notice and this permission notice shall be included in | ||
| 14 | * all copies or substantial portions of the Software. | ||
| 15 | * | ||
| 16 | * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR | ||
| 17 | * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, | ||
| 18 | * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL | ||
| 19 | * THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER | ||
| 20 | * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING | ||
| 21 | * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER | ||
| 22 | * DEALINGS IN THE SOFTWARE. | ||
| 23 | */ | ||
| 24 | |||
| 25 | #include "gk20a.h" | ||
| 26 | #include "gr_gk20a.h" | ||
| 27 | #include "dbg_gpu_gk20a.h" | ||
| 28 | #include "regops_gk20a.h" | ||
| 29 | |||
| 30 | #include <nvgpu/log.h> | ||
| 31 | #include <nvgpu/bsearch.h> | ||
| 32 | #include <nvgpu/bug.h> | ||
| 33 | #include <nvgpu/io.h> | ||
| 34 | |||
| 35 | static int regop_bsearch_range_cmp(const void *pkey, const void *pelem) | ||
| 36 | { | ||
| 37 | u32 key = *(u32 *)pkey; | ||
| 38 | struct regop_offset_range *prange = (struct regop_offset_range *)pelem; | ||
| 39 | if (key < prange->base) { | ||
| 40 | return -1; | ||
| 41 | } else if (prange->base <= key && key < (prange->base + | ||
| 42 | (prange->count * 4U))) { | ||
| 43 | return 0; | ||
| 44 | } | ||
| 45 | return 1; | ||
| 46 | } | ||
| 47 | |||
| 48 | static inline bool linear_search(u32 offset, const u32 *list, int size) | ||
| 49 | { | ||
| 50 | int i; | ||
| 51 | for (i = 0; i < size; i++) { | ||
| 52 | if (list[i] == offset) { | ||
| 53 | return true; | ||
| 54 | } | ||
| 55 | } | ||
| 56 | return false; | ||
| 57 | } | ||
| 58 | |||
| 59 | /* | ||
| 60 | * In order to perform a context relative op the context has | ||
| 61 | * to be created already... which would imply that the | ||
| 62 | * context switch mechanism has already been put in place. | ||
| 63 | * So by the time we perform such an opertation it should always | ||
| 64 | * be possible to query for the appropriate context offsets, etc. | ||
| 65 | * | ||
| 66 | * But note: while the dbg_gpu bind requires the a channel fd, | ||
| 67 | * it doesn't require an allocated gr/compute obj at that point... | ||
| 68 | */ | ||
| 69 | static bool gr_context_info_available(struct gr_gk20a *gr) | ||
| 70 | { | ||
| 71 | int err; | ||
| 72 | |||
| 73 | nvgpu_mutex_acquire(&gr->ctx_mutex); | ||
| 74 | err = !gr->ctx_vars.golden_image_initialized; | ||
| 75 | nvgpu_mutex_release(&gr->ctx_mutex); | ||
| 76 | if (err) { | ||
| 77 | return false; | ||
| 78 | } | ||
| 79 | |||
| 80 | return true; | ||
| 81 | |||
| 82 | } | ||
| 83 | |||
| 84 | static bool validate_reg_ops(struct dbg_session_gk20a *dbg_s, | ||
| 85 | u32 *ctx_rd_count, u32 *ctx_wr_count, | ||
| 86 | struct nvgpu_dbg_reg_op *ops, | ||
| 87 | u32 op_count); | ||
| 88 | |||
| 89 | |||
| 90 | int exec_regops_gk20a(struct dbg_session_gk20a *dbg_s, | ||
| 91 | struct nvgpu_dbg_reg_op *ops, | ||
| 92 | u64 num_ops, | ||
| 93 | bool *is_current_ctx) | ||
| 94 | { | ||
| 95 | int err = 0; | ||
| 96 | unsigned int i; | ||
| 97 | struct channel_gk20a *ch = NULL; | ||
| 98 | struct gk20a *g = dbg_s->g; | ||
| 99 | /*struct gr_gk20a *gr = &g->gr;*/ | ||
| 100 | u32 data32_lo = 0, data32_hi = 0; | ||
| 101 | u32 ctx_rd_count = 0, ctx_wr_count = 0; | ||
| 102 | bool skip_read_lo, skip_read_hi; | ||
| 103 | bool ok; | ||
| 104 | |||
| 105 | nvgpu_log(g, gpu_dbg_fn | gpu_dbg_gpu_dbg, " "); | ||
| 106 | |||
| 107 | ch = nvgpu_dbg_gpu_get_session_channel(dbg_s); | ||
| 108 | |||
| 109 | /* For vgpu, the regops routines need to be handled in the | ||
| 110 | * context of the server and support for that does not exist. | ||
| 111 | * | ||
| 112 | * The two users of the regops interface are the compute driver | ||
| 113 | * and tools. The compute driver will work without a functional | ||
| 114 | * regops implementation, so we return -ENOSYS. This will allow | ||
| 115 | * compute apps to run with vgpu. Tools will not work in this | ||
| 116 | * configuration and are not required to work at this time. */ | ||
| 117 | if (g->is_virtual) { | ||
| 118 | return -ENOSYS; | ||
| 119 | } | ||
| 120 | |||
| 121 | ok = validate_reg_ops(dbg_s, | ||
| 122 | &ctx_rd_count, &ctx_wr_count, | ||
| 123 | ops, num_ops); | ||
| 124 | if (!ok) { | ||
| 125 | nvgpu_err(g, "invalid op(s)"); | ||
| 126 | err = -EINVAL; | ||
| 127 | /* each op has its own err/status */ | ||
| 128 | goto clean_up; | ||
| 129 | } | ||
| 130 | |||
| 131 | /* be sure that ctx info is in place if there are ctx ops */ | ||
| 132 | if (ctx_wr_count | ctx_rd_count) { | ||
| 133 | if (!gr_context_info_available(&g->gr)) { | ||
| 134 | nvgpu_err(g, "gr context data not available"); | ||
| 135 | return -ENODEV; | ||
| 136 | } | ||
| 137 | } | ||
| 138 | |||
| 139 | for (i = 0; i < num_ops; i++) { | ||
| 140 | /* if it isn't global then it is done in the ctx ops... */ | ||
| 141 | if (ops[i].type != REGOP(TYPE_GLOBAL)) { | ||
| 142 | continue; | ||
| 143 | } | ||
| 144 | |||
| 145 | switch (ops[i].op) { | ||
| 146 | |||
| 147 | case REGOP(READ_32): | ||
| 148 | ops[i].value_hi = 0; | ||
| 149 | ops[i].value_lo = gk20a_readl(g, ops[i].offset); | ||
| 150 | nvgpu_log(g, gpu_dbg_gpu_dbg, "read_32 0x%08x from 0x%08x", | ||
| 151 | ops[i].value_lo, ops[i].offset); | ||
| 152 | |||
| 153 | break; | ||
| 154 | |||
| 155 | case REGOP(READ_64): | ||
| 156 | ops[i].value_lo = gk20a_readl(g, ops[i].offset); | ||
| 157 | ops[i].value_hi = | ||
| 158 | gk20a_readl(g, ops[i].offset + 4); | ||
| 159 | |||
| 160 | nvgpu_log(g, gpu_dbg_gpu_dbg, "read_64 0x%08x:%08x from 0x%08x", | ||
| 161 | ops[i].value_hi, ops[i].value_lo, | ||
| 162 | ops[i].offset); | ||
| 163 | break; | ||
| 164 | |||
| 165 | case REGOP(WRITE_32): | ||
| 166 | case REGOP(WRITE_64): | ||
| 167 | /* some of this appears wonky/unnecessary but | ||
| 168 | we've kept it for compat with existing | ||
| 169 | debugger code. just in case... */ | ||
| 170 | skip_read_lo = skip_read_hi = false; | ||
| 171 | if (ops[i].and_n_mask_lo == ~(u32)0) { | ||
| 172 | data32_lo = ops[i].value_lo; | ||
| 173 | skip_read_lo = true; | ||
| 174 | } | ||
| 175 | |||
| 176 | if ((ops[i].op == REGOP(WRITE_64)) && | ||
| 177 | (ops[i].and_n_mask_hi == ~(u32)0)) { | ||
| 178 | data32_hi = ops[i].value_hi; | ||
| 179 | skip_read_hi = true; | ||
| 180 | } | ||
| 181 | |||
| 182 | /* read first 32bits */ | ||
| 183 | if (skip_read_lo == false) { | ||
| 184 | data32_lo = gk20a_readl(g, ops[i].offset); | ||
| 185 | data32_lo &= ~ops[i].and_n_mask_lo; | ||
| 186 | data32_lo |= ops[i].value_lo; | ||
| 187 | } | ||
| 188 | |||
| 189 | /* if desired, read second 32bits */ | ||
| 190 | if ((ops[i].op == REGOP(WRITE_64)) && | ||
| 191 | !skip_read_hi) { | ||
| 192 | data32_hi = gk20a_readl(g, ops[i].offset + 4); | ||
| 193 | data32_hi &= ~ops[i].and_n_mask_hi; | ||
| 194 | data32_hi |= ops[i].value_hi; | ||
| 195 | } | ||
| 196 | |||
| 197 | /* now update first 32bits */ | ||
| 198 | gk20a_writel(g, ops[i].offset, data32_lo); | ||
| 199 | nvgpu_log(g, gpu_dbg_gpu_dbg, "Wrote 0x%08x to 0x%08x ", | ||
| 200 | data32_lo, ops[i].offset); | ||
| 201 | /* if desired, update second 32bits */ | ||
| 202 | if (ops[i].op == REGOP(WRITE_64)) { | ||
| 203 | gk20a_writel(g, ops[i].offset + 4, data32_hi); | ||
| 204 | nvgpu_log(g, gpu_dbg_gpu_dbg, "Wrote 0x%08x to 0x%08x ", | ||
| 205 | data32_hi, ops[i].offset + 4); | ||
| 206 | |||
| 207 | } | ||
| 208 | |||
| 209 | |||
| 210 | break; | ||
| 211 | |||
| 212 | /* shouldn't happen as we've already screened */ | ||
| 213 | default: | ||
| 214 | BUG(); | ||
| 215 | err = -EINVAL; | ||
| 216 | goto clean_up; | ||
| 217 | break; | ||
| 218 | } | ||
