diff options
Diffstat (limited to 'nvdebug.h')
| -rw-r--r-- | nvdebug.h | 719 |
1 files changed, 673 insertions, 46 deletions
| @@ -5,14 +5,18 @@ | |||
| 5 | // TODO(jbakita): Don't depend on these. | 5 | // TODO(jbakita): Don't depend on these. |
| 6 | #include <nvgpu/gk20a.h> // For struct gk20a | 6 | #include <nvgpu/gk20a.h> // For struct gk20a |
| 7 | #include <os/linux/os_linux.h> // For struct nvgpu_os_linux | 7 | #include <os/linux/os_linux.h> // For struct nvgpu_os_linux |
| 8 | #include <linux/proc_fs.h> // For PDE_DATA() macro | ||
| 8 | 9 | ||
| 9 | /* Runlist Channel | 10 | /* Runlist Channel |
| 10 | A timeslice group (TSG) is composed of channels. Each channel is a FIFO queue | 11 | A timeslice group (TSG) is composed of channels. Each channel is a FIFO queue |
| 11 | of GPU commands. These commands are typically queued from userspace. | 12 | of GPU commands. These commands are typically queued from userspace. |
| 12 | 13 | ||
| 13 | `INST_PTR` points to a GPU Instance Block which contains pointers to the GPU | 14 | Prior to Volta, channels could also exist independent of a TSG. These are |
| 14 | virtual address space for this context. All channels in a TSG point to the | 15 | called "bare channels" in the Jetson nvgpu driver. |
| 15 | same GPU Instance Block (?). | 16 | |
| 17 | `INST_PTR` points to a GPU Instance Block which contains FIFO states, virtual | ||
| 18 | address space configuration for this context, and a pointer to the page | ||
| 19 | tables. All channels in a TSG point to the same GPU Instance Block (?). | ||
| 16 | 20 | ||
| 17 | "RUNQUEUE_SELECTOR determines to which runqueue the channel belongs, and | 21 | "RUNQUEUE_SELECTOR determines to which runqueue the channel belongs, and |
| 18 | thereby which PBDMA will run the channel. Increasing values select | 22 | thereby which PBDMA will run the channel. Increasing values select |
| @@ -30,7 +34,13 @@ | |||
| 30 | ENTRY_TYPE (T) : type of this entry: ENTRY_TYPE_CHAN | 34 | ENTRY_TYPE (T) : type of this entry: ENTRY_TYPE_CHAN |
| 31 | CHID (ID) : identifier of the channel to run (overlays ENTRY_ID) | 35 | CHID (ID) : identifier of the channel to run (overlays ENTRY_ID) |
| 32 | RUNQUEUE_SELECTOR (Q) : selects which PBDMA should run this channel if | 36 | RUNQUEUE_SELECTOR (Q) : selects which PBDMA should run this channel if |
| 33 | more than one PBDMA is supported by the runlist | 37 | more than one PBDMA is supported by the runlist, |
| 38 | additionally, "A value of 0 targets the first FE | ||
| 39 | pipe, which can process all FE driven engines: | ||
| 40 | Graphics, Compute, Inline2Memory, and TwoD. A value | ||
| 41 | of 1 targets the second FE pipe, which can only | ||
| 42 | process Compute work. Note that GRCE work is allowed | ||
| 43 | on either runqueue.)" | ||
| 34 | 44 | ||
| 35 | INST_PTR_LO : lower 20 bits of the 4k-aligned instance block pointer | 45 | INST_PTR_LO : lower 20 bits of the 4k-aligned instance block pointer |
| 36 | INST_PTR_HI : upper 32 bit of instance block pointer | 46 | INST_PTR_HI : upper 32 bit of instance block pointer |
| @@ -39,6 +49,9 @@ | |||
| 39 | USERD_PTR_LO : upper 24 bits of the low 32 bits, of the 512-byte-aligned USERD pointer | 49 | USERD_PTR_LO : upper 24 bits of the low 32 bits, of the 512-byte-aligned USERD pointer |
| 40 | USERD_PTR_HI : upper 32 bits of USERD pointer | 50 | USERD_PTR_HI : upper 32 bits of USERD pointer |
| 41 | USERD_TARGET (TGU) : aperture of the USERD data structure | 51 | USERD_TARGET (TGU) : aperture of the USERD data structure |
| 52 | |||
| 53 | Channels were around since at least Fermi, but were rearranged with Volta to | ||
| 54 | add a USERD pointer, a longer INST pointer, and a runqueue selector flag. | ||
| 42 | */ | 55 | */ |
| 43 | enum ENTRY_TYPE {ENTRY_TYPE_CHAN = 0, ENTRY_TYPE_TSG = 1}; | 56 | enum ENTRY_TYPE {ENTRY_TYPE_CHAN = 0, ENTRY_TYPE_TSG = 1}; |
| 44 | enum INST_TARGET {TARGET_VID_MEM = 0, TARGET_SYS_MEM_COHERENT = 2, TARGET_SYS_MEM_NONCOHERENT = 3}; | 57 | enum INST_TARGET {TARGET_VID_MEM = 0, TARGET_SYS_MEM_COHERENT = 2, TARGET_SYS_MEM_NONCOHERENT = 3}; |
| @@ -52,11 +65,12 @@ static inline char* target_to_text(enum INST_TARGET t) { | |||
| 52 | return "SYS_MEM_NONCOHERENT"; | 65 | return "SYS_MEM_NONCOHERENT"; |
| 53 | default: | 66 | default: |
| 54 | printk(KERN_WARNING "[nvdebug] Invalid aperture!\n"); | 67 | printk(KERN_WARNING "[nvdebug] Invalid aperture!\n"); |
| 55 | return NULL; | 68 | return "INVALID"; |
| 56 | } | 69 | } |
| 57 | } | 70 | } |
| 58 | 71 | ||
| 59 | struct runlist_chan { | 72 | // Support: Volta, Ampere, Turing |
| 73 | struct gv100_runlist_chan { | ||
| 60 | // 0:63 | 74 | // 0:63 |
| 61 | enum ENTRY_TYPE entry_type:1; | 75 | enum ENTRY_TYPE entry_type:1; |
| 62 | uint32_t runqueue_selector:1; | 76 | uint32_t runqueue_selector:1; |
| @@ -71,6 +85,20 @@ struct runlist_chan { | |||
| 71 | uint32_t inst_ptr_hi:32; | 85 | uint32_t inst_ptr_hi:32; |
| 72 | } __attribute__((packed)); | 86 | } __attribute__((packed)); |
| 73 | 87 | ||
| 88 | // Support: Fermi, Kepler*, Maxwell, Pascal | ||
| 89 | // *In Kepler, inst fields may be unpopulated? | ||
| 90 | struct gm107_runlist_chan { | ||
| 91 | uint32_t chid:12; | ||
| 92 | uint32_t padding0:1; | ||
| 93 | enum ENTRY_TYPE entry_type:1; | ||
| 94 | uint32_t padding1:18; | ||
| 95 | uint32_t inst_ptr_lo:20; | ||
| 96 | enum INST_TARGET inst_target:2; // Totally guessing on this | ||
| 97 | uint32_t padding2:10; | ||
| 98 | } __attribute__((packed)); | ||
| 99 | |||
| 100 | #define gk110_runlist_chan gm107_runlist_chan | ||
| 101 | |||
| 74 | /* Runlist TSG (TimeSlice Group) | 102 | /* Runlist TSG (TimeSlice Group) |
| 75 | The runlist is composed of timeslice groups (TSG). Each TSG corresponds | 103 | The runlist is composed of timeslice groups (TSG). Each TSG corresponds |
| 76 | to a single virtual address space on the GPU and contains `TSG_LENGTH` | 104 | to a single virtual address space on the GPU and contains `TSG_LENGTH` |
| @@ -85,8 +113,15 @@ struct runlist_chan { | |||
| 85 | TIMESLICE_TIMEOUT : timeout amount for the TSG's timeslice | 113 | TIMESLICE_TIMEOUT : timeout amount for the TSG's timeslice |
| 86 | TSG_LENGTH : number of channels that are part of this timeslice group | 114 | TSG_LENGTH : number of channels that are part of this timeslice group |
| 87 | TSGID : identifier of the Timeslice group (overlays ENTRY_ID) | 115 | TSGID : identifier of the Timeslice group (overlays ENTRY_ID) |
| 116 | |||
| 117 | TSGs appear to have been introduced with Kepler and stayed the same until | ||
| 118 | they were rearranged at the time of channel rearrangement to support longer | ||
| 119 | GPU instance addresses with Volta. | ||
| 88 | */ | 120 | */ |
| 89 | struct entry_tsg { | 121 | |
| 122 | // Support: Volta, Ampere*, Turing* | ||
| 123 | // *These treat the top 8 bits of TSGID as GFID (unused) | ||
| 124 | struct gv100_runlist_tsg { | ||
| 90 | // 0:63 | 125 | // 0:63 |
| 91 | enum ENTRY_TYPE entry_type:1; | 126 | enum ENTRY_TYPE entry_type:1; |
| 92 | uint64_t padding:15; | 127 | uint64_t padding:15; |
| @@ -101,14 +136,28 @@ struct entry_tsg { | |||
| 101 | } __attribute__((packed)); | 136 | } __attribute__((packed)); |
| 102 | #define MAX_TSGID (1 << 12) | 137 | #define MAX_TSGID (1 << 12) |
| 103 | 138 | ||
| 139 | // Support: Kepler (v2?), Maxwell, Pascal | ||
| 140 | // Same fields as Volta except tsg_length is 6 bits rather than 8 | ||
| 141 | // Last 32 bits appear to contain an undocumented inst ptr | ||
| 142 | struct gk110_runlist_tsg { | ||
| 143 | uint32_t tsgid:12; | ||
| 144 | uint32_t padding0:1; | ||
| 145 | enum ENTRY_TYPE entry_type:1; | ||
| 146 | uint32_t timeslice_scale:4; | ||
| 147 | uint32_t timeslice_timeout:8; | ||
| 148 | uint32_t tsg_length:6; | ||
| 149 | uint32_t padding1:32; | ||
| 150 | } __attribute__((packed)); | ||
| 151 | |||
| 152 | |||
| 104 | enum PREEMPT_TYPE {PREEMPT_TYPE_CHANNEL = 0, PREEMPT_TYPE_TSG = 1}; | 153 | enum PREEMPT_TYPE {PREEMPT_TYPE_CHANNEL = 0, PREEMPT_TYPE_TSG = 1}; |
| 105 | 154 | ||
| 106 | /* Preempt a TSG or Channel by ID | 155 | /* Preempt a TSG or Channel by ID |
| 107 | ID/CHID : Id of TSG or channel to preempt | 156 | ID/CHID : Id of TSG or channel to preempt |
| 108 | IS_PENDING : ???? | 157 | IS_PENDING : Is a context switch pending? |
| 109 | TYPE : PREEMPT_TYPE_CHANNEL or PREEMPT_TYPE_TSG | 158 | TYPE : PREEMPT_TYPE_CHANNEL or PREEMPT_TYPE_TSG |
| 110 | 159 | ||
| 111 | Support: Kepler, Maxwell, Pascal, Volta | 160 | Support: Kepler, Maxwell, Pascal, Volta, Turing |
| 112 | */ | 161 | */ |
| 113 | #define NV_PFIFO_PREEMPT 0x00002634 | 162 | #define NV_PFIFO_PREEMPT 0x00002634 |
| 114 | typedef union { | 163 | typedef union { |
| @@ -195,26 +244,36 @@ typedef union { | |||
| 195 | */ | 244 | */ |
| 196 | 245 | ||
| 197 | // Note: This is different with Turing | 246 | // Note: This is different with Turing |
| 198 | // Support: Kepler, Maxwell, Pascal, Volta | 247 | // Support: Fermi, Kepler, Maxwell, Pascal, Volta |
| 199 | #define NV_PFIFO_RUNLIST_BASE 0x00002270 | 248 | #define NV_PFIFO_RUNLIST_BASE 0x00002270 |
| 249 | #define NV_PFIFO_ENG_RUNLIST_BASE(i) (0x00002280+(i)*8) | ||
| 200 | typedef union { | 250 | typedef union { |
| 201 | struct { | 251 | struct { |
| 202 | uint32_t ptr:28; | 252 | uint32_t ptr:28; |
| 203 | uint32_t type:2; | 253 | enum INST_TARGET target:2; |
| 204 | uint32_t padding:2; | 254 | uint32_t padding:2; |
| 205 | } __attribute__((packed)); | 255 | } __attribute__((packed)); |
| 206 | uint32_t raw; | 256 | uint32_t raw; |
| 207 | } runlist_base_t; | 257 | } runlist_base_t; |
| 208 | 258 | ||
| 209 | // Support: Kepler, Maxwell, Pascal, Volta | 259 | // Support: Kepler, Maxwell, Pascal, Volta |
| 260 | // Works on Fermi, but id is one bit longer and is b11111 | ||
| 210 | #define NV_PFIFO_RUNLIST 0x00002274 | 261 | #define NV_PFIFO_RUNLIST 0x00002274 |
| 262 | #define NV_PFIFO_ENG_RUNLIST(i) (0x00002284+(i)*8) | ||
| 211 | typedef union { | 263 | typedef union { |
| 264 | // RUNLIST fields | ||
| 212 | struct { | 265 | struct { |
| 213 | uint32_t len:16; | 266 | uint32_t len:16; |
| 214 | uint32_t padding:4; | 267 | uint32_t padding:4; |
| 215 | uint32_t id:4; | 268 | uint32_t id:4; // Runlist ID (each engine may have a seperate runlist) |
| 216 | uint32_t padding2:8; | 269 | uint32_t padding2:8; |
| 217 | } __attribute__((packed)); | 270 | } __attribute__((packed)); |
| 271 | // ENG_RUNLIST fields that differ | ||
| 272 | struct { | ||
| 273 | uint32_t padding3:20; | ||
| 274 | bool is_pending:1; // Is runlist not yet committed? | ||
| 275 | uint32_t padding4:11; | ||
| 276 | } __attribute__((packed)); | ||
| 218 | uint32_t raw; | 277 | uint32_t raw; |
| 219 | } runlist_info_t; | 278 | } runlist_info_t; |
| 220 | 279 | ||
| @@ -301,63 +360,631 @@ typedef union { | |||
| 301 | uint32_t raw; | 360 | uint32_t raw; |
| 302 | } runlist_disable_t; | 361 | } runlist_disable_t; |
| 303 | 362 | ||
| 363 | /* Read GPU descriptors from the Master Controller (MC) | ||
| 364 | |||
| 365 | MINOR_REVISION : Legacy (only used with Celvin in Nouveau) | ||
| 366 | MAJOR_REVISION : Legacy (only used with Celvin in Nouveau) | ||
| 367 | IMPLEMENTATION : Which implementation of the GPU architecture | ||
| 368 | ARCHITECTURE : Which GPU architecture | ||
| 369 | |||
| 370 | CHIP_ID = IMPLEMENTATION + ARCHITECTURE << 4 | ||
| 371 | CHIP_ID : Unique ID of all chips since Kelvin | ||
| 372 | |||
| 373 | Support: Kelvin, Rankline, Curie, Tesla, Fermi, Kepler, Maxwell, Pascal, | ||
| 374 | Volta, Turing, Ampere | ||
| 375 | */ | ||
| 376 | #define NV_MC_BOOT_0 0x00000000 | ||
| 377 | #define NV_CHIP_ID_GP106 0x136 // Discrete GeForce GTX 1060 | ||
| 378 | #define NV_CHIP_ID_GV11B 0x15B // Jetson Xavier embedded GPU | ||
| 379 | #define NV_CHIP_ID_KEPLER 0x0E0 | ||
| 380 | #define NV_CHIP_ID_VOLTA 0x140 | ||
| 381 | |||
| 382 | inline static const char* ARCH2NAME(uint32_t arch) { | ||
| 383 | switch (arch) { | ||
| 384 | case 0x01: | ||
