/* Copyright 2024 Joshua Bakita
* SPDX-License-Identifier: MIT
*
* File outline:
* - Runlist, preemption, and channel control (FIFO)
* - Basic GPU information (MC)
* - Detailed GPU information (PTOP, FUSE, and CE)
* - PRAMIN, BAR1/2, and page table status
* - Helper functions for nvdebug
*
* This function should not depend on any Linux-internal headers, and may be
* included outside of nvdebug.
*/
#include <linux/types.h>
// Fully defined in include/nvgpu/gk20a.h. We only pass around pointers to
// this, so declare as incomplete type to avoid pulling in the nvgpu headers.
struct gk20a;
/* Runlist Channel
A timeslice group (TSG) is composed of channels. Each channel is a FIFO queue
of GPU commands. These commands are typically queued from userspace.
Prior to Volta, channels could also exist independent of a TSG. These are
called "bare channels" in the Jetson nvgpu driver.
`INST_PTR` points to a GPU Instance Block which contains FIFO states, virtual
address space configuration for this context, and a pointer to the page
tables. All channels in a TSG point to the same GPU Instance Block (?).
"RUNQUEUE_SELECTOR determines to which runqueue the channel belongs, and
thereby which PBDMA will run the channel. Increasing values select
increasingly numbered PBDMA IDs serving the runlist. If the selector value
exceeds the number of PBDMAs on the runlist, the hardware will silently
reassign the channel to run on the first PBDMA as though RUNQUEUE_SELECTOR had
been set to 0. (In current hardware, this is used by SCG on the graphics
runlist only to determine which FE pipe should service a given channel. A
value of 0 targets the first FE pipe, which can process all FE driven engines:
Graphics, Compute, Inline2Memory, and TwoD. A value of 1 targets the second
FE pipe, which can only process Compute work. Note that GRCE work is allowed
on either runqueue." (NVIDIA) Note that it appears runqueue 1 is the default
for CUDA work on the Jetson Xavier.
ENTRY_TYPE (T) : type of this entry: ENTRY_TYPE_CHAN
CHID (ID) : identifier of the channel to run (overlays ENTRY_ID)
RUNQUEUE_SELECTOR (Q) : selects which PBDMA should run this channel if
more than one PBDMA is supported by the runlist,
additionally, "A value of 0 targets the first FE
pipe, which can process all FE driven engines:
Graphics, Compute, Inline2Memory, and TwoD. A value
of 1 targets the second FE pipe, which can only
process Compute work. Note that GRCE work is allowed
on either runqueue.)"
INST_PTR_LO : lower 20 bits of the 4k-aligned instance block pointer
INST_PTR_HI : upper 32 bit of instance block pointer
INST_TARGET (TGI) : aperture of the instance block
USERD_PTR_LO : upper 24 bits of the low 32 bits, of the 512-byte-aligned USERD pointer
USERD_PTR_HI : upper 32 bits of USERD pointer
USERD_TARGET (TGU) : aperture of the USERD data structure
Channels were around since at least Fermi, but were rearranged with Volta to
add a USERD pointer, a longer INST pointer, and a runqueue selector flag.
*/
enum ENTRY_TYPE {ENTRY_TYPE_CHAN = 0, ENTRY_TYPE_TSG = 1};
enum INST_TARGET {TARGET_VID_MEM = 0, TARGET_INVALID = 1, TARGET_SYS_MEM_COHERENT = 2, TARGET_SYS_MEM_NONCOHERENT = 3};
static inline const char *target_to_text(enum INST_TARGET t) {
switch (t) {
case TARGET_VID_MEM:
return "VID_MEM";
case TARGET_SYS_MEM_COHERENT:
return "SYS_MEM_COHERENT";
case TARGET_SYS_MEM_NONCOHERENT:
return "SYS_MEM_NONCOHERENT";
default:
return "INVALID";
}
}
// Support: Volta, Ampere, Turing, Ampere
struct gv100_runlist_chan {
// 0:63
enum ENTRY_TYPE entry_type:1;
uint32_t runqueue_selector:1;
uint32_t padding:2;
enum INST_TARGET inst_target:2;
uint32_t padding2:2;
uint32_t userd_ptr_lo:24;
uint32_t userd_ptr_hi:32;
// 64:128
uint32_t chid:12;
uint32_t inst_ptr_lo:20;
uint32_t inst_ptr_hi:32;
} __attribute__((packed));
// Support: Fermi, Kepler*, Maxwell, Pascal
// *In Kepler, inst fields may be unpopulated?
struct gm107_runlist_chan {
uint32_t chid:12;
uint32_t padding0:1;
enum ENTRY_TYPE entry_type:1;
uint32_t padding1:18;
uint32_t inst_ptr_lo:20;
enum INST_TARGET inst_target:2; // Totally guessing on this
uint32_t padding2:10;
} __attribute__((packed));
#define gk110_runlist_chan gm107_runlist_chan
/* Runlist TSG (TimeSlice Group)
The runlist is composed of timeslice groups (TSG). Each TSG corresponds
to a single virtual address space on the GPU and contains `TSG_LENGTH`
channels. These channels and virtual address space are accessible to the GPU
host unit for use until the timeslice expires or a TSG switch is forcibly
initiated via a write to `NV_PFIFO_PREEMPT`.
timeslice = (TSG_TIMESLICE_TIMEOUT << TSG_TIMESLICE_SCALE) * 1024 nanoseconds
ENTRY_TYPE (T) : type of this entry: ENTRY_TYPE_TSG
TIMESLICE_SCALE : scale factor for the TSG's timeslice
TIMESLICE_TIMEOUT : timeout amount for the TSG's timeslice
TSG_LENGTH : number of channels that are part of this timeslice group
TSGID : identifier of the Timeslice group (overlays ENTRY_ID)
TSGs appear to have been introduced with Kepler and stayed the same until
they were rearranged at the time of channel rearrangement to support longer
GPU instance addresses with Volta.
According to nvgpu, "timeslice is measured with PTIMER [which may be] lower
than 1GHz."
*/
// Support: Volta, Turing*, Ampere*
// *These treat bits 4:11 (8 bits) as GFID (unused)
struct gv100_runlist_tsg {
// 0:63
enum ENTRY_TYPE entry_type:1;
uint64_t padding:15;
uint32_t timeslice_scale:4;
uint64_t padding2:4;
uint32_t timeslice_timeout:8;
uint32_t tsg_length:8;
uint32_t padding3:24;
// 64:128
uint32_t tsgid:12;
uint64_t padding4:52
|