From 293430fcb5d4013b573556c58457ee706e482b7f Mon Sep 17 00:00:00 2001 From: Joshua Bakita Date: Mon, 5 May 2025 03:53:01 -0400 Subject: Snapshot for ECRTS'25 artifact evaluation --- nvdebug.h | 293 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 288 insertions(+), 5 deletions(-) (limited to 'nvdebug.h') diff --git a/nvdebug.h b/nvdebug.h index ca0f514..3ac8db4 100644 --- a/nvdebug.h +++ b/nvdebug.h @@ -2,6 +2,7 @@ * SPDX-License-Identifier: MIT * * File outline: + * - Configuration options * - Runlist, preemption, and channel control (FIFO) * - Basic GPU information (MC) * - Detailed GPU information (PTOP, FUSE, and CE) @@ -20,6 +21,27 @@ // this, so declare as incomplete type to avoid pulling in the nvgpu headers. struct gk20a; +// Uncomment to, upon BAR2 access failure, return a PRAMIN-based runlist pointer +// in get_runlist_iter(). In order for this pointer to remain valid, PRAMIN +// **must** not be moved during runlist traversal. +// - The Jetson TX2 has no BAR2, and stores the runlist in VID_MEM, so this +// must be enabled to print the runlist on the TX2. +// - On the A100 in Google Cloud and H100 in Paperspace, as of Aug 2024, this is +// needed, as nvdebug is not finding (at least) runlist0 mapped in BAR2/3. +// Automatically disables printing Instance Block and Context State while +// traversing the runlist, as these require conflicting uses of PRAMIN (it's +// needed to search the page tables for the Instance Block in BAR2/3, and to +// access anything in the Context State---aka CTXSW). +#define FALLBACK_TO_PRAMIN + +// Starting offset for registers in the corresponding named range +// Programmable First-In First-Out unit; also known as "Host" +#define NV_PFIFO 0x00002000 // 8 KiB long; ends prior to 0x00004000 +// Programmable Channel Control System RAM +#define NV_PCCSR 0x00800000 // 16 KiB long; ends prior to 0x00810000 +// Programmable TOPology registers +#define NV_PTOP 0x00022400 // 1 KiB long; ends prior to 0x00022800 + /* Runlist Channel A timeslice group (TSG) is composed of channels. Each channel is a FIFO queue of GPU commands. These commands are typically queued from userspace. @@ -202,7 +224,7 @@ typedef union { Support: Ampere, Hopper, Ada, [newer untested] */ #define NV_RUNLIST_PREEMPT_GA100 0x098 -#define PREEMPT_TYPE_RUNLIST 0 +#define PREEMPT_TYPE_RUNLIST PREEMPT_TYPE_CHANNEL /* "Initiate a preempt of the engine by writing the bit associated with its @@ -355,6 +377,14 @@ typedef union { uint64_t raw; } runlist_base_tu102_t; +/* + LEN : Read/Write + OFFSET : Read/Write + PREEMPTED_TSGID : Read-only + VALID_PREEMPTED_TSGID : Read-only + IS_PENDING : Read-only + PREEMPTED_OFFSET : Read-only +*/ typedef union { struct { uint16_t len:16; @@ -416,6 +446,27 @@ typedef union { uint32_t raw; } runlist_channel_config_t; +/* Context Switch Timeout Configuration + After a task's budget expires, there's a configurable grace period, a + "timeout", within which the context needs to complete. After this timeout + expires, an interrupt is raised to terminate the task. + + This register configures if such a timeout is enabled and how long the + timeout is (the "period"). + + Support: Volta, Turing +*/ +#define NV_PFIFO_ENG_CTXSW_TIMEOUT 0x00002A0C +// Support: Ampere +#define NV_RUNLIST_ENGINE_CTXSW_TIMEOUT_CONFIG(i) (0x220+(i)*64) +typedef union { + struct { + uint32_t period:31; + bool enabled:1; + } __attribute__((packed)); + uint32_t raw; +} ctxsw_timeout_t; + /* Programmable Channel Control System RAM (PCCSR) 512-entry array of channel control and status data structures. @@ -477,8 +528,15 @@ typedef union { bool busy:1; uint32_t :3; } __attribute__((packed)); + struct { + uint32_t word1; + uint32_t word2; + } __attribute__((packed)); uint64_t raw; -} channel_ctrl_t; +} channel_ctrl_gf100_t; + +// TODO: Remove use of deprecated type name +typedef channel_ctrl_gf100_t channel_ctrl_t; /* CHannel RAM (CHRAM) (PCCSR replacement on Ampere+) Starting with Ampere, channel IDs are no longer unique indexes into the @@ -543,6 +601,8 @@ typedef union { Support: Fermi, Kepler, Maxwell, Pascal, Volta, Turing */ #define NV_PFIFO_SCHED_DISABLE 0x00002630 +// Support: Ampere +#define NV_RUNLIST_SCHED_DISABLE 0x094 typedef union { struct { bool runlist_0:1; @@ -1018,7 +1078,7 @@ typedef union { struct { uint32_t ptr:28; enum INST_TARGET target:2; - uint32_t :1; + uint32_t :1; // disable_cya_debug for BAR2 bool is_virtual:1; } __attribute__((packed)); uint32_t raw; @@ -1091,6 +1151,9 @@ typedef union { Support: Tesla 2.0* through Ampere, Ada *FAULT_REPLAY_* fields are Pascal+ only See also: dev_ram.h (open-gpu-kernel-modules) or dev_ram.ref.txt (open-gpu-doc) + + It appears that on Hopper, IS_VER2 continues to mean IS_VER2, but if unset, the + alternative is VER3. */ #define NV_PRAMIN_PDB_CONFIG_OFF 0x200 typedef union { @@ -1101,7 +1164,7 @@ typedef union { bool fault_replay_tex:1; bool fault_replay_gcc:1; uint32_t :4; - bool is_ver2:1; + bool is_ver2:1; // XXX: Not on Hopper. May be set or not for same page_dir. bool is_64k_big_page:1; // 128Kb otherwise uint32_t page_dir_lo:20; uint32_t page_dir_hi:32; @@ -1421,6 +1484,182 @@ typedef union { } page_tbl_entry_v0_t; */ +/* Fifo Context RAM (RAMFC) and channel INstance RAM (RAMIN) + + Each channel is configured with a 4 KiB instance block. The prefix of this + block is referred to as RAMFC and stores channel-specific state for the Host + (aka PFIFO). + + "A GPU instance block is a block of memory that contains the state + for a GPU context. A GPU context's instance block consists of Host state, + pointers to each engine's state, and memory management state. A GPU instance + block also contains a pointer to a block of memory that contains that part of a + GPU context's state that a user-level driver may access. A GPU instance block + fits within a single 4K-byte page of memory." + + "The NV_RAMFC part of a GPU-instance block contains Host's part of a virtual + GPU's state. Host is referred to as "FIFO". "FC" stands for FIFO Context. + When Host switches from serving one GPU context to serving a second, Host saves + state for the first GPU context to the first GPU context's RAMFC area, and loads + state for the second GPU context from the second GPU context's RAMFC area." + + "Every Host word entry in RAMFC directly corresponds to a PRI-accessible + register. For a description of the contents of a RAMFC entry, please see the + description of the corresponding register in "manuals/dev_pbdma.ref". The + offsets of the fields within each entry in RAMFC match those of the + corresponding register in the associated PBDMA unit's PRI space." + + In summary, RAMFC includes details such as the head and tail of the pushbuffer, + and RAMIN includes details such as the page table configuration(s). + + The instance-global page table (as defined in the PDB field) is only used for + GPU engines which do not support subcontexts (non-VEID engines). + + **Not all documented fields are currently populated below.** + + Support: *Kepler, *Maxwell, *Pascal, Volta, Turing, Ampere, [newer untested] + *Pre-Volta GPUs do not support subcontexts. + See also: dev_ram.ref.txt and dev_pbdma.ref.txt in NVIDIA's open-gpu-doc +*/ + +// 16-byte (128-bit) substructure defining a subcontext configuration +typedef struct { + page_dir_config_t pdb; + uint32_t pasid:20; // Process Address Space ID (PASID) used for ATS + uint32_t :11; + bool enable_ats:1; // Enable Address Translation Services (ATS)? + uint32_t pad; +} __attribute__((packed)) subcontext_ctrl_t; + +typedef struct { +// Start RAMFC (512 bytes) + uint32_t pad[43]; + uint32_t fc_target:5; // NV_RAMFC_TARGET; off 43 + uint32_t :27; + uint32_t pad2[17]; + uint32_t fc_config_l2:1; // NV_RAMFC_CONFIG; off 61 + uint32_t :3; + uint32_t fc_config_ce_split:1; + uint32_t fc_config_ce_no_throttle:1; + uint32_t :2; + uint32_t fc_config_is_priv:1; // ...AUTH_LEVEL + uint32_t :3; + uint32_t fc_config_userd_writeback:1; // ...USERD_WRITEBACK + uint32_t :19; + uint32_t pad3[1]; + uint32_t fc_chan_info_scg:1; // ...SET_CHANNEL_INFO_SCG_TYPE + uint32_t :7; + uint32_t fc_chan_info_veid:6; // ...SET_CHANNEL_INFO_VEID + uint32_t fc_chan_info_chid:12; // ...SET_CHANNEL_INFO_CHID + uint32_t :6; + uint32_t pad4[64]; +// End RAMFC +// Start RAMIN + page_dir_config_t pdb; + uint32_t pad5[2]; + // WFI_TARGET appears to be ignored if WFI_IS_VIRTUAL + uint32_t engine_wfi_target:2; // NV_RAMIN_ENGINE_WFI_TARGET; off 132 + uint32_t engine_wfi_is_virtual:1; + uint32_t :9; + // WFI_PTR points to a CTXSW block (documented below) + uint64_t engine_wfi_ptr:52; // NV_RAMIN_ENGINE_WFI_PTR_LO/_HI; off 132--133 + uint32_t engine_wfi_veid:6; // NV_RAMIN_ENGINE_WFI_VEID; off 134; VEID == Subcontext ID + uint32_t :26; + uint32_t pasid:20; // NV_RAMIN_PASID; off 135; "Process Address Space ID" + uint32_t :11; + bool enable_ats:1; + uint32_t pad6[30]; + uint64_t subcontext_pdb_valid; // NV_RAMIN_SC_PDB_VALID; off 166-167 + subcontext_ctrl_t subcontext[64]; // NV_RAMIN_SC_*; off 168-424 +} __attribute__((packed)) instance_ctrl_t; + +// Context types +enum CTXSW_TYPE { + CTXSW_UNDEFINED = 0x0, + CTXSW_OPENGL = 0x8, + CTXSW_DX9 = 0x10, + CTXSW_DX10 = 0x11, + CTXSW_DX11 = 0x12, + CTXSW_COMPUTE = 0x20, + CTXSW_HEADER = 0x21 // A per-subcontext header +}; +static inline const char *ctxsw_type_to_text(enum CTXSW_TYPE t) { + switch (t) { + case CTXSW_UNDEFINED: + return "[None]"; + case CTXSW_OPENGL: + return "OpenGL"; + case CTXSW_DX9: + case CTXSW_DX10: + case CTXSW_DX11: + return "DirectX"; + case CTXSW_COMPUTE: + return "Compute"; + case CTXSW_HEADER: + return "Header"; + default: + return "UNKNOWN"; + } +} + +// Preemption modes: +// WFI: Wait For Idle (preempt on idle) +// CTA: Cooperative Thread Array-level Preemption (preempt at end of block) +// CILP: Compute-Instruction-Level Preemption (preempt at end of instruction) +enum GRAPHICS_PREEMPT_TYPE {PREEMPT_WFI, PREEMPT_GFXP}; +enum COMPUTE_PREEMPT_TYPE {_PREEMPT_WFI, PREEMPT_CTA, PREEMPT_CILP}; +static inline const char *compute_preempt_type_to_text(enum COMPUTE_PREEMPT_TYPE t) { + switch (t) { + case PREEMPT_WFI: + return "WFI"; + case PREEMPT_CTA: + return "CTA"; + case PREEMPT_CILP: + return "CILP"; + default: + return "INVALID"; + } +} +static inline const char *graphics_preempt_type_to_text(enum COMPUTE_PREEMPT_TYPE t) { + switch (t) { + case PREEMPT_WFI: + return "WFI"; + case PREEMPT_GFXP: + return "GFXP"; + default: + return "INVALID"; + } +} + +/* ConTeXt SWitch control block (CTXSW) + Support: Maxwell*, Pascal**, Volta, Turing, Ampere, Ada + *Nothing except for CONTEXT_ID and TYPE + **Except as noted + See also: manuals/volta/gv100/dev_ctxsw.ref.txt in open-gpu-doc + and hw_ctxsw_prog_*.h in nvgpu +*/ +// (Note that this layout changes some generation-to-generation) +typedef struct context_switch_block { + uint32_t pad[3]; + enum CTXSW_TYPE type:6; // Unused except when type CTXSW_HEADER? + uint32_t :26; + uint32_t pad2[26]; + // The context buffer ptr fields are in an opposite-of-typical order, so we + // can't merge them into a single context_buffer_ptr field. + uint32_t context_buffer_ptr_hi; // Volta+ only + uint32_t context_buffer_ptr_lo; // Volta+ only + enum GRAPHICS_PREEMPT_TYPE graphics_preemption_options:32; + enum COMPUTE_PREEMPT_TYPE compute_preemption_options:32; + uint32_t pad3[18]; + uint32_t num_wfi_save_operations; + uint32_t num_cta_save_operations; + uint32_t num_gfxp_save_operations; + uint32_t num_cilp_save_operations; + uint32_t pad4[4]; + uint32_t context_id; + // [There are more fields not yet added here.] +} __attribute__((packed)) context_switch_ctrl_t; + /* VRAM Information If ECC is disabled: @@ -1452,6 +1691,12 @@ static inline uint64_t memory_range_to_bytes(memory_range_t range) { /* Begin nvdebug types and functions */ +// __iomem is only defined when building as a kernel module, so conditionally +// define it to allow including this header outside the kernel. +#ifndef __iomem +#define __iomem +#endif + // Vendor ID for PCI devices manufactured by NVIDIA #define NV_PCI_VENDOR 0x10de struct nvdebug_state { @@ -1474,6 +1719,10 @@ struct nvdebug_state { struct platform_device *platd; // Pointer to generic device struct (both platform and pcie devices) struct device *dev; +#ifdef __KERNEL__ + // List used by mmu.c to track allocated pages for page directories/tables + struct list_head pd_allocs; +#endif }; // This disgusting macro is a crutch to work around the fact that runlists were @@ -1542,7 +1791,19 @@ int get_runlist_iter( struct runlist_iter *rl_iter /* out */); int preempt_tsg(struct nvdebug_state *g, uint32_t rl_id, uint32_t tsg_id); int preempt_runlist(struct nvdebug_state *g, uint32_t rl_id); -int resubmit_runlist(struct nvdebug_state *g, uint32_t rl_id); +int resubmit_runlist(struct nvdebug_state *g, uint32_t rl_id, uint32_t off); +instance_ctrl_t *instance_deref( + struct nvdebug_state *g, + uint64_t instance_addr, + enum INST_TARGET instance_target); +context_switch_ctrl_t *get_ctxsw( + struct nvdebug_state *g, + instance_ctrl_t *inst); +int set_channel_preemption_mode( + struct nvdebug_state *g, + uint32_t chan_id, + uint32_t rl_id, + enum COMPUTE_PREEMPT_TYPE mode); // Defined in mmu.c uint64_t search_page_directory( @@ -1550,11 +1811,33 @@ uint64_t search_page_directory( page_dir_config_t pd_config, uint64_t addr_to_find, enum INST_TARGET addr_to_find_aperture); +int translate_page_directory( + struct nvdebug_state *g, + page_dir_config_t pd_config, + uint64_t addr_to_find, + uint64_t *found_addr /* out */, + enum INST_TARGET *found_aperture /* out */); +int map_page_directory( + struct nvdebug_state *g, + page_dir_config_t pd_config, + uint64_t paddr_to_map, + uint64_t vaddr_to_find, + enum INST_TARGET paddr_target, + bool huge_page); +int gc_page_directory( + struct nvdebug_state *g, + bool force); uint64_t search_v1_page_directory( struct nvdebug_state *g, page_dir_config_t pd_config, uint64_t addr_to_find, enum INST_TARGET addr_to_find_aperture); +int translate_v1_page_directory( + struct nvdebug_state *g, + page_dir_config_t pd_config, + uint64_t addr_to_find, + uint64_t *found_addr /* out */, + enum INST_TARGET *found_aperture /* out */); // Defined in bus.c int addr_to_pramin_mut(struct nvdebug_state *g, uint64_t addr, enum INST_TARGET target); int get_bar2_pdb(struct nvdebug_state *g, page_dir_config_t* pd /* out */); -- cgit v1.2.2