From 293430fcb5d4013b573556c58457ee706e482b7f Mon Sep 17 00:00:00 2001 From: Joshua Bakita Date: Mon, 5 May 2025 03:53:01 -0400 Subject: Snapshot for ECRTS'25 artifact evaluation --- nvdebug_entry.c | 476 ++++++++++++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 459 insertions(+), 17 deletions(-) (limited to 'nvdebug_entry.c') diff --git a/nvdebug_entry.c b/nvdebug_entry.c index 3a10e13..c0cfa63 100644 --- a/nvdebug_entry.c +++ b/nvdebug_entry.c @@ -15,7 +15,7 @@ // Enable to intercept and log GPU interrupts. Historically used to benchmark // interrupt latency. -#define INTERRUPT_DEBUG 0 +#define INTERRUPT_DEBUG // MIT is GPL-compatible. We need to be GPL-compatible for symbols like // platform_bus_type or bus_find_device_by_name... @@ -28,12 +28,20 @@ extern struct file_operations runlist_file_ops; extern struct file_operations preempt_tsg_file_ops; extern struct file_operations disable_channel_file_ops; extern struct file_operations enable_channel_file_ops; +extern struct file_operations wfi_preempt_channel_file_ops; +extern struct file_operations cta_preempt_channel_file_ops; +extern struct file_operations cil_preempt_channel_file_ops; extern struct file_operations resubmit_runlist_file_ops; +extern struct file_operations preempt_runlist_file_ops; +extern struct file_operations ack_bad_tsg_file_ops; +extern struct file_operations map_mem_chid_file_ops; +extern struct file_operations map_mem_ctxid_file_ops; extern struct file_operations switch_to_tsg_file_ops; // device_info_procfs.c extern struct file_operations device_info_file_ops; extern struct file_operations nvdebug_read_reg32_file_ops; extern struct file_operations nvdebug_read_reg_range_file_ops; +extern struct file_operations nvdebug_read_part_file_ops; extern struct file_operations local_memory_file_ops; // copy_topology_procfs.c extern struct file_operations copy_topology_file_ops; @@ -71,9 +79,271 @@ const struct file_operations* compat_ops(const struct file_operations* ops) { } #endif -#if INTERRUPT_DEBUG +#ifdef INTERRUPT_DEBUG + +void nvdebug_fifo_intr(struct nvdebug_state *g) { + uint32_t fifo_intr_mask;// = nvdebug_readl(g, 0x02100); // PFIFO_INTR_0 + fifo_intr_mask = nvdebug_readl(g, 0x02100); // PFIFO_INTR_0 + if (fifo_intr_mask & 1 << 0) + printk(KERN_INFO "[nvdebug] - Interrupt BIND_ERROR.\n"); + if (fifo_intr_mask & 1 << 1) + printk(KERN_INFO "[nvdebug] - Interrupt CTXSW_TIMEOUT.\n"); + if (fifo_intr_mask & 1 << 4) + printk(KERN_INFO "[nvdebug] - Interrupt RUNLIST_IDLE.\n"); + if (fifo_intr_mask & 1 << 5) + printk(KERN_INFO "[nvdebug] - Interrupt RUNLIST_AND_ENG_IDLE.\n"); + if (fifo_intr_mask & 1 << 6) + printk(KERN_INFO "[nvdebug] - Interrupt RUNLIST_ACQUIRE.\n"); + if (fifo_intr_mask & 1 << 7) + printk(KERN_INFO "[nvdebug] - Interrupt RUNLIST_ACQUIRE_AND_ENG_IDLE.\n"); + if (fifo_intr_mask & 1 << 8) + printk(KERN_INFO "[nvdebug] - Interrupt SCHED_ERROR.\n"); + if (fifo_intr_mask & 1 << 16) + printk(KERN_INFO "[nvdebug] - Interrupt CHSW_ERROR.\n"); + if (fifo_intr_mask & 1 << 23) + printk(KERN_INFO "[nvdebug] - Interrupt MEMOP_TIMEOUT.\n"); + if (fifo_intr_mask & 1 << 24) + printk(KERN_INFO "[nvdebug] - Interrupt LB_ERROR.\n"); + if (fifo_intr_mask & 1 << 25) // OLD; Pascal + printk(KERN_INFO "[nvdebug] - Interrupt REPLAYABLE_FAULT_ERROR.\n"); + if (fifo_intr_mask & 1 << 27) // OLD; Pascal + printk(KERN_INFO "[nvdebug] - Interrupt DROPPED_MMU_FAULT.\n"); + if (fifo_intr_mask & 1 << 28) { // On Pascal, this is MMU_FAULT + if (g->chip_id <= NV_CHIP_ID_VOLTA) // MMU_FAULT on Pascal (nvgpu, l4t/l4t-r28.1:drivers/gpu/nvgpu/include/nvgpu/hw/gp10b/hw_fifo_gp10b.h) + printk(KERN_INFO "[nvdebug] - Interrupt MMU_FAULT.\n"); + else // Repurposed starting with Turing: open-gpu-doc/manuals/turing/tu104/dev_fifo.ref.txt + printk(KERN_INFO "[nvdebug] - Interrupt TSG_PREEMPT_COMPLETE.\n"); + } + if (fifo_intr_mask & 1 << 29) + printk(KERN_INFO "[nvdebug] - Interrupt PBDMA_INTR.\n"); + if (fifo_intr_mask & 1 << 30) { + printk(KERN_INFO "[nvdebug] - Interrupt RUNLIST_EVENT.\n"); + uint32_t fifo_runlist_intr_mask = nvdebug_readl(g, 0x02A00); // PFIFO_INTR_RUNLIST + printk(KERN_INFO "[nvdebug] - Event %#x.\n", fifo_runlist_intr_mask); + } + if (fifo_intr_mask & 1 << 31) + printk(KERN_INFO "[nvdebug] - Interrupt CHANNEL_INTR.\n"); +} + irqreturn_t nvdebug_irq_tap(int irq_num, void * dev) { - printk(KERN_INFO "[nvdebug] Interrupt tap triggered on IRQ %d.\n", irq_num); + struct nvdebug_state *g = dev; + u64 time = ktime_get_raw_ns(); // CLOCK_MONOTONTIC_RAW + // NV_PMC_INTR does not exist on Ada, so use NV_FUNC_PRIV_CPU_INTR_TOP + // Note that this also appears to exist on Turing + if (g->chip_id >= NV_CHIP_ID_TURING) {//AMPERE) { + int i; + // Despite being an indexed register, it is only documented to have on, and could only support two + uint32_t intr_mask0 = nvdebug_readl(g, NV_VIRTUAL_FUNCTION_FULL_PHYS_OFFSET + 0x1600); // NV_FUNC_PRIV_CPU_INTR_TOP(0) + uint32_t intr_mask1 = nvdebug_readl(g, NV_VIRTUAL_FUNCTION_FULL_PHYS_OFFSET + 0x1604); // NV_FUNC_PRIV_CPU_INTR_TOP(1) + printk(KERN_INFO "[nvdebug] Interrupt on IRQ %d with CPU_INTR_TOP(0) %#010x, ...(1) %#010x @ %llu.\n", irq_num, intr_mask0, intr_mask1, time); + for (i = 0; i < 8; i++) { + uint32_t leaf = nvdebug_readl(g, NV_VIRTUAL_FUNCTION_FULL_PHYS_OFFSET + 0x1000 + i*4); // NV_FUNC_PRIV_CPU_INTR_LEAF(0) to ...(7) + if (leaf) + printk(KERN_INFO "[nvdebug] - Interrupt leaf %d: %#010x\n", i, leaf); + // 131-133 & 64 are faults on tu104??? (open-gpu-doc/manuals/turing/tu104/pri_mmu_hub.ref.txt) + if (136 / 32 == i && 1 << (136 % 32) & leaf) // PFIFO0 ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on PFIFO0.\n"); + if (137 / 32 == i && 1 << (137 % 32) & leaf) // PFIFO1 ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on PFIFO1.\n"); + if (148 / 32 == i && 1 << (148 % 32) & leaf) // TIMER ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on PTIMER.\n"); + if (152 / 32 == i && 1 << (152 % 32) & leaf) // PMU ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on PMU.\n"); + if (156 / 32 == i && 1 << (156 % 32) & leaf) { // PBUS ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on PBUS.\n"); + uint32_t bus_intr = nvdebug_readl(g, 0x1100); // BUS_INTR_0 + if (bus_intr & 1 << 2) { + // use timer_pri_timeout_save_0_r + uint32_t SAVE_0 = nvdebug_readl(g, 0x00009084); // NV_PTIMER_PRI_TIMEOUT_SAVE_0 + printk(KERN_INFO "[nvdebug] - Interrupt PRI_FECSERR on %s to address %#010x %stargeting FECS.\n", SAVE_0 & 0x2 ? "write" : "read", SAVE_0 & 0x00fffffc, SAVE_0 & 0x80000000 ? "" : "not "); + uint32_t SAVE_1 = nvdebug_readl(g, 0x00009088); // NV_PTIMER_PRI_TIMEOUT_SAVE_1 + if (SAVE_1) + printk(KERN_INFO "[nvdebug] Data written: %#010x\n", SAVE_1); + uint32_t errcode = readl(g->regs + 0x0000908C); // NV_PTIMER_PRI_TIMEOUT_FECS_ERRCODE + if (errcode) + printk(KERN_INFO "[nvdebug] FECS Error Code: %#010x\n", errcode); + // badf5040 is a "client error" (0) of "no such address" (40) + // See linux-nvgpu/drivers/gpu/nvgpu/hal/priv_ring/priv_ring_ga10b_fusa.c + // for how to decode. + } + if (bus_intr & 1 << 3) + printk(KERN_INFO "[nvdebug] - Interrupt PRI_TIMEOUT.\n"); + if (bus_intr & 1 << 4) + printk(KERN_INFO "[nvdebug] - Interrupt FB_REQ_TIMEOUT.\n"); + if (bus_intr & 1 << 5) + printk(KERN_INFO "[nvdebug] - Interrupt FB_ACK_TIMEOUT.\n"); + if (bus_intr & 1 << 6) + printk(KERN_INFO "[nvdebug] - Interrupt FB_ACK_EXTRA.\n"); + if (bus_intr & 1 << 7) + printk(KERN_INFO "[nvdebug] - Interrupt FB_RDATA_TIMEOUT.\n"); + if (bus_intr & 1 << 8) + printk(KERN_INFO "[nvdebug] - Interrupt FB_RDATA_EXTRA.\n"); + if (bus_intr & 1 << 26) + printk(KERN_INFO "[nvdebug] - Interrupt SW.\n"); + if (bus_intr & 1 << 27) + printk(KERN_INFO "[nvdebug] - Interrupt POSTED_DEADLOCK_TIMEOUT.\n"); + if (bus_intr & 1 << 28) + printk(KERN_INFO "[nvdebug] - Interrupt MPMU.\n"); + if (bus_intr & 1 << 31) + printk(KERN_INFO "[nvdebug] - Interrupt ACCESS_TIMEOUT.\n"); + } + if (158 / 32 == i && 1 << (158 % 32) & leaf) // PRIV_RING ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on PRIV_RING.\n"); + if (192 / 32 == i && 1 << (192 % 32) & leaf) // LEGACY_ENGINE_STALL ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on LEGACY_ENGINE_STALL.\n"); + if (160 / 32 == i && 1 << (160 % 32) & leaf) { // (likely) rl0 ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on RUNLIST0.\n"); + uint32_t off; + get_runlist_ram(g, 0, &off); + uint32_t rl_intr = nvdebug_readl(g, off+0x100); + printk(KERN_INFO "[nvdebug] - RUNLIST_INTR_0: %#x\n", rl_intr); + if (1 << 12 && rl_intr) { // BAD_TSG + printk(KERN_INFO "[nvdebug] - BAD_TSG: %#x\n", nvdebug_readl(g, off+0x174)); + } + } + // Also getting 160, 161, and 162 + + //uint32_t off; + //get_runlist_ram(g, 12, &off); + //printk(KERN_INFO "[nvdebug] - rl10 vector id 0 is %x\n", nvdebug_readl(g, off+0x160)); // NV_RUNLIST_INTR_VECTORID(0) + // 160 is rl0 (C/G, LCE0, LCE1) vector id 0 + // 168 is rl11 (LCE3) vector id 0 + // 169 is rl12 (LCE4) vector id 0 + // 171 is rl1 (SEC) vector id 0 + // 176 is rl10 (LCE2) vector id 0 + // 224 is rl0 vector id 1 + // Only some interrupt vectors are hardcoded + } + // each subtree has two leafs? Each bit at the top corresponds to a subtree? + // So, if bit 0 is set, that means subtree 0 (concept) and leaves 0 and 1 + // So, if bit 1 is set, that means subtree 1 (concept) and leaves 2 and 3 + // the #define'd interrupt vectors all seem to fall in the lower leaf of subtree 2, + // except for INTR_HUB_ACCESS_CNTR_INTR_VECTOR is in the lower leaf of subtree 1 + if (g->chip_id >= NV_CHIP_ID_AMPERE) + return IRQ_NONE; + } + uint32_t intr_mask = nvdebug_readl(g, 0x0100); // NV_PMC_INTR + printk(KERN_INFO "[nvdebug] Interrupt on IRQ %d with MC_INTR %#010x @ %llu.\n", irq_num, intr_mask, time); + // IDs likely changed Ampere+ + //if (g->chip_id >= NV_CHIP_ID_AMPERE) { + // CIC is central interrupt controller + // the u32 passed around nvgpu cic functions is one of the + // enable is nvgpu_cic_mon_intr_stall_unit_config(unit) + // - Calls intr_stall_unit_config(unit) + // - for ga, calls unit = ga10b_intr_map_mc_stall_unit_to_intr_unit(unit) (doesn't do much) + // - for ga, calls nvgpu_cic_mon_intr_get_unit_info() + // - Does *subtree = g->mc.intr_unit_info[unit].subtree; + // *subtree_mask = g->mc.intr_unit_info[unit].subtree_mask; + // - for ga, calls ga10b_intr_config() w/ subtree info + //uint32_t intr_stats = nvdebug_readl(g, 1600 + //return IRQ_NONE; + //} + if (intr_mask & 1 << 5) + printk(KERN_INFO "[nvdebug] - Interrupt on LCE0.\n"); + if (intr_mask & 1 << 6) + printk(KERN_INFO "[nvdebug] - Interrupt on LCE1.\n"); + if (intr_mask & 1 << 7) + printk(KERN_INFO "[nvdebug] - Interrupt on LCE2.\n"); + if (intr_mask & 1 << 8) { + printk(KERN_INFO "[nvdebug] - Interrupt on PFIFO.\n"); + nvdebug_fifo_intr(g); + } + if (intr_mask & 1 << 9) { + printk(KERN_INFO "[nvdebug] - Interrupt on HUB.\n"); // "replayable_fault_pending" in nvgpu on Pascal, "HUB" on Volta+ + // on tu104, if vector is one of the below set in new-style interrupt vector, then MMU fault + // - info_fault (134) + // - nonreplay_fault error (133) + // - nonreplay_fault notify (132) + // - replay_fault error (131) + // - replay_fault notify (64) + // (but the above fault vectors are configurable) + // if it's ecc_error, then not mmu error + // Default fault vectors from open-gpu-doc/manuals/turing/tu104/pri_mmu_hub.ref.txt + // Turing through (at least) Ampere (per nvgpu) + + // on gv100, parse fb_niso_intr_r 0x00100a20U, where bits: + // - hub_access_counter notify (0) + // - hub_access_counter error (1) + // - replay_fault notify (27) + // - replay_fault overflow (28) + // - nonreplay_fault notify (29) + // - nonreplay_fault overflow (30) + // - other_fault notify (31) + // Volta through Turing (per nvgpu) + + // On Pascal, it looks like it's a property of fifo_intr_0??? + if (g->chip_id < NV_CHIP_ID_VOLTA) + nvdebug_fifo_intr(g); + } + if (intr_mask & 1 << 10) + printk(KERN_INFO "[nvdebug] - Interrupt on LCE3.\n"); + if (intr_mask & 1 << 11) + printk(KERN_INFO "[nvdebug] - Interrupt on LCE4.\n"); + if (intr_mask & 1 << 12) { + printk(KERN_INFO "[nvdebug] - Interrupt on Graphics/Compute.\n"); + // Kepler through (at least) Ampere + // From open-gpu-doc/manuals/volta/gv100/dev_graphics.ref.txt + uint32_t graph_intr_mask = nvdebug_readl(g, 0x400100); // NV_PGRAPH_INTR + if (graph_intr_mask & 1 << 0) + printk(KERN_INFO "[nvdebug] - Interrupt NOTIFY.\n"); + if (graph_intr_mask & 1 << 1) + printk(KERN_INFO "[nvdebug] - Interrupt SEMAPHORE.\n"); + if (graph_intr_mask & 1 << 4) + printk(KERN_INFO "[nvdebug] - Interrupt ILLEGAL_METHOD.\n"); + if (graph_intr_mask & 1 << 5) + printk(KERN_INFO "[nvdebug] - Interrupt ILLEGAL_CLASS.\n"); + if (graph_intr_mask & 1 << 6) + printk(KERN_INFO "[nvdebug] - Interrupt ILLEGAL_NOTIFY.\n"); + if (graph_intr_mask & 1 << 7) + printk(KERN_INFO "[nvdebug] - Interrupt DEBUG_METHOD.\n"); + if (graph_intr_mask & 1 << 8) + printk(KERN_INFO "[nvdebug] - Interrupt FIRMWARE_METHOD.\n"); + if (graph_intr_mask & 1 << 16) + printk(KERN_INFO "[nvdebug] - Interrupt BUFFER_NOTIFY.\n"); + if (graph_intr_mask & 1 << 19) + printk(KERN_INFO "[nvdebug] - Interrupt FECS_ERROR.\n"); + if (graph_intr_mask & 1 << 20) + printk(KERN_INFO "[nvdebug] - Interrupt CLASS_ERROR.\n"); + if (graph_intr_mask & 1 << 21) + printk(KERN_INFO "[nvdebug] - Interrupt EXCEPTION.\n"); + } + if (intr_mask & 1 << 13) + printk(KERN_INFO "[nvdebug] - Interrupt on PFB.\n"); + if (intr_mask & 1 << 15) + printk(KERN_INFO "[nvdebug] - Interrupt on SEC.\n"); + if (intr_mask & 1 << 16) + printk(KERN_INFO "[nvdebug] - Interrupt on NVENC0.\n"); + if (intr_mask & 1 << 17) + printk(KERN_INFO "[nvdebug] - Interrupt on NVDEC0.\n"); + if (intr_mask & 1 << 18) + printk(KERN_INFO "[nvdebug] - Interrupt on THERMAL.\n"); + if (intr_mask & 1 << 19) + printk(KERN_INFO "[nvdebug] - Interrupt on HDACODEC.\n"); + if (intr_mask & 1 << 20) + printk(KERN_INFO "[nvdebug] - Interrupt on PTIMER.\n"); + if (intr_mask & 1 << 21) + printk(KERN_INFO "[nvdebug] - Interrupt on PMGR.\n"); + if (intr_mask & 1 << 22) + printk(KERN_INFO "[nvdebug] - Interrupt on IOCTRL.\n"); + if (intr_mask & 1 << 23) + printk(KERN_INFO "[nvdebug] - Interrupt on DFD.\n"); + if (intr_mask & 1 << 24) + printk(KERN_INFO "[nvdebug] - Interrupt on PMU.\n"); + if (intr_mask & 1 << 25) + printk(KERN_INFO "[nvdebug] - Interrupt on LTC.\n"); + if (intr_mask & 1 << 26) + printk(KERN_INFO "[nvdebug] - Interrupt on PDISP.\n"); + if (intr_mask & 1 << 27) + printk(KERN_INFO "[nvdebug] - Interrupt on GSP.\n"); + if (intr_mask & 1 << 28) + printk(KERN_INFO "[nvdebug] - Interrupt on PBUS.\n"); + if (intr_mask & 1 << 29) + printk(KERN_INFO "[nvdebug] - Interrupt on XVE.\n"); + if (intr_mask & 1 << 30) + printk(KERN_INFO "[nvdebug] - Interrupt on PRIV_RING.\n"); + if (intr_mask & 1 << 30) + printk(KERN_INFO "[nvdebug] - Interrupt on SOFTWARE.\n"); + return IRQ_NONE; // We don't actually handle any interrupts. Pass them on. } #endif // INTERRUPT_DEBUG @@ -135,6 +405,7 @@ int probe_and_cache_devices(void) { g_nvdebug_state[i].pcid = NULL; g_nvdebug_state[i].platd = platd; g_nvdebug_state[i].dev = dev; + INIT_LIST_HEAD(&g_nvdebug_state[i].pd_allocs); // Don't check Chip ID until everything else is initalized ids.raw = nvdebug_readl(&g_nvdebug_state[i], NV_MC_BOOT_0); if (ids.raw == -1) { @@ -152,6 +423,11 @@ int probe_and_cache_devices(void) { mc_boot_0_t ids; g_nvdebug_state[i].g = NULL; // Map BAR0 (GPU control registers) + // XXX: Don't use pci_iomap. This adds support for I/O registers, but we do + // not use the required ioread/write functions for those regions. We + // should use pci_ioremap_bar, which is explictly for MMIO regions. + // pci_ioremap_bar -> ioremap_nocache (all platforms) + // pci_iomap -> ioremap_nocache (on x86) g_nvdebug_state[i].regs = pci_iomap(pcid, 0, 0); if (!g_nvdebug_state[i].regs) { pci_err(pcid, "[nvdebug] Unable to map BAR0 on this GPU\n"); @@ -163,9 +439,14 @@ int probe_and_cache_devices(void) { // (vesafb may map the top half for display) if (!g_nvdebug_state[i].bar3) g_nvdebug_state[i].bar3 = pci_iomap(pcid, 3, pci_resource_len(pcid, 3)/2); + // Observed on H100, BAR2, moved it BAR3, was moved to BAR4, and BAR1 + // was moved to BAR2. + if (!g_nvdebug_state[i].bar3) + g_nvdebug_state[i].bar3 = pci_iomap(pcid, 4, 0); g_nvdebug_state[i].pcid = pcid; g_nvdebug_state[i].platd = NULL; g_nvdebug_state[i].dev = &pcid->dev; + INIT_LIST_HEAD(&g_nvdebug_state[i].pd_allocs); // Don't check Chip ID until everything else is initalized ids.raw = nvdebug_readl(&g_nvdebug_state[i], NV_MC_BOOT_0); if (ids.raw == -1) { @@ -175,9 +456,17 @@ int probe_and_cache_devices(void) { g_nvdebug_state[i].chip_id = ids.chip_id; printk(KERN_INFO "[nvdebug] Chip ID %x (architecture %s) detected on PCI bus and initialized.", ids.chip_id, ARCH2NAME(ids.architecture)); -#if INTERRUPT_DEBUG - if (request_irq(pcid->irq, nvdebug_irq_tap, IRQF_SHARED, "nvdebug tap", pcid)) { - printk(KERN_WARNING "[nvdebug] Unable to initialize IRQ tap\n"); +#ifdef INTERRUPT_DEBUG + // For this to work, you must also add IRQF_SHARED to the flags + // argument of the request_threaded_irq() call in the nvidia driver + // (file /usr/src/nvidia.../nvidia/nv.c and nv-msi.c with dkms) + // Then run: + // sudo dkms remove nvidia-srv/VER -k $(uname -r) + // sudo dkms install nvidia-srv/VER -k $(uname -r) --force + // where VER is the version of the nvidia module (eg. 535.216.03) + int err; + if ((err = request_irq(pcid->irq, nvdebug_irq_tap, IRQF_SHARED, "nvdebug tap", &g_nvdebug_state[i]))) { + printk(KERN_WARNING "[nvdebug] Unable to initialize IRQ tap, error %d\n", err); } #endif // INTERRUPT_DEBUG i++; @@ -335,6 +624,40 @@ int __init nvdebug_init(void) { "enable_channel", 0222, chram_scope, compat_ops(&enable_channel_file_ops), (void*)last_runlist)) goto out_nomem; + // Create file `/proc/gpu#/runlist#/wfi_preempt_channel`, world writable + // On Turing and older, `/proc/gpu#/wfi_preempt_channel` + if (!proc_create_data( + "wfi_preempt_channel", 0222, chram_scope, compat_ops(&wfi_preempt_channel_file_ops), + (void*)last_runlist)) + goto out_nomem; + // Create file `/proc/gpu#/runlist#/cta_preempt_channel`, world writable + // On Turing and older, `/proc/gpu#/cta_preempt_channel` + if (!proc_create_data( + "cta_preempt_channel", 0222, chram_scope, compat_ops(&cta_preempt_channel_file_ops), + (void*)last_runlist)) + goto out_nomem; + // Compute-instruction-level (CIL) preemption is only available on Pascal+ + if (g_nvdebug_state[res].chip_id >= NV_CHIP_ID_PASCAL) { + // Create file `/proc/gpu#/runlist#/cil_preempt_channel`, world writable + // On Turing and older, `/proc/gpu#/cil_preempt_channel` + if (!proc_create_data( + "cil_preempt_channel", 0222, chram_scope, compat_ops(&cil_preempt_channel_file_ops), + (void*)last_runlist)) + goto out_nomem; + } + // Create files which enable on-GPU scheduling (Pascal+) + if (g_nvdebug_state[res].chip_id >= NV_CHIP_ID_PASCAL) { + // Create file `/proc/gpu#/map_mem_chid`, root writable + if (!proc_create_data( + "map_mem_chid", 0200, chram_scope, compat_ops(&map_mem_chid_file_ops), + (void*)last_runlist)) + goto out_nomem; + // Create file `/proc/gpu#/map_mem_ctxid`, root writable + if (!proc_create_data( + "map_mem_ctxid", 0222, rl_dir, compat_ops(&map_mem_ctxid_file_ops), + (void*)last_runlist)) + goto out_nomem; + } } // Create file `/proc/gpu#/runlist#/runlist`, world readable if (!proc_create_data( @@ -346,16 +669,26 @@ int __init nvdebug_init(void) { "switch_to_tsg", 0222, rl_dir, compat_ops(&switch_to_tsg_file_ops), (void*)last_runlist)) goto out_nomem; + /* On the TU104, the context scheduler (contained in the Host, aka + * PFIFO, unit) has been observed to sometimes to fail to schedule TSGs + * containing re-enabled channels. Resubmitting the runlist + * configuration appears to remediate this condition, and so this API + * is exposed to help reset GPU scheduling as necessary. + */ + // Create file `/proc/gpu#/resubmit_runlist`, world writable + if (!proc_create_data( + "resubmit_runlist", 0222, rl_dir, compat_ops(&resubmit_runlist_file_ops), + (void*)device_id)) + goto out_nomem; } while (last_runlist-- > 0); - /* On the TU104, the context scheduler (contained in the Host, aka - * PFIFO, unit) has been observed to sometimes to fail to schedule TSGs - * containing re-enabled channels. Resubmitting the runlist - * configuration appears to remediate this condition, and so this API - * is exposed to help reset GPU scheduling as necessary. - */ - // Create file `/proc/gpu#/resubmit_runlist`, world writable + // Create file `/proc/gpu#/preempt_runlist`, world writable if (!proc_create_data( - "resubmit_runlist", 0222, dir, compat_ops(&resubmit_runlist_file_ops), + "preempt_runlist", 0222, dir, compat_ops(&preempt_runlist_file_ops), + (void*)device_id)) + goto out_nomem; + // Create file `/proc/gpu#/ack_bad_tsg`, world writable + if (!proc_create_data( + "ack_bad_tsg", 0222, dir, compat_ops(&ack_bad_tsg_file_ops), (void*)device_id)) goto out_nomem; // Create file `/proc/gpu#/device_info`, world readable @@ -394,6 +727,68 @@ int __init nvdebug_init(void) { (void*)NV_FUSE_GPC_GM107)) goto out_nomem; } + // Create file `/proc/gpu#/CWD_SM_ID#`, world readable (Maxwell+) + // Create file `/proc/gpu#/CWD_GPC_TPC_ID#`, world readable (Maxwell+) + // - 6 entries on Maxwell (nvgpu) + // - 16 entries on Pascal through Ampere (at least) (nvgpu, open-gpu-doc) + // - 24 entries on Hopper through Ada (at least) (XXXX) + // XXX: Only working while a context is active + // XXX: Needed for libsmctrl2; hacky + // Tested on GP104, TU102, GV100, AD102 + if (g_nvdebug_state[res].chip_id >= NV_CHIP_ID_HOPPER) { + char file_name[21]; + long i; + for (i = 0; i < 24; i++) { + snprintf(file_name, 20, "CWD_SM_ID%ld", i); + if (!proc_create_data( + file_name, 0444, dir, compat_ops(&nvdebug_read_reg32_file_ops), + (void*)(0x00405100+4*i))) // XXX: From XXXX + goto out_nomem; + // 18 entries on Ada (RTX 6000 Ada) + // Returns 0xbadf1201 if GPU not active + snprintf(file_name, 20, "CWD_GPC_TPC_ID%ld", i); + if (!proc_create_data( + file_name, 0444, dir, compat_ops(&nvdebug_read_reg32_file_ops), + // Nothing between this location and CWD_SM_ID + (void*)((0x00405000)+4*i))) // Found via reverse search from CWD_SM_ID location on Ada + goto out_nomem; + // Nothing in the following 28 words (before 0x00405220) + } + } else if (g_nvdebug_state[res].chip_id >= NV_CHIP_ID_MAXWELL) { + char file_name[21]; + long i; + union reg_range num_gpc_range; + for (i = 0; i < 16; i++) { + snprintf(file_name, 20, "CWD_SM_ID%ld", i); + if (!proc_create_data( + file_name, 0444, dir, compat_ops(&nvdebug_read_reg32_file_ops), + (void*)(0x00405ba0+4*i))) // NV_PGRAPH_PRI_CWD_SM_ID(i) + goto out_nomem; + // ? entries on Maxwell + // 8 entries on Pascal (test) + // 16 entries on Volta through Ampere (open-gpu-doc) + // Returns 0 if GPU ont active + snprintf(file_name, 20, "CWD_GPC_TPC_ID%ld", i); + if (!proc_create_data( + file_name, 0444, dir, compat_ops(&nvdebug_read_reg32_file_ops), + (void*)(0x00405b60+4*i))) // NV_PGRAPH_PRI_CWD_GPC_TPC_ID(i) + goto out_nomem; + } + num_gpc_range.offset = 0x00405b00; // NV_PGRAPH_PRI_CWD_FS + // Lower eight bits of register are _NUM_GPCS + num_gpc_range.start_bit = 0; + num_gpc_range.stop_bit = 8; + if (!proc_create_data( + "CWD_FS_NUM_GPCS", 0444, dir, compat_ops(&nvdebug_read_reg_range_file_ops), + (void*)(num_gpc_range.raw))) + goto out_nomem; + num_gpc_range.start_bit = 8; + num_gpc_range.stop_bit = 16; + if (!proc_create_data( + "CWD_FS_NUM_TPCS", 0444, dir, compat_ops(&nvdebug_read_reg_range_file_ops), + (void*)(num_gpc_range.raw))) + goto out_nomem; + } // Create file `/proc/gpu#/local_memory`, world readable (Pascal+) if (g_nvdebug_state[res].chip_id >= NV_CHIP_ID_PASCAL) { if (!proc_create_data( @@ -414,6 +809,50 @@ int __init nvdebug_init(void) { (void*)NV_CE_PCE_MAP)) goto out_nomem; } + // Create files exposing subcontext partitioning (Volta+) + // TODO: Make this not a hack with undocumented magic numbers + if (g_nvdebug_state[res].chip_id >= NV_CHIP_ID_VOLTA) { + char file_name[21]; + long i; + // Create file `/proc/gpu#/partition_ctl`, world readable + if (!proc_create_data( + "partition_ctl", 0444, dir, compat_ops(&nvdebug_read_reg32_file_ops), + (void*)0x00405b2c)) + goto out_nomem; + // Create file `/proc/gpu#/partition_data`, world readable + if (!proc_create_data( + "partition_data", 0444, dir, compat_ops(&nvdebug_read_reg32_file_ops), + (void*)0x00405b30)) + goto out_nomem; + // Create file `/proc/gpu#/partition_data#`, world readable + for (i = 0; i < 64; i++) { + snprintf(file_name, 20, "partition_data%ld", i); + if (!proc_create_data( + file_name, 0444, dir, compat_ops(&nvdebug_read_part_file_ops), + (void*)i)) + goto out_nomem; + } + // For debugging what MPS is changing + // Create file `/proc/gpu#/CWD_CG0`, world readable + if (!proc_create_data( + "CWD_CG0", 0444, dir, compat_ops(&nvdebug_read_reg32_file_ops), + (void*)0x00405bf0)) + goto out_nomem; + // Create file `/proc/gpu#/CWD_CG1`, world readable + if (!proc_create_data( + "CWD_CG1", 0444, dir, compat_ops(&nvdebug_read_reg32_file_ops), + (void*)0x00405bf4)) + goto out_nomem; + // Create file `/proc/gpu#/CWD_GPC_TPC_ID#`, world readable + // This does not appear to work on Hopper. Works on Ampere. + /*for (i = 0; i < 16; i++) { + snprintf(file_name, 20, "CWD_GPC_TPC_ID%ld", i); + if (!proc_create_data( + file_name, 0444, dir, compat_ops(&nvdebug_read_reg32_file_ops), + (void*)(0x00405b60+4*i))) + goto out_nomem; + }*/ + } } // (See Makefile if you want to know the origin of GIT_HASH.) printk(KERN_INFO "[nvdebug] Module version "GIT_HASH" initialized\n"); @@ -439,16 +878,19 @@ static void __exit nvdebug_exit(void) { char device_id[7]; snprintf(device_id, 7, "gpu%d", g_nvdebug_devices); remove_proc_subtree(device_id, NULL); + // Force-free associated allocations g = &g_nvdebug_state[g_nvdebug_devices]; + gc_page_directory(g, true); // Free BAR mappings for PCIe devices if (g && g->pcid) { +#ifdef INTERRUPT_DEBUG + // IRQ handler uses g->regs, so free IRQ first + free_irq(g->pcid->irq, g); +#endif // INTERRUPT_DEBUG if (g->regs) pci_iounmap(g->pcid, g->regs); if (g->bar2) pci_iounmap(g->pcid, g->bar2); -#if INTERRUPT_DEBUG - free_irq(g->pcid->irq, g->pcid); -#endif // INTERRUPT_DEBUG } else { if (g->regs) iounmap(g->regs); -- cgit v1.2.2