aboutsummaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
authorJoshua Bakita <bakitajoshua@gmail.com>2025-05-05 03:53:01 -0400
committerJoshua Bakita <bakitajoshua@gmail.com>2025-05-05 03:53:13 -0400
commit293430fcb5d4013b573556c58457ee706e482b7f (patch)
tree9328fa680f55b4e1a08d24714275b8437be3be5d
parent494df296bf4abe9b2b484bde1a4fad28c989afec (diff)
Snapshot for ECRTS'25 artifact evaluation
-rw-r--r--Makefile3
-rw-r--r--README.md10
-rw-r--r--device_info_procfs.c79
-rw-r--r--mmu.c414
-rw-r--r--nvdebug.h293
-rw-r--r--nvdebug_entry.c476
-rw-r--r--nvdebug_linux.h5
-rw-r--r--runlist.c275
-rw-r--r--runlist_procfs.c645
9 files changed, 2154 insertions, 46 deletions
diff --git a/Makefile b/Makefile
index fea3819..9d6d374 100644
--- a/Makefile
+++ b/Makefile
@@ -8,3 +8,6 @@ all:
8 make -C /lib/modules/$(shell uname -r)/build M=$(PWD) modules 8 make -C /lib/modules/$(shell uname -r)/build M=$(PWD) modules
9clean: 9clean:
10 make -C /lib/modules/$(shell uname -r)/build M=$(PWD) clean 10 make -C /lib/modules/$(shell uname -r)/build M=$(PWD) clean
11
12nvdebug_user.so: runlist.c mmu.c bus.c nvdebug_user.c
13 gcc $< -shared -o $@ $(KBULID_CFLAGS)
diff --git a/README.md b/README.md
index da3e5d7..2889b29 100644
--- a/README.md
+++ b/README.md
@@ -59,6 +59,7 @@ Not all these TPCs will necessarially be enabled in every GPC.
59Use `cat gpcX_tpc_mask` to get a bit mask of which TPCs are disabled for GPC X. 59Use `cat gpcX_tpc_mask` to get a bit mask of which TPCs are disabled for GPC X.
60A set bit indicates a disabled TPC. 60A set bit indicates a disabled TPC.
61This API is only available on enabled GPCs. 61This API is only available on enabled GPCs.
62Bits greater than the number of on-chip TPCs per GPC should be ignored (it may appear than non-existent TPCs are "disabled").
62 63
63Example usage: To get the number of on-chip SMs on Volta+ GPUs, multiply the return of `cat num_gpcs` with `cat num_tpc_per_gpc` and multiply by 2 (SMs per TPC). 64Example usage: To get the number of on-chip SMs on Volta+ GPUs, multiply the return of `cat num_gpcs` with `cat num_tpc_per_gpc` and multiply by 2 (SMs per TPC).
64 65
@@ -83,6 +84,13 @@ Use `echo Z > runlistY/switch_to_tsg` to switch the GPU to run only the specifie
83 84
84Use `echo Y > resubmit_runlist` to resubmit runlist Y (useful to prompt newer GPUs to pick up on re-enabled channels). 85Use `echo Y > resubmit_runlist` to resubmit runlist Y (useful to prompt newer GPUs to pick up on re-enabled channels).
85 86
87## Error Interpretation
88First check the kernel log to see if in includes more information about the error.
89The following conventions are used for certain error codes:
90
91- EIO, "Input/Output Error," is returned when an operation fails due to a bad register read.
92- (Other errors may not have a consistent conventional meaning; see the implementation.)
93
86## General Codebase Structure 94## General Codebase Structure
87- `nvdebug.h` defines and describes all GPU data structures. This does not depend on any kernel-internal headers. 95- `nvdebug.h` defines and describes all GPU data structures. This does not depend on any kernel-internal headers.
88- `nvdebug_entry.h` contains module startup, device detection, initialization, and module teardown logic. 96- `nvdebug_entry.h` contains module startup, device detection, initialization, and module teardown logic.
@@ -94,4 +102,4 @@ Use `echo Y > resubmit_runlist` to resubmit runlist Y (useful to prompt newer GP
94 102
95- The runlist-printing API does not work when runlist management is delegated to the GPU System Processor (GSP) (most Turing+ datacenter GPUs). 103- The runlist-printing API does not work when runlist management is delegated to the GPU System Processor (GSP) (most Turing+ datacenter GPUs).
96 To workaround, enable the `FALLBACK_TO_PRAMIN` define in `runlist.c`, or reload the `nvidia` kernel module with the `NVreg_EnableGpuFirmware=0` parameter setting. 104 To workaround, enable the `FALLBACK_TO_PRAMIN` define in `runlist.c`, or reload the `nvidia` kernel module with the `NVreg_EnableGpuFirmware=0` parameter setting.
97 (Eg. on A100: end all GPU-using processes, then `sudo rmmod nvidia_uvm nvidia; sudo modprobe nvidia NVreg_EnableGpuFirmware=0`.) 105 (Eg. on A100: end all GPU-using processes, then `sudo rmmod nvidia_drm nvidia_modeset nvidia_uvm nvidia; sudo modprobe nvidia NVreg_EnableGpuFirmware=0`.)
diff --git a/device_info_procfs.c b/device_info_procfs.c
index 4e4ab03..105e731 100644
--- a/device_info_procfs.c
+++ b/device_info_procfs.c
@@ -18,7 +18,7 @@ static ssize_t nvdebug_reg32_read(struct file *f, char __user *buf, size_t size,
18 return 0; 18 return 0;
19 19
20 if ((read = nvdebug_readl(g, (uintptr_t)pde_data(file_inode(f)))) == -1) 20 if ((read = nvdebug_readl(g, (uintptr_t)pde_data(file_inode(f)))) == -1)
21 return -EOPNOTSUPP; 21 return -EIO;
22 // 32 bit register will always take less than 16 characters to print 22 // 32 bit register will always take less than 16 characters to print
23 chars_written = scnprintf(out, 16, "%#0x\n", read); 23 chars_written = scnprintf(out, 16, "%#0x\n", read);
24 if (copy_to_user(buf, out, chars_written)) 24 if (copy_to_user(buf, out, chars_written))
@@ -32,12 +32,85 @@ struct file_operations nvdebug_read_reg32_file_ops = {
32 .llseek = default_llseek, 32 .llseek = default_llseek,
33}; 33};
34 34
35typedef union {
36 struct {
37 uint8_t partitioning_select:2;
38 uint8_t table_select:2;
39 uint32_t pad_1:12;
40 uint8_t veid_offset:6;
41 uint32_t pad_2:2;
42 uint8_t table_offset:6;
43 uint32_t pad_3:2;
44 };
45 uint32_t raw;
46} partition_ctl_t;
47
48static ssize_t nvdebug_read_part(struct file *f, char __user *buf, size_t size, loff_t *off) {
49 char out[12*64+2];
50 int i, chars_written = 0;
51 partition_ctl_t part_ctl;
52 struct nvdebug_state *g = &g_nvdebug_state[file2parentgpuidx(f)];
53 if (size < 16 || *off != 0)
54 return 0;
55 // 32 bit register will always take less than 16 characters to print
56 part_ctl.raw = nvdebug_readl(g, 0x00405b2c);
57 //part_ctl.partitioning_select = 0; // XXX XXX XXX Temp; 06/18/2024
58 //part_ctl.table_select = 3; // 3 == ???
59 //part_ctl.table_select = 2; // 2 == TBL_SEL_PARTITIONING_LMEM_BLK
60 part_ctl.table_select = 1; // 1 == TBL_SEL_PARTITIONING_ENABLE
61 //part_ctl.table_select = 0; // 0 == TBL_SEL_NONE
62 part_ctl.veid_offset = (uintptr_t)pde_data(file_inode(f)); // Range of [0, 0x3f], aka [0, 63]
63 for (i = 0; i < 64; i++) {
64 // Increment to next table offset in PARTITION_CTL
65 part_ctl.table_offset = i;
66 nvdebug_writel(g, 0x00405b2c, part_ctl.raw);
67 // Verify write applied to PARTITION_CTL
68 part_ctl.raw = nvdebug_readl(g, 0x00405b2c);
69 if (part_ctl.table_offset != i)
70 return -ENOTRECOVERABLE;
71 // Read PARTITION_DATA and print
72 // ---
73 // I get back 0x000000ff on Volta and 0x00000003 on Turing from
74 // PARTITION_DATA for all possible VEID_OFFSET, TBL_OFFSET, and TBL_SEL
75 // combinations.
76 // ---
77 // There's a 48-byte (12-word) gap after the address for PARTITION_DATA.
78 // Exploring this on Turing for TBL_SEL_PARTITIONING_ENABLE, VEID 1, 62, and
79 // 63, with CUDA_MPS_ACTIVE_THREAD_PERCENTAGE=5 for constant_cycles_kernel
80 // running under MPS:
81 // +0x0: 0x3
82 // +0x4: 0
83 // +0x8: 0x100
84 // +0xC: 0
85 // +0x10: 0xffffffff
86 // +0x14: 0
87 // +0x18: 0
88 // +0x1C: 0xffffffff
89 // +0x20: 0
90 // +0x24: 0xffffffff
91 // +0x28: 0xffffffff
92 // +0x2C: 0xffffffff
93 chars_written += scnprintf(out + chars_written, 12, "%#010x ", nvdebug_readl(g, 0x00405b30));
94 }
95 chars_written += scnprintf(out + chars_written, 2, "\n");
96 if (copy_to_user(buf, out, chars_written))
97 printk(KERN_WARNING "Unable to copy all data for %s\n", file_dentry(f)->d_name.name);
98 *off += chars_written;
99 return chars_written;
100}
101
102struct file_operations nvdebug_read_part_file_ops = {
103 .read = nvdebug_read_part,
104 .llseek = default_llseek,
105};
106
35static ssize_t nvdebug_reg_range_read(struct file *f, char __user *buf, size_t size, loff_t *off) { 107static ssize_t nvdebug_reg_range_read(struct file *f, char __user *buf, size_t size, loff_t *off) {
36 char out[12]; 108 char out[12];
37 int chars_written; 109 int chars_written;
38 uint32_t read, mask; 110 uint32_t read, mask;
39 struct nvdebug_state *g = &g_nvdebug_state[file2parentgpuidx(f)]; 111 struct nvdebug_state *g = &g_nvdebug_state[file2parentgpuidx(f)];
40 // See comment in nvdebug_entry.c to understand `union reg_range` 112 // `start_bit` is included, `stop_bit` is not, so to print lower eight bits
113 // from a register, use `start_bit = 0` and `stop_bit = 8`.
41 union reg_range range; 114 union reg_range range;
42 range.raw = (uintptr_t)pde_data(file_inode(f)); 115 range.raw = (uintptr_t)pde_data(file_inode(f));
43 116
@@ -47,7 +120,7 @@ static ssize_t nvdebug_reg_range_read(struct file *f, char __user *buf, size_t s
47 120
48 // Print bits `start_bit` to `stop_bit` from 32 bits at address `offset` 121 // Print bits `start_bit` to `stop_bit` from 32 bits at address `offset`
49 if ((read = nvdebug_readl(g, range.offset)) == -1) 122 if ((read = nvdebug_readl(g, range.offset)) == -1)
50 return -EOPNOTSUPP; 123 return -EIO;
51 // Setup `mask` used to throw out unused upper bits 124 // Setup `mask` used to throw out unused upper bits
52 mask = -1u >> (32 - range.stop_bit + range.start_bit); 125 mask = -1u >> (32 - range.stop_bit + range.start_bit);
53 // Throw out unused lower bits via a shift, apply the mask, and print 126 // Throw out unused lower bits via a shift, apply the mask, and print
diff --git a/mmu.c b/mmu.c
index ababef5..e2b9a91 100644
--- a/mmu.c
+++ b/mmu.c
@@ -1,9 +1,13 @@
1/* Copyright 2024 Joshua Bakita 1/* Copyright 2024 Joshua Bakita
2 * Helpers to deal with NVIDIA's MMU and associated page tables 2 * Helpers to deal with NVIDIA's MMU and associated page tables
3 */ 3 */
4#include <linux/dma-mapping.h> // dma_map_page() and dma_unmap_page()
4#include <linux/err.h> // ERR_PTR() etc. 5#include <linux/err.h> // ERR_PTR() etc.
6#include <linux/gfp.h> // alloc_pages()
5#include <linux/iommu.h> // iommu_get_domain_for_dev() and iommu_iova_to_phys() 7#include <linux/iommu.h> // iommu_get_domain_for_dev() and iommu_iova_to_phys()
6#include <linux/kernel.h> // Kernel types 8#include <linux/kernel.h> // Kernel types
9#include <linux/list.h> // struct list_head and associated functions
10#include <linux/mm.h> // put_page()
7 11
8#include "nvdebug.h" 12#include "nvdebug.h"
9 13
@@ -15,6 +19,11 @@ int g_verbose = 0;
15#define printk_debug if (g_verbose >= 2) printk 19#define printk_debug if (g_verbose >= 2) printk
16#define printk_info if (g_verbose >= 1) printk 20#define printk_info if (g_verbose >= 1) printk
17 21
22// At least map_page_directory() assumes that pages are 4 KiB
23#if PAGE_SIZE != 4096
24#error nvdebug assumes and requires a 4 KiB page size.
25#endif
26
18/* Convert a page directory (PD) pointer and aperture to be kernel-accessible 27/* Convert a page directory (PD) pointer and aperture to be kernel-accessible
19 28
20 I/O MMU handling inspired by amdgpu_iomem_read() in amdgpu_ttm.c of the 29 I/O MMU handling inspired by amdgpu_iomem_read() in amdgpu_ttm.c of the
@@ -22,7 +31,8 @@ int g_verbose = 0;
22 31
23 @param addr Pointer from page directory entry (PDE) 32 @param addr Pointer from page directory entry (PDE)
24 @param pd_ap PD-type aperture (target address space) for `addr` 33 @param pd_ap PD-type aperture (target address space) for `addr`
25 @return A dereferencable kernel address, or an ERR_PTR-wrapped error 34 @return A dereferencable kernel address, 0 if an I/O MMU is in use and has
35 no available mapping for the bus address, or an ERR_PTR-wrapped error
26 */ 36 */
27static void __iomem *pd_deref(struct nvdebug_state *g, uintptr_t addr,