diff options
| author | Joshua Bakita <bakitajoshua@gmail.com> | 2025-05-05 03:53:01 -0400 |
|---|---|---|
| committer | Joshua Bakita <bakitajoshua@gmail.com> | 2025-05-05 03:53:13 -0400 |
| commit | 293430fcb5d4013b573556c58457ee706e482b7f (patch) | |
| tree | 9328fa680f55b4e1a08d24714275b8437be3be5d | |
| parent | 494df296bf4abe9b2b484bde1a4fad28c989afec (diff) | |
Snapshot for ECRTS'25 artifact evaluation
| -rw-r--r-- | Makefile | 3 | ||||
| -rw-r--r-- | README.md | 10 | ||||
| -rw-r--r-- | device_info_procfs.c | 79 | ||||
| -rw-r--r-- | mmu.c | 414 | ||||
| -rw-r--r-- | nvdebug.h | 293 | ||||
| -rw-r--r-- | nvdebug_entry.c | 476 | ||||
| -rw-r--r-- | nvdebug_linux.h | 5 | ||||
| -rw-r--r-- | runlist.c | 275 | ||||
| -rw-r--r-- | runlist_procfs.c | 645 |
9 files changed, 2154 insertions, 46 deletions
| @@ -8,3 +8,6 @@ all: | |||
| 8 | make -C /lib/modules/$(shell uname -r)/build M=$(PWD) modules | 8 | make -C /lib/modules/$(shell uname -r)/build M=$(PWD) modules |
| 9 | clean: | 9 | clean: |
| 10 | make -C /lib/modules/$(shell uname -r)/build M=$(PWD) clean | 10 | make -C /lib/modules/$(shell uname -r)/build M=$(PWD) clean |
| 11 | |||
| 12 | nvdebug_user.so: runlist.c mmu.c bus.c nvdebug_user.c | ||
| 13 | gcc $< -shared -o $@ $(KBULID_CFLAGS) | ||
| @@ -59,6 +59,7 @@ Not all these TPCs will necessarially be enabled in every GPC. | |||
| 59 | Use `cat gpcX_tpc_mask` to get a bit mask of which TPCs are disabled for GPC X. | 59 | Use `cat gpcX_tpc_mask` to get a bit mask of which TPCs are disabled for GPC X. |
| 60 | A set bit indicates a disabled TPC. | 60 | A set bit indicates a disabled TPC. |
| 61 | This API is only available on enabled GPCs. | 61 | This API is only available on enabled GPCs. |
| 62 | Bits greater than the number of on-chip TPCs per GPC should be ignored (it may appear than non-existent TPCs are "disabled"). | ||
| 62 | 63 | ||
| 63 | Example usage: To get the number of on-chip SMs on Volta+ GPUs, multiply the return of `cat num_gpcs` with `cat num_tpc_per_gpc` and multiply by 2 (SMs per TPC). | 64 | Example usage: To get the number of on-chip SMs on Volta+ GPUs, multiply the return of `cat num_gpcs` with `cat num_tpc_per_gpc` and multiply by 2 (SMs per TPC). |
| 64 | 65 | ||
| @@ -83,6 +84,13 @@ Use `echo Z > runlistY/switch_to_tsg` to switch the GPU to run only the specifie | |||
| 83 | 84 | ||
| 84 | Use `echo Y > resubmit_runlist` to resubmit runlist Y (useful to prompt newer GPUs to pick up on re-enabled channels). | 85 | Use `echo Y > resubmit_runlist` to resubmit runlist Y (useful to prompt newer GPUs to pick up on re-enabled channels). |
| 85 | 86 | ||
| 87 | ## Error Interpretation | ||
| 88 | First check the kernel log to see if in includes more information about the error. | ||
| 89 | The following conventions are used for certain error codes: | ||
| 90 | |||
| 91 | - EIO, "Input/Output Error," is returned when an operation fails due to a bad register read. | ||
| 92 | - (Other errors may not have a consistent conventional meaning; see the implementation.) | ||
| 93 | |||
| 86 | ## General Codebase Structure | 94 | ## General Codebase Structure |
| 87 | - `nvdebug.h` defines and describes all GPU data structures. This does not depend on any kernel-internal headers. | 95 | - `nvdebug.h` defines and describes all GPU data structures. This does not depend on any kernel-internal headers. |
| 88 | - `nvdebug_entry.h` contains module startup, device detection, initialization, and module teardown logic. | 96 | - `nvdebug_entry.h` contains module startup, device detection, initialization, and module teardown logic. |
| @@ -94,4 +102,4 @@ Use `echo Y > resubmit_runlist` to resubmit runlist Y (useful to prompt newer GP | |||
| 94 | 102 | ||
| 95 | - The runlist-printing API does not work when runlist management is delegated to the GPU System Processor (GSP) (most Turing+ datacenter GPUs). | 103 | - The runlist-printing API does not work when runlist management is delegated to the GPU System Processor (GSP) (most Turing+ datacenter GPUs). |
| 96 | To workaround, enable the `FALLBACK_TO_PRAMIN` define in `runlist.c`, or reload the `nvidia` kernel module with the `NVreg_EnableGpuFirmware=0` parameter setting. | 104 | To workaround, enable the `FALLBACK_TO_PRAMIN` define in `runlist.c`, or reload the `nvidia` kernel module with the `NVreg_EnableGpuFirmware=0` parameter setting. |
| 97 | (Eg. on A100: end all GPU-using processes, then `sudo rmmod nvidia_uvm nvidia; sudo modprobe nvidia NVreg_EnableGpuFirmware=0`.) | 105 | (Eg. on A100: end all GPU-using processes, then `sudo rmmod nvidia_drm nvidia_modeset nvidia_uvm nvidia; sudo modprobe nvidia NVreg_EnableGpuFirmware=0`.) |
diff --git a/device_info_procfs.c b/device_info_procfs.c index 4e4ab03..105e731 100644 --- a/device_info_procfs.c +++ b/device_info_procfs.c | |||
| @@ -18,7 +18,7 @@ static ssize_t nvdebug_reg32_read(struct file *f, char __user *buf, size_t size, | |||
| 18 | return 0; | 18 | return 0; |
| 19 | 19 | ||
| 20 | if ((read = nvdebug_readl(g, (uintptr_t)pde_data(file_inode(f)))) == -1) | 20 | if ((read = nvdebug_readl(g, (uintptr_t)pde_data(file_inode(f)))) == -1) |
| 21 | return -EOPNOTSUPP; | 21 | return -EIO; |
| 22 | // 32 bit register will always take less than 16 characters to print | 22 | // 32 bit register will always take less than 16 characters to print |
| 23 | chars_written = scnprintf(out, 16, "%#0x\n", read); | 23 | chars_written = scnprintf(out, 16, "%#0x\n", read); |
| 24 | if (copy_to_user(buf, out, chars_written)) | 24 | if (copy_to_user(buf, out, chars_written)) |
| @@ -32,12 +32,85 @@ struct file_operations nvdebug_read_reg32_file_ops = { | |||
| 32 | .llseek = default_llseek, | 32 | .llseek = default_llseek, |
| 33 | }; | 33 | }; |
| 34 | 34 | ||
| 35 | typedef union { | ||
| 36 | struct { | ||
| 37 | uint8_t partitioning_select:2; | ||
| 38 | uint8_t table_select:2; | ||
| 39 | uint32_t pad_1:12; | ||
| 40 | uint8_t veid_offset:6; | ||
| 41 | uint32_t pad_2:2; | ||
| 42 | uint8_t table_offset:6; | ||
| 43 | uint32_t pad_3:2; | ||
| 44 | }; | ||
| 45 | uint32_t raw; | ||
| 46 | } partition_ctl_t; | ||
| 47 | |||
| 48 | static ssize_t nvdebug_read_part(struct file *f, char __user *buf, size_t size, loff_t *off) { | ||
| 49 | char out[12*64+2]; | ||
| 50 | int i, chars_written = 0; | ||
| 51 | partition_ctl_t part_ctl; | ||
| 52 | struct nvdebug_state *g = &g_nvdebug_state[file2parentgpuidx(f)]; | ||
| 53 | if (size < 16 || *off != 0) | ||
| 54 | return 0; | ||
| 55 | // 32 bit register will always take less than 16 characters to print | ||
| 56 | part_ctl.raw = nvdebug_readl(g, 0x00405b2c); | ||
| 57 | //part_ctl.partitioning_select = 0; // XXX XXX XXX Temp; 06/18/2024 | ||
| 58 | //part_ctl.table_select = 3; // 3 == ??? | ||
| 59 | //part_ctl.table_select = 2; // 2 == TBL_SEL_PARTITIONING_LMEM_BLK | ||
| 60 | part_ctl.table_select = 1; // 1 == TBL_SEL_PARTITIONING_ENABLE | ||
| 61 | //part_ctl.table_select = 0; // 0 == TBL_SEL_NONE | ||
| 62 | part_ctl.veid_offset = (uintptr_t)pde_data(file_inode(f)); // Range of [0, 0x3f], aka [0, 63] | ||
| 63 | for (i = 0; i < 64; i++) { | ||
| 64 | // Increment to next table offset in PARTITION_CTL | ||
| 65 | part_ctl.table_offset = i; | ||
| 66 | nvdebug_writel(g, 0x00405b2c, part_ctl.raw); | ||
| 67 | // Verify write applied to PARTITION_CTL | ||
| 68 | part_ctl.raw = nvdebug_readl(g, 0x00405b2c); | ||
| 69 | if (part_ctl.table_offset != i) | ||
| 70 | return -ENOTRECOVERABLE; | ||
| 71 | // Read PARTITION_DATA and print | ||
| 72 | // --- | ||
| 73 | // I get back 0x000000ff on Volta and 0x00000003 on Turing from | ||
| 74 | // PARTITION_DATA for all possible VEID_OFFSET, TBL_OFFSET, and TBL_SEL | ||
| 75 | // combinations. | ||
| 76 | // --- | ||
| 77 | // There's a 48-byte (12-word) gap after the address for PARTITION_DATA. | ||
| 78 | // Exploring this on Turing for TBL_SEL_PARTITIONING_ENABLE, VEID 1, 62, and | ||
| 79 | // 63, with CUDA_MPS_ACTIVE_THREAD_PERCENTAGE=5 for constant_cycles_kernel | ||
| 80 | // running under MPS: | ||
| 81 | // +0x0: 0x3 | ||
| 82 | // +0x4: 0 | ||
| 83 | // +0x8: 0x100 | ||
| 84 | // +0xC: 0 | ||
| 85 | // +0x10: 0xffffffff | ||
| 86 | // +0x14: 0 | ||
| 87 | // +0x18: 0 | ||
| 88 | // +0x1C: 0xffffffff | ||
| 89 | // +0x20: 0 | ||
| 90 | // +0x24: 0xffffffff | ||
| 91 | // +0x28: 0xffffffff | ||
| 92 | // +0x2C: 0xffffffff | ||
| 93 | chars_written += scnprintf(out + chars_written, 12, "%#010x ", nvdebug_readl(g, 0x00405b30)); | ||
| 94 | } | ||
| 95 | chars_written += scnprintf(out + chars_written, 2, "\n"); | ||
| 96 | if (copy_to_user(buf, out, chars_written)) | ||
| 97 | printk(KERN_WARNING "Unable to copy all data for %s\n", file_dentry(f)->d_name.name); | ||
| 98 | *off += chars_written; | ||
| 99 | return chars_written; | ||
| 100 | } | ||
| 101 | |||
| 102 | struct file_operations nvdebug_read_part_file_ops = { | ||
| 103 | .read = nvdebug_read_part, | ||
| 104 | .llseek = default_llseek, | ||
| 105 | }; | ||
| 106 | |||
| 35 | static ssize_t nvdebug_reg_range_read(struct file *f, char __user *buf, size_t size, loff_t *off) { | 107 | static ssize_t nvdebug_reg_range_read(struct file *f, char __user *buf, size_t size, loff_t *off) { |
| 36 | char out[12]; | 108 | char out[12]; |
| 37 | int chars_written; | 109 | int chars_written; |
| 38 | uint32_t read, mask; | 110 | uint32_t read, mask; |
| 39 | struct nvdebug_state *g = &g_nvdebug_state[file2parentgpuidx(f)]; | 111 | struct nvdebug_state *g = &g_nvdebug_state[file2parentgpuidx(f)]; |
| 40 | // See comment in nvdebug_entry.c to understand `union reg_range` | 112 | // `start_bit` is included, `stop_bit` is not, so to print lower eight bits |
| 113 | // from a register, use `start_bit = 0` and `stop_bit = 8`. | ||
| 41 | union reg_range range; | 114 | union reg_range range; |
| 42 | range.raw = (uintptr_t)pde_data(file_inode(f)); | 115 | range.raw = (uintptr_t)pde_data(file_inode(f)); |
| 43 | 116 | ||
| @@ -47,7 +120,7 @@ static ssize_t nvdebug_reg_range_read(struct file *f, char __user *buf, size_t s | |||
| 47 | 120 | ||
| 48 | // Print bits `start_bit` to `stop_bit` from 32 bits at address `offset` | 121 | // Print bits `start_bit` to `stop_bit` from 32 bits at address `offset` |
| 49 | if ((read = nvdebug_readl(g, range.offset)) == -1) | 122 | if ((read = nvdebug_readl(g, range.offset)) == -1) |
| 50 | return -EOPNOTSUPP; | 123 | return -EIO; |
| 51 | // Setup `mask` used to throw out unused upper bits | 124 | // Setup `mask` used to throw out unused upper bits |
| 52 | mask = -1u >> (32 - range.stop_bit + range.start_bit); | 125 | mask = -1u >> (32 - range.stop_bit + range.start_bit); |
| 53 | // Throw out unused lower bits via a shift, apply the mask, and print | 126 | // Throw out unused lower bits via a shift, apply the mask, and print |
| @@ -1,9 +1,13 @@ | |||
| 1 | /* Copyright 2024 Joshua Bakita | 1 | /* Copyright 2024 Joshua Bakita |
| 2 | * Helpers to deal with NVIDIA's MMU and associated page tables | 2 | * Helpers to deal with NVIDIA's MMU and associated page tables |
| 3 | */ | 3 | */ |
| 4 | #include <linux/dma-mapping.h> // dma_map_page() and dma_unmap_page() | ||
| 4 | #include <linux/err.h> // ERR_PTR() etc. | 5 | #include <linux/err.h> // ERR_PTR() etc. |
| 6 | #include <linux/gfp.h> // alloc_pages() | ||
| 5 | #include <linux/iommu.h> // iommu_get_domain_for_dev() and iommu_iova_to_phys() | 7 | #include <linux/iommu.h> // iommu_get_domain_for_dev() and iommu_iova_to_phys() |
| 6 | #include <linux/kernel.h> // Kernel types | 8 | #include <linux/kernel.h> // Kernel types |
| 9 | #include <linux/list.h> // struct list_head and associated functions | ||
| 10 | #include <linux/mm.h> // put_page() | ||
| 7 | 11 | ||
| 8 | #include "nvdebug.h" | 12 | #include "nvdebug.h" |
| 9 | 13 | ||
| @@ -15,6 +19,11 @@ int g_verbose = 0; | |||
| 15 | #define printk_debug if (g_verbose >= 2) printk | 19 | #define printk_debug if (g_verbose >= 2) printk |
| 16 | #define printk_info if (g_verbose >= 1) printk | 20 | #define printk_info if (g_verbose >= 1) printk |
| 17 | 21 | ||
| 22 | // At least map_page_directory() assumes that pages are 4 KiB | ||
| 23 | #if PAGE_SIZE != 4096 | ||
| 24 | #error nvdebug assumes and requires a 4 KiB page size. | ||
| 25 | #endif | ||
| 26 | |||
| 18 | /* Convert a page directory (PD) pointer and aperture to be kernel-accessible | 27 | /* Convert a page directory (PD) pointer and aperture to be kernel-accessible |
| 19 | 28 | ||
| 20 | I/O MMU handling inspired by amdgpu_iomem_read() in amdgpu_ttm.c of the | 29 | I/O MMU handling inspired by amdgpu_iomem_read() in amdgpu_ttm.c of the |
| @@ -22,7 +31,8 @@ int g_verbose = 0; | |||
| 22 | 31 | ||
| 23 | @param addr Pointer from page directory entry (PDE) | 32 | @param addr Pointer from page directory entry (PDE) |
| 24 | @param pd_ap PD-type aperture (target address space) for `addr` | 33 | @param pd_ap PD-type aperture (target address space) for `addr` |
| 25 | @return A dereferencable kernel address, or an ERR_PTR-wrapped error | 34 | @return A dereferencable kernel address, 0 if an I/O MMU is in use and has |
| 35 | no available mapping for the bus address, or an ERR_PTR-wrapped error | ||
| 26 | */ | 36 | */ |
| 27 | static void __iomem *pd_deref(struct nvdebug_state *g, uintptr_t addr, | ||
