From 293430fcb5d4013b573556c58457ee706e482b7f Mon Sep 17 00:00:00 2001 From: Joshua Bakita Date: Mon, 5 May 2025 03:53:01 -0400 Subject: Snapshot for ECRTS'25 artifact evaluation --- Makefile | 3 + README.md | 10 +- device_info_procfs.c | 79 ++++++- mmu.c | 414 ++++++++++++++++++++++++++++++++- nvdebug.h | 293 ++++++++++++++++++++++- nvdebug_entry.c | 476 +++++++++++++++++++++++++++++++++++-- nvdebug_linux.h | 5 + runlist.c | 275 +++++++++++++++++++++- runlist_procfs.c | 645 ++++++++++++++++++++++++++++++++++++++++++++++++++- 9 files changed, 2154 insertions(+), 46 deletions(-) diff --git a/Makefile b/Makefile index fea3819..9d6d374 100644 --- a/Makefile +++ b/Makefile @@ -8,3 +8,6 @@ all: make -C /lib/modules/$(shell uname -r)/build M=$(PWD) modules clean: make -C /lib/modules/$(shell uname -r)/build M=$(PWD) clean + +nvdebug_user.so: runlist.c mmu.c bus.c nvdebug_user.c + gcc $< -shared -o $@ $(KBULID_CFLAGS) diff --git a/README.md b/README.md index da3e5d7..2889b29 100644 --- a/README.md +++ b/README.md @@ -59,6 +59,7 @@ Not all these TPCs will necessarially be enabled in every GPC. Use `cat gpcX_tpc_mask` to get a bit mask of which TPCs are disabled for GPC X. A set bit indicates a disabled TPC. This API is only available on enabled GPCs. +Bits greater than the number of on-chip TPCs per GPC should be ignored (it may appear than non-existent TPCs are "disabled"). Example usage: To get the number of on-chip SMs on Volta+ GPUs, multiply the return of `cat num_gpcs` with `cat num_tpc_per_gpc` and multiply by 2 (SMs per TPC). @@ -83,6 +84,13 @@ Use `echo Z > runlistY/switch_to_tsg` to switch the GPU to run only the specifie Use `echo Y > resubmit_runlist` to resubmit runlist Y (useful to prompt newer GPUs to pick up on re-enabled channels). +## Error Interpretation +First check the kernel log to see if in includes more information about the error. +The following conventions are used for certain error codes: + +- EIO, "Input/Output Error," is returned when an operation fails due to a bad register read. +- (Other errors may not have a consistent conventional meaning; see the implementation.) + ## General Codebase Structure - `nvdebug.h` defines and describes all GPU data structures. This does not depend on any kernel-internal headers. - `nvdebug_entry.h` contains module startup, device detection, initialization, and module teardown logic. @@ -94,4 +102,4 @@ Use `echo Y > resubmit_runlist` to resubmit runlist Y (useful to prompt newer GP - The runlist-printing API does not work when runlist management is delegated to the GPU System Processor (GSP) (most Turing+ datacenter GPUs). To workaround, enable the `FALLBACK_TO_PRAMIN` define in `runlist.c`, or reload the `nvidia` kernel module with the `NVreg_EnableGpuFirmware=0` parameter setting. - (Eg. on A100: end all GPU-using processes, then `sudo rmmod nvidia_uvm nvidia; sudo modprobe nvidia NVreg_EnableGpuFirmware=0`.) + (Eg. on A100: end all GPU-using processes, then `sudo rmmod nvidia_drm nvidia_modeset nvidia_uvm nvidia; sudo modprobe nvidia NVreg_EnableGpuFirmware=0`.) diff --git a/device_info_procfs.c b/device_info_procfs.c index 4e4ab03..105e731 100644 --- a/device_info_procfs.c +++ b/device_info_procfs.c @@ -18,7 +18,7 @@ static ssize_t nvdebug_reg32_read(struct file *f, char __user *buf, size_t size, return 0; if ((read = nvdebug_readl(g, (uintptr_t)pde_data(file_inode(f)))) == -1) - return -EOPNOTSUPP; + return -EIO; // 32 bit register will always take less than 16 characters to print chars_written = scnprintf(out, 16, "%#0x\n", read); if (copy_to_user(buf, out, chars_written)) @@ -32,12 +32,85 @@ struct file_operations nvdebug_read_reg32_file_ops = { .llseek = default_llseek, }; +typedef union { + struct { + uint8_t partitioning_select:2; + uint8_t table_select:2; + uint32_t pad_1:12; + uint8_t veid_offset:6; + uint32_t pad_2:2; + uint8_t table_offset:6; + uint32_t pad_3:2; + }; + uint32_t raw; +} partition_ctl_t; + +static ssize_t nvdebug_read_part(struct file *f, char __user *buf, size_t size, loff_t *off) { + char out[12*64+2]; + int i, chars_written = 0; + partition_ctl_t part_ctl; + struct nvdebug_state *g = &g_nvdebug_state[file2parentgpuidx(f)]; + if (size < 16 || *off != 0) + return 0; + // 32 bit register will always take less than 16 characters to print + part_ctl.raw = nvdebug_readl(g, 0x00405b2c); + //part_ctl.partitioning_select = 0; // XXX XXX XXX Temp; 06/18/2024 + //part_ctl.table_select = 3; // 3 == ??? + //part_ctl.table_select = 2; // 2 == TBL_SEL_PARTITIONING_LMEM_BLK + part_ctl.table_select = 1; // 1 == TBL_SEL_PARTITIONING_ENABLE + //part_ctl.table_select = 0; // 0 == TBL_SEL_NONE + part_ctl.veid_offset = (uintptr_t)pde_data(file_inode(f)); // Range of [0, 0x3f], aka [0, 63] + for (i = 0; i < 64; i++) { + // Increment to next table offset in PARTITION_CTL + part_ctl.table_offset = i; + nvdebug_writel(g, 0x00405b2c, part_ctl.raw); + // Verify write applied to PARTITION_CTL + part_ctl.raw = nvdebug_readl(g, 0x00405b2c); + if (part_ctl.table_offset != i) + return -ENOTRECOVERABLE; + // Read PARTITION_DATA and print + // --- + // I get back 0x000000ff on Volta and 0x00000003 on Turing from + // PARTITION_DATA for all possible VEID_OFFSET, TBL_OFFSET, and TBL_SEL + // combinations. + // --- + // There's a 48-byte (12-word) gap after the address for PARTITION_DATA. + // Exploring this on Turing for TBL_SEL_PARTITIONING_ENABLE, VEID 1, 62, and + // 63, with CUDA_MPS_ACTIVE_THREAD_PERCENTAGE=5 for constant_cycles_kernel + // running under MPS: + // +0x0: 0x3 + // +0x4: 0 + // +0x8: 0x100 + // +0xC: 0 + // +0x10: 0xffffffff + // +0x14: 0 + // +0x18: 0 + // +0x1C: 0xffffffff + // +0x20: 0 + // +0x24: 0xffffffff + // +0x28: 0xffffffff + // +0x2C: 0xffffffff + chars_written += scnprintf(out + chars_written, 12, "%#010x ", nvdebug_readl(g, 0x00405b30)); + } + chars_written += scnprintf(out + chars_written, 2, "\n"); + if (copy_to_user(buf, out, chars_written)) + printk(KERN_WARNING "Unable to copy all data for %s\n", file_dentry(f)->d_name.name); + *off += chars_written; + return chars_written; +} + +struct file_operations nvdebug_read_part_file_ops = { + .read = nvdebug_read_part, + .llseek = default_llseek, +}; + static ssize_t nvdebug_reg_range_read(struct file *f, char __user *buf, size_t size, loff_t *off) { char out[12]; int chars_written; uint32_t read, mask; struct nvdebug_state *g = &g_nvdebug_state[file2parentgpuidx(f)]; - // See comment in nvdebug_entry.c to understand `union reg_range` + // `start_bit` is included, `stop_bit` is not, so to print lower eight bits + // from a register, use `start_bit = 0` and `stop_bit = 8`. union reg_range range; range.raw = (uintptr_t)pde_data(file_inode(f)); @@ -47,7 +120,7 @@ static ssize_t nvdebug_reg_range_read(struct file *f, char __user *buf, size_t s // Print bits `start_bit` to `stop_bit` from 32 bits at address `offset` if ((read = nvdebug_readl(g, range.offset)) == -1) - return -EOPNOTSUPP; + return -EIO; // Setup `mask` used to throw out unused upper bits mask = -1u >> (32 - range.stop_bit + range.start_bit); // Throw out unused lower bits via a shift, apply the mask, and print diff --git a/mmu.c b/mmu.c index ababef5..e2b9a91 100644 --- a/mmu.c +++ b/mmu.c @@ -1,9 +1,13 @@ /* Copyright 2024 Joshua Bakita * Helpers to deal with NVIDIA's MMU and associated page tables */ +#include // dma_map_page() and dma_unmap_page() #include // ERR_PTR() etc. +#include // alloc_pages() #include // iommu_get_domain_for_dev() and iommu_iova_to_phys() #include // Kernel types +#include // struct list_head and associated functions +#include // put_page() #include "nvdebug.h" @@ -15,6 +19,11 @@ int g_verbose = 0; #define printk_debug if (g_verbose >= 2) printk #define printk_info if (g_verbose >= 1) printk +// At least map_page_directory() assumes that pages are 4 KiB +#if PAGE_SIZE != 4096 +#error nvdebug assumes and requires a 4 KiB page size. +#endif + /* Convert a page directory (PD) pointer and aperture to be kernel-accessible I/O MMU handling inspired by amdgpu_iomem_read() in amdgpu_ttm.c of the @@ -22,7 +31,8 @@ int g_verbose = 0; @param addr Pointer from page directory entry (PDE) @param pd_ap PD-type aperture (target address space) for `addr` - @return A dereferencable kernel address, or an ERR_PTR-wrapped error + @return A dereferencable kernel address, 0 if an I/O MMU is in use and has + no available mapping for the bus address, or an ERR_PTR-wrapped error */ static void __iomem *pd_deref(struct nvdebug_state *g, uintptr_t addr, enum PD_TARGET pd_ap) { @@ -56,7 +66,7 @@ static void __iomem *pd_deref(struct nvdebug_state *g, uintptr_t addr, // Check for, and translate through, the I/O MMU (if any) if ((dom = iommu_get_domain_for_dev(g->dev))) { phys = iommu_iova_to_phys(dom, addr); - printk_debug(KERN_DEBUG "[nvdebug] I/O MMU translated SYS_MEM I/O VA %#lx to physical address %#llx.\n", addr, phys); + printk_debug(KERN_DEBUG "[nvdebug] %s: I/O MMU translated SYS_MEM I/O VA %#lx to physical address %#llx.\n", __func__, addr, phys); } else phys = addr; @@ -143,6 +153,327 @@ uint64_t search_page_directory(struct nvdebug_state *g, return 0; } +/* GPU Virtual address -> Physical address ("forward" translation) for V2 tables + Index the page directories and tables used by the GPU MMU to determine which + physical address a given GPU virtual address has been mapped to. + + The page directory and tables may be located in VID_MEM, SYS_MEM, or spread + across multiple apertures. + + @param pd_config Page Directory configuration, containing pointer and + aperture for the start of the PDE3 entries + @param addr_to_find Virtual address to translate to a physical address + @param found_addr Where to store found physical address (0 if unfound) + @param found_aperture Where to store aperture of found physical address + @return 0 on success, -ENXIO if not found, and -errno on error. +*/ +int translate_page_directory(struct nvdebug_state *g, + page_dir_config_t pd_config, + uint64_t addr_to_find, + uint64_t *found_addr /* out */, + enum INST_TARGET *found_aperture /* out */) { + page_dir_entry_t entry; + void __iomem *next_kva; + unsigned int level, pde_idx; + uintptr_t next = (uintptr_t)pd_config.page_dir << 12; + enum PD_TARGET next_target = INST2PD_TARGET(pd_config.target); + + *found_addr = 0; + *found_aperture = TARGET_INVALID; + + // Make sure that the query is page-aligned (likely mistake otherwise) + if (addr_to_find & 0xfff) { + printk(KERN_WARNING "[nvdebug] Attempting to translate unaligned address %#llx in translate_page_directory()!\n", addr_to_find); + return -EINVAL; + } + + printk_info(KERN_INFO "[nvdebug] Translating addr %#018llx in V2 page table with base %#018llx\n", (u64)addr_to_find, (u64)next); + + // Step through each PDE level and the PTE level + for (level = 0; level < 5; level++) { + // Index into this level + pde_idx = (addr_to_find >> NV_MMU_PT_V2_LSB[level]) & (NV_MMU_PT_V2_SZ[level] - 1); + printk_debug(KERN_DEBUG "[nvdebug] Using index %u in lvl %d\n", pde_idx, level); + // Hack to workaround PDE0 being double-size and strangely formatted + if (NV_MMU_PT_V2_ENTRY_SZ[level] == 16) + next += 8; + // Obtain a kernel-dereferencable address + next_kva = pd_deref(g, next, next_target); + if (IS_ERR_OR_NULL(next_kva)) { + printk(KERN_ERR "[nvdebug] %s: Unable to resolve %#lx in GPU %s to a kernel-accessible address. Error %ld.\n", __func__, next, pd_target_to_text(next_target), PTR_ERR(next_kva)); + return PTR_ERR(next_kva); + } + // Obtain entry at this level + entry.raw_w = readq(next_kva + NV_MMU_PT_V2_ENTRY_SZ[level] * pde_idx); + if (entry.target == PD_AND_TARGET_INVALID) + return -ENXIO; + printk_debug(KERN_DEBUG "[nvdebug] Found %s pointing to %#018llx in ap '%s' at lvl %d (raw: %#018llx)\n", entry.is_pte ? "PTE" : "PDE", ((u64)entry.addr) << 12, pd_target_to_text(entry.target), level, entry.raw_w); + // Just return the physical address if this is the PTE level + if (entry.is_pte) { // level == 4 for 4 KiB pages, == 3 for 2 MiB + *found_addr = ((uint64_t)entry.addr) << 12; + *found_aperture = entry.aperture; + return 0; + } + // Otherwise step to the next table level + // TODO: Use addr_w as appropriate + next = (uint64_t)entry.addr << 12; + next_target = entry.target; + } + + return 0; +} + +// This struct is very special. We will never directly allocate this struct; +// its sole purpose is to provide more intuitive names to the offsets at which +// we store data in Linux's struct page. Such (ab)use of struct page is +// explictly permitted (see linux/mm_types.h). This struct is thus used by +// casting a pointer of struct page to a pointer of struct nvdebug_pd_page, +// then accessing the associated fields. This pointer may also be freely cast +// back to a sturct page pointer. +// We have 24 (32-bit) or 44 (64-bit) bytes available in the page struct +// (according to the documentation on struct page). Our comments indicate what +// available parts of struct page we repurpose for our own needs. +struct nvdebug_pd_page { + unsigned long __flags; // From struct page; do not touch! + // Overlaps struct page.lru + struct list_head list; // 4/8 bytes + // Overlaps struct page.mapping (and page.share on 32-bit) + uintptr_t parent_addr; // 8 bytes + // Overlaps struct page.share (page.private on 32-bit) + enum PD_TARGET parent_aperture; // 4 bytes + // Overlaps page.private (page.page_type on 32-bit) + dma_addr_t dma_addr; // 4/8 bytes +}; + +/* Collect and free any now-unused page directory/table allocations + + @param force Deallocate all page directories/tables created by this module, + no matter if they appear to be in-use or not. + @returns Number of freed pages on success, -errno on error. +*/ +int gc_page_directory(struct nvdebug_state *g, bool force) { + struct nvdebug_pd_page *page, *_page; + void __iomem *parent_kva; + page_dir_entry_t parent_entry; + int freed_pages = 0; + + // Depth-first traversal (from perspective of each page table) of page + // allocations. + // (This is depth-first because map_page_directory() always allocates and + // pushes page directory allocations before page table allocations.) + list_for_each_entry_safe_reverse(page, _page, &g->pd_allocs, list) { + printk_debug(KERN_DEBUG "[nvdebug] %s: Checking if page directory/table at %llx (SYS_MEM_?) with parent at %lx (%s) is unused...\n", __func__, page->dma_addr, page->parent_addr, pd_target_to_text(page->parent_aperture)); + // Try to determine if we're still in-use. We consider ourselves + // potentially in-use if our parent still points to us. + parent_kva = pd_deref(g, page->parent_addr, page->parent_aperture); + if (IS_ERR(parent_kva)) { + printk(KERN_ERR "[nvdebug] %s: Error resolving %#lx in GPU %s to a kernel-accessible address. Error %ld.\n", __func__, page->parent_addr, pd_target_to_text(page->parent_aperture), PTR_ERR(parent_kva)); + return -ENOTRECOVERABLE; + } + // A NULL kva indicates parent no longer exists + parent_entry.raw_w = parent_kva ? readq(parent_kva) : 0; + // Page directory/table still in-use; do not free unless forced + if (parent_entry.addr_w == (page->dma_addr >> 12) && !force) + continue; + // Free this page table/directory and delete our parent's pointer to us + if (parent_entry.addr_w == (page->dma_addr >> 12)) { + printk(KERN_WARNING "[nvdebug] %s: Deleting page table/directory at %llx (SYS_MEM_?) with parent at %lx (%s) that may still be in-use!\n", __func__, page->dma_addr, page->parent_addr, pd_target_to_text(page->parent_aperture)); + writeq(0, parent_kva); + } + // Unmap, zero, free, and remove from tracking (these all return void) + dma_unmap_page(g->dev, page->dma_addr, PAGE_SIZE, DMA_TO_DEVICE); + memset(page_to_virt((struct page*)page), 0, PAGE_SIZE); + // Necessary to reset mapcount as we (ab)use its state for other things + page_mapcount_reset((struct page*)page); + // Same reset needed for mapping + ((struct page*)page)->mapping = NULL; + // Remove this page from our list of allocated pages + list_del(&page->list); + // Free the page + put_page((struct page*)page); + freed_pages++; + } + printk_debug(KERN_DEBUG "[nvdebug] %s: Freed %d pages.", __func__, freed_pages); + return freed_pages; +} + +/* Map a GPU virtual address to a physical address in a GPU page table + Search for a mapping for specified GPU virtual address, and create a new one + if none is found. Automatically creates page directories and page table + entries as necessary. + + The page directory and tables may be located in VID_MEM, SYS_MEM, or spread + across multiple apertures. + + @param pd_config Page Directory configuration, containing pointer and + aperture for the start of the PDE3 entries + @param vaddr_to_find Virtual address to check, and map to a physical address + if nothing is already mapped (up to 49 bits long) + @param paddr_to_map Physical address to use (up to 36 bits long if VID_MEM, + and up to 58 bits if SYS_MEM) + @param paddr_target Which space does the physical address refer to? + @param huge_page Set to map a 2 MiB, rather than 4 KiB, page + @return 0 on success, 1 if mapping already exists, -EADDRINUSE if virtual + address is already mapped to something else, and -errno on error +*/ +int map_page_directory(struct nvdebug_state *g, + page_dir_config_t pd_config, + uint64_t vaddr_to_find, + uint64_t paddr_to_map, + enum INST_TARGET paddr_target, + bool huge_page) { + page_dir_entry_t entry; + void __iomem *next_kva; + unsigned int level, pde_idx; + uintptr_t next = (uintptr_t)pd_config.page_dir << 12; + enum PD_TARGET next_target = INST2PD_TARGET(pd_config.target); + + // Make sure that the query is page-aligned (likely mistake otherwise) + if ((vaddr_to_find & 0xfff || paddr_to_map & 0xfff) + || (huge_page && (vaddr_to_find & 0x1fffff || paddr_to_map & 0x1fffff))) { + printk(KERN_WARNING "[nvdebug] %s: Attempting to map an unaligned address (physical %#018llx or virtual %#018llx)! Failing...\n", __func__, paddr_to_map, vaddr_to_find); + return -EINVAL; + } + + // NVIDIA supports up to 49-bit virtual addresss + // Except Jetson Xavier only seems to be able to resolve 47-bit addresses? + if (vaddr_to_find >> 49) { + printk(KERN_WARNING "[nvdebug] %s: vaddr_to_find (%#018llx) is beyond the 49-bit virtual address space supported by the GPU! Failing...\n", __func__, vaddr_to_find); + return -EINVAL; + } + + // NVIDIA supports up to 36-bit VID_MEM addresses + if (paddr_target == TARGET_VID_MEM && paddr_to_map >> 36) { + printk(KERN_WARNING "[nvdebug] %s: paddr_to_map (%#018llx) is beyond the 36-bit VID_MEM address space! Failing...\n", __func__, paddr_to_map); + return -EINVAL; + } + + // NVIDIA supports up to 58-bit SYS_MEM addresses + if ((paddr_target == TARGET_SYS_MEM_COHERENT || + paddr_target == TARGET_SYS_MEM_NONCOHERENT) && paddr_to_map >> 58) { + printk(KERN_WARNING "[nvdebug] %s: paddr_to_map (%#018llx) is beyond the 58-bit SYS_MEM address space! Failing...\n", __func__, paddr_to_map); + return -EINVAL; + } + + // We don't support mapping to PEERs; that requires a PEER ID + if (paddr_target == TARGET_PEER) { + printk(KERN_WARNING "[nvdebug] %s: paddr_target must be SYS_MEM_* or VID_MEM! Failing...\n", __func__); + return -EINVAL; + } + + printk_info(KERN_INFO "[nvdebug] Mapping addr %#018llx in page table with base %#018llx to %s address %#018llx\n", vaddr_to_find, (u64)next, target_to_text(paddr_target), paddr_to_map); + + // Step through each PDE level and the PTE level + for (level = 0; level < 5; level++) { + // Index into this level + pde_idx = (vaddr_to_find >> NV_MMU_PT_V2_LSB[level]) & (NV_MMU_PT_V2_SZ[level] - 1); + printk_debug(KERN_DEBUG "[nvdebug] In table at KVA %#lx, using index %u in lvl %d\n", (uintptr_t)next, pde_idx, level); + // Hack to workaround PDE0 being double-size and strangely formatted + if (NV_MMU_PT_V2_ENTRY_SZ[level] == 16) + next += 8; + // Obtain a kernel-dereferencable address + next_kva = pd_deref(g, next, next_target); + if (IS_ERR_OR_NULL(next_kva)) { + printk(KERN_ERR "[nvdebug] %s: Unable to resolve %#lx in GPU %s to a kernel-accessible address. Error %ld.\n", __func__, next, pd_target_to_text(next_target), PTR_ERR(next_kva)); + return -ENOTRECOVERABLE; + } + // Obtain entry at this level + entry.raw_w = readq(next_kva + NV_MMU_PT_V2_ENTRY_SZ[level] * pde_idx); + // If pointer to next level of the table does not exist + if (entry.target == PD_AND_TARGET_INVALID) { // PTE or PD covered by PD_AND_TARGET_INVALID + if (level == 4 || (huge_page && level == 3)) { + // Create new PTE (allocation, as needed, is handled at level 2 or 3) + // Targets observed in page tables: + // For PCIe: entry.target == PTE_AND_TARGET_VID_MEM; + // For Jetson: entry.target == PTE_AND_TARGET_SYS_MEM_NONCOHERENT; + entry.is_pte = 1; + entry.aperture = paddr_target; + if (paddr_target == TARGET_VID_MEM) + entry.addr = paddr_to_map >> 12; + else + entry.addr_w = paddr_to_map >> 12; + // Set the volatile bit (as NVRM does for SYS_MEM_COHERENT mappings) + // (This does nothing if the target is VID_MEM, but if the target is + // SYS_MEM_*, accesses will bypass the L2.) + entry.is_volatile = 1; + // Leave other fields zero, yielding an unencrypted, unprivileged, r/w, + // volatile mapping with atomics enabled. + + // XXX: Hack to work around PDE0 double-size weirdness. Huge + // page mapping will fault without this. + if (level == 3) + writeq(entry.raw_w, next_kva - 8 + NV_MMU_PT_V2_ENTRY_SZ[level] * pde_idx); + } else { + struct page* page_dir; + struct nvdebug_pd_page* page_dir_reinterpret; + dma_addr_t page_dir_dma; + // Allocate one 4 KiB all-zero (all invalid) page directory/ + // table at the next level + if (!(page_dir = alloc_pages(GFP_KERNEL | __GFP_ZERO, 0))) + return -ENOMEM; + // Obtain a GPU-accessible/bus address for this page (handling + // I/O MMU mappings, etc.) + page_dir_dma = dma_map_page(g->dev, page_dir, 0, PAGE_SIZE, DMA_TO_DEVICE); + // Verify that we were able to create a mapping + if (dma_mapping_error(g->dev, page_dir_dma)) + return dma_mapping_error(g->dev, page_dir_dma); + // Record this allocation for freeing later + // Note: Linux maintains a page struct for every page in the + // system. This struct has available space that drivers + // can use to store their own tracking information. Our + // struct nvdebug_pd_page facilitates this. + page_dir_reinterpret = (struct nvdebug_pd_page*)page_dir; + page_dir_reinterpret->parent_addr = next + NV_MMU_PT_V2_ENTRY_SZ[level] * pde_idx; + page_dir_reinterpret->parent_aperture = next_target; + page_dir_reinterpret->dma_addr = page_dir_dma; + list_add(&page_dir_reinterpret->list, &g->pd_allocs); + // Point this entry to the new directory/table + entry.target = PD_AND_TARGET_SYS_MEM_COHERENT; // Observed in page tables + // Must use addr_w with SYS_MEM targets + entry.addr_w = page_dir_dma >> 12; + // On Jetson and NVRM, all PDEs are marked volatile + entry.is_volatile = 1; + // We don't configure ATS, so disable ATS lookups for speed. + entry.no_ats = 1; + } + writeq(entry.raw_w, next_kva + NV_MMU_PT_V2_ENTRY_SZ[level] * pde_idx); + printk_debug(KERN_DEBUG "[nvdebug] Created %s pointing to %llx in ap '%s' at lvl %d (raw: %#018llx)\n", entry.is_pte ? "PTE" : "PDE", ((u64)entry.addr) << 12, pd_target_to_text(entry.target), level, entry.raw_w); + // Successfully created the requested PTE, so return + if (entry.is_pte) + return 0; + } else { + printk_debug(KERN_DEBUG "[nvdebug] Found %s pointing to %llx in ap '%s' at lvl %d (raw: %#018llx)\n", entry.is_pte ? "PTE" : "PDE", ((u64)entry.addr) << 12, pd_target_to_text(entry.target), level, entry.raw_w); + } + + // If this is the PTE level, return success if the address and target are correct + if (entry.is_pte) { // level == 4 for 4 KiB pages, == 3 for 2 MiB + if (entry.aperture != paddr_target) + return -EADDRINUSE; // Also handles PEER + if (entry.aperture == TARGET_VID_MEM) + return (uint64_t)entry.addr == paddr_to_map >> 12 ? 1 : -EADDRINUSE; + else + return entry.addr_w == paddr_to_map >> 12 ? 1 : -EADDRINUSE; // SYS_MEM is wider + } + + // If mapping a 2 MiB page and we made it here, level 3 had a PDE. This + // means that the requested 2 MiB virtual region already has one or more + // small pages mapped within it---a.k.a., the addresses are in use. + // If we didn't bail out here, the above logic would attempt to fallback + // to a 4 KiB mapping, which would be unexpected behavior. + if (huge_page && level == 3) + return -EADDRINUSE; + + // Otherwise step to the next table level + if (entry.aperture == TARGET_VID_MEM) + next = (uint64_t)entry.addr << 12; + else + next = (uint64_t)entry.addr_w << 12; // SYS_MEM is wider + next_target = entry.target; + } + + return -ENOTRECOVERABLE; // Should be impossible +} + /* GPU Physical address -> Virtual address ("reverse" translation) for V1 tables (See `search_page_directory()` for documentation.) */ @@ -187,7 +518,7 @@ uint64_t search_v1_page_directory(struct nvdebug_state *g, // Verify PDE is present if (pde.target == PD_TARGET_INVALID && pde.alt_target == PD_TARGET_INVALID) continue; -// printk(KERN_INFO "[nvdebug] Found %s PDE pointing to PTEs @ %llx in ap '%d' (raw: %llx)\n", pde.is_volatile ? "volatile" : "non-volatile", ((u64)pde.addr) << 12, pde.target, pde.raw); + // TODO: Handle huge pages printk_debug(KERN_DEBUG "[nvdebug] Found %s PDE at index %lld pointing to PTEs @ %#018llx in ap '%d' (raw: %#018llx)\n", pde.alt_is_volatile ? "volatile" : "non-volatile", i, ((u64)pde.alt_addr) << 12, pde.alt_target, pde.raw); // For each PTE for (j = 0; j < NV_MMU_PT_V1_SZ[1]; j++) { @@ -215,7 +546,84 @@ uint64_t search_v1_page_directory(struct nvdebug_state *g, return 0; } +/* GPU Virtual address -> Physical address ("forward" translation) for V1 tables + (See `translate_page_directory()` for documentation.) +*/ +int translate_v1_page_directory(struct nvdebug_state *g, + page_dir_config_t pd_config, + uint64_t addr_to_find, + uint64_t *found_addr /* out */, + enum INST_TARGET *found_aperture /* out */) { + page_dir_entry_v1_t pde; + page_tbl_entry_v1_t pte; + uintptr_t pde_idx, pde_phys, pte_idx, pte_phys; + void __iomem *pte_kva, *pde_kva; + + *found_addr = 0; + *found_aperture = TARGET_INVALID; + + // Make sure that the query is page-aligned (likely mistake otherwise) + if (addr_to_find & 0xfff) { + printk(KERN_WARNING "[nvdebug] Attempting to translate unaligned address %#llx in translate_v1_page_directory()!\n", addr_to_find); + return -EINVAL; + } + + // This function only understands the Page Table Version 1 format + if (pd_config.is_ver2) { + printk(KERN_ERR "[nvdebug] Passed a Version 2 page table at %#018llx to translate_v1_page_directory()!\n", (uint64_t)pd_config.page_dir << 12); + return -EINVAL; + } + + // We only understand the Version 1 format when 128 KiB huge pages are in-use + if (pd_config.is_64k_big_page) { + printk(KERN_ERR "[nvdebug] Page Table Version 1 with 64 KiB huge pages is unsupported!\n"); + return -EINVAL; + } + + printk_info(KERN_INFO "[nvdebug] Translating addr %#018llx in V1 page table with base %#018llx\n", (uint64_t)addr_to_find, (uint64_t)pd_config.page_dir << 12); + + // Shift bits which define PDE index to start at bit 0, and mask other bits + pde_idx = (addr_to_find >> NV_MMU_PT_V1_LSB[0]) & (NV_MMU_PT_V1_SZ[0] - 1); + // Compute VID_MEM/SYS_MEM address of page directory entry + pde_phys = ((uint64_t)pd_config.page_dir << 12) + pde_idx * sizeof(page_dir_entry_v1_t); + // Convert VID_MEM/SYS_MEM address to Kernel-accessible Virtual Address (KVA) + pde_kva = pd_deref(g, pde_phys, INST2PD_TARGET(pd_config.target)); + if (IS_ERR_OR_NULL(pde_kva)) { + printk(KERN_ERR "[nvdebug] %s: Unable to resolve %#lx in GPU %s to a kernel-accessible address. Error %ld.\n", __func__, pde_phys, target_to_text(pd_config.target), PTR_ERR(pde_kva)); + return PTR_ERR(pde_kva); + } + // Read page directory entry (readq seems to work fine; tested on GM204) + pde.raw = readq(pde_kva); + // Verify this PDE points to an array of page table entries + if (pde.target == PD_TARGET_INVALID && pde.alt_target == PD_TARGET_INVALID) + return -ENXIO; + // TODO: Check for and handle huge pages + printk_debug(KERN_DEBUG "[nvdebug] Found %s PDE pointing to PTEs @ %llx in ap '%d' (raw: %llx)\n", pde.alt_is_volatile ? "volatile" : "non-volatile", ((u64)pde.alt_addr) << 12, pde.alt_target, pde.raw); + + // Shift bits which define PTE index to start at bit 0, and mask other bits + pte_idx = (addr_to_find >> NV_MMU_PT_V1_LSB[1]) & (NV_MMU_PT_V1_SZ[1] - 1); + // Compute VID_MEM/SYS_MEM address of page table entry + pte_phys = ((uint64_t)pde.alt_addr << 12) + pte_idx * sizeof(page_tbl_entry_v1_t); + // Convert VID_MEM/SYS_MEM address to Kernel-accessible Virtual Address (KVA) + pte_kva = pd_deref(g, pte_phys, V12PD_TARGET(pde.alt_target)); + if (IS_ERR_OR_NULL(pde_kva)) { + printk(KERN_ERR "[nvdebug] %s: Unable to resolve %#lx in GPU %s to a kernel-accessible address. Error %ld.\n", __func__, pte_phys, pd_target_to_text(pde.alt_target), PTR_ERR(pte_kva)); + return PTR_ERR(pte_kva); + } + // Read page table entry + pte.raw = readq(pte_kva); + // XXX: The above readq() is bogus on gk104 (returns -1). Potential issue of pd_deref's move of PRAMIN racing with the driver? + if (!pte.is_present) + return -ENXIO; + printk_debug(KERN_DEBUG "[nvdebug] PTE for phy addr %#018llx, ap '%s', vol '%d', priv '%d', ro '%d', no_atomics '%d' (raw: %#018llx)\n", ((u64)pte.addr) << 12, target_to_text(pte.target), pte.is_volatile, pte.is_privileged, pte.is_readonly, pte.atomics_disabled, pte.raw); + // Access PTE and return physical address + *found_addr = (uint64_t)pte.addr << 12; + *found_aperture = pte.target; + return 0; +} + /* *** UNTESTED *** +// This is only relevant on pre-Kepler GPUs; not a current priority #define NV_MMU_PT_V0_SZ 2048 #define NV_MMU_PT_V0_LSB 29 uint64_t search_v0_page_directory(struct nvdebug_state *g, diff --git a/nvdebug.h b/nvdebug.h index ca0f514..3ac8db4 100644 --- a/nvdebug.h +++ b/nvdebug.h @@ -2,6 +2,7 @@ * SPDX-License-Identifier: MIT * * File outline: + * - Configuration options * - Runlist, preemption, and channel control (FIFO) * - Basic GPU information (MC) * - Detailed GPU information (PTOP, FUSE, and CE) @@ -20,6 +21,27 @@ // this, so declare as incomplete type to avoid pulling in the nvgpu headers. struct gk20a; +// Uncomment to, upon BAR2 access failure, return a PRAMIN-based runlist pointer +// in get_runlist_iter(). In order for this pointer to remain valid, PRAMIN +// **must** not be moved during runlist traversal. +// - The Jetson TX2 has no BAR2, and stores the runlist in VID_MEM, so this +// must be enabled to print the runlist on the TX2. +// - On the A100 in Google Cloud and H100 in Paperspace, as of Aug 2024, this is +// needed, as nvdebug is not finding (at least) runlist0 mapped in BAR2/3. +// Automatically disables printing Instance Block and Context State while +// traversing the runlist, as these require conflicting uses of PRAMIN (it's +// needed to search the page tables for the Instance Block in BAR2/3, and to +// access anything in the Context State---aka CTXSW). +#define FALLBACK_TO_PRAMIN + +// Starting offset for registers in the corresponding named range +// Programmable First-In First-Out unit; also known as "Host" +#define NV_PFIFO 0x00002000 // 8 KiB long; ends prior to 0x00004000 +// Programmable Channel Control System RAM +#define NV_PCCSR 0x00800000 // 16 KiB long; ends prior to 0x00810000 +// Programmable TOPology registers +#define NV_PTOP 0x00022400 // 1 KiB long; ends prior to 0x00022800 + /* Runlist Channel A timeslice group (TSG) is composed of channels. Each channel is a FIFO queue of GPU commands. These commands are typically queued from userspace. @@ -202,7 +224,7 @@ typedef union { Support: Ampere, Hopper, Ada, [newer untested] */ #define NV_RUNLIST_PREEMPT_GA100 0x098 -#define PREEMPT_TYPE_RUNLIST 0 +#define PREEMPT_TYPE_RUNLIST PREEMPT_TYPE_CHANNEL /* "Initiate a preempt of the engine by writing the bit associated with its @@ -355,6 +377,14 @@ typedef union { uint64_t raw; } runlist_base_tu102_t; +/* + LEN : Read/Write + OFFSET : Read/Write + PREEMPTED_TSGID : Read-only + VALID_PREEMPTED_TSGID : Read-only + IS_PENDING : Read-only + PREEMPTED_OFFSET : Read-only +*/ typedef union { struct { uint16_t len:16; @@ -416,6 +446,27 @@ typedef union { uint32_t raw; } runlist_channel_config_t; +/* Context Switch Timeout Configuration + After a task's budget expires, there's a configurable grace period, a + "timeout", within which the context needs to complete. After this timeout + expires, an interrupt is raised to terminate the task. + + This register configures if such a timeout is enabled and how long the + timeout is (the "period"). + + Support: Volta, Turing +*/ +#define NV_PFIFO_ENG_CTXSW_TIMEOUT 0x00002A0C +// Support: Ampere +#define NV_RUNLIST_ENGINE_CTXSW_TIMEOUT_CONFIG(i) (0x220+(i)*64) +typedef union { + struct { + uint32_t period:31; + bool enabled:1; + } __attribute__((packed)); + uint32_t raw; +} ctxsw_timeout_t; + /* Programmable Channel Control System RAM (PCCSR) 512-entry array of channel control and status data structures. @@ -477,8 +528,15 @@ typedef union { bool busy:1; uint32_t :3; } __attribute__((packed)); + struct { + uint32_t word1; + uint32_t word2; + } __attribute__((packed)); uint64_t raw; -} channel_ctrl_t; +} channel_ctrl_gf100_t; + +// TODO: Remove use of deprecated type name +typedef channel_ctrl_gf100_t channel_ctrl_t; /* CHannel RAM (CHRAM) (PCCSR replacement on Ampere+) Starting with Ampere, channel IDs are no longer unique indexes into the @@ -543,6 +601,8 @@ typedef union { Support: Fermi, Kepler, Maxwell, Pascal, Volta, Turing */ #define NV_PFIFO_SCHED_DISABLE 0x00002630 +// Support: Ampere +#define NV_RUNLIST_SCHED_DISABLE 0x094 typedef union { struct { bool runlist_0:1; @@ -1018,7 +1078,7 @@ typedef union { struct { uint32_t ptr:28; enum INST_TARGET target:2; - uint32_t :1; + uint32_t :1; // disable_cya_debug for BAR2 bool is_virtual:1; } __attribute__((packed)); uint32_t raw; @@ -1091,6 +1151,9 @@ typedef union { Support: Tesla 2.0* through Ampere, Ada *FAULT_REPLAY_* fields are Pascal+ only See also: dev_ram.h (open-gpu-kernel-modules) or dev_ram.ref.txt (open-gpu-doc) + + It appears that on Hopper, IS_VER2 continues to mean IS_VER2, but if unset, the + alternative is VER3. */ #define NV_PRAMIN_PDB_CONFIG_OFF 0x200 typedef union { @@ -1101,7 +1164,7 @@ typedef union { bool fault_replay_tex:1; bool fault_replay_gcc:1; uint32_t :4; - bool is_ver2:1; + bool is_ver2:1; // XXX: Not on Hopper. May be set or not for same page_dir. bool is_64k_big_page:1; // 128Kb otherwise uint32_t page_dir_lo:20; uint32_t page_dir_hi:32; @@ -1421,6 +1484,182 @@ typedef union { } page_tbl_entry_v0_t; */ +/* Fifo Context RAM (RAMFC) and channel INstance RAM (RAMIN) + + Each channel is configured with a 4 KiB instance block. The prefix of this + block is referred to as RAMFC and stores channel-specific state for the Host + (aka PFIFO). + + "A GPU instance block is a block of memory that contains the state + for a GPU context. A GPU context's instance block consists of Host state, + pointers to each engine's state, and memory management state. A GPU instance + block also contains a pointer to a block of memory that contains that part of a + GPU context's state that a user-level driver may access. A GPU instance block + fits within a single 4K-byte page of memory." + + "The NV_RAMFC part of a GPU-instance block contains Host's part of a virtual + GPU's state. Host is referred to as "FIFO". "FC" stands for FIFO Context. + When Host switches from serving one GPU context to serving a second, Host saves + state for the first GPU context to the first GPU context's RAMFC area, and loads + state for the second GPU context from the second GPU context's RAMFC area." + + "Every Host word entry in RAMFC directly corresponds to a PRI-accessible + register. For a description of the contents of a RAMFC entry, please see the + description of the corresponding register in "manuals/dev_pbdma.ref". The + offsets of the fields within each entry in RAMFC match those of the + corresponding register in the associated PBDMA unit's PRI space." + + In summary, RAMFC includes details such as the head and tail of the pushbuffer, + and RAMIN includes details such as the page table configuration(s). + + The instance-global page table (as defined in the PDB field) is only used for + GPU engines which do not support subcontexts (non-VEID engines). + + **Not all documented fields are currently populated below.** + + Support: *Kepler, *Maxwell, *Pascal, Volta, Turing, Ampere, [newer untested] + *Pre-Volta GPUs do not support subcontexts. + See also: dev_ram.ref.txt and dev_pbdma.ref.txt in NVIDIA's open-gpu-doc +*/ + +// 16-byte (128-bit) substructure defining a subcontext configuration +typedef struct { + page_dir_config_t pdb; + uint32_t pasid:20; // Process Address Space ID (PASID) used for ATS + uint32_t :11; + bool enable_ats:1; // Enable Address Translation Services (ATS)? + uint32_t pad; +} __attribute__((packed)) subcontext_ctrl_t; + +typedef struct { +// Start RAMFC (512 bytes) + uint32_t pad[43]; + uint32_t fc_target:5; // NV_RAMFC_TARGET; off 43 + uint32_t :27; + uint32_t pad2[17]; + uint32_t fc_config_l2:1; // NV_RAMFC_CONFIG; off 61 + uint32_t :3; + uint32_t fc_config_ce_split:1; + uint32_t fc_config_ce_no_throttle:1; + uint32_t :2; + uint32_t fc_config_is_priv:1; // ...AUTH_LEVEL + uint32_t :3; + uint32_t fc_config_userd_writeback:1; // ...USERD_WRITEBACK + uint32_t :19; + uint32_t pad3[1]; + uint32_t fc_chan_info_scg:1; // ...SET_CHANNEL_INFO_SCG_TYPE + uint32_t :7; + uint32_t fc_chan_info_veid:6; // ...SET_CHANNEL_INFO_VEID + uint32_t fc_chan_info_chid:12; // ...SET_CHANNEL_INFO_CHID + uint32_t :6; + uint32_t pad4[64]; +// End RAMFC +// Start RAMIN + page_dir_config_t pdb; + uint32_t pad5[2]; + // WFI_TARGET appears to be ignored if WFI_IS_VIRTUAL + uint32_t engine_wfi_target:2; // NV_RAMIN_ENGINE_WFI_TARGET; off 132 + uint32_t engine_wfi_is_virtual:1; + uint32_t :9; + // WFI_PTR points to a CTXSW block (documented below) + uint64_t engine_wfi_ptr:52; // NV_RAMIN_ENGINE_WFI_PTR_LO/_HI; off 132--133 + uint32_t engine_wfi_veid:6; // NV_RAMIN_ENGINE_WFI_VEID; off 134; VEID == Subcontext ID + uint32_t :26; + uint32_t pasid:20; // NV_RAMIN_PASID; off 135; "Process Address Space ID" + uint32_t :11; + bool enable_ats:1; + uint32_t pad6[30]; + uint64_t subcontext_pdb_valid; // NV_RAMIN_SC_PDB_VALID; off 166-167 + subcontext_ctrl_t subcontext[64]; // NV_RAMIN_SC_*; off 168-424 +} __attribute__((packed)) instance_ctrl_t; + +// Context types +enum CTXSW_TYPE { + CTXSW_UNDEFINED = 0x0, + CTXSW_OPENGL = 0x8, + CTXSW_DX9 = 0x10, + CTXSW_DX10 = 0x11, + CTXSW_DX11 = 0x12, + CTXSW_COMPUTE = 0x20, + CTXSW_HEADER = 0x21 // A per-subcontext header +}; +static inline const char *ctxsw_type_to_text(enum CTXSW_TYPE t) { + switch (t) { + case CTXSW_UNDEFINED: + return "[None]"; + case CTXSW_OPENGL: + return "OpenGL"; + case CTXSW_DX9: + case CTXSW_DX10: + case CTXSW_DX11: + return "DirectX"; + case CTXSW_COMPUTE: + return "Compute"; + case CTXSW_HEADER: + return "Header"; + default: + return "UNKNOWN"; + } +} + +// Preemption modes: +// WFI: Wait For Idle (preempt on idle) +// CTA: Cooperative Thread Array-level Preemption (preempt at end of block) +// CILP: Compute-Instruction-Level Preemption (preempt at end of instruction) +enum GRAPHICS_PREEMPT_TYPE {PREEMPT_WFI, PREEMPT_GFXP}; +enum COMPUTE_PREEMPT_TYPE {_PREEMPT_WFI, PREEMPT_CTA, PREEMPT_CILP}; +static inline const char *compute_preempt_type_to_text(enum COMPUTE_PREEMPT_TYPE t) { + switch (t) { + case PREEMPT_WFI: + return "WFI"; + case PREEMPT_CTA: + return "CTA"; + case PREEMPT_CILP: + return "CILP"; + default: + return "INVALID"; + } +} +static inline const char *graphics_preempt_type_to_text(enum COMPUTE_PREEMPT_TYPE t) { + switch (t) { + case PREEMPT_WFI: + return "WFI"; + case PREEMPT_GFXP: + return "GFXP"; + default: + return "INVALID"; + } +} + +/* ConTeXt SWitch control block (CTXSW) + Support: Maxwell*, Pascal**, Volta, Turing, Ampere, Ada + *Nothing except for CONTEXT_ID and TYPE + **Except as noted + See also: manuals/volta/gv100/dev_ctxsw.ref.txt in open-gpu-doc + and hw_ctxsw_prog_*.h in nvgpu +*/ +// (Note that this layout changes some generation-to-generation) +typedef struct context_switch_block { + uint32_t pad[3]; + enum CTXSW_TYPE type:6; // Unused except when type CTXSW_HEADER? + uint32_t :26; + uint32_t pad2[26]; + // The context buffer ptr fields are in an opposite-of-typical order, so we + // can't merge them into a single context_buffer_ptr field. + uint32_t context_buffer_ptr_hi; // Volta+ only + uint32_t context_buffer_ptr_lo; // Volta+ only + enum GRAPHICS_PREEMPT_TYPE graphics_preemption_options:32; + enum COMPUTE_PREEMPT_TYPE compute_preemption_options:32; + uint32_t pad3[18]; + uint32_t num_wfi_save_operations; + uint32_t num_cta_save_operations; + uint32_t num_gfxp_save_operations; + uint32_t num_cilp_save_operations; + uint32_t pad4[4]; + uint32_t context_id; + // [There are more fields not yet added here.] +} __attribute__((packed)) context_switch_ctrl_t; + /* VRAM Information If ECC is disabled: @@ -1452,6 +1691,12 @@ static inline uint64_t memory_range_to_bytes(memory_range_t range) { /* Begin nvdebug types and functions */ +// __iomem is only defined when building as a kernel module, so conditionally +// define it to allow including this header outside the kernel. +#ifndef __iomem +#define __iomem +#endif + // Vendor ID for PCI devices manufactured by NVIDIA #define NV_PCI_VENDOR 0x10de struct nvdebug_state { @@ -1474,6 +1719,10 @@ struct nvdebug_state { struct platform_device *platd; // Pointer to generic device struct (both platform and pcie devices) struct device *dev; +#ifdef __KERNEL__ + // List used by mmu.c to track allocated pages for page directories/tables + struct list_head pd_allocs; +#endif }; // This disgusting macro is a crutch to work around the fact that runlists were @@ -1542,7 +1791,19 @@ int get_runlist_iter( struct runlist_iter *rl_iter /* out */); int preempt_tsg(struct nvdebug_state *g, uint32_t rl_id, uint32_t tsg_id); int preempt_runlist(struct nvdebug_state *g, uint32_t rl_id); -int resubmit_runlist(struct nvdebug_state *g, uint32_t rl_id); +int resubmit_runlist(struct nvdebug_state *g, uint32_t rl_id, uint32_t off); +instance_ctrl_t *instance_deref( + struct nvdebug_state *g, + uint64_t instance_addr, + enum INST_TARGET instance_target); +context_switch_ctrl_t *get_ctxsw( + struct nvdebug_state *g, + instance_ctrl_t *inst); +int set_channel_preemption_mode( + struct nvdebug_state *g, + uint32_t chan_id, + uint32_t rl_id, + enum COMPUTE_PREEMPT_TYPE mode); // Defined in mmu.c uint64_t search_page_directory( @@ -1550,11 +1811,33 @@ uint64_t search_page_directory( page_dir_config_t pd_config, uint64_t addr_to_find, enum INST_TARGET addr_to_find_aperture); +int translate_page_directory( + struct nvdebug_state *g, + page_dir_config_t pd_config, + uint64_t addr_to_find, + uint64_t *found_addr /* out */, + enum INST_TARGET *found_aperture /* out */); +int map_page_directory( + struct nvdebug_state *g, + page_dir_config_t pd_config, + uint64_t paddr_to_map, + uint64_t vaddr_to_find, + enum INST_TARGET paddr_target, + bool huge_page); +int gc_page_directory( + struct nvdebug_state *g, + bool force); uint64_t search_v1_page_directory( struct nvdebug_state *g, page_dir_config_t pd_config, uint64_t addr_to_find, enum INST_TARGET addr_to_find_aperture); +int translate_v1_page_directory( + struct nvdebug_state *g, + page_dir_config_t pd_config, + uint64_t addr_to_find, + uint64_t *found_addr /* out */, + enum INST_TARGET *found_aperture /* out */); // Defined in bus.c int addr_to_pramin_mut(struct nvdebug_state *g, uint64_t addr, enum INST_TARGET target); int get_bar2_pdb(struct nvdebug_state *g, page_dir_config_t* pd /* out */); diff --git a/nvdebug_entry.c b/nvdebug_entry.c index 3a10e13..c0cfa63 100644 --- a/nvdebug_entry.c +++ b/nvdebug_entry.c @@ -15,7 +15,7 @@ // Enable to intercept and log GPU interrupts. Historically used to benchmark // interrupt latency. -#define INTERRUPT_DEBUG 0 +#define INTERRUPT_DEBUG // MIT is GPL-compatible. We need to be GPL-compatible for symbols like // platform_bus_type or bus_find_device_by_name... @@ -28,12 +28,20 @@ extern struct file_operations runlist_file_ops; extern struct file_operations preempt_tsg_file_ops; extern struct file_operations disable_channel_file_ops; extern struct file_operations enable_channel_file_ops; +extern struct file_operations wfi_preempt_channel_file_ops; +extern struct file_operations cta_preempt_channel_file_ops; +extern struct file_operations cil_preempt_channel_file_ops; extern struct file_operations resubmit_runlist_file_ops; +extern struct file_operations preempt_runlist_file_ops; +extern struct file_operations ack_bad_tsg_file_ops; +extern struct file_operations map_mem_chid_file_ops; +extern struct file_operations map_mem_ctxid_file_ops; extern struct file_operations switch_to_tsg_file_ops; // device_info_procfs.c extern struct file_operations device_info_file_ops; extern struct file_operations nvdebug_read_reg32_file_ops; extern struct file_operations nvdebug_read_reg_range_file_ops; +extern struct file_operations nvdebug_read_part_file_ops; extern struct file_operations local_memory_file_ops; // copy_topology_procfs.c extern struct file_operations copy_topology_file_ops; @@ -71,9 +79,271 @@ const struct file_operations* compat_ops(const struct file_operations* ops) { } #endif -#if INTERRUPT_DEBUG +#ifdef INTERRUPT_DEBUG + +void nvdebug_fifo_intr(struct nvdebug_state *g) { + uint32_t fifo_intr_mask;// = nvdebug_readl(g, 0x02100); // PFIFO_INTR_0 + fifo_intr_mask = nvdebug_readl(g, 0x02100); // PFIFO_INTR_0 + if (fifo_intr_mask & 1 << 0) + printk(KERN_INFO "[nvdebug] - Interrupt BIND_ERROR.\n"); + if (fifo_intr_mask & 1 << 1) + printk(KERN_INFO "[nvdebug] - Interrupt CTXSW_TIMEOUT.\n"); + if (fifo_intr_mask & 1 << 4) + printk(KERN_INFO "[nvdebug] - Interrupt RUNLIST_IDLE.\n"); + if (fifo_intr_mask & 1 << 5) + printk(KERN_INFO "[nvdebug] - Interrupt RUNLIST_AND_ENG_IDLE.\n"); + if (fifo_intr_mask & 1 << 6) + printk(KERN_INFO "[nvdebug] - Interrupt RUNLIST_ACQUIRE.\n"); + if (fifo_intr_mask & 1 << 7) + printk(KERN_INFO "[nvdebug] - Interrupt RUNLIST_ACQUIRE_AND_ENG_IDLE.\n"); + if (fifo_intr_mask & 1 << 8) + printk(KERN_INFO "[nvdebug] - Interrupt SCHED_ERROR.\n"); + if (fifo_intr_mask & 1 << 16) + printk(KERN_INFO "[nvdebug] - Interrupt CHSW_ERROR.\n"); + if (fifo_intr_mask & 1 << 23) + printk(KERN_INFO "[nvdebug] - Interrupt MEMOP_TIMEOUT.\n"); + if (fifo_intr_mask & 1 << 24) + printk(KERN_INFO "[nvdebug] - Interrupt LB_ERROR.\n"); + if (fifo_intr_mask & 1 << 25) // OLD; Pascal + printk(KERN_INFO "[nvdebug] - Interrupt REPLAYABLE_FAULT_ERROR.\n"); + if (fifo_intr_mask & 1 << 27) // OLD; Pascal + printk(KERN_INFO "[nvdebug] - Interrupt DROPPED_MMU_FAULT.\n"); + if (fifo_intr_mask & 1 << 28) { // On Pascal, this is MMU_FAULT + if (g->chip_id <= NV_CHIP_ID_VOLTA) // MMU_FAULT on Pascal (nvgpu, l4t/l4t-r28.1:drivers/gpu/nvgpu/include/nvgpu/hw/gp10b/hw_fifo_gp10b.h) + printk(KERN_INFO "[nvdebug] - Interrupt MMU_FAULT.\n"); + else // Repurposed starting with Turing: open-gpu-doc/manuals/turing/tu104/dev_fifo.ref.txt + printk(KERN_INFO "[nvdebug] - Interrupt TSG_PREEMPT_COMPLETE.\n"); + } + if (fifo_intr_mask & 1 << 29) + printk(KERN_INFO "[nvdebug] - Interrupt PBDMA_INTR.\n"); + if (fifo_intr_mask & 1 << 30) { + printk(KERN_INFO "[nvdebug] - Interrupt RUNLIST_EVENT.\n"); + uint32_t fifo_runlist_intr_mask = nvdebug_readl(g, 0x02A00); // PFIFO_INTR_RUNLIST + printk(KERN_INFO "[nvdebug] - Event %#x.\n", fifo_runlist_intr_mask); + } + if (fifo_intr_mask & 1 << 31) + printk(KERN_INFO "[nvdebug] - Interrupt CHANNEL_INTR.\n"); +} + irqreturn_t nvdebug_irq_tap(int irq_num, void * dev) { - printk(KERN_INFO "[nvdebug] Interrupt tap triggered on IRQ %d.\n", irq_num); + struct nvdebug_state *g = dev; + u64 time = ktime_get_raw_ns(); // CLOCK_MONOTONTIC_RAW + // NV_PMC_INTR does not exist on Ada, so use NV_FUNC_PRIV_CPU_INTR_TOP + // Note that this also appears to exist on Turing + if (g->chip_id >= NV_CHIP_ID_TURING) {//AMPERE) { + int i; + // Despite being an indexed register, it is only documented to have on, and could only support two + uint32_t intr_mask0 = nvdebug_readl(g, NV_VIRTUAL_FUNCTION_FULL_PHYS_OFFSET + 0x1600); // NV_FUNC_PRIV_CPU_INTR_TOP(0) + uint32_t intr_mask1 = nvdebug_readl(g, NV_VIRTUAL_FUNCTION_FULL_PHYS_OFFSET + 0x1604); // NV_FUNC_PRIV_CPU_INTR_TOP(1) + printk(KERN_INFO "[nvdebug] Interrupt on IRQ %d with CPU_INTR_TOP(0) %#010x, ...(1) %#010x @ %llu.\n", irq_num, intr_mask0, intr_mask1, time); + for (i = 0; i < 8; i++) { + uint32_t leaf = nvdebug_readl(g, NV_VIRTUAL_FUNCTION_FULL_PHYS_OFFSET + 0x1000 + i*4); // NV_FUNC_PRIV_CPU_INTR_LEAF(0) to ...(7) + if (leaf) + printk(KERN_INFO "[nvdebug] - Interrupt leaf %d: %#010x\n", i, leaf); + // 131-133 & 64 are faults on tu104??? (open-gpu-doc/manuals/turing/tu104/pri_mmu_hub.ref.txt) + if (136 / 32 == i && 1 << (136 % 32) & leaf) // PFIFO0 ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on PFIFO0.\n"); + if (137 / 32 == i && 1 << (137 % 32) & leaf) // PFIFO1 ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on PFIFO1.\n"); + if (148 / 32 == i && 1 << (148 % 32) & leaf) // TIMER ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on PTIMER.\n"); + if (152 / 32 == i && 1 << (152 % 32) & leaf) // PMU ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on PMU.\n"); + if (156 / 32 == i && 1 << (156 % 32) & leaf) { // PBUS ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on PBUS.\n"); + uint32_t bus_intr = nvdebug_readl(g, 0x1100); // BUS_INTR_0 + if (bus_intr & 1 << 2) { + // use timer_pri_timeout_save_0_r + uint32_t SAVE_0 = nvdebug_readl(g, 0x00009084); // NV_PTIMER_PRI_TIMEOUT_SAVE_0 + printk(KERN_INFO "[nvdebug] - Interrupt PRI_FECSERR on %s to address %#010x %stargeting FECS.\n", SAVE_0 & 0x2 ? "write" : "read", SAVE_0 & 0x00fffffc, SAVE_0 & 0x80000000 ? "" : "not "); + uint32_t SAVE_1 = nvdebug_readl(g, 0x00009088); // NV_PTIMER_PRI_TIMEOUT_SAVE_1 + if (SAVE_1) + printk(KERN_INFO "[nvdebug] Data written: %#010x\n", SAVE_1); + uint32_t errcode = readl(g->regs + 0x0000908C); // NV_PTIMER_PRI_TIMEOUT_FECS_ERRCODE + if (errcode) + printk(KERN_INFO "[nvdebug] FECS Error Code: %#010x\n", errcode); + // badf5040 is a "client error" (0) of "no such address" (40) + // See linux-nvgpu/drivers/gpu/nvgpu/hal/priv_ring/priv_ring_ga10b_fusa.c + // for how to decode. + } + if (bus_intr & 1 << 3) + printk(KERN_INFO "[nvdebug] - Interrupt PRI_TIMEOUT.\n"); + if (bus_intr & 1 << 4) + printk(KERN_INFO "[nvdebug] - Interrupt FB_REQ_TIMEOUT.\n"); + if (bus_intr & 1 << 5) + printk(KERN_INFO "[nvdebug] - Interrupt FB_ACK_TIMEOUT.\n"); + if (bus_intr & 1 << 6) + printk(KERN_INFO "[nvdebug] - Interrupt FB_ACK_EXTRA.\n"); + if (bus_intr & 1 << 7) + printk(KERN_INFO "[nvdebug] - Interrupt FB_RDATA_TIMEOUT.\n"); + if (bus_intr & 1 << 8) + printk(KERN_INFO "[nvdebug] - Interrupt FB_RDATA_EXTRA.\n"); + if (bus_intr & 1 << 26) + printk(KERN_INFO "[nvdebug] - Interrupt SW.\n"); + if (bus_intr & 1 << 27) + printk(KERN_INFO "[nvdebug] - Interrupt POSTED_DEADLOCK_TIMEOUT.\n"); + if (bus_intr & 1 << 28) + printk(KERN_INFO "[nvdebug] - Interrupt MPMU.\n"); + if (bus_intr & 1 << 31) + printk(KERN_INFO "[nvdebug] - Interrupt ACCESS_TIMEOUT.\n"); + } + if (158 / 32 == i && 1 << (158 % 32) & leaf) // PRIV_RING ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on PRIV_RING.\n"); + if (192 / 32 == i && 1 << (192 % 32) & leaf) // LEGACY_ENGINE_STALL ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on LEGACY_ENGINE_STALL.\n"); + if (160 / 32 == i && 1 << (160 % 32) & leaf) { // (likely) rl0 ga100 + printk(KERN_INFO "[nvdebug] - Interrupt on RUNLIST0.\n"); + uint32_t off; + get_runlist_ram(g, 0, &off); + uint32_t rl_intr = nvdebug_readl(g, off+0x100); + printk(KERN_INFO "[nvdebug] - RUNLIST_INTR_0: %#x\n", rl_intr); + if (1 << 12 && rl_intr) { // BAD_TSG + printk(KERN_INFO "[nvdebug] - BAD_TSG: %#x\n", nvdebug_readl(g, off+0x174)); + } + } + // Also getting 160, 161, and 162 + + //uint32_t off; + //get_runlist_ram(g, 12, &off); + //printk(KERN_INFO "[nvdebug] - rl10 vector id 0 is %x\n", nvdebug_readl(g, off+0x160)); // NV_RUNLIST_INTR_VECTORID(0) + // 160 is rl0 (C/G, LCE0, LCE1) vector id 0 + // 168 is rl11 (LCE3) vector id 0 + // 169 is rl12 (LCE4) vector id 0 + // 171 is rl1 (SEC) vector id 0 + // 176 is rl10 (LCE2) vector id 0 + // 224 is rl0 vector id 1 + // Only some interrupt vectors are hardcoded + } + // each subtree has two leafs? Each bit at the top corresponds to a subtree? + // So, if bit 0 is set, that means subtree 0 (concept) and leaves 0 and 1 + // So, if bit 1 is set, that means subtree 1 (concept) and leaves 2 and 3 + // the #define'd interrupt vectors all seem to fall in the lower leaf of subtree 2, + // except for INTR_HUB_ACCESS_CNTR_INTR_VECTOR is in the lower leaf of subtree 1 + if (g->chip_id >= NV_CHIP_ID_AMPERE) + return IRQ_NONE; + } + uint32_t intr_mask = nvdebug_readl(g, 0x0100); // NV_PMC_INTR + printk(KERN_INFO "[nvdebug] Interrupt on IRQ %d with MC_INTR %#010x @ %llu.\n", irq_num, intr_mask, time); + // IDs likely changed Ampere+ + //if (g->chip_id >= NV_CHIP_ID_AMPERE) { + // CIC is central interrupt controller + // the u32 passed around nvgpu cic functions is one of the + // enable is nvgpu_cic_mon_intr_stall_unit_config(unit) + // - Calls intr_stall_unit_config(unit) + // - for ga, calls unit = ga10b_intr_map_mc_stall_unit_to_intr_unit(unit) (doesn't do much) + // - for ga, calls nvgpu_cic_mon_intr_get_unit_info() + // - Does *subtree = g->mc.intr_unit_info[unit].subtree; + // *subtree_mask = g->mc.intr_unit_info[unit].subtree_mask; + // - for ga, calls ga10b_intr_config() w/ subtree info + //uint32_t intr_stats = nvdebug_readl(g, 1600 + //return IRQ_NONE; + //} + if (intr_mask & 1 << 5) + printk(KERN_INFO "[nvdebug] - Interrupt on LCE0.\n"); + if (intr_mask & 1 << 6) + printk(KERN_INFO "[nvdebug] - Interrupt on LCE1.\n"); + if (intr_mask & 1 << 7) + printk(KERN_INFO "[nvdebug] - Interrupt on LCE2.\n"); + if (intr_mask & 1 << 8) { + printk(KERN_INFO "[nvdebug] - Interrupt on PFIFO.\n"); + nvdebug_fifo_intr(g); + } + if (intr_mask & 1 << 9) { + printk(KERN_INFO "[nvdebug] - Interrupt on HUB.\n"); // "replayable_fault_pending" in nvgpu on Pascal, "HUB" on Volta+ + // on tu104, if vector is one of the below set in new-style interrupt vector, then MMU fault + // - info_fault (134) + // - nonreplay_fault error (133) + // - nonreplay_fault notify (132) + // - replay_fault error (131) + // - replay_fault notify (64) + // (but the above fault vectors are configurable) + // if it's ecc_error, then not mmu error + // Default fault vectors from open-gpu-doc/manuals/turing/tu104/pri_mmu_hub.ref.txt + // Turing through (at least) Ampere (per nvgpu) + + // on gv100, parse fb_niso_intr_r 0x00100a20U, where bits: + // - hub_access_counter notify (0) + // - hub_access_counter error (1) + // - replay_fault notify (27) + // - replay_fault overflow (28) + // - nonreplay_fault notify (29) + // - nonreplay_fault overflow (30) + // - other_fault notify (31) + // Volta through Turing (per nvgpu) + + // On Pascal, it looks like it's a property of fifo_intr_0??? + if (g->chip_id < NV_CHIP_ID_VOLTA) + nvdebug_fifo_intr(g); + } + if (intr_mask & 1 << 10) + printk(KERN_INFO "[nvdebug] - Interrupt on LCE3.\n"); + if (intr_mask & 1 << 11) + printk(KERN_INFO "[nvdebug] - Interrupt on LCE4.\n"); + if (intr_mask & 1 << 12) { + printk(KERN_INFO "[nvdebug] - Interrupt on Graphics/Compute.\n"); + // Kepler through (at least) Ampere + // From open-gpu-doc/manuals/volta/gv100/dev_graphics.ref.txt + uint32_t graph_intr_mask = nvdebug_readl(g, 0x400100); // NV_PGRAPH_INTR + if (graph_intr_mask & 1 << 0) + printk(KERN_INFO "[nvdebug] - Interrupt NOTIFY.\n"); + if (graph_intr_mask & 1 << 1) + printk(KERN_INFO "[nvdebug] - Interrupt SEMAPHORE.\n"); + if (graph_intr_mask & 1 << 4) + printk(KERN_INFO "[nvdebug] - Interrupt ILLEGAL_METHOD.\n"); + if (graph_intr_mask & 1 << 5) + printk(KERN_INFO "[nvdebug] - Interrupt ILLEGAL_CLASS.\n"); + if (graph_intr_mask & 1 << 6) + printk(KERN_INFO "[nvdebug] - Interrupt ILLEGAL_NOTIFY.\n"); + if (graph_intr_mask & 1 << 7) + printk(KERN_INFO "[nvdebug] - Interrupt DEBUG_METHOD.\n"); + if (graph_intr_mask & 1 << 8) + printk(KERN_INFO "[nvdebug] - Interrupt FIRMWARE_METHOD.\n"); + if (graph_intr_mask & 1 << 16) + printk(KERN_INFO "[nvdebug] - Interrupt BUFFER_NOTIFY.\n"); + if (graph_intr_mask & 1 << 19) + printk(KERN_INFO "[nvdebug] - Interrupt FECS_ERROR.\n"); + if (graph_intr_mask & 1 << 20) + printk(KERN_INFO "[nvdebug] - Interrupt CLASS_ERROR.\n"); + if (graph_intr_mask & 1 << 21) + printk(KERN_INFO "[nvdebug] - Interrupt EXCEPTION.\n"); + } + if (intr_mask & 1 << 13) + printk(KERN_INFO "[nvdebug] - Interrupt on PFB.\n"); + if (intr_mask & 1 << 15) + printk(KERN_INFO "[nvdebug] - Interrupt on SEC.\n"); + if (intr_mask & 1 << 16) + printk(KERN_INFO "[nvdebug] - Interrupt on NVENC0.\n"); + if (intr_mask & 1 << 17) + printk(KERN_INFO "[nvdebug] - Interrupt on NVDEC0.\n"); + if (intr_mask & 1 << 18) + printk(KERN_INFO "[nvdebug] - Interrupt on THERMAL.\n"); + if (intr_mask & 1 << 19) + printk(KERN_INFO "[nvdebug] - Interrupt on HDACODEC.\n"); + if (intr_mask & 1 << 20) + printk(KERN_INFO "[nvdebug] - Interrupt on PTIMER.\n"); + if (intr_mask & 1 << 21) + printk(KERN_INFO "[nvdebug] - Interrupt on PMGR.\n"); + if (intr_mask & 1 << 22) + printk(KERN_INFO "[nvdebug] - Interrupt on IOCTRL.\n"); + if (intr_mask & 1 << 23) + printk(KERN_INFO "[nvdebug] - Interrupt on DFD.\n"); + if (intr_mask & 1 << 24) + printk(KERN_INFO "[nvdebug] - Interrupt on PMU.\n"); + if (intr_mask & 1 << 25) + printk(KERN_INFO "[nvdebug] - Interrupt on LTC.\n"); + if (intr_mask & 1 << 26) + printk(KERN_INFO "[nvdebug] - Interrupt on PDISP.\n"); + if (intr_mask & 1 << 27) + printk(KERN_INFO "[nvdebug] - Interrupt on GSP.\n"); + if (intr_mask & 1 << 28) + printk(KERN_INFO "[nvdebug] - Interrupt on PBUS.\n"); + if (intr_mask & 1 << 29) + printk(KERN_INFO "[nvdebug] - Interrupt on XVE.\n"); + if (intr_mask & 1 << 30) + printk(KERN_INFO "[nvdebug] - Interrupt on PRIV_RING.\n"); + if (intr_mask & 1 << 30) + printk(KERN_INFO "[nvdebug] - Interrupt on SOFTWARE.\n"); + return IRQ_NONE; // We don't actually handle any interrupts. Pass them on. } #endif // INTERRUPT_DEBUG @@ -135,6 +405,7 @@ int probe_and_cache_devices(void) { g_nvdebug_state[i].pcid = NULL; g_nvdebug_state[i].platd = platd; g_nvdebug_state[i].dev = dev; + INIT_LIST_HEAD(&g_nvdebug_state[i].pd_allocs); // Don't check Chip ID until everything else is initalized ids.raw = nvdebug_readl(&g_nvdebug_state[i], NV_MC_BOOT_0); if (ids.raw == -1) { @@ -152,6 +423,11 @@ int probe_and_cache_devices(void) { mc_boot_0_t ids; g_nvdebug_state[i].g = NULL; // Map BAR0 (GPU control registers) + // XXX: Don't use pci_iomap. This adds support for I/O registers, but we do + // not use the required ioread/write functions for those regions. We + // should use pci_ioremap_bar, which is explictly for MMIO regions. + // pci_ioremap_bar -> ioremap_nocache (all platforms) + // pci_iomap -> ioremap_nocache (on x86) g_nvdebug_state[i].regs = pci_iomap(pcid, 0, 0); if (!g_nvdebug_state[i].regs) { pci_err(pcid, "[nvdebug] Unable to map BAR0 on this GPU\n"); @@ -163,9 +439,14 @@ int probe_and_cache_devices(void) { // (vesafb may map the top half for display) if (!g_nvdebug_state[i].bar3) g_nvdebug_state[i].bar3 = pci_iomap(pcid, 3, pci_resource_len(pcid, 3)/2); + // Observed on H100, BAR2, moved it BAR3, was moved to BAR4, and BAR1 + // was moved to BAR2. + if (!g_nvdebug_state[i].bar3) + g_nvdebug_state[i].bar3 = pci_iomap(pcid, 4, 0); g_nvdebug_state[i].pcid = pcid; g_nvdebug_state[i].platd = NULL; g_nvdebug_state[i].dev = &pcid->dev; + INIT_LIST_HEAD(&g_nvdebug_state[i].pd_allocs); // Don't check Chip ID until everything else is initalized ids.raw = nvdebug_readl(&g_nvdebug_state[i], NV_MC_BOOT_0); if (ids.raw == -1) { @@ -175,9 +456,17 @@ int probe_and_cache_devices(void) { g_nvdebug_state[i].chip_id = ids.chip_id; printk(KERN_INFO "[nvdebug] Chip ID %x (architecture %s) detected on PCI bus and initialized.", ids.chip_id, ARCH2NAME(ids.architecture)); -#if INTERRUPT_DEBUG - if (request_irq(pcid->irq, nvdebug_irq_tap, IRQF_SHARED, "nvdebug tap", pcid)) { - printk(KERN_WARNING "[nvdebug] Unable to initialize IRQ tap\n"); +#ifdef INTERRUPT_DEBUG + // For this to work, you must also add IRQF_SHARED to the flags + // argument of the request_threaded_irq() call in the nvidia driver + // (file /usr/src/nvidia.../nvidia/nv.c and nv-msi.c with dkms) + // Then run: + // sudo dkms remove nvidia-srv/VER -k $(uname -r) + // sudo dkms install nvidia-srv/VER -k $(uname -r) --force + // where VER is the version of the nvidia module (eg. 535.216.03) + int err; + if ((err = request_irq(pcid->irq, nvdebug_irq_tap, IRQF_SHARED, "nvdebug tap", &g_nvdebug_state[i]))) { + printk(KERN_WARNING "[nvdebug] Unable to initialize IRQ tap, error %d\n", err); } #endif // INTERRUPT_DEBUG i++; @@ -335,6 +624,40 @@ int __init nvdebug_init(void) { "enable_channel", 0222, chram_scope, compat_ops(&enable_channel_file_ops), (void*)last_runlist)) goto out_nomem; + // Create file `/proc/gpu#/runlist#/wfi_preempt_channel`, world writable + // On Turing and older, `/proc/gpu#/wfi_preempt_channel` + if (!proc_create_data( + "wfi_preempt_channel", 0222, chram_scope, compat_ops(&wfi_preempt_channel_file_ops), + (void*)last_runlist)) + goto out_nomem; + // Create file `/proc/gpu#/runlist#/cta_preempt_channel`, world writable + // On Turing and older, `/proc/gpu#/cta_preempt_channel` + if (!proc_create_data( + "cta_preempt_channel", 0222, chram_scope, compat_ops(&cta_preempt_channel_file_ops), + (void*)last_runlist)) + goto out_nomem; + // Compute-instruction-level (CIL) preemption is only available on Pascal+ + if (g_nvdebug_state[res].chip_id >= NV_CHIP_ID_PASCAL) { + // Create file `/proc/gpu#/runlist#/cil_preempt_channel`, world writable + // On Turing and older, `/proc/gpu#/cil_preempt_channel` + if (!proc_create_data( + "cil_preempt_channel", 0222, chram_scope, compat_ops(&cil_preempt_channel_file_ops), + (void*)last_runlist)) + goto out_nomem; + } + // Create files which enable on-GPU scheduling (Pascal+) + if (g_nvdebug_state[res].chip_id >= NV_CHIP_ID_PASCAL) { + // Create file `/proc/gpu#/map_mem_chid`, root writable + if (!proc_create_data( + "map_mem_chid", 0200, chram_scope, compat_ops(&map_mem_chid_file_ops), + (void*)last_runlist)) + goto out_nomem; + // Create file `/proc/gpu#/map_mem_ctxid`, root writable + if (!proc_create_data( + "map_mem_ctxid", 0222, rl_dir, compat_ops(&map_mem_ctxid