aboutsummaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
authorJoshua Bakita <bakitajoshua@gmail.com>2024-04-11 12:23:18 -0400
committerJoshua Bakita <jbakita@cs.unc.edu>2024-04-11 13:03:20 -0400
commita8fd5a8dee066d0008e7667b0c9e6a60cd5f3a2e (patch)
treef05095d4b6458a709034a182649e6d16b6a8558a
parent5ea953292441e31e37ae074e48d8b3b5ce1d9440 (diff)
Support page directories outside PRAMIN or in SYS_MEM
- Re-read PRAMIN configuration after update to verify change applies - Return a page_dir_config_t rather than just an address and page. table version from `get_bar2_pdb()`. - Less verbose logging for MMU-related functions by default. - Perform all conversion from SYS_MEM/VID_MEM addresses to kernel addresses inside the translation functions, via the new function 'pd_deref()`. - Support use of an I/O MMU, page tables/directories outside the current PRAMIN window, and page tables/directories arbitrarially located in SYS_MEM or VID_MEM on different levels of the same tree. - Heavily improve documentation and add references for Version 1 and Version 0 page tables. - Improve logging in `runlist.c` to include runlist and chip IDs. - Update all users of search_page_directory* to use the new API. - Remove now-unused supporting functions from `mmu.c`. Tested on GTX 970, GTX 1060 3GB, Jetson TX2, Titan V, Jetson Xavier, and RTX 2080 Ti.
-rw-r--r--bus.c50
-rw-r--r--mmu.c206
-rw-r--r--nvdebug.h158
-rw-r--r--nvdebug_entry.c2
-rw-r--r--runlist.c33
5 files changed, 268 insertions, 181 deletions
diff --git a/bus.c b/bus.c
index 802b6df..951ac77 100644
--- a/bus.c
+++ b/bus.c
@@ -57,35 +57,25 @@ relocate:
57 window.base = (u32)(addr >> 16); // Safe, due to above range check 57 window.base = (u32)(addr >> 16); // Safe, due to above range check
58 window.target = target; 58 window.target = target;
59 nvdebug_writel(g, NV_PBUS_BAR0_WINDOW, window.raw); 59 nvdebug_writel(g, NV_PBUS_BAR0_WINDOW, window.raw);
60 // Wait for the window to move by re-reading (as done in nvgpu driver)
61 (void) nvdebug_readl(g, NV_PBUS_BAR0_WINDOW);
60 return (int)(addr & 0xffffull); 62 return (int)(addr & 0xffffull);
61} 63}
62 64
63 65/* Get a copy of the BAR2 page directory configuration (base and aperture)
64/* Get a persistent pointer to the page directory base 66 @param pd Pointer at which to store the configuration, including a pointer
65 @param pdb Dereferencable pointer to the zeroeth entry of top-level page 67 and aperture for the zeroth entry of the top-level page directory
66 directory (PD3) for the BAR2 register region. 68 (PD3 for V2 page tables). This pointer **may not** be directly
67 Note: The returned pointer will be into the PRAMIN space. If the PRAMIN 69 dereferencable, and the caller may need to shift the BAR2 window.
68 window is moved to a region that does not cover the BAR2 page table, 70 @return 0 on success, -errno on error.
69 this ***will move the window***. 71 Note: This may move the PRAMIN window.
70 Note: Even if the page table is located in SYS_MEM, we route reads/writes via
71 PRAMIN. This ensures that we always see what the GPU sees, and that
72 includes any passes through I/O MMUs or IOVA spaces.
73*/ 72*/
74int get_bar2_pdb(struct nvdebug_state *g, void **pdb, bool *is_v2_pdb) { 73int get_bar2_pdb(struct nvdebug_state *g, page_dir_config_t* pd) {
75 static void* cached_pdb = NULL;
76 static bool cached_is_v2_pdb = false;
77 static long pd_hash = 0;
78 int ret; 74 int ret;
79 bar_config_block_t bar2_block; 75 bar_config_block_t bar2_block;
80 page_dir_config_t pd_config;
81 uint64_t pdb_vram;
82 76
83 // Use cached base as long as it's still pointing to the same thing 77 if (!pd)
84 if (cached_pdb && readl(cached_pdb) == pd_hash) { 78 return -EINVAL;
85 *pdb = cached_pdb;
86 *is_v2_pdb = cached_is_v2_pdb;
87 return 0;
88 }
89 79
90 if (!g->bar2) 80 if (!g->bar2)
91 return -ENXIO; 81 return -ENXIO;
@@ -107,24 +97,10 @@ int get_bar2_pdb(struct nvdebug_state *g, void **pdb, bool *is_v2_pdb) {
107 } 97 }
108 printk(KERN_INFO "[nvdebug] BAR2 inst block at off %x in PRAMIN\n", ret); 98 printk(KERN_INFO "[nvdebug] BAR2 inst block at off %x in PRAMIN\n", ret);
109 // Pull the page directory base configuration from the instance block 99 // Pull the page directory base configuration from the instance block
110 if ((pd_config.raw = nvdebug_readq(g, NV_PRAMIN + ret + NV_PRAMIN_PDB_CONFIG_OFF)) == -1) { 100 if ((pd->raw = nvdebug_readq(g, NV_PRAMIN + ret + NV_PRAMIN_PDB_CONFIG_OFF)) == -1) {
111 printk(KERN_ERR "[nvdebug] Unable to read BAR2/3 PDB configuration! BAR2/3 inaccessible.\n"); 101 printk(KERN_ERR "[nvdebug] Unable to read BAR2/3 PDB configuration! BAR2/3 inaccessible.\n");
112 return -ENOTSUPP; 102 return -ENOTSUPP;
113 } 103 }
114 pdb_vram = pd_config.page_dir_hi;
115 pdb_vram <<= 20;
116 pdb_vram |= pd_config.page_dir_lo;
117 pdb_vram <<= 12;
118 printk(KERN_INFO "[nvdebug] BAR2 PDB @ %llx (config raw: %llx)\n", pdb_vram, pd_config.raw);
119 // Setup PRAMIN to point at the page directory
120 if ((ret = addr_to_pramin_mut(g, pdb_vram, pd_config.target)) < 0) {
121 printk(KERN_ERR "[nvdebug] Invalid BAR2/3 PDB configuration! BAR2/3 inaccessible.\n");
122 return ret;
123 }
124
125 *pdb = cached_pdb = g->regs + NV_PRAMIN + ret;
126 pd_hash = readl(cached_pdb);
127 *is_v2_pdb = cached_is_v2_pdb = pd_config.is_ver2;
128 104
129 return 0; 105 return 0;
130} 106}
diff --git a/mmu.c b/mmu.c
index e420864..70c00f9 100644
--- a/mmu.c
+++ b/mmu.c
@@ -1,117 +1,129 @@
1// Helpers to deal with NVIDIA's MMU and associated page tables 1/* Copyright 2024 Joshua Bakita
2 * Helpers to deal with NVIDIA's MMU and associated page tables
3 */
4#include <linux/err.h> // ERR_PTR() etc.
5#include <linux/iommu.h> // iommu_get_domain_for_dev() and iommu_iova_to_phys()
2#include <linux/kernel.h> // Kernel types 6#include <linux/kernel.h> // Kernel types
3 7
4#include "nvdebug.h" 8#include "nvdebug.h"
5 9
6/* One of the oldest ways to access video memory on NVIDIA GPUs is by using 10// Uncomment to print every PDE and PTE walked for debugging
7 a configurable 1MB window into VRAM which is mapped into BAR0 (register) 11//#define DEBUG
8 space starting at offset NV_PRAMIN. This is still supported on NVIDIA GPUs 12#ifdef DEBUG
9 and appear to be used today to bootstrap page table configuration. 13#define printk_debug printk
14#else
15#define printk_debug(...)
16#endif
10 17
11 Why is it mapped at a location called NVIDIA Private RAM Instance? Because 18/* Convert a page directory (PD) pointer and aperture to be kernel-accessible
12 this used to point to the entirety of intance RAM, which was seperate from
13 VRAM on older NVIDIA GPUs.
14*/
15 19
16/* Convert a physical VRAM address to an offset in the PRAMIN window 20 I/O MMU handling inspired by amdgpu_iomem_read() in amdgpu_ttm.c of the
17 @param addr VRAM address to convert 21 AMDGPU driver.
18 @return -errno on error, PRAMIN offset on success
19 22
20 Note: Use off2PRAMIN() instead if you want a dereferenceable address 23 @param addr Pointer from page directory entry (PDE)
21 Note: PRAMIN window is only 1MB, so returning an int is safe 24 @param pd_ap PD-type aperture (target address space) for `addr`
22*/ 25 @return A dereferencable kernel address, or an ERR_PTR-wrapped error
23static int vram2PRAMIN(struct nvdebug_state *g, uint64_t addr) { 26 */
24 uint64_t pramin_base_va; 27void __iomem *pd_deref(struct nvdebug_state *g, uintptr_t addr, enum PD_TARGET pd_ap) {
25 bar0_window_t window; 28 struct iommu_domain *dom;
26 window.raw = nvdebug_readl(g, NV_PBUS_BAR0_WINDOW); 29 phys_addr_t phys;
27 // Check if the address is valid (49 bits are addressable on-GPU) 30
28 if (addr & ~0x0001ffffffffffff) { 31 // Validate arguments
29 printk(KERN_ERR "[nvdebug] Invalid address %llx passed to %s!\n", 32 if (unlikely(!IS_PD_TARGET(pd_ap) || pd_ap == PD_AND_TARGET_INVALID || !addr))
30 addr, __func__); 33 return ERR_PTR(-EINVAL);
31 return -EINVAL; 34
35 // VID_MEM accesses are the simple common-case
36 if (pd_ap == PD_AND_TARGET_VID_MEM) {
37 // Using BAR2 requires a page-table traversal. As this function is part
38 // of the page-table traversal process, it must instead use PRAMIN.
39 int off = addr_to_pramin_mut(g, addr, TARGET_VID_MEM);
40 if (off < 0)
41 return ERR_PTR(off);
42 return g->regs + NV_PRAMIN + off;
32 } 43 }
33 // For unclear (debugging?) reasons, PRAMIN can point to SYSMEM 44 /* SYS_MEM accesses are rare. Only nvgpu (Jetson driver), nouveau, and this
34 if (window.target != TARGET_VID_MEM) 45 * driver are known to create page directory entries in SYS_MEM.
35 return -EFAULT; 46 *
36 pramin_base_va = ((uint64_t)window.base) << 16; 47 * On systems using an I/O MMU, or some other I/O virtual address space,
37 // Protect against out-of-bounds accesses 48 * these are **not** physical addresses, and must first be translated
38 if (addr < pramin_base_va || addr > pramin_base_va + NV_PRAMIN_LEN) 49 * through the I/O MMU before use.
39 return -ERANGE; 50 * Example default meaning of a SYS_MEM address for a few CPUs:
40 return addr - pramin_base_va; 51 * - Jetson Xavier : physical address
41} 52 * - AMD 3950X : I/O MMU address
53 * - Phenom II x4 : physical address
54 */
55 // Check for, and translate through, the I/O MMU (if any)
56 if ((dom = iommu_get_domain_for_dev(g->dev))) {
57 phys = iommu_iova_to_phys(dom, addr);
58 printk(KERN_ERR "[nvdebug] I/O MMU translated SYS_MEM I/O VA %#lx to physical address %llx.\n", addr, phys);
59 } else
60 phys = addr;
42 61
43// Convert a GPU physical address to CPU virtual address via the PRAMIN window 62 if (!phys)
44// @return A dereferencable address, or 0 (an invalid physical address) on err
45void __iomem *phy2PRAMIN(struct nvdebug_state* g, uint64_t phy) {
46 int off = vram2PRAMIN(g, phy);
47 if (off == -ERANGE)
48 printk(KERN_ERR "[nvdebug] Page table walk off end of PRAMIN!\n");
49 if (off < 0)
50 return 0; 63 return 0;
51 return g->regs + NV_PRAMIN + vram2PRAMIN(g, phy);
52}
53 64
54/* FIXME 65 return phys_to_virt(addr);
55void __iomem *off2BAR2(struct nvdebug_state* g, uint32_t off) {
56 return g->bar2 + off;
57} 66}
58*/
59 67
60// Internal helper for search_page_directory(). 68// Internal helper for search_page_directory().
61uint64_t search_page_directory_subtree(struct nvdebug_state *g, 69uint64_t search_page_directory_subtree(struct nvdebug_state *g,
62 void __iomem *pde_offset, 70 uintptr_t pde_addr,
63 void __iomem *(*off2addr)(struct nvdebug_state*, uint64_t), 71 enum PD_TARGET pde_target,
64 uint64_t addr_to_find, 72 uint64_t addr_to_find,
65 uint32_t level) { 73 uint32_t level) {
66 uint64_t res, i; 74 uint64_t res, i;
67 void __iomem *next; 75 void __iomem *pde_kern;
68 page_dir_entry_t entry; 76 page_dir_entry_t entry;
69 if (level > sizeof(NV_MMU_PT_V2_SZ)) 77 if (level > sizeof(NV_MMU_PT_V2_SZ))
70 return 0; 78 return 0;
71 // Hack to workaround PDE0 being double-size and strangely formatted 79 // Hack to workaround PDE0 being double-size and strangely formatted
72 if (NV_MMU_PT_V2_ENTRY_SZ[level] == 16) 80 if (NV_MMU_PT_V2_ENTRY_SZ[level] == 16)
73 pde_offset += 8; 81 pde_addr += 8;
74 entry.raw_w = readq(pde_offset); 82 // Translate a VID_MEM/SYS_MEM-space address to something kernel-accessible
83 pde_kern = pd_deref(g, pde_addr, pde_target);
84 if (IS_ERR_OR_NULL(pde_kern)) {
85 printk(KERN_ERR "[nvdebug] %s: Unable to resolve %#lx in GPU %s to a kernel-accessible address. Error %ld.\n", __func__, pde_addr, pd_target_to_text(pde_target), PTR_ERR(pde_kern));
86 return 0;
87 }