aboutsummaryrefslogtreecommitdiffstats
path: root/runlist.c
diff options
context:
space:
mode:
Diffstat (limited to 'runlist.c')
-rw-r--r--runlist.c275
1 files changed, 267 insertions, 8 deletions
diff --git a/runlist.c b/runlist.c
index 7bb2ee4..3076d27 100644
--- a/runlist.c
+++ b/runlist.c
@@ -1,19 +1,13 @@
1/* Copyright 2024 Joshua Bakita 1/* Copyright 2024 Joshua Bakita
2 * Helpers for dealing with the runlist and other Host (PFIFO) registers 2 * Helpers for dealing with the runlist and other Host (PFIFO) registers
3 */ 3 */
4#include <linux/iommu.h> // iommu_get_domain_for_dev() and iommu_iova_to_phys()
4#include <linux/printk.h> // For printk() 5#include <linux/printk.h> // For printk()
5#include <asm/errno.h> // For error defines 6#include <asm/errno.h> // For error defines
6#include <asm/io.h> // For phys_to_virt() 7#include <asm/io.h> // For phys_to_virt()
7 8
8#include "nvdebug.h" 9#include "nvdebug.h"
9 10
10// Uncomment to, upon BAR2 access failure, return a PRAMIN-based runlist pointer
11// in get_runlist_iter(). In order for this pointer to remain valid, PRAMIN
12// **must** not be moved during runlist traversal.
13// The Jetson TX2 has no BAR2, and stores the runlist in VID_MEM, so this must
14// be enabled to print the runlist on the TX2.
15//#define FALLBACK_TO_PRAMIN
16
17/* Get RunList RAM (RLRAM) offset for a runlist from the device topology 11/* Get RunList RAM (RLRAM) offset for a runlist from the device topology
18 @param rl_id Which runlist to obtain [numbered in order of appearance in 12 @param rl_id Which runlist to obtain [numbered in order of appearance in
19 the device topology (PTOP) registers] 13 the device topology (PTOP) registers]
@@ -116,6 +110,7 @@ int get_runlist_iter(struct nvdebug_state *g, int rl_id, struct runlist_iter *rl
116 runlist_len = submit.len; 110 runlist_len = submit.len;
117 printk(KERN_INFO "[nvdebug] Runlist %d for %x: %d entries @ %llx in %s (config raw: %#018llx %#018llx)\n", 111 printk(KERN_INFO "[nvdebug] Runlist %d for %x: %d entries @ %llx in %s (config raw: %#018llx %#018llx)\n",
118 rl_id, g->chip_id, submit.len, runlist_iova, target_to_text(runlist_target), base.raw, submit.raw); 112 rl_id, g->chip_id, submit.len, runlist_iova, target_to_text(runlist_target), base.raw, submit.raw);
113 printk(KERN_INFO "[nvdebug] Runlist offset is %d\n", submit.offset);
119 rl_iter->runlist_pri_base = runlist_pri_base; 114 rl_iter->runlist_pri_base = runlist_pri_base;
120 } 115 }
121 // Return early on an empty runlist 116 // Return early on an empty runlist
@@ -130,6 +125,12 @@ int get_runlist_iter(struct nvdebug_state *g, int rl_id, struct runlist_iter *rl
130 if ((err = get_bar2_pdb(g, &pd_config)) < 0) 125 if ((err = get_bar2_pdb(g, &pd_config)) < 0)
131 goto attempt_pramin_access; 126 goto attempt_pramin_access;
132 127
128 // XXX: PD version detection not working on Hopper [is_ver2 errantly (?) unset]
129 if (g->chip_id >= NV_CHIP_ID_HOPPER && g->chip_id < NV_CHIP_ID_ADA) {
130 printk(KERN_WARNING "[nvdebug] V3 page tables do not currently work on Hopper! Mystery config: %llx\n", pd_config.raw);
131 err = -EOPNOTSUPP;
132 goto attempt_pramin_access;
133 }
133 if (pd_config.is_ver2) 134 if (pd_config.is_ver2)
134 runlist_bar_vaddr = search_page_directory(g, pd_config, runlist_iova, TARGET_VID_MEM); 135 runlist_bar_vaddr = search_page_directory(g, pd_config, runlist_iova, TARGET_VID_MEM);
135 else 136 else
@@ -233,7 +234,7 @@ int preempt_runlist(struct nvdebug_state *g, uint32_t rl_id) {
233} 234}
234 235
235// Read and write runlist configuration, triggering a resubmit 236// Read and write runlist configuration, triggering a resubmit
236int resubmit_runlist(struct nvdebug_state *g, uint32_t rl_id) { 237int resubmit_runlist(struct nvdebug_state *g, uint32_t rl_id, uint32_t off) {
237 // Necessary registers do not exist pre-Fermi 238 // Necessary registers do not exist pre-Fermi
238 if (g->chip_id < NV_CHIP_ID_FERMI) 239 if (g->chip_id < NV_CHIP_ID_FERMI)
239 return -EOPNOTSUPP; 240 return -EOPNOTSUPP;
@@ -252,6 +253,9 @@ int resubmit_runlist(struct nvdebug_state *g, uint32_t rl_id) {
252 return -EINVAL; 253 return -EINVAL;
253 if ((submit.raw = nvdebug_readq(g, NV_PFIFO_RUNLIST_SUBMIT_TU102(rl_id))) == -1) 254 if ((submit.raw = nvdebug_readq(g, NV_PFIFO_RUNLIST_SUBMIT_TU102(rl_id))) == -1)
254 return -EIO; 255 return -EIO;
256 preempt_runlist(g, rl_id);
257 if (off != -1)
258 submit.offset = off;
255 nvdebug_writeq(g, NV_PFIFO_RUNLIST_SUBMIT_TU102(rl_id), submit.raw); 259 nvdebug_writeq(g, NV_PFIFO_RUNLIST_SUBMIT_TU102(rl_id), submit.raw);
256 } else { 260 } else {
257 int err; 261 int err;
@@ -261,6 +265,9 @@ int resubmit_runlist(struct nvdebug_state *g, uint32_t rl_id) {
261 return err; 265 return err;
262 if ((submit.raw = nvdebug_readq(g, runlist_pri_base + NV_RUNLIST_SUBMIT_GA100)) == -1) 266 if ((submit.raw = nvdebug_readq(g, runlist_pri_base + NV_RUNLIST_SUBMIT_GA100)) == -1)
263 return -EIO; 267 return -EIO;
268 preempt_runlist(g, rl_id);
269 if (off != -1)
270 submit.offset = off;
264 // On Ampere, this does not appear to trigger a preempt of the 271 // On Ampere, this does not appear to trigger a preempt of the
265 // currently-running channel (even if the currently running channel 272 // currently-running channel (even if the currently running channel
266 // becomes disabled), but will cause newly re-enabled channels 273 // becomes disabled), but will cause newly re-enabled channels
@@ -270,3 +277,255 @@ int resubmit_runlist(struct nvdebug_state *g, uint32_t rl_id) {
270 } 277 }
271 return 0; 278 return 0;
272} 279}
280
281/* Get a CPU-accessible pointer to an arbitrary-address-space instance block
282 @param instance_addr Address of instance block
283 @param intasce_target Aperture/taget of instance block address
284 @return A dereferencable KVA, NULL if not found, or an ERR_PTR-wrapped error
285
286 Note: The returned address will be a BAR2 or physical address, mapped into
287 kernel space, /not/ a PRAMIN-derived address. Thus, the returned
288 address will have an indefinite lifetime, and will be uneffected by use
289 of PRAMIN elsewhere (such as to read the CTXSW block).
290*/
291instance_ctrl_t *instance_deref(struct nvdebug_state *g, uint64_t instance_addr,
292 enum INST_TARGET instance_target) {
293 if (!instance_addr || instance_target == TARGET_INVALID)
294 return ERR_PTR(-EINVAL);
295 if (instance_target == TARGET_VID_MEM) {
296 int err;
297 uint64_t inst_bar_vaddr;
298 page_dir_config_t pd_config;
299 // Only access VID_MEM via BAR2; do not fall back to PRAMIN
300 if (!g->bar2)
301 return NULL;
302 // Find page tables which define how BAR2/3 offsets are translated to
303 // physical VID/SYS_MEM addresses.
304 if ((err = get_bar2_pdb(g, &pd_config)) < 0) {
305 printk(KERN_ERR "[nvdebug] Error: Unable to access page directory "
306 "configuration for BAR2/3. Error %d.\n", err);
307 return ERR_PTR(err);
308 }
309 // Search the BAR2/3 page tables for the offset at which the instance
310 // block is mapped (reverse translation).
311 if (pd_config.is_ver2)
312 inst_bar_vaddr = search_page_directory(g, pd_config, instance_addr, instance_target);
313 else
314 inst_bar_vaddr = search_v1_page_directory(g, pd_config, instance_addr, instance_target);
315 if (!inst_bar_vaddr) {
316 printk(KERN_WARNING "[nvdebug] Warning: Instance block %#018llx "
317 "(%s) appears unmapped in BAR2/3.\n", instance_addr,
318 target_to_text(instance_target));
319 return NULL;
320 }
321 return g->bar2 + inst_bar_vaddr;
322 } else {
323 struct iommu_domain *dom;
324 // SYS_MEM addresses are physical addresses *from the perspective of
325 // the device* ("bus addresses"), and may not necessarially correspond
326 // to physical addresses from the perspective of the CPU. The I/O MMU
327 // is responsible for mapping bus addresses to CPU-relative physical
328 // addresses when there is no direct correspondence. If an I/O MMU is
329 // enabled on this GPU, ask it to translate the bus address to a
330 // CPU-relative physical address.
331 if ((dom = iommu_get_domain_for_dev(g->dev))) {
332 // XXX: As of Aug 2024, this is not tested, so include extra logging
333 printk(KERN_DEBUG "[nvdebug] I/O MMU translated SYS_MEM I/O VA %#llx for instance block", instance_addr);
334 if (!(instance_addr = iommu_iova_to_phys(dom, instance_addr))) {
335 printk(KERN_ERR "[nvdebug] Error: I/O MMU failed to translate "
336 "%#018llx (%s) to a CPU-relative physical address.\n",
337 instance_addr, target_to_text(instance_target));
338 return ERR_PTR(-EADDRNOTAVAIL);
339 }
340 printk(KERN_DEBUG " to physical address %#llx.\n", instance_addr);
341 }
342 // Convert from a physical address to a kernel virtual address (KVA)
343 return phys_to_virt(instance_addr);
344 }
345}
346
347/* Get a CPU-accessible pointer to the CTXSW block for a channel intance block
348 @param inst Dereferencable pointer to the start of a complete instance block
349 @return A dereferencable KVA, NULL if not found, or an ERR_PTR-wrapped error
350
351 Note: The returned address **will** be a PRAMIN-based address. Any changes to
352 PRAMIN **will** invalidate the returned pointer. `inst` **cannot** be a
353 pointer into the PRAMIN space.
354*/
355context_switch_ctrl_t *get_ctxsw(struct nvdebug_state *g,
356 instance_ctrl_t *inst) {
357 int err;
358 context_switch_ctrl_t *wfi = NULL;
359 uint64_t wfi_virt, wfi_phys, ctxsw_virt, ctxsw_phys;
360 enum INST_TARGET wfi_phys_aperture, ctxsw_phys_aperture;
361
362 // The WFI block contains a pointer to the CTXSW block, which contains the
363 // preemption mode configuration for the context. (As best I can tell, the WFI
364 // block is subcontext-specific, whereas the CTXSW block is context-wide.
365 wfi_virt = (uint64_t)inst->engine_wfi_ptr << 12;
366
367 // WFI may not be configured
368 if (!wfi_virt)
369 goto out;
370
371 // Determine the physical location of the WFI block
372 if (inst->engine_wfi_is_virtual) {
373 if (inst->pdb.is_ver2)
374 err = translate_page_directory(g, inst->pdb, wfi_virt, &wfi_phys, &wfi_phys_aperture);
375 else
376 err = translate_v1_page_directory(g, inst->pdb, wfi_virt, &wfi_phys, &wfi_phys_aperture);
377 if (err) {
378 printk(KERN_ERR "[nvdebug] Critical: Inconsistent GPU state; WFI block "
379 "pointer %#018llx (virt) cannot be found in process page tables! "
380 "Translation error %d.\n", wfi_virt, -err);
381 return ERR_PTR(-ENOTRECOVERABLE);
382 }
383 } else {
384 wfi_phys = (uint64_t)inst->engine_wfi_ptr << 12;
385 wfi_phys_aperture = inst->engine_wfi_target;
386 }
387
388 // Get a dereferencible pointer to the WFI block (the WFI and CTXSW blocks
389 // have not been observed as mapped in BAR2/3, so we use the PRAMIN window).
390 // Note: On Jetson boards, we could attempt to avoid PRAMIN since CTXSW is in
391 // SYS_MEM, but this function will always need to use PRAMIN to work
392 // around the WFI and CTXSW blocks not being accessible via BAR2/3 on
393 // PCIe GPU, so always use PRAMIN for simplicity.
394 if ((wfi_phys = addr_to_pramin_mut(g, wfi_phys, wfi_phys_aperture)) == -1)
395 goto out;
396 wfi = g->regs + wfi_phys + NV_PRAMIN;
397
398// XXX
399// return wfi;
400// End XXX
401
402 // While the WFI block uses the same layout as the context switch (CTXSW)
403 // control block, it is mostly unpopulated except for a few pointers on GPUs
404 // after Volta. This appears to be related to subcontexts, where each
405 // subcontext has its own WFI block containing a pointer to the overarching
406 // CTXSW block. Only attempt to find the overarching CTXSW block if at least