1 files changed, 150 insertions, 134 deletions
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c
index 06f24322e7c3..ded57dd538e2 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c
@@ -116,38 +116,43 @@ void amdgpu_vm_get_pd_bo(struct amdgpu_vm *vm,
 }
 /**
- * amdgpu_vm_get_bos - add the vm BOs to a duplicates list
+ * amdgpu_vm_validate_pt_bos - validate the page table BOs
 *
 * @adev: amdgpu device pointer
 * @vm: vm providing the BOs
- * @duplicates: head of duplicates list
+ * @validate: callback to do the validation
+ * @param: parameter for the validation callback
 *
- * Add the page directory to the BO duplicates list
+ * Validate the page table BOs on command submission if neccessary.
- * for command submission.
 */
-void amdgpu_vm_get_pt_bos(struct amdgpu_device *adev, struct amdgpu_vm *vm,
+int amdgpu_vm_validate_pt_bos(struct amdgpu_device *adev, struct amdgpu_vm *vm,
-                          struct list_head *duplicates)
+                              int (*validate)(void *p, struct amdgpu_bo *bo),
+                              void *param)
 {
        uint64_t num_evictions;
        unsigned i;
+        int r;
        /* We only need to validate the page tables
         * if they aren't already valid.
         */
        num_evictions = atomic64_read(&adev->num_evictions);
        if (num_evictions == vm->last_eviction_counter)
-                return;
+                return 0;
        /* add the vm page table to the list */
        for (i = 0; i <= vm->max_pde_used; ++i) {
-                struct amdgpu_bo_list_entry *entry = &vm->page_tables[i].entry;
+                struct amdgpu_bo *bo = vm->page_tables[i].bo;
-                if (!entry->robj)
+                if (!bo)
                        continue;
-                list_add(&entry->tv.head, duplicates);
+                r = validate(param, bo);
+                if (r)
+                        return r;
        }
+        return 0;
 }
 /**
@@ -166,12 +171,12 @@ void amdgpu_vm_move_pt_bos_in_lru(struct amdgpu_device *adev,
        spin_lock(&glob->lru_lock);
        for (i = 0; i <= vm->max_pde_used; ++i) {
-                struct amdgpu_bo_list_entry *entry = &vm->page_tables[i].entry;
+                struct amdgpu_bo *bo = vm->page_tables[i].bo;
-                if (!entry->robj)
+                if (!bo)
                        continue;
-                ttm_bo_move_to_lru_tail(&entry->robj->tbo);
+                ttm_bo_move_to_lru_tail(&bo->tbo);
        }
        spin_unlock(&glob->lru_lock);
 }
@@ -341,9 +346,9 @@ error:
 static bool amdgpu_vm_ring_has_compute_vm_bug(struct amdgpu_ring *ring)
 {
        struct amdgpu_device *adev = ring->adev;
-        const struct amdgpu_ip_block_version *ip_block;
+        const struct amdgpu_ip_block *ip_block;
-        if (ring->type != AMDGPU_RING_TYPE_COMPUTE)
+        if (ring->funcs->type != AMDGPU_RING_TYPE_COMPUTE)
                /* only compute rings */
                return false;
@@ -351,10 +356,10 @@ static bool amdgpu_vm_ring_has_compute_vm_bug(struct amdgpu_ring *ring)
        if (!ip_block)
                return false;
-        if (ip_block->major <= 7) {
+        if (ip_block->version->major <= 7) {
                /* gfx7 has no workaround */
                return true;
-        } else if (ip_block->major == 8) {
+        } else if (ip_block->version->major == 8) {
                if (adev->gfx.mec_fw_version >= 673)
                        /* gfx8 is fixed in MEC firmware 673 */
                        return false;
@@ -612,16 +617,26 @@ static uint64_t amdgpu_vm_map_gart(const dma_addr_t *pages_addr, uint64_t addr)
        return result;
 }
-static int amdgpu_vm_update_pd_or_shadow(struct amdgpu_device *adev,
+/*
-                                         struct amdgpu_vm *vm,
+ * amdgpu_vm_update_pdes - make sure that page directory is valid
-                                         bool shadow)
+ *
+ * @adev: amdgpu_device pointer
+ * @vm: requested vm
+ * @start: start of GPU address range
+ * @end: end of GPU address range
+ *
+ * Allocates new page tables if necessary
+ * and updates the page directory.
+ * Returns 0 for success, error for failure.
+ */
+int amdgpu_vm_update_page_directory(struct amdgpu_device *adev,
+                                    struct amdgpu_vm *vm)
 {
+        struct amdgpu_bo *shadow;
        struct amdgpu_ring *ring;
-        struct amdgpu_bo *pd = shadow ? vm->page_directory->shadow :
+        uint64_t pd_addr, shadow_addr;
-                vm->page_directory;
-        uint64_t pd_addr;
        uint32_t incr = AMDGPU_VM_PTE_COUNT * 8;
-        uint64_t last_pde = ~0, last_pt = ~0;
+        uint64_t last_pde = ~0, last_pt = ~0, last_shadow = ~0;
        unsigned count = 0, pt_idx, ndw;
        struct amdgpu_job *job;
        struct amdgpu_pte_update_params params;
@@ -629,15 +644,8 @@ static int amdgpu_vm_update_pd_or_shadow(struct amdgpu_device *adev,
        int r;
-        if (!pd)
-                return 0;
-        r = amdgpu_ttm_bind(&pd->tbo, &pd->tbo.mem);
-        if (r)
-                return r;
-        pd_addr = amdgpu_bo_gpu_offset(pd);
        ring = container_of(vm->entity.sched, struct amdgpu_ring, sched);
+        shadow = vm->page_directory->shadow;
        /* padding, etc. */
        ndw = 64;
@@ -645,6 +653,17 @@ static int amdgpu_vm_update_pd_or_shadow(struct amdgpu_device *adev,
        /* assume the worst case */
        ndw += vm->max_pde_used * 6;
+        pd_addr = amdgpu_bo_gpu_offset(vm->page_directory);
+        if (shadow) {
+                r = amdgpu_ttm_bind(&shadow->tbo, &shadow->tbo.mem);
+                if (r)
+                        return r;
+                shadow_addr = amdgpu_bo_gpu_offset(shadow);
+                ndw *= 2;
+        } else {
+                shadow_addr = 0;
+        }
        r = amdgpu_job_alloc_with_ib(adev, ndw * 4, &job);
        if (r)
                return r;
@@ -655,30 +674,26 @@ static int amdgpu_vm_update_pd_or_shadow(struct amdgpu_device *adev,
        /* walk over the address space and update the page directory */
        for (pt_idx = 0; pt_idx <= vm->max_pde_used; ++pt_idx) {
-                struct amdgpu_bo *bo = vm->page_tables[pt_idx].entry.robj;
+                struct amdgpu_bo *bo = vm->page_tables[pt_idx].bo;
                uint64_t pde, pt;
                if (bo == NULL)
                        continue;
                if (bo->shadow) {
-                        struct amdgpu_bo *shadow = bo->shadow;
+                        struct amdgpu_bo *pt_shadow = bo->shadow;
-                        r = amdgpu_ttm_bind(&shadow->tbo, &shadow->tbo.mem);
+                        r = amdgpu_ttm_bind(&pt_shadow->tbo,
+                                            &pt_shadow->tbo.mem);
                        if (r)
                                return r;
                }
                pt = amdgpu_bo_gpu_offset(bo);
-                if (!shadow) {
+                if (vm->page_tables[pt_idx].addr == pt)
-                        if (vm->page_tables[pt_idx].addr == pt)
+                        continue;
-                                continue;
-                        vm->page_tables[pt_idx].addr = pt;
+                vm->page_tables[pt_idx].addr = pt;
-                } else {
-                        if (vm->page_tables[pt_idx].shadow_addr == pt)
-                                continue;
-                        vm->page_tables[pt_idx].shadow_addr = pt;
-                }
                pde = pd_addr + pt_idx * 8;
                if (((last_pde + 8 * count) != pde) ||
@@ -686,6 +701,13 @@ static int amdgpu_vm_update_pd_or_shadow(struct amdgpu_device *adev,
                    (count == AMDGPU_VM_MAX_UPDATE_SIZE)) {
                        if (count) {
+                                if (shadow)
+                                        amdgpu_vm_do_set_ptes(&params,
+                                                              last_shadow,
+                                                              last_pt, count,
+                                                              incr,
+                                                              AMDGPU_PTE_VALID);
                                amdgpu_vm_do_set_ptes(&params, last_pde,
                                                      last_pt, count, incr,
                                                      AMDGPU_PTE_VALID);
@@ -693,34 +715,44 @@ static int amdgpu_vm_update_pd_or_shadow(struct amdgpu_device *adev,
                        count = 1;
                        last_pde = pde;
+                        last_shadow = shadow_addr + pt_idx * 8;
                        last_pt = pt;
                } else {
                        ++count;
                }
        }
-        if (count)
+        if (count) {
+                if (vm->page_directory->shadow)
+                        amdgpu_vm_do_set_ptes(&params, last_shadow, last_pt,
+                                              count, incr, AMDGPU_PTE_VALID);
                amdgpu_vm_do_set_ptes(&params, last_pde, last_pt,
                                      count, incr, AMDGPU_PTE_VALID);
+        }
-        if (params.ib->length_dw != 0) {
+        if (params.ib->length_dw == 0) {
-                amdgpu_ring_pad_ib(ring, params.ib);
+                amdgpu_job_free(job);
-                amdgpu_sync_resv(adev, &job->sync, pd->tbo.resv,
+                return 0;
+        }
+        amdgpu_ring_pad_ib(ring, params.ib);
+        amdgpu_sync_resv(adev, &job->sync, vm->page_directory->tbo.resv,
+                         AMDGPU_FENCE_OWNER_VM);
+        if (shadow)
+                amdgpu_sync_resv(adev, &job->sync, shadow->tbo.resv,
                                 AMDGPU_FENCE_OWNER_VM);
-                WARN_ON(params.ib->length_dw > ndw);
-                r = amdgpu_job_submit(job, ring, &vm->entity,
-                                      AMDGPU_FENCE_OWNER_VM, &fence);
-                if (r)
-                        goto error_free;
-                amdgpu_bo_fence(pd, fence, true);
+        WARN_ON(params.ib->length_dw > ndw);
-                fence_put(vm->page_directory_fence);
+        r = amdgpu_job_submit(job, ring, &vm->entity,
-                vm->page_directory_fence = fence_get(fence);
+                              AMDGPU_FENCE_OWNER_VM, &fence);
-                fence_put(fence);
+        if (r)
+                goto error_free;
-        } else {
+        amdgpu_bo_fence(vm->page_directory, fence, true);
-                amdgpu_job_free(job);
+        fence_put(vm->page_directory_fence);
-        }
+        vm->page_directory_fence = fence_get(fence);
+        fence_put(fence);
        return 0;
@@ -729,29 +761,6 @@ error_free:
        return r;
 }
-/*
- * amdgpu_vm_update_pdes - make sure that page directory is valid
- *
- * @adev: amdgpu_device pointer
- * @vm: requested vm
- * @start: start of GPU address range
- * @end: end of GPU address range
- *
- * Allocates new page tables if necessary
- * and updates the page directory.
- * Returns 0 for success, error for failure.
- */
-int amdgpu_vm_update_page_directory(struct amdgpu_device *adev,
-                                   struct amdgpu_vm *vm)
-{
-        int r;
-        r = amdgpu_vm_update_pd_or_shadow(adev, vm, true);
-        if (r)
-                return r;
-        return amdgpu_vm_update_pd_or_shadow(adev, vm, false);
-}
 /**
 * amdgpu_vm_update_ptes - make sure that page tables are valid
 *
@@ -781,11 +790,11 @@ static void amdgpu_vm_update_ptes(struct amdgpu_pte_update_params *params,
        /* initialize the variables */
        addr = start;
        pt_idx = addr >> amdgpu_vm_block_size;
-        pt = vm->page_tables[pt_idx].entry.robj;
+        pt = vm->page_tables[pt_idx].bo;
        if (params->shadow) {
                if (!pt->shadow)
                        return;
-                pt = vm->page_tables[pt_idx].entry.robj->shadow;
+                pt = pt->shadow;
        }
        if ((addr & ~mask) == (end & ~mask))
                nptes = end - addr;
@@ -804,11 +813,11 @@ static void amdgpu_vm_update_ptes(struct amdgpu_pte_update_params *params,
        /* walk over the address space and update the page tables */
        while (addr < end) {
                pt_idx = addr >> amdgpu_vm_block_size;
-                pt = vm->page_tables[pt_idx].entry.robj;
+                pt = vm->page_tables[pt_idx].bo;
                if (params->shadow) {
                        if (!pt->shadow)
                                return;
-                        pt = vm->page_tables[pt_idx].entry.robj->shadow;
+                        pt = pt->shadow;
                }
                if ((addr & ~mask) == (end & ~mask))
@@ -1065,8 +1074,8 @@ error_free:
 * @pages_addr: DMA addresses to use for mapping
 * @vm: requested vm
 * @mapping: mapped range and flags to use for the update
- * @addr: addr to set the area to
 * @flags: HW flags for the mapping
+ * @nodes: array of drm_mm_nodes with the MC addresses
 * @fence: optional resulting fence
 *
 * Split the mapping into smaller chunks so that each update fits
@@ -1079,12 +1088,11 @@ static int amdgpu_vm_bo_split_mapping(struct amdgpu_device *adev,
                                      dma_addr_t *pages_addr,
                                      struct amdgpu_vm *vm,
                                      struct amdgpu_bo_va_mapping *mapping,
-                                      uint32_t flags, uint64_t addr,
+                                      uint32_t flags,
+                                      struct drm_mm_node *nodes,
                                      struct fence **fence)
 {
-        const uint64_t max_size = 64ULL * 1024ULL * 1024ULL / AMDGPU_GPU_PAGE_SIZE;
+        uint64_t pfn, src = 0, start = mapping->it.start;
-        uint64_t src = 0, start = mapping->it.start;
        int r;
        /* normally,bo_va->flags only contians READABLE and WIRTEABLE bit go here
@@ -1097,23 +1105,40 @@ static int amdgpu_vm_bo_split_mapping(struct amdgpu_device *adev,
        trace_amdgpu_vm_bo_update(mapping);
-        if (pages_addr) {
+        pfn = mapping->offset >> PAGE_SHIFT;
-                if (flags == gtt_flags)
+        if (nodes) {
-                        src = adev->gart.table_addr + (addr >> 12) * 8;
+                while (pfn >= nodes->size) {
-                addr = 0;
+                        pfn -= nodes->size;
+                        ++nodes;
+                }
        }
-        addr += mapping->offset;
-        if (!pages_addr || src)
+        do {
-                return amdgpu_vm_bo_update_mapping(adev, exclusive,
+                uint64_t max_entries;
-                                                   src, pages_addr, vm,
+                uint64_t addr, last;
-                                                   start, mapping->it.last,
-                                                   flags, addr, fence);
+                if (nodes) {
+                        addr = nodes->start << PAGE_SHIFT;
+                        max_entries = (nodes->size - pfn) *
+                                (PAGE_SIZE / AMDGPU_GPU_PAGE_SIZE);
+                } else {
+                        addr = 0;
+                        max_entries = S64_MAX;
+                }
-        while (start != mapping->it.last + 1) {
+                if (pages_addr) {
-                uint64_t last;
+                        if (flags == gtt_flags)
+                                src = adev->gart.table_addr +
+                                        (addr >> AMDGPU_GPU_PAGE_SHIFT) * 8;
+                        else
+                                max_entries = min(max_entries, 16ull * 1024ull);
+                        addr = 0;
+                } else if (flags & AMDGPU_PTE_VALID) {
+                        addr += adev->vm_manager.vram_base_offset;
+                }
+                addr += pfn << PAGE_SHIFT;
-                last = min((uint64_t)mapping->it.last, start + max_size - 1);
+                last = min((uint64_t)mapping->it.last, start + max_entries - 1);
                r = amdgpu_vm_bo_update_mapping(adev, exclusive,
                                                src, pages_addr, vm,
                                                start, last, flags, addr,
@@ -1121,9 +1146,14 @@ static int amdgpu_vm_bo_split_mapping(struct amdgpu_device *adev,
                if (r)
                        return r;
+                pfn += last - start + 1;
+                if (nodes && nodes->size == pfn) {
+                        pfn = 0;
+                        ++nodes;
+                }
                start = last + 1;
-                addr += max_size * AMDGPU_GPU_PAGE_SIZE;
-        }
+        } while (unlikely(start != mapping->it.last + 1));
        return 0;
 }
@@ -1147,40 +1177,30 @@ int amdgpu_vm_bo_update(struct amdgpu_device *adev,
        dma_addr_t *pages_addr = NULL;
        uint32_t gtt_flags, flags;
        struct ttm_mem_reg *mem;
+        struct drm_mm_node *nodes;
        struct fence *exclusive;
-        uint64_t addr;
        int r;
        if (clear) {
                mem = NULL;
-                addr = 0;
+                nodes = NULL;
                exclusive = NULL;
        } else {
                struct ttm_dma_tt *ttm;
                mem = &bo_va->bo->tbo.mem;
-                addr = (u64)mem->start << PAGE_SHIFT;
+                nodes = mem->mm_node;
-                switch (mem->mem_type) {
+                if (mem->mem_type == TTM_PL_TT) {
-                case TTM_PL_TT:
                        ttm = container_of(bo_va->bo->tbo.ttm, struct
                                           ttm_dma_tt, ttm);
                        pages_addr = ttm->dma_address;
-                        break;
-                case TTM_PL_VRAM:
-                        addr += adev->vm_manager.vram_base_offset;
-                        break;
-                default:
-                        break;
                }
                exclusive = reservation_object_get_excl(bo_va->bo->tbo.resv);
        }
        flags = amdgpu_ttm_tt_pte_flags(adev, bo_va->bo->tbo.ttm, mem);
        gtt_flags = (amdgpu_ttm_is_bound(bo_va->bo->tbo.ttm) &&
-                adev == bo_va->bo->adev) ? flags : 0;
+                adev == amdgpu_ttm_adev(bo_va->bo->tbo.bdev)) ? flags : 0;
        spin_lock(&vm->status_lock);
        if (!list_empty(&bo_va->vm_status))
@@ -1190,7 +1210,7 @@ int amdgpu_vm_bo_update(struct amdgpu_device *adev,
        list_for_each_entry(mapping, &bo_va->invalids, list) {
                r = amdgpu_vm_bo_split_mapping(adev, exclusive,
                                               gtt_flags, pages_addr, vm,
-                                               mapping, flags, addr,
+                                               mapping, flags, nodes,
                                               &bo_va->last_pt_update);
                if (r)
                        return r;
@@ -1405,18 +1425,17 @@ int amdgpu_vm_bo_map(struct amdgpu_device *adev,
        /* walk over the address space and allocate the page tables */
        for (pt_idx = saddr; pt_idx <= eaddr; ++pt_idx) {
                struct reservation_object *resv = vm->page_directory->tbo.resv;
-                struct amdgpu_bo_list_entry *entry;
                struct amdgpu_bo *pt;
-                entry = &vm->page_tables[pt_idx].entry;
+                if (vm->page_tables[pt_idx].bo)
-                if (entry->robj)
                        continue;
                r = amdgpu_bo_create(adev, AMDGPU_VM_PTE_COUNT * 8,
                                     AMDGPU_GPU_PAGE_SIZE, true,
                                     AMDGPU_GEM_DOMAIN_VRAM,
                                     AMDGPU_GEM_CREATE_NO_CPU_ACCESS |
-                                     AMDGPU_GEM_CREATE_SHADOW,
+                                     AMDGPU_GEM_CREATE_SHADOW |
+                                     AMDGPU_GEM_CREATE_VRAM_CONTIGUOUS,
                                     NULL, resv, &pt);
                if (r)
                        goto error_free;
@@ -1442,11 +1461,7 @@ int amdgpu_vm_bo_map(struct amdgpu_device *adev,
                        }
                }
-                entry->robj = pt;
+                vm->page_tables[pt_idx].bo = pt;
-                entry->priority = 0;
-                entry->tv.bo = &entry->robj->tbo;
-                entry->tv.shared = true;
-                entry->user_pages = NULL;
                vm->page_tables[pt_idx].addr = 0;
        }
@@ -1626,7 +1641,8 @@ int amdgpu_vm_init(struct amdgpu_device *adev, struct amdgpu_vm *vm)
        r = amdgpu_bo_create(adev, pd_size, align, true,
                             AMDGPU_GEM_DOMAIN_VRAM,
                             AMDGPU_GEM_CREATE_NO_CPU_ACCESS |
-                             AMDGPU_GEM_CREATE_SHADOW,
+                             AMDGPU_GEM_CREATE_SHADOW |
+                             AMDGPU_GEM_CREATE_VRAM_CONTIGUOUS,
                             NULL, NULL, &vm->page_directory);
        if (r)
                goto error_free_sched_entity;
@@ -1697,7 +1713,7 @@ void amdgpu_vm_fini(struct amdgpu_device *adev, struct amdgpu_vm *vm)
        }
        for (i = 0; i < amdgpu_vm_num_pdes(adev); i++) {
-                struct amdgpu_bo *pt = vm->page_tables[i].entry.robj;
+                struct amdgpu_bo *pt = vm->page_tables[i].bo;
                if (!pt)
                        continue;