/* Impose limit on how much memory KFD can use */ staticstruct {
uint64_t max_system_mem_limit;
uint64_t max_ttm_mem_limit;
int64_t system_mem_used;
int64_t ttm_mem_used;
spinlock_t mem_limit_lock;
} kfd_mem_limit;
if (kfd_mem_limit.system_mem_used + system_mem_needed >
kfd_mem_limit.max_system_mem_limit) {
pr_debug("Set no_system_mem_limit=1 if using shared memory\n"); if (!no_system_mem_limit) {
ret = -ENOMEM; goto release;
}
}
if (kfd_mem_limit.ttm_mem_used + ttm_mem_needed >
kfd_mem_limit.max_ttm_mem_limit) {
ret = -ENOMEM; goto release;
}
/*if is_app_apu is false and apu_prefer_gtt is true, it is an APU with *carveout<gtt.Inthatcase,VRAMallocationwillgotogttdomain,skip *VRAMchecksincettm_mem_limitcheckalreadycoverthisallocation
*/
/* TODO: Instead of block before we should use the fence of the page *tableupdateandTLBflushheredirectly.
*/
replacement = dma_fence_get_stub();
dma_resv_replace_fences(bo->tbo.base.resv, ef->base.context,
replacement, DMA_RESV_USAGE_BOOKKEEP);
dma_fence_put(replacement); return0;
}
if (WARN_ON(ttm->num_pages != src_ttm->num_pages)) return -EINVAL;
ttm->sg = kmalloc(sizeof(*ttm->sg), GFP_KERNEL); if (unlikely(!ttm->sg)) return -ENOMEM;
/* Same sequence as in amdgpu_ttm_tt_pin_userptr */
ret = sg_alloc_table_from_pages(ttm->sg, src_ttm->pages,
ttm->num_pages, 0,
(u64)ttm->num_pages << PAGE_SHIFT,
GFP_KERNEL); if (unlikely(ret)) goto free_sg;
ret = dma_map_sgtable(adev->dev, ttm->sg, direction, 0); if (unlikely(ret)) goto release_sg;
amdgpu_bo_placement_from_domain(bo, AMDGPU_GEM_DOMAIN_GTT);
ret = ttm_bo_validate(&bo->tbo, &bo->placement, &ctx); if (ret) goto unmap_sg;
/* Expect SG Table of dmapmap BO to be NULL */
mmio = (mem->alloc_flags & KFD_IOC_ALLOC_MEM_FLAGS_MMIO_REMAP); if (unlikely(ttm->sg)) {
pr_err("SG Table of %d BO for peer device is UNEXPECTEDLY NON-NULL", mmio); return -EINVAL;
}
dir = mem->alloc_flags & KFD_IOC_ALLOC_MEM_FLAGS_WRITABLE ?
DMA_BIDIRECTIONAL : DMA_TO_DEVICE;
dma_addr = mem->bo->tbo.sg->sgl->dma_address;
pr_debug("%d BO size: %d\n", mmio, mem->bo->tbo.sg->sgl->length);
pr_debug("%d BO address before DMA mapping: %llx\n", mmio, dma_addr);
dma_addr = dma_map_resource(adev->dev, dma_addr,
mem->bo->tbo.sg->sgl->length, dir, DMA_ATTR_SKIP_CPU_SYNC);
ret = dma_mapping_error(adev->dev, dma_addr); if (unlikely(ret)) return ret;
pr_debug("%d BO address after DMA mapping: %llx\n", mmio, dma_addr);
ttm->sg = create_sg_table(dma_addr, mem->bo->tbo.sg->sgl->length); if (unlikely(!ttm->sg)) {
ret = -ENOMEM; goto unmap_sg;
}
amdgpu_bo_placement_from_domain(bo, AMDGPU_GEM_DOMAIN_GTT);
ret = ttm_bo_validate(&bo->tbo, &bo->placement, &ctx); if (unlikely(ret)) goto free_sg;
staticvoid
kfd_mem_dmaunmap_dmabuf(struct kfd_mem_attachment *attachment)
{ /* This is a no-op. We don't want to trigger eviction fences when *unmappingDMABufs.Thereforetheinvalidation(movingtosystem *domain)isdoneinkfd_mem_dmamap_dmabuf.
*/
}
/* kfd_mem_attach - Add a BO to a VM * *EverythingthatneedstobodoneonlyoncewhenaBOisfirstadded *toaVM.Itcanlaterbemappedandunmappedmanytimeswithout *repeatingthesesteps. * *0.CreateBOforDMAmapping,ifneeded *1.AllocateandinitializeBOVAentrydatastructure *2.AddBOtotheVM *3.DetermineASIC-specificPTEflags *4.Allocpagetablesanddirectoriesifneeded *4a.Validatenewpagetablesanddirectories
*/ staticint kfd_mem_attach(struct amdgpu_device *adev, struct kgd_mem *mem, struct amdgpu_vm *vm, bool is_aql)
{ struct amdgpu_device *bo_adev = amdgpu_ttm_adev(mem->bo->tbo.bdev); unsignedlong bo_size = mem->bo->tbo.base.size;
uint64_t va = mem->va; struct kfd_mem_attachment *attachment[2] = {NULL, NULL}; struct amdgpu_bo *bo[2] = {NULL, NULL}; struct amdgpu_bo_va *bo_va; bool same_hive = false; int i, ret;
if (!va) {
pr_err("Invalid VA when adding BO to VM\n"); return -EINVAL;
}
/* Determine access to VRAM, MMIO and DOORBELL BOs of peer devices * *TheaccesspathofMMIOandDOORBELLBOsofisalwaysoverPCIe. *IncontrasttheaccesspathofVRAMBOsdepensuponthetypeof *linkthatconnectsthepeerdevice.AccessoverPCIeisallowed *ifpeerdevicehaslargeBAR.Incontrast,accessoverxGMIis *allowedforbothsmallandlargeBARconfigurationsofpeerdevice
*/ if ((adev != bo_adev && !adev->apu_prefer_gtt) &&
((mem->domain == AMDGPU_GEM_DOMAIN_VRAM) ||
(mem->alloc_flags & KFD_IOC_ALLOC_MEM_FLAGS_DOORBELL) ||
(mem->alloc_flags & KFD_IOC_ALLOC_MEM_FLAGS_MMIO_REMAP))) { if (mem->domain == AMDGPU_GEM_DOMAIN_VRAM)
same_hive = amdgpu_xgmi_same_hive(adev, bo_adev); if (!same_hive && !amdgpu_device_is_peer_accessible(bo_adev, adev)) return -EINVAL;
}
for (i = 0; i <= is_aql; i++) {
attachment[i] = kzalloc(sizeof(*attachment[i]), GFP_KERNEL); if (unlikely(!attachment[i])) {
ret = -ENOMEM; goto unwind;
}
pr_debug("\t add VA 0x%llx - 0x%llx to vm %p\n", va,
va + bo_size, vm);
if ((adev == bo_adev && !(mem->alloc_flags & KFD_IOC_ALLOC_MEM_FLAGS_MMIO_REMAP)) ||
(amdgpu_ttm_tt_get_usermm(mem->bo->tbo.ttm) && reuse_dmamap(adev, bo_adev)) ||
(mem->domain == AMDGPU_GEM_DOMAIN_GTT && reuse_dmamap(adev, bo_adev)) ||
same_hive) { /* Mappings on the local GPU, or VRAM mappings in the *localhive,oruserptr,orGTTmappingcanreusedmamap *addressspacesharetheoriginalBO
*/
attachment[i]->type = KFD_MEM_ATT_SHARED;
bo[i] = mem->bo;
drm_gem_object_get(&bo[i]->tbo.base);
} elseif (i > 0) { /* Multiple mappings on the same GPU share the BO */
attachment[i]->type = KFD_MEM_ATT_SHARED;
bo[i] = bo[0];
drm_gem_object_get(&bo[i]->tbo.base);
} elseif (amdgpu_ttm_tt_get_usermm(mem->bo->tbo.ttm)) { /* Create an SG BO to DMA-map userptrs on other GPUs */
attachment[i]->type = KFD_MEM_ATT_USERPTR;
ret = create_dmamap_sg_bo(adev, mem, &bo[i]); if (ret) goto unwind; /* Handle DOORBELL BOs of peer devices and MMIO BOs of local and peer devices */
} elseif (mem->bo->tbo.type == ttm_bo_type_sg) {
WARN_ONCE(!(mem->alloc_flags & KFD_IOC_ALLOC_MEM_FLAGS_DOORBELL ||
mem->alloc_flags & KFD_IOC_ALLOC_MEM_FLAGS_MMIO_REMAP), "Handing invalid SG BO in ATTACH request");
attachment[i]->type = KFD_MEM_ATT_SG;
ret = create_dmamap_sg_bo(adev, mem, &bo[i]); if (ret) goto unwind; /* Enable acces to GTT and VRAM BOs of peer devices */
} elseif (mem->domain == AMDGPU_GEM_DOMAIN_GTT ||
mem->domain == AMDGPU_GEM_DOMAIN_VRAM) {
attachment[i]->type = KFD_MEM_ATT_DMABUF;
ret = kfd_mem_attach_dmabuf(adev, mem, &bo[i]); if (ret) goto unwind;
pr_debug("Employ DMABUF mechanism to enable peer GPU access\n");
} else {
WARN_ONCE(true, "Handling invalid ATTACH request");
ret = -EINVAL; goto unwind;
}
/* Add BO to VM internal data structures */
ret = amdgpu_bo_reserve(bo[i], false); if (ret) {
pr_debug("Unable to reserve BO during memory attach"); goto unwind;
}
bo_va = amdgpu_vm_bo_find(vm, bo[i]); if (!bo_va)
bo_va = amdgpu_vm_bo_add(adev, vm, bo[i]); else
++bo_va->ref_count;
attachment[i]->bo_va = bo_va;
amdgpu_bo_unreserve(bo[i]); if (unlikely(!attachment[i]->bo_va)) {
ret = -ENOMEM;
pr_err("Failed to add BO object to VM. ret == %d\n",
ret); goto unwind;
}
attachment[i]->va = va;
attachment[i]->pte_flags = get_pte_flags(adev, mem);
attachment[i]->adev = adev;
list_add(&attachment[i]->list, &mem->attachments);
va += bo_size;
}
return0;
unwind: for (; i >= 0; i--) { if (!attachment[i]) continue; if (attachment[i]->bo_va) {
(void)amdgpu_bo_reserve(bo[i], true); if (--attachment[i]->bo_va->ref_count == 0)
amdgpu_vm_bo_del(adev, attachment[i]->bo_va);
amdgpu_bo_unreserve(bo[i]);
list_del(&attachment[i]->list);
} if (bo[i])
drm_gem_object_put(&bo[i]->tbo.base);
kfree(attachment[i]);
} return ret;
}
ret = amdgpu_ttm_tt_get_user_pages(bo, bo->tbo.ttm->pages, &range); if (ret) { if (ret == -EAGAIN)
pr_debug("Failed to get user pages, try again\n"); else
pr_err("%s: Failed to get user pages: %d\n", __func__, ret); goto unregister_out;
}
ret = amdgpu_bo_reserve(bo, true); if (ret) {
pr_err("%s: Failed to reserve BO\n", __func__); goto release_out;
}
amdgpu_bo_placement_from_domain(bo, mem->domain);
ret = ttm_bo_validate(&bo->tbo, &bo->placement, &ctx); if (ret)
pr_err("%s: failed to validate BO\n", __func__);
amdgpu_bo_unreserve(bo);
/* Reserving a BO and its page table BOs must happen atomically to *avoiddeadlocks.SomeoperationsupdatemultipleVMsatonce.Track *allthereservationinfoinacontextstructure.Optionallyasync *objectcantrackVMupdates.
*/ struct bo_vm_reservation_context { /* DRM execution context for the reservation */ struct drm_exec exec; /* Number of VMs reserved */ unsignedint n_vms; /* Pointer to sync object */ struct amdgpu_sync *sync;
};
enum bo_vm_match {
BO_VM_NOT_MAPPED = 0, /* Match VMs where a BO is not mapped */
BO_VM_MAPPED, /* Match VMs where a BO is mapped */
BO_VM_ALL, /* Match all VMs a BO was added to */
};
/* Set virtual address for the allocation */
ret = amdgpu_vm_bo_map(entry->adev, entry->bo_va, entry->va, 0,
amdgpu_bo_size(entry->bo_va->base.bo),
entry->pte_flags); if (ret) {
pr_err("Failed to map VA 0x%llx in vm. ret %d\n",
entry->va, ret); return ret;
}
if (no_update_pte) return0;
ret = update_gpuvm_pte(mem, entry, sync); if (ret) {
pr_err("update_gpuvm_pte() failed\n"); goto update_gpuvm_pte_failed;
}
/* Validate page directory and attach eviction fence */
ret = amdgpu_bo_reserve(vm->root.bo, true); if (ret) goto reserve_pd_fail;
ret = vm_validate_pt_pd_bos(vm, NULL); if (ret) {
pr_err("validate_pt_pd_bos() failed\n"); goto validate_pd_fail;
}
ret = amdgpu_bo_sync_wait(vm->root.bo,
AMDGPU_FENCE_OWNER_KFD, false); if (ret) goto wait_pd_fail;
ret = dma_resv_reserve_fences(vm->root.bo->tbo.base.resv, 1); if (ret) goto reserve_shared_fail;
dma_resv_add_fence(vm->root.bo->tbo.base.resv,
&vm->process_info->eviction_fence->base,
DMA_RESV_USAGE_BOOKKEEP);
amdgpu_bo_unreserve(vm->root.bo);
/* Update process info */
mutex_lock(&vm->process_info->lock);
list_add_tail(&vm->vm_list_node,
&(vm->process_info->vm_list_head));
vm->process_info->n_vms++; if (ef)
*ef = dma_fence_get(&vm->process_info->eviction_fence->base);
mutex_unlock(&vm->process_info->lock);
/* Update process info */
mutex_lock(&process_info->lock);
process_info->n_vms--;
list_del(&vm->vm_list_node);
mutex_unlock(&process_info->lock);
vm->process_info = NULL;
/* Release per-process resources when last compute VM is destroyed */ if (!process_info->n_vms) {
WARN_ON(!list_empty(&process_info->kfd_bo_list));
WARN_ON(!list_empty(&process_info->userptr_valid_list));
WARN_ON(!list_empty(&process_info->userptr_inval_list));
/* Unpin MMIO/DOORBELL BO's that were pinned during allocation */ if (mem->alloc_flags &
(KFD_IOC_ALLOC_MEM_FLAGS_DOORBELL |
KFD_IOC_ALLOC_MEM_FLAGS_MMIO_REMAP)) {
amdgpu_amdkfd_gpuvm_unpin_bo(mem->bo);
}
mapped_to_gpu_memory = mem->mapped_to_gpu_memory;
is_imported = mem->is_imported;
mutex_unlock(&mem->lock); /* lock is not needed after this, since mem is unused and will *befreedanyway
*/
if (mapped_to_gpu_memory > 0) {
pr_debug("BO VA 0x%llx size 0x%lx is still mapped.\n",
mem->va, bo_size); return -EBUSY;
}
/* Make sure restore workers don't access the BO any more */
mutex_lock(&process_info->lock);
list_del(&mem->validate_list);
mutex_unlock(&process_info->lock);
/* Cleanup user pages and MMU notifiers */ if (amdgpu_ttm_tt_get_usermm(mem->bo->tbo.ttm)) {
amdgpu_hmm_unregister(mem->bo);
mutex_lock(&process_info->notifier_lock);
amdgpu_ttm_tt_discard_user_pages(mem->bo->tbo.ttm, mem->range);
mutex_unlock(&process_info->notifier_lock);
}
ret = reserve_bo_and_cond_vms(mem, NULL, BO_VM_ALL, &ctx); if (unlikely(ret)) return ret;
/* Remove from VM internal data structures */
list_for_each_entry_safe(entry, tmp, &mem->attachments, list) {
kfd_mem_dmaunmap_attachment(mem, entry);
kfd_mem_detach(entry);
}
ret = unreserve_bo_and_vms(&ctx, false, false);
/* Free the sync object */
amdgpu_sync_free(&mem->sync);
/* If the SG is not NULL, it's one we created for a doorbell or mmio *remapBO.Weneedtofreeit.
*/ if (mem->bo->tbo.sg) {
sg_free_table(mem->bo->tbo.sg);
kfree(mem->bo->tbo.sg);
}
/* Update the size of the BO being freed if it was allocated from *VRAMandisnotimported.ForAPPAPUVRAMallocationsaredone *inGTTdomain
*/ if (size) { if (!is_imported &&
(mem->bo->preferred_domains == AMDGPU_GEM_DOMAIN_VRAM ||
(adev->apu_prefer_gtt &&
mem->bo->preferred_domains == AMDGPU_GEM_DOMAIN_GTT)))
*size = bo_size; else
*size = 0;
}
/* Free the BO*/
drm_vma_node_revoke(&mem->bo->tbo.base.vma_node, drm_priv);
drm_gem_handle_delete(adev->kfd.client.file, mem->gem_handle); if (mem->dmabuf) {
dma_buf_put(mem->dmabuf);
mem->dmabuf = NULL;
}
mutex_destroy(&mem->lock);
/* If this releases the last reference, it will end up calling *amdgpu_amdkfd_release_notifyandkfreethememstruct.That'swhy *thisneedstobethelastcallhere.
*/
drm_gem_object_put(&mem->bo->tbo.base);
/* *Forkgd_memallocatedinamdgpu_amdkfd_gpuvm_import_dmabuf(), *explicitlyfreeithere.
*/ if (!use_release_notifier)
kfree(mem);
bo = mem->bo; if (!bo) {
pr_err("Invalid BO when mapping memory to GPU\n"); return -EINVAL;
}
/* Make sure restore is not running concurrently. Since we *don'tmapinvaliduserptrBOs,werelyonthenextrestore *workertodothemapping
*/
mutex_lock(&mem->process_info->lock);
/* Lock notifier lock. If we find an invalid userptr BO, we can be *surethattheMMUnotifierisnolongerrunning *concurrentlyandthequeuesareactuallystopped
*/ if (amdgpu_ttm_tt_get_usermm(bo->tbo.ttm)) {
mutex_lock(&mem->process_info->notifier_lock);
is_invalid_userptr = !!mem->invalid;
mutex_unlock(&mem->process_info->notifier_lock);
}
pr_debug("Map VA 0x%llx - 0x%llx to vm %p domain %s\n",
mem->va,
mem->va + bo_size * (1 + mem->aql_queue),
avm, domain_string(domain));
if (!kfd_mem_is_attached(avm, mem)) {
ret = kfd_mem_attach(adev, mem, avm, mem->aql_queue); if (ret) goto out;
}
ret = reserve_bo_and_vm(mem, avm, &ctx); if (unlikely(ret)) goto out;
/* Userptr can be marked as "not invalid", but not actually be *validatedyet(stillinthesystemdomain).Inthatcase *thequeuesarestillstoppedandwecanleavemappingfor *thenextrestoreworker
*/ if (amdgpu_ttm_tt_get_usermm(bo->tbo.ttm) &&
bo->tbo.resource->mem_type == TTM_PL_SYSTEM)
is_invalid_userptr = true;
ret = vm_validate_pt_pd_bos(avm, NULL); if (unlikely(ret)) goto out_unreserve;
ret = reserve_bo_and_cond_vms(mem, avm, BO_VM_MAPPED, &ctx); if (unlikely(ret)) goto out; /* If no VMs were reserved, it means the BO wasn't actually mapped */ if (ctx.n_vms == 0) {
ret = -EINVAL; goto unreserve_out;
}
ret = vm_validate_pt_pd_bos(avm, NULL); if (unlikely(ret)) goto unreserve_out;
pr_debug("Unmap VA 0x%llx - 0x%llx from vm %p\n",
mem->va,
mem->va + bo_size * (1 + mem->aql_queue),
avm);
bo = gem_to_amdgpu_bo(obj); if (!(bo->preferred_domains & (AMDGPU_GEM_DOMAIN_VRAM |
AMDGPU_GEM_DOMAIN_GTT))) /* Only VRAM and GTT BOs are supported */ return -EINVAL;
*mem = kzalloc(sizeof(struct kgd_mem), GFP_KERNEL); if (!*mem) return -ENOMEM;
ret = drm_vma_node_allow(&obj->vma_node, drm_priv); if (ret) goto err_free_mem;
if (size)
*size = amdgpu_bo_size(bo);
if (mmap_offset)
*mmap_offset = amdgpu_bo_mmap_offset(bo);
mutex_lock(&avm->process_info->lock); if (avm->process_info->eviction_fence &&
!dma_fence_is_signaled(&avm->process_info->eviction_fence->base))
ret = amdgpu_amdkfd_bo_validate_and_fence(bo, (*mem)->domain,
&avm->process_info->eviction_fence->base);
mutex_unlock(&avm->process_info->lock); if (ret) goto err_remove_mem;
/* Evict a userptr BO by stopping the queues if necessary * *RunsinMMUnotifier,maybeinRECLAIM_FScontext.Thismeansit *cannotdoanymemoryallocations,andcannottakeanylocksthat *areheldelsewherewhileallocatingmemory. * *Itdoesn'tdoanythingtotheBOitself.Therealworkhappensin *restore,wherewegetupdatedpageaddresses.Thisfunctiononly *ensuresthatGPUaccesstotheBOisstopped.
*/ int amdgpu_amdkfd_evict_userptr(struct mmu_interval_notifier *mni, unsignedlong cur_seq, struct kgd_mem *mem)
{ struct amdkfd_process_info *process_info = mem->process_info; int r = 0;
/* Do not process MMU notifications during CRIU restore until *KFD_CRIU_OP_RESUMEIOCTLisreceived
*/ if (READ_ONCE(process_info->block_mmu_notifications)) return0;
mem->invalid++; if (++process_info->evicted_bos == 1) { /* First eviction, stop the queues */
r = kgd2kfd_quiesce_mm(mni->mm,
KFD_QUEUE_EVICTION_TRIGGER_USERPTR);
if (r && r != -ESRCH)
pr_err("Failed to quiesce KFD\n");
if (r != -ESRCH)
queue_delayed_work(system_freezable_wq,
&process_info->restore_userptr_work,
msecs_to_jiffies(AMDGPU_USERPTR_RESTORE_DELAY_MS));
}
mutex_unlock(&process_info->notifier_lock);
/* Move all invalidated BOs to the userptr_inval_list */
list_for_each_entry_safe(mem, tmp_mem,
&process_info->userptr_valid_list,
validate_list) if (mem->invalid)
list_move_tail(&mem->validate_list,
&process_info->userptr_inval_list);
/* Go through userptr_inval_list and update any invalid user_pages */
list_for_each_entry(mem, &process_info->userptr_inval_list,
validate_list) {
invalid = mem->invalid; if (!invalid) /* BO hasn't been invalidated since the last *revalidationattempt.Keepitspagelist.
*/ continue;
/* BO reservations and getting user pages (hmm_range_fault) *musthappenoutsidethenotifierlock
*/
mutex_unlock(&process_info->notifier_lock);
/* Move the BO to system (CPU) domain if necessary to unmap *andfreetheSGtable
*/ if (bo->tbo.resource->mem_type != TTM_PL_SYSTEM) { if (amdgpu_bo_reserve(bo, true)) return -EAGAIN;
amdgpu_bo_placement_from_domain(bo, AMDGPU_GEM_DOMAIN_CPU);
ret = ttm_bo_validate(&bo->tbo, &bo->placement, &ctx);
amdgpu_bo_unreserve(bo); if (ret) {
pr_err("%s: Failed to invalidate userptr BO\n",
__func__); return -EAGAIN;
}
}
/* Get updated user pages */
ret = amdgpu_ttm_tt_get_user_pages(bo, bo->tbo.ttm->pages,
&mem->range); if (ret) {
pr_debug("Failed %d to get user pages\n", ret);
/* Return -EFAULT bad address error as success. It will *faillaterwithaVMfaultiftheGPUtriestoaccess *it.Betterthanhangingindefinitelywithstalled *usermodequeues. * *Returnothererror-EBUSYor-ENOMEMtoretryrestore
*/ if (ret != -EFAULT) return ret;
/* If applications unmap memory before destroying the userptr *fromtheKFD,triggerasegmentationfaultinVMdebugmode.
*/ if (amdgpu_ttm_adev(bo->tbo.bdev)->debug_vm_userptr) { struct kfd_process *p;
pr_err("Pid %d unmapped memory before destroying userptr at GPU addr 0x%llx\n",
pid_nr(process_info->pid), mem->va);
// Send GPU VM fault to user space
p = kfd_lookup_process_by_pid(process_info->pid); if (p) {
kfd_signal_vm_fault_event_with_userptr(p, mem->va);
kfd_unref_process(p);
}
}
ret = 0;
}
mutex_lock(&process_info->notifier_lock);
/* Mark the BO as valid unless it was invalidated *againconcurrently.
*/ if (mem->invalid != invalid) {
ret = -EAGAIN; goto unlock_out;
} /* set mem valid if mem has hmm range associated */ if (mem->range)
mem->invalid = 0;
}
/* Validate the BO if we got user pages */ if (bo->tbo.ttm->pages[0]) {
amdgpu_bo_placement_from_domain(bo, mem->domain);
ret = ttm_bo_validate(&bo->tbo, &bo->placement, &ctx); if (ret) {
pr_err("%s: failed to validate BO\n", __func__); goto unreserve_out;
}
}
/* Update mapping. If the BO was not validated *(becausewecouldn'tgetuserpages),thiswill *clearthepagetableentries,whichwillresultin *VMfaultsiftheGPUtriestoaccesstheinvalid *memory.
*/
list_for_each_entry(attachment, &mem->attachments, list) { if (!attachment->is_mapped) continue;
kfd_mem_dmaunmap_attachment(mem, attachment);
ret = update_gpuvm_pte(mem, attachment, &sync); if (ret) {
pr_err("%s: update PTE failed\n", __func__); /* make sure this gets validated again */
mutex_lock(&process_info->notifier_lock);
mem->invalid++;
mutex_unlock(&process_info->notifier_lock); goto unreserve_out;
}
}
}
/* Update page directories */
ret = process_update_pds(process_info, &sync);
/* Confirm that all user pages are valid while holding the notifier lock * *MovesvalidBOsfromtheuserptr_inval_listbacktouserptr_val_list.
*/ staticint confirm_valid_user_pages_locked(struct amdkfd_process_info *process_info)
{ struct kgd_mem *mem, *tmp_mem; int ret = 0;
mutex_lock(&process_info->notifier_lock);
evicted_bos = process_info->evicted_bos;
mutex_unlock(&process_info->notifier_lock); if (!evicted_bos) return;
/* Reference task and mm in case of concurrent process termination */
usertask = get_pid_task(process_info->pid, PIDTYPE_PID); if (!usertask) return;
mm = get_task_mm(usertask); if (!mm) {
put_task_struct(usertask); return;
}
mutex_lock(&process_info->lock);
if (update_invalid_user_pages(process_info, mm)) goto unlock_out; /* userptr_inval_list can be empty if all evicted userptr BOs *havebeenfreed.Inthatcasethereisnothingtovalidate *andwecanjustrestartthequeues.
*/ if (!list_empty(&process_info->userptr_inval_list)) { if (validate_invalid_user_pages(process_info)) goto unlock_out;
} /* Final check for concurrent evicton and atomic update. If *anotherevictionhappensaftersuccessfulupdate,itwill *beafirstevictionthatcallsquiesce_mm.Theeviction *referencecountinginsideKFDwillhandlethiscase.
*/
mutex_lock(&process_info->notifier_lock); if (process_info->evicted_bos != evicted_bos) goto unlock_notifier_out;
if (kgd2kfd_resume_mm(mm)) {
pr_err("%s: Failed to resume KFD\n", __func__); /* No recovery from this failure. Probably the CP is *hanging.Nopointtryingagain.
*/
}
/* If validation failed, reschedule another attempt */ if (evicted_bos) {
queue_delayed_work(system_freezable_wq,
&process_info->restore_userptr_work,
msecs_to_jiffies(AMDGPU_USERPTR_RESTORE_DELAY_MS));
drm_exec_init(&exec, DRM_EXEC_IGNORE_DUPLICATES, 0);
drm_exec_until_all_locked(&exec) {
list_for_each_entry(peer_vm, &process_info->vm_list_head,
vm_list_node) {
ret = amdgpu_vm_lock_pd(peer_vm, &exec, 2);
drm_exec_retry_on_contention(&exec); if (unlikely(ret)) {
pr_err("Locking VM PD failed, ret: %d\n", ret); goto ttm_reserve_fail;
}
}
/* Reserve all BOs and page tables/directory. Add all BOs from *kfd_bo_listtoctx.list
*/
list_for_each_entry(mem, &process_info->kfd_bo_list,
validate_list) { struct drm_gem_object *gobj;
/* Sync with fences on all the page tables. They implicitly depend on any *movefencesfromamdgpu_vm_handle_movedabove.
*/
ret = process_sync_pds_resv(process_info, &sync_obj); if (ret) {
pr_debug("Memory eviction: Failed to sync to PD BO moving fence. Try again\n"); goto validate_map_fail;
}
/* Wait for validate and PT updates to finish */
amdgpu_sync_wait(&sync_obj, false);
/* The old eviction fence may be unsignaled if restore happens *afteraGPUresetorsuspend/resume.Keeptheoldfenceinthat *case.Otherwisereleasetheoldevictionfenceandcreatenew *one,becausefenceonlygoesfromunsignaledtosignaledonce *andcannotbereused.Usecontextandmmfromtheoldfence. * *Ifanoldevictionfencesignalsafterthischeck,that'sOK. *Anyonesignalinganevictionfencemuststopthequeuesfirst *andscheduleanotherrestoreworker.
*/ if (dma_fence_is_signaled(&process_info->eviction_fence->base)) { struct amdgpu_amdkfd_fence *new_fence =
amdgpu_amdkfd_fence_create(
process_info->eviction_fence->base.context,
process_info->eviction_fence->mm,
NULL);
if (!new_fence) {
pr_err("Failed to create eviction fence\n");
ret = -ENOMEM; goto validate_map_fail;
}
dma_fence_put(&process_info->eviction_fence->base);
process_info->eviction_fence = new_fence;
replace_eviction_fence(ef, dma_fence_get(&new_fence->base));
} else {
WARN_ONCE(*ef != &process_info->eviction_fence->base, "KFD eviction fence doesn't match KGD process_info");
}
/* Attach new eviction fence to all BOs except pinned ones */
list_for_each_entry(mem, &process_info->kfd_bo_list, validate_list) { if (mem->bo->tbo.pin_count) continue;
/* Validate gws bo the first time it is added to process */
mutex_lock(&(*mem)->process_info->lock);
ret = amdgpu_bo_reserve(gws_bo, false); if (unlikely(ret)) {
pr_err("Reserve gws bo failed %d\n", ret); goto bo_reservation_failure;
}
ret = amdgpu_amdkfd_bo_validate(gws_bo, AMDGPU_GEM_DOMAIN_GWS, true); if (ret) {
pr_err("GWS BO validate failed %d\n", ret); goto bo_validation_failure;
} /* GWS resource is shared b/t amdgpu and amdkfd *Addprocessevictionfencetobosotheycan *evicteachother.
*/
ret = dma_resv_reserve_fences(gws_bo->tbo.base.resv, 1); if (ret) goto reserve_shared_fail;
dma_resv_add_fence(gws_bo->tbo.base.resv,
&process_info->eviction_fence->base,
DMA_RESV_USAGE_BOOKKEEP);
amdgpu_bo_unreserve(gws_bo);
mutex_unlock(&(*mem)->process_info->lock);
int kfd_debugfs_kfd_mem_limits(struct seq_file *m, void *data)
{
spin_lock(&kfd_mem_limit.mem_limit_lock);
seq_printf(m, "System mem used %lldM out of %lluM\n",
(kfd_mem_limit.system_mem_used >> 20),
(kfd_mem_limit.max_system_mem_limit >> 20));
seq_printf(m, "TTM mem used %lldM out of %lluM\n",
(kfd_mem_limit.ttm_mem_used >> 20),
(kfd_mem_limit.max_ttm_mem_limit >> 20));
spin_unlock(&kfd_mem_limit.mem_limit_lock);
return0;
}
#endif
Messung V0.5 in Prozent
¤ Dauer der Verarbeitung: 0.99 Sekunden
(vorverarbeitet am 2026-10-02)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.