diff options
| author | Mark Brown <broonie@kernel.org> | 2026-09-14 15:21:31 +0100 |
|---|---|---|
| committer | Mark Brown <broonie@kernel.org> | 2026-09-14 15:21:31 +0100 |
| commit | 62c0e6c9d9a275040a145a92dd42316a110e484e (patch) | |
| tree | 4ffc56e93ffd8b44bfb6245428a94cd827edab27 /drivers/gpu/drm | |
| parent | 4f6280141c1c87d1e82713ceb2751ecfab47eb3a (diff) | |
| parent | 7dbd24e9d5311b6afdbdd875dd247ab5ff54edca (diff) | |
| download | linux-next-62c0e6c9d9a275040a145a92dd42316a110e484e.tar.gz linux-next-62c0e6c9d9a275040a145a92dd42316a110e484e.zip | |
Merge branch 'drm-xe-next' of https://gitlab.freedesktop.org/drm/xe/kernel.git
Diffstat (limited to 'drivers/gpu/drm')
57 files changed, 2234 insertions, 478 deletions
diff --git a/drivers/gpu/drm/drm_gpusvm.c b/drivers/gpu/drm/drm_gpusvm.c index a93eee7ddb9e..b6c9d3a07dc8 100644 --- a/drivers/gpu/drm/drm_gpusvm.c +++ b/drivers/gpu/drm/drm_gpusvm.c @@ -80,6 +80,13 @@ * }; * }; * + * static struct drm_gpusvm_pages * + * driver_pages(struct driver_range *drange) + * { + * return drange->num_pages == 1 ? &drange->inline_pages : + * drange->pages; + * } + * * In the N:1 case the driver allocates the pages array with a zeroing * allocator (e.g. kcalloc(num_pages, ...)), initialises each entry with * drm_gpusvm_init_pages(), and frees each entry with @@ -89,6 +96,28 @@ * Each drm_gpusvm_pages must be zero-initialised and initialised with * drm_gpusvm_init_pages(), called once per entry. * + * The 1:1 examples below pass @num_pages == 1 and &drange->pages. In the + * N:1 case the driver instead passes the whole array and its count, so a + * single call faults the CPU range once and DMA maps it for every owning + * drm_device, e.g.: + * + * .. code-block:: c + * + * // GPU fault handler: one fault, one DMA mapping per device + * err = drm_gpusvm_get_pages(gpusvm, driver_pages(drange), + * drange->num_pages, gpusvm->mm, + * &range->notifier->notifier, + * drm_gpusvm_range_start(range), + * drm_gpusvm_range_end(range), &ctx); + * + * // Notifier callback: mark every instance unmapped in one call + * drm_gpusvm_range_set_unmapped(range, driver_pages(drange), + * drange->num_pages, mmu_range); + * + * The unmap and free paths stay per-instance: iterate @num_pages over + * driver_pages(drange) and call drm_gpusvm_unmap_pages() / + * drm_gpusvm_free_pages() for each entry. + * * - Operations: * Define the interface for driver-specific GPU SVM operations such as * range allocation, notifier allocation, and invalidations. @@ -232,7 +261,7 @@ * goto retry; * } * - * err = drm_gpusvm_get_pages(gpusvm, &drange->pages, + * err = drm_gpusvm_get_pages(gpusvm, &drange->pages, 1, * gpusvm->mm, &range->notifier->notifier, * drm_gpusvm_range_start(range), * drm_gpusvm_range_end(range), &ctx); @@ -1212,6 +1241,8 @@ static void __drm_gpusvm_unmap_pages(struct drm_gpusvm *gpusvm, struct drm_gpusvm_pages_flags flags = { .__flags = svm_pages->flags.__flags, }; + const struct drm_pagemap_addr *addrs = + drm_gpusvm_pages_first_dma(svm_pages, NULL); bool use_iova = dma_use_iova(&svm_pages->state); /* @@ -1224,12 +1255,20 @@ static void __drm_gpusvm_unmap_pages(struct drm_gpusvm *gpusvm, if (svm_pages->state_offset) dma_iova_unlink(dev, &svm_pages->state, 0, svm_pages->state_offset, - svm_pages->dma_addr[0].dir, 0); + addrs[0].dir, 0); dma_iova_free(dev, &svm_pages->state); } - for (i = 0, j = 0; i < npages; j++) { - struct drm_pagemap_addr *addr = &svm_pages->dma_addr[j]; + /* + * With IOVA and no device page the unlink above tore every + * entry down, and that is also when the range may be folded + * to one entry, which must not be walked per entry. dpagemap + * is set before the first device_map(), so it is also right + * on the error path, where the flags are not published yet. + */ + for (i = 0, j = 0; + (!use_iova || dpagemap) && i < npages; j++) { + const struct drm_pagemap_addr *addr = &addrs[j]; if (addr->proto == DRM_INTERCONNECT_SYSTEM) { /* @@ -1270,6 +1309,18 @@ static void __drm_gpusvm_free_pages(struct drm_gpusvm *gpusvm, { lockdep_assert_held(&gpusvm->notifier_lock); + if (svm_pages->flags.inline_dma_mapping) { + struct drm_gpusvm_pages_flags flags = { + .__flags = svm_pages->flags.__flags, + }; + + svm_pages->inline_addr = (struct drm_pagemap_addr){}; + flags.inline_dma_mapping = false; + /* WRITE_ONCE pairs with READ_ONCE for opportunistic checks */ + WRITE_ONCE(svm_pages->flags.__flags, flags.__flags); + return; + } + if (svm_pages->dma_addr) { kvfree(svm_pages->dma_addr); svm_pages->dma_addr = NULL; @@ -1417,144 +1468,105 @@ EXPORT_SYMBOL_GPL(drm_gpusvm_pages_valid); /** * drm_gpusvm_pages_valid_unlocked() - GPU SVM pages valid unlocked * @gpusvm: Pointer to the GPU SVM structure - * @svm_pages: Pointer to the GPU SVM pages structure + * @svm_pages: Array of GPU SVM pages structures + * @num_pages: Number of drm_gpusvm_pages instances in @svm_pages * - * This function determines if a GPU SVM pages are valid. Expected be called - * without holding gpusvm->notifier_lock. + * This function determines if every GPU SVM pages instance is valid, resetting + * every instance which is not so that get_pages() maps it afresh. It therefore + * has to walk them all. Expected be called without holding + * gpusvm->notifier_lock. * - * Return: True if GPU SVM pages are valid, False otherwise + * Return: True if all GPU SVM pages are valid, False otherwise */ static bool drm_gpusvm_pages_valid_unlocked(struct drm_gpusvm *gpusvm, - struct drm_gpusvm_pages *svm_pages) + struct drm_gpusvm_pages *svm_pages, + unsigned int num_pages) { - bool pages_valid; - - if (!svm_pages->dma_addr) - return false; + bool pages_valid = true; + unsigned int p; drm_gpusvm_notifier_lock(gpusvm); - pages_valid = drm_gpusvm_pages_valid(gpusvm, svm_pages); - if (!pages_valid) - __drm_gpusvm_free_pages(gpusvm, svm_pages); + for (p = 0; p < num_pages; ++p) { + if (drm_gpusvm_pages_valid(gpusvm, &svm_pages[p])) + continue; + __drm_gpusvm_free_pages(gpusvm, &svm_pages[p]); + pages_valid = false; + } drm_gpusvm_notifier_unlock(gpusvm); return pages_valid; } /** - * drm_gpusvm_get_pages() - Get pages and populate GPU SVM pages struct + * drm_gpusvm_pages_inlinable() - Whether the dma address can be inlined + * @svm_pages: The SVM pages instance that was just mapped + * @nentries: Number of entries the mapping loop produced + * @npages: Number of pages in the CPU range + * + * A THP maps as one huge page, and an IOVA reservation links every page of + * the range at the next offset, so the device addresses run contiguously from + * entry 0. Either way one entry describes the whole range, so the dma_addr + * array can be freed and the address kept inline. + * + * state_offset advances only on the IOVA branch, so reaching the full range + * length proves no device page was mapped in between. Only single page + * entries fold, so the order kept is 0 and describes the range truthfully. + * Larger chunks, several huge pages among them, stay an array that is + * already short and that a consumer places with one PTE each. + * + * Return: True if the mapping fits in a single drm_pagemap_addr. + */ +static bool drm_gpusvm_pages_inlinable(struct drm_gpusvm_pages *svm_pages, + unsigned long nentries, + unsigned long npages) +{ + if (nentries == 1) + return true; + + return nentries == npages && dma_use_iova(&svm_pages->state) && + svm_pages->state_offset == npages * PAGE_SIZE; +} + +/** + * drm_gpusvm_dma_map_pages() - DMA map one drm_gpusvm_pages instance * @gpusvm: Pointer to the GPU SVM structure - * @svm_pages: The SVM pages to populate. This will contain the dma-addresses - * @mm: The mm corresponding to the CPU range - * @notifier: The corresponding notifier for the given CPU range - * @pages_start: Start CPU address for the pages - * @pages_end: End CPU address for the pages (exclusive) + * @svm_pages: The SVM pages instance to populate with dma-addresses + * @pfns: The already-faulted pfn array (size @npages) + * @npages: Number of pages in the CPU range * @ctx: GPU SVM context + * @dma_dir: DMA data direction for the mappings * - * This function gets and maps pages for CPU range and ensures they are - * mapped for DMA access. + * Map the faulted @pfns into @svm_pages for DMA access through its owning + * drm_device. Must be called under the notifier lock and only for an instance + * without a live mapping. On failure this unwinds the partial mapping of this + * instance before returning. * * Return: 0 on success, negative error code on failure. */ -int drm_gpusvm_get_pages(struct drm_gpusvm *gpusvm, - struct drm_gpusvm_pages *svm_pages, - struct mm_struct *mm, - struct mmu_interval_notifier *notifier, - unsigned long pages_start, unsigned long pages_end, - const struct drm_gpusvm_ctx *ctx) +static int drm_gpusvm_dma_map_pages(struct drm_gpusvm *gpusvm, + struct drm_gpusvm_pages *svm_pages, + unsigned long *pfns, + unsigned long npages, + const struct drm_gpusvm_ctx *ctx, + enum dma_data_direction dma_dir) { - struct hmm_range hmm_range = { - .default_flags = HMM_PFN_REQ_FAULT | (ctx->read_only ? 0 : - HMM_PFN_REQ_WRITE), - .notifier = notifier, - .start = pages_start, - .end = pages_end, - .dev_private_owner = ctx->device_private_page_owner, - }; - void *zdd; - unsigned long timeout = - jiffies + msecs_to_jiffies(HMM_RANGE_DEFAULT_TIMEOUT); - unsigned long remaining; + void *zdd = NULL; unsigned long i, j; - unsigned long npages = npages_in_range(pages_start, pages_end); - unsigned long num_dma_mapped; + unsigned long num_dma_mapped = 0; unsigned int order = 0; - unsigned long *pfns; int err = 0; - struct dev_pagemap *pagemap; + struct dev_pagemap *pagemap = NULL; struct drm_pagemap *dpagemap; struct drm_gpusvm_pages_flags flags; - enum dma_data_direction dma_dir = ctx->read_only ? DMA_TO_DEVICE : - DMA_BIDIRECTIONAL; struct dma_iova_state *state = &svm_pages->state; - if (!svm_pages->drm) - return -EINVAL; - -retry: - remaining = timeout - jiffies; - - if (time_after_eq(jiffies, timeout)) - return -EBUSY; - - hmm_range.notifier_seq = mmu_interval_read_begin(notifier); - if (drm_gpusvm_pages_valid_unlocked(gpusvm, svm_pages)) - goto set_seqno; - - pfns = kvmalloc_array(npages, sizeof(*pfns), GFP_KERNEL); - if (!pfns) - return -ENOMEM; - - if (!mmget_not_zero(mm)) { - err = -EFAULT; - goto err_free; - } - - hmm_range.hmm_pfns = pfns; - err = hmm_range_fault_unlocked_timeout(&hmm_range, remaining); - mmput(mm); - if (err) - goto err_free; + lockdep_assert_held(&gpusvm->notifier_lock); *state = (struct dma_iova_state){}; svm_pages->state_offset = 0; -map_pages: - /* - * Perform all dma mappings under the notifier lock to not - * access freed pages. A notifier will either block on - * the notifier lock or unmap dma. - */ - drm_gpusvm_notifier_lock(gpusvm); - flags.__flags = svm_pages->flags.__flags; - if (flags.unmapped) { - drm_gpusvm_notifier_unlock(gpusvm); - err = -EFAULT; - goto err_free; - } - - if (mmu_interval_read_retry(notifier, hmm_range.notifier_seq)) { - drm_gpusvm_notifier_unlock(gpusvm); - kvfree(pfns); - goto retry; - } - - if (!svm_pages->dma_addr) { - /* Unlock and restart mapping to allocate memory. */ - drm_gpusvm_notifier_unlock(gpusvm); - svm_pages->dma_addr = - kvzalloc_objs(*svm_pages->dma_addr, npages); - if (!svm_pages->dma_addr) { - err = -ENOMEM; - goto err_free; - } - goto map_pages; - } - zdd = NULL; - pagemap = NULL; - num_dma_mapped = 0; for (i = 0, j = 0; i < npages; ++j) { struct page *page = hmm_pfn_to_page(pfns[i]); @@ -1667,20 +1679,186 @@ map_pages: if (pagemap) flags.has_devmem_pages = true; + if (drm_gpusvm_pages_inlinable(svm_pages, j, npages)) { + struct drm_pagemap_addr addr = svm_pages->dma_addr[0]; + + kvfree(svm_pages->dma_addr); + svm_pages->inline_addr = addr; + flags.inline_dma_mapping = true; + } + /* WRITE_ONCE pairs with READ_ONCE for opportunistic checks */ WRITE_ONCE(svm_pages->flags.__flags, flags.__flags); - drm_gpusvm_notifier_unlock(gpusvm); - kvfree(pfns); -set_seqno: - svm_pages->notifier_seq = hmm_range.notifier_seq; - return 0; err_unmap: svm_pages->flags.has_dma_mapping = true; __drm_gpusvm_unmap_pages(gpusvm, svm_pages, num_dma_mapped); + return err; +} + +/** + * drm_gpusvm_get_pages() - Get pages and populate GPU SVM pages struct + * @gpusvm: Pointer to the GPU SVM structure + * @svm_pages: Array of SVM pages instances to populate with dma addresses + * @num_pages: Number of drm_gpusvm_pages instances in @svm_pages, must not be 0 + * @mm: The mm corresponding to the CPU range + * @notifier: The corresponding notifier for the given CPU range + * @pages_start: Start CPU address for the pages + * @pages_end: End CPU address for the pages (exclusive) + * @ctx: GPU SVM context + * + * This function gets and maps pages for a CPU range and ensures they are + * mapped for DMA access. The HMM fault for the CPU range is performed once, + * the DMA mapping by drm_gpusvm_dma_map_pages() is then done per instance, + * one per owning drm_device. The retry against notifier races is kept here + * in common code so drivers never open code it. + * + * On error the instances mapped before the failing one stay mapped, so the + * caller must unmap and free every instance regardless of the return value. + * + * With &drm_gpusvm_ctx.no_dma_map no mapping state is recorded, so + * drm_gpusvm_pages_valid() never returns true and success is only a snapshot: + * the caller must recheck mmu_interval_read_retry() against the recorded + * &drm_gpusvm_pages.notifier_seq under the notifier lock, and hold it until + * its work is visible to invalidation. + * + * Return: 0 on success, negative error code on failure. + */ +int drm_gpusvm_get_pages(struct drm_gpusvm *gpusvm, + struct drm_gpusvm_pages *svm_pages, + unsigned int num_pages, + struct mm_struct *mm, + struct mmu_interval_notifier *notifier, + unsigned long pages_start, unsigned long pages_end, + const struct drm_gpusvm_ctx *ctx) +{ + struct hmm_range hmm_range = { + .default_flags = HMM_PFN_REQ_FAULT | (ctx->read_only ? 0 : + HMM_PFN_REQ_WRITE), + .notifier = notifier, + .start = pages_start, + .end = pages_end, + .dev_private_owner = ctx->device_private_page_owner, + }; + unsigned long timeout = + jiffies + msecs_to_jiffies(HMM_RANGE_DEFAULT_TIMEOUT); + unsigned long remaining; + unsigned long npages = npages_in_range(pages_start, pages_end); + unsigned long *pfns; + int err = 0; + enum dma_data_direction dma_dir = ctx->read_only ? DMA_TO_DEVICE : + DMA_BIDIRECTIONAL; + const bool map_dma = !ctx->no_dma_map; + unsigned int p; + + if (!num_pages) + return -EINVAL; + + if (ctx->no_dma_map && ctx->devmem_only) + return -EINVAL; + + if (map_dma) { + for (p = 0; p < num_pages; ++p) { + if (!svm_pages[p].drm) + return -EINVAL; + } + } + +retry: + remaining = timeout - jiffies; + + if (time_after_eq(jiffies, timeout)) + return -EBUSY; + + hmm_range.notifier_seq = mmu_interval_read_begin(notifier); + + if (map_dma && + drm_gpusvm_pages_valid_unlocked(gpusvm, svm_pages, num_pages)) + goto set_seqno; + + pfns = kvmalloc_array(npages, sizeof(*pfns), GFP_KERNEL); + if (!pfns) + return -ENOMEM; + + if (!mmget_not_zero(mm)) { + err = -EFAULT; + goto err_free; + } + + hmm_range.hmm_pfns = pfns; + err = hmm_range_fault_unlocked_timeout(&hmm_range, remaining); + mmput(mm); + if (err) + goto err_free; + + if (map_dma) { + for (p = 0; p < num_pages; ++p) { + if (drm_gpusvm_pages_first_dma(&svm_pages[p], NULL)) + continue; + svm_pages[p].dma_addr = + kvzalloc_objs(*svm_pages[p].dma_addr, npages); + if (!svm_pages[p].dma_addr) { + err = -ENOMEM; + goto err_free; + } + } + } + + /* + * Perform all dma mappings under the notifier lock to not + * access freed pages. A notifier will either block on + * the notifier lock or unmap dma. + */ + drm_gpusvm_notifier_lock(gpusvm); + + /* + * drm_gpusvm_range_set_unmapped() flags the whole array in one go under + * the write lock, so any instance answers for all of them here. + */ + if (svm_pages[0].flags.unmapped) { + drm_gpusvm_notifier_unlock(gpusvm); + err = -EFAULT; + goto err_free; + } + + if (mmu_interval_read_retry(notifier, hmm_range.notifier_seq)) { + drm_gpusvm_notifier_unlock(gpusvm); + kvfree(pfns); + goto retry; + } + + if (!map_dma) + goto done_mapping; + + for (p = 0; p < num_pages; ++p) { + if (drm_gpusvm_pages_valid(gpusvm, &svm_pages[p])) + continue; + + err = drm_gpusvm_dma_map_pages(gpusvm, &svm_pages[p], pfns, + npages, ctx, dma_dir); + if (err) { + /* + * The failing instance was unwound by the helper. Keep + * the ones mapped earlier: the -EAGAIN retry reuses + * them, and the driver unmaps every instance with the + * range on the other error paths. + */ + drm_gpusvm_notifier_unlock(gpusvm); + goto err_free; + } + } + +done_mapping: drm_gpusvm_notifier_unlock(gpusvm); + kvfree(pfns); +set_seqno: + for (p = 0; p < num_pages; ++p) + svm_pages[p].notifier_seq = hmm_range.notifier_seq; + + return 0; + err_free: kvfree(pfns); if (err == -EAGAIN) diff --git a/drivers/gpu/drm/xe/abi/xe_log_abi.h b/drivers/gpu/drm/xe/abi/xe_log_abi.h index d6105520173e..526aadf85fe9 100644 --- a/drivers/gpu/drm/xe/abi/xe_log_abi.h +++ b/drivers/gpu/drm/xe/abi/xe_log_abi.h @@ -144,6 +144,7 @@ enum xe_log_location_bits { define(DRIVER, 4, RTP, SW, "Register Table Processing") \ define(DRIVER, 5, WA, SW, "Workarounds") \ define(DRIVER, 6, PAGEFAULT, MEM_FAULT, "Page Fault") \ + define(DRIVER, 7, GUCSUBMIT, GT_TDR, "GuC Submission") \ /* */ \ define(DRIVER_HARDWARE, 1, REGS, IO_BUS, "Registers") \ define(DRIVER_HARDWARE, 2, GGTT, IO_BUS, "Global GTT") \ diff --git a/drivers/gpu/drm/xe/display/xe_dsb_buffer.c b/drivers/gpu/drm/xe/display/xe_dsb_buffer.c index a7158c73a14c..82974e933e8d 100644 --- a/drivers/gpu/drm/xe/display/xe_dsb_buffer.c +++ b/drivers/gpu/drm/xe/display/xe_dsb_buffer.c @@ -88,7 +88,7 @@ static void xe_dsb_buffer_flush_map(struct intel_dsb_buffer *dsb_buf) * both for weak ordering archs and discrete cards. */ xe_device_wmb(xe); - xe_device_l2_flush(xe); + xe_device_l2_flush(xe, false); } const struct intel_display_dsb_interface xe_display_dsb_interface = { diff --git a/drivers/gpu/drm/xe/display/xe_fb_pin.c b/drivers/gpu/drm/xe/display/xe_fb_pin.c index 73469ea5f333..b46a2c32ac07 100644 --- a/drivers/gpu/drm/xe/display/xe_fb_pin.c +++ b/drivers/gpu/drm/xe/display/xe_fb_pin.c @@ -203,7 +203,7 @@ static int __xe_pin_fb_vma_dpt(struct drm_gem_object *obj, vma->node = dpt->ggtt_node[tile0->id]; /* Ensure DPT writes are flushed */ - xe_device_l2_flush(xe); + xe_device_l2_flush(xe, false); return 0; } diff --git a/drivers/gpu/drm/xe/display/xe_panic.c b/drivers/gpu/drm/xe/display/xe_panic.c index 12c6fb99015d..1a6cee25e9d7 100644 --- a/drivers/gpu/drm/xe/display/xe_panic.c +++ b/drivers/gpu/drm/xe/display/xe_panic.c @@ -52,7 +52,8 @@ static void xe_panic_page_set_pixel(struct drm_scanout_buffer *sb, unsigned int if (new_page != panic->page) { if (xe_bo_is_vram(bo)) { /* Display is always mapped on root tile */ - struct xe_vram_region *vram = xe_bo_device(bo)->mem.vram; + struct xe_vram_region *vram = + xe_device_get_root_tile(xe_bo_device(bo))->mem.vram; if (panic->page < 0 || new_page < panic->page) { xe_res_first(bo->ttm.resource, new_page * PAGE_SIZE, diff --git a/drivers/gpu/drm/xe/tests/xe_pci.c b/drivers/gpu/drm/xe/tests/xe_pci.c index bb0393475524..5aefc03c00c7 100644 --- a/drivers/gpu/drm/xe/tests/xe_pci.c +++ b/drivers/gpu/drm/xe/tests/xe_pci.c @@ -311,10 +311,11 @@ const void *xe_pci_id_gen_param(struct kunit *test, const void *prev, char *desc EXPORT_SYMBOL_IF_KUNIT(xe_pci_id_gen_param); static int fake_probe_info(struct xe_device *xe, - const struct xe_device_desc *desc, struct xe_pci_fake_data *data, struct xe_probed_info *probed_info) { + const struct xe_device_desc *desc = xe->desc; + probed_info->tile_count = 1 + desc->max_remote_tiles; if (!data || desc->pre_gmdid_graphics_ip) { @@ -351,7 +352,7 @@ int xe_pci_fake_device_init(struct xe_device *xe) if (!data) { desc = (const void *)ent->driver_data; - subplatform_desc = NULL; + subplatform_desc = desc->subplatforms; goto done; } @@ -364,25 +365,37 @@ int xe_pci_fake_device_init(struct xe_device *xe) if (!ent->device) return -ENODEV; + if (data->subplatform == XE_SUBPLATFORM_NONE) { + subplatform_desc = NULL; + goto done; + } + + if (data->subplatform == XE_SUBPLATFORM_UNINITIALIZED) { + subplatform_desc = desc->subplatforms; + goto done; + } + for (subplatform_desc = desc->subplatforms; subplatform_desc && subplatform_desc->subplatform; subplatform_desc++) if (subplatform_desc->subplatform == data->subplatform) break; - if (data->subplatform != XE_SUBPLATFORM_NONE && !subplatform_desc) + if (!subplatform_desc || !subplatform_desc->subplatform) return -ENODEV; done: + xe->desc = desc; + xe->subplatform_desc = subplatform_desc; xe->sriov.__mode = data && data->sriov_mode ? data->sriov_mode : XE_SRIOV_MODE_NONE; - err = fake_probe_info(xe, desc, data, &probed_info); + err = fake_probe_info(xe, data, &probed_info); if (err) return err; - xe_info_init_early(xe, desc, subplatform_desc, &probed_info); - xe_info_init(xe, desc, &probed_info); + xe_info_init_early(xe, &probed_info); + xe_info_init(xe, &probed_info); return 0; } diff --git a/drivers/gpu/drm/xe/xe_bo.c b/drivers/gpu/drm/xe/xe_bo.c index dde309821237..9f3f0cb95afa 100644 --- a/drivers/gpu/drm/xe/xe_bo.c +++ b/drivers/gpu/drm/xe/xe_bo.c @@ -28,6 +28,7 @@ #include "xe_ggtt.h" #include "xe_map.h" #include "xe_migrate.h" +#include "xe_mmio_gem.h" #include "xe_pat.h" #include "xe_pm.h" #include "xe_preempt_fence.h" @@ -104,13 +105,16 @@ static bool resource_is_vram(struct ttm_resource *res) bool xe_bo_is_vram(struct xe_bo *bo) { - return resource_is_vram(bo->ttm.resource) || - resource_is_stolen_vram(xe_bo_device(bo), bo->ttm.resource); + struct ttm_resource *res = bo->ttm.resource; + + return res && (resource_is_vram(res) || resource_is_stolen_vram(xe_bo_device(bo), res)); } bool xe_bo_is_stolen(struct xe_bo *bo) { - return bo->ttm.resource->mem_type == XE_PL_STOLEN; + struct ttm_resource *res = bo->ttm.resource; + + return res && res->mem_type == XE_PL_STOLEN; } /** @@ -158,7 +162,13 @@ bool xe_bo_is_vm_bound(struct xe_bo *bo) return !list_empty(&bo->ttm.base.gpuva.list); } -static bool xe_bo_is_user(struct xe_bo *bo) +/** + * xe_bo_is_user - Check if BO is user-created + * @bo: The BO + * + * Returns: true if @bo was created by userspace + */ +bool xe_bo_is_user(struct xe_bo *bo) { return bo->flags & XE_BO_FLAG_USER; } @@ -921,7 +931,7 @@ void xe_bo_set_purgeable_state(struct xe_bo *bo, * * Return: 0 on success, negative error code on failure */ -static int xe_ttm_bo_purge(struct ttm_buffer_object *ttm_bo, struct ttm_operation_ctx *ctx) +int xe_ttm_bo_purge(struct ttm_buffer_object *ttm_bo, struct ttm_operation_ctx *ctx) { struct xe_bo *bo = ttm_to_xe_bo(ttm_bo); struct ttm_placement place = {}; @@ -929,9 +939,6 @@ static int xe_ttm_bo_purge(struct ttm_buffer_object *ttm_bo, struct ttm_operatio xe_bo_assert_held(bo); - if (!ttm_bo->ttm) - return 0; - if (!xe_bo_madv_is_dontneed(bo)) return 0; @@ -3261,6 +3268,9 @@ void xe_bo_unpin(struct xe_bo *bo) struct ttm_place *place = &bo->placements[0]; struct xe_device *xe = xe_bo_device(bo); + if (xe_bo_is_purged(bo)) + return; + xe_assert(xe, !bo->ttm.base.import_attach); xe_assert(xe, xe_bo_is_pinned(bo)); @@ -3655,6 +3665,39 @@ out_vm: return err; } +static int xe_gem_pci_barrier_mmap_offset(struct xe_device *xe, struct drm_file *file, + struct drm_xe_gem_mmap_offset *args) +{ + struct xe_file *xef = file->driver_priv; + struct xe_mmio_gem **barrier = &xef->mmio_gem.pci_barrier; + + if (XE_IOCTL_DBG(xe, !IS_DGFX(xe))) + return -EINVAL; + + if (XE_IOCTL_DBG(xe, args->handle)) + return -EINVAL; + + scoped_guard(mutex, &xef->mmio_gem.lock) { + if (!*barrier) { + phys_addr_t phys_addr; + +#define LAST_DB_PAGE_OFFSET 0x7ff000 + phys_addr = pci_resource_start(to_pci_dev(xe->drm.dev), 0) + + LAST_DB_PAGE_OFFSET; + *barrier = xe_mmio_gem_create(xe, file, phys_addr, SZ_4K); + if (IS_ERR(*barrier)) { + int err = PTR_ERR(*barrier); + + *barrier = NULL; + return err; + } + } + + args->offset = xe_mmio_gem_mmap_offset(*barrier); + } + return 0; +} + int xe_gem_mmap_offset_ioctl(struct drm_device *dev, void *data, struct drm_file *file) { @@ -3670,21 +3713,8 @@ int xe_gem_mmap_offset_ioctl(struct drm_device *dev, void *data, ~DRM_XE_MMAP_OFFSET_FLAG_PCI_BARRIER)) return -EINVAL; - if (args->flags & DRM_XE_MMAP_OFFSET_FLAG_PCI_BARRIER) { - if (XE_IOCTL_DBG(xe, !IS_DGFX(xe))) - return -EINVAL; - - if (XE_IOCTL_DBG(xe, args->handle)) - return -EINVAL; - - if (XE_IOCTL_DBG(xe, PAGE_SIZE > SZ_4K)) - return -EINVAL; - - BUILD_BUG_ON(((XE_PCI_BARRIER_MMAP_OFFSET >> XE_PTE_SHIFT) + - SZ_4K) >= DRM_FILE_PAGE_OFFSET_START); - args->offset = XE_PCI_BARRIER_MMAP_OFFSET; - return 0; - } + if (args->flags & DRM_XE_MMAP_OFFSET_FLAG_PCI_BARRIER) + return xe_gem_pci_barrier_mmap_offset(xe, file, args); gem_obj = drm_gem_object_lookup(file, args->handle); if (XE_IOCTL_DBG(xe, !gem_obj)) diff --git a/drivers/gpu/drm/xe/xe_bo.h b/drivers/gpu/drm/xe/xe_bo.h index e8081af5bfc1..290ca624e2a7 100644 --- a/drivers/gpu/drm/xe/xe_bo.h +++ b/drivers/gpu/drm/xe/xe_bo.h @@ -87,7 +87,6 @@ #define XE_BO_PROPS_INVALID (-1) -#define XE_PCI_BARRIER_MMAP_OFFSET (0x50 << XE_PTE_SHIFT) /** * enum xe_madv_purgeable_state - Buffer object purgeable state enumeration @@ -600,6 +599,8 @@ struct xe_bo_shrink_flags { long xe_bo_shrink(struct ttm_operation_ctx *ctx, struct ttm_buffer_object *bo, const struct xe_bo_shrink_flags flags, unsigned long *scanned); +int xe_ttm_bo_purge(struct ttm_buffer_object *ttm_bo, struct ttm_operation_ctx *ctx); +bool xe_bo_is_user(struct xe_bo *bo); /** * xe_bo_is_mem_type - Whether the bo currently resides in the given diff --git a/drivers/gpu/drm/xe/xe_bo_types.h b/drivers/gpu/drm/xe/xe_bo_types.h index e45f24301050..0eb93052c1d5 100644 --- a/drivers/gpu/drm/xe/xe_bo_types.h +++ b/drivers/gpu/drm/xe/xe_bo_types.h @@ -20,6 +20,7 @@ struct xe_device; struct xe_mem_pool_node; struct xe_vm; +struct xe_exec_queue; #define XE_BO_MAX_PLACEMENTS 3 @@ -42,6 +43,13 @@ struct xe_bo { u32 flags; /** @vm: VM this BO is attached to, for extobj this will be NULL */ struct xe_vm *vm; + /** + * @q: Queue this BO is attached to, mostly for LRC BO, NULL otherwise. + * Protected by the BO dma_resv: readers must hold it across both the + * read and xe_exec_queue_get_unless_zero(). The BO holds no reference + * on the queue. + */ + struct xe_exec_queue *q; /** @tile: Tile this BO is attached to (kernel BO only) */ struct xe_tile *tile; /** @placements: valid placements for this BO */ diff --git a/drivers/gpu/drm/xe/xe_configfs.c b/drivers/gpu/drm/xe/xe_configfs.c index 052cce962161..f5c828cf7e8f 100644 --- a/drivers/gpu/drm/xe/xe_configfs.c +++ b/drivers/gpu/drm/xe/xe_configfs.c @@ -61,7 +61,8 @@ * ├── survivability_mode * ├── gt_types_allowed * ├── engines_allowed - * └── enable_psmi + * ├── enable_psmi + * └── disable_vram_page_offline * * After configuring the attributes as per next section, the device can be * probed with:: @@ -159,6 +160,18 @@ * * This attribute can only be set before binding to the device. * + * Disable VRAM page offline: + * ---------------------------- + * + * 0, n, N, false - Do not disable (Offlining is active - default) + * 1, y, Y, true - Disable vram page offline (Logging only) + * + * Example to disable VRAM offline:: + * + * # echo 1 > /sys/kernel/config/xe/0000:03:00.0/disable_vram_page_offline + * + * This attribute can only be set on CRI before binding to the device. + * * Context restore BB * ------------------ * @@ -275,6 +288,7 @@ struct xe_config_group_device { bool survivability_mode; bool enable_psmi; bool enable_multi_queue; + bool disable_vram_page_offline; struct { unsigned int max_vfs; bool admin_only_pf; @@ -295,6 +309,7 @@ static const struct xe_config_device device_defaults = { .survivability_mode = false, .enable_psmi = false, .enable_multi_queue = true, + .disable_vram_page_offline = false, .sriov = { .max_vfs = XE_DEFAULT_MAX_VFS, .admin_only_pf = XE_DEFAULT_ADMIN_ONLY_PF, @@ -616,6 +631,33 @@ static ssize_t enable_multi_queue_store(struct config_item *item, const char *pa return len; } +static ssize_t disable_vram_page_offline_show(struct config_item *item, char *page) +{ + struct xe_config_device *dev = to_xe_config_device(item); + + return sprintf(page, "%s\n", str_yes_no(dev->disable_vram_page_offline)); +} + +static ssize_t disable_vram_page_offline_store(struct config_item *item, + const char *page, size_t len) +{ + struct xe_config_group_device *dev = to_xe_config_group_device(item); + bool val; + int ret; + + ret = kstrtobool(page, &val); + if (ret) + return ret; + + guard(mutex)(&dev->lock); + if (is_bound(dev)) + return -EBUSY; + + dev->config.disable_vram_page_offline = val; + + return len; +} + static bool wa_bb_read_advance(bool dereference, char **p, const char *append, size_t len, size_t *max_size) @@ -855,6 +897,7 @@ CONFIGFS_ATTR(, ctx_restore_mid_bb); CONFIGFS_ATTR(, ctx_restore_post_bb); CONFIGFS_ATTR(, enable_multi_queue); CONFIGFS_ATTR(, enable_psmi); +CONFIGFS_ATTR(, disable_vram_page_offline); CONFIGFS_ATTR(, engines_allowed); CONFIGFS_ATTR(, gt_types_allowed); CONFIGFS_ATTR(, survivability_mode); @@ -864,6 +907,7 @@ static struct configfs_attribute *xe_config_device_attrs[] = { &attr_ctx_restore_post_bb, &attr_enable_multi_queue, &attr_enable_psmi, + &attr_disable_vram_page_offline, &attr_engines_allowed, &attr_gt_types_allowed, &attr_survivability_mode, @@ -895,6 +939,11 @@ static bool xe_config_device_is_visible(struct config_item *item, return false; } + if (attr == &attr_disable_vram_page_offline) { + if (!dev->desc->is_dgfx || dev->desc->platform != XE_CRESCENTISLAND) + return false; + } + return true; } @@ -1142,6 +1191,7 @@ static void dump_custom_dev_config(struct pci_dev *pdev, PRI_CUSTOM_ATTR("%llx", engines_allowed); PRI_CUSTOM_ATTR("%d", enable_multi_queue); PRI_CUSTOM_ATTR("%d", enable_psmi); + PRI_CUSTOM_ATTR("%d", disable_vram_page_offline); PRI_CUSTOM_ATTR("%d", survivability_mode); PRI_CUSTOM_ATTR("%u", sriov.admin_only_pf); @@ -1291,6 +1341,26 @@ bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev) } /** + * xe_configfs_get_disable_vram_page_offline - get configfs disable_vram_page_offline setting + * @pdev: pci device + * + * Return: disable_vram_page_offline setting in configfs + */ +bool xe_configfs_get_disable_vram_page_offline(struct pci_dev *pdev) +{ + struct xe_config_group_device *dev = find_xe_config_group_device(pdev); + bool ret; + + if (!dev) + return device_defaults.disable_vram_page_offline; + + ret = dev->config.disable_vram_page_offline; + config_group_put(&dev->group); + + return ret; +} + +/** * xe_configfs_get_ctx_restore_mid_bb - get configfs ctx_restore_mid_bb setting * @pdev: pci device * @class: hw engine class diff --git a/drivers/gpu/drm/xe/xe_configfs.h b/drivers/gpu/drm/xe/xe_configfs.h index 4fbbeafba473..42cd1a491d01 100644 --- a/drivers/gpu/drm/xe/xe_configfs.h +++ b/drivers/gpu/drm/xe/xe_configfs.h @@ -24,6 +24,7 @@ bool xe_configfs_media_gt_allowed(struct pci_dev *pdev); u64 xe_configfs_get_engines_allowed(struct pci_dev *pdev); bool xe_configfs_get_psmi_enabled(struct pci_dev *pdev); bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev); +bool xe_configfs_get_disable_vram_page_offline(struct pci_dev *pdev); u32 xe_configfs_get_ctx_restore_mid_bb(struct pci_dev *pdev, enum xe_engine_class class, const u32 **cs); @@ -44,6 +45,7 @@ static inline bool xe_configfs_media_gt_allowed(struct pci_dev *pdev) { return t static inline u64 xe_configfs_get_engines_allowed(struct pci_dev *pdev) { return U64_MAX; } static inline bool xe_configfs_get_psmi_enabled(struct pci_dev *pdev) { return false; } static inline bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev) { return true; } +static inline bool xe_configfs_get_disable_vram_page_offline(struct pci_dev *pdev) { return false; } static inline u32 xe_configfs_get_ctx_restore_mid_bb(struct pci_dev *pdev, enum xe_engine_class class, const u32 **cs) { return 0; } diff --git a/drivers/gpu/drm/xe/xe_debugfs.c b/drivers/gpu/drm/xe/xe_debugfs.c index 28135f84e286..80f62634fae5 100644 --- a/drivers/gpu/drm/xe/xe_debugfs.c +++ b/drivers/gpu/drm/xe/xe_debugfs.c @@ -32,6 +32,7 @@ #include "xe_sriov_vf.h" #include "xe_step.h" #include "xe_tile_debugfs.h" +#include "xe_ttm_vram_mgr.h" #include "xe_vsec.h" #include "xe_wa.h" @@ -44,12 +45,18 @@ DECLARE_FAULT_ATTR(gt_reset_failure); DECLARE_FAULT_ATTR(inject_csc_hw_error); DECLARE_FAULT_ATTR(wedge_cold_reset); +DECLARE_FAULT_ATTR(inject_mempage_offline); static bool csc_hw_error_available(struct xe_device *xe) { return !IS_SRIOV_VF(xe) && xe->info.platform == XE_BATTLEMAGE; } +static bool is_crescent_island_pf(struct xe_device *xe) +{ + return !IS_SRIOV_VF(xe) && xe->info.platform == XE_CRESCENTISLAND; +} + /* * Fault injection table. Each entry registers a debugfs attribute; add a * matching FAULT_ACTION() below for every entry added here. @@ -66,6 +73,9 @@ static struct { .is_visible = csc_hw_error_available }, { .name = "wedge_cold_reset", .attr = &wedge_cold_reset }, + { .name = "inject_mempage_offline", + .attr = &inject_mempage_offline, + .is_visible = is_crescent_island_pf }, }; /* @@ -81,6 +91,40 @@ bool xe_fault_##name(void) \ FAULT_ACTION(gt_reset, gt_reset_failure) FAULT_ACTION(csc_hw_error, inject_csc_hw_error) FAULT_ACTION(wedge_cold_reset, wedge_cold_reset) +FAULT_ACTION(mempage_offline, inject_mempage_offline) + +static ssize_t inject_mempage_offline_trigger(struct file *f, + const char __user *ubuf, + size_t size, loff_t *pos) +{ + struct xe_device *xe = file_inode(f)->i_private; + struct xe_tile *tile = xe_device_get_root_tile(xe); + struct xe_vram_region *vr = tile->mem.vram; + u64 pfn; + int ret; + + if (!vr) + return -ENODEV; + + ret = kstrtou64_from_user(ubuf, size, 0, &pfn); + if (ret) + return ret; + + if (!xe_fault_mempage_offline()) + return size; + + xe_warn(xe, "Page offlining test interface accessed. Notice: Offlined or reserved memory pages cannot be reclaimed dynamically. A driver rebind (unbind and bind loop) is required post-test to clean up.\n"); + if (pfn == 0) + return xe_ttm_vram_inject_fault(xe) ?: size; + + /* User provided PFN - convert to DPA and inject */ + return xe_ttm_vram_handle_addr_fault(xe, pfn << PAGE_SHIFT) ?: size; +} + +static const struct file_operations inject_mempage_offline_fops = { + .owner = THIS_MODULE, + .write = inject_mempage_offline_trigger, +}; static void xe_fault_inject_debugfs_register(struct xe_device *xe, struct dentry *root) @@ -95,6 +139,11 @@ static void xe_fault_inject_debugfs_register(struct xe_device *xe, fault_create_debugfs_attr(xe_fault_inject_entry[i].name, root, xe_fault_inject_entry[i].attr); } + + if (is_crescent_island_pf(xe)) { + debugfs_create_file("inject_mempage_offline_trigger", 0200, + root, xe, &inject_mempage_offline_fops); + } } static void read_residency_counter(struct xe_device *xe, struct xe_mmio *mmio, @@ -773,6 +822,8 @@ void xe_debugfs_register(struct xe_device *xe) if (man) ttm_resource_manager_create_debugfs(man, root, "stolen_mm"); + xe_ttm_vram_debugfs_init(xe, root); + for_each_tile(tile, xe, tile_id) xe_tile_debugfs_register(tile); diff --git a/drivers/gpu/drm/xe/xe_debugfs.h b/drivers/gpu/drm/xe/xe_debugfs.h index 0dcd28fd7dc0..88d91c78036b 100644 --- a/drivers/gpu/drm/xe/xe_debugfs.h +++ b/drivers/gpu/drm/xe/xe_debugfs.h @@ -14,11 +14,13 @@ struct xe_device; bool xe_fault_gt_reset(void); bool xe_fault_csc_hw_error(void); bool xe_fault_wedge_cold_reset(void); +bool xe_fault_mempage_offline(void); void xe_debugfs_register(struct xe_device *xe); #else static inline bool xe_fault_gt_reset(void) { return false; } static inline bool xe_fault_csc_hw_error(void) { return false; } static inline bool xe_fault_wedge_cold_reset(void) { return false; } +static inline bool xe_fault_mempage_offline(void) { return false; } static inline void xe_debugfs_register(struct xe_device *xe) { } #endif diff --git a/drivers/gpu/drm/xe/xe_device.c b/drivers/gpu/drm/xe/xe_device.c index 396d02eb2af8..205cb4e7f9e8 100644 --- a/drivers/gpu/drm/xe/xe_device.c +++ b/drivers/gpu/drm/xe/xe_device.c @@ -50,6 +50,7 @@ #include "xe_late_bind_fw.h" #include "xe_log.h" #include "xe_mmio.h" +#include "xe_mmio_gem.h" #include "xe_module.h" #include "xe_nvm.h" #include "xe_oa.h" @@ -111,6 +112,8 @@ static int xe_file_open(struct drm_device *dev, struct drm_file *file) mutex_init(&xef->exec_queue.lock); xa_init_flags(&xef->exec_queue.xa, XA_FLAGS_ALLOC1); + mutex_init(&xef->mmio_gem.lock); + file->driver_priv = xef; kref_init(&xef->refcount); @@ -133,6 +136,8 @@ static void xe_file_destroy(struct kref *ref) xa_destroy(&xef->vm.xa); mutex_destroy(&xef->vm.lock); + mutex_destroy(&xef->mmio_gem.lock); + xe_drm_client_put(xef->client); kfree(xef->process_name); kfree(xef); @@ -189,6 +194,13 @@ static void xe_file_close(struct drm_device *dev, struct drm_file *file) xa_for_each(&xef->vm.xa, idx, vm) xe_vm_close_and_put(vm); + scoped_guard(mutex, &xef->mmio_gem.lock) { + if (xef->mmio_gem.pci_barrier) { + xe_mmio_gem_destroy(xef->mmio_gem.pci_barrier, file); + xef->mmio_gem.pci_barrier = NULL; + } + } + xe_file_put(xef); } @@ -258,95 +270,6 @@ static long xe_drm_compat_ioctl(struct file *file, unsigned int cmd, unsigned lo #define xe_drm_compat_ioctl NULL #endif -static void barrier_open(struct vm_area_struct *vma) -{ - drm_dev_get(vma->vm_private_data); -} - -static void barrier_close(struct vm_area_struct *vma) -{ - drm_dev_put(vma->vm_private_data); -} - -static void barrier_release_dummy_page(struct drm_device *dev, void *res) -{ - struct page *dummy_page = (struct page *)res; - - __free_page(dummy_page); -} - -static vm_fault_t barrier_fault(struct vm_fault *vmf) -{ - struct drm_device *dev = vmf->vma->vm_private_data; - struct vm_area_struct *vma = vmf->vma; - vm_fault_t ret = VM_FAULT_NOPAGE; - pgprot_t prot; - int idx; - - prot = vma_get_page_prot(vma); - - if (drm_dev_enter(dev, &idx)) { - unsigned long pfn; - -#define LAST_DB_PAGE_OFFSET 0x7ff001 - pfn = PHYS_PFN(pci_resource_start(to_pci_dev(dev->dev), 0) + - LAST_DB_PAGE_OFFSET); - ret = vmf_insert_pfn_prot(vma, vma->vm_start, pfn, - pgprot_noncached(prot)); - drm_dev_exit(idx); - } else { - struct page *page; - - /* Allocate new dummy page to map all the VA range in this VMA to it*/ - page = alloc_page(GFP_KERNEL | __GFP_ZERO); - if (!page) - return VM_FAULT_OOM; - - /* Set the page to be freed using drmm release action */ - if (drmm_add_action_or_reset(dev, barrier_release_dummy_page, page)) - return VM_FAULT_OOM; - - ret = vmf_insert_pfn_prot(vma, vma->vm_start, page_to_pfn(page), - prot); - } - - return ret; -} - -static const struct vm_operations_struct vm_ops_barrier = { - .open = barrier_open, - .close = barrier_close, - .fault = barrier_fault, -}; - -static int xe_pci_barrier_mmap(struct file *filp, - struct vm_area_struct *vma) -{ - struct drm_file *priv = filp->private_data; - struct drm_device *dev = priv->minor->dev; - struct xe_device *xe = to_xe_device(dev); - - if (!IS_DGFX(xe)) - return -EINVAL; - - if (vma->vm_end - vma->vm_start > SZ_4K) - return -EINVAL; - - if (vma_is_cow_mapping(vma)) - return -EINVAL; - - if (vma->vm_flags & (VM_READ | VM_EXEC)) - return -EINVAL; - - vm_flags_clear(vma, VM_MAYREAD | VM_MAYEXEC); - vm_flags_set(vma, VM_PFNMAP | VM_DONTEXPAND | VM_DONTDUMP | VM_IO); - vma->vm_ops = &vm_ops_barrier; - vma->vm_private_data = dev; - drm_dev_get(vma->vm_private_data); - - return 0; -} - static int xe_mmap(struct file *filp, struct vm_area_struct *vma) { struct drm_file *priv = filp->private_data; @@ -355,11 +278,6 @@ static int xe_mmap(struct file *filp, struct vm_area_struct *vma) if (drm_dev_is_unplugged(dev)) return -ENODEV; - switch (vma->vm_pgoff) { - case XE_PCI_BARRIER_MMAP_OFFSET >> XE_PTE_SHIFT: - return xe_pci_barrier_mmap(filp, vma); - } - return drm_gem_mmap(filp, vma); } @@ -1051,6 +969,10 @@ int xe_device_probe(struct xe_device *xe) if (err) return err; + err = xe_vram_reserve_memtest_bo(xe); + if (err) + return err; + for_each_tile(tile, xe, id) { err = xe_tile_init(tile); if (err) @@ -1067,6 +989,10 @@ int xe_device_probe(struct xe_device *xe) return err; } + err = xe_vram_memtest(xe); + if (err) + return err; + err = xe_pagefault_init(xe); if (err) return err; @@ -1270,7 +1196,7 @@ bool xe_device_is_l2_flush_optimized(struct xe_device *xe) return false; } -void xe_device_l2_flush(struct xe_device *xe) +void xe_device_l2_flush(struct xe_device *xe, bool force) { struct xe_gt *gt; @@ -1278,7 +1204,7 @@ void xe_device_l2_flush(struct xe_device *xe) if (!gt) return; - if (!XE_GT_WA(gt, 16023588340)) + if (!force && !XE_GT_WA(gt, 16023588340)) return; CLASS(xe_force_wake, fw_ref)(gt_to_fw(gt), XE_FW_GT); @@ -1333,7 +1259,7 @@ void xe_device_td_flush(struct xe_device *xe) if (XE_GT_WA(root_gt, 16023588340)) { /* A transient flush is not sufficient: flush the L2 */ - xe_device_l2_flush(xe); + xe_device_l2_flush(xe, false); } else { xe_guc_pc_apply_flush_freq_limit(&root_gt->uc.guc.pc); tdf_request_sync(xe); diff --git a/drivers/gpu/drm/xe/xe_device.h b/drivers/gpu/drm/xe/xe_device.h index 6c4cfaebc44a..6d3d6d5eba29 100644 --- a/drivers/gpu/drm/xe/xe_device.h +++ b/drivers/gpu/drm/xe/xe_device.h @@ -205,7 +205,7 @@ u64 xe_device_uncanonicalize_addr(struct xe_device *xe, u64 address); bool xe_device_is_l2_flush_optimized(struct xe_device *xe); void xe_device_td_flush(struct xe_device *xe); -void xe_device_l2_flush(struct xe_device *xe); +void xe_device_l2_flush(struct xe_device *xe, bool force); static inline bool xe_device_wedged(struct xe_device *xe) { diff --git a/drivers/gpu/drm/xe/xe_device_types.h b/drivers/gpu/drm/xe/xe_device_types.h index 180d450a6deb..4661bfce2f4e 100644 --- a/drivers/gpu/drm/xe/xe_device_types.h +++ b/drivers/gpu/drm/xe/xe_device_types.h @@ -40,6 +40,7 @@ struct intel_display; struct intel_dg_nvm_dev; struct xe_ggtt; struct xe_i2c; +struct xe_mmio_gem; struct xe_pat_ops; struct xe_pxp; struct xe_ttm_stolen_mgr; @@ -116,6 +117,12 @@ struct xe_device { /** @devcoredump: device coredump */ struct xe_devcoredump devcoredump; + /** @desc: device descriptor */ + const struct xe_device_desc *desc; + + /** @subplatform_desc: subplatform descriptor */ + const struct xe_subplatform_desc *subplatform_desc; + /** @info: device info */ struct intel_device_info { /** @info.platform_name: platform name */ @@ -676,6 +683,18 @@ struct xe_file { /** @refcount: ref count of this xe file */ struct kref refcount; + + /** @mmio_gem: MMIO GEM objects for this xe file */ + struct { + /** + * @mmio_gem.lock: Protects allocation and attach of MMIO + * GEM objects on first use (singleton). All MMIO GEM access + * should be guarded by this lock. Prefer scoped_guard(). + */ + struct mutex lock; + /** @mmio_gem.pci_barrier: MMIO GEM object for PCI barrier mmap. */ + struct xe_mmio_gem *pci_barrier; + } mmio_gem; }; #endif diff --git a/drivers/gpu/drm/xe/xe_dma_buf.c b/drivers/gpu/drm/xe/xe_dma_buf.c index bf0728838ead..5d9f1cd24b7f 100644 --- a/drivers/gpu/drm/xe/xe_dma_buf.c +++ b/drivers/gpu/drm/xe/xe_dma_buf.c @@ -104,6 +104,9 @@ static struct sg_table *xe_dma_buf_map(struct dma_buf_attachment *attach, struct sg_table *sgt; int r = 0; + if (xe_bo_is_purged(bo)) + return ERR_PTR(-ENOENT); + if (!attach->peer2peer && !xe_bo_can_migrate(bo, XE_PL_TT)) return ERR_PTR(-EOPNOTSUPP); diff --git a/drivers/gpu/drm/xe/xe_drm_ras_types.h b/drivers/gpu/drm/xe/xe_drm_ras_types.h index 8d729ad6a264..0be218ba2db7 100644 --- a/drivers/gpu/drm/xe/xe_drm_ras_types.h +++ b/drivers/gpu/drm/xe/xe_drm_ras_types.h @@ -43,6 +43,9 @@ struct xe_drm_ras { /** @info: info array for all types of errors */ struct xe_drm_ras_counter *info[DRM_XE_RAS_ERR_SEV_MAX]; + + /** @disable_vram_page_offline: cached configfs policy, immutable after init */ + bool disable_vram_page_offline; }; #endif diff --git a/drivers/gpu/drm/xe/xe_exec_queue.c b/drivers/gpu/drm/xe/xe_exec_queue.c index c4213bb9c137..e63559a2f582 100644 --- a/drivers/gpu/drm/xe/xe_exec_queue.c +++ b/drivers/gpu/drm/xe/xe_exec_queue.c @@ -322,10 +322,66 @@ struct xe_lrc *xe_exec_queue_lrc(struct xe_exec_queue *q) return q->lrc[0]; } +/* + * Publish the queue back-pointer in the LRC BOs. + * + * The BO holds no reference on the queue; the queue owns the LRCs, and + * therefore the BOs, instead. The back-pointer is made safe by two rules: + * + * - It is published only once the queue is fully constructed and can no + * longer be destroyed by an error path that bypasses the kref (see + * xe_exec_queue_create()), so a reader that successfully takes a + * reference can never be handed a queue that is freed without going + * through xe_exec_queue_destroy(). + * + * - It is written and cleared under the BO dma_resv. Readers must hold + * the same lock across both the read and + * xe_exec_queue_get_unless_zero(), which serializes them against + * xe_exec_queue_clear_lrc_bo_backpointer() below. + * + * For a multi-queue group the LRC BOs point at the primary queue, which is + * kept alive by the reference every secondary holds on it. + */ +static void xe_exec_queue_set_lrc_bo_backpointer(struct xe_exec_queue *q) +{ + struct xe_exec_queue *primary = xe_exec_queue_multi_queue_primary(q); + int i; + + for (i = 0; i < q->width; ++i) { + struct xe_bo *bo = q->lrc[i]->bo; + + xe_bo_lock(bo, false); + bo->q = primary; + xe_bo_unlock(bo); + } +} + +/* + * Drop the queue back-pointer before anything belonging to the queue is + * torn down. This must happen before q->ops->fini(), otherwise a reader + * could take a reference and then operate on an already destroyed backend. + */ +static void xe_exec_queue_clear_lrc_bo_backpointer(struct xe_exec_queue *q) +{ + int i; + + for (i = 0; i < q->width; ++i) { + struct xe_bo *bo = q->lrc[i] ? q->lrc[i]->bo : NULL; + + if (!bo) + continue; + + xe_bo_lock(bo, false); + bo->q = NULL; + xe_bo_unlock(bo); + } +} + static void __xe_exec_queue_fini(struct xe_exec_queue *q) { int i; + xe_exec_queue_clear_lrc_bo_backpointer(q); q->ops->fini(q); for (i = 0; i < q->width; ++i) @@ -450,6 +506,14 @@ struct xe_exec_queue *xe_exec_queue_create(struct xe_device *xe, struct xe_vm *v goto err_post_init; } + /* + * Publish the LRC BO back-pointers last: past this point the queue can + * only be destroyed through xe_exec_queue_destroy(), so a concurrent + * reader that takes a reference via bo->q cannot race with the + * kref-bypassing error paths below. + */ + xe_exec_queue_set_lrc_bo_backpointer(q); + return q; err_post_init: @@ -1566,8 +1630,12 @@ void xe_exec_queue_update_run_ticks(struct xe_exec_queue *q) * errors. */ lrc = q->lrc[0]; - new_ts = xe_lrc_update_timestamp(lrc, &old_ts); - q->xef->run_ticks[q->class] += (new_ts - old_ts) * q->width; + xe_bo_lock(lrc->bo, false); + if (!xe_bo_is_purged(lrc->bo)) { + new_ts = xe_lrc_update_timestamp(lrc, &old_ts); + q->xef->run_ticks[q->class] += (new_ts - old_ts) * q->width; + } + xe_bo_unlock(lrc->bo); drm_dev_exit(idx); } diff --git a/drivers/gpu/drm/xe/xe_exec_queue_types.h b/drivers/gpu/drm/xe/xe_exec_queue_types.h index 95f75d61a647..836f88fc0faa 100644 --- a/drivers/gpu/drm/xe/xe_exec_queue_types.h +++ b/drivers/gpu/drm/xe/xe_exec_queue_types.h @@ -154,6 +154,9 @@ struct xe_exec_queue { */ unsigned long flags; + /** @ban_reason: Bitmask of ban reasons (DRM_XE_EXEC_QUEUE_BAN_REASON_*) */ + atomic_t ban_reason; + union { /** @multi_gt_list: list head for VM bind engines if multi-GT */ struct list_head multi_gt_list; @@ -348,8 +351,8 @@ struct xe_exec_queue_ops { * signalled when this function is called. */ void (*resume)(struct xe_exec_queue *q); - /** @reset_status: check exec queue reset status */ - bool (*reset_status)(struct xe_exec_queue *q); + /** @reset_status: check exec queue ban status, returns ban reason bitmask */ + u64 (*reset_status)(struct xe_exec_queue *q); }; #endif diff --git a/drivers/gpu/drm/xe/xe_execlist.c b/drivers/gpu/drm/xe/xe_execlist.c index 0d0db66c6ea2..a36db39dcda8 100644 --- a/drivers/gpu/drm/xe/xe_execlist.c +++ b/drivers/gpu/drm/xe/xe_execlist.c @@ -453,10 +453,10 @@ static void execlist_exec_queue_resume(struct xe_exec_queue *q) /* NIY */ } -static bool execlist_exec_queue_reset_status(struct xe_exec_queue *q) +static u64 execlist_exec_queue_reset_status(struct xe_exec_queue *q) { /* NIY */ - return false; + return 0; } static const struct xe_exec_queue_ops execlist_exec_queue_ops = { diff --git a/drivers/gpu/drm/xe/xe_gt.c b/drivers/gpu/drm/xe/xe_gt.c index 478e047031f4..775c826b68b4 100644 --- a/drivers/gpu/drm/xe/xe_gt.c +++ b/drivers/gpu/drm/xe/xe_gt.c @@ -70,6 +70,8 @@ #include "xe_wa.h" #include "xe_wopcm.h" +#define GRDOM_RESET_TIMEOUT_MS 5 + struct xe_gt *xe_gt_alloc(struct xe_tile *tile) { struct xe_device *xe = tile_to_xe(tile); @@ -828,10 +830,13 @@ static int do_gt_reset(struct xe_gt *gt) xe_gsc_wa_14015076503(gt, true); xe_mmio_write32(>->mmio, GDRST, GRDOM_FULL); - err = xe_mmio_wait32(>->mmio, GDRST, GRDOM_FULL, 0, 5000, NULL, false); + err = xe_mmio_wait32(>->mmio, GDRST, GRDOM_FULL, 0, + GRDOM_RESET_TIMEOUT_MS * USEC_PER_MSEC, + NULL, false); if (err) - xe_gt_err(gt, "failed to clear GRDOM_FULL (%pe)\n", - ERR_PTR(err)); + xe_log_err(gt, GT, err, + "full graphics reset not completed in %u ms\n", + GRDOM_RESET_TIMEOUT_MS); xe_gsc_wa_14015076503(gt, false); diff --git a/drivers/gpu/drm/xe/xe_gt_debugfs.c b/drivers/gpu/drm/xe/xe_gt_debugfs.c index 361a70234d1f..bb09e70ee44c 100644 --- a/drivers/gpu/drm/xe/xe_gt_debugfs.c +++ b/drivers/gpu/drm/xe/xe_gt_debugfs.c @@ -13,6 +13,7 @@ #include <drm/drm_managed.h> #include <linux/math.h> +#include "regs/xe_engine_regs.h" #include "regs/xe_gt_regs.h" #include "xe_device.h" #include "xe_force_wake.h" @@ -25,6 +26,7 @@ #include "xe_gt_stats.h" #include "xe_gt_topology.h" #include "xe_guc_hwconfig.h" +#include "xe_guc_submit.h" #include "xe_hw_engine.h" #include "xe_lrc.h" #include "xe_mmio.h" @@ -130,6 +132,48 @@ static int hw_engines(struct xe_gt *gt, struct drm_printer *p) return 0; } +static int multi_queue_active_lrca(struct xe_gt *gt, struct drm_printer *p) +{ + struct xe_guc *guc = >->uc.guc; + struct xe_hw_engine *hwe; + enum xe_hw_engine_id id; + + for_each_hw_engine(hwe, gt, id) { + u32 cur_lrca, active_id, lrca; + unsigned int fw_ref; + + if (!xe_gt_supports_multi_queue(gt, hwe->class)) + continue; + + /* + * Forcewake is dropped before xe_guc_submit_active_multi_queue_lrca() + * below, which takes guc->submission_state.lock, to avoid holding a + * GT forcewake ref across a mutex acquired elsewhere in the opposite + * order. + */ + fw_ref = xe_force_wake_get(gt_to_fw(gt), XE_FORCEWAKE_ALL); + if (!xe_force_wake_ref_has_domain(fw_ref, XE_FORCEWAKE_ALL)) { + drm_printf(p, "%s\tforcewake failed, skipping\n", hwe->name); + xe_force_wake_put(gt_to_fw(gt), fw_ref); + continue; + } + + cur_lrca = xe_mmio_read32(>->mmio, + RING_CURRENT_LRCA(hwe->mmio_base)); + active_id = xe_lrc_get_multi_queue_active_queue_id(hwe); + + xe_force_wake_put(gt_to_fw(gt), fw_ref); + + lrca = xe_guc_submit_active_multi_queue_lrca(guc, hwe, cur_lrca, + active_id); + + drm_printf(p, "%s\tactive_queue_id %u\tcurrent_lrca 0x%08x\tactive_lrca 0x%08x\n", + hwe->name, active_id, cur_lrca, lrca); + } + + return 0; +} + static int steering(struct xe_gt *gt, struct drm_printer *p) { xe_gt_mcr_steering_dump(gt, p); @@ -254,6 +298,11 @@ static const struct drm_info_list pf_only_debugfs_list[] = { { "steering", .show = xe_gt_debugfs_show_with_rpm, .data = steering }, }; +static const struct drm_info_list multi_queue_debugfs_list[] = { + { "multi_queue_active_lrca", + .show = xe_gt_debugfs_show_with_rpm, .data = multi_queue_active_lrca }, +}; + static ssize_t write_to_gt_call(const char __user *userbuf, size_t count, loff_t *ppos, void (*call)(struct xe_gt *), struct xe_gt *gt) { @@ -521,6 +570,11 @@ void xe_gt_debugfs_register(struct xe_gt *gt) ARRAY_SIZE(pf_only_debugfs_list), root, minor); + if (!IS_SRIOV_VF(xe) && xe_gt_has_multi_queue(gt)) + drm_debugfs_create_files(multi_queue_debugfs_list, + ARRAY_SIZE(multi_queue_debugfs_list), + root, minor); + if (xe_gt_is_main_type(gt) && !IS_DGFX(xe) && !IS_SRIOV_VF(xe)) debugfs_create_file("gt_ia_bias", 0600, root, gt, >_ia_bias_fops); diff --git a/drivers/gpu/drm/xe/xe_guc_pagefault.c b/drivers/gpu/drm/xe/xe_guc_pagefault.c index 8f8210a732e9..df237fd40551 100644 --- a/drivers/gpu/drm/xe/xe_guc_pagefault.c +++ b/drivers/gpu/drm/xe/xe_guc_pagefault.c @@ -108,7 +108,13 @@ int xe_guc_pagefault_handler(struct xe_guc *guc, u32 *msg, u32 len) << PFD_VIRTUAL_ADDR_HI_SHIFT) | (FIELD_GET(PFD_VIRTUAL_ADDR_LO, msg[2]) << PFD_VIRTUAL_ADDR_LO_SHIFT); - pf.consumer.asid = FIELD_GET(PFD_ASID, msg[1]); + + BUILD_BUG_ON(XE_MAX_ASID > XE_PAGEFAULT_ASID_MASK); + + pf.consumer.id = FIELD_PREP(XE_PAGEFAULT_ASID_MASK, + FIELD_GET(PFD_ASID, msg[1])) | + FIELD_PREP(XE_PAGEFAULT_SRCID_MASK, + FIELD_GET(PFD_SRC_ID, msg[0])); pf.consumer.access_type = FIELD_GET(PFD_ACCESS_TYPE, msg[2]) | (FIELD_GET(PFD_PREFETCH, msg[2]) ? XE_PAGEFAULT_ACCESS_PREFETCH : 0); if (FIELD_GET(XE2_PFD_TRVA_FAULT, msg[0])) diff --git a/drivers/gpu/drm/xe/xe_guc_submit.c b/drivers/gpu/drm/xe/xe_guc_submit.c index 99d8c807ff05..f3ba8abfc228 100644 --- a/drivers/gpu/drm/xe/xe_guc_submit.c +++ b/drivers/gpu/drm/xe/xe_guc_submit.c @@ -6,6 +6,7 @@ #include "xe_guc_submit.h" #include <linux/bitfield.h> +#include <uapi/drm/xe_drm.h> #include <linux/bitmap.h> #include <linux/circ_buf.h> #include <linux/dma-fence-array.h> @@ -34,6 +35,7 @@ #include "xe_guc_klv_helpers.h" #include "xe_guc_submit_types.h" #include "xe_hw_engine.h" +#include "xe_log.h" #include "xe_lrc.h" #include "xe_macros.h" #include "xe_map.h" @@ -1599,6 +1601,12 @@ guc_exec_queue_timedout_job(struct drm_sched_job *drm_job) else wedged = xe_device_wedged(xe); + /* + * Only tag as GPU hang if this is the original timeout, not a + * consequence of a prior kill (e.g., page-offline). + */ + if (!exec_queue_killed(q)) + atomic_or(DRM_XE_EXEC_QUEUE_BAN_REASON_GPU_HANG, &q->ban_reason); set_exec_queue_banned(q); /* Kick job / queue off hardware */ @@ -1682,6 +1690,9 @@ trigger_reset: if (timeout_needs_gt_reset(q, job, skip_timeout_check)) { if (!xe_sched_invalidate_job(job, 2)) { clear_exec_queue_banned(q); + /* protect concurrent page offline reasons */ + atomic_andnot(DRM_XE_EXEC_QUEUE_BAN_REASON_GPU_HANG, + &q->ban_reason); xe_gt_reset_async(q->gt); goto rearm; } @@ -2580,13 +2591,29 @@ static void guc_exec_queue_multi_queue_drop_suspend(struct xe_exec_queue *q) } } -static bool guc_exec_queue_reset_status(struct xe_exec_queue *q) +static u64 guc_exec_queue_reset_status(struct xe_exec_queue *q) { - if (xe_exec_queue_is_multi_queue_secondary(q) && - guc_exec_queue_reset_status(xe_exec_queue_multi_queue_primary(q))) - return true; + /* TODO: In case of multiqueue, if a secondary queue is banned due to + * page offlining, checking only the primary queue's GuC reset status + * may mask the true reason or race with it. + */ + if (xe_exec_queue_is_multi_queue_secondary(q)) { + u64 status = guc_exec_queue_reset_status(xe_exec_queue_multi_queue_primary(q)); - return exec_queue_reset(q) || exec_queue_killed_or_banned_or_wedged(q); + if (status) + return status; + } + + if (exec_queue_reset(q) || exec_queue_killed_or_banned_or_wedged(q)) { + u64 reason = atomic_read_acquire(&q->ban_reason); + + /* If no specific reason was recorded, default to GPU hang */ + if (!reason) + reason = DRM_XE_EXEC_QUEUE_BAN_REASON_GPU_HANG; + return reason; + } + + return 0; } /* @@ -3494,8 +3521,9 @@ int xe_guc_exec_queue_reset_failure_handler(struct xe_guc *guc, u32 *msg, u32 le reason = msg[2]; /* Unexpected failure of a hardware feature, log an actual error */ - xe_gt_err(gt, "GuC engine reset request failed on %d:%d because 0x%08X", - guc_class, instance, reason); + xe_log_err(gt, GUCSUBMIT, -EIO, + "engine reset failed on %u:%u, reason=%#x\n", + guc_class, instance, reason); xe_gt_reset_async(gt); @@ -3854,6 +3882,79 @@ bool xe_guc_has_registered_mlrc_queues(struct xe_guc *guc) } /** + * xe_guc_submit_active_multi_queue_lrca() - Resolve the LRCA of the active + * queue in the multi-queue group currently running on an engine. + * @guc: the &xe_guc managing the exec queues + * @hwe: the &xe_hw_engine whose active queue is being resolved + * @cur_lrca: value read from RING_CURRENT_LRCA, identifies the running group + * @active_id: current Active Queue ID read from CSMQDEBUG (position in group) + * + * The running group is identified by matching @cur_lrca against the group's + * primary LRCA; @active_id then selects the active queue within that group. + * + * Return: the LRCA of the active queue, or 0 if no matching queue is found. + */ +u32 xe_guc_submit_active_multi_queue_lrca(struct xe_guc *guc, + struct xe_hw_engine *hwe, + u32 cur_lrca, u32 active_id) +{ + struct xe_exec_queue *q; + unsigned long index; + u32 lrca = 0; + + /* + * submission_state.lock also protects exec_queue teardown: an exec + * queue is removed from exec_queue_lookup before its group/primary + * are freed, so any q found in the xarray below has a live group + * and primary for as long as we hold the lock. + */ + guard(mutex)(&guc->submission_state.lock); + + xa_for_each(&guc->submission_state.exec_queue_lookup, index, q) { + struct xe_exec_queue_group *group = q->multi_queue.group; + struct xe_lrc *active_lrc; + struct xe_lrc *primary_lrc; + + if (!q->multi_queue.valid || !group || !group->primary) + continue; + /* + * Multi-queue exec queues are bound to a hw engine class; + * GuC dynamically schedules them onto one of the class's + * physical instances, so there is no fixed queue-to-instance + * mapping to filter on here. + */ + if (q->class != hwe->class) + continue; + if (q->multi_queue.pos != active_id) + continue; + /* + * LRCAs are page-aligned (4K) addresses in GGTT; the low + * bits reported by RING_CURRENT_LRCA are not meaningful, so + * only compare bits [31:12]. + */ + primary_lrc = xe_exec_queue_get_lrc(group->primary, 0); + if (!primary_lrc) + continue; + + if ((xe_lrc_ggtt_addr(primary_lrc) ^ cur_lrca) & GENMASK(31, 12)) { + xe_lrc_put(primary_lrc); + continue; + } + + active_lrc = xe_exec_queue_get_lrc(q, 0); + xe_lrc_put(primary_lrc); + if (!active_lrc) + continue; + + lrca = xe_lrc_ggtt_addr(active_lrc); + xe_lrc_put(active_lrc); + break; + } + + return lrca; +} + +/** * xe_guc_contexts_hwsp_rebase - Re-compute GGTT references within all * exec queues registered to given GuC. * @guc: the &xe_guc struct instance diff --git a/drivers/gpu/drm/xe/xe_guc_submit.h b/drivers/gpu/drm/xe/xe_guc_submit.h index ccade320dc69..29abf07d8f04 100644 --- a/drivers/gpu/drm/xe/xe_guc_submit.h +++ b/drivers/gpu/drm/xe/xe_guc_submit.h @@ -11,6 +11,7 @@ struct drm_printer; struct xe_exec_queue; struct xe_guc; +struct xe_hw_engine; int xe_guc_submit_init(struct xe_guc *guc, unsigned int num_ids); int xe_guc_submit_enable(struct xe_guc *guc); @@ -55,6 +56,10 @@ void xe_guc_register_vf_exec_queue(struct xe_exec_queue *q, int ctx_type); bool xe_guc_has_registered_mlrc_queues(struct xe_guc *guc); +u32 xe_guc_submit_active_multi_queue_lrca(struct xe_guc *guc, + struct xe_hw_engine *hwe, + u32 cur_lrca, u32 active_id); + int xe_guc_contexts_hwsp_rebase(struct xe_guc *guc, void *scratch); #endif diff --git a/drivers/gpu/drm/xe/xe_hwmon.c b/drivers/gpu/drm/xe/xe_hwmon.c index 5284cab6703d..5edeac961ec3 100644 --- a/drivers/gpu/drm/xe/xe_hwmon.c +++ b/drivers/gpu/drm/xe/xe_hwmon.c @@ -101,11 +101,6 @@ enum sensor_attr_power { #define PWR_ATTR_TO_STR(attr) (((attr) == hwmon_power_max) ? "PL1" : "PL2") -/* - * Timeout for power limit write mailbox command. - */ -#define PL_WRITE_MBX_TIMEOUT_MS (1) - /* Index of memory controller in READ_THERMAL_DATA output */ #define TEMP_INDEX_MCTRL 2 @@ -252,7 +247,7 @@ static int xe_hwmon_pcode_rmw_power_limit(const struct xe_hwmon *hwmon, u32 attr (channel == CHANNEL_CARD) ? WRITE_PSYSGPU_POWER_LIMIT : WRITE_PACKAGE_POWER_LIMIT, 0), - val0, val1, PL_WRITE_MBX_TIMEOUT_MS); + val0, val1, PCODE_DEFAULT_TIMEOUT_MS); if (ret) drm_dbg(&hwmon->xe->drm, "write failed ch %d val0 0x%08x, val1 0x%08x, ret %d\n", channel, val0, val1, ret); diff --git a/drivers/gpu/drm/xe/xe_log.c b/drivers/gpu/drm/xe/xe_log.c index 5549ef6966fd..29eb16db3320 100644 --- a/drivers/gpu/drm/xe/xe_log.c +++ b/drivers/gpu/drm/xe/xe_log.c @@ -10,6 +10,7 @@ #include "xe_device.h" #include "xe_log.h" +#include "xe_pci_types.h" #include "xe_printk.h" static void log_emit_cper(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid, @@ -52,18 +53,24 @@ static const char *log_component_prefix(u32 component) return component ? log_unknown_component_prefix(component) : ""; } -static struct xe_gt *get_gt_safe(struct pci_dev *pdev, u8 id) +static bool allowed_tile_id(struct xe_device *xe, u8 tile_id) { - struct xe_device *xe = pdev_to_xe_device(pdev); + return tile_id < 1 + xe->desc->max_remote_tiles; +} - return xe ? xe_device_get_gt(xe, id) : NULL; +static bool allowed_gt_id(struct xe_device *xe, u8 gt_id) +{ + return gt_id < (1 + xe->desc->max_remote_tiles) * xe->desc->max_gt_per_tile; } -static struct xe_tile *get_tile_safe(struct pci_dev *pdev, u8 id) +static u8 gt_id_to_tile_id(struct xe_device *xe, u8 gt_id) { - struct xe_device *xe = pdev_to_xe_device(pdev); + return gt_id / xe->desc->max_gt_per_tile; +} - return xe && id < xe->info.tile_count ? &xe->tiles[id] : NULL; +static const char *location_suffix(bool valid) +{ + return valid ? ":" : "?"; } static const char *log_location_prefix(struct pci_dev *pdev, u32 location, char *buf, size_t size) @@ -76,17 +83,22 @@ static const char *log_location_prefix(struct pci_dev *pdev, u32 location, char goto unrecognized; strscpy(buf, "", size); } else if (type == XE_LOG_LOCATION_TYPE_TILE) { - struct xe_tile *tile = get_tile_safe(pdev, id); + struct xe_device *xe = xe_any_to_xe(pdev); + bool valid = xe ? allowed_tile_id(xe, id) : false; + const char *pad = location_suffix(valid); - if (!tile) - goto unrecognized; - snprintf(buf, size, "Tile%u: ", id); + pci_WARN(pdev, !valid && IS_ENABLED(CONFIG_DRM_XE_DEBUG), + "LOG: invalid tile identifier: %u\n", id); + snprintf(buf, size, "Tile%u%s ", id, pad); } else if (type == XE_LOG_LOCATION_TYPE_GT) { - struct xe_gt *gt = get_gt_safe(pdev, id); - - if (!gt) - goto unrecognized; - snprintf(buf, size, "Tile%u: GT%u: ", gt->tile->id, id); + struct xe_device *xe = xe_any_to_xe(pdev); + bool valid = xe ? allowed_gt_id(xe, id) : false; + const char *pad = location_suffix(valid); + u8 tile_id = xe ? gt_id_to_tile_id(xe, id) : 0; + + pci_WARN(pdev, !valid && IS_ENABLED(CONFIG_DRM_XE_DEBUG), + "LOG: invalid GT identifier: %u\n", id); + snprintf(buf, size, "Tile%u%s GT%u%s ", tile_id, pad, id, pad); } else { goto unrecognized; } diff --git a/drivers/gpu/drm/xe/xe_lrc.c b/drivers/gpu/drm/xe/xe_lrc.c index 25fe9dbc9141..f1cf1463f1b2 100644 --- a/drivers/gpu/drm/xe/xe_lrc.c +++ b/drivers/gpu/drm/xe/xe_lrc.c @@ -2705,7 +2705,7 @@ static u64 get_queue_timestamp(struct xe_hw_engine *hwe) RING_QUEUE_TIMESTAMP(hwe->mmio_base)); } -static u32 get_multi_queue_active_queue_id(struct xe_hw_engine *hwe) +u32 xe_lrc_get_multi_queue_active_queue_id(struct xe_hw_engine *hwe) { u32 val = xe_mmio_read32(&hwe->gt->mmio, RING_CSMQDEBUG(hwe->mmio_base)); @@ -2739,14 +2739,14 @@ static u64 xe_lrc_multi_queue_timestamp(struct xe_lrc *lrc) if (!hwe) return xe_lrc_queue_timestamp(lrc); - if (get_multi_queue_active_queue_id(hwe) != lrc->multi_queue.pos) + if (xe_lrc_get_multi_queue_active_queue_id(hwe) != lrc->multi_queue.pos) return xe_lrc_queue_timestamp(lrc); /* queue is active, so store the queue timestamp register */ reg_queue_ts = get_queue_timestamp(hwe); /* double check queue and primary queue are both still active */ - if (get_multi_queue_active_queue_id(hwe) != lrc->multi_queue.pos || + if (xe_lrc_get_multi_queue_active_queue_id(hwe) != lrc->multi_queue.pos || !context_active(primary_lrc)) return xe_lrc_queue_timestamp(lrc); diff --git a/drivers/gpu/drm/xe/xe_lrc.h b/drivers/gpu/drm/xe/xe_lrc.h index 7be5e3da8bc8..a8ff4e59a1f4 100644 --- a/drivers/gpu/drm/xe/xe_lrc.h +++ b/drivers/gpu/drm/xe/xe_lrc.h @@ -158,6 +158,7 @@ int xe_lrc_lookup_default_reg_value(struct xe_gt *gt, u32 *xe_lrc_emit_hwe_state_instructions(struct xe_exec_queue *q, u32 *cs); void xe_lrc_set_multi_queue_priority(struct xe_lrc *lrc, enum xe_multi_queue_priority priority); +u32 xe_lrc_get_multi_queue_active_queue_id(struct xe_hw_engine *hwe); struct xe_lrc_snapshot *xe_lrc_snapshot_capture(struct xe_lrc *lrc); void xe_lrc_snapshot_capture_delayed(struct xe_lrc_snapshot *snapshot); diff --git a/drivers/gpu/drm/xe/xe_migrate.c b/drivers/gpu/drm/xe/xe_migrate.c index 75b83687f1b5..ff45c24d8889 100644 --- a/drivers/gpu/drm/xe/xe_migrate.c +++ b/drivers/gpu/drm/xe/xe_migrate.c @@ -87,7 +87,7 @@ struct xe_migrate { #define MAX_PREEMPTDISABLE_TRANSFER SZ_8M /* Around 1ms. */ #define MAX_CCS_LIMITED_TRANSFER SZ_4M /* XE_PAGE_SIZE * (FIELD_MAX(XE2_CCS_SIZE_MASK) + 1) */ #define NUM_KERNEL_PDE 15 -#define NUM_PT_SLOTS 32 +#define NUM_PT_SLOTS 48 #define LEVEL0_PAGE_TABLE_ENCODE_SIZE SZ_2M #define MAX_NUM_PTE 512 #define IDENTITY_OFFSET 256ULL @@ -163,22 +163,20 @@ static u64 xe_migrate_vram_ofs(struct xe_device *xe, u64 addr, bool is_comp_pte) } static void xe_migrate_program_identity(struct xe_device *xe, struct xe_vm *vm, struct xe_bo *bo, - u64 map_ofs, u64 vram_offset, u16 pat_index, u64 pt_2m_ofs) + u64 map_ofs, u64 vram_offset, u16 pat_index, u64 pt_2m_ofs, + u64 pt_4k_ofs) { struct xe_vram_region *vram = xe->mem.vram; resource_size_t dpa_base = xe_vram_region_dpa_base(vram); u64 pos, ofs, flags; u64 entry; - /* XXX: Unclear if this should be usable_size? */ - u64 vram_limit = xe_vram_region_actual_physical_size(vram) + dpa_base; + u64 vram_limit = xe_vram_region_usable_size(vram) + dpa_base; u32 level = 2; ofs = map_ofs + XE_PAGE_SIZE * level + vram_offset * 8; flags = vm->pt_ops->pte_encode_addr(xe, 0, pat_index, level, true, 0); - xe_assert(xe, IS_ALIGNED(xe_vram_region_usable_size(vram), SZ_2M)); - /* * Use 1GB pages when possible, last chunk always use 2M * pages as mixing reserved memory (stolen, WOCPM) with a single @@ -196,8 +194,24 @@ static void xe_migrate_program_identity(struct xe_device *xe, struct xe_vm *vm, true, 0); for (ofs = pt_2m_ofs; pos < vram_limit; - pos += SZ_2M, ofs += 8) + pos += SZ_2M, ofs += 8) { + if (pos + SZ_2M > vram_limit) { + entry = vm->pt_ops->pde_encode_bo(bo, pt_4k_ofs); + xe_map_wr(xe, &bo->vmap, ofs, u64, entry); + + flags = vm->pt_ops->pte_encode_addr(xe, 0, + pat_index, + level - 2, + true, 0); + + for (ofs = pt_4k_ofs; pos < vram_limit; + pos += SZ_4K, ofs += 8) + xe_map_wr(xe, &bo->vmap, ofs, u64, pos | flags); + break; + } + xe_map_wr(xe, &bo->vmap, ofs, u64, pos | flags); + } break; /* Ensure pos == vram_limit assert correct */ } @@ -242,16 +256,17 @@ static void xe_migrate_prepare_vm(struct xe_tile *tile, struct xe_migrate *m, u16 pat_index = xe_cache_pat_idx(xe, XE_CACHE_WB); u8 id = tile->id; u32 num_entries = NUM_PT_SLOTS, num_level = vm->pt_root[id]->level; -#define VRAM_IDENTITY_MAP_COUNT 2 - u32 num_setup = num_level + VRAM_IDENTITY_MAP_COUNT; -#undef VRAM_IDENTITY_MAP_COUNT +#define VRAM_IDENTITY_MAP_PT_COUNT 4 + u32 num_setup = num_level + VRAM_IDENTITY_MAP_PT_COUNT; +#undef VRAM_IDENTITY_MAP_PT_COUNT u32 map_ofs, level, i; struct xe_bo *bo = m->pt_bo, *batch = tile->mem.kernel_bb_pool->bo; - u64 entry, pt29_ofs; + u64 entry; - /* PT30 & PT31 reserved for 2M identity map */ - pt29_ofs = xe_bo_size(bo) - 3 * XE_PAGE_SIZE; - entry = vm->pt_ops->pde_encode_bo(bo, pt29_ofs); + /* PT44..PT47 reserved for 4K and 2M identity map */ + u64 l1_pt_ofs = xe_bo_size(bo) - 5 * XE_PAGE_SIZE; + + entry = vm->pt_ops->pde_encode_bo(bo, l1_pt_ofs); xe_pt_write(xe, &vm->pt_root[id]->bo->vmap, 0, entry); map_ofs = (num_entries - num_setup) * XE_PAGE_SIZE; @@ -347,11 +362,12 @@ static void xe_migrate_prepare_vm(struct xe_tile *tile, struct xe_migrate *m, /* Identity map the entire vram at 256GiB offset */ if (IS_DGFX(xe)) { - u64 pt30_ofs = xe_bo_size(bo) - 2 * XE_PAGE_SIZE; + u64 pt46_ofs = xe_bo_size(bo) - 2 * XE_PAGE_SIZE; resource_size_t actual_phy_size = xe_vram_region_actual_physical_size(xe->mem.vram); + u64 pt44_ofs = xe_bo_size(bo) - 4 * XE_PAGE_SIZE; xe_migrate_program_identity(xe, vm, bo, map_ofs, IDENTITY_OFFSET, - pat_index, pt30_ofs); + pat_index, pt46_ofs, pt44_ofs); xe_assert(xe, actual_phy_size <= (MAX_NUM_PTE - IDENTITY_OFFSET) * SZ_1G); /* @@ -362,12 +378,13 @@ static void xe_migrate_prepare_vm(struct xe_tile *tile, struct xe_migrate *m, u16 comp_pat_index = xe_cache_pat_idx(xe, XE_CACHE_NONE_COMPRESSION); u64 vram_offset = IDENTITY_OFFSET + DIV_ROUND_UP_ULL(actual_phy_size, SZ_1G); - u64 pt31_ofs = xe_bo_size(bo) - XE_PAGE_SIZE; + u64 pt47_ofs = xe_bo_size(bo) - XE_PAGE_SIZE; xe_assert(xe, actual_phy_size <= (MAX_NUM_PTE - IDENTITY_OFFSET - IDENTITY_OFFSET / 2) * SZ_1G); + u64 pt45_ofs = xe_bo_size(bo) - 3 * XE_PAGE_SIZE; xe_migrate_program_identity(xe, vm, bo, map_ofs, vram_offset, - comp_pat_index, pt31_ofs); + comp_pat_index, pt47_ofs, pt45_ofs); } } @@ -381,8 +398,8 @@ static void xe_migrate_suballoc_manager_init(struct xe_migrate *m, u32 map_ofs) * Example layout created above, with root level = 3: * [PT0...PT7]: kernel PT's for copy/clear; 64 or 4KiB PTE's * [PT8]: Kernel PT for VM_BIND, 4 KiB PTE's - * [PT9...PT26]: Userspace PT's for VM_BIND, 4 KiB PTE's - * [PT27 = PDE 0] [PT28 = PDE 1] [PT29 = PDE 2] [PT30 & PT31 = 2M vram identity map] + * [PT9...PT40]: Userspace PT's for VM_BIND, 4 KiB PTE's + * [PT41 = PDE 0] [PT44...PT47 = 4K and 2M vram identity maps] * * This makes the lowest part of the VM point to the pagetables. * Hence the lowest 2M in the vm should point to itself, with a few writes @@ -2633,3 +2650,67 @@ void xe_migrate_job_lock_assert(struct xe_exec_queue *q) #if IS_ENABLED(CONFIG_DRM_XE_KUNIT_TEST) #include "tests/xe_migrate.c" #endif + +#if IS_ENABLED(CONFIG_DRM_XE_DEBUG_MEM) +int xe_migrate_debug_ccs_overlap(struct xe_migrate *m, + struct xe_bo *scratch_bo, + bool write_to_ccs) +{ + struct xe_device *xe = tile_to_xe(m->tile); + struct xe_gt *gt = m->tile->primary_gt; + struct dma_fence *fence; + struct xe_bb *bb; + struct xe_sched_job *job; + u64 first_page_dpa, clear_L0_ofs, scratch_dpa, scratch_L0_ofs; + + if (!xe_device_has_flat_ccs(xe)) + return -EINVAL; + + first_page_dpa = xe_vram_region_dpa_base(m->tile->mem.vram); + clear_L0_ofs = xe_migrate_vram_ofs(xe, first_page_dpa, true); + + scratch_dpa = xe_bo_addr(scratch_bo, 0, XE_PAGE_SIZE); + scratch_L0_ofs = xe_migrate_vram_ofs(xe, scratch_dpa, false); + + bb = xe_bb_new(gt, EMIT_COPY_CCS_DW + 1, xe->info.has_usm); + if (IS_ERR(bb)) { + drm_warn(&xe->drm, "Failed to create bb for VRAM overlap check\n"); + return PTR_ERR(bb); + } + + /* 4MB payload = 8KB CCS metadata */ + if (write_to_ccs) { + emit_copy_ccs(gt, bb, clear_L0_ofs, true, + scratch_L0_ofs, false, SZ_4M); + } else { + emit_copy_ccs(gt, bb, scratch_L0_ofs, false, + clear_L0_ofs, true, SZ_4M); + } + + bb->cs[bb->len++] = MI_BATCH_BUFFER_END; + + job = xe_bb_create_migration_job(m->q, bb, + xe_migrate_batch_base(m, xe->info.has_usm), + 0); + if (!IS_ERR(job)) { + xe_sched_job_add_migrate_flush(job, MI_FLUSH_DW_CCS); + + mutex_lock(&m->job_mutex); + xe_sched_job_arm(job); + + fence = dma_fence_get(&job->drm.s_fence->finished); + xe_sched_job_push(job); + mutex_unlock(&m->job_mutex); + + dma_fence_wait(fence, false); + dma_fence_put(fence); + } else { + drm_warn(&xe->drm, "Failed to create job for VRAM overlap check\n"); + xe_bb_free(bb, NULL); + return PTR_ERR(job); + } + + xe_bb_free(bb, NULL); + return 0; +} +#endif diff --git a/drivers/gpu/drm/xe/xe_migrate.h b/drivers/gpu/drm/xe/xe_migrate.h index c3a268b01768..a9acc62f78f0 100644 --- a/drivers/gpu/drm/xe/xe_migrate.h +++ b/drivers/gpu/drm/xe/xe_migrate.h @@ -182,4 +182,10 @@ static inline void xe_migrate_job_lock_assert(struct xe_exec_queue *q) void xe_migrate_job_lock(struct xe_migrate *m, struct xe_exec_queue *q); void xe_migrate_job_unlock(struct xe_migrate *m, struct xe_exec_queue *q); +#if IS_ENABLED(CONFIG_DRM_XE_DEBUG_MEM) +int xe_migrate_debug_ccs_overlap(struct xe_migrate *m, + struct xe_bo *scratch_bo, + bool write_to_ccs); +#endif + #endif diff --git a/drivers/gpu/drm/xe/xe_mmio_gem.c b/drivers/gpu/drm/xe/xe_mmio_gem.c index 3741ae60f532..3cdc9538957d 100644 --- a/drivers/gpu/drm/xe/xe_mmio_gem.c +++ b/drivers/gpu/drm/xe/xe_mmio_gem.c @@ -5,9 +5,9 @@ #include "xe_mmio_gem.h" +#include <linux/dma-resv.h> #include <drm/drm_drv.h> #include <drm/drm_gem.h> -#include <drm/drm_managed.h> #include "xe_device_types.h" @@ -37,12 +37,24 @@ static vm_fault_t xe_mmio_gem_vm_fault(struct vm_fault *); struct xe_mmio_gem { struct drm_gem_object base; phys_addr_t phys_addr; + struct page *dummy_page; /* protected by the GEM's dma_resv */ + bool destroyed; /* protected by the GEM's dma_resv */ }; +static int xe_mmio_gem_vm_may_split(struct vm_area_struct *area, unsigned long addr) +{ + /* + * Forbid splitting. Together with VM_DONTEXPAND, this keeps the VMA + * matching the GEM object exactly. + */ + return -EINVAL; +} + static const struct vm_operations_struct vm_ops = { .open = drm_gem_vm_open, .close = drm_gem_vm_close, .fault = xe_mmio_gem_vm_fault, + .may_split = xe_mmio_gem_vm_may_split, }; static const struct drm_gem_object_funcs xe_mmio_gem_funcs = { @@ -121,6 +133,8 @@ static void xe_mmio_gem_free(struct drm_gem_object *base) { struct xe_mmio_gem *obj = to_xe_mmio_gem(base); + if (obj->dummy_page) + __free_page(obj->dummy_page); drm_gem_object_release(base); kfree(obj); } @@ -128,15 +142,31 @@ static void xe_mmio_gem_free(struct drm_gem_object *base) /** * xe_mmio_gem_destroy - Destroy the GEM object that exposes an MMIO region * @gem: the GEM object to destroy + * @file: DRM file descriptor previously passed to xe_mmio_gem_create() * * This function releases resources associated with the GEM object created by * xe_mmio_gem_create(). * * See: "Exposing MMIO regions to userspace" */ -void xe_mmio_gem_destroy(struct xe_mmio_gem *gem) +void xe_mmio_gem_destroy(struct xe_mmio_gem *gem, struct drm_file *file) { - xe_mmio_gem_free(&gem->base); + struct drm_gem_object *base = &gem->base; + struct drm_device *dev = base->dev; + + drm_vma_node_revoke(&base->vma_node, file); + + dma_resv_lock(base->resv, NULL); + gem->destroyed = true; + dma_resv_unlock(base->resv); + /* + * Setting 'destroyed' under lock takes care of the subsequent faults. + * Zap the existing PTEs to cut off access to the real MMIO through + * currently mapped pages. + */ + drm_vma_node_unmap(&base->vma_node, dev->anon_inode->i_mapping); + + drm_gem_object_put(base); } static int xe_mmio_gem_mmap(struct drm_gem_object *base, struct vm_area_struct *vma) @@ -147,61 +177,58 @@ static int xe_mmio_gem_mmap(struct drm_gem_object *base, struct vm_area_struct * if ((vma->vm_flags & VM_SHARED) == 0) return -EINVAL; - /* Set vm_pgoff (used as a fake buffer offset by DRM) to 0 */ - vma->vm_pgoff = 0; + if (vma->vm_flags & VM_EXEC) + return -EINVAL; + vma->vm_page_prot = pgprot_noncached(vma_get_page_prot(vma)); - vm_flags_set(vma, VM_IO | VM_PFNMAP | VM_DONTEXPAND | VM_DONTDUMP | - VM_DONTCOPY | VM_NORESERVE); + vm_flags_mod(vma, VM_IO | VM_PFNMAP | VM_DONTEXPAND | VM_DONTDUMP | + VM_NORESERVE, VM_MAYEXEC); /* Defer actual mapping to the fault handler. */ return 0; } -static void xe_mmio_gem_release_dummy_page(struct drm_device *dev, void *res) +static int alloc_dummy_page_if_needed(struct drm_gem_object *base) { - __free_page((struct page *)res); + struct xe_mmio_gem *obj = to_xe_mmio_gem(base); + + dma_resv_assert_held(base->resv); + if (!obj->dummy_page) + obj->dummy_page = alloc_page(GFP_KERNEL | __GFP_ZERO); + + return obj->dummy_page ? 0 : -ENOMEM; } -static vm_fault_t xe_mmio_gem_vm_fault_dummy_page(struct vm_area_struct *vma) +static vm_fault_t xe_mmio_gem_vm_fault_dummy_page(struct vm_fault *vmf) { + struct vm_area_struct *vma = vmf->vma; struct drm_gem_object *base = vma->vm_private_data; - struct drm_device *dev = base->dev; - vm_fault_t ret = VM_FAULT_NOPAGE; - struct page *page; + struct xe_mmio_gem *obj = to_xe_mmio_gem(base); unsigned long pfn; - unsigned long i; - - page = alloc_page(GFP_KERNEL | __GFP_ZERO); - if (!page) - return VM_FAULT_OOM; - if (drmm_add_action_or_reset(dev, xe_mmio_gem_release_dummy_page, page)) + if (alloc_dummy_page_if_needed(base)) return VM_FAULT_OOM; - pfn = page_to_pfn(page); - - /* Map the entire VMA to the same dummy page */ - for (i = 0; i < base->size; i += PAGE_SIZE) { - unsigned long addr = vma->vm_start + i; + pfn = page_to_pfn(obj->dummy_page); - ret = vmf_insert_pfn(vma, addr, pfn); - if (ret & VM_FAULT_ERROR) - break; - } - - return ret; + return vmf_insert_pfn_prot(vma, vmf->address, pfn, + vm_get_page_prot(vma->vm_flags)); } -static vm_fault_t xe_mmio_gem_vm_fault(struct vm_fault *vmf) +static vm_fault_t xe_mmio_gem_vm_fault_locked(struct vm_fault *vmf) { struct vm_area_struct *vma = vmf->vma; struct drm_gem_object *base = vma->vm_private_data; struct xe_mmio_gem *obj = to_xe_mmio_gem(base); struct drm_device *dev = base->dev; vm_fault_t ret = VM_FAULT_NOPAGE; - unsigned long i; + unsigned long addr, pfn; int idx; + dma_resv_assert_held(base->resv); + if (obj->destroyed) + return VM_FAULT_SIGBUS; + if (!drm_dev_enter(dev, &idx)) { /* * Provide a dummy page to avoid SIGBUS for events such as hot-unplug. @@ -209,18 +236,30 @@ static vm_fault_t xe_mmio_gem_vm_fault(struct vm_fault *vmf) * It is assumed the userspace will receive the notification via some * other channel (e.g. drm uevent). */ - return xe_mmio_gem_vm_fault_dummy_page(vma); + return xe_mmio_gem_vm_fault_dummy_page(vmf); } - for (i = 0; i < base->size; i += PAGE_SIZE) { - unsigned long addr = vma->vm_start + i; - unsigned long phys_addr = obj->phys_addr + i; - - ret = vmf_insert_pfn(vma, addr, PHYS_PFN(phys_addr)); + pfn = PHYS_PFN(obj->phys_addr); + for (addr = vma->vm_start; addr < vma->vm_end; addr += PAGE_SIZE) { + ret = vmf_insert_pfn(vma, addr, pfn); if (ret & VM_FAULT_ERROR) break; + + pfn++; } drm_dev_exit(idx); return ret; } + +static vm_fault_t xe_mmio_gem_vm_fault(struct vm_fault *vmf) +{ + struct vm_area_struct *vma = vmf->vma; + struct drm_gem_object *base = vma->vm_private_data; + vm_fault_t ret; + + dma_resv_lock(base->resv, NULL); + ret = xe_mmio_gem_vm_fault_locked(vmf); + dma_resv_unlock(base->resv); + return ret; +} diff --git a/drivers/gpu/drm/xe/xe_mmio_gem.h b/drivers/gpu/drm/xe/xe_mmio_gem.h index 4b76d5586ebb..80d7795f07c8 100644 --- a/drivers/gpu/drm/xe/xe_mmio_gem.h +++ b/drivers/gpu/drm/xe/xe_mmio_gem.h @@ -15,6 +15,6 @@ struct xe_mmio_gem; struct xe_mmio_gem *xe_mmio_gem_create(struct xe_device *xe, struct drm_file *file, phys_addr_t phys_addr, size_t size); u64 xe_mmio_gem_mmap_offset(struct xe_mmio_gem *gem); -void xe_mmio_gem_destroy(struct xe_mmio_gem *gem); +void xe_mmio_gem_destroy(struct xe_mmio_gem *gem, struct drm_file *file); #endif /* _XE_MMIO_GEM_H_ */ diff --git a/drivers/gpu/drm/xe/xe_pagefault.c b/drivers/gpu/drm/xe/xe_pagefault.c index d348ec408204..aeb56ff5d58e 100644 --- a/drivers/gpu/drm/xe/xe_pagefault.c +++ b/drivers/gpu/drm/xe/xe_pagefault.c @@ -253,12 +253,13 @@ static int xe_pagefault_service(struct xe_pagefault *pf) struct xe_vma *vma = NULL; int err; bool atomic; + u32 asid = FIELD_GET(XE_PAGEFAULT_ASID_MASK, pf->consumer.id); /* Producer flagged this fault to be nacked */ if (pf->consumer.fault_type_level == XE_PAGEFAULT_TYPE_LEVEL_NACK) return -EFAULT; - vm = xe_pagefault_asid_to_vm(xe, pf->consumer.asid); + vm = xe_pagefault_asid_to_vm(xe, asid); if (IS_ERR(vm)) return PTR_ERR(vm); @@ -375,7 +376,7 @@ static bool xe_pagefault_match(struct xe_pagefault *pf, u64 start, { struct xe_device *xe = gt_to_xe(pf->gt); u64 page_addr = pf->consumer.page_addr; - u32 pf_asid = pf->consumer.asid; + u32 pf_asid = FIELD_GET(XE_PAGEFAULT_ASID_MASK, pf->consumer.id); xe_assert(xe, pf->consumer.alloc_state != XE_PAGEFAULT_ALLOC_STATE_FREE); @@ -500,7 +501,7 @@ static bool xe_pagefault_queue_pop(struct xe_pagefault_queue *pf_queue, align = SZ_4K; pf_work->cache.start = ALIGN_DOWN(lpf->consumer.page_addr, align); pf_work->cache.end = pf_work->cache.start + align; - pf_work->cache.asid = lpf->consumer.asid; + pf_work->cache.asid = FIELD_GET(XE_PAGEFAULT_ASID_MASK, lpf->consumer.id); pf_work->cache.pf = lpf; lpf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_ACTIVE; @@ -546,14 +547,16 @@ static void xe_pagefault_print(struct xe_pagefault *pf) u8 engine_class = FIELD_GET(XE_PAGEFAULT_ENGINE_CLASS_MASK, pf->consumer.engine_class_instance); - xe_gt_info(pf->gt, "\n\tASID: %d\n" + xe_gt_info(pf->gt, "\n\tASID: %lu\n" "\tFaulted Address: 0x%08x%08x\n" "\tFaultType: %lu\n" "\tAccessType: %lu\n" "\tFaultLevel: %lu\n" "\tEngineClass: %d %s\n" - "\tEngineInstance: %lu\n", - pf->consumer.asid, + "\tEngineInstance: %lu\n" + "\tSRCID: 0x%02lx\n", + FIELD_GET(XE_PAGEFAULT_ASID_MASK, + pf->consumer.id), upper_32_bits(pf->consumer.page_addr), lower_32_bits(pf->consumer.page_addr), FIELD_GET(XE_PAGEFAULT_TYPE_MASK, @@ -565,7 +568,9 @@ static void xe_pagefault_print(struct xe_pagefault *pf) engine_class, xe_hw_engine_class_to_str(engine_class), FIELD_GET(XE_PAGEFAULT_ENGINE_INSTANCE_MASK, - pf->consumer.engine_class_instance)); + pf->consumer.engine_class_instance), + FIELD_GET(XE_PAGEFAULT_SRCID_MASK, + pf->consumer.id)); } static void xe_pagefault_save_to_vm(struct xe_device *xe, struct xe_pagefault *pf) @@ -578,7 +583,8 @@ static void xe_pagefault_save_to_vm(struct xe_device *xe, struct xe_pagefault *p * mode, return VM anyways. */ down_read(&xe->usm.lock); - vm = xa_load(&xe->usm.asid_to_vm, pf->consumer.asid); + vm = xa_load(&xe->usm.asid_to_vm, + FIELD_GET(XE_PAGEFAULT_ASID_MASK, pf->consumer.id)); if (vm) xe_vm_get(vm); else @@ -619,7 +625,7 @@ static void xe_pagefault_queue_work(struct work_struct *w) const struct xe_pagefault_ops *ops = pf->producer.ops; void *private = pf->producer.private; struct xe_gt *gt = pf->gt; - u32 asid = pf->consumer.asid; + u32 asid = FIELD_GET(XE_PAGEFAULT_ASID_MASK, pf->consumer.id); int err = 0; bool invalidated = false; diff --git a/drivers/gpu/drm/xe/xe_pagefault_types.h b/drivers/gpu/drm/xe/xe_pagefault_types.h index 185d0813fd30..907189b73286 100644 --- a/drivers/gpu/drm/xe/xe_pagefault_types.h +++ b/drivers/gpu/drm/xe/xe_pagefault_types.h @@ -113,8 +113,13 @@ struct xe_pagefault { u8 engine_class_instance; #define XE_PAGEFAULT_ENGINE_CLASS_MASK GENMASK(3, 0) #define XE_PAGEFAULT_ENGINE_INSTANCE_MASK GENMASK(7, 4) - /** @consumer.asid: address space ID */ - u32 asid; + /** + * @consumer.id: address space ID and SRCID, folded into one + * to keep size compact + */ + u32 id; +#define XE_PAGEFAULT_ASID_MASK GENMASK(23, 0) +#define XE_PAGEFAULT_SRCID_MASK GENMASK(31, 24) }; /** * @consumer.end_addr: end address of page fault, diff --git a/drivers/gpu/drm/xe/xe_pci.c b/drivers/gpu/drm/xe/xe_pci.c index 1e04e8ef2611..00b2b5dfb8f6 100644 --- a/drivers/gpu/drm/xe/xe_pci.c +++ b/drivers/gpu/drm/xe/xe_pci.c @@ -752,7 +752,6 @@ struct xe_probed_info { * Probe from the hardware the info required by xe_info_init_early(). */ static int xe_probe_info_early(struct xe_device *xe, - const struct xe_device_desc *desc, struct xe_probed_info *probed_info) { struct pci_dev *pdev = to_pci_dev(xe->drm.dev); @@ -760,7 +759,7 @@ static int xe_probe_info_early(struct xe_device *xe, probed_info->devid = pdev->device; probed_info->revid = pdev->revision; - xe_step_platform_get(desc->platform, probed_info->revid, &probed_info->step); + xe_step_platform_get(xe->desc->platform, probed_info->revid, &probed_info->step); return 0; } @@ -770,10 +769,10 @@ static int xe_probe_info_early(struct xe_device *xe, * passed to the driver at probe time from PCI ID table. */ static int xe_info_init_early(struct xe_device *xe, - const struct xe_device_desc *desc, - const struct xe_subplatform_desc *subplatform_desc, struct xe_probed_info *probed_info) { + const struct xe_subplatform_desc *subplatform_desc = xe->subplatform_desc; + const struct xe_device_desc *desc = xe->desc; int err; xe->info.devid = probed_info->devid; @@ -836,14 +835,13 @@ static int xe_info_init_early(struct xe_device *xe, } static void xe_probe_tile_count(struct xe_device *xe, - const struct xe_device_desc *desc, struct xe_probed_info *probed_info) { struct xe_mmio *mmio; u8 tile_count; u32 mtcfg; - probed_info->tile_count = 1 + desc->max_remote_tiles; + probed_info->tile_count = 1 + xe->desc->max_remote_tiles; /* * Probe for tile count only for platforms that support multiple @@ -946,9 +944,10 @@ static struct xe_gt *alloc_media_gt(struct xe_tile *tile, } static int xe_probe_ips(struct xe_device *xe, - const struct xe_device_desc *desc, struct xe_probed_info *probed_info) { + const struct xe_device_desc *desc = xe->desc; + /* * If this platform supports GMD_ID, we'll detect the proper IP * descriptor to use from hardware registers. @@ -989,14 +988,13 @@ static int xe_probe_ips(struct xe_device *xe, * Probe from the hardware the info required by xe_info_init(). */ static int xe_probe_info(struct xe_device *xe, - const struct xe_device_desc *desc, struct xe_probed_info *probed_info) { int err; - xe_probe_tile_count(xe, desc, probed_info); + xe_probe_tile_count(xe, probed_info); - err = xe_probe_ips(xe, desc, probed_info); + err = xe_probe_ips(xe, probed_info); if (err) return err; @@ -1010,7 +1008,6 @@ static int xe_probe_info(struct xe_device *xe, * present in device info. */ static int xe_info_init(struct xe_device *xe, - const struct xe_device_desc *desc, struct xe_probed_info *probed_info) { const struct xe_ip *graphics_ip; @@ -1208,6 +1205,8 @@ static int __xe_pci_probe(struct pci_dev *pdev, const struct xe_device_desc *des if (IS_ERR(xe)) return PTR_ERR(xe); + xe->desc = desc; + xe->subplatform_desc = subplatform_desc; xe->devres_group = group; pci_set_drvdata(pdev, &xe->drm); @@ -1216,11 +1215,11 @@ static int __xe_pci_probe(struct pci_dev *pdev, const struct xe_device_desc *des pci_set_master(pdev); - err = xe_probe_info_early(xe, desc, &probed_info); + err = xe_probe_info_early(xe, &probed_info); if (err) return err; - err = xe_info_init_early(xe, desc, subplatform_desc, &probed_info); + err = xe_info_init_early(xe, &probed_info); if (err) return err; @@ -1239,11 +1238,11 @@ static int __xe_pci_probe(struct pci_dev *pdev, const struct xe_device_desc *des if (err) return err; - err = xe_probe_info(xe, desc, &probed_info); + err = xe_probe_info(xe, &probed_info); if (err) return err; - err = xe_info_init(xe, desc, &probed_info); + err = xe_info_init(xe, &probed_info); if (err) return err; diff --git a/drivers/gpu/drm/xe/xe_pcode.c b/drivers/gpu/drm/xe/xe_pcode.c index d502205bb72a..266deecbb100 100644 --- a/drivers/gpu/drm/xe/xe_pcode.c +++ b/drivers/gpu/drm/xe/xe_pcode.c @@ -140,7 +140,7 @@ int xe_pcode_read(struct xe_tile *tile, u32 mbox, u32 *val0, u32 *val1) int err; mutex_lock(&tile->pcode.lock); - err = pcode_mailbox_rw(tile, mbox, val0, val1, 1, true, false); + err = pcode_mailbox_rw(tile, mbox, val0, val1, PCODE_DEFAULT_TIMEOUT_MS, true, false); mutex_unlock(&tile->pcode.lock); return err; diff --git a/drivers/gpu/drm/xe/xe_pcode.h b/drivers/gpu/drm/xe/xe_pcode.h index ba8a1d1b2152..7e43792b0037 100644 --- a/drivers/gpu/drm/xe/xe_pcode.h +++ b/drivers/gpu/drm/xe/xe_pcode.h @@ -18,6 +18,8 @@ struct xe_pcode_version { u32 engg; }; +#define PCODE_DEFAULT_TIMEOUT_MS 10 + int xe_pcode_init_early(struct xe_tile *tile); int xe_pcode_probe_early(struct xe_device *xe); int xe_pcode_ready(struct xe_device *xe, bool locked); @@ -31,7 +33,7 @@ int xe_pcode_write64_timeout(struct xe_tile *tile, u32 mbox, u32 data0, int xe_get_pcode_version(struct xe_device *xe, struct xe_pcode_version *version); #define xe_pcode_write(tile, mbox, val) \ - xe_pcode_write_timeout(tile, mbox, val, 1) + xe_pcode_write_timeout(tile, mbox, val, PCODE_DEFAULT_TIMEOUT_MS) int xe_pcode_request(struct xe_tile *tile, u32 mbox, u32 request, u32 reply_mask, u32 reply, int timeout_ms); diff --git a/drivers/gpu/drm/xe/xe_pt.c b/drivers/gpu/drm/xe/xe_pt.c index 5d990c1c3740..4b351dbf6572 100644 --- a/drivers/gpu/drm/xe/xe_pt.c +++ b/drivers/gpu/drm/xe/xe_pt.c @@ -236,9 +236,11 @@ void xe_pt_destroy(struct xe_pt *pt, u32 flags, struct llist_head *deferred) */ void xe_pt_clear(struct xe_device *xe, struct xe_pt *pt) { - struct iosys_map *map = &pt->bo->vmap; + struct xe_bo *bo = pt->bo; - xe_map_memset(xe, map, 0, 0, SZ_4K); + xe_bo_assert_held(bo); + if (!iosys_map_is_null(&bo->vmap)) + xe_map_memset(xe, &bo->vmap, 0, 0, SZ_4K); } /** @@ -831,9 +833,12 @@ xe_pt_stage_bind(struct xe_tile *tile, struct xe_vma *vma, return -EAGAIN; } if (xe_svm_range_has_dma_mapping(range)) { - xe_res_first_dma(range->pages.dma_addr, 0, - xe_svm_range_size(range), - &curs); + const struct drm_pagemap_addr *addr; + bool contiguous; + + addr = xe_svm_range_first_dma(range, &contiguous); + xe_res_first_dma(addr, 0, xe_svm_range_size(range), + contiguous, &curs); xe_svm_range_debug(range, "BIND PREPARE - MIXED"); } else { xe_assert(xe, false); @@ -865,10 +870,15 @@ xe_pt_stage_bind(struct xe_tile *tile, struct xe_vma *vma, xe_bo_assert_held(bo); if (!xe_vma_is_null(vma) && !range && !is_purged) { - if (xe_vma_is_userptr(vma)) - xe_res_first_dma(to_userptr_vma(vma)->userptr.pages.dma_addr, 0, - xe_vma_size(vma), &curs); - else if (xe_bo_is_vram(bo) || xe_bo_is_stolen(bo)) + if (xe_vma_is_userptr(vma)) { + const struct drm_pagemap_addr *addr; + bool contiguous; + + addr = drm_gpusvm_pages_first_dma(&to_userptr_vma(vma)->userptr.pages, + &contiguous); + xe_res_first_dma(addr, 0, xe_vma_size(vma), contiguous, + &curs); + } else if (xe_bo_is_vram(bo) || xe_bo_is_stolen(bo)) xe_res_first(bo->ttm.resource, xe_vma_bo_offset(vma), xe_vma_size(vma), &curs); else diff --git a/drivers/gpu/drm/xe/xe_ras.c b/drivers/gpu/drm/xe/xe_ras.c index de4cb9ef7355..7a85735c57d5 100644 --- a/drivers/gpu/drm/xe/xe_ras.c +++ b/drivers/gpu/drm/xe/xe_ras.c @@ -3,6 +3,7 @@ * Copyright © 2026 Intel Corporation */ +#include "xe_configfs.h" #include "xe_debugfs.h" #include "xe_device.h" #include "xe_drm_ras.h" @@ -915,6 +916,14 @@ void xe_ras_init(struct xe_device *xe) { int ret; + /* + * TODO: Replace platform check with xe->info.has_disable_vram_page_offline + * once the feature flag is plumbed through device info. + */ + if (xe->info.platform == XE_CRESCENTISLAND) + xe->ras.disable_vram_page_offline = + xe_configfs_get_disable_vram_page_offline(to_pci_dev(xe->drm.dev)); + xe_drm_ras_init(xe); if (!xe->info.has_sysctrl) diff --git a/drivers/gpu/drm/xe/xe_res_cursor.h b/drivers/gpu/drm/xe/xe_res_cursor.h index 0522caafd89d..c3a037e5f34e 100644 --- a/drivers/gpu/drm/xe/xe_res_cursor.h +++ b/drivers/gpu/drm/xe/xe_res_cursor.h @@ -233,12 +233,13 @@ static inline void xe_res_first_sg(const struct sg_table *sg, * @dma_addr: struct drm_pagemap_addr array to walk * @start: Start of the range * @size: Size of the range + * @contiguous: Whether one entry describes the whole range * @cur: cursor object to initialize * * Start walking over the range of allocations between @start and @size. */ static inline void xe_res_first_dma(const struct drm_pagemap_addr *dma_addr, - u64 start, u64 size, + u64 start, u64 size, bool contiguous, struct xe_res_cursor *cur) { XE_WARN_ON(!dma_addr); @@ -248,7 +249,7 @@ static inline void xe_res_first_dma(const struct drm_pagemap_addr *dma_addr, cur->node = NULL; cur->start = start; cur->remaining = size; - cur->dma_seg_size = PAGE_SIZE << dma_addr->order; + cur->dma_seg_size = contiguous ? start + size : PAGE_SIZE << dma_addr->order; cur->dma_start = 0; cur->size = 0; cur->dma_addr = dma_addr; diff --git a/drivers/gpu/drm/xe/xe_shrinker.c b/drivers/gpu/drm/xe/xe_shrinker.c index 83374cd57660..deb4378c1ec1 100644 --- a/drivers/gpu/drm/xe/xe_shrinker.c +++ b/drivers/gpu/drm/xe/xe_shrinker.c @@ -54,13 +54,40 @@ xe_shrinker_mod_pages(struct xe_shrinker *shrinker, long shrinkable, long purgea write_unlock(&shrinker->lock); } -static s64 __xe_shrinker_walk(struct xe_device *xe, +static bool __xe_shrinker_runtime_pm_get(struct xe_shrinker *shrinker) +{ + struct xe_device *xe = shrinker->xe; + + if (xe_pm_runtime_get_if_active(xe)) + return true; + + if (xe_rpm_reclaim_safe(xe) && !ttm_bo_shrink_avoid_wait()) { + xe_pm_runtime_get(xe); + return true; + } + + queue_work(xe->unordered_wq, &shrinker->pm_worker); + + return false; +} + +static void xe_shrinker_runtime_pm_put(struct xe_shrinker *shrinker, bool runtime_pm) +{ + if (runtime_pm) + xe_pm_runtime_put(shrinker->xe); +} + +static int __xe_shrinker_walk(struct xe_shrinker *shrinker, struct ttm_operation_ctx *ctx, const struct xe_bo_shrink_flags flags, - unsigned long to_scan, unsigned long *scanned) + unsigned long to_scan, unsigned long *scanned, + unsigned long *freed) { + struct xe_device *xe = shrinker->xe; unsigned int mem_type; - s64 freed = 0, lret; + bool rpm = false; + int ret = 0; + s64 lret; for (mem_type = XE_PL_SYSTEM; mem_type <= XE_PL_TT; ++mem_type) { struct ttm_resource_manager *man = ttm_manager_type(&xe->ttm, mem_type); @@ -74,23 +101,35 @@ static s64 __xe_shrinker_walk(struct xe_device *xe, if (!man || !man->use_tt) continue; + if (mem_type != XE_PL_SYSTEM && !rpm && + xe_device_is_l2_flush_optimized(xe)) { + if (!__xe_shrinker_runtime_pm_get(shrinker)) + break; + rpm = true; + } + ttm_bo_lru_for_each_reserved_guarded(&curs, man, &arg, ttm_bo) { if (!ttm_bo_shrink_suitable(ttm_bo, ctx)) continue; lret = xe_bo_shrink(ctx, ttm_bo, flags, scanned); - if (lret < 0) - return lret; + if (lret < 0) { + ret = lret; + goto out; + } - freed += lret; + *freed += lret; if (*scanned >= to_scan) - break; + goto out; } /* Trylocks should never error, just fail. */ xe_assert(xe, !IS_ERR(ttm_bo)); } - return freed; +out: + xe_shrinker_runtime_pm_put(shrinker, rpm); + + return ret; } /* @@ -99,40 +138,36 @@ static s64 __xe_shrinker_walk(struct xe_device *xe, * add writeback. This avoids stalls and explicit writebacks with light or * moderate memory pressure. */ -static s64 xe_shrinker_walk(struct xe_device *xe, +static int xe_shrinker_walk(struct xe_shrinker *shrinker, struct ttm_operation_ctx *ctx, const struct xe_bo_shrink_flags flags, - unsigned long to_scan, unsigned long *scanned) + unsigned long to_scan, unsigned long *scanned, + unsigned long *freed) { bool no_wait_gpu = true; struct xe_bo_shrink_flags save_flags = flags; - s64 lret, freed; + int ret; swap(no_wait_gpu, ctx->no_wait_gpu); save_flags.writeback = false; - lret = __xe_shrinker_walk(xe, ctx, save_flags, to_scan, scanned); + ret = __xe_shrinker_walk(shrinker, ctx, save_flags, to_scan, scanned, + freed); swap(no_wait_gpu, ctx->no_wait_gpu); - if (lret < 0 || *scanned >= to_scan) - return lret; + if (ret || *scanned >= to_scan) + return ret; - freed = lret; if (!ctx->no_wait_gpu) { - lret = __xe_shrinker_walk(xe, ctx, save_flags, to_scan, scanned); - if (lret < 0) - return lret; - freed += lret; - if (*scanned >= to_scan) - return freed; + ret = __xe_shrinker_walk(shrinker, ctx, save_flags, to_scan, scanned, + freed); + if (ret || *scanned >= to_scan) + return ret; } - if (flags.writeback) { - lret = __xe_shrinker_walk(xe, ctx, flags, to_scan, scanned); - if (lret < 0) - return lret; - freed += lret; - } + if (flags.writeback) + ret = __xe_shrinker_walk(shrinker, ctx, flags, to_scan, scanned, + freed); - return freed; + return ret; } static unsigned long @@ -180,22 +215,7 @@ static bool xe_shrinker_runtime_pm_get(struct xe_shrinker *shrinker, bool force, return false; } - if (!xe_pm_runtime_get_if_active(xe)) { - if (xe_rpm_reclaim_safe(xe) && !ttm_bo_shrink_avoid_wait()) { - xe_pm_runtime_get(xe); - return true; - } - queue_work(xe->unordered_wq, &shrinker->pm_worker); - return false; - } - - return true; -} - -static void xe_shrinker_runtime_pm_put(struct xe_shrinker *shrinker, bool runtime_pm) -{ - if (runtime_pm) - xe_pm_runtime_put(shrinker->xe); + return __xe_shrinker_runtime_pm_get(shrinker); } static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_control *sc) @@ -214,7 +234,6 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con bool runtime_pm; bool purgeable; bool can_backup = !!(sc->gfp_mask & __GFP_FS); - s64 lret; nr_to_scan = sc->nr_to_scan; @@ -225,12 +244,9 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con /* Might need runtime PM. Try to wake early if it looks like it. */ runtime_pm = xe_shrinker_runtime_pm_get(shrinker, false, nr_to_scan, can_backup); - if (purgeable && nr_scanned < nr_to_scan) { - lret = xe_shrinker_walk(shrinker->xe, &ctx, shrink_flags, - nr_to_scan, &nr_scanned); - if (lret >= 0) - freed += lret; - } + if (purgeable && nr_scanned < nr_to_scan) + xe_shrinker_walk(shrinker, &ctx, shrink_flags, + nr_to_scan, &nr_scanned, &freed); sc->nr_scanned = nr_scanned; if (nr_scanned >= nr_to_scan || !can_backup) @@ -242,10 +258,8 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con shrink_flags.purge = false; - lret = xe_shrinker_walk(shrinker->xe, &ctx, shrink_flags, - nr_to_scan, &nr_scanned); - if (lret >= 0) - freed += lret; + xe_shrinker_walk(shrinker, &ctx, shrink_flags, + nr_to_scan, &nr_scanned, &freed); sc->nr_scanned = nr_scanned; out: diff --git a/drivers/gpu/drm/xe/xe_svm.c b/drivers/gpu/drm/xe/xe_svm.c index 627a741293d5..6c3033fc4db7 100644 --- a/drivers/gpu/drm/xe/xe_svm.c +++ b/drivers/gpu/drm/xe/xe_svm.c @@ -13,6 +13,7 @@ #include "xe_bo.h" #include "xe_exec_queue_types.h" #include "xe_gt_stats.h" +#include "xe_log.h" #include "xe_migrate.h" #include "xe_module.h" #include "xe_pagefault.h" @@ -1361,9 +1362,9 @@ retry: else goto retry; } else { - drm_err(&vm->xe->drm, - "VRAM allocation failed, retry count exceeded, asid=%u, errno=%pe\n", - vm->usm.asid, ERR_PTR(err)); + xe_log_err(gt, PAGEFAULT, err, + "VRAM allocation failed, retry count exceeded, ASID=%u\n", + vm->usm.asid); goto err_out; } } @@ -1384,9 +1385,9 @@ get_pages: range_debug(range, "PAGE FAULT - RETRY PAGES"); goto retry; } else { - drm_err(&vm->xe->drm, - "Get pages failed, retry count exceeded, asid=%u, gpusvm=%p, errno=%pe\n", - vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err)); + xe_log_err(gt, PAGEFAULT, err, + "Get pages failed, retry count exceeded, ASID=%u, GPUVM=%s\n", + vm->usm.asid, vm->svm.gpusvm.name); } } if (err) { @@ -1598,7 +1599,7 @@ int xe_svm_range_get_pages(struct xe_vm *vm, struct xe_svm_range *range, lockdep_assert_held(&range->lock); - err = drm_gpusvm_get_pages(&vm->svm.gpusvm, &range->pages, + err = drm_gpusvm_get_pages(&vm->svm.gpusvm, &range->pages, 1, vm->svm.gpusvm.mm, &range->base.notifier->notifier, drm_gpusvm_range_start(&range->base), diff --git a/drivers/gpu/drm/xe/xe_svm.h b/drivers/gpu/drm/xe/xe_svm.h index 2a0dc0d125c9..2ef4ef026ccd 100644 --- a/drivers/gpu/drm/xe/xe_svm.h +++ b/drivers/gpu/drm/xe/xe_svm.h @@ -220,6 +220,19 @@ static inline unsigned long xe_svm_range_size(struct xe_svm_range *range) return drm_gpusvm_range_size(&range->base); } +/** + * xe_svm_range_first_dma() - Resolve the device address array of a SVM range + * @range: SVM range + * @contiguous: Where to store whether one entry spans the whole range + * + * Return: Pointer to the first device address, NULL if none is populated. + */ +static inline const struct drm_pagemap_addr * +xe_svm_range_first_dma(struct xe_svm_range *range, bool *contiguous) +{ + return drm_gpusvm_pages_first_dma(&range->pages, contiguous); +} + void xe_svm_flush(struct xe_vm *vm); int xe_pagemap_shrinker_create(struct xe_device *xe); @@ -436,6 +449,13 @@ static inline bool xe_svm_range_is_removed(struct xe_svm_range *range) return false; } +static inline const struct drm_pagemap_addr * +xe_svm_range_first_dma(struct xe_svm_range *range, bool *contiguous) +{ + *contiguous = false; + return NULL; +} + #define xe_svm_range_has_dma_mapping(...) false #endif /* CONFIG_DRM_XE_GPUSVM */ diff --git a/drivers/gpu/drm/xe/xe_sysctrl.c b/drivers/gpu/drm/xe/xe_sysctrl.c index 62ccc9be71b4..4067e1dfdcd5 100644 --- a/drivers/gpu/drm/xe/xe_sysctrl.c +++ b/drivers/gpu/drm/xe/xe_sysctrl.c @@ -13,9 +13,11 @@ #include "xe_device.h" #include "xe_mmio.h" #include "xe_pm.h" +#include "xe_printk.h" #include "xe_soc_remapper.h" #include "xe_sysctrl.h" #include "xe_sysctrl_mailbox.h" +#include "xe_sysctrl_mailbox_types.h" #include "xe_sysctrl_types.h" /** @@ -29,6 +31,21 @@ * This module provides initialization and support code for interacting * with System Controller through the mailbox interface. */ + +/* Application status flags reported in xe_sysctrl_app_status_resp.flags */ +#define XE_SYSCTRL_APP_RESP_VALID BIT(0) +#define XE_SYSCTRL_APP_RESP_BOOTED BIT(1) +#define XE_SYSCTRL_APP_RESP_INITIALIZED BIT(2) + +/* + * Known System Controller application identifiers, keyed by firmware + * application ID. + */ +enum xe_sysctrl_app_id { + XE_SYSCTRL_APP_OCODE = 0x0C, + XE_SYSCTRL_APP_DIAG = 0x0D, +}; + static void sysctrl_fini(void *arg) { struct xe_device *xe = arg; @@ -125,3 +142,78 @@ void xe_sysctrl_pm_resume(struct xe_device *xe) xe->soc_remapper.set_sysctrl_region(xe, SYSCTRL_MAILBOX_INDEX); } + +static enum xe_sysctrl_fw_status +xe_sysctrl_check_app_status(struct xe_device *xe, enum xe_sysctrl_app_id app_id) +{ + struct xe_sysctrl_app_status_req req = {}; + struct xe_sysctrl_app_status_resp resp = {}; + struct xe_sysctrl_mailbox_command cmd = {}; + size_t out_len = 0; + u32 flags; + int ret; + + req.app_id = (u8)app_id; + + xe_sysctrl_create_command(&cmd, XE_SYSCTRL_GROUP_CORE, XE_SYSCTRL_CMD_GET_APP_STATUS_BY_ID, + &req, sizeof(req), &resp, sizeof(resp)); + + ret = xe_sysctrl_send_command(&xe->sc, &cmd, &out_len); + if (ret) + return XE_SYSCTRL_FIRMWARE_COMM_FAILURE; + + if (out_len != sizeof(resp)) { + xe_err(xe, "sysctrl: unexpected get app status response length %zu (expected %zu)\n", + out_len, sizeof(resp)); + return XE_SYSCTRL_FIRMWARE_COMM_FAILURE; + } + + flags = resp.flags; + + if (!(flags & XE_SYSCTRL_APP_RESP_VALID)) + return XE_SYSCTRL_FIRMWARE_APP_INVALID; + + if (!(flags & XE_SYSCTRL_APP_RESP_BOOTED)) + return XE_SYSCTRL_FIRMWARE_APP_NOT_LOADED; + + if (!(flags & XE_SYSCTRL_APP_RESP_INITIALIZED)) + return XE_SYSCTRL_FIRMWARE_APP_BOOTED; + + return XE_SYSCTRL_FIRMWARE_APP_INITIALIZED; +} + +/** + * xe_sysctrl_is_oobmsm_fw_ready() - Check if oCode firmware is fully initialized + * @xe: xe device instance + * + * Returns true if oCode firmware has reached the initialized state, indicating + * it is ready to handle requests. + * + * Callers must only invoke this on platforms where System Controller is + * present (xe->info.has_sysctrl). + * + * Return: true if oCode firmware is initialized, false otherwise + */ +bool xe_sysctrl_is_oobmsm_fw_ready(struct xe_device *xe) +{ + return xe_sysctrl_check_app_status(xe, XE_SYSCTRL_APP_OCODE) == + XE_SYSCTRL_FIRMWARE_APP_INITIALIZED; +} + +/** + * xe_sysctrl_is_diag_fw_ready() - Check if diag firmware is fully initialized + * @xe: xe device instance + * + * Returns true if diag firmware has reached the initialized state, indicating + * it is ready to handle requests. + * + * Callers must only invoke this on platforms where System Controller is + * present (xe->info.has_sysctrl). + * + * Return: true if diag firmware is initialized, false otherwise + */ +bool xe_sysctrl_is_diag_fw_ready(struct xe_device *xe) +{ + return xe_sysctrl_check_app_status(xe, XE_SYSCTRL_APP_DIAG) == + XE_SYSCTRL_FIRMWARE_APP_INITIALIZED; +} diff --git a/drivers/gpu/drm/xe/xe_sysctrl.h b/drivers/gpu/drm/xe/xe_sysctrl.h index 090dffb6d55f..b69a3f474236 100644 --- a/drivers/gpu/drm/xe/xe_sysctrl.h +++ b/drivers/gpu/drm/xe/xe_sysctrl.h @@ -20,5 +20,7 @@ void xe_sysctrl_event(struct xe_sysctrl *sc); int xe_sysctrl_init(struct xe_device *xe); void xe_sysctrl_irq_handler(struct xe_device *xe, u32 master_ctl); void xe_sysctrl_pm_resume(struct xe_device *xe); +bool xe_sysctrl_is_oobmsm_fw_ready(struct xe_device *xe); +bool xe_sysctrl_is_diag_fw_ready(struct xe_device *xe); #endif diff --git a/drivers/gpu/drm/xe/xe_sysctrl_mailbox_types.h b/drivers/gpu/drm/xe/xe_sysctrl_mailbox_types.h index 66e7cbcc3f91..c236e5377f30 100644 --- a/drivers/gpu/drm/xe/xe_sysctrl_mailbox_types.h +++ b/drivers/gpu/drm/xe/xe_sysctrl_mailbox_types.h @@ -14,9 +14,11 @@ * enum xe_sysctrl_group - System Controller command groups * * @XE_SYSCTRL_GROUP_GFSP: GFSP group + * @XE_SYSCTRL_GROUP_CORE: Core group */ enum xe_sysctrl_group { XE_SYSCTRL_GROUP_GFSP = 0x01, + XE_SYSCTRL_GROUP_CORE = 0xFF, }; /** @@ -43,6 +45,49 @@ enum xe_sysctrl_gfsp_cmd { }; /** + * enum xe_sysctrl_core_cmd - Commands supported by Core group + * + * @XE_SYSCTRL_CMD_GET_APP_STATUS_BY_ID: Retrieve application status by ID + */ +enum xe_sysctrl_core_cmd { + XE_SYSCTRL_CMD_GET_APP_STATUS_BY_ID = 0x05, +}; + +/** + * struct xe_sysctrl_app_status_req - Get application status request + * + * @app_id: Application ID for which to retrieve status + */ +struct xe_sysctrl_app_status_req { + u8 app_id; +} __packed; + +/** + * struct xe_sysctrl_app_status_resp - Get application status response + * @flags: Application status flags interpreted by xe_sysctrl_check_app_status() + */ +struct xe_sysctrl_app_status_resp { + u32 flags; +} __packed; + +/** + * enum xe_sysctrl_fw_status - System Controller firmware application lifecycle states + * + * @XE_SYSCTRL_FIRMWARE_APP_INVALID: app_id is not recognized by firmware + * @XE_SYSCTRL_FIRMWARE_APP_NOT_LOADED: application is known but has not yet booted + * @XE_SYSCTRL_FIRMWARE_APP_BOOTED: boot sequence completed, post-boot init pending + * @XE_SYSCTRL_FIRMWARE_APP_INITIALIZED: application fully operational + * @XE_SYSCTRL_FIRMWARE_COMM_FAILURE: communication with System Controller firmware failed + */ +enum xe_sysctrl_fw_status { + XE_SYSCTRL_FIRMWARE_APP_INVALID, + XE_SYSCTRL_FIRMWARE_APP_NOT_LOADED, + XE_SYSCTRL_FIRMWARE_APP_BOOTED, + XE_SYSCTRL_FIRMWARE_APP_INITIALIZED, + XE_SYSCTRL_FIRMWARE_COMM_FAILURE, +}; + +/** * struct xe_sysctrl_mailbox_command - System Controller mailbox command */ struct xe_sysctrl_mailbox_command { diff --git a/drivers/gpu/drm/xe/xe_tile_types.h b/drivers/gpu/drm/xe/xe_tile_types.h index 0048100ccb72..e1368c04846a 100644 --- a/drivers/gpu/drm/xe/xe_tile_types.h +++ b/drivers/gpu/drm/xe/xe_tile_types.h @@ -97,6 +97,10 @@ struct xe_tile { * Only main GT has page reclaim list allocations. */ struct xe_sa_manager *reclaim_pool; +#if IS_ENABLED(CONFIG_DRM_XE_DEBUG_MEM) + /** @mem.memtest_bo: VRAM overlap check BO */ + struct xe_bo *memtest_bo; +#endif } mem; /** @sriov: tile level virtualization data */ diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c index 05911904c1f9..9a514d983e90 100644 --- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c +++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c @@ -5,18 +5,27 @@ */ #include <linux/cgroup_dmem.h> +#include <linux/debugfs.h> #include <drm/drm_managed.h> #include <drm/drm_drv.h> #include <drm/drm_buddy.h> +#include <uapi/drm/xe_drm.h> #include <drm/ttm/ttm_placement.h> #include <drm/ttm/ttm_range_manager.h> +#include "regs/xe_regs.h" #include "xe_bo.h" +#include "xe_configfs.h" #include "xe_device.h" +#include "xe_exec_queue.h" +#include "xe_lrc.h" +#include "xe_mmio.h" #include "xe_pm.h" +#include "xe_printk.h" #include "xe_res_cursor.h" +#include "xe_ttm_stolen_mgr.h" #include "xe_ttm_vram_mgr.h" #include "xe_vram_types.h" @@ -49,6 +58,48 @@ static inline bool xe_is_vram_mgr_blocks_contiguous(struct gpu_buddy *mm, return true; } +static int xe_ttm_vram_buddy_alloc(struct xe_ttm_vram_mgr *mgr, u64 start, + u64 end, u64 size, u64 min_page_size, + struct list_head *blocks, unsigned long flags, + struct ttm_resource *res, u64 *used_visible) +{ + struct gpu_buddy *mm = &mgr->mm; + struct gpu_buddy_block *block; + int err; + + err = gpu_buddy_alloc_blocks(mm, start, end, size, min_page_size, blocks, flags); + if (err) + return err; + + /* + * Track the owning resource, never the owning BO. A BO backpointer + * cached here goes stale the moment TTM hands the resource to a ghost + * object (ttm_buffer_object_transfer()), which happens on every + * accelerated move and on pipelined gutting. The resource, in + * contrast, has exactly the same lifetime as these blocks and TTM + * keeps &ttm_resource.bo pointing at the current owner for us. + */ + list_for_each_entry(block, blocks, link) + block->private = res; + + if (end <= mgr->visible_size) { + *used_visible = size; + } else { + list_for_each_entry(block, blocks, link) { + u64 blk_start = gpu_buddy_block_offset(block); + + if (blk_start < mgr->visible_size) { + u64 blk_end = blk_start + gpu_buddy_block_size(mm, block); + + *used_visible += min(blk_end, mgr->visible_size) - blk_start; + } + } + } + + mgr->visible_avail -= *used_visible; + return 0; +} + static int xe_ttm_vram_mgr_new(struct ttm_resource_manager *man, struct ttm_buffer_object *tbo, const struct ttm_place *place, @@ -117,30 +168,12 @@ static int xe_ttm_vram_mgr_new(struct ttm_resource_manager *man, goto error_unlock; } - err = gpu_buddy_alloc_blocks(mm, (u64)place->fpfn << PAGE_SHIFT, - (u64)lpfn << PAGE_SHIFT, size, - min_page_size, &vres->blocks, vres->flags); + err = xe_ttm_vram_buddy_alloc(mgr, (u64)place->fpfn << PAGE_SHIFT, + (u64)lpfn << PAGE_SHIFT, size, + min_page_size, &vres->blocks, vres->flags, + &vres->base, &vres->used_visible_size); if (err) goto error_unlock; - - if (lpfn <= mgr->visible_size >> PAGE_SHIFT) { - vres->used_visible_size = size; - } else { - struct gpu_buddy_block *block; - - list_for_each_entry(block, &vres->blocks, link) { - u64 start = gpu_buddy_block_offset(block); - - if (start < mgr->visible_size) { - u64 end = start + gpu_buddy_block_size(mm, block); - - vres->used_visible_size += - min(end, mgr->visible_size) - start; - } - } - } - - mgr->visible_avail -= vres->used_visible_size; mutex_unlock(&mgr->lock); if (!(vres->base.placement & TTM_PL_FLAG_CONTIGUOUS) && @@ -172,17 +205,61 @@ error_fini: return err; } +static void xe_ttm_vram_buddy_free(struct xe_ttm_vram_mgr *mgr, + struct list_head *blocks, + u64 used_visible) +{ + struct gpu_buddy_block *block; + + list_for_each_entry(block, blocks, link) + block->private = NULL; + gpu_buddy_free_list(&mgr->mm, blocks, 0); + mgr->visible_avail += used_visible; +} + +/* + * Retry pending page-offline reservations. + * + * A reservation can fail because the blocks backing the bad page are still + * allocated: either the owning BO could not be purged, or the purge was + * pipelined and TTM handed the resource to a ghost object which frees it + * only once the move fences signal. Rather than giving up, entries stay on + * @queued_pages and are retried here every time VRAM blocks come back. + * + * Called with @mgr->lock held. + */ +static void xe_ttm_vram_retry_queued_pages(struct xe_ttm_vram_mgr *mgr) +{ + struct xe_ttm_vram_offline_resource *pos, *n; + + lockdep_assert_held(&mgr->lock); + + list_for_each_entry_safe(pos, n, &mgr->queued_pages, queued_link) { + if (xe_ttm_vram_buddy_alloc(mgr, pos->addr, pos->addr + PAGE_SIZE, + PAGE_SIZE, PAGE_SIZE, &pos->blocks, + GPU_BUDDY_RANGE_ALLOCATION, NULL, + &pos->used_visible_size)) { + pos->status = XE_PAGE_RESERVE_FAIL; + continue; + } + --mgr->n_queued_pages; + list_del_rcu(&pos->queued_link); + ++mgr->n_offlined_pages; + list_add_rcu(&pos->offlined_link, &mgr->offlined_pages); + } +} + static void xe_ttm_vram_mgr_del(struct ttm_resource_manager *man, struct ttm_resource *res) { struct xe_ttm_vram_mgr_resource *vres = to_xe_ttm_vram_mgr_resource(res); struct xe_ttm_vram_mgr *mgr = to_xe_ttm_vram_mgr(man); - struct gpu_buddy *mm = &mgr->mm; mutex_lock(&mgr->lock); - gpu_buddy_free_list(mm, &vres->blocks, 0); - mgr->visible_avail += vres->used_visible_size; + xe_ttm_vram_buddy_free(mgr, &vres->blocks, vres->used_visible_size); + if (unlikely(!list_empty(&mgr->queued_pages))) + xe_ttm_vram_retry_queued_pages(mgr); mutex_unlock(&mgr->lock); ttm_resource_fini(man, res); @@ -312,12 +389,35 @@ static void xe_ttm_vram_mgr_set_unused(struct drm_device *dev, void *arg) ttm_resource_manager_set_used(man, false); } +static void xe_ttm_vram_free_bad_pages(struct xe_ttm_vram_mgr *mgr) +{ + struct xe_ttm_vram_offline_resource *pos, *n; + + list_for_each_entry_safe(pos, n, &mgr->offlined_pages, offlined_link) { + list_del_rcu(&pos->offlined_link); + xe_ttm_vram_buddy_free(mgr, &pos->blocks, pos->used_visible_size); + --mgr->n_offlined_pages; + kfree_rcu(pos, rcu); + } + list_for_each_entry_safe(pos, n, &mgr->queued_pages, queued_link) { + list_del_rcu(&pos->queued_link); + /* queued entries have no buddy reservation yet */ + xe_ttm_vram_buddy_free(mgr, &pos->blocks, 0); + --mgr->n_queued_pages; + kfree_rcu(pos, rcu); + } +} + static void xe_ttm_vram_mgr_fini(struct drm_device *dev, void *arg) { struct xe_device *xe = to_xe_device(dev); struct xe_ttm_vram_mgr *mgr = arg; struct ttm_resource_manager *man = &mgr->manager; + mutex_lock(&mgr->lock); + xe_ttm_vram_free_bad_pages(mgr); + mutex_unlock(&mgr->lock); + if (ttm_resource_manager_evict_all(&xe->ttm, man)) return; @@ -344,6 +444,8 @@ int __xe_ttm_vram_mgr_init(struct xe_device *xe, struct xe_ttm_vram_mgr *mgr, err = drmm_mutex_init(&xe->drm, &mgr->lock); if (err) return err; + INIT_LIST_HEAD(&mgr->offlined_pages); + INIT_LIST_HEAD(&mgr->queued_pages); mgr->default_page_size = default_page_size; mgr->visible_size = io_size; mgr->visible_avail = io_size; @@ -521,3 +623,451 @@ u64 xe_ttm_vram_get_avail(struct ttm_resource_manager *man) return avail; } + +static int xe_ttm_vram_purge_page(struct xe_device *xe, struct xe_bo *bo) +{ + u32 q_flag = DRM_XE_EXEC_QUEUE_BAN_REASON_PAGE_OFFLINE; + struct ttm_operation_ctx ctx = {}; + struct xe_exec_queue *q_to_put = NULL; + struct xe_exec_queue *q = NULL; + struct xe_vm *vm = NULL; + u32 flags; + int ret = 0; + + xe_bo_lock(bo, false); + if (bo->vm) + vm = xe_vm_get(bo->vm); + flags = bo->flags; + xe_bo_unlock(bo); + /* Ban VM if BO is PPGTT */ + if (vm && (flags & XE_BO_FLAG_PAGETABLE)) { + struct xe_exec_queue *eq; + int id; + + down_write(&vm->lock); + if (xe->info.has_ctx_tlb_inval) { + /* + * Must be the write lock: send_tlb_inval_ctx_ppgtt() + * mutates this list (list_move_tail() onto an on-stack + * head) while holding only the read lock, relying on + * tlb_inval->seqno_lock to keep itself the sole + * mutator. Traversing it under down_read() would let + * this walk follow entries onto that stack list. + */ + down_write(&vm->exec_queues.lock); + for (id = 0; id < ARRAY_SIZE(vm->exec_queues.list); id++) + list_for_each_entry(eq, &vm->exec_queues.list[id], + vm_exec_queue_link) + atomic_or(q_flag, &eq->ban_reason); + up_write(&vm->exec_queues.lock); + } else { + list_for_each_entry(eq, &vm->preempt.exec_queues, lr.link) + atomic_or(q_flag, &eq->ban_reason); + } + smp_wmb(); /* Force all queue bits to be visible before killing the VM */ + xe_vm_kill(vm, true); + up_write(&vm->lock); + } + if (vm) + xe_vm_put(vm); + + xe_bo_lock(bo, false); + q = READ_ONCE(bo->q); + /* Ban exec queue if BO is lrc */ + if (q && xe_exec_queue_get_unless_zero(q)) { + /* ban queue */ + atomic_or(q_flag, &q->ban_reason); + smp_wmb(); /* Force bit change to finish before state change triggers */ + q_to_put = q; + } + + if (bo->purgeable.state == XE_MADV_PURGEABLE_PURGED) { + /* Already purged by shrinker during unlocked window — nothing to do */ + xe_bo_unlock(bo); + goto out; + } + + xe_bo_set_purgeable_state(bo, XE_MADV_PURGEABLE_DONTNEED); + ttm_bo_unmap_virtual(&bo->ttm); /* nuke CPU mmap + VRAM IO mappings */ + if (xe_bo_is_pinned(bo)) + xe_bo_unpin(bo); + ret = xe_ttm_bo_purge(&bo->ttm, &ctx); + xe_bo_unlock(bo); + +out: + if (q_to_put) { + xe_exec_queue_kill(q_to_put); + xe_exec_queue_put(q_to_put); + } + + return ret; +} + +static bool xe_ttm_vram_page_already_processed(struct xe_ttm_vram_mgr *mgr, + u64 addr) +{ + struct xe_ttm_vram_offline_resource *pos; + + lockdep_assert_held(&mgr->lock); + + list_for_each_entry(pos, &mgr->offlined_pages, offlined_link) { + if (pos->addr == addr) + return true; + } + + list_for_each_entry(pos, &mgr->queued_pages, queued_link) { + if (pos->addr == addr) + return true; + } + + return false; +} + +/* + * Resolve the BO currently owning @block and take a reference on it. + * + * Called with @mgr->lock held, which serializes against + * xe_ttm_vram_buddy_free() clearing block->private. + * + * Returns NULL when there is no xe_bo we can act on: either the block is + * free, or the resource is temporarily owned by a TTM ghost object because + * a move or a pipelined gutting is still in flight. In both cases the + * blocks will hit xe_ttm_vram_mgr_del() on their own and the pending + * reservation is retried from there. + */ +static struct xe_bo *xe_ttm_vram_block_owner_get(struct xe_device *xe, + struct gpu_buddy_block *block) +{ + struct ttm_resource *res = block->private; + struct ttm_buffer_object *tbo; + struct xe_bo *bo; + + if (!res) + return NULL; + + guard(spinlock)(&xe->ttm.lru_lock); + + /* + * res->bo is updated under bdev->lru_lock by ttm_resource_set_bo(). + * Racing with a ghost transfer here is benign: we either see the old + * owner (whose purge is a no-op and the retry path recovers) or the + * ghost (rejected below). + * + * A ghost is a bare ttm_transfer_obj, not an xe_bo, so ttm_to_xe_bo() + * on one would be out of bounds. xe_bo_is_xe_bo() rejects it since + * only our own BOs carry xe_ttm_bo_destroy(). + */ + tbo = READ_ONCE(res->bo); + if (!tbo || !xe_bo_is_xe_bo(tbo)) + return NULL; + + bo = ttm_to_xe_bo(tbo); + + /* The BO may already be in teardown with a zero refcount */ + return xe_bo_get_unless_zero(bo) ? bo : NULL; +} + +static int xe_ttm_vram_reserve_page_at_addr(struct xe_device *xe, u64 addr, + struct xe_ttm_vram_mgr *vram_mgr, struct gpu_buddy *mm) +{ + struct xe_ttm_vram_offline_resource *nentry; + struct xe_bo *pbo_to_put = NULL; + struct xe_bo *pbo = NULL; + struct gpu_buddy_block *block; + u64 size = PAGE_SIZE; + int ret = 0; + + scoped_guard(mutex, &vram_mgr->lock) { + if (xe_ttm_vram_page_already_processed(vram_mgr, addr)) + return -EEXIST; + block = gpu_buddy_allocated_addr_to_block(mm, addr); + if (WARN_ON(IS_ERR(block))) + return PTR_ERR(block); + + nentry = kzalloc_obj(*nentry); + if (!nentry) + return -ENOMEM; + INIT_LIST_HEAD(&nentry->blocks); + nentry->status = XE_PAGE_RESERVE_PENDING; + nentry->addr = addr; + + if (block) { + pbo = xe_ttm_vram_block_owner_get(xe, block); + + /* + * Critical kernel BO? Best-effort check without resv lock; + * worst case a concurrent pin causes reset path unnecessarily. + */ + if (pbo && ((pbo->ttm.type == ttm_bo_type_kernel && + !(pbo->flags & XE_BO_FLAG_PINNED_LATE_RESTORE)) || + (xe_bo_is_user(pbo) && xe_bo_is_pinned(pbo)))) { + kfree(nentry); + pbo_to_put = pbo; + drm_err(&xe->drm, + "%s: addr: 0x%llx is critical kernel bo, requesting SBR\n", + __func__, addr); + break; + } + /* Queue free(to-be-purged) pages */ + ++vram_mgr->n_queued_pages; + list_add_rcu(&nentry->queued_link, &vram_mgr->queued_pages); + } else { + /* Immediately offline unoccupied pages */ + /* Queue free(to-be-reserved) pages */ + ret = xe_ttm_vram_buddy_alloc(vram_mgr, addr, addr + size, + size, size, &nentry->blocks, + GPU_BUDDY_RANGE_ALLOCATION, + NULL, &nentry->used_visible_size); + if (ret) { + nentry->status = XE_PAGE_RESERVE_FAIL; + drm_dbg(&xe->drm, + "Page at addr:0x%llx still busy (%d), deferring reservation\n", + addr, ret); + ++vram_mgr->n_queued_pages; + list_add_rcu(&nentry->queued_link, &vram_mgr->queued_pages); + return 0; + } + ++vram_mgr->n_offlined_pages; + list_add_rcu(&nentry->offlined_link, &vram_mgr->offlined_pages); + return ret; + } + } + + /* Deferred put outside lock to avoid recursive deadlock */ + if (pbo_to_put) { + xe_bo_put(pbo_to_put); + /* Hint System controller driver for reset with -EIO */ + return -EIO; + } + + if (pbo) { + /* + * Purge BO containing address - reference held from above. + * This does not necessarily free the blocks synchronously: if + * the BO is not idle, ttm_bo_pipeline_gutting() hands the + * resource to a ghost object and it is released only once the + * move fences signal. The reservation below then fails and is + * retried from xe_ttm_vram_mgr_del(). + */ + ret = xe_ttm_vram_purge_page(xe, pbo); + xe_bo_put(pbo); + if (ret) + drm_warn(&xe->drm, "Purge failed at addr:0x%llx, ret:%d\n", addr, ret); + } + + return 0; +} + +static struct xe_vram_region *xe_ttm_vram_addr_to_region(struct xe_device *xe, u64 addr) +{ + struct xe_tile *tile; + u8 id; + + for_each_tile(tile, xe, id) { + struct xe_vram_region *vr = tile->mem.vram; + + if (!vr) + continue; + + if (addr >= vr->dpa_base && addr < (vr->dpa_base + vr->usable_size)) + return vr; + + /* CCS, GSM, or DSM — infrastructure zone, needs reset */ + if (addr >= (vr->dpa_base + vr->usable_size) && + addr < (vr->dpa_base + vr->actual_physical_size)) + return NULL; + } + + /* + * Return an explicit error pointer so the caller knows the addr + * is invalid and should be ignored, NOT SBR. + */ + return ERR_PTR(-EOPNOTSUPP); +} + +/** + * xe_ttm_vram_handle_addr_fault - Handle vram physical address error flaged + * @xe: pointer to parent device + * @addr: physical faulty address + * + * Handle the physcial faulty address error on specific tile. + * + * Returns 0 for success, negative error code otherwise as follow: + * * %-EIO - critical BO or address outside any VRAM region; next action is reset. + * * %-EOPNOTSUPP - log-only policy or unknown address; no further action. + * * %-ENOMEM - allocation failure; next action is reset. + * * %-ENXIO - address not found in buddy; no further action. + * * %-EEXIST - address already processed; no further action. + * + * A return of 0 means the page is tracked. It may still be listed as + * pending if the blocks backing it could not be freed immediately; the + * reservation is then completed from xe_ttm_vram_mgr_del(). + */ +int xe_ttm_vram_handle_addr_fault(struct xe_device *xe, u64 addr) +{ + struct xe_ttm_vram_mgr *vram_mgr; + struct xe_vram_region *vr; + struct gpu_buddy *mm; + + /* Assert that the address is PAGE_SIZE aligned */ + if (WARN_ON_ONCE(!IS_ALIGNED(addr, PAGE_SIZE))) { + drm_err(&xe->drm, "Address %llx is not %lu aligned!\n", addr, PAGE_SIZE); + return -EINVAL; + } + + vr = xe_ttm_vram_addr_to_region(xe, addr); + if (IS_ERR(vr)) { + /* + * The addr is outside VRAM and GSM. + * Log a debug message if needed, and safely exit/ignore. + */ + drm_dbg(&xe->drm, "Address %llx is out of bounds, ignoring fault.\n", addr); + return PTR_ERR(vr); + } + if (!vr) { + drm_err(&xe->drm, "%s:%d GSM addr:%llx error requesting SBR\n", + __func__, __LINE__, addr); + /* Hint System controller driver for reset with -EIO */ + return -EIO; + } + vram_mgr = &vr->ttm; + mm = &vram_mgr->mm; + + if (xe->ras.disable_vram_page_offline) { + xe_err(xe, "0x%llx is reported as corrupted address by HW\n", + addr); + return -EOPNOTSUPP; + } + + /* Reserve page at address */ + return xe_ttm_vram_reserve_page_at_addr(xe, addr - vr->dpa_base, vram_mgr, mm); +} +EXPORT_SYMBOL(xe_ttm_vram_handle_addr_fault); + +/** + * xe_ttm_vram_inject_fault - Inject a VRAM page fault for testing + * @xe: xe device instance + * + * Picks the last unallocated VRAM page and reports it as faulted + * via xe_ttm_vram_handle_addr_fault(). Used by the fault-inject + * debugfs interface for testing page offlining. + * + * Note: Executing this test will permanently retire the allocated + * memory tracking pages. The driver must be rebinded (unbind and bind) + * post-test execution to reclaim the reserved space, as these pages + * cannot be freed or reclaimed dynamically while the current instance + * remains active. + * + * Return: 0 on success, negative error code on failure. + */ +int xe_ttm_vram_inject_fault(struct xe_device *xe) +{ + struct xe_tile *tile = xe_device_get_root_tile(xe); + struct xe_vram_region *vr = tile->mem.vram; + struct xe_ttm_vram_mgr *vram_mgr = &vr->ttm; + struct gpu_buddy *mm = &vram_mgr->mm; + u64 addr; + + if (vr->actual_physical_size < PAGE_SIZE) + return -ENOSPC; + + addr = vr->actual_physical_size - PAGE_SIZE; + while (addr < vr->actual_physical_size) { + struct gpu_buddy_block *block; + bool found = false; + + scoped_guard(mutex, &vram_mgr->lock) { + block = gpu_buddy_allocated_addr_to_block(mm, addr); + if (!block) + found = true; + } + + /* + * Intentional race window: xe_ttm_vram_handle_addr_fault() + * re-acquires vram_mgr->lock internally, so we cannot hold + * it here. A concurrent allocation claiming this page between + * the two calls is an acceptable false negative for this + * test-only path. + */ + if (found) + return xe_ttm_vram_handle_addr_fault(xe, addr + vr->dpa_base); + + cond_resched(); + if (addr == 0) + break; + addr -= PAGE_SIZE; + } + + return -ENOSPC; +} +EXPORT_SYMBOL(xe_ttm_vram_inject_fault); + +static int vram_bad_pages_show(struct seq_file *m, void *unused) +{ + struct xe_device *xe = m->private; + struct xe_ttm_vram_offline_resource *pos; + struct ttm_resource_manager *man; + struct xe_ttm_vram_mgr *mgr; + struct xe_tile *tile; + u8 id; + + man = ttm_manager_type(&xe->ttm, XE_PL_VRAM0); + if (man) + /* TODO Hook with RAS to show max_pages fetched from FW */ + seq_printf(m, "max_pages: %d\n", + to_xe_ttm_vram_mgr(man)->max_pages); + + for_each_tile(tile, xe, id) { + struct xe_vram_region *vr = tile->mem.vram; + + man = ttm_manager_type(&xe->ttm, XE_PL_VRAM0 + id); + if (!man || !vr) + continue; + mgr = to_xe_ttm_vram_mgr(man); + + rcu_read_lock(); + + list_for_each_entry_rcu(pos, &mgr->offlined_pages, offlined_link) { + u64 pfn; + + pfn = (pos->addr + vr->dpa_base) >> PAGE_SHIFT; + seq_printf(m, "0x%016llx : 0x%016lx : R\n", pfn, PAGE_SIZE); + } + + list_for_each_entry_rcu(pos, &mgr->queued_pages, queued_link) { + u64 pfn; + + pfn = (pos->addr + vr->dpa_base) >> PAGE_SHIFT; + seq_printf(m, "0x%016llx : 0x%016lx : %c\n", + pfn, PAGE_SIZE, pos->status ? 'F' : 'P'); + } + + rcu_read_unlock(); + } + + return 0; +} +DEFINE_SHOW_ATTRIBUTE(vram_bad_pages); + +/** + * xe_ttm_vram_debugfs_init - Initialize VRAM debugfs interfaces + * @xe: The xe device structure pointer + * @root: The root dentry of the debugfs directory + * + * This function registers platform-specific VRAM debugfs files used for + * testing and debugging. Currently, it exposes the "vram_bad_pages" interface + * to inspect marked faulty memory pages, restricted specifically to the + * %XE_CRESCENTISLAND platform. + * + * Return: Void. + */ +void xe_ttm_vram_debugfs_init(struct xe_device *xe, struct dentry *root) +{ + /* + * TODO: Replace platform check with xe->info + * once the feature flag is plumbed through device info. + */ + if (xe->info.platform != XE_CRESCENTISLAND) + return; + debugfs_create_file("vram_bad_pages", 0444, root, xe, &vram_bad_pages_fops); +} diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.h b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.h index 87b7fae5edba..d77f067d197b 100644 --- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.h +++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.h @@ -17,6 +17,7 @@ int __xe_ttm_vram_mgr_init(struct xe_device *xe, struct xe_ttm_vram_mgr *mgr, u32 mem_type, u64 size, u64 io_size, u64 default_page_size); int xe_ttm_vram_mgr_init(struct xe_device *xe, struct xe_vram_region *vram); +void xe_ttm_vram_debugfs_init(struct xe_device *xe, struct dentry *root); int xe_ttm_vram_mgr_alloc_sgt(struct xe_device *xe, struct ttm_resource *res, u64 offset, u64 length, @@ -30,6 +31,8 @@ u64 xe_ttm_vram_get_avail(struct ttm_resource_manager *man); u64 xe_ttm_vram_get_cpu_visible_size(struct ttm_resource_manager *man); void xe_ttm_vram_get_used(struct ttm_resource_manager *man, u64 *used, u64 *used_visible); +int xe_ttm_vram_handle_addr_fault(struct xe_device *xe, u64 addr); +int xe_ttm_vram_inject_fault(struct xe_device *xe); static inline struct xe_ttm_vram_mgr_resource * to_xe_ttm_vram_mgr_resource(struct ttm_resource *res) diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h b/drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h index 9106da056b49..efcf3e1d4e80 100644 --- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h +++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h @@ -19,6 +19,14 @@ struct xe_ttm_vram_mgr { struct ttm_resource_manager manager; /** @mm: DRM buddy allocator which manages the VRAM */ struct gpu_buddy mm; + /** @offlined_pages: List of offlined pages */ + struct list_head offlined_pages; + /** @n_offlined_pages: Number of offlined pages */ + u16 n_offlined_pages; + /** @queued_pages: List of queued pages */ + struct list_head queued_pages; + /** @n_queued_pages: Number of queued pages */ + u16 n_queued_pages; /** @visible_size: Proped size of the CPU visible portion */ u64 visible_size; /** @visible_avail: CPU visible portion still unallocated */ @@ -29,6 +37,8 @@ struct xe_ttm_vram_mgr { struct mutex lock; /** @mem_type: The TTM memory type */ u32 mem_type; + /** @max_pages: max pages that can be in offline queue retrieved from FW */ + u16 max_pages; }; /** @@ -45,4 +55,34 @@ struct xe_ttm_vram_mgr_resource { unsigned long flags; }; +/** + * enum xe_page_reserve_status - Buddy reservation status + * @XE_PAGE_RESERVE_PENDING: reservation in progress + * @XE_PAGE_RESERVE_FAIL: reservation failed + */ +enum xe_page_reserve_status { + XE_PAGE_RESERVE_PENDING = 0, + XE_PAGE_RESERVE_FAIL, +}; + +/** + * struct xe_ttm_vram_offline_resource - Tracks a single offlined VRAM page + */ +struct xe_ttm_vram_offline_resource { + /** @offlined_link: Link into mgr->offlined_pages */ + struct list_head offlined_link; + /** @queued_link: Link into mgr->queued_pages */ + struct list_head queued_link; + /** @blocks: Buddy blocks reserved for this page */ + struct list_head blocks; + /** @used_visible_size: CPU-visible bytes consumed */ + u64 used_visible_size; + /** @addr: Faulty DPA reported by HW */ + u64 addr; + /** @status: buddy reservation status */ + enum xe_page_reserve_status status; + /** @rcu: RCU head for deferred freeing */ + struct rcu_head rcu; +}; + #endif diff --git a/drivers/gpu/drm/xe/xe_userptr.c b/drivers/gpu/drm/xe/xe_userptr.c index 90ac141fc12d..9c1dac0fce6f 100644 --- a/drivers/gpu/drm/xe/xe_userptr.c +++ b/drivers/gpu/drm/xe/xe_userptr.c @@ -91,7 +91,7 @@ int xe_vma_userptr_pin_pages(struct xe_userptr_vma *uvma) if (vma->gpuva.flags & XE_VMA_DESTROYED) return 0; - return drm_gpusvm_get_pages(&vm->svm.gpusvm, &uvma->userptr.pages, + return drm_gpusvm_get_pages(&vm->svm.gpusvm, &uvma->userptr.pages, 1, uvma->userptr.notifier.mm, &uvma->userptr.notifier, xe_vma_userptr(vma), diff --git a/drivers/gpu/drm/xe/xe_vm.c b/drivers/gpu/drm/xe/xe_vm.c index e97b061be82a..efa5ff6cc823 100644 --- a/drivers/gpu/drm/xe/xe_vm.c +++ b/drivers/gpu/drm/xe/xe_vm.c @@ -655,6 +655,7 @@ void xe_vm_add_fault_entry_pf(struct xe_vm *vm, struct xe_pagefault *pf) pf->consumer.fault_type_level); e->fault_level = FIELD_GET(XE_PAGEFAULT_LEVEL_MASK, pf->consumer.fault_type_level); + e->srcid = FIELD_GET(XE_PAGEFAULT_SRCID_MASK, pf->consumer.id); list_add_tail(&e->list, &vm->faults.list); vm->faults.len++; @@ -1875,6 +1876,8 @@ static void xe_vm_close(struct xe_vm *vm) bound = drm_dev_enter(&xe->drm, &idx); down_write(&vm->lock); + xe_vm_lock(vm, false); + if (xe_vm_in_fault_mode(vm)) xe_svm_notifier_lock(vm); @@ -1902,6 +1905,8 @@ static void xe_vm_close(struct xe_vm *vm) if (xe_vm_in_fault_mode(vm)) xe_svm_notifier_unlock(vm); + + xe_vm_unlock(vm); up_write(&vm->lock); if (bound) @@ -4277,6 +4282,11 @@ static u8 xe_to_user_fault_level(u8 fault_level) return fault_level; } +static u8 xe_to_user_srcid(u8 srcid) +{ + return srcid; +} + static int fill_faults(struct xe_vm *vm, struct drm_xe_vm_get_property *args) { @@ -4304,6 +4314,8 @@ static int fill_faults(struct xe_vm *vm, fault_entry.fault_type = xe_to_user_fault_type(entry->fault_type); fault_entry.fault_level = xe_to_user_fault_level(entry->fault_level); + fault_entry.srcid = xe_to_user_srcid(entry->srcid); + memcpy(&fault_list[i], &fault_entry, entry_size); i++; diff --git a/drivers/gpu/drm/xe/xe_vm_types.h b/drivers/gpu/drm/xe/xe_vm_types.h index 68588b624212..648031e64145 100644 --- a/drivers/gpu/drm/xe/xe_vm_types.h +++ b/drivers/gpu/drm/xe/xe_vm_types.h @@ -202,6 +202,7 @@ struct xe_device; * @access_type: type of address access that resulted in fault * @fault_type: type of fault reported * @fault_level: fault level of the fault + * @srcid: ID of the faulting hardware unit */ struct xe_vm_fault_entry { struct list_head list; @@ -210,6 +211,7 @@ struct xe_vm_fault_entry { u8 access_type; u8 fault_type; u8 fault_level; + u8 srcid; }; struct xe_vm { diff --git a/drivers/gpu/drm/xe/xe_vram.c b/drivers/gpu/drm/xe/xe_vram.c index 56cff1e44530..dcedd8cfd731 100644 --- a/drivers/gpu/drm/xe/xe_vram.c +++ b/drivers/gpu/drm/xe/xe_vram.c @@ -17,8 +17,11 @@ #include "xe_device.h" #include "xe_force_wake.h" #include "xe_gt_mcr.h" +#include "xe_map.h" +#include "xe_migrate.h" #include "xe_mmio.h" #include "xe_sriov.h" +#include "xe_tile.h" #include "xe_tile_sriov_vf.h" #include "xe_ttm_vram_mgr.h" #include "xe_vram.h" @@ -55,9 +58,6 @@ static int determine_lmem_bar_size(struct xe_device *xe, struct xe_vram_region * /* XXX: Need to change when xe link code is ready */ lmem_bar->dpa_base = 0; - /* set up a map to the total memory area. */ - lmem_bar->mapping = devm_ioremap_wc(&pdev->dev, lmem_bar->io_start, lmem_bar->io_size); - return 0; } @@ -196,7 +196,7 @@ static void vram_fini(void *arg) struct xe_tile *tile; int id; - xe->mem.vram->mapping = NULL; + xe_assert(xe, !xe->mem.vram->mapping); for_each_tile(tile, xe, id) { tile->mem.vram->mapping = NULL; @@ -257,8 +257,15 @@ static int vram_region_init(struct xe_device *xe, struct xe_vram_region *vram, return -ENODEV; } + if (vram != xe->mem.vram) { + struct pci_dev *pdev = to_pci_dev(xe->drm.dev); + + vram->mapping = devm_ioremap_wc(&pdev->dev, vram->io_start, vram->io_size); + if (!vram->mapping) + return -ENOMEM; + } + vram->dpa_base = lmem_bar->dpa_base + offset; - vram->mapping = lmem_bar->mapping + offset; vram->usable_size = usable_size; print_vram_region_info(xe, vram); @@ -403,3 +410,241 @@ resource_size_t xe_vram_region_actual_physical_size(const struct xe_vram_region return vram ? vram->actual_physical_size : 0; } EXPORT_SYMBOL_IF_KUNIT(xe_vram_region_actual_physical_size); + +#if IS_ENABLED(CONFIG_DRM_XE_DEBUG_MEM) +static void memtest_bo_cleanup(void *arg) +{ + struct xe_device *xe = arg; + + xe_vram_free_memtest_bos(xe); +} + +int xe_vram_reserve_memtest_bo(struct xe_device *xe) +{ + struct xe_tile *tile; + u8 id; + + if (IS_SRIOV_VF(xe)) + return 0; + + for_each_tile(tile, xe, id) { + u64 vram_size; + + if (!tile->mem.vram) + continue; + + if (tile->mem.vram->io_size < tile->mem.vram->usable_size) { + drm_info(&xe->drm, + "Tile %d: Small-BAR system detected, skipping VRAM memtest\n", + id); + continue; + } + + vram_size = tile->mem.vram->usable_size; + + tile->mem.memtest_bo = xe_bo_create_pin_map_at_novm(xe, tile, SZ_64K, + vram_size - SZ_64K, + ttm_bo_type_kernel, + XE_BO_FLAG_VRAM_IF_DGFX(tile), + 0, false); + if (IS_ERR(tile->mem.memtest_bo)) { + drm_warn(&xe->drm, "Tile %d: Failed to reserve memtest BO\n", id); + tile->mem.memtest_bo = NULL; + continue; + } + + drm_info(&xe->drm, "Tile %d: Reserved memtest BO at offset 0x%llx\n", + id, vram_size - SZ_64K); + } + + return devm_add_action_or_reset(xe->drm.dev, memtest_bo_cleanup, xe); +} + +void xe_vram_free_memtest_bos(struct xe_device *xe) +{ + struct xe_tile *tile; + u8 id; + + for_each_tile(tile, xe, id) { + if (tile->mem.memtest_bo) { + xe_bo_unpin_map_no_vm(tile->mem.memtest_bo); + tile->mem.memtest_bo = NULL; + } + } +} + +int xe_vram_memtest(struct xe_device *xe) +{ + struct xe_tile *tile; + u8 id; + int err = 0; + + if (IS_SRIOV_VF(xe)) + return 0; + + for_each_tile(tile, xe, id) { + struct xe_bo *last_page_bo = tile->mem.memtest_bo; + struct dma_fence *fence; + bool overlap = false; + int i; + u8 val; + + if (!last_page_bo || !tile->migrate) + continue; + + drm_info(&xe->drm, "Tile %d: Running VRAM memtest...\n", id); + + /* CPU write and readback first and last byte of the last page */ + xe_map_wr(xe, &last_page_bo->vmap, 0, u8, 0xA5); + xe_map_wr(xe, &last_page_bo->vmap, SZ_64K - 1, u8, 0x5A); + + val = xe_map_rd(xe, &last_page_bo->vmap, 0, u8); + if (drm_WARN(&xe->drm, val != 0xA5, + "Tile %d: CPU memtest failed at offset 0 (expected 0xA5, got 0x%02x)\n", + id, val)) { + err = -EIO; + goto unpin; + } + + val = xe_map_rd(xe, &last_page_bo->vmap, SZ_64K - 1, u8); + if (drm_WARN(&xe->drm, val != 0x5A, + "Tile %d: CPU memtest failed at offset 65535 (expected 0x5A, got 0x%02x)\n", + id, val)) { + err = -EIO; + goto unpin; + } + + /* Non-CCS access via GPU on the last page */ + xe_bo_lock(last_page_bo, false); + fence = xe_migrate_clear(tile->migrate, last_page_bo, + last_page_bo->ttm.resource, + XE_MIGRATE_CLEAR_FLAG_BO_DATA); + xe_bo_unlock(last_page_bo); + + if (!IS_ERR(fence)) { + dma_fence_wait(fence, false); + dma_fence_put(fence); + } else { + err = PTR_ERR(fence); + goto unpin; + } + + val = xe_map_rd(xe, &last_page_bo->vmap, 0, u8); + if (drm_WARN(&xe->drm, val != 0x00, + "Tile %d: GPU memtest clear failed at offset 0 (expected 0x00, got 0x%02x)\n", + id, val)) { + err = -EIO; + goto unpin; + } + + /* + * Check for CCS overlap on the root tile. + * + * TODO: maybe extend if we ever get multi-tile + CCS. Pay + * special attention to the l2 flush below. Currently that is + * hard coded to the root tile. + */ + if (!id && xe_device_has_flat_ccs(xe) && + GRAPHICS_VERx100(xe) >= 2000) { + struct xe_bo *scratch_bo_before; + struct xe_bo *scratch_bo_after; + + scratch_bo_before = xe_bo_create_pin_map_novm(xe, tile, SZ_64K, + ttm_bo_type_kernel, + XE_BO_FLAG_VRAM_IF_DGFX(tile), + false); + if (IS_ERR(scratch_bo_before)) { + err = PTR_ERR(scratch_bo_before); + goto unpin; + } + + scratch_bo_after = xe_bo_create_pin_map_novm(xe, tile, SZ_64K, + ttm_bo_type_kernel, + XE_BO_FLAG_VRAM_IF_DGFX(tile), + false); + if (IS_ERR(scratch_bo_after)) { + xe_bo_unpin_map_no_vm(scratch_bo_before); + err = PTR_ERR(scratch_bo_after); + goto unpin; + } + + /* Save original CCS metadata for PA 0 + */ + err = xe_migrate_debug_ccs_overlap(tile->migrate, scratch_bo_before, false); + if (err) { + xe_bo_unpin_map_no_vm(scratch_bo_before); + xe_bo_unpin_map_no_vm(scratch_bo_after); + goto unpin; + } + + /* + * Fill last page. If there is CCS overlap in the last + * page this will snag the raw CCS storage. + */ + xe_map_memset(xe, &last_page_bo->vmap, 0, 0x5A, SZ_64K); + xe_device_wmb(xe); + + /* + * Global invalidation. Some BMG SKUs will cache the BAR + * writes in the GPU side VRAM cache. Make sure above + * writes are fully flushed out to VRAM, so this is + * hopefully more well behaved with the CCS unit, if + * there is indeed CCS overlap with normal VRAM. Since + * there is a separate CCS cache, the CCS unit might not + * respect the GPU VRAM cache for CCS accesses, so opt + * for being super careful here. + */ + xe_device_l2_flush(xe, true); + + /* Use GPU to clear CCS state for PA 0 */ + xe_map_memset(xe, &scratch_bo_after->vmap, 0, 0x00, SZ_64K); + err = xe_migrate_debug_ccs_overlap(tile->migrate, scratch_bo_after, true); + if (err) { + xe_bo_unpin_map_no_vm(scratch_bo_before); + xe_bo_unpin_map_no_vm(scratch_bo_after); + goto unpin; + } + /* + * Global invalidation. Ensure CCS caches really are + * nuked and the raw CCS data is visible in VRAM, for + * the below access. + */ + xe_device_l2_flush(xe, true); + + /* Check if last_page_bo was corrupted by the GPU CCS clear */ + for (i = 0; i < SZ_64K; i += 8) { + u64 payload = xe_map_rd(xe, &last_page_bo->vmap, i, u64); + + if (payload != 0x5A5A5A5A5A5A5A5AULL) { + overlap = true; + break; + } + } + + /* Restore original CCS metadata for PA 0 + */ + err = xe_migrate_debug_ccs_overlap(tile->migrate, scratch_bo_before, true); + if (err) + drm_warn(&xe->drm, "Failed to restore CCS metadata\n"); + + xe_bo_unpin_map_no_vm(scratch_bo_before); + xe_bo_unpin_map_no_vm(scratch_bo_after); + } + + if (drm_WARN(&xe->drm, overlap, + "Tile %d: VRAM bounds overlap CCS region! VRAM sizing is incorrect.\n", + id)) { + err = -EINVAL; + goto unpin; + } + + drm_info(&xe->drm, "Tile %d: VRAM memtest completed.\n", id); + +unpin: + if (err) + break; + } + + xe_vram_free_memtest_bos(xe); + + return err; +} +#endif diff --git a/drivers/gpu/drm/xe/xe_vram.h b/drivers/gpu/drm/xe/xe_vram.h index dd1c8bf17922..38425d81f777 100644 --- a/drivers/gpu/drm/xe/xe_vram.h +++ b/drivers/gpu/drm/xe/xe_vram.h @@ -23,4 +23,14 @@ resource_size_t xe_vram_region_dpa_base(const struct xe_vram_region *vram); resource_size_t xe_vram_region_usable_size(const struct xe_vram_region *vram); resource_size_t xe_vram_region_actual_physical_size(const struct xe_vram_region *vram); +#if IS_ENABLED(CONFIG_DRM_XE_DEBUG_MEM) +int xe_vram_reserve_memtest_bo(struct xe_device *xe); +void xe_vram_free_memtest_bos(struct xe_device *xe); +int xe_vram_memtest(struct xe_device *xe); +#else +static inline int xe_vram_reserve_memtest_bo(struct xe_device *xe) { return 0; } +static inline void xe_vram_free_memtest_bos(struct xe_device *xe) {} +static inline int xe_vram_memtest(struct xe_device *xe) { return 0; } +#endif + #endif |
