summaryrefslogtreecommitdiff
path: root/drivers/gpu/drm
diff options
context:
space:
mode:
authorMark Brown <broonie@kernel.org>2026-09-14 15:21:31 +0100
committerMark Brown <broonie@kernel.org>2026-09-14 15:21:31 +0100
commit62c0e6c9d9a275040a145a92dd42316a110e484e (patch)
tree4ffc56e93ffd8b44bfb6245428a94cd827edab27 /drivers/gpu/drm
parent4f6280141c1c87d1e82713ceb2751ecfab47eb3a (diff)
parent7dbd24e9d5311b6afdbdd875dd247ab5ff54edca (diff)
downloadlinux-next-62c0e6c9d9a275040a145a92dd42316a110e484e.tar.gz
linux-next-62c0e6c9d9a275040a145a92dd42316a110e484e.zip
Merge branch 'drm-xe-next' of https://gitlab.freedesktop.org/drm/xe/kernel.git
Diffstat (limited to 'drivers/gpu/drm')
-rw-r--r--drivers/gpu/drm/drm_gpusvm.c406
-rw-r--r--drivers/gpu/drm/xe/abi/xe_log_abi.h1
-rw-r--r--drivers/gpu/drm/xe/display/xe_dsb_buffer.c2
-rw-r--r--drivers/gpu/drm/xe/display/xe_fb_pin.c2
-rw-r--r--drivers/gpu/drm/xe/display/xe_panic.c3
-rw-r--r--drivers/gpu/drm/xe/tests/xe_pci.c25
-rw-r--r--drivers/gpu/drm/xe/xe_bo.c76
-rw-r--r--drivers/gpu/drm/xe/xe_bo.h3
-rw-r--r--drivers/gpu/drm/xe/xe_bo_types.h8
-rw-r--r--drivers/gpu/drm/xe/xe_configfs.c72
-rw-r--r--drivers/gpu/drm/xe/xe_configfs.h2
-rw-r--r--drivers/gpu/drm/xe/xe_debugfs.c51
-rw-r--r--drivers/gpu/drm/xe/xe_debugfs.h2
-rw-r--r--drivers/gpu/drm/xe/xe_device.c120
-rw-r--r--drivers/gpu/drm/xe/xe_device.h2
-rw-r--r--drivers/gpu/drm/xe/xe_device_types.h19
-rw-r--r--drivers/gpu/drm/xe/xe_dma_buf.c3
-rw-r--r--drivers/gpu/drm/xe/xe_drm_ras_types.h3
-rw-r--r--drivers/gpu/drm/xe/xe_exec_queue.c72
-rw-r--r--drivers/gpu/drm/xe/xe_exec_queue_types.h7
-rw-r--r--drivers/gpu/drm/xe/xe_execlist.c4
-rw-r--r--drivers/gpu/drm/xe/xe_gt.c11
-rw-r--r--drivers/gpu/drm/xe/xe_gt_debugfs.c54
-rw-r--r--drivers/gpu/drm/xe/xe_guc_pagefault.c8
-rw-r--r--drivers/gpu/drm/xe/xe_guc_submit.c115
-rw-r--r--drivers/gpu/drm/xe/xe_guc_submit.h5
-rw-r--r--drivers/gpu/drm/xe/xe_hwmon.c7
-rw-r--r--drivers/gpu/drm/xe/xe_log.c42
-rw-r--r--drivers/gpu/drm/xe/xe_lrc.c6
-rw-r--r--drivers/gpu/drm/xe/xe_lrc.h1
-rw-r--r--drivers/gpu/drm/xe/xe_migrate.c121
-rw-r--r--drivers/gpu/drm/xe/xe_migrate.h6
-rw-r--r--drivers/gpu/drm/xe/xe_mmio_gem.c115
-rw-r--r--drivers/gpu/drm/xe/xe_mmio_gem.h2
-rw-r--r--drivers/gpu/drm/xe/xe_pagefault.c24
-rw-r--r--drivers/gpu/drm/xe/xe_pagefault_types.h9
-rw-r--r--drivers/gpu/drm/xe/xe_pci.c29
-rw-r--r--drivers/gpu/drm/xe/xe_pcode.c2
-rw-r--r--drivers/gpu/drm/xe/xe_pcode.h4
-rw-r--r--drivers/gpu/drm/xe/xe_pt.c28
-rw-r--r--drivers/gpu/drm/xe/xe_ras.c9
-rw-r--r--drivers/gpu/drm/xe/xe_res_cursor.h5
-rw-r--r--drivers/gpu/drm/xe/xe_shrinker.c124
-rw-r--r--drivers/gpu/drm/xe/xe_svm.c15
-rw-r--r--drivers/gpu/drm/xe/xe_svm.h20
-rw-r--r--drivers/gpu/drm/xe/xe_sysctrl.c92
-rw-r--r--drivers/gpu/drm/xe/xe_sysctrl.h2
-rw-r--r--drivers/gpu/drm/xe/xe_sysctrl_mailbox_types.h45
-rw-r--r--drivers/gpu/drm/xe/xe_tile_types.h4
-rw-r--r--drivers/gpu/drm/xe/xe_ttm_vram_mgr.c600
-rw-r--r--drivers/gpu/drm/xe/xe_ttm_vram_mgr.h3
-rw-r--r--drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h40
-rw-r--r--drivers/gpu/drm/xe/xe_userptr.c2
-rw-r--r--drivers/gpu/drm/xe/xe_vm.c12
-rw-r--r--drivers/gpu/drm/xe/xe_vm_types.h2
-rw-r--r--drivers/gpu/drm/xe/xe_vram.c255
-rw-r--r--drivers/gpu/drm/xe/xe_vram.h10
57 files changed, 2234 insertions, 478 deletions
diff --git a/drivers/gpu/drm/drm_gpusvm.c b/drivers/gpu/drm/drm_gpusvm.c
index a93eee7ddb9e..b6c9d3a07dc8 100644
--- a/drivers/gpu/drm/drm_gpusvm.c
+++ b/drivers/gpu/drm/drm_gpusvm.c
@@ -80,6 +80,13 @@
* };
* };
*
+ * static struct drm_gpusvm_pages *
+ * driver_pages(struct driver_range *drange)
+ * {
+ * return drange->num_pages == 1 ? &drange->inline_pages :
+ * drange->pages;
+ * }
+ *
* In the N:1 case the driver allocates the pages array with a zeroing
* allocator (e.g. kcalloc(num_pages, ...)), initialises each entry with
* drm_gpusvm_init_pages(), and frees each entry with
@@ -89,6 +96,28 @@
* Each drm_gpusvm_pages must be zero-initialised and initialised with
* drm_gpusvm_init_pages(), called once per entry.
*
+ * The 1:1 examples below pass @num_pages == 1 and &drange->pages. In the
+ * N:1 case the driver instead passes the whole array and its count, so a
+ * single call faults the CPU range once and DMA maps it for every owning
+ * drm_device, e.g.:
+ *
+ * .. code-block:: c
+ *
+ * // GPU fault handler: one fault, one DMA mapping per device
+ * err = drm_gpusvm_get_pages(gpusvm, driver_pages(drange),
+ * drange->num_pages, gpusvm->mm,
+ * &range->notifier->notifier,
+ * drm_gpusvm_range_start(range),
+ * drm_gpusvm_range_end(range), &ctx);
+ *
+ * // Notifier callback: mark every instance unmapped in one call
+ * drm_gpusvm_range_set_unmapped(range, driver_pages(drange),
+ * drange->num_pages, mmu_range);
+ *
+ * The unmap and free paths stay per-instance: iterate @num_pages over
+ * driver_pages(drange) and call drm_gpusvm_unmap_pages() /
+ * drm_gpusvm_free_pages() for each entry.
+ *
* - Operations:
* Define the interface for driver-specific GPU SVM operations such as
* range allocation, notifier allocation, and invalidations.
@@ -232,7 +261,7 @@
* goto retry;
* }
*
- * err = drm_gpusvm_get_pages(gpusvm, &drange->pages,
+ * err = drm_gpusvm_get_pages(gpusvm, &drange->pages, 1,
* gpusvm->mm, &range->notifier->notifier,
* drm_gpusvm_range_start(range),
* drm_gpusvm_range_end(range), &ctx);
@@ -1212,6 +1241,8 @@ static void __drm_gpusvm_unmap_pages(struct drm_gpusvm *gpusvm,
struct drm_gpusvm_pages_flags flags = {
.__flags = svm_pages->flags.__flags,
};
+ const struct drm_pagemap_addr *addrs =
+ drm_gpusvm_pages_first_dma(svm_pages, NULL);
bool use_iova = dma_use_iova(&svm_pages->state);
/*
@@ -1224,12 +1255,20 @@ static void __drm_gpusvm_unmap_pages(struct drm_gpusvm *gpusvm,
if (svm_pages->state_offset)
dma_iova_unlink(dev, &svm_pages->state, 0,
svm_pages->state_offset,
- svm_pages->dma_addr[0].dir, 0);
+ addrs[0].dir, 0);
dma_iova_free(dev, &svm_pages->state);
}
- for (i = 0, j = 0; i < npages; j++) {
- struct drm_pagemap_addr *addr = &svm_pages->dma_addr[j];
+ /*
+ * With IOVA and no device page the unlink above tore every
+ * entry down, and that is also when the range may be folded
+ * to one entry, which must not be walked per entry. dpagemap
+ * is set before the first device_map(), so it is also right
+ * on the error path, where the flags are not published yet.
+ */
+ for (i = 0, j = 0;
+ (!use_iova || dpagemap) && i < npages; j++) {
+ const struct drm_pagemap_addr *addr = &addrs[j];
if (addr->proto == DRM_INTERCONNECT_SYSTEM) {
/*
@@ -1270,6 +1309,18 @@ static void __drm_gpusvm_free_pages(struct drm_gpusvm *gpusvm,
{
lockdep_assert_held(&gpusvm->notifier_lock);
+ if (svm_pages->flags.inline_dma_mapping) {
+ struct drm_gpusvm_pages_flags flags = {
+ .__flags = svm_pages->flags.__flags,
+ };
+
+ svm_pages->inline_addr = (struct drm_pagemap_addr){};
+ flags.inline_dma_mapping = false;
+ /* WRITE_ONCE pairs with READ_ONCE for opportunistic checks */
+ WRITE_ONCE(svm_pages->flags.__flags, flags.__flags);
+ return;
+ }
+
if (svm_pages->dma_addr) {
kvfree(svm_pages->dma_addr);
svm_pages->dma_addr = NULL;
@@ -1417,144 +1468,105 @@ EXPORT_SYMBOL_GPL(drm_gpusvm_pages_valid);
/**
* drm_gpusvm_pages_valid_unlocked() - GPU SVM pages valid unlocked
* @gpusvm: Pointer to the GPU SVM structure
- * @svm_pages: Pointer to the GPU SVM pages structure
+ * @svm_pages: Array of GPU SVM pages structures
+ * @num_pages: Number of drm_gpusvm_pages instances in @svm_pages
*
- * This function determines if a GPU SVM pages are valid. Expected be called
- * without holding gpusvm->notifier_lock.
+ * This function determines if every GPU SVM pages instance is valid, resetting
+ * every instance which is not so that get_pages() maps it afresh. It therefore
+ * has to walk them all. Expected be called without holding
+ * gpusvm->notifier_lock.
*
- * Return: True if GPU SVM pages are valid, False otherwise
+ * Return: True if all GPU SVM pages are valid, False otherwise
*/
static bool drm_gpusvm_pages_valid_unlocked(struct drm_gpusvm *gpusvm,
- struct drm_gpusvm_pages *svm_pages)
+ struct drm_gpusvm_pages *svm_pages,
+ unsigned int num_pages)
{
- bool pages_valid;
-
- if (!svm_pages->dma_addr)
- return false;
+ bool pages_valid = true;
+ unsigned int p;
drm_gpusvm_notifier_lock(gpusvm);
- pages_valid = drm_gpusvm_pages_valid(gpusvm, svm_pages);
- if (!pages_valid)
- __drm_gpusvm_free_pages(gpusvm, svm_pages);
+ for (p = 0; p < num_pages; ++p) {
+ if (drm_gpusvm_pages_valid(gpusvm, &svm_pages[p]))
+ continue;
+ __drm_gpusvm_free_pages(gpusvm, &svm_pages[p]);
+ pages_valid = false;
+ }
drm_gpusvm_notifier_unlock(gpusvm);
return pages_valid;
}
/**
- * drm_gpusvm_get_pages() - Get pages and populate GPU SVM pages struct
+ * drm_gpusvm_pages_inlinable() - Whether the dma address can be inlined
+ * @svm_pages: The SVM pages instance that was just mapped
+ * @nentries: Number of entries the mapping loop produced
+ * @npages: Number of pages in the CPU range
+ *
+ * A THP maps as one huge page, and an IOVA reservation links every page of
+ * the range at the next offset, so the device addresses run contiguously from
+ * entry 0. Either way one entry describes the whole range, so the dma_addr
+ * array can be freed and the address kept inline.
+ *
+ * state_offset advances only on the IOVA branch, so reaching the full range
+ * length proves no device page was mapped in between. Only single page
+ * entries fold, so the order kept is 0 and describes the range truthfully.
+ * Larger chunks, several huge pages among them, stay an array that is
+ * already short and that a consumer places with one PTE each.
+ *
+ * Return: True if the mapping fits in a single drm_pagemap_addr.
+ */
+static bool drm_gpusvm_pages_inlinable(struct drm_gpusvm_pages *svm_pages,
+ unsigned long nentries,
+ unsigned long npages)
+{
+ if (nentries == 1)
+ return true;
+
+ return nentries == npages && dma_use_iova(&svm_pages->state) &&
+ svm_pages->state_offset == npages * PAGE_SIZE;
+}
+
+/**
+ * drm_gpusvm_dma_map_pages() - DMA map one drm_gpusvm_pages instance
* @gpusvm: Pointer to the GPU SVM structure
- * @svm_pages: The SVM pages to populate. This will contain the dma-addresses
- * @mm: The mm corresponding to the CPU range
- * @notifier: The corresponding notifier for the given CPU range
- * @pages_start: Start CPU address for the pages
- * @pages_end: End CPU address for the pages (exclusive)
+ * @svm_pages: The SVM pages instance to populate with dma-addresses
+ * @pfns: The already-faulted pfn array (size @npages)
+ * @npages: Number of pages in the CPU range
* @ctx: GPU SVM context
+ * @dma_dir: DMA data direction for the mappings
*
- * This function gets and maps pages for CPU range and ensures they are
- * mapped for DMA access.
+ * Map the faulted @pfns into @svm_pages for DMA access through its owning
+ * drm_device. Must be called under the notifier lock and only for an instance
+ * without a live mapping. On failure this unwinds the partial mapping of this
+ * instance before returning.
*
* Return: 0 on success, negative error code on failure.
*/
-int drm_gpusvm_get_pages(struct drm_gpusvm *gpusvm,
- struct drm_gpusvm_pages *svm_pages,
- struct mm_struct *mm,
- struct mmu_interval_notifier *notifier,
- unsigned long pages_start, unsigned long pages_end,
- const struct drm_gpusvm_ctx *ctx)
+static int drm_gpusvm_dma_map_pages(struct drm_gpusvm *gpusvm,
+ struct drm_gpusvm_pages *svm_pages,
+ unsigned long *pfns,
+ unsigned long npages,
+ const struct drm_gpusvm_ctx *ctx,
+ enum dma_data_direction dma_dir)
{
- struct hmm_range hmm_range = {
- .default_flags = HMM_PFN_REQ_FAULT | (ctx->read_only ? 0 :
- HMM_PFN_REQ_WRITE),
- .notifier = notifier,
- .start = pages_start,
- .end = pages_end,
- .dev_private_owner = ctx->device_private_page_owner,
- };
- void *zdd;
- unsigned long timeout =
- jiffies + msecs_to_jiffies(HMM_RANGE_DEFAULT_TIMEOUT);
- unsigned long remaining;
+ void *zdd = NULL;
unsigned long i, j;
- unsigned long npages = npages_in_range(pages_start, pages_end);
- unsigned long num_dma_mapped;
+ unsigned long num_dma_mapped = 0;
unsigned int order = 0;
- unsigned long *pfns;
int err = 0;
- struct dev_pagemap *pagemap;
+ struct dev_pagemap *pagemap = NULL;
struct drm_pagemap *dpagemap;
struct drm_gpusvm_pages_flags flags;
- enum dma_data_direction dma_dir = ctx->read_only ? DMA_TO_DEVICE :
- DMA_BIDIRECTIONAL;
struct dma_iova_state *state = &svm_pages->state;
- if (!svm_pages->drm)
- return -EINVAL;
-
-retry:
- remaining = timeout - jiffies;
-
- if (time_after_eq(jiffies, timeout))
- return -EBUSY;
-
- hmm_range.notifier_seq = mmu_interval_read_begin(notifier);
- if (drm_gpusvm_pages_valid_unlocked(gpusvm, svm_pages))
- goto set_seqno;
-
- pfns = kvmalloc_array(npages, sizeof(*pfns), GFP_KERNEL);
- if (!pfns)
- return -ENOMEM;
-
- if (!mmget_not_zero(mm)) {
- err = -EFAULT;
- goto err_free;
- }
-
- hmm_range.hmm_pfns = pfns;
- err = hmm_range_fault_unlocked_timeout(&hmm_range, remaining);
- mmput(mm);
- if (err)
- goto err_free;
+ lockdep_assert_held(&gpusvm->notifier_lock);
*state = (struct dma_iova_state){};
svm_pages->state_offset = 0;
-map_pages:
- /*
- * Perform all dma mappings under the notifier lock to not
- * access freed pages. A notifier will either block on
- * the notifier lock or unmap dma.
- */
- drm_gpusvm_notifier_lock(gpusvm);
-
flags.__flags = svm_pages->flags.__flags;
- if (flags.unmapped) {
- drm_gpusvm_notifier_unlock(gpusvm);
- err = -EFAULT;
- goto err_free;
- }
-
- if (mmu_interval_read_retry(notifier, hmm_range.notifier_seq)) {
- drm_gpusvm_notifier_unlock(gpusvm);
- kvfree(pfns);
- goto retry;
- }
-
- if (!svm_pages->dma_addr) {
- /* Unlock and restart mapping to allocate memory. */
- drm_gpusvm_notifier_unlock(gpusvm);
- svm_pages->dma_addr =
- kvzalloc_objs(*svm_pages->dma_addr, npages);
- if (!svm_pages->dma_addr) {
- err = -ENOMEM;
- goto err_free;
- }
- goto map_pages;
- }
- zdd = NULL;
- pagemap = NULL;
- num_dma_mapped = 0;
for (i = 0, j = 0; i < npages; ++j) {
struct page *page = hmm_pfn_to_page(pfns[i]);
@@ -1667,20 +1679,186 @@ map_pages:
if (pagemap)
flags.has_devmem_pages = true;
+ if (drm_gpusvm_pages_inlinable(svm_pages, j, npages)) {
+ struct drm_pagemap_addr addr = svm_pages->dma_addr[0];
+
+ kvfree(svm_pages->dma_addr);
+ svm_pages->inline_addr = addr;
+ flags.inline_dma_mapping = true;
+ }
+
/* WRITE_ONCE pairs with READ_ONCE for opportunistic checks */
WRITE_ONCE(svm_pages->flags.__flags, flags.__flags);
- drm_gpusvm_notifier_unlock(gpusvm);
- kvfree(pfns);
-set_seqno:
- svm_pages->notifier_seq = hmm_range.notifier_seq;
-
return 0;
err_unmap:
svm_pages->flags.has_dma_mapping = true;
__drm_gpusvm_unmap_pages(gpusvm, svm_pages, num_dma_mapped);
+ return err;
+}
+
+/**
+ * drm_gpusvm_get_pages() - Get pages and populate GPU SVM pages struct
+ * @gpusvm: Pointer to the GPU SVM structure
+ * @svm_pages: Array of SVM pages instances to populate with dma addresses
+ * @num_pages: Number of drm_gpusvm_pages instances in @svm_pages, must not be 0
+ * @mm: The mm corresponding to the CPU range
+ * @notifier: The corresponding notifier for the given CPU range
+ * @pages_start: Start CPU address for the pages
+ * @pages_end: End CPU address for the pages (exclusive)
+ * @ctx: GPU SVM context
+ *
+ * This function gets and maps pages for a CPU range and ensures they are
+ * mapped for DMA access. The HMM fault for the CPU range is performed once,
+ * the DMA mapping by drm_gpusvm_dma_map_pages() is then done per instance,
+ * one per owning drm_device. The retry against notifier races is kept here
+ * in common code so drivers never open code it.
+ *
+ * On error the instances mapped before the failing one stay mapped, so the
+ * caller must unmap and free every instance regardless of the return value.
+ *
+ * With &drm_gpusvm_ctx.no_dma_map no mapping state is recorded, so
+ * drm_gpusvm_pages_valid() never returns true and success is only a snapshot:
+ * the caller must recheck mmu_interval_read_retry() against the recorded
+ * &drm_gpusvm_pages.notifier_seq under the notifier lock, and hold it until
+ * its work is visible to invalidation.
+ *
+ * Return: 0 on success, negative error code on failure.
+ */
+int drm_gpusvm_get_pages(struct drm_gpusvm *gpusvm,
+ struct drm_gpusvm_pages *svm_pages,
+ unsigned int num_pages,
+ struct mm_struct *mm,
+ struct mmu_interval_notifier *notifier,
+ unsigned long pages_start, unsigned long pages_end,
+ const struct drm_gpusvm_ctx *ctx)
+{
+ struct hmm_range hmm_range = {
+ .default_flags = HMM_PFN_REQ_FAULT | (ctx->read_only ? 0 :
+ HMM_PFN_REQ_WRITE),
+ .notifier = notifier,
+ .start = pages_start,
+ .end = pages_end,
+ .dev_private_owner = ctx->device_private_page_owner,
+ };
+ unsigned long timeout =
+ jiffies + msecs_to_jiffies(HMM_RANGE_DEFAULT_TIMEOUT);
+ unsigned long remaining;
+ unsigned long npages = npages_in_range(pages_start, pages_end);
+ unsigned long *pfns;
+ int err = 0;
+ enum dma_data_direction dma_dir = ctx->read_only ? DMA_TO_DEVICE :
+ DMA_BIDIRECTIONAL;
+ const bool map_dma = !ctx->no_dma_map;
+ unsigned int p;
+
+ if (!num_pages)
+ return -EINVAL;
+
+ if (ctx->no_dma_map && ctx->devmem_only)
+ return -EINVAL;
+
+ if (map_dma) {
+ for (p = 0; p < num_pages; ++p) {
+ if (!svm_pages[p].drm)
+ return -EINVAL;
+ }
+ }
+
+retry:
+ remaining = timeout - jiffies;
+
+ if (time_after_eq(jiffies, timeout))
+ return -EBUSY;
+
+ hmm_range.notifier_seq = mmu_interval_read_begin(notifier);
+
+ if (map_dma &&
+ drm_gpusvm_pages_valid_unlocked(gpusvm, svm_pages, num_pages))
+ goto set_seqno;
+
+ pfns = kvmalloc_array(npages, sizeof(*pfns), GFP_KERNEL);
+ if (!pfns)
+ return -ENOMEM;
+
+ if (!mmget_not_zero(mm)) {
+ err = -EFAULT;
+ goto err_free;
+ }
+
+ hmm_range.hmm_pfns = pfns;
+ err = hmm_range_fault_unlocked_timeout(&hmm_range, remaining);
+ mmput(mm);
+ if (err)
+ goto err_free;
+
+ if (map_dma) {
+ for (p = 0; p < num_pages; ++p) {
+ if (drm_gpusvm_pages_first_dma(&svm_pages[p], NULL))
+ continue;
+ svm_pages[p].dma_addr =
+ kvzalloc_objs(*svm_pages[p].dma_addr, npages);
+ if (!svm_pages[p].dma_addr) {
+ err = -ENOMEM;
+ goto err_free;
+ }
+ }
+ }
+
+ /*
+ * Perform all dma mappings under the notifier lock to not
+ * access freed pages. A notifier will either block on
+ * the notifier lock or unmap dma.
+ */
+ drm_gpusvm_notifier_lock(gpusvm);
+
+ /*
+ * drm_gpusvm_range_set_unmapped() flags the whole array in one go under
+ * the write lock, so any instance answers for all of them here.
+ */
+ if (svm_pages[0].flags.unmapped) {
+ drm_gpusvm_notifier_unlock(gpusvm);
+ err = -EFAULT;
+ goto err_free;
+ }
+
+ if (mmu_interval_read_retry(notifier, hmm_range.notifier_seq)) {
+ drm_gpusvm_notifier_unlock(gpusvm);
+ kvfree(pfns);
+ goto retry;
+ }
+
+ if (!map_dma)
+ goto done_mapping;
+
+ for (p = 0; p < num_pages; ++p) {
+ if (drm_gpusvm_pages_valid(gpusvm, &svm_pages[p]))
+ continue;
+
+ err = drm_gpusvm_dma_map_pages(gpusvm, &svm_pages[p], pfns,
+ npages, ctx, dma_dir);
+ if (err) {
+ /*
+ * The failing instance was unwound by the helper. Keep
+ * the ones mapped earlier: the -EAGAIN retry reuses
+ * them, and the driver unmaps every instance with the
+ * range on the other error paths.
+ */
+ drm_gpusvm_notifier_unlock(gpusvm);
+ goto err_free;
+ }
+ }
+
+done_mapping:
drm_gpusvm_notifier_unlock(gpusvm);
+ kvfree(pfns);
+set_seqno:
+ for (p = 0; p < num_pages; ++p)
+ svm_pages[p].notifier_seq = hmm_range.notifier_seq;
+
+ return 0;
+
err_free:
kvfree(pfns);
if (err == -EAGAIN)
diff --git a/drivers/gpu/drm/xe/abi/xe_log_abi.h b/drivers/gpu/drm/xe/abi/xe_log_abi.h
index d6105520173e..526aadf85fe9 100644
--- a/drivers/gpu/drm/xe/abi/xe_log_abi.h
+++ b/drivers/gpu/drm/xe/abi/xe_log_abi.h
@@ -144,6 +144,7 @@ enum xe_log_location_bits {
define(DRIVER, 4, RTP, SW, "Register Table Processing") \
define(DRIVER, 5, WA, SW, "Workarounds") \
define(DRIVER, 6, PAGEFAULT, MEM_FAULT, "Page Fault") \
+ define(DRIVER, 7, GUCSUBMIT, GT_TDR, "GuC Submission") \
/* */ \
define(DRIVER_HARDWARE, 1, REGS, IO_BUS, "Registers") \
define(DRIVER_HARDWARE, 2, GGTT, IO_BUS, "Global GTT") \
diff --git a/drivers/gpu/drm/xe/display/xe_dsb_buffer.c b/drivers/gpu/drm/xe/display/xe_dsb_buffer.c
index a7158c73a14c..82974e933e8d 100644
--- a/drivers/gpu/drm/xe/display/xe_dsb_buffer.c
+++ b/drivers/gpu/drm/xe/display/xe_dsb_buffer.c
@@ -88,7 +88,7 @@ static void xe_dsb_buffer_flush_map(struct intel_dsb_buffer *dsb_buf)
* both for weak ordering archs and discrete cards.
*/
xe_device_wmb(xe);
- xe_device_l2_flush(xe);
+ xe_device_l2_flush(xe, false);
}
const struct intel_display_dsb_interface xe_display_dsb_interface = {
diff --git a/drivers/gpu/drm/xe/display/xe_fb_pin.c b/drivers/gpu/drm/xe/display/xe_fb_pin.c
index 73469ea5f333..b46a2c32ac07 100644
--- a/drivers/gpu/drm/xe/display/xe_fb_pin.c
+++ b/drivers/gpu/drm/xe/display/xe_fb_pin.c
@@ -203,7 +203,7 @@ static int __xe_pin_fb_vma_dpt(struct drm_gem_object *obj,
vma->node = dpt->ggtt_node[tile0->id];
/* Ensure DPT writes are flushed */
- xe_device_l2_flush(xe);
+ xe_device_l2_flush(xe, false);
return 0;
}
diff --git a/drivers/gpu/drm/xe/display/xe_panic.c b/drivers/gpu/drm/xe/display/xe_panic.c
index 12c6fb99015d..1a6cee25e9d7 100644
--- a/drivers/gpu/drm/xe/display/xe_panic.c
+++ b/drivers/gpu/drm/xe/display/xe_panic.c
@@ -52,7 +52,8 @@ static void xe_panic_page_set_pixel(struct drm_scanout_buffer *sb, unsigned int
if (new_page != panic->page) {
if (xe_bo_is_vram(bo)) {
/* Display is always mapped on root tile */
- struct xe_vram_region *vram = xe_bo_device(bo)->mem.vram;
+ struct xe_vram_region *vram =
+ xe_device_get_root_tile(xe_bo_device(bo))->mem.vram;
if (panic->page < 0 || new_page < panic->page) {
xe_res_first(bo->ttm.resource, new_page * PAGE_SIZE,
diff --git a/drivers/gpu/drm/xe/tests/xe_pci.c b/drivers/gpu/drm/xe/tests/xe_pci.c
index bb0393475524..5aefc03c00c7 100644
--- a/drivers/gpu/drm/xe/tests/xe_pci.c
+++ b/drivers/gpu/drm/xe/tests/xe_pci.c
@@ -311,10 +311,11 @@ const void *xe_pci_id_gen_param(struct kunit *test, const void *prev, char *desc
EXPORT_SYMBOL_IF_KUNIT(xe_pci_id_gen_param);
static int fake_probe_info(struct xe_device *xe,
- const struct xe_device_desc *desc,
struct xe_pci_fake_data *data,
struct xe_probed_info *probed_info)
{
+ const struct xe_device_desc *desc = xe->desc;
+
probed_info->tile_count = 1 + desc->max_remote_tiles;
if (!data || desc->pre_gmdid_graphics_ip) {
@@ -351,7 +352,7 @@ int xe_pci_fake_device_init(struct xe_device *xe)
if (!data) {
desc = (const void *)ent->driver_data;
- subplatform_desc = NULL;
+ subplatform_desc = desc->subplatforms;
goto done;
}
@@ -364,25 +365,37 @@ int xe_pci_fake_device_init(struct xe_device *xe)
if (!ent->device)
return -ENODEV;
+ if (data->subplatform == XE_SUBPLATFORM_NONE) {
+ subplatform_desc = NULL;
+ goto done;
+ }
+
+ if (data->subplatform == XE_SUBPLATFORM_UNINITIALIZED) {
+ subplatform_desc = desc->subplatforms;
+ goto done;
+ }
+
for (subplatform_desc = desc->subplatforms;
subplatform_desc && subplatform_desc->subplatform;
subplatform_desc++)
if (subplatform_desc->subplatform == data->subplatform)
break;
- if (data->subplatform != XE_SUBPLATFORM_NONE && !subplatform_desc)
+ if (!subplatform_desc || !subplatform_desc->subplatform)
return -ENODEV;
done:
+ xe->desc = desc;
+ xe->subplatform_desc = subplatform_desc;
xe->sriov.__mode = data && data->sriov_mode ?
data->sriov_mode : XE_SRIOV_MODE_NONE;
- err = fake_probe_info(xe, desc, data, &probed_info);
+ err = fake_probe_info(xe, data, &probed_info);
if (err)
return err;
- xe_info_init_early(xe, desc, subplatform_desc, &probed_info);
- xe_info_init(xe, desc, &probed_info);
+ xe_info_init_early(xe, &probed_info);
+ xe_info_init(xe, &probed_info);
return 0;
}
diff --git a/drivers/gpu/drm/xe/xe_bo.c b/drivers/gpu/drm/xe/xe_bo.c
index dde309821237..9f3f0cb95afa 100644
--- a/drivers/gpu/drm/xe/xe_bo.c
+++ b/drivers/gpu/drm/xe/xe_bo.c
@@ -28,6 +28,7 @@
#include "xe_ggtt.h"
#include "xe_map.h"
#include "xe_migrate.h"
+#include "xe_mmio_gem.h"
#include "xe_pat.h"
#include "xe_pm.h"
#include "xe_preempt_fence.h"
@@ -104,13 +105,16 @@ static bool resource_is_vram(struct ttm_resource *res)
bool xe_bo_is_vram(struct xe_bo *bo)
{
- return resource_is_vram(bo->ttm.resource) ||
- resource_is_stolen_vram(xe_bo_device(bo), bo->ttm.resource);
+ struct ttm_resource *res = bo->ttm.resource;
+
+ return res && (resource_is_vram(res) || resource_is_stolen_vram(xe_bo_device(bo), res));
}
bool xe_bo_is_stolen(struct xe_bo *bo)
{
- return bo->ttm.resource->mem_type == XE_PL_STOLEN;
+ struct ttm_resource *res = bo->ttm.resource;
+
+ return res && res->mem_type == XE_PL_STOLEN;
}
/**
@@ -158,7 +162,13 @@ bool xe_bo_is_vm_bound(struct xe_bo *bo)
return !list_empty(&bo->ttm.base.gpuva.list);
}
-static bool xe_bo_is_user(struct xe_bo *bo)
+/**
+ * xe_bo_is_user - Check if BO is user-created
+ * @bo: The BO
+ *
+ * Returns: true if @bo was created by userspace
+ */
+bool xe_bo_is_user(struct xe_bo *bo)
{
return bo->flags & XE_BO_FLAG_USER;
}
@@ -921,7 +931,7 @@ void xe_bo_set_purgeable_state(struct xe_bo *bo,
*
* Return: 0 on success, negative error code on failure
*/
-static int xe_ttm_bo_purge(struct ttm_buffer_object *ttm_bo, struct ttm_operation_ctx *ctx)
+int xe_ttm_bo_purge(struct ttm_buffer_object *ttm_bo, struct ttm_operation_ctx *ctx)
{
struct xe_bo *bo = ttm_to_xe_bo(ttm_bo);
struct ttm_placement place = {};
@@ -929,9 +939,6 @@ static int xe_ttm_bo_purge(struct ttm_buffer_object *ttm_bo, struct ttm_operatio
xe_bo_assert_held(bo);
- if (!ttm_bo->ttm)
- return 0;
-
if (!xe_bo_madv_is_dontneed(bo))
return 0;
@@ -3261,6 +3268,9 @@ void xe_bo_unpin(struct xe_bo *bo)
struct ttm_place *place = &bo->placements[0];
struct xe_device *xe = xe_bo_device(bo);
+ if (xe_bo_is_purged(bo))
+ return;
+
xe_assert(xe, !bo->ttm.base.import_attach);
xe_assert(xe, xe_bo_is_pinned(bo));
@@ -3655,6 +3665,39 @@ out_vm:
return err;
}
+static int xe_gem_pci_barrier_mmap_offset(struct xe_device *xe, struct drm_file *file,
+ struct drm_xe_gem_mmap_offset *args)
+{
+ struct xe_file *xef = file->driver_priv;
+ struct xe_mmio_gem **barrier = &xef->mmio_gem.pci_barrier;
+
+ if (XE_IOCTL_DBG(xe, !IS_DGFX(xe)))
+ return -EINVAL;
+
+ if (XE_IOCTL_DBG(xe, args->handle))
+ return -EINVAL;
+
+ scoped_guard(mutex, &xef->mmio_gem.lock) {
+ if (!*barrier) {
+ phys_addr_t phys_addr;
+
+#define LAST_DB_PAGE_OFFSET 0x7ff000
+ phys_addr = pci_resource_start(to_pci_dev(xe->drm.dev), 0) +
+ LAST_DB_PAGE_OFFSET;
+ *barrier = xe_mmio_gem_create(xe, file, phys_addr, SZ_4K);
+ if (IS_ERR(*barrier)) {
+ int err = PTR_ERR(*barrier);
+
+ *barrier = NULL;
+ return err;
+ }
+ }
+
+ args->offset = xe_mmio_gem_mmap_offset(*barrier);
+ }
+ return 0;
+}
+
int xe_gem_mmap_offset_ioctl(struct drm_device *dev, void *data,
struct drm_file *file)
{
@@ -3670,21 +3713,8 @@ int xe_gem_mmap_offset_ioctl(struct drm_device *dev, void *data,
~DRM_XE_MMAP_OFFSET_FLAG_PCI_BARRIER))
return -EINVAL;
- if (args->flags & DRM_XE_MMAP_OFFSET_FLAG_PCI_BARRIER) {
- if (XE_IOCTL_DBG(xe, !IS_DGFX(xe)))
- return -EINVAL;
-
- if (XE_IOCTL_DBG(xe, args->handle))
- return -EINVAL;
-
- if (XE_IOCTL_DBG(xe, PAGE_SIZE > SZ_4K))
- return -EINVAL;
-
- BUILD_BUG_ON(((XE_PCI_BARRIER_MMAP_OFFSET >> XE_PTE_SHIFT) +
- SZ_4K) >= DRM_FILE_PAGE_OFFSET_START);
- args->offset = XE_PCI_BARRIER_MMAP_OFFSET;
- return 0;
- }
+ if (args->flags & DRM_XE_MMAP_OFFSET_FLAG_PCI_BARRIER)
+ return xe_gem_pci_barrier_mmap_offset(xe, file, args);
gem_obj = drm_gem_object_lookup(file, args->handle);
if (XE_IOCTL_DBG(xe, !gem_obj))
diff --git a/drivers/gpu/drm/xe/xe_bo.h b/drivers/gpu/drm/xe/xe_bo.h
index e8081af5bfc1..290ca624e2a7 100644
--- a/drivers/gpu/drm/xe/xe_bo.h
+++ b/drivers/gpu/drm/xe/xe_bo.h
@@ -87,7 +87,6 @@
#define XE_BO_PROPS_INVALID (-1)
-#define XE_PCI_BARRIER_MMAP_OFFSET (0x50 << XE_PTE_SHIFT)
/**
* enum xe_madv_purgeable_state - Buffer object purgeable state enumeration
@@ -600,6 +599,8 @@ struct xe_bo_shrink_flags {
long xe_bo_shrink(struct ttm_operation_ctx *ctx, struct ttm_buffer_object *bo,
const struct xe_bo_shrink_flags flags,
unsigned long *scanned);
+int xe_ttm_bo_purge(struct ttm_buffer_object *ttm_bo, struct ttm_operation_ctx *ctx);
+bool xe_bo_is_user(struct xe_bo *bo);
/**
* xe_bo_is_mem_type - Whether the bo currently resides in the given
diff --git a/drivers/gpu/drm/xe/xe_bo_types.h b/drivers/gpu/drm/xe/xe_bo_types.h
index e45f24301050..0eb93052c1d5 100644
--- a/drivers/gpu/drm/xe/xe_bo_types.h
+++ b/drivers/gpu/drm/xe/xe_bo_types.h
@@ -20,6 +20,7 @@
struct xe_device;
struct xe_mem_pool_node;
struct xe_vm;
+struct xe_exec_queue;
#define XE_BO_MAX_PLACEMENTS 3
@@ -42,6 +43,13 @@ struct xe_bo {
u32 flags;
/** @vm: VM this BO is attached to, for extobj this will be NULL */
struct xe_vm *vm;
+ /**
+ * @q: Queue this BO is attached to, mostly for LRC BO, NULL otherwise.
+ * Protected by the BO dma_resv: readers must hold it across both the
+ * read and xe_exec_queue_get_unless_zero(). The BO holds no reference
+ * on the queue.
+ */
+ struct xe_exec_queue *q;
/** @tile: Tile this BO is attached to (kernel BO only) */
struct xe_tile *tile;
/** @placements: valid placements for this BO */
diff --git a/drivers/gpu/drm/xe/xe_configfs.c b/drivers/gpu/drm/xe/xe_configfs.c
index 052cce962161..f5c828cf7e8f 100644
--- a/drivers/gpu/drm/xe/xe_configfs.c
+++ b/drivers/gpu/drm/xe/xe_configfs.c
@@ -61,7 +61,8 @@
* ├── survivability_mode
* ├── gt_types_allowed
* ├── engines_allowed
- * └── enable_psmi
+ * ├── enable_psmi
+ * └── disable_vram_page_offline
*
* After configuring the attributes as per next section, the device can be
* probed with::
@@ -159,6 +160,18 @@
*
* This attribute can only be set before binding to the device.
*
+ * Disable VRAM page offline:
+ * ----------------------------
+ *
+ * 0, n, N, false - Do not disable (Offlining is active - default)
+ * 1, y, Y, true - Disable vram page offline (Logging only)
+ *
+ * Example to disable VRAM offline::
+ *
+ * # echo 1 > /sys/kernel/config/xe/0000:03:00.0/disable_vram_page_offline
+ *
+ * This attribute can only be set on CRI before binding to the device.
+ *
* Context restore BB
* ------------------
*
@@ -275,6 +288,7 @@ struct xe_config_group_device {
bool survivability_mode;
bool enable_psmi;
bool enable_multi_queue;
+ bool disable_vram_page_offline;
struct {
unsigned int max_vfs;
bool admin_only_pf;
@@ -295,6 +309,7 @@ static const struct xe_config_device device_defaults = {
.survivability_mode = false,
.enable_psmi = false,
.enable_multi_queue = true,
+ .disable_vram_page_offline = false,
.sriov = {
.max_vfs = XE_DEFAULT_MAX_VFS,
.admin_only_pf = XE_DEFAULT_ADMIN_ONLY_PF,
@@ -616,6 +631,33 @@ static ssize_t enable_multi_queue_store(struct config_item *item, const char *pa
return len;
}
+static ssize_t disable_vram_page_offline_show(struct config_item *item, char *page)
+{
+ struct xe_config_device *dev = to_xe_config_device(item);
+
+ return sprintf(page, "%s\n", str_yes_no(dev->disable_vram_page_offline));
+}
+
+static ssize_t disable_vram_page_offline_store(struct config_item *item,
+ const char *page, size_t len)
+{
+ struct xe_config_group_device *dev = to_xe_config_group_device(item);
+ bool val;
+ int ret;
+
+ ret = kstrtobool(page, &val);
+ if (ret)
+ return ret;
+
+ guard(mutex)(&dev->lock);
+ if (is_bound(dev))
+ return -EBUSY;
+
+ dev->config.disable_vram_page_offline = val;
+
+ return len;
+}
+
static bool wa_bb_read_advance(bool dereference, char **p,
const char *append, size_t len,
size_t *max_size)
@@ -855,6 +897,7 @@ CONFIGFS_ATTR(, ctx_restore_mid_bb);
CONFIGFS_ATTR(, ctx_restore_post_bb);
CONFIGFS_ATTR(, enable_multi_queue);
CONFIGFS_ATTR(, enable_psmi);
+CONFIGFS_ATTR(, disable_vram_page_offline);
CONFIGFS_ATTR(, engines_allowed);
CONFIGFS_ATTR(, gt_types_allowed);
CONFIGFS_ATTR(, survivability_mode);
@@ -864,6 +907,7 @@ static struct configfs_attribute *xe_config_device_attrs[] = {
&attr_ctx_restore_post_bb,
&attr_enable_multi_queue,
&attr_enable_psmi,
+ &attr_disable_vram_page_offline,
&attr_engines_allowed,
&attr_gt_types_allowed,
&attr_survivability_mode,
@@ -895,6 +939,11 @@ static bool xe_config_device_is_visible(struct config_item *item,
return false;
}
+ if (attr == &attr_disable_vram_page_offline) {
+ if (!dev->desc->is_dgfx || dev->desc->platform != XE_CRESCENTISLAND)
+ return false;
+ }
+
return true;
}
@@ -1142,6 +1191,7 @@ static void dump_custom_dev_config(struct pci_dev *pdev,
PRI_CUSTOM_ATTR("%llx", engines_allowed);
PRI_CUSTOM_ATTR("%d", enable_multi_queue);
PRI_CUSTOM_ATTR("%d", enable_psmi);
+ PRI_CUSTOM_ATTR("%d", disable_vram_page_offline);
PRI_CUSTOM_ATTR("%d", survivability_mode);
PRI_CUSTOM_ATTR("%u", sriov.admin_only_pf);
@@ -1291,6 +1341,26 @@ bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev)
}
/**
+ * xe_configfs_get_disable_vram_page_offline - get configfs disable_vram_page_offline setting
+ * @pdev: pci device
+ *
+ * Return: disable_vram_page_offline setting in configfs
+ */
+bool xe_configfs_get_disable_vram_page_offline(struct pci_dev *pdev)
+{
+ struct xe_config_group_device *dev = find_xe_config_group_device(pdev);
+ bool ret;
+
+ if (!dev)
+ return device_defaults.disable_vram_page_offline;
+
+ ret = dev->config.disable_vram_page_offline;
+ config_group_put(&dev->group);
+
+ return ret;
+}
+
+/**
* xe_configfs_get_ctx_restore_mid_bb - get configfs ctx_restore_mid_bb setting
* @pdev: pci device
* @class: hw engine class
diff --git a/drivers/gpu/drm/xe/xe_configfs.h b/drivers/gpu/drm/xe/xe_configfs.h
index 4fbbeafba473..42cd1a491d01 100644
--- a/drivers/gpu/drm/xe/xe_configfs.h
+++ b/drivers/gpu/drm/xe/xe_configfs.h
@@ -24,6 +24,7 @@ bool xe_configfs_media_gt_allowed(struct pci_dev *pdev);
u64 xe_configfs_get_engines_allowed(struct pci_dev *pdev);
bool xe_configfs_get_psmi_enabled(struct pci_dev *pdev);
bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev);
+bool xe_configfs_get_disable_vram_page_offline(struct pci_dev *pdev);
u32 xe_configfs_get_ctx_restore_mid_bb(struct pci_dev *pdev,
enum xe_engine_class class,
const u32 **cs);
@@ -44,6 +45,7 @@ static inline bool xe_configfs_media_gt_allowed(struct pci_dev *pdev) { return t
static inline u64 xe_configfs_get_engines_allowed(struct pci_dev *pdev) { return U64_MAX; }
static inline bool xe_configfs_get_psmi_enabled(struct pci_dev *pdev) { return false; }
static inline bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev) { return true; }
+static inline bool xe_configfs_get_disable_vram_page_offline(struct pci_dev *pdev) { return false; }
static inline u32 xe_configfs_get_ctx_restore_mid_bb(struct pci_dev *pdev,
enum xe_engine_class class,
const u32 **cs) { return 0; }
diff --git a/drivers/gpu/drm/xe/xe_debugfs.c b/drivers/gpu/drm/xe/xe_debugfs.c
index 28135f84e286..80f62634fae5 100644
--- a/drivers/gpu/drm/xe/xe_debugfs.c
+++ b/drivers/gpu/drm/xe/xe_debugfs.c
@@ -32,6 +32,7 @@
#include "xe_sriov_vf.h"
#include "xe_step.h"
#include "xe_tile_debugfs.h"
+#include "xe_ttm_vram_mgr.h"
#include "xe_vsec.h"
#include "xe_wa.h"
@@ -44,12 +45,18 @@
DECLARE_FAULT_ATTR(gt_reset_failure);
DECLARE_FAULT_ATTR(inject_csc_hw_error);
DECLARE_FAULT_ATTR(wedge_cold_reset);
+DECLARE_FAULT_ATTR(inject_mempage_offline);
static bool csc_hw_error_available(struct xe_device *xe)
{
return !IS_SRIOV_VF(xe) && xe->info.platform == XE_BATTLEMAGE;
}
+static bool is_crescent_island_pf(struct xe_device *xe)
+{
+ return !IS_SRIOV_VF(xe) && xe->info.platform == XE_CRESCENTISLAND;
+}
+
/*
* Fault injection table. Each entry registers a debugfs attribute; add a
* matching FAULT_ACTION() below for every entry added here.
@@ -66,6 +73,9 @@ static struct {
.is_visible = csc_hw_error_available },
{ .name = "wedge_cold_reset",
.attr = &wedge_cold_reset },
+ { .name = "inject_mempage_offline",
+ .attr = &inject_mempage_offline,
+ .is_visible = is_crescent_island_pf },
};
/*
@@ -81,6 +91,40 @@ bool xe_fault_##name(void) \
FAULT_ACTION(gt_reset, gt_reset_failure)
FAULT_ACTION(csc_hw_error, inject_csc_hw_error)
FAULT_ACTION(wedge_cold_reset, wedge_cold_reset)
+FAULT_ACTION(mempage_offline, inject_mempage_offline)
+
+static ssize_t inject_mempage_offline_trigger(struct file *f,
+ const char __user *ubuf,
+ size_t size, loff_t *pos)
+{
+ struct xe_device *xe = file_inode(f)->i_private;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_vram_region *vr = tile->mem.vram;
+ u64 pfn;
+ int ret;
+
+ if (!vr)
+ return -ENODEV;
+
+ ret = kstrtou64_from_user(ubuf, size, 0, &pfn);
+ if (ret)
+ return ret;
+
+ if (!xe_fault_mempage_offline())
+ return size;
+
+ xe_warn(xe, "Page offlining test interface accessed. Notice: Offlined or reserved memory pages cannot be reclaimed dynamically. A driver rebind (unbind and bind loop) is required post-test to clean up.\n");
+ if (pfn == 0)
+ return xe_ttm_vram_inject_fault(xe) ?: size;
+
+ /* User provided PFN - convert to DPA and inject */
+ return xe_ttm_vram_handle_addr_fault(xe, pfn << PAGE_SHIFT) ?: size;
+}
+
+static const struct file_operations inject_mempage_offline_fops = {
+ .owner = THIS_MODULE,
+ .write = inject_mempage_offline_trigger,
+};
static void xe_fault_inject_debugfs_register(struct xe_device *xe,
struct dentry *root)
@@ -95,6 +139,11 @@ static void xe_fault_inject_debugfs_register(struct xe_device *xe,
fault_create_debugfs_attr(xe_fault_inject_entry[i].name, root,
xe_fault_inject_entry[i].attr);
}
+
+ if (is_crescent_island_pf(xe)) {
+ debugfs_create_file("inject_mempage_offline_trigger", 0200,
+ root, xe, &inject_mempage_offline_fops);
+ }
}
static void read_residency_counter(struct xe_device *xe, struct xe_mmio *mmio,
@@ -773,6 +822,8 @@ void xe_debugfs_register(struct xe_device *xe)
if (man)
ttm_resource_manager_create_debugfs(man, root, "stolen_mm");
+ xe_ttm_vram_debugfs_init(xe, root);
+
for_each_tile(tile, xe, tile_id)
xe_tile_debugfs_register(tile);
diff --git a/drivers/gpu/drm/xe/xe_debugfs.h b/drivers/gpu/drm/xe/xe_debugfs.h
index 0dcd28fd7dc0..88d91c78036b 100644
--- a/drivers/gpu/drm/xe/xe_debugfs.h
+++ b/drivers/gpu/drm/xe/xe_debugfs.h
@@ -14,11 +14,13 @@ struct xe_device;
bool xe_fault_gt_reset(void);
bool xe_fault_csc_hw_error(void);
bool xe_fault_wedge_cold_reset(void);
+bool xe_fault_mempage_offline(void);
void xe_debugfs_register(struct xe_device *xe);
#else
static inline bool xe_fault_gt_reset(void) { return false; }
static inline bool xe_fault_csc_hw_error(void) { return false; }
static inline bool xe_fault_wedge_cold_reset(void) { return false; }
+static inline bool xe_fault_mempage_offline(void) { return false; }
static inline void xe_debugfs_register(struct xe_device *xe) { }
#endif
diff --git a/drivers/gpu/drm/xe/xe_device.c b/drivers/gpu/drm/xe/xe_device.c
index 396d02eb2af8..205cb4e7f9e8 100644
--- a/drivers/gpu/drm/xe/xe_device.c
+++ b/drivers/gpu/drm/xe/xe_device.c
@@ -50,6 +50,7 @@
#include "xe_late_bind_fw.h"
#include "xe_log.h"
#include "xe_mmio.h"
+#include "xe_mmio_gem.h"
#include "xe_module.h"
#include "xe_nvm.h"
#include "xe_oa.h"
@@ -111,6 +112,8 @@ static int xe_file_open(struct drm_device *dev, struct drm_file *file)
mutex_init(&xef->exec_queue.lock);
xa_init_flags(&xef->exec_queue.xa, XA_FLAGS_ALLOC1);
+ mutex_init(&xef->mmio_gem.lock);
+
file->driver_priv = xef;
kref_init(&xef->refcount);
@@ -133,6 +136,8 @@ static void xe_file_destroy(struct kref *ref)
xa_destroy(&xef->vm.xa);
mutex_destroy(&xef->vm.lock);
+ mutex_destroy(&xef->mmio_gem.lock);
+
xe_drm_client_put(xef->client);
kfree(xef->process_name);
kfree(xef);
@@ -189,6 +194,13 @@ static void xe_file_close(struct drm_device *dev, struct drm_file *file)
xa_for_each(&xef->vm.xa, idx, vm)
xe_vm_close_and_put(vm);
+ scoped_guard(mutex, &xef->mmio_gem.lock) {
+ if (xef->mmio_gem.pci_barrier) {
+ xe_mmio_gem_destroy(xef->mmio_gem.pci_barrier, file);
+ xef->mmio_gem.pci_barrier = NULL;
+ }
+ }
+
xe_file_put(xef);
}
@@ -258,95 +270,6 @@ static long xe_drm_compat_ioctl(struct file *file, unsigned int cmd, unsigned lo
#define xe_drm_compat_ioctl NULL
#endif
-static void barrier_open(struct vm_area_struct *vma)
-{
- drm_dev_get(vma->vm_private_data);
-}
-
-static void barrier_close(struct vm_area_struct *vma)
-{
- drm_dev_put(vma->vm_private_data);
-}
-
-static void barrier_release_dummy_page(struct drm_device *dev, void *res)
-{
- struct page *dummy_page = (struct page *)res;
-
- __free_page(dummy_page);
-}
-
-static vm_fault_t barrier_fault(struct vm_fault *vmf)
-{
- struct drm_device *dev = vmf->vma->vm_private_data;
- struct vm_area_struct *vma = vmf->vma;
- vm_fault_t ret = VM_FAULT_NOPAGE;
- pgprot_t prot;
- int idx;
-
- prot = vma_get_page_prot(vma);
-
- if (drm_dev_enter(dev, &idx)) {
- unsigned long pfn;
-
-#define LAST_DB_PAGE_OFFSET 0x7ff001
- pfn = PHYS_PFN(pci_resource_start(to_pci_dev(dev->dev), 0) +
- LAST_DB_PAGE_OFFSET);
- ret = vmf_insert_pfn_prot(vma, vma->vm_start, pfn,
- pgprot_noncached(prot));
- drm_dev_exit(idx);
- } else {
- struct page *page;
-
- /* Allocate new dummy page to map all the VA range in this VMA to it*/
- page = alloc_page(GFP_KERNEL | __GFP_ZERO);
- if (!page)
- return VM_FAULT_OOM;
-
- /* Set the page to be freed using drmm release action */
- if (drmm_add_action_or_reset(dev, barrier_release_dummy_page, page))
- return VM_FAULT_OOM;
-
- ret = vmf_insert_pfn_prot(vma, vma->vm_start, page_to_pfn(page),
- prot);
- }
-
- return ret;
-}
-
-static const struct vm_operations_struct vm_ops_barrier = {
- .open = barrier_open,
- .close = barrier_close,
- .fault = barrier_fault,
-};
-
-static int xe_pci_barrier_mmap(struct file *filp,
- struct vm_area_struct *vma)
-{
- struct drm_file *priv = filp->private_data;
- struct drm_device *dev = priv->minor->dev;
- struct xe_device *xe = to_xe_device(dev);
-
- if (!IS_DGFX(xe))
- return -EINVAL;
-
- if (vma->vm_end - vma->vm_start > SZ_4K)
- return -EINVAL;
-
- if (vma_is_cow_mapping(vma))
- return -EINVAL;
-
- if (vma->vm_flags & (VM_READ | VM_EXEC))
- return -EINVAL;
-
- vm_flags_clear(vma, VM_MAYREAD | VM_MAYEXEC);
- vm_flags_set(vma, VM_PFNMAP | VM_DONTEXPAND | VM_DONTDUMP | VM_IO);
- vma->vm_ops = &vm_ops_barrier;
- vma->vm_private_data = dev;
- drm_dev_get(vma->vm_private_data);
-
- return 0;
-}
-
static int xe_mmap(struct file *filp, struct vm_area_struct *vma)
{
struct drm_file *priv = filp->private_data;
@@ -355,11 +278,6 @@ static int xe_mmap(struct file *filp, struct vm_area_struct *vma)
if (drm_dev_is_unplugged(dev))
return -ENODEV;
- switch (vma->vm_pgoff) {
- case XE_PCI_BARRIER_MMAP_OFFSET >> XE_PTE_SHIFT:
- return xe_pci_barrier_mmap(filp, vma);
- }
-
return drm_gem_mmap(filp, vma);
}
@@ -1051,6 +969,10 @@ int xe_device_probe(struct xe_device *xe)
if (err)
return err;
+ err = xe_vram_reserve_memtest_bo(xe);
+ if (err)
+ return err;
+
for_each_tile(tile, xe, id) {
err = xe_tile_init(tile);
if (err)
@@ -1067,6 +989,10 @@ int xe_device_probe(struct xe_device *xe)
return err;
}
+ err = xe_vram_memtest(xe);
+ if (err)
+ return err;
+
err = xe_pagefault_init(xe);
if (err)
return err;
@@ -1270,7 +1196,7 @@ bool xe_device_is_l2_flush_optimized(struct xe_device *xe)
return false;
}
-void xe_device_l2_flush(struct xe_device *xe)
+void xe_device_l2_flush(struct xe_device *xe, bool force)
{
struct xe_gt *gt;
@@ -1278,7 +1204,7 @@ void xe_device_l2_flush(struct xe_device *xe)
if (!gt)
return;
- if (!XE_GT_WA(gt, 16023588340))
+ if (!force && !XE_GT_WA(gt, 16023588340))
return;
CLASS(xe_force_wake, fw_ref)(gt_to_fw(gt), XE_FW_GT);
@@ -1333,7 +1259,7 @@ void xe_device_td_flush(struct xe_device *xe)
if (XE_GT_WA(root_gt, 16023588340)) {
/* A transient flush is not sufficient: flush the L2 */
- xe_device_l2_flush(xe);
+ xe_device_l2_flush(xe, false);
} else {
xe_guc_pc_apply_flush_freq_limit(&root_gt->uc.guc.pc);
tdf_request_sync(xe);
diff --git a/drivers/gpu/drm/xe/xe_device.h b/drivers/gpu/drm/xe/xe_device.h
index 6c4cfaebc44a..6d3d6d5eba29 100644
--- a/drivers/gpu/drm/xe/xe_device.h
+++ b/drivers/gpu/drm/xe/xe_device.h
@@ -205,7 +205,7 @@ u64 xe_device_uncanonicalize_addr(struct xe_device *xe, u64 address);
bool xe_device_is_l2_flush_optimized(struct xe_device *xe);
void xe_device_td_flush(struct xe_device *xe);
-void xe_device_l2_flush(struct xe_device *xe);
+void xe_device_l2_flush(struct xe_device *xe, bool force);
static inline bool xe_device_wedged(struct xe_device *xe)
{
diff --git a/drivers/gpu/drm/xe/xe_device_types.h b/drivers/gpu/drm/xe/xe_device_types.h
index 180d450a6deb..4661bfce2f4e 100644
--- a/drivers/gpu/drm/xe/xe_device_types.h
+++ b/drivers/gpu/drm/xe/xe_device_types.h
@@ -40,6 +40,7 @@ struct intel_display;
struct intel_dg_nvm_dev;
struct xe_ggtt;
struct xe_i2c;
+struct xe_mmio_gem;
struct xe_pat_ops;
struct xe_pxp;
struct xe_ttm_stolen_mgr;
@@ -116,6 +117,12 @@ struct xe_device {
/** @devcoredump: device coredump */
struct xe_devcoredump devcoredump;
+ /** @desc: device descriptor */
+ const struct xe_device_desc *desc;
+
+ /** @subplatform_desc: subplatform descriptor */
+ const struct xe_subplatform_desc *subplatform_desc;
+
/** @info: device info */
struct intel_device_info {
/** @info.platform_name: platform name */
@@ -676,6 +683,18 @@ struct xe_file {
/** @refcount: ref count of this xe file */
struct kref refcount;
+
+ /** @mmio_gem: MMIO GEM objects for this xe file */
+ struct {
+ /**
+ * @mmio_gem.lock: Protects allocation and attach of MMIO
+ * GEM objects on first use (singleton). All MMIO GEM access
+ * should be guarded by this lock. Prefer scoped_guard().
+ */
+ struct mutex lock;
+ /** @mmio_gem.pci_barrier: MMIO GEM object for PCI barrier mmap. */
+ struct xe_mmio_gem *pci_barrier;
+ } mmio_gem;
};
#endif
diff --git a/drivers/gpu/drm/xe/xe_dma_buf.c b/drivers/gpu/drm/xe/xe_dma_buf.c
index bf0728838ead..5d9f1cd24b7f 100644
--- a/drivers/gpu/drm/xe/xe_dma_buf.c
+++ b/drivers/gpu/drm/xe/xe_dma_buf.c
@@ -104,6 +104,9 @@ static struct sg_table *xe_dma_buf_map(struct dma_buf_attachment *attach,
struct sg_table *sgt;
int r = 0;
+ if (xe_bo_is_purged(bo))
+ return ERR_PTR(-ENOENT);
+
if (!attach->peer2peer && !xe_bo_can_migrate(bo, XE_PL_TT))
return ERR_PTR(-EOPNOTSUPP);
diff --git a/drivers/gpu/drm/xe/xe_drm_ras_types.h b/drivers/gpu/drm/xe/xe_drm_ras_types.h
index 8d729ad6a264..0be218ba2db7 100644
--- a/drivers/gpu/drm/xe/xe_drm_ras_types.h
+++ b/drivers/gpu/drm/xe/xe_drm_ras_types.h
@@ -43,6 +43,9 @@ struct xe_drm_ras {
/** @info: info array for all types of errors */
struct xe_drm_ras_counter *info[DRM_XE_RAS_ERR_SEV_MAX];
+
+ /** @disable_vram_page_offline: cached configfs policy, immutable after init */
+ bool disable_vram_page_offline;
};
#endif
diff --git a/drivers/gpu/drm/xe/xe_exec_queue.c b/drivers/gpu/drm/xe/xe_exec_queue.c
index c4213bb9c137..e63559a2f582 100644
--- a/drivers/gpu/drm/xe/xe_exec_queue.c
+++ b/drivers/gpu/drm/xe/xe_exec_queue.c
@@ -322,10 +322,66 @@ struct xe_lrc *xe_exec_queue_lrc(struct xe_exec_queue *q)
return q->lrc[0];
}
+/*
+ * Publish the queue back-pointer in the LRC BOs.
+ *
+ * The BO holds no reference on the queue; the queue owns the LRCs, and
+ * therefore the BOs, instead. The back-pointer is made safe by two rules:
+ *
+ * - It is published only once the queue is fully constructed and can no
+ * longer be destroyed by an error path that bypasses the kref (see
+ * xe_exec_queue_create()), so a reader that successfully takes a
+ * reference can never be handed a queue that is freed without going
+ * through xe_exec_queue_destroy().
+ *
+ * - It is written and cleared under the BO dma_resv. Readers must hold
+ * the same lock across both the read and
+ * xe_exec_queue_get_unless_zero(), which serializes them against
+ * xe_exec_queue_clear_lrc_bo_backpointer() below.
+ *
+ * For a multi-queue group the LRC BOs point at the primary queue, which is
+ * kept alive by the reference every secondary holds on it.
+ */
+static void xe_exec_queue_set_lrc_bo_backpointer(struct xe_exec_queue *q)
+{
+ struct xe_exec_queue *primary = xe_exec_queue_multi_queue_primary(q);
+ int i;
+
+ for (i = 0; i < q->width; ++i) {
+ struct xe_bo *bo = q->lrc[i]->bo;
+
+ xe_bo_lock(bo, false);
+ bo->q = primary;
+ xe_bo_unlock(bo);
+ }
+}
+
+/*
+ * Drop the queue back-pointer before anything belonging to the queue is
+ * torn down. This must happen before q->ops->fini(), otherwise a reader
+ * could take a reference and then operate on an already destroyed backend.
+ */
+static void xe_exec_queue_clear_lrc_bo_backpointer(struct xe_exec_queue *q)
+{
+ int i;
+
+ for (i = 0; i < q->width; ++i) {
+ struct xe_bo *bo = q->lrc[i] ? q->lrc[i]->bo : NULL;
+
+ if (!bo)
+ continue;
+
+ xe_bo_lock(bo, false);
+ bo->q = NULL;
+ xe_bo_unlock(bo);
+ }
+}
+
static void __xe_exec_queue_fini(struct xe_exec_queue *q)
{
int i;
+ xe_exec_queue_clear_lrc_bo_backpointer(q);
q->ops->fini(q);
for (i = 0; i < q->width; ++i)
@@ -450,6 +506,14 @@ struct xe_exec_queue *xe_exec_queue_create(struct xe_device *xe, struct xe_vm *v
goto err_post_init;
}
+ /*
+ * Publish the LRC BO back-pointers last: past this point the queue can
+ * only be destroyed through xe_exec_queue_destroy(), so a concurrent
+ * reader that takes a reference via bo->q cannot race with the
+ * kref-bypassing error paths below.
+ */
+ xe_exec_queue_set_lrc_bo_backpointer(q);
+
return q;
err_post_init:
@@ -1566,8 +1630,12 @@ void xe_exec_queue_update_run_ticks(struct xe_exec_queue *q)
* errors.
*/
lrc = q->lrc[0];
- new_ts = xe_lrc_update_timestamp(lrc, &old_ts);
- q->xef->run_ticks[q->class] += (new_ts - old_ts) * q->width;
+ xe_bo_lock(lrc->bo, false);
+ if (!xe_bo_is_purged(lrc->bo)) {
+ new_ts = xe_lrc_update_timestamp(lrc, &old_ts);
+ q->xef->run_ticks[q->class] += (new_ts - old_ts) * q->width;
+ }
+ xe_bo_unlock(lrc->bo);
drm_dev_exit(idx);
}
diff --git a/drivers/gpu/drm/xe/xe_exec_queue_types.h b/drivers/gpu/drm/xe/xe_exec_queue_types.h
index 95f75d61a647..836f88fc0faa 100644
--- a/drivers/gpu/drm/xe/xe_exec_queue_types.h
+++ b/drivers/gpu/drm/xe/xe_exec_queue_types.h
@@ -154,6 +154,9 @@ struct xe_exec_queue {
*/
unsigned long flags;
+ /** @ban_reason: Bitmask of ban reasons (DRM_XE_EXEC_QUEUE_BAN_REASON_*) */
+ atomic_t ban_reason;
+
union {
/** @multi_gt_list: list head for VM bind engines if multi-GT */
struct list_head multi_gt_list;
@@ -348,8 +351,8 @@ struct xe_exec_queue_ops {
* signalled when this function is called.
*/
void (*resume)(struct xe_exec_queue *q);
- /** @reset_status: check exec queue reset status */
- bool (*reset_status)(struct xe_exec_queue *q);
+ /** @reset_status: check exec queue ban status, returns ban reason bitmask */
+ u64 (*reset_status)(struct xe_exec_queue *q);
};
#endif
diff --git a/drivers/gpu/drm/xe/xe_execlist.c b/drivers/gpu/drm/xe/xe_execlist.c
index 0d0db66c6ea2..a36db39dcda8 100644
--- a/drivers/gpu/drm/xe/xe_execlist.c
+++ b/drivers/gpu/drm/xe/xe_execlist.c
@@ -453,10 +453,10 @@ static void execlist_exec_queue_resume(struct xe_exec_queue *q)
/* NIY */
}
-static bool execlist_exec_queue_reset_status(struct xe_exec_queue *q)
+static u64 execlist_exec_queue_reset_status(struct xe_exec_queue *q)
{
/* NIY */
- return false;
+ return 0;
}
static const struct xe_exec_queue_ops execlist_exec_queue_ops = {
diff --git a/drivers/gpu/drm/xe/xe_gt.c b/drivers/gpu/drm/xe/xe_gt.c
index 478e047031f4..775c826b68b4 100644
--- a/drivers/gpu/drm/xe/xe_gt.c
+++ b/drivers/gpu/drm/xe/xe_gt.c
@@ -70,6 +70,8 @@
#include "xe_wa.h"
#include "xe_wopcm.h"
+#define GRDOM_RESET_TIMEOUT_MS 5
+
struct xe_gt *xe_gt_alloc(struct xe_tile *tile)
{
struct xe_device *xe = tile_to_xe(tile);
@@ -828,10 +830,13 @@ static int do_gt_reset(struct xe_gt *gt)
xe_gsc_wa_14015076503(gt, true);
xe_mmio_write32(&gt->mmio, GDRST, GRDOM_FULL);
- err = xe_mmio_wait32(&gt->mmio, GDRST, GRDOM_FULL, 0, 5000, NULL, false);
+ err = xe_mmio_wait32(&gt->mmio, GDRST, GRDOM_FULL, 0,
+ GRDOM_RESET_TIMEOUT_MS * USEC_PER_MSEC,
+ NULL, false);
if (err)
- xe_gt_err(gt, "failed to clear GRDOM_FULL (%pe)\n",
- ERR_PTR(err));
+ xe_log_err(gt, GT, err,
+ "full graphics reset not completed in %u ms\n",
+ GRDOM_RESET_TIMEOUT_MS);
xe_gsc_wa_14015076503(gt, false);
diff --git a/drivers/gpu/drm/xe/xe_gt_debugfs.c b/drivers/gpu/drm/xe/xe_gt_debugfs.c
index 361a70234d1f..bb09e70ee44c 100644
--- a/drivers/gpu/drm/xe/xe_gt_debugfs.c
+++ b/drivers/gpu/drm/xe/xe_gt_debugfs.c
@@ -13,6 +13,7 @@
#include <drm/drm_managed.h>
#include <linux/math.h>
+#include "regs/xe_engine_regs.h"
#include "regs/xe_gt_regs.h"
#include "xe_device.h"
#include "xe_force_wake.h"
@@ -25,6 +26,7 @@
#include "xe_gt_stats.h"
#include "xe_gt_topology.h"
#include "xe_guc_hwconfig.h"
+#include "xe_guc_submit.h"
#include "xe_hw_engine.h"
#include "xe_lrc.h"
#include "xe_mmio.h"
@@ -130,6 +132,48 @@ static int hw_engines(struct xe_gt *gt, struct drm_printer *p)
return 0;
}
+static int multi_queue_active_lrca(struct xe_gt *gt, struct drm_printer *p)
+{
+ struct xe_guc *guc = &gt->uc.guc;
+ struct xe_hw_engine *hwe;
+ enum xe_hw_engine_id id;
+
+ for_each_hw_engine(hwe, gt, id) {
+ u32 cur_lrca, active_id, lrca;
+ unsigned int fw_ref;
+
+ if (!xe_gt_supports_multi_queue(gt, hwe->class))
+ continue;
+
+ /*
+ * Forcewake is dropped before xe_guc_submit_active_multi_queue_lrca()
+ * below, which takes guc->submission_state.lock, to avoid holding a
+ * GT forcewake ref across a mutex acquired elsewhere in the opposite
+ * order.
+ */
+ fw_ref = xe_force_wake_get(gt_to_fw(gt), XE_FORCEWAKE_ALL);
+ if (!xe_force_wake_ref_has_domain(fw_ref, XE_FORCEWAKE_ALL)) {
+ drm_printf(p, "%s\tforcewake failed, skipping\n", hwe->name);
+ xe_force_wake_put(gt_to_fw(gt), fw_ref);
+ continue;
+ }
+
+ cur_lrca = xe_mmio_read32(&gt->mmio,
+ RING_CURRENT_LRCA(hwe->mmio_base));
+ active_id = xe_lrc_get_multi_queue_active_queue_id(hwe);
+
+ xe_force_wake_put(gt_to_fw(gt), fw_ref);
+
+ lrca = xe_guc_submit_active_multi_queue_lrca(guc, hwe, cur_lrca,
+ active_id);
+
+ drm_printf(p, "%s\tactive_queue_id %u\tcurrent_lrca 0x%08x\tactive_lrca 0x%08x\n",
+ hwe->name, active_id, cur_lrca, lrca);
+ }
+
+ return 0;
+}
+
static int steering(struct xe_gt *gt, struct drm_printer *p)
{
xe_gt_mcr_steering_dump(gt, p);
@@ -254,6 +298,11 @@ static const struct drm_info_list pf_only_debugfs_list[] = {
{ "steering", .show = xe_gt_debugfs_show_with_rpm, .data = steering },
};
+static const struct drm_info_list multi_queue_debugfs_list[] = {
+ { "multi_queue_active_lrca",
+ .show = xe_gt_debugfs_show_with_rpm, .data = multi_queue_active_lrca },
+};
+
static ssize_t write_to_gt_call(const char __user *userbuf, size_t count, loff_t *ppos,
void (*call)(struct xe_gt *), struct xe_gt *gt)
{
@@ -521,6 +570,11 @@ void xe_gt_debugfs_register(struct xe_gt *gt)
ARRAY_SIZE(pf_only_debugfs_list),
root, minor);
+ if (!IS_SRIOV_VF(xe) && xe_gt_has_multi_queue(gt))
+ drm_debugfs_create_files(multi_queue_debugfs_list,
+ ARRAY_SIZE(multi_queue_debugfs_list),
+ root, minor);
+
if (xe_gt_is_main_type(gt) && !IS_DGFX(xe) && !IS_SRIOV_VF(xe))
debugfs_create_file("gt_ia_bias", 0600, root, gt, &gt_ia_bias_fops);
diff --git a/drivers/gpu/drm/xe/xe_guc_pagefault.c b/drivers/gpu/drm/xe/xe_guc_pagefault.c
index 8f8210a732e9..df237fd40551 100644
--- a/drivers/gpu/drm/xe/xe_guc_pagefault.c
+++ b/drivers/gpu/drm/xe/xe_guc_pagefault.c
@@ -108,7 +108,13 @@ int xe_guc_pagefault_handler(struct xe_guc *guc, u32 *msg, u32 len)
<< PFD_VIRTUAL_ADDR_HI_SHIFT) |
(FIELD_GET(PFD_VIRTUAL_ADDR_LO, msg[2]) <<
PFD_VIRTUAL_ADDR_LO_SHIFT);
- pf.consumer.asid = FIELD_GET(PFD_ASID, msg[1]);
+
+ BUILD_BUG_ON(XE_MAX_ASID > XE_PAGEFAULT_ASID_MASK);
+
+ pf.consumer.id = FIELD_PREP(XE_PAGEFAULT_ASID_MASK,
+ FIELD_GET(PFD_ASID, msg[1])) |
+ FIELD_PREP(XE_PAGEFAULT_SRCID_MASK,
+ FIELD_GET(PFD_SRC_ID, msg[0]));
pf.consumer.access_type = FIELD_GET(PFD_ACCESS_TYPE, msg[2]) |
(FIELD_GET(PFD_PREFETCH, msg[2]) ? XE_PAGEFAULT_ACCESS_PREFETCH : 0);
if (FIELD_GET(XE2_PFD_TRVA_FAULT, msg[0]))
diff --git a/drivers/gpu/drm/xe/xe_guc_submit.c b/drivers/gpu/drm/xe/xe_guc_submit.c
index 99d8c807ff05..f3ba8abfc228 100644
--- a/drivers/gpu/drm/xe/xe_guc_submit.c
+++ b/drivers/gpu/drm/xe/xe_guc_submit.c
@@ -6,6 +6,7 @@
#include "xe_guc_submit.h"
#include <linux/bitfield.h>
+#include <uapi/drm/xe_drm.h>
#include <linux/bitmap.h>
#include <linux/circ_buf.h>
#include <linux/dma-fence-array.h>
@@ -34,6 +35,7 @@
#include "xe_guc_klv_helpers.h"
#include "xe_guc_submit_types.h"
#include "xe_hw_engine.h"
+#include "xe_log.h"
#include "xe_lrc.h"
#include "xe_macros.h"
#include "xe_map.h"
@@ -1599,6 +1601,12 @@ guc_exec_queue_timedout_job(struct drm_sched_job *drm_job)
else
wedged = xe_device_wedged(xe);
+ /*
+ * Only tag as GPU hang if this is the original timeout, not a
+ * consequence of a prior kill (e.g., page-offline).
+ */
+ if (!exec_queue_killed(q))
+ atomic_or(DRM_XE_EXEC_QUEUE_BAN_REASON_GPU_HANG, &q->ban_reason);
set_exec_queue_banned(q);
/* Kick job / queue off hardware */
@@ -1682,6 +1690,9 @@ trigger_reset:
if (timeout_needs_gt_reset(q, job, skip_timeout_check)) {
if (!xe_sched_invalidate_job(job, 2)) {
clear_exec_queue_banned(q);
+ /* protect concurrent page offline reasons */
+ atomic_andnot(DRM_XE_EXEC_QUEUE_BAN_REASON_GPU_HANG,
+ &q->ban_reason);
xe_gt_reset_async(q->gt);
goto rearm;
}
@@ -2580,13 +2591,29 @@ static void guc_exec_queue_multi_queue_drop_suspend(struct xe_exec_queue *q)
}
}
-static bool guc_exec_queue_reset_status(struct xe_exec_queue *q)
+static u64 guc_exec_queue_reset_status(struct xe_exec_queue *q)
{
- if (xe_exec_queue_is_multi_queue_secondary(q) &&
- guc_exec_queue_reset_status(xe_exec_queue_multi_queue_primary(q)))
- return true;
+ /* TODO: In case of multiqueue, if a secondary queue is banned due to
+ * page offlining, checking only the primary queue's GuC reset status
+ * may mask the true reason or race with it.
+ */
+ if (xe_exec_queue_is_multi_queue_secondary(q)) {
+ u64 status = guc_exec_queue_reset_status(xe_exec_queue_multi_queue_primary(q));
- return exec_queue_reset(q) || exec_queue_killed_or_banned_or_wedged(q);
+ if (status)
+ return status;
+ }
+
+ if (exec_queue_reset(q) || exec_queue_killed_or_banned_or_wedged(q)) {
+ u64 reason = atomic_read_acquire(&q->ban_reason);
+
+ /* If no specific reason was recorded, default to GPU hang */
+ if (!reason)
+ reason = DRM_XE_EXEC_QUEUE_BAN_REASON_GPU_HANG;
+ return reason;
+ }
+
+ return 0;
}
/*
@@ -3494,8 +3521,9 @@ int xe_guc_exec_queue_reset_failure_handler(struct xe_guc *guc, u32 *msg, u32 le
reason = msg[2];
/* Unexpected failure of a hardware feature, log an actual error */
- xe_gt_err(gt, "GuC engine reset request failed on %d:%d because 0x%08X",
- guc_class, instance, reason);
+ xe_log_err(gt, GUCSUBMIT, -EIO,
+ "engine reset failed on %u:%u, reason=%#x\n",
+ guc_class, instance, reason);
xe_gt_reset_async(gt);
@@ -3854,6 +3882,79 @@ bool xe_guc_has_registered_mlrc_queues(struct xe_guc *guc)
}
/**
+ * xe_guc_submit_active_multi_queue_lrca() - Resolve the LRCA of the active
+ * queue in the multi-queue group currently running on an engine.
+ * @guc: the &xe_guc managing the exec queues
+ * @hwe: the &xe_hw_engine whose active queue is being resolved
+ * @cur_lrca: value read from RING_CURRENT_LRCA, identifies the running group
+ * @active_id: current Active Queue ID read from CSMQDEBUG (position in group)
+ *
+ * The running group is identified by matching @cur_lrca against the group's
+ * primary LRCA; @active_id then selects the active queue within that group.
+ *
+ * Return: the LRCA of the active queue, or 0 if no matching queue is found.
+ */
+u32 xe_guc_submit_active_multi_queue_lrca(struct xe_guc *guc,
+ struct xe_hw_engine *hwe,
+ u32 cur_lrca, u32 active_id)
+{
+ struct xe_exec_queue *q;
+ unsigned long index;
+ u32 lrca = 0;
+
+ /*
+ * submission_state.lock also protects exec_queue teardown: an exec
+ * queue is removed from exec_queue_lookup before its group/primary
+ * are freed, so any q found in the xarray below has a live group
+ * and primary for as long as we hold the lock.
+ */
+ guard(mutex)(&guc->submission_state.lock);
+
+ xa_for_each(&guc->submission_state.exec_queue_lookup, index, q) {
+ struct xe_exec_queue_group *group = q->multi_queue.group;
+ struct xe_lrc *active_lrc;
+ struct xe_lrc *primary_lrc;
+
+ if (!q->multi_queue.valid || !group || !group->primary)
+ continue;
+ /*
+ * Multi-queue exec queues are bound to a hw engine class;
+ * GuC dynamically schedules them onto one of the class's
+ * physical instances, so there is no fixed queue-to-instance
+ * mapping to filter on here.
+ */
+ if (q->class != hwe->class)
+ continue;
+ if (q->multi_queue.pos != active_id)
+ continue;
+ /*
+ * LRCAs are page-aligned (4K) addresses in GGTT; the low
+ * bits reported by RING_CURRENT_LRCA are not meaningful, so
+ * only compare bits [31:12].
+ */
+ primary_lrc = xe_exec_queue_get_lrc(group->primary, 0);
+ if (!primary_lrc)
+ continue;
+
+ if ((xe_lrc_ggtt_addr(primary_lrc) ^ cur_lrca) & GENMASK(31, 12)) {
+ xe_lrc_put(primary_lrc);
+ continue;
+ }
+
+ active_lrc = xe_exec_queue_get_lrc(q, 0);
+ xe_lrc_put(primary_lrc);
+ if (!active_lrc)
+ continue;
+
+ lrca = xe_lrc_ggtt_addr(active_lrc);
+ xe_lrc_put(active_lrc);
+ break;
+ }
+
+ return lrca;
+}
+
+/**
* xe_guc_contexts_hwsp_rebase - Re-compute GGTT references within all
* exec queues registered to given GuC.
* @guc: the &xe_guc struct instance
diff --git a/drivers/gpu/drm/xe/xe_guc_submit.h b/drivers/gpu/drm/xe/xe_guc_submit.h
index ccade320dc69..29abf07d8f04 100644
--- a/drivers/gpu/drm/xe/xe_guc_submit.h
+++ b/drivers/gpu/drm/xe/xe_guc_submit.h
@@ -11,6 +11,7 @@
struct drm_printer;
struct xe_exec_queue;
struct xe_guc;
+struct xe_hw_engine;
int xe_guc_submit_init(struct xe_guc *guc, unsigned int num_ids);
int xe_guc_submit_enable(struct xe_guc *guc);
@@ -55,6 +56,10 @@ void xe_guc_register_vf_exec_queue(struct xe_exec_queue *q, int ctx_type);
bool xe_guc_has_registered_mlrc_queues(struct xe_guc *guc);
+u32 xe_guc_submit_active_multi_queue_lrca(struct xe_guc *guc,
+ struct xe_hw_engine *hwe,
+ u32 cur_lrca, u32 active_id);
+
int xe_guc_contexts_hwsp_rebase(struct xe_guc *guc, void *scratch);
#endif
diff --git a/drivers/gpu/drm/xe/xe_hwmon.c b/drivers/gpu/drm/xe/xe_hwmon.c
index 5284cab6703d..5edeac961ec3 100644
--- a/drivers/gpu/drm/xe/xe_hwmon.c
+++ b/drivers/gpu/drm/xe/xe_hwmon.c
@@ -101,11 +101,6 @@ enum sensor_attr_power {
#define PWR_ATTR_TO_STR(attr) (((attr) == hwmon_power_max) ? "PL1" : "PL2")
-/*
- * Timeout for power limit write mailbox command.
- */
-#define PL_WRITE_MBX_TIMEOUT_MS (1)
-
/* Index of memory controller in READ_THERMAL_DATA output */
#define TEMP_INDEX_MCTRL 2
@@ -252,7 +247,7 @@ static int xe_hwmon_pcode_rmw_power_limit(const struct xe_hwmon *hwmon, u32 attr
(channel == CHANNEL_CARD) ?
WRITE_PSYSGPU_POWER_LIMIT :
WRITE_PACKAGE_POWER_LIMIT, 0),
- val0, val1, PL_WRITE_MBX_TIMEOUT_MS);
+ val0, val1, PCODE_DEFAULT_TIMEOUT_MS);
if (ret)
drm_dbg(&hwmon->xe->drm, "write failed ch %d val0 0x%08x, val1 0x%08x, ret %d\n",
channel, val0, val1, ret);
diff --git a/drivers/gpu/drm/xe/xe_log.c b/drivers/gpu/drm/xe/xe_log.c
index 5549ef6966fd..29eb16db3320 100644
--- a/drivers/gpu/drm/xe/xe_log.c
+++ b/drivers/gpu/drm/xe/xe_log.c
@@ -10,6 +10,7 @@
#include "xe_device.h"
#include "xe_log.h"
+#include "xe_pci_types.h"
#include "xe_printk.h"
static void log_emit_cper(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid,
@@ -52,18 +53,24 @@ static const char *log_component_prefix(u32 component)
return component ? log_unknown_component_prefix(component) : "";
}
-static struct xe_gt *get_gt_safe(struct pci_dev *pdev, u8 id)
+static bool allowed_tile_id(struct xe_device *xe, u8 tile_id)
{
- struct xe_device *xe = pdev_to_xe_device(pdev);
+ return tile_id < 1 + xe->desc->max_remote_tiles;
+}
- return xe ? xe_device_get_gt(xe, id) : NULL;
+static bool allowed_gt_id(struct xe_device *xe, u8 gt_id)
+{
+ return gt_id < (1 + xe->desc->max_remote_tiles) * xe->desc->max_gt_per_tile;
}
-static struct xe_tile *get_tile_safe(struct pci_dev *pdev, u8 id)
+static u8 gt_id_to_tile_id(struct xe_device *xe, u8 gt_id)
{
- struct xe_device *xe = pdev_to_xe_device(pdev);
+ return gt_id / xe->desc->max_gt_per_tile;
+}
- return xe && id < xe->info.tile_count ? &xe->tiles[id] : NULL;
+static const char *location_suffix(bool valid)
+{
+ return valid ? ":" : "?";
}
static const char *log_location_prefix(struct pci_dev *pdev, u32 location, char *buf, size_t size)
@@ -76,17 +83,22 @@ static const char *log_location_prefix(struct pci_dev *pdev, u32 location, char
goto unrecognized;
strscpy(buf, "", size);
} else if (type == XE_LOG_LOCATION_TYPE_TILE) {
- struct xe_tile *tile = get_tile_safe(pdev, id);
+ struct xe_device *xe = xe_any_to_xe(pdev);
+ bool valid = xe ? allowed_tile_id(xe, id) : false;
+ const char *pad = location_suffix(valid);
- if (!tile)
- goto unrecognized;
- snprintf(buf, size, "Tile%u: ", id);
+ pci_WARN(pdev, !valid && IS_ENABLED(CONFIG_DRM_XE_DEBUG),
+ "LOG: invalid tile identifier: %u\n", id);
+ snprintf(buf, size, "Tile%u%s ", id, pad);
} else if (type == XE_LOG_LOCATION_TYPE_GT) {
- struct xe_gt *gt = get_gt_safe(pdev, id);
-
- if (!gt)
- goto unrecognized;
- snprintf(buf, size, "Tile%u: GT%u: ", gt->tile->id, id);
+ struct xe_device *xe = xe_any_to_xe(pdev);
+ bool valid = xe ? allowed_gt_id(xe, id) : false;
+ const char *pad = location_suffix(valid);
+ u8 tile_id = xe ? gt_id_to_tile_id(xe, id) : 0;
+
+ pci_WARN(pdev, !valid && IS_ENABLED(CONFIG_DRM_XE_DEBUG),
+ "LOG: invalid GT identifier: %u\n", id);
+ snprintf(buf, size, "Tile%u%s GT%u%s ", tile_id, pad, id, pad);
} else {
goto unrecognized;
}
diff --git a/drivers/gpu/drm/xe/xe_lrc.c b/drivers/gpu/drm/xe/xe_lrc.c
index 25fe9dbc9141..f1cf1463f1b2 100644
--- a/drivers/gpu/drm/xe/xe_lrc.c
+++ b/drivers/gpu/drm/xe/xe_lrc.c
@@ -2705,7 +2705,7 @@ static u64 get_queue_timestamp(struct xe_hw_engine *hwe)
RING_QUEUE_TIMESTAMP(hwe->mmio_base));
}
-static u32 get_multi_queue_active_queue_id(struct xe_hw_engine *hwe)
+u32 xe_lrc_get_multi_queue_active_queue_id(struct xe_hw_engine *hwe)
{
u32 val = xe_mmio_read32(&hwe->gt->mmio,
RING_CSMQDEBUG(hwe->mmio_base));
@@ -2739,14 +2739,14 @@ static u64 xe_lrc_multi_queue_timestamp(struct xe_lrc *lrc)
if (!hwe)
return xe_lrc_queue_timestamp(lrc);
- if (get_multi_queue_active_queue_id(hwe) != lrc->multi_queue.pos)
+ if (xe_lrc_get_multi_queue_active_queue_id(hwe) != lrc->multi_queue.pos)
return xe_lrc_queue_timestamp(lrc);
/* queue is active, so store the queue timestamp register */
reg_queue_ts = get_queue_timestamp(hwe);
/* double check queue and primary queue are both still active */
- if (get_multi_queue_active_queue_id(hwe) != lrc->multi_queue.pos ||
+ if (xe_lrc_get_multi_queue_active_queue_id(hwe) != lrc->multi_queue.pos ||
!context_active(primary_lrc))
return xe_lrc_queue_timestamp(lrc);
diff --git a/drivers/gpu/drm/xe/xe_lrc.h b/drivers/gpu/drm/xe/xe_lrc.h
index 7be5e3da8bc8..a8ff4e59a1f4 100644
--- a/drivers/gpu/drm/xe/xe_lrc.h
+++ b/drivers/gpu/drm/xe/xe_lrc.h
@@ -158,6 +158,7 @@ int xe_lrc_lookup_default_reg_value(struct xe_gt *gt,
u32 *xe_lrc_emit_hwe_state_instructions(struct xe_exec_queue *q, u32 *cs);
void xe_lrc_set_multi_queue_priority(struct xe_lrc *lrc, enum xe_multi_queue_priority priority);
+u32 xe_lrc_get_multi_queue_active_queue_id(struct xe_hw_engine *hwe);
struct xe_lrc_snapshot *xe_lrc_snapshot_capture(struct xe_lrc *lrc);
void xe_lrc_snapshot_capture_delayed(struct xe_lrc_snapshot *snapshot);
diff --git a/drivers/gpu/drm/xe/xe_migrate.c b/drivers/gpu/drm/xe/xe_migrate.c
index 75b83687f1b5..ff45c24d8889 100644
--- a/drivers/gpu/drm/xe/xe_migrate.c
+++ b/drivers/gpu/drm/xe/xe_migrate.c
@@ -87,7 +87,7 @@ struct xe_migrate {
#define MAX_PREEMPTDISABLE_TRANSFER SZ_8M /* Around 1ms. */
#define MAX_CCS_LIMITED_TRANSFER SZ_4M /* XE_PAGE_SIZE * (FIELD_MAX(XE2_CCS_SIZE_MASK) + 1) */
#define NUM_KERNEL_PDE 15
-#define NUM_PT_SLOTS 32
+#define NUM_PT_SLOTS 48
#define LEVEL0_PAGE_TABLE_ENCODE_SIZE SZ_2M
#define MAX_NUM_PTE 512
#define IDENTITY_OFFSET 256ULL
@@ -163,22 +163,20 @@ static u64 xe_migrate_vram_ofs(struct xe_device *xe, u64 addr, bool is_comp_pte)
}
static void xe_migrate_program_identity(struct xe_device *xe, struct xe_vm *vm, struct xe_bo *bo,
- u64 map_ofs, u64 vram_offset, u16 pat_index, u64 pt_2m_ofs)
+ u64 map_ofs, u64 vram_offset, u16 pat_index, u64 pt_2m_ofs,
+ u64 pt_4k_ofs)
{
struct xe_vram_region *vram = xe->mem.vram;
resource_size_t dpa_base = xe_vram_region_dpa_base(vram);
u64 pos, ofs, flags;
u64 entry;
- /* XXX: Unclear if this should be usable_size? */
- u64 vram_limit = xe_vram_region_actual_physical_size(vram) + dpa_base;
+ u64 vram_limit = xe_vram_region_usable_size(vram) + dpa_base;
u32 level = 2;
ofs = map_ofs + XE_PAGE_SIZE * level + vram_offset * 8;
flags = vm->pt_ops->pte_encode_addr(xe, 0, pat_index, level,
true, 0);
- xe_assert(xe, IS_ALIGNED(xe_vram_region_usable_size(vram), SZ_2M));
-
/*
* Use 1GB pages when possible, last chunk always use 2M
* pages as mixing reserved memory (stolen, WOCPM) with a single
@@ -196,8 +194,24 @@ static void xe_migrate_program_identity(struct xe_device *xe, struct xe_vm *vm,
true, 0);
for (ofs = pt_2m_ofs; pos < vram_limit;
- pos += SZ_2M, ofs += 8)
+ pos += SZ_2M, ofs += 8) {
+ if (pos + SZ_2M > vram_limit) {
+ entry = vm->pt_ops->pde_encode_bo(bo, pt_4k_ofs);
+ xe_map_wr(xe, &bo->vmap, ofs, u64, entry);
+
+ flags = vm->pt_ops->pte_encode_addr(xe, 0,
+ pat_index,
+ level - 2,
+ true, 0);
+
+ for (ofs = pt_4k_ofs; pos < vram_limit;
+ pos += SZ_4K, ofs += 8)
+ xe_map_wr(xe, &bo->vmap, ofs, u64, pos | flags);
+ break;
+ }
+
xe_map_wr(xe, &bo->vmap, ofs, u64, pos | flags);
+ }
break; /* Ensure pos == vram_limit assert correct */
}
@@ -242,16 +256,17 @@ static void xe_migrate_prepare_vm(struct xe_tile *tile, struct xe_migrate *m,
u16 pat_index = xe_cache_pat_idx(xe, XE_CACHE_WB);
u8 id = tile->id;
u32 num_entries = NUM_PT_SLOTS, num_level = vm->pt_root[id]->level;
-#define VRAM_IDENTITY_MAP_COUNT 2
- u32 num_setup = num_level + VRAM_IDENTITY_MAP_COUNT;
-#undef VRAM_IDENTITY_MAP_COUNT
+#define VRAM_IDENTITY_MAP_PT_COUNT 4
+ u32 num_setup = num_level + VRAM_IDENTITY_MAP_PT_COUNT;
+#undef VRAM_IDENTITY_MAP_PT_COUNT
u32 map_ofs, level, i;
struct xe_bo *bo = m->pt_bo, *batch = tile->mem.kernel_bb_pool->bo;
- u64 entry, pt29_ofs;
+ u64 entry;
- /* PT30 & PT31 reserved for 2M identity map */
- pt29_ofs = xe_bo_size(bo) - 3 * XE_PAGE_SIZE;
- entry = vm->pt_ops->pde_encode_bo(bo, pt29_ofs);
+ /* PT44..PT47 reserved for 4K and 2M identity map */
+ u64 l1_pt_ofs = xe_bo_size(bo) - 5 * XE_PAGE_SIZE;
+
+ entry = vm->pt_ops->pde_encode_bo(bo, l1_pt_ofs);
xe_pt_write(xe, &vm->pt_root[id]->bo->vmap, 0, entry);
map_ofs = (num_entries - num_setup) * XE_PAGE_SIZE;
@@ -347,11 +362,12 @@ static void xe_migrate_prepare_vm(struct xe_tile *tile, struct xe_migrate *m,
/* Identity map the entire vram at 256GiB offset */
if (IS_DGFX(xe)) {
- u64 pt30_ofs = xe_bo_size(bo) - 2 * XE_PAGE_SIZE;
+ u64 pt46_ofs = xe_bo_size(bo) - 2 * XE_PAGE_SIZE;
resource_size_t actual_phy_size = xe_vram_region_actual_physical_size(xe->mem.vram);
+ u64 pt44_ofs = xe_bo_size(bo) - 4 * XE_PAGE_SIZE;
xe_migrate_program_identity(xe, vm, bo, map_ofs, IDENTITY_OFFSET,
- pat_index, pt30_ofs);
+ pat_index, pt46_ofs, pt44_ofs);
xe_assert(xe, actual_phy_size <= (MAX_NUM_PTE - IDENTITY_OFFSET) * SZ_1G);
/*
@@ -362,12 +378,13 @@ static void xe_migrate_prepare_vm(struct xe_tile *tile, struct xe_migrate *m,
u16 comp_pat_index = xe_cache_pat_idx(xe, XE_CACHE_NONE_COMPRESSION);
u64 vram_offset = IDENTITY_OFFSET +
DIV_ROUND_UP_ULL(actual_phy_size, SZ_1G);
- u64 pt31_ofs = xe_bo_size(bo) - XE_PAGE_SIZE;
+ u64 pt47_ofs = xe_bo_size(bo) - XE_PAGE_SIZE;
xe_assert(xe, actual_phy_size <= (MAX_NUM_PTE - IDENTITY_OFFSET -
IDENTITY_OFFSET / 2) * SZ_1G);
+ u64 pt45_ofs = xe_bo_size(bo) - 3 * XE_PAGE_SIZE;
xe_migrate_program_identity(xe, vm, bo, map_ofs, vram_offset,
- comp_pat_index, pt31_ofs);
+ comp_pat_index, pt47_ofs, pt45_ofs);
}
}
@@ -381,8 +398,8 @@ static void xe_migrate_suballoc_manager_init(struct xe_migrate *m, u32 map_ofs)
* Example layout created above, with root level = 3:
* [PT0...PT7]: kernel PT's for copy/clear; 64 or 4KiB PTE's
* [PT8]: Kernel PT for VM_BIND, 4 KiB PTE's
- * [PT9...PT26]: Userspace PT's for VM_BIND, 4 KiB PTE's
- * [PT27 = PDE 0] [PT28 = PDE 1] [PT29 = PDE 2] [PT30 & PT31 = 2M vram identity map]
+ * [PT9...PT40]: Userspace PT's for VM_BIND, 4 KiB PTE's
+ * [PT41 = PDE 0] [PT44...PT47 = 4K and 2M vram identity maps]
*
* This makes the lowest part of the VM point to the pagetables.
* Hence the lowest 2M in the vm should point to itself, with a few writes
@@ -2633,3 +2650,67 @@ void xe_migrate_job_lock_assert(struct xe_exec_queue *q)
#if IS_ENABLED(CONFIG_DRM_XE_KUNIT_TEST)
#include "tests/xe_migrate.c"
#endif
+
+#if IS_ENABLED(CONFIG_DRM_XE_DEBUG_MEM)
+int xe_migrate_debug_ccs_overlap(struct xe_migrate *m,
+ struct xe_bo *scratch_bo,
+ bool write_to_ccs)
+{
+ struct xe_device *xe = tile_to_xe(m->tile);
+ struct xe_gt *gt = m->tile->primary_gt;
+ struct dma_fence *fence;
+ struct xe_bb *bb;
+ struct xe_sched_job *job;
+ u64 first_page_dpa, clear_L0_ofs, scratch_dpa, scratch_L0_ofs;
+
+ if (!xe_device_has_flat_ccs(xe))
+ return -EINVAL;
+
+ first_page_dpa = xe_vram_region_dpa_base(m->tile->mem.vram);
+ clear_L0_ofs = xe_migrate_vram_ofs(xe, first_page_dpa, true);
+
+ scratch_dpa = xe_bo_addr(scratch_bo, 0, XE_PAGE_SIZE);
+ scratch_L0_ofs = xe_migrate_vram_ofs(xe, scratch_dpa, false);
+
+ bb = xe_bb_new(gt, EMIT_COPY_CCS_DW + 1, xe->info.has_usm);
+ if (IS_ERR(bb)) {
+ drm_warn(&xe->drm, "Failed to create bb for VRAM overlap check\n");
+ return PTR_ERR(bb);
+ }
+
+ /* 4MB payload = 8KB CCS metadata */
+ if (write_to_ccs) {
+ emit_copy_ccs(gt, bb, clear_L0_ofs, true,
+ scratch_L0_ofs, false, SZ_4M);
+ } else {
+ emit_copy_ccs(gt, bb, scratch_L0_ofs, false,
+ clear_L0_ofs, true, SZ_4M);
+ }
+
+ bb->cs[bb->len++] = MI_BATCH_BUFFER_END;
+
+ job = xe_bb_create_migration_job(m->q, bb,
+ xe_migrate_batch_base(m, xe->info.has_usm),
+ 0);
+ if (!IS_ERR(job)) {
+ xe_sched_job_add_migrate_flush(job, MI_FLUSH_DW_CCS);
+
+ mutex_lock(&m->job_mutex);
+ xe_sched_job_arm(job);
+
+ fence = dma_fence_get(&job->drm.s_fence->finished);
+ xe_sched_job_push(job);
+ mutex_unlock(&m->job_mutex);
+
+ dma_fence_wait(fence, false);
+ dma_fence_put(fence);
+ } else {
+ drm_warn(&xe->drm, "Failed to create job for VRAM overlap check\n");
+ xe_bb_free(bb, NULL);
+ return PTR_ERR(job);
+ }
+
+ xe_bb_free(bb, NULL);
+ return 0;
+}
+#endif
diff --git a/drivers/gpu/drm/xe/xe_migrate.h b/drivers/gpu/drm/xe/xe_migrate.h
index c3a268b01768..a9acc62f78f0 100644
--- a/drivers/gpu/drm/xe/xe_migrate.h
+++ b/drivers/gpu/drm/xe/xe_migrate.h
@@ -182,4 +182,10 @@ static inline void xe_migrate_job_lock_assert(struct xe_exec_queue *q)
void xe_migrate_job_lock(struct xe_migrate *m, struct xe_exec_queue *q);
void xe_migrate_job_unlock(struct xe_migrate *m, struct xe_exec_queue *q);
+#if IS_ENABLED(CONFIG_DRM_XE_DEBUG_MEM)
+int xe_migrate_debug_ccs_overlap(struct xe_migrate *m,
+ struct xe_bo *scratch_bo,
+ bool write_to_ccs);
+#endif
+
#endif
diff --git a/drivers/gpu/drm/xe/xe_mmio_gem.c b/drivers/gpu/drm/xe/xe_mmio_gem.c
index 3741ae60f532..3cdc9538957d 100644
--- a/drivers/gpu/drm/xe/xe_mmio_gem.c
+++ b/drivers/gpu/drm/xe/xe_mmio_gem.c
@@ -5,9 +5,9 @@
#include "xe_mmio_gem.h"
+#include <linux/dma-resv.h>
#include <drm/drm_drv.h>
#include <drm/drm_gem.h>
-#include <drm/drm_managed.h>
#include "xe_device_types.h"
@@ -37,12 +37,24 @@ static vm_fault_t xe_mmio_gem_vm_fault(struct vm_fault *);
struct xe_mmio_gem {
struct drm_gem_object base;
phys_addr_t phys_addr;
+ struct page *dummy_page; /* protected by the GEM's dma_resv */
+ bool destroyed; /* protected by the GEM's dma_resv */
};
+static int xe_mmio_gem_vm_may_split(struct vm_area_struct *area, unsigned long addr)
+{
+ /*
+ * Forbid splitting. Together with VM_DONTEXPAND, this keeps the VMA
+ * matching the GEM object exactly.
+ */
+ return -EINVAL;
+}
+
static const struct vm_operations_struct vm_ops = {
.open = drm_gem_vm_open,
.close = drm_gem_vm_close,
.fault = xe_mmio_gem_vm_fault,
+ .may_split = xe_mmio_gem_vm_may_split,
};
static const struct drm_gem_object_funcs xe_mmio_gem_funcs = {
@@ -121,6 +133,8 @@ static void xe_mmio_gem_free(struct drm_gem_object *base)
{
struct xe_mmio_gem *obj = to_xe_mmio_gem(base);
+ if (obj->dummy_page)
+ __free_page(obj->dummy_page);
drm_gem_object_release(base);
kfree(obj);
}
@@ -128,15 +142,31 @@ static void xe_mmio_gem_free(struct drm_gem_object *base)
/**
* xe_mmio_gem_destroy - Destroy the GEM object that exposes an MMIO region
* @gem: the GEM object to destroy
+ * @file: DRM file descriptor previously passed to xe_mmio_gem_create()
*
* This function releases resources associated with the GEM object created by
* xe_mmio_gem_create().
*
* See: "Exposing MMIO regions to userspace"
*/
-void xe_mmio_gem_destroy(struct xe_mmio_gem *gem)
+void xe_mmio_gem_destroy(struct xe_mmio_gem *gem, struct drm_file *file)
{
- xe_mmio_gem_free(&gem->base);
+ struct drm_gem_object *base = &gem->base;
+ struct drm_device *dev = base->dev;
+
+ drm_vma_node_revoke(&base->vma_node, file);
+
+ dma_resv_lock(base->resv, NULL);
+ gem->destroyed = true;
+ dma_resv_unlock(base->resv);
+ /*
+ * Setting 'destroyed' under lock takes care of the subsequent faults.
+ * Zap the existing PTEs to cut off access to the real MMIO through
+ * currently mapped pages.
+ */
+ drm_vma_node_unmap(&base->vma_node, dev->anon_inode->i_mapping);
+
+ drm_gem_object_put(base);
}
static int xe_mmio_gem_mmap(struct drm_gem_object *base, struct vm_area_struct *vma)
@@ -147,61 +177,58 @@ static int xe_mmio_gem_mmap(struct drm_gem_object *base, struct vm_area_struct *
if ((vma->vm_flags & VM_SHARED) == 0)
return -EINVAL;
- /* Set vm_pgoff (used as a fake buffer offset by DRM) to 0 */
- vma->vm_pgoff = 0;
+ if (vma->vm_flags & VM_EXEC)
+ return -EINVAL;
+
vma->vm_page_prot = pgprot_noncached(vma_get_page_prot(vma));
- vm_flags_set(vma, VM_IO | VM_PFNMAP | VM_DONTEXPAND | VM_DONTDUMP |
- VM_DONTCOPY | VM_NORESERVE);
+ vm_flags_mod(vma, VM_IO | VM_PFNMAP | VM_DONTEXPAND | VM_DONTDUMP |
+ VM_NORESERVE, VM_MAYEXEC);
/* Defer actual mapping to the fault handler. */
return 0;
}
-static void xe_mmio_gem_release_dummy_page(struct drm_device *dev, void *res)
+static int alloc_dummy_page_if_needed(struct drm_gem_object *base)
{
- __free_page((struct page *)res);
+ struct xe_mmio_gem *obj = to_xe_mmio_gem(base);
+
+ dma_resv_assert_held(base->resv);
+ if (!obj->dummy_page)
+ obj->dummy_page = alloc_page(GFP_KERNEL | __GFP_ZERO);
+
+ return obj->dummy_page ? 0 : -ENOMEM;
}
-static vm_fault_t xe_mmio_gem_vm_fault_dummy_page(struct vm_area_struct *vma)
+static vm_fault_t xe_mmio_gem_vm_fault_dummy_page(struct vm_fault *vmf)
{
+ struct vm_area_struct *vma = vmf->vma;
struct drm_gem_object *base = vma->vm_private_data;
- struct drm_device *dev = base->dev;
- vm_fault_t ret = VM_FAULT_NOPAGE;
- struct page *page;
+ struct xe_mmio_gem *obj = to_xe_mmio_gem(base);
unsigned long pfn;
- unsigned long i;
-
- page = alloc_page(GFP_KERNEL | __GFP_ZERO);
- if (!page)
- return VM_FAULT_OOM;
- if (drmm_add_action_or_reset(dev, xe_mmio_gem_release_dummy_page, page))
+ if (alloc_dummy_page_if_needed(base))
return VM_FAULT_OOM;
- pfn = page_to_pfn(page);
-
- /* Map the entire VMA to the same dummy page */
- for (i = 0; i < base->size; i += PAGE_SIZE) {
- unsigned long addr = vma->vm_start + i;
+ pfn = page_to_pfn(obj->dummy_page);
- ret = vmf_insert_pfn(vma, addr, pfn);
- if (ret & VM_FAULT_ERROR)
- break;
- }
-
- return ret;
+ return vmf_insert_pfn_prot(vma, vmf->address, pfn,
+ vm_get_page_prot(vma->vm_flags));
}
-static vm_fault_t xe_mmio_gem_vm_fault(struct vm_fault *vmf)
+static vm_fault_t xe_mmio_gem_vm_fault_locked(struct vm_fault *vmf)
{
struct vm_area_struct *vma = vmf->vma;
struct drm_gem_object *base = vma->vm_private_data;
struct xe_mmio_gem *obj = to_xe_mmio_gem(base);
struct drm_device *dev = base->dev;
vm_fault_t ret = VM_FAULT_NOPAGE;
- unsigned long i;
+ unsigned long addr, pfn;
int idx;
+ dma_resv_assert_held(base->resv);
+ if (obj->destroyed)
+ return VM_FAULT_SIGBUS;
+
if (!drm_dev_enter(dev, &idx)) {
/*
* Provide a dummy page to avoid SIGBUS for events such as hot-unplug.
@@ -209,18 +236,30 @@ static vm_fault_t xe_mmio_gem_vm_fault(struct vm_fault *vmf)
* It is assumed the userspace will receive the notification via some
* other channel (e.g. drm uevent).
*/
- return xe_mmio_gem_vm_fault_dummy_page(vma);
+ return xe_mmio_gem_vm_fault_dummy_page(vmf);
}
- for (i = 0; i < base->size; i += PAGE_SIZE) {
- unsigned long addr = vma->vm_start + i;
- unsigned long phys_addr = obj->phys_addr + i;
-
- ret = vmf_insert_pfn(vma, addr, PHYS_PFN(phys_addr));
+ pfn = PHYS_PFN(obj->phys_addr);
+ for (addr = vma->vm_start; addr < vma->vm_end; addr += PAGE_SIZE) {
+ ret = vmf_insert_pfn(vma, addr, pfn);
if (ret & VM_FAULT_ERROR)
break;
+
+ pfn++;
}
drm_dev_exit(idx);
return ret;
}
+
+static vm_fault_t xe_mmio_gem_vm_fault(struct vm_fault *vmf)
+{
+ struct vm_area_struct *vma = vmf->vma;
+ struct drm_gem_object *base = vma->vm_private_data;
+ vm_fault_t ret;
+
+ dma_resv_lock(base->resv, NULL);
+ ret = xe_mmio_gem_vm_fault_locked(vmf);
+ dma_resv_unlock(base->resv);
+ return ret;
+}
diff --git a/drivers/gpu/drm/xe/xe_mmio_gem.h b/drivers/gpu/drm/xe/xe_mmio_gem.h
index 4b76d5586ebb..80d7795f07c8 100644
--- a/drivers/gpu/drm/xe/xe_mmio_gem.h
+++ b/drivers/gpu/drm/xe/xe_mmio_gem.h
@@ -15,6 +15,6 @@ struct xe_mmio_gem;
struct xe_mmio_gem *xe_mmio_gem_create(struct xe_device *xe, struct drm_file *file,
phys_addr_t phys_addr, size_t size);
u64 xe_mmio_gem_mmap_offset(struct xe_mmio_gem *gem);
-void xe_mmio_gem_destroy(struct xe_mmio_gem *gem);
+void xe_mmio_gem_destroy(struct xe_mmio_gem *gem, struct drm_file *file);
#endif /* _XE_MMIO_GEM_H_ */
diff --git a/drivers/gpu/drm/xe/xe_pagefault.c b/drivers/gpu/drm/xe/xe_pagefault.c
index d348ec408204..aeb56ff5d58e 100644
--- a/drivers/gpu/drm/xe/xe_pagefault.c
+++ b/drivers/gpu/drm/xe/xe_pagefault.c
@@ -253,12 +253,13 @@ static int xe_pagefault_service(struct xe_pagefault *pf)
struct xe_vma *vma = NULL;
int err;
bool atomic;
+ u32 asid = FIELD_GET(XE_PAGEFAULT_ASID_MASK, pf->consumer.id);
/* Producer flagged this fault to be nacked */
if (pf->consumer.fault_type_level == XE_PAGEFAULT_TYPE_LEVEL_NACK)
return -EFAULT;
- vm = xe_pagefault_asid_to_vm(xe, pf->consumer.asid);
+ vm = xe_pagefault_asid_to_vm(xe, asid);
if (IS_ERR(vm))
return PTR_ERR(vm);
@@ -375,7 +376,7 @@ static bool xe_pagefault_match(struct xe_pagefault *pf, u64 start,
{
struct xe_device *xe = gt_to_xe(pf->gt);
u64 page_addr = pf->consumer.page_addr;
- u32 pf_asid = pf->consumer.asid;
+ u32 pf_asid = FIELD_GET(XE_PAGEFAULT_ASID_MASK, pf->consumer.id);
xe_assert(xe, pf->consumer.alloc_state !=
XE_PAGEFAULT_ALLOC_STATE_FREE);
@@ -500,7 +501,7 @@ static bool xe_pagefault_queue_pop(struct xe_pagefault_queue *pf_queue,
align = SZ_4K;
pf_work->cache.start = ALIGN_DOWN(lpf->consumer.page_addr, align);
pf_work->cache.end = pf_work->cache.start + align;
- pf_work->cache.asid = lpf->consumer.asid;
+ pf_work->cache.asid = FIELD_GET(XE_PAGEFAULT_ASID_MASK, lpf->consumer.id);
pf_work->cache.pf = lpf;
lpf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_ACTIVE;
@@ -546,14 +547,16 @@ static void xe_pagefault_print(struct xe_pagefault *pf)
u8 engine_class = FIELD_GET(XE_PAGEFAULT_ENGINE_CLASS_MASK,
pf->consumer.engine_class_instance);
- xe_gt_info(pf->gt, "\n\tASID: %d\n"
+ xe_gt_info(pf->gt, "\n\tASID: %lu\n"
"\tFaulted Address: 0x%08x%08x\n"
"\tFaultType: %lu\n"
"\tAccessType: %lu\n"
"\tFaultLevel: %lu\n"
"\tEngineClass: %d %s\n"
- "\tEngineInstance: %lu\n",
- pf->consumer.asid,
+ "\tEngineInstance: %lu\n"
+ "\tSRCID: 0x%02lx\n",
+ FIELD_GET(XE_PAGEFAULT_ASID_MASK,
+ pf->consumer.id),
upper_32_bits(pf->consumer.page_addr),
lower_32_bits(pf->consumer.page_addr),
FIELD_GET(XE_PAGEFAULT_TYPE_MASK,
@@ -565,7 +568,9 @@ static void xe_pagefault_print(struct xe_pagefault *pf)
engine_class,
xe_hw_engine_class_to_str(engine_class),
FIELD_GET(XE_PAGEFAULT_ENGINE_INSTANCE_MASK,
- pf->consumer.engine_class_instance));
+ pf->consumer.engine_class_instance),
+ FIELD_GET(XE_PAGEFAULT_SRCID_MASK,
+ pf->consumer.id));
}
static void xe_pagefault_save_to_vm(struct xe_device *xe, struct xe_pagefault *pf)
@@ -578,7 +583,8 @@ static void xe_pagefault_save_to_vm(struct xe_device *xe, struct xe_pagefault *p
* mode, return VM anyways.
*/
down_read(&xe->usm.lock);
- vm = xa_load(&xe->usm.asid_to_vm, pf->consumer.asid);
+ vm = xa_load(&xe->usm.asid_to_vm,
+ FIELD_GET(XE_PAGEFAULT_ASID_MASK, pf->consumer.id));
if (vm)
xe_vm_get(vm);
else
@@ -619,7 +625,7 @@ static void xe_pagefault_queue_work(struct work_struct *w)
const struct xe_pagefault_ops *ops = pf->producer.ops;
void *private = pf->producer.private;
struct xe_gt *gt = pf->gt;
- u32 asid = pf->consumer.asid;
+ u32 asid = FIELD_GET(XE_PAGEFAULT_ASID_MASK, pf->consumer.id);
int err = 0;
bool invalidated = false;
diff --git a/drivers/gpu/drm/xe/xe_pagefault_types.h b/drivers/gpu/drm/xe/xe_pagefault_types.h
index 185d0813fd30..907189b73286 100644
--- a/drivers/gpu/drm/xe/xe_pagefault_types.h
+++ b/drivers/gpu/drm/xe/xe_pagefault_types.h
@@ -113,8 +113,13 @@ struct xe_pagefault {
u8 engine_class_instance;
#define XE_PAGEFAULT_ENGINE_CLASS_MASK GENMASK(3, 0)
#define XE_PAGEFAULT_ENGINE_INSTANCE_MASK GENMASK(7, 4)
- /** @consumer.asid: address space ID */
- u32 asid;
+ /**
+ * @consumer.id: address space ID and SRCID, folded into one
+ * to keep size compact
+ */
+ u32 id;
+#define XE_PAGEFAULT_ASID_MASK GENMASK(23, 0)
+#define XE_PAGEFAULT_SRCID_MASK GENMASK(31, 24)
};
/**
* @consumer.end_addr: end address of page fault,
diff --git a/drivers/gpu/drm/xe/xe_pci.c b/drivers/gpu/drm/xe/xe_pci.c
index 1e04e8ef2611..00b2b5dfb8f6 100644
--- a/drivers/gpu/drm/xe/xe_pci.c
+++ b/drivers/gpu/drm/xe/xe_pci.c
@@ -752,7 +752,6 @@ struct xe_probed_info {
* Probe from the hardware the info required by xe_info_init_early().
*/
static int xe_probe_info_early(struct xe_device *xe,
- const struct xe_device_desc *desc,
struct xe_probed_info *probed_info)
{
struct pci_dev *pdev = to_pci_dev(xe->drm.dev);
@@ -760,7 +759,7 @@ static int xe_probe_info_early(struct xe_device *xe,
probed_info->devid = pdev->device;
probed_info->revid = pdev->revision;
- xe_step_platform_get(desc->platform, probed_info->revid, &probed_info->step);
+ xe_step_platform_get(xe->desc->platform, probed_info->revid, &probed_info->step);
return 0;
}
@@ -770,10 +769,10 @@ static int xe_probe_info_early(struct xe_device *xe,
* passed to the driver at probe time from PCI ID table.
*/
static int xe_info_init_early(struct xe_device *xe,
- const struct xe_device_desc *desc,
- const struct xe_subplatform_desc *subplatform_desc,
struct xe_probed_info *probed_info)
{
+ const struct xe_subplatform_desc *subplatform_desc = xe->subplatform_desc;
+ const struct xe_device_desc *desc = xe->desc;
int err;
xe->info.devid = probed_info->devid;
@@ -836,14 +835,13 @@ static int xe_info_init_early(struct xe_device *xe,
}
static void xe_probe_tile_count(struct xe_device *xe,
- const struct xe_device_desc *desc,
struct xe_probed_info *probed_info)
{
struct xe_mmio *mmio;
u8 tile_count;
u32 mtcfg;
- probed_info->tile_count = 1 + desc->max_remote_tiles;
+ probed_info->tile_count = 1 + xe->desc->max_remote_tiles;
/*
* Probe for tile count only for platforms that support multiple
@@ -946,9 +944,10 @@ static struct xe_gt *alloc_media_gt(struct xe_tile *tile,
}
static int xe_probe_ips(struct xe_device *xe,
- const struct xe_device_desc *desc,
struct xe_probed_info *probed_info)
{
+ const struct xe_device_desc *desc = xe->desc;
+
/*
* If this platform supports GMD_ID, we'll detect the proper IP
* descriptor to use from hardware registers.
@@ -989,14 +988,13 @@ static int xe_probe_ips(struct xe_device *xe,
* Probe from the hardware the info required by xe_info_init().
*/
static int xe_probe_info(struct xe_device *xe,
- const struct xe_device_desc *desc,
struct xe_probed_info *probed_info)
{
int err;
- xe_probe_tile_count(xe, desc, probed_info);
+ xe_probe_tile_count(xe, probed_info);
- err = xe_probe_ips(xe, desc, probed_info);
+ err = xe_probe_ips(xe, probed_info);
if (err)
return err;
@@ -1010,7 +1008,6 @@ static int xe_probe_info(struct xe_device *xe,
* present in device info.
*/
static int xe_info_init(struct xe_device *xe,
- const struct xe_device_desc *desc,
struct xe_probed_info *probed_info)
{
const struct xe_ip *graphics_ip;
@@ -1208,6 +1205,8 @@ static int __xe_pci_probe(struct pci_dev *pdev, const struct xe_device_desc *des
if (IS_ERR(xe))
return PTR_ERR(xe);
+ xe->desc = desc;
+ xe->subplatform_desc = subplatform_desc;
xe->devres_group = group;
pci_set_drvdata(pdev, &xe->drm);
@@ -1216,11 +1215,11 @@ static int __xe_pci_probe(struct pci_dev *pdev, const struct xe_device_desc *des
pci_set_master(pdev);
- err = xe_probe_info_early(xe, desc, &probed_info);
+ err = xe_probe_info_early(xe, &probed_info);
if (err)
return err;
- err = xe_info_init_early(xe, desc, subplatform_desc, &probed_info);
+ err = xe_info_init_early(xe, &probed_info);
if (err)
return err;
@@ -1239,11 +1238,11 @@ static int __xe_pci_probe(struct pci_dev *pdev, const struct xe_device_desc *des
if (err)
return err;
- err = xe_probe_info(xe, desc, &probed_info);
+ err = xe_probe_info(xe, &probed_info);
if (err)
return err;
- err = xe_info_init(xe, desc, &probed_info);
+ err = xe_info_init(xe, &probed_info);
if (err)
return err;
diff --git a/drivers/gpu/drm/xe/xe_pcode.c b/drivers/gpu/drm/xe/xe_pcode.c
index d502205bb72a..266deecbb100 100644
--- a/drivers/gpu/drm/xe/xe_pcode.c
+++ b/drivers/gpu/drm/xe/xe_pcode.c
@@ -140,7 +140,7 @@ int xe_pcode_read(struct xe_tile *tile, u32 mbox, u32 *val0, u32 *val1)
int err;
mutex_lock(&tile->pcode.lock);
- err = pcode_mailbox_rw(tile, mbox, val0, val1, 1, true, false);
+ err = pcode_mailbox_rw(tile, mbox, val0, val1, PCODE_DEFAULT_TIMEOUT_MS, true, false);
mutex_unlock(&tile->pcode.lock);
return err;
diff --git a/drivers/gpu/drm/xe/xe_pcode.h b/drivers/gpu/drm/xe/xe_pcode.h
index ba8a1d1b2152..7e43792b0037 100644
--- a/drivers/gpu/drm/xe/xe_pcode.h
+++ b/drivers/gpu/drm/xe/xe_pcode.h
@@ -18,6 +18,8 @@ struct xe_pcode_version {
u32 engg;
};
+#define PCODE_DEFAULT_TIMEOUT_MS 10
+
int xe_pcode_init_early(struct xe_tile *tile);
int xe_pcode_probe_early(struct xe_device *xe);
int xe_pcode_ready(struct xe_device *xe, bool locked);
@@ -31,7 +33,7 @@ int xe_pcode_write64_timeout(struct xe_tile *tile, u32 mbox, u32 data0,
int xe_get_pcode_version(struct xe_device *xe, struct xe_pcode_version *version);
#define xe_pcode_write(tile, mbox, val) \
- xe_pcode_write_timeout(tile, mbox, val, 1)
+ xe_pcode_write_timeout(tile, mbox, val, PCODE_DEFAULT_TIMEOUT_MS)
int xe_pcode_request(struct xe_tile *tile, u32 mbox, u32 request,
u32 reply_mask, u32 reply, int timeout_ms);
diff --git a/drivers/gpu/drm/xe/xe_pt.c b/drivers/gpu/drm/xe/xe_pt.c
index 5d990c1c3740..4b351dbf6572 100644
--- a/drivers/gpu/drm/xe/xe_pt.c
+++ b/drivers/gpu/drm/xe/xe_pt.c
@@ -236,9 +236,11 @@ void xe_pt_destroy(struct xe_pt *pt, u32 flags, struct llist_head *deferred)
*/
void xe_pt_clear(struct xe_device *xe, struct xe_pt *pt)
{
- struct iosys_map *map = &pt->bo->vmap;
+ struct xe_bo *bo = pt->bo;
- xe_map_memset(xe, map, 0, 0, SZ_4K);
+ xe_bo_assert_held(bo);
+ if (!iosys_map_is_null(&bo->vmap))
+ xe_map_memset(xe, &bo->vmap, 0, 0, SZ_4K);
}
/**
@@ -831,9 +833,12 @@ xe_pt_stage_bind(struct xe_tile *tile, struct xe_vma *vma,
return -EAGAIN;
}
if (xe_svm_range_has_dma_mapping(range)) {
- xe_res_first_dma(range->pages.dma_addr, 0,
- xe_svm_range_size(range),
- &curs);
+ const struct drm_pagemap_addr *addr;
+ bool contiguous;
+
+ addr = xe_svm_range_first_dma(range, &contiguous);
+ xe_res_first_dma(addr, 0, xe_svm_range_size(range),
+ contiguous, &curs);
xe_svm_range_debug(range, "BIND PREPARE - MIXED");
} else {
xe_assert(xe, false);
@@ -865,10 +870,15 @@ xe_pt_stage_bind(struct xe_tile *tile, struct xe_vma *vma,
xe_bo_assert_held(bo);
if (!xe_vma_is_null(vma) && !range && !is_purged) {
- if (xe_vma_is_userptr(vma))
- xe_res_first_dma(to_userptr_vma(vma)->userptr.pages.dma_addr, 0,
- xe_vma_size(vma), &curs);
- else if (xe_bo_is_vram(bo) || xe_bo_is_stolen(bo))
+ if (xe_vma_is_userptr(vma)) {
+ const struct drm_pagemap_addr *addr;
+ bool contiguous;
+
+ addr = drm_gpusvm_pages_first_dma(&to_userptr_vma(vma)->userptr.pages,
+ &contiguous);
+ xe_res_first_dma(addr, 0, xe_vma_size(vma), contiguous,
+ &curs);
+ } else if (xe_bo_is_vram(bo) || xe_bo_is_stolen(bo))
xe_res_first(bo->ttm.resource, xe_vma_bo_offset(vma),
xe_vma_size(vma), &curs);
else
diff --git a/drivers/gpu/drm/xe/xe_ras.c b/drivers/gpu/drm/xe/xe_ras.c
index de4cb9ef7355..7a85735c57d5 100644
--- a/drivers/gpu/drm/xe/xe_ras.c
+++ b/drivers/gpu/drm/xe/xe_ras.c
@@ -3,6 +3,7 @@
* Copyright © 2026 Intel Corporation
*/
+#include "xe_configfs.h"
#include "xe_debugfs.h"
#include "xe_device.h"
#include "xe_drm_ras.h"
@@ -915,6 +916,14 @@ void xe_ras_init(struct xe_device *xe)
{
int ret;
+ /*
+ * TODO: Replace platform check with xe->info.has_disable_vram_page_offline
+ * once the feature flag is plumbed through device info.
+ */
+ if (xe->info.platform == XE_CRESCENTISLAND)
+ xe->ras.disable_vram_page_offline =
+ xe_configfs_get_disable_vram_page_offline(to_pci_dev(xe->drm.dev));
+
xe_drm_ras_init(xe);
if (!xe->info.has_sysctrl)
diff --git a/drivers/gpu/drm/xe/xe_res_cursor.h b/drivers/gpu/drm/xe/xe_res_cursor.h
index 0522caafd89d..c3a037e5f34e 100644
--- a/drivers/gpu/drm/xe/xe_res_cursor.h
+++ b/drivers/gpu/drm/xe/xe_res_cursor.h
@@ -233,12 +233,13 @@ static inline void xe_res_first_sg(const struct sg_table *sg,
* @dma_addr: struct drm_pagemap_addr array to walk
* @start: Start of the range
* @size: Size of the range
+ * @contiguous: Whether one entry describes the whole range
* @cur: cursor object to initialize
*
* Start walking over the range of allocations between @start and @size.
*/
static inline void xe_res_first_dma(const struct drm_pagemap_addr *dma_addr,
- u64 start, u64 size,
+ u64 start, u64 size, bool contiguous,
struct xe_res_cursor *cur)
{
XE_WARN_ON(!dma_addr);
@@ -248,7 +249,7 @@ static inline void xe_res_first_dma(const struct drm_pagemap_addr *dma_addr,
cur->node = NULL;
cur->start = start;
cur->remaining = size;
- cur->dma_seg_size = PAGE_SIZE << dma_addr->order;
+ cur->dma_seg_size = contiguous ? start + size : PAGE_SIZE << dma_addr->order;
cur->dma_start = 0;
cur->size = 0;
cur->dma_addr = dma_addr;
diff --git a/drivers/gpu/drm/xe/xe_shrinker.c b/drivers/gpu/drm/xe/xe_shrinker.c
index 83374cd57660..deb4378c1ec1 100644
--- a/drivers/gpu/drm/xe/xe_shrinker.c
+++ b/drivers/gpu/drm/xe/xe_shrinker.c
@@ -54,13 +54,40 @@ xe_shrinker_mod_pages(struct xe_shrinker *shrinker, long shrinkable, long purgea
write_unlock(&shrinker->lock);
}
-static s64 __xe_shrinker_walk(struct xe_device *xe,
+static bool __xe_shrinker_runtime_pm_get(struct xe_shrinker *shrinker)
+{
+ struct xe_device *xe = shrinker->xe;
+
+ if (xe_pm_runtime_get_if_active(xe))
+ return true;
+
+ if (xe_rpm_reclaim_safe(xe) && !ttm_bo_shrink_avoid_wait()) {
+ xe_pm_runtime_get(xe);
+ return true;
+ }
+
+ queue_work(xe->unordered_wq, &shrinker->pm_worker);
+
+ return false;
+}
+
+static void xe_shrinker_runtime_pm_put(struct xe_shrinker *shrinker, bool runtime_pm)
+{
+ if (runtime_pm)
+ xe_pm_runtime_put(shrinker->xe);
+}
+
+static int __xe_shrinker_walk(struct xe_shrinker *shrinker,
struct ttm_operation_ctx *ctx,
const struct xe_bo_shrink_flags flags,
- unsigned long to_scan, unsigned long *scanned)
+ unsigned long to_scan, unsigned long *scanned,
+ unsigned long *freed)
{
+ struct xe_device *xe = shrinker->xe;
unsigned int mem_type;
- s64 freed = 0, lret;
+ bool rpm = false;
+ int ret = 0;
+ s64 lret;
for (mem_type = XE_PL_SYSTEM; mem_type <= XE_PL_TT; ++mem_type) {
struct ttm_resource_manager *man = ttm_manager_type(&xe->ttm, mem_type);
@@ -74,23 +101,35 @@ static s64 __xe_shrinker_walk(struct xe_device *xe,
if (!man || !man->use_tt)
continue;
+ if (mem_type != XE_PL_SYSTEM && !rpm &&
+ xe_device_is_l2_flush_optimized(xe)) {
+ if (!__xe_shrinker_runtime_pm_get(shrinker))
+ break;
+ rpm = true;
+ }
+
ttm_bo_lru_for_each_reserved_guarded(&curs, man, &arg, ttm_bo) {
if (!ttm_bo_shrink_suitable(ttm_bo, ctx))
continue;
lret = xe_bo_shrink(ctx, ttm_bo, flags, scanned);
- if (lret < 0)
- return lret;
+ if (lret < 0) {
+ ret = lret;
+ goto out;
+ }
- freed += lret;
+ *freed += lret;
if (*scanned >= to_scan)
- break;
+ goto out;
}
/* Trylocks should never error, just fail. */
xe_assert(xe, !IS_ERR(ttm_bo));
}
- return freed;
+out:
+ xe_shrinker_runtime_pm_put(shrinker, rpm);
+
+ return ret;
}
/*
@@ -99,40 +138,36 @@ static s64 __xe_shrinker_walk(struct xe_device *xe,
* add writeback. This avoids stalls and explicit writebacks with light or
* moderate memory pressure.
*/
-static s64 xe_shrinker_walk(struct xe_device *xe,
+static int xe_shrinker_walk(struct xe_shrinker *shrinker,
struct ttm_operation_ctx *ctx,
const struct xe_bo_shrink_flags flags,
- unsigned long to_scan, unsigned long *scanned)
+ unsigned long to_scan, unsigned long *scanned,
+ unsigned long *freed)
{
bool no_wait_gpu = true;
struct xe_bo_shrink_flags save_flags = flags;
- s64 lret, freed;
+ int ret;
swap(no_wait_gpu, ctx->no_wait_gpu);
save_flags.writeback = false;
- lret = __xe_shrinker_walk(xe, ctx, save_flags, to_scan, scanned);
+ ret = __xe_shrinker_walk(shrinker, ctx, save_flags, to_scan, scanned,
+ freed);
swap(no_wait_gpu, ctx->no_wait_gpu);
- if (lret < 0 || *scanned >= to_scan)
- return lret;
+ if (ret || *scanned >= to_scan)
+ return ret;
- freed = lret;
if (!ctx->no_wait_gpu) {
- lret = __xe_shrinker_walk(xe, ctx, save_flags, to_scan, scanned);
- if (lret < 0)
- return lret;
- freed += lret;
- if (*scanned >= to_scan)
- return freed;
+ ret = __xe_shrinker_walk(shrinker, ctx, save_flags, to_scan, scanned,
+ freed);
+ if (ret || *scanned >= to_scan)
+ return ret;
}
- if (flags.writeback) {
- lret = __xe_shrinker_walk(xe, ctx, flags, to_scan, scanned);
- if (lret < 0)
- return lret;
- freed += lret;
- }
+ if (flags.writeback)
+ ret = __xe_shrinker_walk(shrinker, ctx, flags, to_scan, scanned,
+ freed);
- return freed;
+ return ret;
}
static unsigned long
@@ -180,22 +215,7 @@ static bool xe_shrinker_runtime_pm_get(struct xe_shrinker *shrinker, bool force,
return false;
}
- if (!xe_pm_runtime_get_if_active(xe)) {
- if (xe_rpm_reclaim_safe(xe) && !ttm_bo_shrink_avoid_wait()) {
- xe_pm_runtime_get(xe);
- return true;
- }
- queue_work(xe->unordered_wq, &shrinker->pm_worker);
- return false;
- }
-
- return true;
-}
-
-static void xe_shrinker_runtime_pm_put(struct xe_shrinker *shrinker, bool runtime_pm)
-{
- if (runtime_pm)
- xe_pm_runtime_put(shrinker->xe);
+ return __xe_shrinker_runtime_pm_get(shrinker);
}
static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_control *sc)
@@ -214,7 +234,6 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con
bool runtime_pm;
bool purgeable;
bool can_backup = !!(sc->gfp_mask & __GFP_FS);
- s64 lret;
nr_to_scan = sc->nr_to_scan;
@@ -225,12 +244,9 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con
/* Might need runtime PM. Try to wake early if it looks like it. */
runtime_pm = xe_shrinker_runtime_pm_get(shrinker, false, nr_to_scan, can_backup);
- if (purgeable && nr_scanned < nr_to_scan) {
- lret = xe_shrinker_walk(shrinker->xe, &ctx, shrink_flags,
- nr_to_scan, &nr_scanned);
- if (lret >= 0)
- freed += lret;
- }
+ if (purgeable && nr_scanned < nr_to_scan)
+ xe_shrinker_walk(shrinker, &ctx, shrink_flags,
+ nr_to_scan, &nr_scanned, &freed);
sc->nr_scanned = nr_scanned;
if (nr_scanned >= nr_to_scan || !can_backup)
@@ -242,10 +258,8 @@ static unsigned long xe_shrinker_scan(struct shrinker *shrink, struct shrink_con
shrink_flags.purge = false;
- lret = xe_shrinker_walk(shrinker->xe, &ctx, shrink_flags,
- nr_to_scan, &nr_scanned);
- if (lret >= 0)
- freed += lret;
+ xe_shrinker_walk(shrinker, &ctx, shrink_flags,
+ nr_to_scan, &nr_scanned, &freed);
sc->nr_scanned = nr_scanned;
out:
diff --git a/drivers/gpu/drm/xe/xe_svm.c b/drivers/gpu/drm/xe/xe_svm.c
index 627a741293d5..6c3033fc4db7 100644
--- a/drivers/gpu/drm/xe/xe_svm.c
+++ b/drivers/gpu/drm/xe/xe_svm.c
@@ -13,6 +13,7 @@
#include "xe_bo.h"
#include "xe_exec_queue_types.h"
#include "xe_gt_stats.h"
+#include "xe_log.h"
#include "xe_migrate.h"
#include "xe_module.h"
#include "xe_pagefault.h"
@@ -1361,9 +1362,9 @@ retry:
else
goto retry;
} else {
- drm_err(&vm->xe->drm,
- "VRAM allocation failed, retry count exceeded, asid=%u, errno=%pe\n",
- vm->usm.asid, ERR_PTR(err));
+ xe_log_err(gt, PAGEFAULT, err,
+ "VRAM allocation failed, retry count exceeded, ASID=%u\n",
+ vm->usm.asid);
goto err_out;
}
}
@@ -1384,9 +1385,9 @@ get_pages:
range_debug(range, "PAGE FAULT - RETRY PAGES");
goto retry;
} else {
- drm_err(&vm->xe->drm,
- "Get pages failed, retry count exceeded, asid=%u, gpusvm=%p, errno=%pe\n",
- vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err));
+ xe_log_err(gt, PAGEFAULT, err,
+ "Get pages failed, retry count exceeded, ASID=%u, GPUVM=%s\n",
+ vm->usm.asid, vm->svm.gpusvm.name);
}
}
if (err) {
@@ -1598,7 +1599,7 @@ int xe_svm_range_get_pages(struct xe_vm *vm, struct xe_svm_range *range,
lockdep_assert_held(&range->lock);
- err = drm_gpusvm_get_pages(&vm->svm.gpusvm, &range->pages,
+ err = drm_gpusvm_get_pages(&vm->svm.gpusvm, &range->pages, 1,
vm->svm.gpusvm.mm,
&range->base.notifier->notifier,
drm_gpusvm_range_start(&range->base),
diff --git a/drivers/gpu/drm/xe/xe_svm.h b/drivers/gpu/drm/xe/xe_svm.h
index 2a0dc0d125c9..2ef4ef026ccd 100644
--- a/drivers/gpu/drm/xe/xe_svm.h
+++ b/drivers/gpu/drm/xe/xe_svm.h
@@ -220,6 +220,19 @@ static inline unsigned long xe_svm_range_size(struct xe_svm_range *range)
return drm_gpusvm_range_size(&range->base);
}
+/**
+ * xe_svm_range_first_dma() - Resolve the device address array of a SVM range
+ * @range: SVM range
+ * @contiguous: Where to store whether one entry spans the whole range
+ *
+ * Return: Pointer to the first device address, NULL if none is populated.
+ */
+static inline const struct drm_pagemap_addr *
+xe_svm_range_first_dma(struct xe_svm_range *range, bool *contiguous)
+{
+ return drm_gpusvm_pages_first_dma(&range->pages, contiguous);
+}
+
void xe_svm_flush(struct xe_vm *vm);
int xe_pagemap_shrinker_create(struct xe_device *xe);
@@ -436,6 +449,13 @@ static inline bool xe_svm_range_is_removed(struct xe_svm_range *range)
return false;
}
+static inline const struct drm_pagemap_addr *
+xe_svm_range_first_dma(struct xe_svm_range *range, bool *contiguous)
+{
+ *contiguous = false;
+ return NULL;
+}
+
#define xe_svm_range_has_dma_mapping(...) false
#endif /* CONFIG_DRM_XE_GPUSVM */
diff --git a/drivers/gpu/drm/xe/xe_sysctrl.c b/drivers/gpu/drm/xe/xe_sysctrl.c
index 62ccc9be71b4..4067e1dfdcd5 100644
--- a/drivers/gpu/drm/xe/xe_sysctrl.c
+++ b/drivers/gpu/drm/xe/xe_sysctrl.c
@@ -13,9 +13,11 @@
#include "xe_device.h"
#include "xe_mmio.h"
#include "xe_pm.h"
+#include "xe_printk.h"
#include "xe_soc_remapper.h"
#include "xe_sysctrl.h"
#include "xe_sysctrl_mailbox.h"
+#include "xe_sysctrl_mailbox_types.h"
#include "xe_sysctrl_types.h"
/**
@@ -29,6 +31,21 @@
* This module provides initialization and support code for interacting
* with System Controller through the mailbox interface.
*/
+
+/* Application status flags reported in xe_sysctrl_app_status_resp.flags */
+#define XE_SYSCTRL_APP_RESP_VALID BIT(0)
+#define XE_SYSCTRL_APP_RESP_BOOTED BIT(1)
+#define XE_SYSCTRL_APP_RESP_INITIALIZED BIT(2)
+
+/*
+ * Known System Controller application identifiers, keyed by firmware
+ * application ID.
+ */
+enum xe_sysctrl_app_id {
+ XE_SYSCTRL_APP_OCODE = 0x0C,
+ XE_SYSCTRL_APP_DIAG = 0x0D,
+};
+
static void sysctrl_fini(void *arg)
{
struct xe_device *xe = arg;
@@ -125,3 +142,78 @@ void xe_sysctrl_pm_resume(struct xe_device *xe)
xe->soc_remapper.set_sysctrl_region(xe, SYSCTRL_MAILBOX_INDEX);
}
+
+static enum xe_sysctrl_fw_status
+xe_sysctrl_check_app_status(struct xe_device *xe, enum xe_sysctrl_app_id app_id)
+{
+ struct xe_sysctrl_app_status_req req = {};
+ struct xe_sysctrl_app_status_resp resp = {};
+ struct xe_sysctrl_mailbox_command cmd = {};
+ size_t out_len = 0;
+ u32 flags;
+ int ret;
+
+ req.app_id = (u8)app_id;
+
+ xe_sysctrl_create_command(&cmd, XE_SYSCTRL_GROUP_CORE, XE_SYSCTRL_CMD_GET_APP_STATUS_BY_ID,
+ &req, sizeof(req), &resp, sizeof(resp));
+
+ ret = xe_sysctrl_send_command(&xe->sc, &cmd, &out_len);
+ if (ret)
+ return XE_SYSCTRL_FIRMWARE_COMM_FAILURE;
+
+ if (out_len != sizeof(resp)) {
+ xe_err(xe, "sysctrl: unexpected get app status response length %zu (expected %zu)\n",
+ out_len, sizeof(resp));
+ return XE_SYSCTRL_FIRMWARE_COMM_FAILURE;
+ }
+
+ flags = resp.flags;
+
+ if (!(flags & XE_SYSCTRL_APP_RESP_VALID))
+ return XE_SYSCTRL_FIRMWARE_APP_INVALID;
+
+ if (!(flags & XE_SYSCTRL_APP_RESP_BOOTED))
+ return XE_SYSCTRL_FIRMWARE_APP_NOT_LOADED;
+
+ if (!(flags & XE_SYSCTRL_APP_RESP_INITIALIZED))
+ return XE_SYSCTRL_FIRMWARE_APP_BOOTED;
+
+ return XE_SYSCTRL_FIRMWARE_APP_INITIALIZED;
+}
+
+/**
+ * xe_sysctrl_is_oobmsm_fw_ready() - Check if oCode firmware is fully initialized
+ * @xe: xe device instance
+ *
+ * Returns true if oCode firmware has reached the initialized state, indicating
+ * it is ready to handle requests.
+ *
+ * Callers must only invoke this on platforms where System Controller is
+ * present (xe->info.has_sysctrl).
+ *
+ * Return: true if oCode firmware is initialized, false otherwise
+ */
+bool xe_sysctrl_is_oobmsm_fw_ready(struct xe_device *xe)
+{
+ return xe_sysctrl_check_app_status(xe, XE_SYSCTRL_APP_OCODE) ==
+ XE_SYSCTRL_FIRMWARE_APP_INITIALIZED;
+}
+
+/**
+ * xe_sysctrl_is_diag_fw_ready() - Check if diag firmware is fully initialized
+ * @xe: xe device instance
+ *
+ * Returns true if diag firmware has reached the initialized state, indicating
+ * it is ready to handle requests.
+ *
+ * Callers must only invoke this on platforms where System Controller is
+ * present (xe->info.has_sysctrl).
+ *
+ * Return: true if diag firmware is initialized, false otherwise
+ */
+bool xe_sysctrl_is_diag_fw_ready(struct xe_device *xe)
+{
+ return xe_sysctrl_check_app_status(xe, XE_SYSCTRL_APP_DIAG) ==
+ XE_SYSCTRL_FIRMWARE_APP_INITIALIZED;
+}
diff --git a/drivers/gpu/drm/xe/xe_sysctrl.h b/drivers/gpu/drm/xe/xe_sysctrl.h
index 090dffb6d55f..b69a3f474236 100644
--- a/drivers/gpu/drm/xe/xe_sysctrl.h
+++ b/drivers/gpu/drm/xe/xe_sysctrl.h
@@ -20,5 +20,7 @@ void xe_sysctrl_event(struct xe_sysctrl *sc);
int xe_sysctrl_init(struct xe_device *xe);
void xe_sysctrl_irq_handler(struct xe_device *xe, u32 master_ctl);
void xe_sysctrl_pm_resume(struct xe_device *xe);
+bool xe_sysctrl_is_oobmsm_fw_ready(struct xe_device *xe);
+bool xe_sysctrl_is_diag_fw_ready(struct xe_device *xe);
#endif
diff --git a/drivers/gpu/drm/xe/xe_sysctrl_mailbox_types.h b/drivers/gpu/drm/xe/xe_sysctrl_mailbox_types.h
index 66e7cbcc3f91..c236e5377f30 100644
--- a/drivers/gpu/drm/xe/xe_sysctrl_mailbox_types.h
+++ b/drivers/gpu/drm/xe/xe_sysctrl_mailbox_types.h
@@ -14,9 +14,11 @@
* enum xe_sysctrl_group - System Controller command groups
*
* @XE_SYSCTRL_GROUP_GFSP: GFSP group
+ * @XE_SYSCTRL_GROUP_CORE: Core group
*/
enum xe_sysctrl_group {
XE_SYSCTRL_GROUP_GFSP = 0x01,
+ XE_SYSCTRL_GROUP_CORE = 0xFF,
};
/**
@@ -43,6 +45,49 @@ enum xe_sysctrl_gfsp_cmd {
};
/**
+ * enum xe_sysctrl_core_cmd - Commands supported by Core group
+ *
+ * @XE_SYSCTRL_CMD_GET_APP_STATUS_BY_ID: Retrieve application status by ID
+ */
+enum xe_sysctrl_core_cmd {
+ XE_SYSCTRL_CMD_GET_APP_STATUS_BY_ID = 0x05,
+};
+
+/**
+ * struct xe_sysctrl_app_status_req - Get application status request
+ *
+ * @app_id: Application ID for which to retrieve status
+ */
+struct xe_sysctrl_app_status_req {
+ u8 app_id;
+} __packed;
+
+/**
+ * struct xe_sysctrl_app_status_resp - Get application status response
+ * @flags: Application status flags interpreted by xe_sysctrl_check_app_status()
+ */
+struct xe_sysctrl_app_status_resp {
+ u32 flags;
+} __packed;
+
+/**
+ * enum xe_sysctrl_fw_status - System Controller firmware application lifecycle states
+ *
+ * @XE_SYSCTRL_FIRMWARE_APP_INVALID: app_id is not recognized by firmware
+ * @XE_SYSCTRL_FIRMWARE_APP_NOT_LOADED: application is known but has not yet booted
+ * @XE_SYSCTRL_FIRMWARE_APP_BOOTED: boot sequence completed, post-boot init pending
+ * @XE_SYSCTRL_FIRMWARE_APP_INITIALIZED: application fully operational
+ * @XE_SYSCTRL_FIRMWARE_COMM_FAILURE: communication with System Controller firmware failed
+ */
+enum xe_sysctrl_fw_status {
+ XE_SYSCTRL_FIRMWARE_APP_INVALID,
+ XE_SYSCTRL_FIRMWARE_APP_NOT_LOADED,
+ XE_SYSCTRL_FIRMWARE_APP_BOOTED,
+ XE_SYSCTRL_FIRMWARE_APP_INITIALIZED,
+ XE_SYSCTRL_FIRMWARE_COMM_FAILURE,
+};
+
+/**
* struct xe_sysctrl_mailbox_command - System Controller mailbox command
*/
struct xe_sysctrl_mailbox_command {
diff --git a/drivers/gpu/drm/xe/xe_tile_types.h b/drivers/gpu/drm/xe/xe_tile_types.h
index 0048100ccb72..e1368c04846a 100644
--- a/drivers/gpu/drm/xe/xe_tile_types.h
+++ b/drivers/gpu/drm/xe/xe_tile_types.h
@@ -97,6 +97,10 @@ struct xe_tile {
* Only main GT has page reclaim list allocations.
*/
struct xe_sa_manager *reclaim_pool;
+#if IS_ENABLED(CONFIG_DRM_XE_DEBUG_MEM)
+ /** @mem.memtest_bo: VRAM overlap check BO */
+ struct xe_bo *memtest_bo;
+#endif
} mem;
/** @sriov: tile level virtualization data */
diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
index 05911904c1f9..9a514d983e90 100644
--- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
+++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
@@ -5,18 +5,27 @@
*/
#include <linux/cgroup_dmem.h>
+#include <linux/debugfs.h>
#include <drm/drm_managed.h>
#include <drm/drm_drv.h>
#include <drm/drm_buddy.h>
+#include <uapi/drm/xe_drm.h>
#include <drm/ttm/ttm_placement.h>
#include <drm/ttm/ttm_range_manager.h>
+#include "regs/xe_regs.h"
#include "xe_bo.h"
+#include "xe_configfs.h"
#include "xe_device.h"
+#include "xe_exec_queue.h"
+#include "xe_lrc.h"
+#include "xe_mmio.h"
#include "xe_pm.h"
+#include "xe_printk.h"
#include "xe_res_cursor.h"
+#include "xe_ttm_stolen_mgr.h"
#include "xe_ttm_vram_mgr.h"
#include "xe_vram_types.h"
@@ -49,6 +58,48 @@ static inline bool xe_is_vram_mgr_blocks_contiguous(struct gpu_buddy *mm,
return true;
}
+static int xe_ttm_vram_buddy_alloc(struct xe_ttm_vram_mgr *mgr, u64 start,
+ u64 end, u64 size, u64 min_page_size,
+ struct list_head *blocks, unsigned long flags,
+ struct ttm_resource *res, u64 *used_visible)
+{
+ struct gpu_buddy *mm = &mgr->mm;
+ struct gpu_buddy_block *block;
+ int err;
+
+ err = gpu_buddy_alloc_blocks(mm, start, end, size, min_page_size, blocks, flags);
+ if (err)
+ return err;
+
+ /*
+ * Track the owning resource, never the owning BO. A BO backpointer
+ * cached here goes stale the moment TTM hands the resource to a ghost
+ * object (ttm_buffer_object_transfer()), which happens on every
+ * accelerated move and on pipelined gutting. The resource, in
+ * contrast, has exactly the same lifetime as these blocks and TTM
+ * keeps &ttm_resource.bo pointing at the current owner for us.
+ */
+ list_for_each_entry(block, blocks, link)
+ block->private = res;
+
+ if (end <= mgr->visible_size) {
+ *used_visible = size;
+ } else {
+ list_for_each_entry(block, blocks, link) {
+ u64 blk_start = gpu_buddy_block_offset(block);
+
+ if (blk_start < mgr->visible_size) {
+ u64 blk_end = blk_start + gpu_buddy_block_size(mm, block);
+
+ *used_visible += min(blk_end, mgr->visible_size) - blk_start;
+ }
+ }
+ }
+
+ mgr->visible_avail -= *used_visible;
+ return 0;
+}
+
static int xe_ttm_vram_mgr_new(struct ttm_resource_manager *man,
struct ttm_buffer_object *tbo,
const struct ttm_place *place,
@@ -117,30 +168,12 @@ static int xe_ttm_vram_mgr_new(struct ttm_resource_manager *man,
goto error_unlock;
}
- err = gpu_buddy_alloc_blocks(mm, (u64)place->fpfn << PAGE_SHIFT,
- (u64)lpfn << PAGE_SHIFT, size,
- min_page_size, &vres->blocks, vres->flags);
+ err = xe_ttm_vram_buddy_alloc(mgr, (u64)place->fpfn << PAGE_SHIFT,
+ (u64)lpfn << PAGE_SHIFT, size,
+ min_page_size, &vres->blocks, vres->flags,
+ &vres->base, &vres->used_visible_size);
if (err)
goto error_unlock;
-
- if (lpfn <= mgr->visible_size >> PAGE_SHIFT) {
- vres->used_visible_size = size;
- } else {
- struct gpu_buddy_block *block;
-
- list_for_each_entry(block, &vres->blocks, link) {
- u64 start = gpu_buddy_block_offset(block);
-
- if (start < mgr->visible_size) {
- u64 end = start + gpu_buddy_block_size(mm, block);
-
- vres->used_visible_size +=
- min(end, mgr->visible_size) - start;
- }
- }
- }
-
- mgr->visible_avail -= vres->used_visible_size;
mutex_unlock(&mgr->lock);
if (!(vres->base.placement & TTM_PL_FLAG_CONTIGUOUS) &&
@@ -172,17 +205,61 @@ error_fini:
return err;
}
+static void xe_ttm_vram_buddy_free(struct xe_ttm_vram_mgr *mgr,
+ struct list_head *blocks,
+ u64 used_visible)
+{
+ struct gpu_buddy_block *block;
+
+ list_for_each_entry(block, blocks, link)
+ block->private = NULL;
+ gpu_buddy_free_list(&mgr->mm, blocks, 0);
+ mgr->visible_avail += used_visible;
+}
+
+/*
+ * Retry pending page-offline reservations.
+ *
+ * A reservation can fail because the blocks backing the bad page are still
+ * allocated: either the owning BO could not be purged, or the purge was
+ * pipelined and TTM handed the resource to a ghost object which frees it
+ * only once the move fences signal. Rather than giving up, entries stay on
+ * @queued_pages and are retried here every time VRAM blocks come back.
+ *
+ * Called with @mgr->lock held.
+ */
+static void xe_ttm_vram_retry_queued_pages(struct xe_ttm_vram_mgr *mgr)
+{
+ struct xe_ttm_vram_offline_resource *pos, *n;
+
+ lockdep_assert_held(&mgr->lock);
+
+ list_for_each_entry_safe(pos, n, &mgr->queued_pages, queued_link) {
+ if (xe_ttm_vram_buddy_alloc(mgr, pos->addr, pos->addr + PAGE_SIZE,
+ PAGE_SIZE, PAGE_SIZE, &pos->blocks,
+ GPU_BUDDY_RANGE_ALLOCATION, NULL,
+ &pos->used_visible_size)) {
+ pos->status = XE_PAGE_RESERVE_FAIL;
+ continue;
+ }
+ --mgr->n_queued_pages;
+ list_del_rcu(&pos->queued_link);
+ ++mgr->n_offlined_pages;
+ list_add_rcu(&pos->offlined_link, &mgr->offlined_pages);
+ }
+}
+
static void xe_ttm_vram_mgr_del(struct ttm_resource_manager *man,
struct ttm_resource *res)
{
struct xe_ttm_vram_mgr_resource *vres =
to_xe_ttm_vram_mgr_resource(res);
struct xe_ttm_vram_mgr *mgr = to_xe_ttm_vram_mgr(man);
- struct gpu_buddy *mm = &mgr->mm;
mutex_lock(&mgr->lock);
- gpu_buddy_free_list(mm, &vres->blocks, 0);
- mgr->visible_avail += vres->used_visible_size;
+ xe_ttm_vram_buddy_free(mgr, &vres->blocks, vres->used_visible_size);
+ if (unlikely(!list_empty(&mgr->queued_pages)))
+ xe_ttm_vram_retry_queued_pages(mgr);
mutex_unlock(&mgr->lock);
ttm_resource_fini(man, res);
@@ -312,12 +389,35 @@ static void xe_ttm_vram_mgr_set_unused(struct drm_device *dev, void *arg)
ttm_resource_manager_set_used(man, false);
}
+static void xe_ttm_vram_free_bad_pages(struct xe_ttm_vram_mgr *mgr)
+{
+ struct xe_ttm_vram_offline_resource *pos, *n;
+
+ list_for_each_entry_safe(pos, n, &mgr->offlined_pages, offlined_link) {
+ list_del_rcu(&pos->offlined_link);
+ xe_ttm_vram_buddy_free(mgr, &pos->blocks, pos->used_visible_size);
+ --mgr->n_offlined_pages;
+ kfree_rcu(pos, rcu);
+ }
+ list_for_each_entry_safe(pos, n, &mgr->queued_pages, queued_link) {
+ list_del_rcu(&pos->queued_link);
+ /* queued entries have no buddy reservation yet */
+ xe_ttm_vram_buddy_free(mgr, &pos->blocks, 0);
+ --mgr->n_queued_pages;
+ kfree_rcu(pos, rcu);
+ }
+}
+
static void xe_ttm_vram_mgr_fini(struct drm_device *dev, void *arg)
{
struct xe_device *xe = to_xe_device(dev);
struct xe_ttm_vram_mgr *mgr = arg;
struct ttm_resource_manager *man = &mgr->manager;
+ mutex_lock(&mgr->lock);
+ xe_ttm_vram_free_bad_pages(mgr);
+ mutex_unlock(&mgr->lock);
+
if (ttm_resource_manager_evict_all(&xe->ttm, man))
return;
@@ -344,6 +444,8 @@ int __xe_ttm_vram_mgr_init(struct xe_device *xe, struct xe_ttm_vram_mgr *mgr,
err = drmm_mutex_init(&xe->drm, &mgr->lock);
if (err)
return err;
+ INIT_LIST_HEAD(&mgr->offlined_pages);
+ INIT_LIST_HEAD(&mgr->queued_pages);
mgr->default_page_size = default_page_size;
mgr->visible_size = io_size;
mgr->visible_avail = io_size;
@@ -521,3 +623,451 @@ u64 xe_ttm_vram_get_avail(struct ttm_resource_manager *man)
return avail;
}
+
+static int xe_ttm_vram_purge_page(struct xe_device *xe, struct xe_bo *bo)
+{
+ u32 q_flag = DRM_XE_EXEC_QUEUE_BAN_REASON_PAGE_OFFLINE;
+ struct ttm_operation_ctx ctx = {};
+ struct xe_exec_queue *q_to_put = NULL;
+ struct xe_exec_queue *q = NULL;
+ struct xe_vm *vm = NULL;
+ u32 flags;
+ int ret = 0;
+
+ xe_bo_lock(bo, false);
+ if (bo->vm)
+ vm = xe_vm_get(bo->vm);
+ flags = bo->flags;
+ xe_bo_unlock(bo);
+ /* Ban VM if BO is PPGTT */
+ if (vm && (flags & XE_BO_FLAG_PAGETABLE)) {
+ struct xe_exec_queue *eq;
+ int id;
+
+ down_write(&vm->lock);
+ if (xe->info.has_ctx_tlb_inval) {
+ /*
+ * Must be the write lock: send_tlb_inval_ctx_ppgtt()
+ * mutates this list (list_move_tail() onto an on-stack
+ * head) while holding only the read lock, relying on
+ * tlb_inval->seqno_lock to keep itself the sole
+ * mutator. Traversing it under down_read() would let
+ * this walk follow entries onto that stack list.
+ */
+ down_write(&vm->exec_queues.lock);
+ for (id = 0; id < ARRAY_SIZE(vm->exec_queues.list); id++)
+ list_for_each_entry(eq, &vm->exec_queues.list[id],
+ vm_exec_queue_link)
+ atomic_or(q_flag, &eq->ban_reason);
+ up_write(&vm->exec_queues.lock);
+ } else {
+ list_for_each_entry(eq, &vm->preempt.exec_queues, lr.link)
+ atomic_or(q_flag, &eq->ban_reason);
+ }
+ smp_wmb(); /* Force all queue bits to be visible before killing the VM */
+ xe_vm_kill(vm, true);
+ up_write(&vm->lock);
+ }
+ if (vm)
+ xe_vm_put(vm);
+
+ xe_bo_lock(bo, false);
+ q = READ_ONCE(bo->q);
+ /* Ban exec queue if BO is lrc */
+ if (q && xe_exec_queue_get_unless_zero(q)) {
+ /* ban queue */
+ atomic_or(q_flag, &q->ban_reason);
+ smp_wmb(); /* Force bit change to finish before state change triggers */
+ q_to_put = q;
+ }
+
+ if (bo->purgeable.state == XE_MADV_PURGEABLE_PURGED) {
+ /* Already purged by shrinker during unlocked window — nothing to do */
+ xe_bo_unlock(bo);
+ goto out;
+ }
+
+ xe_bo_set_purgeable_state(bo, XE_MADV_PURGEABLE_DONTNEED);
+ ttm_bo_unmap_virtual(&bo->ttm); /* nuke CPU mmap + VRAM IO mappings */
+ if (xe_bo_is_pinned(bo))
+ xe_bo_unpin(bo);
+ ret = xe_ttm_bo_purge(&bo->ttm, &ctx);
+ xe_bo_unlock(bo);
+
+out:
+ if (q_to_put) {
+ xe_exec_queue_kill(q_to_put);
+ xe_exec_queue_put(q_to_put);
+ }
+
+ return ret;
+}
+
+static bool xe_ttm_vram_page_already_processed(struct xe_ttm_vram_mgr *mgr,
+ u64 addr)
+{
+ struct xe_ttm_vram_offline_resource *pos;
+
+ lockdep_assert_held(&mgr->lock);
+
+ list_for_each_entry(pos, &mgr->offlined_pages, offlined_link) {
+ if (pos->addr == addr)
+ return true;
+ }
+
+ list_for_each_entry(pos, &mgr->queued_pages, queued_link) {
+ if (pos->addr == addr)
+ return true;
+ }
+
+ return false;
+}
+
+/*
+ * Resolve the BO currently owning @block and take a reference on it.
+ *
+ * Called with @mgr->lock held, which serializes against
+ * xe_ttm_vram_buddy_free() clearing block->private.
+ *
+ * Returns NULL when there is no xe_bo we can act on: either the block is
+ * free, or the resource is temporarily owned by a TTM ghost object because
+ * a move or a pipelined gutting is still in flight. In both cases the
+ * blocks will hit xe_ttm_vram_mgr_del() on their own and the pending
+ * reservation is retried from there.
+ */
+static struct xe_bo *xe_ttm_vram_block_owner_get(struct xe_device *xe,
+ struct gpu_buddy_block *block)
+{
+ struct ttm_resource *res = block->private;
+ struct ttm_buffer_object *tbo;
+ struct xe_bo *bo;
+
+ if (!res)
+ return NULL;
+
+ guard(spinlock)(&xe->ttm.lru_lock);
+
+ /*
+ * res->bo is updated under bdev->lru_lock by ttm_resource_set_bo().
+ * Racing with a ghost transfer here is benign: we either see the old
+ * owner (whose purge is a no-op and the retry path recovers) or the
+ * ghost (rejected below).
+ *
+ * A ghost is a bare ttm_transfer_obj, not an xe_bo, so ttm_to_xe_bo()
+ * on one would be out of bounds. xe_bo_is_xe_bo() rejects it since
+ * only our own BOs carry xe_ttm_bo_destroy().
+ */
+ tbo = READ_ONCE(res->bo);
+ if (!tbo || !xe_bo_is_xe_bo(tbo))
+ return NULL;
+
+ bo = ttm_to_xe_bo(tbo);
+
+ /* The BO may already be in teardown with a zero refcount */
+ return xe_bo_get_unless_zero(bo) ? bo : NULL;
+}
+
+static int xe_ttm_vram_reserve_page_at_addr(struct xe_device *xe, u64 addr,
+ struct xe_ttm_vram_mgr *vram_mgr, struct gpu_buddy *mm)
+{
+ struct xe_ttm_vram_offline_resource *nentry;
+ struct xe_bo *pbo_to_put = NULL;
+ struct xe_bo *pbo = NULL;
+ struct gpu_buddy_block *block;
+ u64 size = PAGE_SIZE;
+ int ret = 0;
+
+ scoped_guard(mutex, &vram_mgr->lock) {
+ if (xe_ttm_vram_page_already_processed(vram_mgr, addr))
+ return -EEXIST;
+ block = gpu_buddy_allocated_addr_to_block(mm, addr);
+ if (WARN_ON(IS_ERR(block)))
+ return PTR_ERR(block);
+
+ nentry = kzalloc_obj(*nentry);
+ if (!nentry)
+ return -ENOMEM;
+ INIT_LIST_HEAD(&nentry->blocks);
+ nentry->status = XE_PAGE_RESERVE_PENDING;
+ nentry->addr = addr;
+
+ if (block) {
+ pbo = xe_ttm_vram_block_owner_get(xe, block);
+
+ /*
+ * Critical kernel BO? Best-effort check without resv lock;
+ * worst case a concurrent pin causes reset path unnecessarily.
+ */
+ if (pbo && ((pbo->ttm.type == ttm_bo_type_kernel &&
+ !(pbo->flags & XE_BO_FLAG_PINNED_LATE_RESTORE)) ||
+ (xe_bo_is_user(pbo) && xe_bo_is_pinned(pbo)))) {
+ kfree(nentry);
+ pbo_to_put = pbo;
+ drm_err(&xe->drm,
+ "%s: addr: 0x%llx is critical kernel bo, requesting SBR\n",
+ __func__, addr);
+ break;
+ }
+ /* Queue free(to-be-purged) pages */
+ ++vram_mgr->n_queued_pages;
+ list_add_rcu(&nentry->queued_link, &vram_mgr->queued_pages);
+ } else {
+ /* Immediately offline unoccupied pages */
+ /* Queue free(to-be-reserved) pages */
+ ret = xe_ttm_vram_buddy_alloc(vram_mgr, addr, addr + size,
+ size, size, &nentry->blocks,
+ GPU_BUDDY_RANGE_ALLOCATION,
+ NULL, &nentry->used_visible_size);
+ if (ret) {
+ nentry->status = XE_PAGE_RESERVE_FAIL;
+ drm_dbg(&xe->drm,
+ "Page at addr:0x%llx still busy (%d), deferring reservation\n",
+ addr, ret);
+ ++vram_mgr->n_queued_pages;
+ list_add_rcu(&nentry->queued_link, &vram_mgr->queued_pages);
+ return 0;
+ }
+ ++vram_mgr->n_offlined_pages;
+ list_add_rcu(&nentry->offlined_link, &vram_mgr->offlined_pages);
+ return ret;
+ }
+ }
+
+ /* Deferred put outside lock to avoid recursive deadlock */
+ if (pbo_to_put) {
+ xe_bo_put(pbo_to_put);
+ /* Hint System controller driver for reset with -EIO */
+ return -EIO;
+ }
+
+ if (pbo) {
+ /*
+ * Purge BO containing address - reference held from above.
+ * This does not necessarily free the blocks synchronously: if
+ * the BO is not idle, ttm_bo_pipeline_gutting() hands the
+ * resource to a ghost object and it is released only once the
+ * move fences signal. The reservation below then fails and is
+ * retried from xe_ttm_vram_mgr_del().
+ */
+ ret = xe_ttm_vram_purge_page(xe, pbo);
+ xe_bo_put(pbo);
+ if (ret)
+ drm_warn(&xe->drm, "Purge failed at addr:0x%llx, ret:%d\n", addr, ret);
+ }
+
+ return 0;
+}
+
+static struct xe_vram_region *xe_ttm_vram_addr_to_region(struct xe_device *xe, u64 addr)
+{
+ struct xe_tile *tile;
+ u8 id;
+
+ for_each_tile(tile, xe, id) {
+ struct xe_vram_region *vr = tile->mem.vram;
+
+ if (!vr)
+ continue;
+
+ if (addr >= vr->dpa_base && addr < (vr->dpa_base + vr->usable_size))
+ return vr;
+
+ /* CCS, GSM, or DSM — infrastructure zone, needs reset */
+ if (addr >= (vr->dpa_base + vr->usable_size) &&
+ addr < (vr->dpa_base + vr->actual_physical_size))
+ return NULL;
+ }
+
+ /*
+ * Return an explicit error pointer so the caller knows the addr
+ * is invalid and should be ignored, NOT SBR.
+ */
+ return ERR_PTR(-EOPNOTSUPP);
+}
+
+/**
+ * xe_ttm_vram_handle_addr_fault - Handle vram physical address error flaged
+ * @xe: pointer to parent device
+ * @addr: physical faulty address
+ *
+ * Handle the physcial faulty address error on specific tile.
+ *
+ * Returns 0 for success, negative error code otherwise as follow:
+ * * %-EIO - critical BO or address outside any VRAM region; next action is reset.
+ * * %-EOPNOTSUPP - log-only policy or unknown address; no further action.
+ * * %-ENOMEM - allocation failure; next action is reset.
+ * * %-ENXIO - address not found in buddy; no further action.
+ * * %-EEXIST - address already processed; no further action.
+ *
+ * A return of 0 means the page is tracked. It may still be listed as
+ * pending if the blocks backing it could not be freed immediately; the
+ * reservation is then completed from xe_ttm_vram_mgr_del().
+ */
+int xe_ttm_vram_handle_addr_fault(struct xe_device *xe, u64 addr)
+{
+ struct xe_ttm_vram_mgr *vram_mgr;
+ struct xe_vram_region *vr;
+ struct gpu_buddy *mm;
+
+ /* Assert that the address is PAGE_SIZE aligned */
+ if (WARN_ON_ONCE(!IS_ALIGNED(addr, PAGE_SIZE))) {
+ drm_err(&xe->drm, "Address %llx is not %lu aligned!\n", addr, PAGE_SIZE);
+ return -EINVAL;
+ }
+
+ vr = xe_ttm_vram_addr_to_region(xe, addr);
+ if (IS_ERR(vr)) {
+ /*
+ * The addr is outside VRAM and GSM.
+ * Log a debug message if needed, and safely exit/ignore.
+ */
+ drm_dbg(&xe->drm, "Address %llx is out of bounds, ignoring fault.\n", addr);
+ return PTR_ERR(vr);
+ }
+ if (!vr) {
+ drm_err(&xe->drm, "%s:%d GSM addr:%llx error requesting SBR\n",
+ __func__, __LINE__, addr);
+ /* Hint System controller driver for reset with -EIO */
+ return -EIO;
+ }
+ vram_mgr = &vr->ttm;
+ mm = &vram_mgr->mm;
+
+ if (xe->ras.disable_vram_page_offline) {
+ xe_err(xe, "0x%llx is reported as corrupted address by HW\n",
+ addr);
+ return -EOPNOTSUPP;
+ }
+
+ /* Reserve page at address */
+ return xe_ttm_vram_reserve_page_at_addr(xe, addr - vr->dpa_base, vram_mgr, mm);
+}
+EXPORT_SYMBOL(xe_ttm_vram_handle_addr_fault);
+
+/**
+ * xe_ttm_vram_inject_fault - Inject a VRAM page fault for testing
+ * @xe: xe device instance
+ *
+ * Picks the last unallocated VRAM page and reports it as faulted
+ * via xe_ttm_vram_handle_addr_fault(). Used by the fault-inject
+ * debugfs interface for testing page offlining.
+ *
+ * Note: Executing this test will permanently retire the allocated
+ * memory tracking pages. The driver must be rebinded (unbind and bind)
+ * post-test execution to reclaim the reserved space, as these pages
+ * cannot be freed or reclaimed dynamically while the current instance
+ * remains active.
+ *
+ * Return: 0 on success, negative error code on failure.
+ */
+int xe_ttm_vram_inject_fault(struct xe_device *xe)
+{
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_vram_region *vr = tile->mem.vram;
+ struct xe_ttm_vram_mgr *vram_mgr = &vr->ttm;
+ struct gpu_buddy *mm = &vram_mgr->mm;
+ u64 addr;
+
+ if (vr->actual_physical_size < PAGE_SIZE)
+ return -ENOSPC;
+
+ addr = vr->actual_physical_size - PAGE_SIZE;
+ while (addr < vr->actual_physical_size) {
+ struct gpu_buddy_block *block;
+ bool found = false;
+
+ scoped_guard(mutex, &vram_mgr->lock) {
+ block = gpu_buddy_allocated_addr_to_block(mm, addr);
+ if (!block)
+ found = true;
+ }
+
+ /*
+ * Intentional race window: xe_ttm_vram_handle_addr_fault()
+ * re-acquires vram_mgr->lock internally, so we cannot hold
+ * it here. A concurrent allocation claiming this page between
+ * the two calls is an acceptable false negative for this
+ * test-only path.
+ */
+ if (found)
+ return xe_ttm_vram_handle_addr_fault(xe, addr + vr->dpa_base);
+
+ cond_resched();
+ if (addr == 0)
+ break;
+ addr -= PAGE_SIZE;
+ }
+
+ return -ENOSPC;
+}
+EXPORT_SYMBOL(xe_ttm_vram_inject_fault);
+
+static int vram_bad_pages_show(struct seq_file *m, void *unused)
+{
+ struct xe_device *xe = m->private;
+ struct xe_ttm_vram_offline_resource *pos;
+ struct ttm_resource_manager *man;
+ struct xe_ttm_vram_mgr *mgr;
+ struct xe_tile *tile;
+ u8 id;
+
+ man = ttm_manager_type(&xe->ttm, XE_PL_VRAM0);
+ if (man)
+ /* TODO Hook with RAS to show max_pages fetched from FW */
+ seq_printf(m, "max_pages: %d\n",
+ to_xe_ttm_vram_mgr(man)->max_pages);
+
+ for_each_tile(tile, xe, id) {
+ struct xe_vram_region *vr = tile->mem.vram;
+
+ man = ttm_manager_type(&xe->ttm, XE_PL_VRAM0 + id);
+ if (!man || !vr)
+ continue;
+ mgr = to_xe_ttm_vram_mgr(man);
+
+ rcu_read_lock();
+
+ list_for_each_entry_rcu(pos, &mgr->offlined_pages, offlined_link) {
+ u64 pfn;
+
+ pfn = (pos->addr + vr->dpa_base) >> PAGE_SHIFT;
+ seq_printf(m, "0x%016llx : 0x%016lx : R\n", pfn, PAGE_SIZE);
+ }
+
+ list_for_each_entry_rcu(pos, &mgr->queued_pages, queued_link) {
+ u64 pfn;
+
+ pfn = (pos->addr + vr->dpa_base) >> PAGE_SHIFT;
+ seq_printf(m, "0x%016llx : 0x%016lx : %c\n",
+ pfn, PAGE_SIZE, pos->status ? 'F' : 'P');
+ }
+
+ rcu_read_unlock();
+ }
+
+ return 0;
+}
+DEFINE_SHOW_ATTRIBUTE(vram_bad_pages);
+
+/**
+ * xe_ttm_vram_debugfs_init - Initialize VRAM debugfs interfaces
+ * @xe: The xe device structure pointer
+ * @root: The root dentry of the debugfs directory
+ *
+ * This function registers platform-specific VRAM debugfs files used for
+ * testing and debugging. Currently, it exposes the "vram_bad_pages" interface
+ * to inspect marked faulty memory pages, restricted specifically to the
+ * %XE_CRESCENTISLAND platform.
+ *
+ * Return: Void.
+ */
+void xe_ttm_vram_debugfs_init(struct xe_device *xe, struct dentry *root)
+{
+ /*
+ * TODO: Replace platform check with xe->info
+ * once the feature flag is plumbed through device info.
+ */
+ if (xe->info.platform != XE_CRESCENTISLAND)
+ return;
+ debugfs_create_file("vram_bad_pages", 0444, root, xe, &vram_bad_pages_fops);
+}
diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.h b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.h
index 87b7fae5edba..d77f067d197b 100644
--- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.h
+++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.h
@@ -17,6 +17,7 @@ int __xe_ttm_vram_mgr_init(struct xe_device *xe, struct xe_ttm_vram_mgr *mgr,
u32 mem_type, u64 size, u64 io_size,
u64 default_page_size);
int xe_ttm_vram_mgr_init(struct xe_device *xe, struct xe_vram_region *vram);
+void xe_ttm_vram_debugfs_init(struct xe_device *xe, struct dentry *root);
int xe_ttm_vram_mgr_alloc_sgt(struct xe_device *xe,
struct ttm_resource *res,
u64 offset, u64 length,
@@ -30,6 +31,8 @@ u64 xe_ttm_vram_get_avail(struct ttm_resource_manager *man);
u64 xe_ttm_vram_get_cpu_visible_size(struct ttm_resource_manager *man);
void xe_ttm_vram_get_used(struct ttm_resource_manager *man,
u64 *used, u64 *used_visible);
+int xe_ttm_vram_handle_addr_fault(struct xe_device *xe, u64 addr);
+int xe_ttm_vram_inject_fault(struct xe_device *xe);
static inline struct xe_ttm_vram_mgr_resource *
to_xe_ttm_vram_mgr_resource(struct ttm_resource *res)
diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h b/drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h
index 9106da056b49..efcf3e1d4e80 100644
--- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h
+++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h
@@ -19,6 +19,14 @@ struct xe_ttm_vram_mgr {
struct ttm_resource_manager manager;
/** @mm: DRM buddy allocator which manages the VRAM */
struct gpu_buddy mm;
+ /** @offlined_pages: List of offlined pages */
+ struct list_head offlined_pages;
+ /** @n_offlined_pages: Number of offlined pages */
+ u16 n_offlined_pages;
+ /** @queued_pages: List of queued pages */
+ struct list_head queued_pages;
+ /** @n_queued_pages: Number of queued pages */
+ u16 n_queued_pages;
/** @visible_size: Proped size of the CPU visible portion */
u64 visible_size;
/** @visible_avail: CPU visible portion still unallocated */
@@ -29,6 +37,8 @@ struct xe_ttm_vram_mgr {
struct mutex lock;
/** @mem_type: The TTM memory type */
u32 mem_type;
+ /** @max_pages: max pages that can be in offline queue retrieved from FW */
+ u16 max_pages;
};
/**
@@ -45,4 +55,34 @@ struct xe_ttm_vram_mgr_resource {
unsigned long flags;
};
+/**
+ * enum xe_page_reserve_status - Buddy reservation status
+ * @XE_PAGE_RESERVE_PENDING: reservation in progress
+ * @XE_PAGE_RESERVE_FAIL: reservation failed
+ */
+enum xe_page_reserve_status {
+ XE_PAGE_RESERVE_PENDING = 0,
+ XE_PAGE_RESERVE_FAIL,
+};
+
+/**
+ * struct xe_ttm_vram_offline_resource - Tracks a single offlined VRAM page
+ */
+struct xe_ttm_vram_offline_resource {
+ /** @offlined_link: Link into mgr->offlined_pages */
+ struct list_head offlined_link;
+ /** @queued_link: Link into mgr->queued_pages */
+ struct list_head queued_link;
+ /** @blocks: Buddy blocks reserved for this page */
+ struct list_head blocks;
+ /** @used_visible_size: CPU-visible bytes consumed */
+ u64 used_visible_size;
+ /** @addr: Faulty DPA reported by HW */
+ u64 addr;
+ /** @status: buddy reservation status */
+ enum xe_page_reserve_status status;
+ /** @rcu: RCU head for deferred freeing */
+ struct rcu_head rcu;
+};
+
#endif
diff --git a/drivers/gpu/drm/xe/xe_userptr.c b/drivers/gpu/drm/xe/xe_userptr.c
index 90ac141fc12d..9c1dac0fce6f 100644
--- a/drivers/gpu/drm/xe/xe_userptr.c
+++ b/drivers/gpu/drm/xe/xe_userptr.c
@@ -91,7 +91,7 @@ int xe_vma_userptr_pin_pages(struct xe_userptr_vma *uvma)
if (vma->gpuva.flags & XE_VMA_DESTROYED)
return 0;
- return drm_gpusvm_get_pages(&vm->svm.gpusvm, &uvma->userptr.pages,
+ return drm_gpusvm_get_pages(&vm->svm.gpusvm, &uvma->userptr.pages, 1,
uvma->userptr.notifier.mm,
&uvma->userptr.notifier,
xe_vma_userptr(vma),
diff --git a/drivers/gpu/drm/xe/xe_vm.c b/drivers/gpu/drm/xe/xe_vm.c
index e97b061be82a..efa5ff6cc823 100644
--- a/drivers/gpu/drm/xe/xe_vm.c
+++ b/drivers/gpu/drm/xe/xe_vm.c
@@ -655,6 +655,7 @@ void xe_vm_add_fault_entry_pf(struct xe_vm *vm, struct xe_pagefault *pf)
pf->consumer.fault_type_level);
e->fault_level = FIELD_GET(XE_PAGEFAULT_LEVEL_MASK,
pf->consumer.fault_type_level);
+ e->srcid = FIELD_GET(XE_PAGEFAULT_SRCID_MASK, pf->consumer.id);
list_add_tail(&e->list, &vm->faults.list);
vm->faults.len++;
@@ -1875,6 +1876,8 @@ static void xe_vm_close(struct xe_vm *vm)
bound = drm_dev_enter(&xe->drm, &idx);
down_write(&vm->lock);
+ xe_vm_lock(vm, false);
+
if (xe_vm_in_fault_mode(vm))
xe_svm_notifier_lock(vm);
@@ -1902,6 +1905,8 @@ static void xe_vm_close(struct xe_vm *vm)
if (xe_vm_in_fault_mode(vm))
xe_svm_notifier_unlock(vm);
+
+ xe_vm_unlock(vm);
up_write(&vm->lock);
if (bound)
@@ -4277,6 +4282,11 @@ static u8 xe_to_user_fault_level(u8 fault_level)
return fault_level;
}
+static u8 xe_to_user_srcid(u8 srcid)
+{
+ return srcid;
+}
+
static int fill_faults(struct xe_vm *vm,
struct drm_xe_vm_get_property *args)
{
@@ -4304,6 +4314,8 @@ static int fill_faults(struct xe_vm *vm,
fault_entry.fault_type = xe_to_user_fault_type(entry->fault_type);
fault_entry.fault_level = xe_to_user_fault_level(entry->fault_level);
+ fault_entry.srcid = xe_to_user_srcid(entry->srcid);
+
memcpy(&fault_list[i], &fault_entry, entry_size);
i++;
diff --git a/drivers/gpu/drm/xe/xe_vm_types.h b/drivers/gpu/drm/xe/xe_vm_types.h
index 68588b624212..648031e64145 100644
--- a/drivers/gpu/drm/xe/xe_vm_types.h
+++ b/drivers/gpu/drm/xe/xe_vm_types.h
@@ -202,6 +202,7 @@ struct xe_device;
* @access_type: type of address access that resulted in fault
* @fault_type: type of fault reported
* @fault_level: fault level of the fault
+ * @srcid: ID of the faulting hardware unit
*/
struct xe_vm_fault_entry {
struct list_head list;
@@ -210,6 +211,7 @@ struct xe_vm_fault_entry {
u8 access_type;
u8 fault_type;
u8 fault_level;
+ u8 srcid;
};
struct xe_vm {
diff --git a/drivers/gpu/drm/xe/xe_vram.c b/drivers/gpu/drm/xe/xe_vram.c
index 56cff1e44530..dcedd8cfd731 100644
--- a/drivers/gpu/drm/xe/xe_vram.c
+++ b/drivers/gpu/drm/xe/xe_vram.c
@@ -17,8 +17,11 @@
#include "xe_device.h"
#include "xe_force_wake.h"
#include "xe_gt_mcr.h"
+#include "xe_map.h"
+#include "xe_migrate.h"
#include "xe_mmio.h"
#include "xe_sriov.h"
+#include "xe_tile.h"
#include "xe_tile_sriov_vf.h"
#include "xe_ttm_vram_mgr.h"
#include "xe_vram.h"
@@ -55,9 +58,6 @@ static int determine_lmem_bar_size(struct xe_device *xe, struct xe_vram_region *
/* XXX: Need to change when xe link code is ready */
lmem_bar->dpa_base = 0;
- /* set up a map to the total memory area. */
- lmem_bar->mapping = devm_ioremap_wc(&pdev->dev, lmem_bar->io_start, lmem_bar->io_size);
-
return 0;
}
@@ -196,7 +196,7 @@ static void vram_fini(void *arg)
struct xe_tile *tile;
int id;
- xe->mem.vram->mapping = NULL;
+ xe_assert(xe, !xe->mem.vram->mapping);
for_each_tile(tile, xe, id) {
tile->mem.vram->mapping = NULL;
@@ -257,8 +257,15 @@ static int vram_region_init(struct xe_device *xe, struct xe_vram_region *vram,
return -ENODEV;
}
+ if (vram != xe->mem.vram) {
+ struct pci_dev *pdev = to_pci_dev(xe->drm.dev);
+
+ vram->mapping = devm_ioremap_wc(&pdev->dev, vram->io_start, vram->io_size);
+ if (!vram->mapping)
+ return -ENOMEM;
+ }
+
vram->dpa_base = lmem_bar->dpa_base + offset;
- vram->mapping = lmem_bar->mapping + offset;
vram->usable_size = usable_size;
print_vram_region_info(xe, vram);
@@ -403,3 +410,241 @@ resource_size_t xe_vram_region_actual_physical_size(const struct xe_vram_region
return vram ? vram->actual_physical_size : 0;
}
EXPORT_SYMBOL_IF_KUNIT(xe_vram_region_actual_physical_size);
+
+#if IS_ENABLED(CONFIG_DRM_XE_DEBUG_MEM)
+static void memtest_bo_cleanup(void *arg)
+{
+ struct xe_device *xe = arg;
+
+ xe_vram_free_memtest_bos(xe);
+}
+
+int xe_vram_reserve_memtest_bo(struct xe_device *xe)
+{
+ struct xe_tile *tile;
+ u8 id;
+
+ if (IS_SRIOV_VF(xe))
+ return 0;
+
+ for_each_tile(tile, xe, id) {
+ u64 vram_size;
+
+ if (!tile->mem.vram)
+ continue;
+
+ if (tile->mem.vram->io_size < tile->mem.vram->usable_size) {
+ drm_info(&xe->drm,
+ "Tile %d: Small-BAR system detected, skipping VRAM memtest\n",
+ id);
+ continue;
+ }
+
+ vram_size = tile->mem.vram->usable_size;
+
+ tile->mem.memtest_bo = xe_bo_create_pin_map_at_novm(xe, tile, SZ_64K,
+ vram_size - SZ_64K,
+ ttm_bo_type_kernel,
+ XE_BO_FLAG_VRAM_IF_DGFX(tile),
+ 0, false);
+ if (IS_ERR(tile->mem.memtest_bo)) {
+ drm_warn(&xe->drm, "Tile %d: Failed to reserve memtest BO\n", id);
+ tile->mem.memtest_bo = NULL;
+ continue;
+ }
+
+ drm_info(&xe->drm, "Tile %d: Reserved memtest BO at offset 0x%llx\n",
+ id, vram_size - SZ_64K);
+ }
+
+ return devm_add_action_or_reset(xe->drm.dev, memtest_bo_cleanup, xe);
+}
+
+void xe_vram_free_memtest_bos(struct xe_device *xe)
+{
+ struct xe_tile *tile;
+ u8 id;
+
+ for_each_tile(tile, xe, id) {
+ if (tile->mem.memtest_bo) {
+ xe_bo_unpin_map_no_vm(tile->mem.memtest_bo);
+ tile->mem.memtest_bo = NULL;
+ }
+ }
+}
+
+int xe_vram_memtest(struct xe_device *xe)
+{
+ struct xe_tile *tile;
+ u8 id;
+ int err = 0;
+
+ if (IS_SRIOV_VF(xe))
+ return 0;
+
+ for_each_tile(tile, xe, id) {
+ struct xe_bo *last_page_bo = tile->mem.memtest_bo;
+ struct dma_fence *fence;
+ bool overlap = false;
+ int i;
+ u8 val;
+
+ if (!last_page_bo || !tile->migrate)
+ continue;
+
+ drm_info(&xe->drm, "Tile %d: Running VRAM memtest...\n", id);
+
+ /* CPU write and readback first and last byte of the last page */
+ xe_map_wr(xe, &last_page_bo->vmap, 0, u8, 0xA5);
+ xe_map_wr(xe, &last_page_bo->vmap, SZ_64K - 1, u8, 0x5A);
+
+ val = xe_map_rd(xe, &last_page_bo->vmap, 0, u8);
+ if (drm_WARN(&xe->drm, val != 0xA5,
+ "Tile %d: CPU memtest failed at offset 0 (expected 0xA5, got 0x%02x)\n",
+ id, val)) {
+ err = -EIO;
+ goto unpin;
+ }
+
+ val = xe_map_rd(xe, &last_page_bo->vmap, SZ_64K - 1, u8);
+ if (drm_WARN(&xe->drm, val != 0x5A,
+ "Tile %d: CPU memtest failed at offset 65535 (expected 0x5A, got 0x%02x)\n",
+ id, val)) {
+ err = -EIO;
+ goto unpin;
+ }
+
+ /* Non-CCS access via GPU on the last page */
+ xe_bo_lock(last_page_bo, false);
+ fence = xe_migrate_clear(tile->migrate, last_page_bo,
+ last_page_bo->ttm.resource,
+ XE_MIGRATE_CLEAR_FLAG_BO_DATA);
+ xe_bo_unlock(last_page_bo);
+
+ if (!IS_ERR(fence)) {
+ dma_fence_wait(fence, false);
+ dma_fence_put(fence);
+ } else {
+ err = PTR_ERR(fence);
+ goto unpin;
+ }
+
+ val = xe_map_rd(xe, &last_page_bo->vmap, 0, u8);
+ if (drm_WARN(&xe->drm, val != 0x00,
+ "Tile %d: GPU memtest clear failed at offset 0 (expected 0x00, got 0x%02x)\n",
+ id, val)) {
+ err = -EIO;
+ goto unpin;
+ }
+
+ /*
+ * Check for CCS overlap on the root tile.
+ *
+ * TODO: maybe extend if we ever get multi-tile + CCS. Pay
+ * special attention to the l2 flush below. Currently that is
+ * hard coded to the root tile.
+ */
+ if (!id && xe_device_has_flat_ccs(xe) &&
+ GRAPHICS_VERx100(xe) >= 2000) {
+ struct xe_bo *scratch_bo_before;
+ struct xe_bo *scratch_bo_after;
+
+ scratch_bo_before = xe_bo_create_pin_map_novm(xe, tile, SZ_64K,
+ ttm_bo_type_kernel,
+ XE_BO_FLAG_VRAM_IF_DGFX(tile),
+ false);
+ if (IS_ERR(scratch_bo_before)) {
+ err = PTR_ERR(scratch_bo_before);
+ goto unpin;
+ }
+
+ scratch_bo_after = xe_bo_create_pin_map_novm(xe, tile, SZ_64K,
+ ttm_bo_type_kernel,
+ XE_BO_FLAG_VRAM_IF_DGFX(tile),
+ false);
+ if (IS_ERR(scratch_bo_after)) {
+ xe_bo_unpin_map_no_vm(scratch_bo_before);
+ err = PTR_ERR(scratch_bo_after);
+ goto unpin;
+ }
+
+ /* Save original CCS metadata for PA 0 + */
+ err = xe_migrate_debug_ccs_overlap(tile->migrate, scratch_bo_before, false);
+ if (err) {
+ xe_bo_unpin_map_no_vm(scratch_bo_before);
+ xe_bo_unpin_map_no_vm(scratch_bo_after);
+ goto unpin;
+ }
+
+ /*
+ * Fill last page. If there is CCS overlap in the last
+ * page this will snag the raw CCS storage.
+ */
+ xe_map_memset(xe, &last_page_bo->vmap, 0, 0x5A, SZ_64K);
+ xe_device_wmb(xe);
+
+ /*
+ * Global invalidation. Some BMG SKUs will cache the BAR
+ * writes in the GPU side VRAM cache. Make sure above
+ * writes are fully flushed out to VRAM, so this is
+ * hopefully more well behaved with the CCS unit, if
+ * there is indeed CCS overlap with normal VRAM. Since
+ * there is a separate CCS cache, the CCS unit might not
+ * respect the GPU VRAM cache for CCS accesses, so opt
+ * for being super careful here.
+ */
+ xe_device_l2_flush(xe, true);
+
+ /* Use GPU to clear CCS state for PA 0 */
+ xe_map_memset(xe, &scratch_bo_after->vmap, 0, 0x00, SZ_64K);
+ err = xe_migrate_debug_ccs_overlap(tile->migrate, scratch_bo_after, true);
+ if (err) {
+ xe_bo_unpin_map_no_vm(scratch_bo_before);
+ xe_bo_unpin_map_no_vm(scratch_bo_after);
+ goto unpin;
+ }
+ /*
+ * Global invalidation. Ensure CCS caches really are
+ * nuked and the raw CCS data is visible in VRAM, for
+ * the below access.
+ */
+ xe_device_l2_flush(xe, true);
+
+ /* Check if last_page_bo was corrupted by the GPU CCS clear */
+ for (i = 0; i < SZ_64K; i += 8) {
+ u64 payload = xe_map_rd(xe, &last_page_bo->vmap, i, u64);
+
+ if (payload != 0x5A5A5A5A5A5A5A5AULL) {
+ overlap = true;
+ break;
+ }
+ }
+
+ /* Restore original CCS metadata for PA 0 + */
+ err = xe_migrate_debug_ccs_overlap(tile->migrate, scratch_bo_before, true);
+ if (err)
+ drm_warn(&xe->drm, "Failed to restore CCS metadata\n");
+
+ xe_bo_unpin_map_no_vm(scratch_bo_before);
+ xe_bo_unpin_map_no_vm(scratch_bo_after);
+ }
+
+ if (drm_WARN(&xe->drm, overlap,
+ "Tile %d: VRAM bounds overlap CCS region! VRAM sizing is incorrect.\n",
+ id)) {
+ err = -EINVAL;
+ goto unpin;
+ }
+
+ drm_info(&xe->drm, "Tile %d: VRAM memtest completed.\n", id);
+
+unpin:
+ if (err)
+ break;
+ }
+
+ xe_vram_free_memtest_bos(xe);
+
+ return err;
+}
+#endif
diff --git a/drivers/gpu/drm/xe/xe_vram.h b/drivers/gpu/drm/xe/xe_vram.h
index dd1c8bf17922..38425d81f777 100644
--- a/drivers/gpu/drm/xe/xe_vram.h
+++ b/drivers/gpu/drm/xe/xe_vram.h
@@ -23,4 +23,14 @@ resource_size_t xe_vram_region_dpa_base(const struct xe_vram_region *vram);
resource_size_t xe_vram_region_usable_size(const struct xe_vram_region *vram);
resource_size_t xe_vram_region_actual_physical_size(const struct xe_vram_region *vram);
+#if IS_ENABLED(CONFIG_DRM_XE_DEBUG_MEM)
+int xe_vram_reserve_memtest_bo(struct xe_device *xe);
+void xe_vram_free_memtest_bos(struct xe_device *xe);
+int xe_vram_memtest(struct xe_device *xe);
+#else
+static inline int xe_vram_reserve_memtest_bo(struct xe_device *xe) { return 0; }
+static inline void xe_vram_free_memtest_bos(struct xe_device *xe) {}
+static inline int xe_vram_memtest(struct xe_device *xe) { return 0; }
+#endif
+
#endif