summaryrefslogtreecommitdiff
path: root/drivers/gpu
diff options
context:
space:
mode:
Diffstat (limited to 'drivers/gpu')
-rw-r--r--drivers/gpu/drm/drm_drv.c2
-rw-r--r--drivers/gpu/drm/drm_ras.c101
-rw-r--r--drivers/gpu/drm/drm_ras_nl.c6
-rw-r--r--drivers/gpu/drm/drm_ras_nl.h4
-rw-r--r--drivers/gpu/drm/xe/Makefile1
-rw-r--r--drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h1
-rw-r--r--drivers/gpu/drm/xe/abi/xe_log_abi.h199
-rw-r--r--drivers/gpu/drm/xe/abi/xe_sigid_abi.h172
-rw-r--r--drivers/gpu/drm/xe/regs/xe_gt_regs.h6
-rw-r--r--drivers/gpu/drm/xe/tests/Makefile1
-rw-r--r--drivers/gpu/drm/xe/tests/xe_any_kunit.c213
-rw-r--r--drivers/gpu/drm/xe/tests/xe_kunit_helpers.c4
-rw-r--r--drivers/gpu/drm/xe/tests/xe_log_kunit.c553
-rw-r--r--drivers/gpu/drm/xe/tests/xe_migrate.c3
-rw-r--r--drivers/gpu/drm/xe/xe_any.h137
-rw-r--r--drivers/gpu/drm/xe/xe_debugfs.c15
-rw-r--r--drivers/gpu/drm/xe/xe_debugfs.h2
-rw-r--r--drivers/gpu/drm/xe/xe_defaults.h1
-rw-r--r--drivers/gpu/drm/xe/xe_device.c30
-rw-r--r--drivers/gpu/drm/xe/xe_device_types.h23
-rw-r--r--drivers/gpu/drm/xe/xe_drm_ras.c31
-rw-r--r--drivers/gpu/drm/xe/xe_drm_ras.h3
-rw-r--r--drivers/gpu/drm/xe/xe_exec_queue.c6
-rw-r--r--drivers/gpu/drm/xe/xe_exec_queue_types.h21
-rw-r--r--drivers/gpu/drm/xe/xe_gsc.c3
-rw-r--r--drivers/gpu/drm/xe/xe_gt.c7
-rw-r--r--drivers/gpu/drm/xe/xe_gt_debugfs.c80
-rw-r--r--drivers/gpu/drm/xe/xe_gt_idle.c3
-rw-r--r--drivers/gpu/drm/xe/xe_gt_printk.h3
-rw-r--r--drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c111
-rw-r--r--drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h6
-rw-r--r--drivers/gpu/drm/xe/xe_gt_sriov_printk.h3
-rw-r--r--drivers/gpu/drm/xe/xe_gt_stats.c7
-rw-r--r--drivers/gpu/drm/xe/xe_gt_stats_types.h22
-rw-r--r--drivers/gpu/drm/xe/xe_guc.c18
-rw-r--r--drivers/gpu/drm/xe/xe_guc_ct.c95
-rw-r--r--drivers/gpu/drm/xe/xe_guc_ct.h38
-rw-r--r--drivers/gpu/drm/xe/xe_guc_exec_queue_types.h32
-rw-r--r--drivers/gpu/drm/xe/xe_guc_pagefault.c39
-rw-r--r--drivers/gpu/drm/xe/xe_guc_pc.c15
-rw-r--r--drivers/gpu/drm/xe/xe_guc_submit.c271
-rw-r--r--drivers/gpu/drm/xe/xe_guc_types.h6
-rw-r--r--drivers/gpu/drm/xe/xe_hwmon.c88
-rw-r--r--drivers/gpu/drm/xe/xe_log.c235
-rw-r--r--drivers/gpu/drm/xe/xe_log.h194
-rw-r--r--drivers/gpu/drm/xe/xe_mert.c2
-rw-r--r--drivers/gpu/drm/xe/xe_migrate.c2
-rw-r--r--drivers/gpu/drm/xe/xe_module.c4
-rw-r--r--drivers/gpu/drm/xe/xe_module.h1
-rw-r--r--drivers/gpu/drm/xe/xe_oa.c7
-rw-r--r--drivers/gpu/drm/xe/xe_pagefault.c700
-rw-r--r--drivers/gpu/drm/xe/xe_pagefault.h74
-rw-r--r--drivers/gpu/drm/xe/xe_pagefault_types.h112
-rw-r--r--drivers/gpu/drm/xe/xe_pci.c32
-rw-r--r--drivers/gpu/drm/xe/xe_pci_error.c13
-rw-r--r--drivers/gpu/drm/xe/xe_pcode.c14
-rw-r--r--drivers/gpu/drm/xe/xe_pm.c5
-rw-r--r--drivers/gpu/drm/xe/xe_printk.h3
-rw-r--r--drivers/gpu/drm/xe/xe_pt.c6
-rw-r--r--drivers/gpu/drm/xe/xe_pxp_submit.c5
-rw-r--r--drivers/gpu/drm/xe/xe_ras.c125
-rw-r--r--drivers/gpu/drm/xe/xe_sriov_printk.h3
-rw-r--r--drivers/gpu/drm/xe/xe_sriov_vf_ccs.c1
-rw-r--r--drivers/gpu/drm/xe/xe_survivability_mode.c85
-rw-r--r--drivers/gpu/drm/xe/xe_svm.c146
-rw-r--r--drivers/gpu/drm/xe/xe_svm.h59
-rw-r--r--drivers/gpu/drm/xe/xe_tile_printk.h3
-rw-r--r--drivers/gpu/drm/xe/xe_tile_sriov_printk.h3
-rw-r--r--drivers/gpu/drm/xe/xe_userptr.c19
-rw-r--r--drivers/gpu/drm/xe/xe_vm.c299
-rw-r--r--drivers/gpu/drm/xe/xe_vm_types.h42
-rw-r--r--drivers/gpu/drm/xe/xe_wa_oob.rules2
72 files changed, 4042 insertions, 536 deletions
diff --git a/drivers/gpu/drm/drm_drv.c b/drivers/gpu/drm/drm_drv.c
index c808958a2188..cb53baa70995 100644
--- a/drivers/gpu/drm/drm_drv.c
+++ b/drivers/gpu/drm/drm_drv.c
@@ -545,6 +545,8 @@ static const char *drm_get_wedge_recovery(unsigned int opt)
return "bus-reset";
case DRM_WEDGE_RECOVERY_VENDOR:
return "vendor-specific";
+ case DRM_WEDGE_RECOVERY_COLD_RESET:
+ return "cold-reset";
default:
return NULL;
}
diff --git a/drivers/gpu/drm/drm_ras.c b/drivers/gpu/drm/drm_ras.c
index d6eab29a1394..39155fb514de 100644
--- a/drivers/gpu/drm/drm_ras.c
+++ b/drivers/gpu/drm/drm_ras.c
@@ -41,6 +41,11 @@
* Userspace must provide Node ID, Error ID.
* Clears specific error counter of a node if supported.
*
+ * 4. ERROR_REPORT: Subscribe to this multicast group to receive error events
+ *
+ * 5. ERROR_EVENT: Report an error event to userspace. The event contains device, node
+ * and error information that triggered the event.
+ *
* Node registration:
*
* - drm_ras_node_register(): Registers a new node and assigns
@@ -186,6 +191,34 @@ static int msg_reply_value(struct sk_buff *msg, u32 error_id,
value);
}
+static int msg_put_error_event_attrs(struct sk_buff *msg, struct drm_ras_node *node,
+ u32 error_id, const char *error_name, u32 value)
+{
+ int ret;
+
+ ret = nla_put_string(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_DEVICE_NAME, node->device_name);
+ if (ret)
+ return ret;
+
+ ret = nla_put_u32(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_NODE_ID, node->id);
+ if (ret)
+ return ret;
+
+ ret = nla_put_string(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_NODE_NAME, node->node_name);
+ if (ret)
+ return ret;
+
+ ret = nla_put_u32(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_ERROR_ID, error_id);
+ if (ret)
+ return ret;
+
+ ret = nla_put_string(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_ERROR_NAME, error_name);
+ if (ret)
+ return ret;
+
+ return nla_put_u32(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_ERROR_VALUE, value);
+}
+
static int doit_reply_value(struct genl_info *info, u32 node_id,
u32 error_id)
{
@@ -223,6 +256,74 @@ static int doit_reply_value(struct genl_info *info, u32 node_id,
}
/**
+ * drm_ras_nl_error_event() - Report an error event
+ * @node: Node structure
+ * @error_id: ID of the error
+ * @error_name: Name of the error
+ * @value: Value of the error counter
+ *
+ * Report an error-event to userspace using the error-report multicast group.
+ *
+ * Context: Process context only. Uses %GFP_KERNEL and multicasts to all
+ * netns, which is unbounded work; callers in interrupt handlers
+ * or other atomic context must defer event.
+ *
+ * Return: 0 on success, or negative errno on failure.
+ */
+int drm_ras_nl_error_event(struct drm_ras_node *node, u32 error_id, const char *error_name,
+ u32 value)
+{
+ struct genl_info info;
+ struct sk_buff *msg;
+ struct nlattr *hdr;
+ int ret;
+
+ if (!node || !error_name)
+ return -EINVAL;
+
+ /* Check the node is currently registered */
+ if (xa_load(&drm_ras_xa, node->id) != node)
+ return -ENOENT;
+
+ /* Currently only Error Counter events are supported */
+ if (node->type != DRM_RAS_NODE_TYPE_ERROR_COUNTER)
+ return -EOPNOTSUPP;
+
+ /* Check the error ID is within the valid range */
+ if (error_id < node->error_counter_range.first ||
+ error_id > node->error_counter_range.last)
+ return -EINVAL;
+
+ genl_info_init_ntf(&info, &drm_ras_nl_family, DRM_RAS_CMD_ERROR_EVENT);
+
+ msg = genlmsg_new(NLMSG_GOODSIZE, GFP_KERNEL);
+ if (!msg)
+ return -ENOMEM;
+
+ hdr = genlmsg_iput(msg, &info);
+ if (!hdr) {
+ ret = -EMSGSIZE;
+ goto free_msg;
+ }
+
+ ret = msg_put_error_event_attrs(msg, node, error_id, error_name, value);
+ if (ret)
+ goto cancel_msg;
+
+ genlmsg_end(msg, hdr);
+ genlmsg_multicast_allns(&drm_ras_nl_family, msg, 0, DRM_RAS_NLGRP_ERROR_REPORT);
+
+ return 0;
+
+cancel_msg:
+ genlmsg_cancel(msg, hdr);
+free_msg:
+ nlmsg_free(msg);
+ return ret;
+}
+EXPORT_SYMBOL(drm_ras_nl_error_event);
+
+/**
* drm_ras_nl_get_error_counter_dumpit() - Dump all Error Counters
* @skb: Netlink message buffer
* @cb: Callback context for multi-part dumps
diff --git a/drivers/gpu/drm/drm_ras_nl.c b/drivers/gpu/drm/drm_ras_nl.c
index dea1c1b2494e..9d3123cc9f9c 100644
--- a/drivers/gpu/drm/drm_ras_nl.c
+++ b/drivers/gpu/drm/drm_ras_nl.c
@@ -58,6 +58,10 @@ static const struct genl_split_ops drm_ras_nl_ops[] = {
},
};
+static const struct genl_multicast_group drm_ras_nl_mcgrps[] = {
+ [DRM_RAS_NLGRP_ERROR_REPORT] = { "error-report", },
+};
+
struct genl_family drm_ras_nl_family __ro_after_init = {
.name = DRM_RAS_FAMILY_NAME,
.version = DRM_RAS_FAMILY_VERSION,
@@ -66,4 +70,6 @@ struct genl_family drm_ras_nl_family __ro_after_init = {
.module = THIS_MODULE,
.split_ops = drm_ras_nl_ops,
.n_split_ops = ARRAY_SIZE(drm_ras_nl_ops),
+ .mcgrps = drm_ras_nl_mcgrps,
+ .n_mcgrps = ARRAY_SIZE(drm_ras_nl_mcgrps),
};
diff --git a/drivers/gpu/drm/drm_ras_nl.h b/drivers/gpu/drm/drm_ras_nl.h
index a398643572a5..03ec275aca92 100644
--- a/drivers/gpu/drm/drm_ras_nl.h
+++ b/drivers/gpu/drm/drm_ras_nl.h
@@ -21,6 +21,10 @@ int drm_ras_nl_get_error_counter_dumpit(struct sk_buff *skb,
int drm_ras_nl_clear_error_counter_doit(struct sk_buff *skb,
struct genl_info *info);
+enum {
+ DRM_RAS_NLGRP_ERROR_REPORT,
+};
+
extern struct genl_family drm_ras_nl_family;
#endif /* _LINUX_DRM_RAS_GEN_H */
diff --git a/drivers/gpu/drm/xe/Makefile b/drivers/gpu/drm/xe/Makefile
index 67ada1d6c2fb..7ac3954737f9 100644
--- a/drivers/gpu/drm/xe/Makefile
+++ b/drivers/gpu/drm/xe/Makefile
@@ -87,6 +87,7 @@ xe-y += xe_bb.o \
xe_hw_fence.o \
xe_irq.o \
xe_late_bind_fw.o \
+ xe_log.o \
xe_lrc.o \
xe_mem_pool.o \
xe_migrate.o \
diff --git a/drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h b/drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h
index ce5c59517528..57baa68bcba4 100644
--- a/drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h
+++ b/drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h
@@ -119,6 +119,7 @@ enum slpc_param_id {
SLPC_PARAM_STRATEGIES = 26,
SLPC_PARAM_POWER_PROFILE = 27,
SLPC_PARAM_IGNORE_EFFICIENT_FREQUENCY = 28,
+ SLPC_PARAM_SET_IBC_VERSION = 29,
SLPC_MAX_PARAM = 32,
};
diff --git a/drivers/gpu/drm/xe/abi/xe_log_abi.h b/drivers/gpu/drm/xe/abi/xe_log_abi.h
new file mode 100644
index 000000000000..d6105520173e
--- /dev/null
+++ b/drivers/gpu/drm/xe/abi/xe_log_abi.h
@@ -0,0 +1,199 @@
+/* SPDX-License-Identifier: MIT */
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#ifndef _ABI_XE_LOG_ABI_H_
+#define _ABI_XE_LOG_ABI_H_
+
+#include <linux/bits.h>
+#include <linux/bitfield.h>
+
+#include "abi/xe_sigid_abi.h"
+
+/**
+ * enum xe_log_component_bits - bits for components structure definitions
+ *
+ * Component identifiers are structured based on::
+ *
+ * COMPONENT = CLASS(8b).TYPE(8b)
+ *
+ * and the structure looks like this::
+ *
+ * ├── SYSTEM(0)
+ * │ └── ...
+ * ├── DRIVER(1)
+ * │ └── ...
+ * ├── FEATURE(2)
+ * │ └── ...
+ * ├── FIRMWARE(4)
+ * │ └── ...
+ * └── HARDWARE(8)
+ * └── ...
+ *
+ * Examples::
+ *
+ * COMPONENT(0.type) = SYSTEM.type = system component
+ * COMPONENT(1.type) = DRIVER.type = driver core component
+ * COMPONENT(3.type) = DRIVER_FEATURE.type = driver feature
+ * COMPONENT(5.type) = DRIVER_FIRMWARE.type = firmware driver component
+ * COMPONENT(9.type) = DRIVER_HARDWARE.type = hardware driver component
+ *
+ */
+enum xe_log_component_bits {
+ /* private: */
+ XE_LOG_COMPONENT_CLASS_MASK = GENMASK_U16(7, 0),
+ XE_LOG_COMPONENT_TYPE_MASK = GENMASK_U16(15, 8),
+ /* private: component classes */
+ XE_LOG_COMPONENT_CLASS_SYSTEM = 0u,
+ XE_LOG_COMPONENT_CLASS_DRIVER = 1u,
+ XE_LOG_COMPONENT_CLASS_FEATURE = 2u,
+ XE_LOG_COMPONENT_CLASS_FIRMWARE = 4u,
+ XE_LOG_COMPONENT_CLASS_HARDWARE = 8u,
+ XE_LOG_COMPONENT_CLASS_DRIVER_FEATURE = XE_LOG_COMPONENT_CLASS_DRIVER |
+ XE_LOG_COMPONENT_CLASS_FEATURE,
+ XE_LOG_COMPONENT_CLASS_DRIVER_FIRMWARE = XE_LOG_COMPONENT_CLASS_DRIVER |
+ XE_LOG_COMPONENT_CLASS_FIRMWARE,
+ XE_LOG_COMPONENT_CLASS_DRIVER_HARDWARE = XE_LOG_COMPONENT_CLASS_DRIVER |
+ XE_LOG_COMPONENT_CLASS_HARDWARE,
+ /* private: reserved identifiers */
+ XE_LOG_COMPONENT_NONE = 0u,
+};
+
+#define MAKE_XE_LOG_COMPONENT(_CLASS, type) \
+ (FIELD_PREP_CONST(XE_LOG_COMPONENT_CLASS_MASK, \
+ XE_LOG_COMPONENT_CLASS_##_CLASS) | \
+ FIELD_PREP_CONST(XE_LOG_COMPONENT_TYPE_MASK, (type)))
+
+/**
+ * enum xe_log_location_bits - bits for location structure definitions
+ *
+ * Location identifiers are structured based on::
+ *
+ * LOCATION = TYPE(8b).ID(8b)
+ *
+ * and the structure looks like this::
+ *
+ * ├── DEVICE(0)
+ * │ └── MBZ(0)
+ * ├── TILE(1)
+ * │ ├── Tile0(0)
+ * │ ├── ...
+ * │ └── TileN(n)
+ * ├── GT(1)
+ * │ ├── GT0(0)
+ * │ ├── ...
+ * │ └── GTn(n)
+ * └── ...
+ *
+ * Examples::
+ *
+ * LOCATION(0.0) = NONE
+ * LOCATION(1.0) = DEVICE.0 = "Device"
+ * LOCATION(2.1) = TILE.1 = "Tile1"
+ * LOCATION(3.2) = GT.2 = "GT2"
+ *
+ */
+enum xe_log_location_bits {
+ /* private: */
+ XE_LOG_LOCATION_TYPE_MASK = GENMASK_U16(7, 0),
+ XE_LOG_LOCATION_ID_MASK = GENMASK_U16(15, 8),
+ /* private: location types */
+ XE_LOG_LOCATION_TYPE_DEVICE = 1u,
+ XE_LOG_LOCATION_TYPE_TILE = 2u,
+ XE_LOG_LOCATION_TYPE_GT = 3u,
+ /* private: reserved identifiers */
+ XE_LOG_LOCATION_NONE = 0u,
+};
+
+#define PREP_XE_LOG_LOCATION(type, id) \
+ (FIELD_PREP(XE_LOG_LOCATION_TYPE_MASK, (type)) | \
+ FIELD_PREP(XE_LOG_LOCATION_ID_MASK, (id)))
+
+#define MAKE_XE_LOG_LOCATION(_TYPE, id) \
+ PREP_XE_LOG_LOCATION(XE_LOG_LOCATION_TYPE_##_TYPE, (id))
+
+/**
+ * DEFINE_XE_LOG_COMPONENTS() - Define log components.
+ * @define: name of the inner macro to expand.
+ *
+ * Use this super macro to define custom code for the log components.
+ * The following parameters are available for each component::
+ *
+ * define(CLASS, ID, TAG, SIGID, NAME)
+ *
+ * where:
+ *
+ * @CLASS is the component class name (without the XE_LOG_COMPONENT_CLASS_ prefix)
+ * @ID is the unique component identifier within @CLASS
+ * @TAG is unique component tag (across all components)
+ * @SIGID is the default xe_sigid for the component (without the XE_SIGID_ prefix)
+ */
+#define DEFINE_XE_LOG_COMPONENTS(define) \
+ DEFINE_XE_LOG_SOFTWARE_COMPONENTS(define) \
+ DEFINE_XE_LOG_HARDWARE_COMPONENTS(define)
+
+#define DEFINE_XE_LOG_SOFTWARE_COMPONENTS(define) \
+ /* */ \
+ define(SYSTEM, 1, PCI, IO_BUS, "Linux PCI Subsystem") \
+ define(SYSTEM, 2, DRM, SW, "DRM") \
+ /* */ \
+ define(DRIVER, 1, XE, SW, "Xe Driver") \
+ define(DRIVER, 2, PROBE, PROBE, "Driver Initialization") \
+ define(DRIVER, 3, WEDGED, WEDGED, "Device Malfunction") \
+ define(DRIVER, 4, RTP, SW, "Register Table Processing") \
+ define(DRIVER, 5, WA, SW, "Workarounds") \
+ define(DRIVER, 6, PAGEFAULT, MEM_FAULT, "Page Fault") \
+ /* */ \
+ define(DRIVER_HARDWARE, 1, REGS, IO_BUS, "Registers") \
+ define(DRIVER_HARDWARE, 2, GGTT, IO_BUS, "Global GTT") \
+ define(DRIVER_HARDWARE, 3, GT, GT_TDR, "Graphics Technology") \
+ define(DRIVER_HARDWARE, 4, LMTT, IO_BUS, "LMEM Translation Table") \
+ define(DRIVER_HARDWARE, 5, MEMIRQ, IO_BUS, "Memory Based IRQ") \
+ /* */ \
+ define(DRIVER_FEATURE, 1, PF, SW, "SR-IOV Physical Function") \
+ define(DRIVER_FEATURE, 2, VF, SW, "SR-IOV Virtual Function") \
+ define(DRIVER_FEATURE, 3, SURVIVABILITY, SURVIVABILITY, "Survivability") \
+ define(DRIVER_FEATURE, 4, RAS, SW, "Reliability, Accessibility, Serviceability") \
+ /* */ \
+ define(DRIVER_FIRMWARE, 1, GUC, RUNTIME_FW, "GuC") \
+ define(DRIVER_FIRMWARE, 2, HUC, RUNTIME_FW, "HuC") \
+ define(DRIVER_FIRMWARE, 3, GSC, RUNTIME_FW, "GSC") \
+ define(DRIVER_FIRMWARE, 16, PCODE, DEVICE_FW, "PCode") \
+ define(DRIVER_FIRMWARE, 17, SYSCTRL, DEVICE_FW, "System Controller") \
+
+#define DEFINE_XE_LOG_HARDWARE_COMPONENTS(define) \
+ define(HARDWARE, 1, DEVICE_MEMORY, DEVICE_MEMORY, "Device Memory") \
+ define(HARDWARE, 2, CORE_COMPUTE, CORE_COMPUTE, "Core Compute") \
+ /* HARDWARE, 3, RESERVED */ \
+ define(HARDWARE, 4, PCIE, PCIE, "PCIe Interface") \
+ define(HARDWARE, 5, FABRIC, FABRIC, "Fabric") \
+ define(HARDWARE, 6, SOC_INTERNAL, SOC_INTERNAL, "SoC Internal") \
+ /* eod */
+
+/**
+ * enum xe_log_component_tags - TAGs of all supported components
+ */
+enum xe_log_component_tags {
+ /* private: */
+#define MAKE_XE_LOG_COMPONENT_ENUM(_CLASS, _ID, _TAG, _SIG, _NAME) \
+ XE_LOG_COMPONENT_##_TAG = MAKE_XE_LOG_COMPONENT(_CLASS, (_ID)), \
+ XE_LOG_COMPONENT_##_CLASS##_##_ID = XE_LOG_COMPONENT_##_TAG, \
+ /* eod */
+ DEFINE_XE_LOG_COMPONENTS(MAKE_XE_LOG_COMPONENT_ENUM)
+#undef MAKE_XE_LOG_COMPONENT_ENUM
+};
+
+/**
+ * enum xe_log_component_sigids - SIGIDs of all supported components
+ */
+enum xe_log_component_sigids {
+ /* private: */
+#define MAKE_XE_LOG_COMPONENT_SIGID(_CLASS, _ID, _TAG, _SIG, _NAME) \
+ XE_LOG_COMPONENT_##_TAG##_SIGID = XE_SIGID_##_SIG, \
+ /* eod */
+ DEFINE_XE_LOG_COMPONENTS(MAKE_XE_LOG_COMPONENT_SIGID)
+#undef MAKE_XE_LOG_COMPONENT_SIGID
+};
+
+#endif
diff --git a/drivers/gpu/drm/xe/abi/xe_sigid_abi.h b/drivers/gpu/drm/xe/abi/xe_sigid_abi.h
new file mode 100644
index 000000000000..2f3440505f69
--- /dev/null
+++ b/drivers/gpu/drm/xe/abi/xe_sigid_abi.h
@@ -0,0 +1,172 @@
+/* SPDX-License-Identifier: MIT */
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#ifndef _ABI_XE_SIGID_ABI_H_
+#define _ABI_XE_SIGID_ABI_H_
+
+/**
+ * DOC: Xe Error Signatures (SIGID)
+ *
+ * What SIGID stands for
+ * ---------------------
+ *
+ * SIGID is short for *Signature Identifier*. It is a small, stable integer
+ * that names one of the *recognised fault sites* -- nothing more. It is the
+ * primary handle used for triage and maps directly to a specific report site.
+ *
+ * Numbering
+ * ---------
+ *
+ * SIGIDs are a single flat list numbered sequentially within the assigned range,
+ * in the order the fault sites were introduced. Values are stable: once assigned
+ * they are only ever appended, never renumbered or reused. A retired fault site
+ * SIGID value is deprecated in place, never re-purposed.
+ *
+ * Why this exists
+ * ---------------
+ *
+ * Today the driver reports faults with ad-hoc ``xe_err()`` / ``xe_gt_err()``
+ * strings that have no stable shape. That is fine for a human reading dmesg,
+ * but it gives fleet tooling nothing durable to match on: the wording changes
+ * between releases, lines can be rate-limited or dropped under an error storm,
+ * and there is no consistent way to ask "which recognised fault just happened?"
+ *
+ * A SIGID answers exactly that one question, identically across driver and
+ * firmware versions, and (eventually) across other Intel devices in a node.
+ *
+ * What a SIGID is not
+ * -------------------
+ *
+ * SIGID deliberately does not encode the detailed reason or the outcome. Those
+ * are carried alongside it::
+ *
+ * SIGID -> which recognised fault site is being reported
+ * severity -> how serious this instance is
+ * errno -> the failing operation's error, if available, shown with %pe
+ * message -> free-form human-readable context
+ *
+ * Severity is independent of the SIGID. The same SIGID can be reported at
+ * different severities depending on the instance and the recovery taken.
+ *
+ * When to use SIGID logging
+ * -------------------------
+ *
+ * The xe_log_*() helpers are for these recognised fault sites only --
+ * important, operator-relevant faults and events. The driver's only job is to
+ * emit the right SIGID next to the usual human-readable text.
+
+ * They are not a replacement for ``xe_info()`` / ``xe_dbg()`` / tracing, nor
+ * for one-off diagnostics; using them for ordinary logging would dilute the
+ * fault stream. Not every ``xe_err()`` needs to become a SIGID report -- only
+ * those that correspond to a published fault site.
+ *
+ * SIGID log output (dmesg vs. the machine record)
+ * -----------------------------------------------
+ *
+ * The dmesg line stays close to a normal xe error message so it remains
+ * readable for admins; the only stable, machine-matchable token on it is
+ * ``SIGID=<n>`` (``dmesg | grep SIGID=``).
+ *
+ * The full dmesg line is not an ABI: the surrounding text may change freely,
+ * and lines may be dropped. The durable record for tooling is the CPER record
+ * carrying the same SIGID (generation is a planned follow-up).
+ *
+ * How to pick a SIGID (the uniqueness rule)
+ * -----------------------------------------
+ *
+ * Pick per *report site*, not per incident. Each site emits the single most
+ * specific recognised SIGID *for that site* -- so the question is never
+ * "classify this whole failure", it is "what does this site detect?", which has
+ * one answer. A single underlying failure therefore legitimately produces a
+ * *chain* of reports from different layers, each with its own SIGID -- e.g. a
+ * GuC communication failure is reported as %XE_SIGID_RUNTIME_FW by the firmware
+ * path, the failed recovery as %XE_SIGID_GT_TDR by the reset path, and an
+ * aborted bind as %XE_SIGID_PROBE by the probe path. That chain lets triage
+ * follow a fault from origin to final effect; it is not a duplicate.
+ *
+ * If a site does not match any defined SIGID, keep using the ordinary
+ * ``xe_err()`` / ``xe_gt_err()`` logging rather than forcing a SIGID: a wrong
+ * or over-broad classification is harder to retire than a missing one. When a
+ * new report site is genuinely worth triaging, add it to the list below.
+ *
+ * Usage of the existing SIGID reports must reevaluated according to this section
+ * after making significant changes to the site that emits this SIGID.
+ *
+ * Scope: software vs hardware emitted signatures
+ * ----------------------------------------------
+ *
+ * Some SIGID represents fault sites that the *driver itself* detects and
+ * reports from the software POV: probe abort, wedged, survivability, driver-
+ * detected firmware failures, engine TDR, memory faults and IO/bus faults.
+ * These are the only values the driver assigns on its own.
+ *
+ * Signatures that *originate* in firmware or hardware are a different thing:
+ * they are produced and identified by the firmware or the hardware itself
+ * (e.g. via their own records or error counters), and the driver merely logs
+ * them as they are given to us. They are deliberately enumerated separately.
+ *
+ * The two driver-detected firmware report sites below (%XE_SIGID_RUNTIME_FW,
+ * %XE_SIGID_DEVICE_FW) are software signatures: they mark that *the driver*
+ * observed a firmware problem, not a signature reported by the firmware.
+ */
+
+/*
+ * Top level Intel Error Signature Identifiers.
+ */
+#define INTEL_SIGID_INVALID 0
+#define INTEL_SIGID_BATCH 100
+#define INTEL_SIGID_RANGE_START(n) ((n) * INTEL_SIGID_BATCH)
+#define INTEL_SIGID_RANGE_END(n) (INTEL_SIGID_RANGE_START((n) + 1) - 1)
+
+/* SIGIDs 1xx are reserved for Xe GPU software and 2xx for Xe GPU hardware */
+#define INTEL_SIGID_GPU_XE_SOFTWARE_START INTEL_SIGID_RANGE_START(1)
+#define INTEL_SIGID_GPU_XE_SOFTWARE_END INTEL_SIGID_RANGE_END(1)
+#define INTEL_SIGID_GPU_XE_HARDWARE_START INTEL_SIGID_RANGE_START(2)
+#define INTEL_SIGID_GPU_XE_HARDWARE_END INTEL_SIGID_RANGE_END(2)
+
+/**
+ * enum xe_sigid - Stable Xe Error Signature Identifiers (SIGID).
+ * @XE_SIGID_SW: Software component failure.
+ * @XE_SIGID_PROBE: Device probe/bind was aborted.
+ * @XE_SIGID_WEDGED: Device was declared wedged and is no longer usable.
+ * @XE_SIGID_SURVIVABILITY: Device entered survivability mode.
+ * @XE_SIGID_RUNTIME_FW: Driver-detected runtime firmware failure, GuC/HuC/GSC.
+ * @XE_SIGID_DEVICE_FW: Driver-detected device firmware failure, PCODE/sysctrl.
+ * @XE_SIGID_GT_TDR: Engine hang / timeout detection and recovery (reset).
+ * @XE_SIGID_MEM_FAULT: VM bind, page fault or GTT fault.
+ * @XE_SIGID_IO_BUS: Runtime PCIe / IOMMU / MMIO access fault.
+ * @XE_SIGID_HW: Generic hardware failure.
+ * @XE_SIGID_PCIE: PCIe interface errors.
+ * @XE_SIGID_DEVICE_MEMORY: Device memory errors.
+ * @XE_SIGID_CORE_COMPUTE: Compute/shader core errors.
+ * @XE_SIGID_FABRIC: Fabric errors.
+ * @XE_SIGID_SOC_INTERNAL: SoC-internal errors.
+ *
+ * Each SIGID represents the report sites the driver detects and reports.
+ * Values are numbered sequentially, are only ever appended, and are never
+ * renumbered or reused.
+ *
+ * Firmware- and hardware-originated signatures are numbered separately.
+ */
+enum xe_sigid {
+ XE_SIGID_SW = INTEL_SIGID_GPU_XE_SOFTWARE_START,
+ XE_SIGID_PROBE = INTEL_SIGID_GPU_XE_SOFTWARE_START + 1,
+ XE_SIGID_WEDGED = INTEL_SIGID_GPU_XE_SOFTWARE_START + 2,
+ XE_SIGID_SURVIVABILITY = INTEL_SIGID_GPU_XE_SOFTWARE_START + 3,
+ XE_SIGID_RUNTIME_FW = INTEL_SIGID_GPU_XE_SOFTWARE_START + 4,
+ XE_SIGID_DEVICE_FW = INTEL_SIGID_GPU_XE_SOFTWARE_START + 5,
+ XE_SIGID_GT_TDR = INTEL_SIGID_GPU_XE_SOFTWARE_START + 6,
+ XE_SIGID_MEM_FAULT = INTEL_SIGID_GPU_XE_SOFTWARE_START + 7,
+ XE_SIGID_IO_BUS = INTEL_SIGID_GPU_XE_SOFTWARE_START + 8,
+
+ XE_SIGID_HW = INTEL_SIGID_GPU_XE_HARDWARE_START,
+ XE_SIGID_PCIE = INTEL_SIGID_GPU_XE_HARDWARE_START + 1,
+ XE_SIGID_DEVICE_MEMORY = INTEL_SIGID_GPU_XE_HARDWARE_START + 2,
+ XE_SIGID_CORE_COMPUTE = INTEL_SIGID_GPU_XE_HARDWARE_START + 3,
+ XE_SIGID_FABRIC = INTEL_SIGID_GPU_XE_HARDWARE_START + 4,
+ XE_SIGID_SOC_INTERNAL = INTEL_SIGID_GPU_XE_HARDWARE_START + 5,
+};
+
+#endif
diff --git a/drivers/gpu/drm/xe/regs/xe_gt_regs.h b/drivers/gpu/drm/xe/regs/xe_gt_regs.h
index 08251c7a1a4b..48c515d91882 100644
--- a/drivers/gpu/drm/xe/regs/xe_gt_regs.h
+++ b/drivers/gpu/drm/xe/regs/xe_gt_regs.h
@@ -634,6 +634,12 @@
#define GT_GFX_RC6_LOCKED XE_REG(0x138104)
#define GT_GFX_RC6 XE_REG(0x138108)
+#define GT_IA_PERF_BIAS_REG XE_REG(0x138158)
+#define GT_BIAS REG_GENMASK(31, 16)
+#define IA_BIAS REG_GENMASK(15, 0)
+#define GT_BIAS_DEFAULT 0x10
+#define IA_BIAS_DEFAULT 0x10
+
#define GT0_PERF_LIMIT_REASONS XE_REG(0x1381a8)
/* Common performance limit reason bits - available on all platforms */
#define GT0_PERF_LIMIT_REASONS_MASK 0xde3
diff --git a/drivers/gpu/drm/xe/tests/Makefile b/drivers/gpu/drm/xe/tests/Makefile
index f7aa47f11a36..0b809a252cfa 100644
--- a/drivers/gpu/drm/xe/tests/Makefile
+++ b/drivers/gpu/drm/xe/tests/Makefile
@@ -7,6 +7,7 @@ xe_live_test-y = xe_live_test_mod.o
# Normal kunit tests
obj-$(CONFIG_DRM_XE_KUNIT_TEST) += xe_test.o
xe_test-y = xe_test_mod.o \
+ xe_any_kunit.o \
xe_args_test.o \
xe_pci_test.o \
xe_rtp_tables_test.o \
diff --git a/drivers/gpu/drm/xe/tests/xe_any_kunit.c b/drivers/gpu/drm/xe/tests/xe_any_kunit.c
new file mode 100644
index 000000000000..0a5f28894cd2
--- /dev/null
+++ b/drivers/gpu/drm/xe/tests/xe_any_kunit.c
@@ -0,0 +1,213 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#include <kunit/test.h>
+
+#include "tests/xe_kunit_helpers.h"
+#include "tests/xe_pci_test.h"
+#include "xe_any.h"
+#include "xe_device.h"
+
+static void test_to_xe(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+ const struct xe_device *cxe = xe;
+ const struct xe_tile *ctile = tile;
+ const struct xe_gt *cgt = gt;
+
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(xe));
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(tile));
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(gt));
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(drm));
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(dev));
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(pdev));
+
+ KUNIT_EXPECT_PTR_EQ(test, cxe, xe_any_to_xe(cxe));
+ KUNIT_EXPECT_PTR_EQ(test, cxe, xe_any_to_xe(ctile));
+ KUNIT_EXPECT_PTR_EQ(test, cxe, xe_any_to_xe(cgt));
+}
+
+static void test_to_pdev(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(xe));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(tile));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(gt));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(drm));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(dev));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(pdev));
+
+ /* mimic early probe stage */
+ dev_set_drvdata(xe->drm.dev, NULL);
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(dev));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(pdev));
+}
+
+static void test_to_dev(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(xe));
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(tile));
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(gt));
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(drm));
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(dev));
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(pdev));
+
+ /* mimic early probe stage */
+ dev_set_drvdata(xe->drm.dev, NULL);
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(dev));
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(pdev));
+}
+
+static void test_to_drm(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(xe));
+ KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(tile));
+ KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(gt));
+ KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(drm));
+ KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(dev));
+ KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(pdev));
+}
+
+static void test_if_pdev(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_TRUE(test, xe && drm && tile && gt && dev && pdev);
+ KUNIT_EXPECT_NULL(test, xe_any_if_pdev(xe));
+ KUNIT_EXPECT_NULL(test, xe_any_if_pdev(tile));
+ KUNIT_EXPECT_NULL(test, xe_any_if_pdev(gt));
+ KUNIT_EXPECT_NULL(test, xe_any_if_pdev(drm));
+ KUNIT_EXPECT_NULL(test, xe_any_if_pdev(dev));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_if_pdev(pdev));
+}
+
+static void test_if_xe(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_TRUE(test, xe && drm && tile && gt && dev && pdev);
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_if_xe(xe));
+ KUNIT_EXPECT_NULL(test, xe_any_if_xe(tile));
+ KUNIT_EXPECT_NULL(test, xe_any_if_xe(gt));
+ KUNIT_EXPECT_NULL(test, xe_any_if_xe(drm));
+ KUNIT_EXPECT_NULL(test, xe_any_if_xe(dev));
+ KUNIT_EXPECT_NULL(test, xe_any_if_xe(pdev));
+}
+
+static void test_if_tile(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_TRUE(test, xe && drm && tile && gt && dev && pdev);
+ KUNIT_EXPECT_NULL(test, xe_any_if_tile(xe));
+ KUNIT_EXPECT_PTR_EQ(test, tile, xe_any_if_tile(tile));
+ KUNIT_EXPECT_NULL(test, xe_any_if_tile(gt));
+ KUNIT_EXPECT_NULL(test, xe_any_if_tile(drm));
+ KUNIT_EXPECT_NULL(test, xe_any_if_tile(dev));
+ KUNIT_EXPECT_NULL(test, xe_any_if_tile(pdev));
+}
+
+static void test_if_gt(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_TRUE(test, xe && drm && tile && gt && dev && pdev);
+ KUNIT_EXPECT_NULL(test, xe_any_if_gt(xe));
+ KUNIT_EXPECT_NULL(test, xe_any_if_gt(tile));
+ KUNIT_EXPECT_PTR_EQ(test, gt, xe_any_if_gt(gt));
+ KUNIT_EXPECT_NULL(test, xe_any_if_gt(drm));
+ KUNIT_EXPECT_NULL(test, xe_any_if_gt(dev));
+ KUNIT_EXPECT_NULL(test, xe_any_if_gt(pdev));
+}
+
+static void test_to_id(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+ const struct xe_device *cxe = xe;
+ const struct xe_tile *ctile = tile;
+ const struct xe_gt *cgt = gt;
+
+ tile->id = 1;
+ gt->info.id = 2;
+
+ KUNIT_EXPECT_EQ(test, 0, xe_any_id(xe));
+ KUNIT_EXPECT_EQ(test, 1, xe_any_id(tile));
+ KUNIT_EXPECT_EQ(test, 2, xe_any_id(gt));
+ KUNIT_EXPECT_EQ(test, 0, xe_any_id(dev));
+ KUNIT_EXPECT_EQ(test, 0, xe_any_id(pdev));
+ KUNIT_EXPECT_EQ(test, 0, xe_any_id(cxe));
+ KUNIT_EXPECT_EQ(test, 1, xe_any_id(ctile));
+ KUNIT_EXPECT_EQ(test, 2, xe_any_id(cgt));
+}
+
+static struct kunit_case xe_any_tests[] = {
+ KUNIT_CASE(test_to_xe),
+ KUNIT_CASE(test_to_dev),
+ KUNIT_CASE(test_to_pdev),
+ KUNIT_CASE(test_to_drm),
+ KUNIT_CASE(test_if_pdev),
+ KUNIT_CASE(test_if_xe),
+ KUNIT_CASE(test_if_tile),
+ KUNIT_CASE(test_if_gt),
+ KUNIT_CASE(test_to_id),
+ {}
+};
+
+static struct kunit_suite xe_any_test_suite = {
+ .name = "xe_any",
+ .test_cases = xe_any_tests,
+ .init = xe_kunit_helper_xe_device_test_init,
+};
+
+kunit_test_suite(xe_any_test_suite);
diff --git a/drivers/gpu/drm/xe/tests/xe_kunit_helpers.c b/drivers/gpu/drm/xe/tests/xe_kunit_helpers.c
index bc5156966ce9..27740b40c8ae 100644
--- a/drivers/gpu/drm/xe/tests/xe_kunit_helpers.c
+++ b/drivers/gpu/drm/xe/tests/xe_kunit_helpers.c
@@ -39,6 +39,10 @@ struct xe_device *xe_kunit_helper_alloc_xe_device(struct kunit *test,
struct xe_device,
drm, DRIVER_GEM);
KUNIT_ASSERT_NOT_ERR_OR_NULL(test, xe);
+
+ dev_set_drvdata(xe->drm.dev, &xe->drm);
+ KUNIT_ASSERT_PTR_EQ(test, xe, kdev_to_xe_device(dev));
+
return xe;
}
EXPORT_SYMBOL_IF_KUNIT(xe_kunit_helper_alloc_xe_device);
diff --git a/drivers/gpu/drm/xe/tests/xe_log_kunit.c b/drivers/gpu/drm/xe/tests/xe_log_kunit.c
new file mode 100644
index 000000000000..56c36c8a08e1
--- /dev/null
+++ b/drivers/gpu/drm/xe/tests/xe_log_kunit.c
@@ -0,0 +1,553 @@
+// SPDX-License-Identifier: GPL-2.0 AND MIT
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#include <kunit/static_stub.h>
+#include <kunit/test.h>
+#include <kunit/test-bug.h>
+
+#include "tests/xe_kunit_helpers.h"
+#include "tests/xe_pci_test.h"
+#include "xe_device.h"
+#include "xe_log.h"
+
+static void nop_dmesg_vprintk(struct pci_dev *pdev, int cper_sev, struct va_format *vaf)
+{
+}
+
+static void nop_emit_cper(struct pci_dev *pdev, int cper_sev,
+ enum xe_sigid sigid, u32 component, u32 location,
+ const void *data, size_t len, struct va_format *vaf)
+{
+}
+
+static const char *component_name(u32 component)
+{
+ switch (component) {
+#define make_component_tag_case(_CLASS, _ID, _TAG, _SIG, _NAME) \
+ case XE_LOG_COMPONENT_##_TAG: return _NAME;
+ DEFINE_XE_LOG_COMPONENTS(make_component_tag_case)
+#undef make_component_tag_case
+ }
+ return component ? "???" : "";
+}
+
+static const char *location_type(u32 location)
+{
+ u32 type = FIELD_GET(XE_LOG_LOCATION_TYPE_MASK, location);
+
+ return type == XE_LOG_LOCATION_TYPE_DEVICE ? "DEVICE" :
+ type == XE_LOG_LOCATION_TYPE_TILE ? "TILE" :
+ type == XE_LOG_LOCATION_TYPE_GT ? "GT" :
+ location ? "?" : "";
+}
+
+static void fake_emit_cper(struct pci_dev *pdev, int cper_sev,
+ enum xe_sigid sigid, u32 component, u32 location,
+ const void *data, size_t len, struct va_format *vaf)
+{
+ char msg[64];
+ int n;
+
+ pr_info("\n");
+ pr_info("CPER SEV=%u SIGID=%u\n", cper_sev, sigid);
+ pr_info("CPER DEVICE=%s\n", dev_name(&pdev->dev));
+ if (location)
+ pr_info("CPER LOCATION=%#x \t# %s.%u\n",
+ location, location_type(location),
+ FIELD_GET(XE_LOG_LOCATION_ID_MASK, location));
+ if (component)
+ pr_info("CPER COMPONENT=%#x \t# %s\n",
+ component, component_name(component));
+ if (IS_ERR(data))
+ pr_info("CPER ERR=%ld \t\t# %pe\n", PTR_ERR(data), data);
+ else if (len)
+ print_hex_dump(KERN_INFO, "CPER BIN=", DUMP_PREFIX_OFFSET,
+ 16, 1, data, len, false);
+
+ n = vscnprintf(msg, sizeof(msg), vaf->fmt, *vaf->va);
+ print_hex_dump(KERN_INFO, "CPER MSG=", DUMP_PREFIX_OFFSET, 16, 1, msg, n, true);
+ pr_info("CPER END\n");
+}
+
+static const u8 blob[] = { 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12 };
+static const u8 dead[] = { 0xde, 0xad, 0xbe, 0xef };
+
+static struct xe_tile *to_tile_safe(struct xe_device *xe)
+{
+ return xe ? &xe->tiles[1] : NULL;
+}
+
+static struct xe_gt *to_gt_safe(struct xe_device *xe)
+{
+ return xe ? to_tile_safe(xe)->primary_gt : NULL;
+}
+
+static struct pci_dev *to_pdev_safe(struct xe_device *xe)
+{
+ return xe ? xe_any_to_pdev(xe) : NULL;
+}
+
+static void demo(struct xe_device *xe)
+{
+ struct pci_dev *pdev = xe_any_to_pdev(xe);
+ struct xe_tile *tile = to_tile_safe(xe);
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ /* SW errno */
+ xe_log_emit(pdev, CPER_SEV_FATAL, XE_SIGID_PROBE,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ ERR_PTR(-ENODEV), 0, "testing %s signature\n", "software");
+
+ xe_log_err(tile, PROBE, -ENODEV, "testing %s signature\n", "software");
+ xe_log_err_corrected(gt, PROBE, -ENODEV, "testing %s signature\n", "software");
+ xe_log_info(gt, PROBE, "testing %s signature\n", "software");
+
+ /* HW data */
+ xe_log_emit(pdev, CPER_SEV_FATAL, XE_SIGID_PCIE,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ blob, sizeof(blob), "testing %s signature\n", "HARDWARE");
+ xe_log_emit_recoverable(pdev, XE_SIGID_FABRIC,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ blob, sizeof(blob), "testing %s signature\n", "HARDWARE");
+ xe_log_from_corrected(tile, XE_SIGID_DEVICE_MEMORY, XE_LOG_COMPONENT_NONE,
+ blob, sizeof(blob), "testing %s signature\n", "HARDWARE");
+ xe_log_from_info(gt, XE_SIGID_CORE_COMPUTE, XE_LOG_COMPONENT_NONE,
+ blob, sizeof(blob), "testing %s signature\n", "HARDWARE");
+}
+
+static void demo_dmesg(struct kunit *test)
+{
+ kunit_activate_static_stub(test, log_emit_cper, nop_emit_cper);
+ demo(test->priv);
+}
+
+static void demo_cper(struct kunit *test)
+{
+ kunit_activate_static_stub(test, log_emit_cper, fake_emit_cper);
+ kunit_activate_static_stub(test, log_dmesg_vprintk, nop_dmesg_vprintk);
+ demo(test->priv);
+}
+
+static const char *test_fatal(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit_fatal(pdev, XE_SIGID_PROBE,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ NULL, 0, "testing %d\n", 123);
+ return "SIGID=101 FATAL testing 123\n";
+}
+
+static const char *test_fatal_tile(struct xe_device *xe)
+{
+ struct xe_tile *tile = to_tile_safe(xe);
+
+ if (tile)
+ xe_log_from_fatal(tile, XE_SIGID_PROBE, XE_LOG_COMPONENT_NONE,
+ ERR_PTR(-ENODEV), 0, "testing %d\n", 123);
+ return "SIGID=101 FATAL (-ENODEV) Tile1: testing 123\n";
+}
+
+static const char *test_fatal_gt(struct xe_device *xe)
+{
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (gt)
+ xe_log_from_fatal(gt, XE_SIGID_PROBE, XE_LOG_COMPONENT_NONE,
+ ERR_PTR(-ENODEV), 0, "testing %d\n", 123);
+ return "SIGID=101 FATAL (-ENODEV) Tile1: GT1: testing 123\n";
+}
+
+static const char *test_fatal_comp(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_err_fatal(pdev, PROBE, -ENODEV, "testing %d\n", 123);
+ return "SIGID=101 FATAL (-ENODEV) PROBE: testing 123\n";
+}
+
+static const char *test_fatal_comp_tile(struct xe_device *xe)
+{
+ struct xe_tile *tile = to_tile_safe(xe);
+
+ if (tile)
+ xe_log_err_fatal(tile, PROBE, -ENODEV, "testing %d\n", 123);
+ return "SIGID=101 FATAL (-ENODEV) Tile1: PROBE: testing 123\n";
+}
+
+static const char *test_fatal_comp_gt(struct xe_device *xe)
+{
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (gt)
+ xe_log_err_fatal(gt, PROBE, -ENODEV, "testing %d\n", 123);
+ return "SIGID=101 FATAL (-ENODEV) Tile1: GT1: PROBE: testing 123\n";
+}
+
+static const char *test_fatal_all(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (pdev)
+ xe_log_emit(pdev, CPER_SEV_FATAL, XE_SIGID_PROBE,
+ XE_LOG_COMPONENT_PROBE, MAKE_XE_LOG_LOCATION(GT, gt->info.id),
+ ERR_PTR(-ENODEV), 0, "testing %d\n", 123);
+ return "SIGID=101 FATAL (-ENODEV) Tile1: GT1: PROBE: testing 123\n";
+}
+
+static const char *test_recoverable(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit_recoverable(pdev, XE_SIGID_RUNTIME_FW,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ ERR_PTR(-EIO), 0, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) testing 123\n";
+}
+
+static const char *test_recoverable_tile(struct xe_device *xe)
+{
+ struct xe_tile *tile = to_tile_safe(xe);
+
+ if (tile)
+ xe_log_from_recoverable(tile, XE_SIGID_RUNTIME_FW, XE_LOG_COMPONENT_NONE,
+ ERR_PTR(-EIO), 0, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) Tile1: testing 123\n";
+}
+
+static const char *test_recoverable_gt(struct xe_device *xe)
+{
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (gt)
+ xe_log_from_recoverable(gt, XE_SIGID_RUNTIME_FW, XE_LOG_COMPONENT_NONE,
+ ERR_PTR(-EIO), 0, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) Tile1: GT1: testing 123\n";
+}
+
+static const char *test_recoverable_comp(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_err(pdev, GUC, -EIO, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) GUC: testing 123\n";
+}
+
+static const char *test_recoverable_comp_tile(struct xe_device *xe)
+{
+ struct xe_tile *tile = to_tile_safe(xe);
+
+ if (tile)
+ xe_log_err(tile, GUC, -EIO, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) Tile1: GUC: testing 123\n";
+}
+
+static const char *test_recoverable_comp_gt(struct xe_device *xe)
+{
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (gt)
+ xe_log_err(gt, GUC, -EIO, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) Tile1: GT1: GUC: testing 123\n";
+}
+
+static const char *test_recoverable_all(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (pdev)
+ xe_log_emit(pdev, CPER_SEV_RECOVERABLE, XE_SIGID_RUNTIME_FW,
+ XE_LOG_COMPONENT_GUC, MAKE_XE_LOG_LOCATION(GT, gt->info.id),
+ ERR_PTR(-EIO), 0, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) Tile1: GT1: GUC: testing 123\n";
+}
+
+static const char *test_info(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit(pdev, CPER_SEV_INFORMATIONAL, XE_SIGID_DEVICE_FW,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ NULL, 0, "testing %d\n", 123);
+ return "SIGID=105 testing 123\n";
+}
+
+static const char *test_info_tile(struct xe_device *xe)
+{
+ struct xe_tile *tile = to_tile_safe(xe);
+
+ if (tile)
+ xe_log_from_info(tile, XE_SIGID_DEVICE_FW, XE_LOG_COMPONENT_NONE,
+ NULL, 0, "testing %d\n", 123);
+ return "SIGID=105 Tile1: testing 123\n";
+}
+
+static const char *test_info_gt(struct xe_device *xe)
+{
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (gt)
+ xe_log_from_info(gt, XE_SIGID_DEVICE_FW, XE_LOG_COMPONENT_NONE,
+ NULL, 0, "testing %d\n", 123);
+ return "SIGID=105 Tile1: GT1: testing 123\n";
+}
+
+static const char *test_info_err(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_err_info(pdev, PCODE, -EPROTO, "testing %d\n", 123);
+ return "SIGID=105 (-EPROTO) PCODE: testing 123\n";
+}
+
+static const char *test_info_comp(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_info(pdev, PCODE, "testing %d\n", 123);
+ return "SIGID=105 PCODE: testing 123\n";
+}
+
+static const char *test_info_comp_tile(struct xe_device *xe)
+{
+ struct xe_tile *tile = to_tile_safe(xe);
+
+ if (tile)
+ xe_log_info(tile, PCODE, "testing %d\n", 123);
+ return "SIGID=105 Tile1: PCODE: testing 123\n";
+}
+
+static const char *test_info_comp_gt(struct xe_device *xe)
+{
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (gt)
+ xe_log_info(gt, PCODE, "testing %d\n", 123);
+ return "SIGID=105 Tile1: GT1: PCODE: testing 123\n";
+}
+
+static const char *test_info_all(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (pdev)
+ xe_log_emit(pdev, CPER_SEV_INFORMATIONAL, XE_SIGID_DEVICE_FW,
+ XE_LOG_COMPONENT_PCODE, MAKE_XE_LOG_LOCATION(GT, gt->info.id),
+ NULL, 0, "testing %d\n", 123);
+ return "SIGID=105 Tile1: GT1: PCODE: testing 123\n";
+}
+
+static const char *test_hw_fatal(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit_fatal(pdev, XE_SIGID_PCIE,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ dead, sizeof(dead), "testing %d\n", 123);
+ return "SIGID=201 FATAL (deadbeef) " HW_ERR "testing 123\n";
+}
+
+static const char *test_hw_recoverable(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit_recoverable(pdev, XE_SIGID_DEVICE_MEMORY,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ dead, sizeof(dead), "testing %d\n", 123);
+ return "SIGID=202 (deadbeef) " HW_ERR "testing 123\n";
+}
+
+static const char *test_hw_corrected(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit_corrected(pdev, XE_SIGID_DEVICE_MEMORY,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ dead, sizeof(dead), "testing %d\n", 123);
+ return "SIGID=202 CORRECTED (deadbeef) " HW_ERR "testing 123\n";
+}
+
+static const char *test_hw_informational(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit_info(pdev, XE_SIGID_DEVICE_MEMORY,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ dead, sizeof(dead), "testing %d\n", 123);
+ return "SIGID=202 (deadbeef) testing 123\n";
+}
+
+static const struct log_test_param {
+ const char *(*func)(struct xe_device *xe);
+} log_test_params[] = {
+ { .func = test_fatal },
+ { .func = test_fatal_tile },
+ { .func = test_fatal_gt },
+ { .func = test_fatal_comp },
+ { .func = test_fatal_comp_tile },
+ { .func = test_fatal_comp_gt },
+ { .func = test_fatal_all },
+ { .func = test_recoverable },
+ { .func = test_recoverable_tile },
+ { .func = test_recoverable_gt },
+ { .func = test_recoverable_comp },
+ { .func = test_recoverable_comp_tile },
+ { .func = test_recoverable_comp_gt },
+ { .func = test_recoverable_all },
+ { .func = test_info },
+ { .func = test_info_tile },
+ { .func = test_info_gt },
+ { .func = test_info_err },
+ { .func = test_info_comp },
+ { .func = test_info_comp_tile },
+ { .func = test_info_comp_gt },
+ { .func = test_info_all },
+ { .func = test_hw_fatal },
+ { .func = test_hw_recoverable },
+ { .func = test_hw_corrected },
+ { .func = test_hw_informational },
+};
+
+static void log_param_get_desc(const struct log_test_param *p, char *desc)
+{
+ snprintf(desc, KUNIT_PARAM_DESC_SIZE, "%ps", p->func);
+}
+
+KUNIT_ARRAY_PARAM(log, log_test_params, log_param_get_desc);
+
+static void check_dmesg_vprintk(struct pci_dev *pdev, int cper_sev, struct va_format *vaf)
+{
+ struct kunit *test = kunit_get_current_test();
+ const struct log_test_param *param = test->param_value;
+ const char *exp = param->func(NULL);
+ char msg[64];
+ int n;
+
+ n = vsnprintf(msg, sizeof(msg), vaf->fmt, *vaf->va);
+ KUNIT_EXPECT_LT(test, n, sizeof(msg));
+ KUNIT_EXPECT_STREQ(test, msg, exp);
+}
+
+static void test_dmesg(struct kunit *test)
+{
+ const struct log_test_param *param = test->param_value;
+
+ kunit_activate_static_stub(test, log_emit_cper, nop_emit_cper);
+ kunit_activate_static_stub(test, log_dmesg_vprintk, check_dmesg_vprintk);
+
+ param->func(test->priv);
+}
+
+#define INVALID_TILEID (XE_MAX_TILES_PER_DEVICE + 1)
+#define INVALID_GTID (XE_MAX_TILES_PER_DEVICE * XE_MAX_GT_PER_TILE + 1)
+
+#define PREP_TEST_LOCATION(type, id) \
+ (FIELD_PREP_CONST(XE_LOG_LOCATION_TYPE_MASK, (type)) | \
+ FIELD_PREP_CONST(XE_LOG_LOCATION_ID_MASK, (id)))
+#define PREP_TEST_COMPONENT(class, type) \
+ (FIELD_PREP_CONST(XE_LOG_COMPONENT_CLASS_MASK, (class)) | \
+ FIELD_PREP_CONST(XE_LOG_COMPONENT_TYPE_MASK, (type)))
+
+static const struct {
+ u32 comp;
+ u32 loc;
+ const char *name;
+} invalid_params[] = {
+ { .name = "no-component no-location no-warn" },
+ { .loc = PREP_TEST_LOCATION(0, 1),
+ .name = "reserved location" },
+ { .loc = PREP_TEST_LOCATION(255, 0),
+ .name = "unknown location" },
+ { .loc = PREP_TEST_LOCATION(XE_LOG_LOCATION_TYPE_DEVICE, 1),
+ .name = "nonzero-device-id location" },
+ { .loc = PREP_TEST_LOCATION(XE_LOG_LOCATION_TYPE_TILE, INVALID_TILEID),
+ .name = "invalid-tile-id location" },
+ { .loc = PREP_TEST_LOCATION(XE_LOG_LOCATION_TYPE_GT, INVALID_GTID),
+ .name = "invalid-gt-id location" },
+ { .comp = PREP_TEST_COMPONENT(255, 0),
+ .name = "unknown component class" },
+ { .comp = PREP_TEST_COMPONENT(XE_LOG_COMPONENT_CLASS_SYSTEM, 255),
+ .name = "unknown system component" },
+ { .comp = PREP_TEST_COMPONENT(XE_LOG_COMPONENT_CLASS_HARDWARE, 255),
+ .name = "unknown hardware component" },
+ { .comp = PREP_TEST_COMPONENT(255, 1),
+ .loc = PREP_TEST_LOCATION(255, 1),
+ .name = "unknown component and location" },
+};
+
+KUNIT_ARRAY_PARAM_DESC(invalid_param, invalid_params, name);
+
+static void test_invalid(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ typeof(invalid_params[0]) *param = test->param_value;
+
+ struct pci_dev *pdev = xe_any_to_pdev(xe);
+
+ if (!IS_ENABLED(CONFIG_DRM_XE_DEBUG))
+ kunit_skip(test, "requires CONFIG_DRM_XE_DEBUG\n");
+
+ kunit_activate_static_stub(test, log_emit_cper, nop_emit_cper);
+ kunit_activate_static_stub(test, log_dmesg_vprintk, nop_dmesg_vprintk);
+
+ kunit_warning_suppress(test) {
+ xe_log_emit(pdev, CPER_SEV_FATAL, XE_SIGID_PROBE,
+ param->comp, param->loc,
+ NULL, 0, "testing %s\n", param->name);
+ KUNIT_EXPECT_SUPPRESSED_WARNING_COUNT(test, !!param->comp + !!param->loc);
+ }
+}
+
+static int xe_log_test_init(struct kunit *test)
+{
+ struct xe_pci_fake_data fake = {
+ .platform = XE_PVC, /* with max_remote_tiles != 0 */
+ .subplatform = XE_SUBPLATFORM_NONE,
+ .graphics_verx100 = 2001,
+ .media_verx100 = 2001,
+ };
+ struct xe_device *xe;
+
+ test->priv = &fake;
+ xe_kunit_helper_xe_device_test_init(test);
+ xe = test->priv;
+
+ KUNIT_ASSERT_NOT_NULL(test, to_tile_safe(xe));
+ KUNIT_ASSERT_NOT_NULL(test, to_gt_safe(xe));
+ KUNIT_EXPECT_EQ(test, 1, xe_any_id(to_tile_safe(xe)));
+ KUNIT_EXPECT_EQ(test, 1, xe_any_id(to_gt_safe(xe)));
+
+ return 0;
+}
+
+static struct kunit_case xe_log_test_cases[] = {
+ KUNIT_CASE(demo_cper),
+ KUNIT_CASE(demo_dmesg),
+ KUNIT_CASE_PARAM(test_dmesg, log_gen_params),
+ KUNIT_CASE_PARAM(test_invalid, invalid_param_gen_params),
+ {}
+};
+
+static struct kunit_suite xe_log_suite = {
+ .name = "xe_log",
+ .test_cases = xe_log_test_cases,
+ .init = xe_log_test_init,
+};
+
+kunit_test_suites(&xe_log_suite);
diff --git a/drivers/gpu/drm/xe/tests/xe_migrate.c b/drivers/gpu/drm/xe/tests/xe_migrate.c
index 3c1be809be82..f10d9513747b 100644
--- a/drivers/gpu/drm/xe/tests/xe_migrate.c
+++ b/drivers/gpu/drm/xe/tests/xe_migrate.c
@@ -198,8 +198,7 @@ static void xe_migrate_sanity_test(struct xe_migrate *m, struct kunit *test,
err = xe_bo_vmap(bo);
if (err) {
- KUNIT_FAIL(test, "Failed to vmap our pagetables: %li\n",
- PTR_ERR(bo));
+ KUNIT_FAIL(test, "Failed to vmap our pagetables: %d\n", err);
return;
}
diff --git a/drivers/gpu/drm/xe/xe_any.h b/drivers/gpu/drm/xe/xe_any.h
new file mode 100644
index 000000000000..5d97afa76915
--- /dev/null
+++ b/drivers/gpu/drm/xe/xe_any.h
@@ -0,0 +1,137 @@
+/* SPDX-License-Identifier: MIT */
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#ifndef _XE_ANY_H_
+#define _XE_ANY_H_
+
+#include "xe_device.h"
+
+#define __xe_any_to_self_assoc(type, any) \
+ const type * : (any), \
+ type * : (any)
+
+/**
+ * xe_any_if_type() - Get the pointer only if it is @type pointer.
+ * @any: any pointer
+ * @type: data type to look for
+ *
+ * Return: the @type pointer or NULL.
+ */
+#define xe_any_if_type(any, type) \
+ _Generic((any), \
+ __xe_any_to_self_assoc(type, (any)), \
+ default : NULL)
+
+/**
+ * xe_any_if_gt() - Get the pointer only if it is &xe_gt.
+ * @any: any pointer
+ *
+ * Return: the @xe_gt pointer or NULL.
+ */
+#define xe_any_if_gt(any) xe_any_if_type((any), struct xe_gt)
+
+/**
+ * xe_any_if_tile() - Get the pointer only if it is &xe_tile.
+ * @any: any pointer
+ *
+ * Return: the @xe_tile pointer or NULL.
+ */
+#define xe_any_if_tile(any) xe_any_if_type((any), struct xe_tile)
+
+/**
+ * xe_any_if_xe() - Get the pointer only if it is &xe_device.
+ * @any: any pointer
+ *
+ * Return: the @xe_device pointer or NULL.
+ */
+#define xe_any_if_xe(any) xe_any_if_type((any), struct xe_device)
+
+/**
+ * xe_any_if_pdev() - Get the pointer only if it is &pci_dev.
+ * @any: any pointer
+ *
+ * Return: the @pci_dev pointer or NULL.
+ */
+#define xe_any_if_pdev(any) xe_any_if_type((any), struct pci_dev)
+
+#define __xe_any_to_other_assoc(const, from, other, p) \
+ const struct from * : __##from##_to_##other((const struct from *)(p))
+
+#define __xe_tile_to_xe_device(p) tile_to_xe(p)
+#define __xe_gt_to_xe_device(p) gt_to_xe(p)
+#define __pci_dev_to_xe_device(p) pdev_to_xe_device(p)
+#define __device_to_xe_device(p) kdev_to_xe_device(p)
+#define __drm_device_to_xe_device(p) to_xe_device(p)
+#define __pci_dev_to_device(p) (&(p)->dev)
+
+/**
+ * xe_any_to_xe() - Obtain the &xe_device pointer.
+ * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer
+ *
+ * Return: the @xe_device pointer or backpointer.
+ */
+#define xe_any_to_xe(any) \
+ _Generic((any), \
+ __xe_any_to_self_assoc(struct xe_device, (any)), \
+ __xe_any_to_other_assoc(/* */, xe_tile, xe_device, (any)), \
+ __xe_any_to_other_assoc(const, xe_tile, xe_device, (any)), \
+ __xe_any_to_other_assoc(/* */, xe_gt, xe_device, (any)), \
+ __xe_any_to_other_assoc(const, xe_gt, xe_device, (any)), \
+ __xe_any_to_other_assoc(, drm_device, xe_device, (any)), \
+ __xe_any_to_other_assoc(, pci_dev, xe_device, (any)), \
+ __xe_any_to_other_assoc(, device, xe_device, (any)))
+
+/**
+ * xe_any_to_drm() - Obtain the &drm_device pointer.
+ * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer
+ *
+ * Return: the @drm_device pointer or backpointer.
+ */
+#define xe_any_to_drm(any) \
+ _Generic((any), \
+ __xe_any_to_self_assoc(struct drm_device, (any)), \
+ default : &xe_any_to_xe(any)->drm)
+
+/**
+ * xe_any_to_dev() - Obtain the &device pointer.
+ * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer
+ *
+ * Return: the @device pointer or backpointer.
+ */
+#define xe_any_to_dev(any) \
+ _Generic((any), \
+ __xe_any_to_self_assoc(struct device, (any)), \
+ __xe_any_to_other_assoc(, pci_dev, device, (any)), \
+ default : xe_any_to_drm(any)->dev)
+
+/**
+ * xe_any_to_pdev() - Obtain the &pci_dev pointer.
+ * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer
+ *
+ * Return: the @pci_dev pointer or backpointer.
+ */
+#define xe_any_to_pdev(any) \
+ _Generic((any), \
+ __xe_any_to_self_assoc(struct pci_dev, (any)), \
+ default : to_pci_dev(xe_any_to_dev(any)))
+
+#define __xe_tile_to_id(p) ((p)->id)
+#define __xe_gt_to_id(p) ((p)->info.id)
+
+/**
+ * xe_any_id() - Get the identifier of the underlying object.
+ * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer
+ *
+ * Return: the identifier of the object, or 0 if not applicable/available.
+ */
+#define xe_any_id(any) \
+ _Generic((any), \
+ __xe_any_to_other_assoc(/* */, xe_tile, id, (any)), \
+ __xe_any_to_other_assoc(const, xe_tile, id, (any)), \
+ __xe_any_to_other_assoc(/* */, xe_gt, id, (any)), \
+ __xe_any_to_other_assoc(const, xe_gt, id, (any)), \
+ default : 0)
+
+#endif
diff --git a/drivers/gpu/drm/xe/xe_debugfs.c b/drivers/gpu/drm/xe/xe_debugfs.c
index 8de78cd0aa03..28135f84e286 100644
--- a/drivers/gpu/drm/xe/xe_debugfs.c
+++ b/drivers/gpu/drm/xe/xe_debugfs.c
@@ -22,6 +22,7 @@
#include "xe_guc_ads.h"
#include "xe_hw_engine.h"
#include "xe_mmio.h"
+#include "xe_pagefault.h"
#include "xe_pcode.h"
#include "xe_pm.h"
#include "xe_psmi.h"
@@ -42,6 +43,7 @@
DECLARE_FAULT_ATTR(gt_reset_failure);
DECLARE_FAULT_ATTR(inject_csc_hw_error);
+DECLARE_FAULT_ATTR(wedge_cold_reset);
static bool csc_hw_error_available(struct xe_device *xe)
{
@@ -62,6 +64,8 @@ static struct {
{ .name = "inject_csc_hw_error",
.attr = &inject_csc_hw_error,
.is_visible = csc_hw_error_available },
+ { .name = "wedge_cold_reset",
+ .attr = &wedge_cold_reset },
};
/*
@@ -76,6 +80,7 @@ bool xe_fault_##name(void) \
FAULT_ACTION(gt_reset, gt_reset_failure)
FAULT_ACTION(csc_hw_error, inject_csc_hw_error)
+FAULT_ACTION(wedge_cold_reset, wedge_cold_reset)
static void xe_fault_inject_debugfs_register(struct xe_device *xe,
struct dentry *root)
@@ -194,6 +199,15 @@ static int sriov_info(struct seq_file *m, void *data)
return 0;
}
+static int pagefault_info(struct seq_file *m, void *data)
+{
+ struct xe_device *xe = node_to_xe(m->private);
+ struct drm_printer p = drm_seq_file_printer(m);
+
+ xe_pagefault_print_info(xe, &p);
+ return 0;
+}
+
static int workarounds(struct xe_device *xe, struct drm_printer *p)
{
guard(xe_pm_runtime)(xe);
@@ -285,6 +299,7 @@ static const struct drm_info_list debugfs_list[] = {
{"info", info, 0},
{ .name = "sriov_info", .show = sriov_info, },
{ .name = "workarounds", .show = workaround_info, },
+ { .name = "pagefault_info", .show = pagefault_info, },
};
static const struct drm_info_list pcode_info_debugfs[] = {
diff --git a/drivers/gpu/drm/xe/xe_debugfs.h b/drivers/gpu/drm/xe/xe_debugfs.h
index cd56f7442b99..0dcd28fd7dc0 100644
--- a/drivers/gpu/drm/xe/xe_debugfs.h
+++ b/drivers/gpu/drm/xe/xe_debugfs.h
@@ -13,10 +13,12 @@ struct xe_device;
#ifdef CONFIG_DEBUG_FS
bool xe_fault_gt_reset(void);
bool xe_fault_csc_hw_error(void);
+bool xe_fault_wedge_cold_reset(void);
void xe_debugfs_register(struct xe_device *xe);
#else
static inline bool xe_fault_gt_reset(void) { return false; }
static inline bool xe_fault_csc_hw_error(void) { return false; }
+static inline bool xe_fault_wedge_cold_reset(void) { return false; }
static inline void xe_debugfs_register(struct xe_device *xe) { }
#endif
diff --git a/drivers/gpu/drm/xe/xe_defaults.h b/drivers/gpu/drm/xe/xe_defaults.h
index c8ae1d5f3d60..0884224ef7c7 100644
--- a/drivers/gpu/drm/xe/xe_defaults.h
+++ b/drivers/gpu/drm/xe/xe_defaults.h
@@ -22,5 +22,6 @@
#define XE_DEFAULT_WEDGED_MODE XE_WEDGED_MODE_UPON_CRITICAL_ERROR
#define XE_DEFAULT_WEDGED_MODE_STR "upon-critical-error"
#define XE_DEFAULT_SVM_NOTIFIER_SIZE 512
+#define XE_DEFAULT_NUM_PF_WORK 2
#endif
diff --git a/drivers/gpu/drm/xe/xe_device.c b/drivers/gpu/drm/xe/xe_device.c
index ee732e5495f7..396d02eb2af8 100644
--- a/drivers/gpu/drm/xe/xe_device.c
+++ b/drivers/gpu/drm/xe/xe_device.c
@@ -48,6 +48,7 @@
#include "xe_i2c.h"
#include "xe_irq.h"
#include "xe_late_bind_fw.h"
+#include "xe_log.h"
#include "xe_mmio.h"
#include "xe_module.h"
#include "xe_nvm.h"
@@ -513,6 +514,17 @@ struct xe_device *xe_device_create(struct pci_dev *pdev)
}
ALLOW_ERROR_INJECTION(xe_device_create, ERRNO); /* See xe_pci_probe() */
+static void xe_device_parse_modparam(struct xe_device *xe)
+{
+ xe->atomic_svm_timeslice_ms = 5;
+ xe->min_run_period_lr_ms = 5;
+ xe->info.num_pf_work = xe_modparam.num_pf_work;
+ if (xe->info.num_pf_work < 1)
+ xe->info.num_pf_work = 1;
+ else if (xe->info.num_pf_work > XE_PAGEFAULT_WORK_MAX)
+ xe->info.num_pf_work = XE_PAGEFAULT_WORK_MAX;
+}
+
/**
* xe_device_init_early() - Initialize a new &xe_device instance
* @xe: the &xe_device to initialize
@@ -539,8 +551,7 @@ int xe_device_init_early(struct xe_device *xe)
if (err)
return err;
- xe->atomic_svm_timeslice_ms = 5;
- xe->min_run_period_lr_ms = 5;
+ xe_device_parse_modparam(xe);
err = xe_irq_init(xe);
if (err)
@@ -1432,6 +1443,9 @@ void xe_device_set_wedged_method(struct xe_device *xe, unsigned long method)
xe->wedged.method = method;
}
+#define WEDGED_URL "https://docs.kernel.org/gpu/drm-uapi.html#device-wedging"
+#define XE_BUG_URL "https://gitlab.freedesktop.org/drm/xe/kernel/issues/new"
+
/**
* xe_device_declare_wedged - Declare device wedged
* @xe: xe device instance
@@ -1463,12 +1477,12 @@ void xe_device_declare_wedged(struct xe_device *xe)
if (!atomic_xchg(&xe->wedged.flag, 1)) {
xe->needs_flr_on_fini = true;
xe_pm_runtime_get_noresume(xe);
- drm_err(&xe->drm,
- "CRITICAL: Xe has declared device %s as wedged.\n"
- "IOCTLs and executions are blocked.\n"
- "For recovery procedure, refer to https://docs.kernel.org/gpu/drm-uapi.html#device-wedging\n"
- "Please file a _new_ bug report at https://gitlab.freedesktop.org/drm/xe/kernel/issues/new\n",
- dev_name(xe->drm.dev));
+
+ xe_log_err_fatal(xe, WEDGED, -EIO, "Device declared wedged!\n");
+ xe_err_once(xe, "IOCTLs and executions are now blocked!\n"
+ "For recovery procedure, refer to %s\n"
+ "Please file a _new_ bug report at %s\n",
+ WEDGED_URL, XE_BUG_URL);
}
for_each_gt(gt, xe, id)
diff --git a/drivers/gpu/drm/xe/xe_device_types.h b/drivers/gpu/drm/xe/xe_device_types.h
index 03a7bb08adf7..180d450a6deb 100644
--- a/drivers/gpu/drm/xe/xe_device_types.h
+++ b/drivers/gpu/drm/xe/xe_device_types.h
@@ -140,6 +140,8 @@ struct xe_device {
u8 revid;
/** @info.step: stepping information for each IP */
struct xe_step_info step;
+ /** @info.num_pf_work: Number of page fault work thread */
+ int num_pf_work;
/** @info.dma_mask_size: DMA address bits */
u8 dma_mask_size;
/** @info.vram_flags: Vram flags */
@@ -320,18 +322,19 @@ struct xe_device {
struct xarray asid_to_vm;
/** @usm.next_asid: next ASID, used to cyclical alloc asids */
u32 next_asid;
+ /** @usm.current_pf_work: current page fault work item */
+ u32 current_pf_work;
/** @usm.lock: protects UM state */
struct rw_semaphore lock;
- /** @usm.pf_wq: page fault work queue, unbound, high priority */
- struct workqueue_struct *pf_wq;
- /*
- * We pick 4 here because, in the current implementation, it
- * yields the best bandwidth utilization of the kernel paging
- * engine.
- */
-#define XE_PAGEFAULT_QUEUE_COUNT 4
- /** @usm.pf_queue: Page fault queues */
- struct xe_pagefault_queue pf_queue[XE_PAGEFAULT_QUEUE_COUNT];
+ /** @usm.pagefault_wq: page fault work queue, unbound, high priority */
+ struct workqueue_struct *pagefault_wq;
+ /** @usm.prefetch_wq: threaded prefetch work queue, unbound */
+ struct workqueue_struct *prefetch_wq;
+#define XE_PAGEFAULT_WORK_MAX 8
+ /** @usm.pf_workers: Page fault workers */
+ struct xe_pagefault_work pf_workers[XE_PAGEFAULT_WORK_MAX];
+ /** @usm.pf_queue: Page fault queue */
+ struct xe_pagefault_queue pf_queue;
#if IS_ENABLED(CONFIG_DRM_XE_PAGEMAP)
/** @usm.dpagemap_shrinker: Shrinker for unused pagemaps */
struct drm_pagemap_shrinker *dpagemap_shrinker;
diff --git a/drivers/gpu/drm/xe/xe_drm_ras.c b/drivers/gpu/drm/xe/xe_drm_ras.c
index 984fa67b9015..78184b6ea7d4 100644
--- a/drivers/gpu/drm/xe/xe_drm_ras.c
+++ b/drivers/gpu/drm/xe/xe_drm_ras.c
@@ -186,6 +186,37 @@ null_info:
}
/**
+ * xe_drm_ras_event() - Report drm-ras error event to userspace
+ * @xe: xe device structure
+ * @component: error component (see &enum drm_xe_ras_error_component)
+ * @severity: error severity (see &enum drm_xe_ras_error_severity)
+ * @value: value of error counter
+ *
+ * Report an error-event to userspace.
+ */
+void xe_drm_ras_event(struct xe_device *xe, u32 component, u32 severity, u32 value)
+{
+ struct xe_drm_ras *ras = &xe->ras;
+ struct xe_drm_ras_counter *info = ras->info[severity];
+ struct drm_ras_node *node;
+ int ret;
+
+ /* Event is supported only if drm-ras is enabled */
+ if (!xe->info.has_drm_ras)
+ return;
+
+ node = &ras->node[severity];
+
+ if (!info || !info[component].name)
+ return;
+
+ ret = drm_ras_nl_error_event(node, component, info[component].name, value);
+ if (ret)
+ drm_err_ratelimited(&xe->drm, "drm-ras error-event failed: %d for %s %s\n", ret,
+ info[component].name, error_severity[severity]);
+}
+
+/**
* xe_drm_ras_init() - Initialize DRM RAS
* @xe: xe device instance
*
diff --git a/drivers/gpu/drm/xe/xe_drm_ras.h b/drivers/gpu/drm/xe/xe_drm_ras.h
index 365c70e93e82..723229b2cffb 100644
--- a/drivers/gpu/drm/xe/xe_drm_ras.h
+++ b/drivers/gpu/drm/xe/xe_drm_ras.h
@@ -5,11 +5,14 @@
#ifndef _XE_DRM_RAS_H_
#define _XE_DRM_RAS_H_
+#include <linux/types.h>
+
struct xe_device;
#define for_each_error_severity(i) \
for (i = 0; i < DRM_XE_RAS_ERR_SEV_MAX; i++)
int xe_drm_ras_init(struct xe_device *xe);
+void xe_drm_ras_event(struct xe_device *xe, u32 component, u32 severity, u32 value);
#endif
diff --git a/drivers/gpu/drm/xe/xe_exec_queue.c b/drivers/gpu/drm/xe/xe_exec_queue.c
index cd053de3c6b5..c4213bb9c137 100644
--- a/drivers/gpu/drm/xe/xe_exec_queue.c
+++ b/drivers/gpu/drm/xe/xe_exec_queue.c
@@ -207,9 +207,6 @@ static struct xe_exec_queue *__xe_exec_queue_alloc(struct xe_device *xe,
struct xe_gt *gt = hwe->gt;
int err;
- /* only kernel queues can be permanent */
- XE_WARN_ON((flags & EXEC_QUEUE_FLAG_PERMANENT) && !(flags & EXEC_QUEUE_FLAG_KERNEL));
-
q = kzalloc_flex(*q, lrc, width);
if (!q)
return ERR_PTR(-ENOMEM);
@@ -1121,6 +1118,9 @@ static int exec_queue_user_ext_set_property(struct xe_device *xe,
if (!exec_queue_set_property_funcs[idx])
return -EINVAL;
+ if (XE_IOCTL_DBG(xe, *properties & BIT_ULL(idx)))
+ return -EINVAL;
+
*properties |= BIT_ULL(idx);
err = exec_queue_user_ext_check(q, *properties);
if (XE_IOCTL_DBG(xe, err))
diff --git a/drivers/gpu/drm/xe/xe_exec_queue_types.h b/drivers/gpu/drm/xe/xe_exec_queue_types.h
index 53b6c0bf4849..95f75d61a647 100644
--- a/drivers/gpu/drm/xe/xe_exec_queue_types.h
+++ b/drivers/gpu/drm/xe/xe_exec_queue_types.h
@@ -70,6 +70,13 @@ struct xe_exec_queue_group {
spinlock_t suspend_lock;
/** @sync_pending: CGP_SYNC_DONE g2h response pending */
bool sync_pending;
+ /**
+ * @cgp_update_q: Queue that issued the currently outstanding (sent)
+ * CGP_SYNC or REGISTER_CONTEXT_MULTI_QUEUE; NULL when none is
+ * outstanding. Used during VF recovery to identify and replay the
+ * message whose CGP_SYNC_DONE was not received.
+ */
+ struct xe_exec_queue *cgp_update_q;
/** @banned: Group banned */
bool banned;
/** @stopped: Group is stopped, protected by list_lock */
@@ -128,20 +135,18 @@ struct xe_exec_queue {
/* queue used for kernel submission only */
#define EXEC_QUEUE_FLAG_KERNEL BIT(0)
-/* kernel engine only destroyed at driver unload */
-#define EXEC_QUEUE_FLAG_PERMANENT BIT(1)
/* for VM jobs. Caller needs to hold rpm ref when creating queue with this flag */
-#define EXEC_QUEUE_FLAG_VM BIT(2)
+#define EXEC_QUEUE_FLAG_VM BIT(1)
/* child of VM queue for multi-tile VM jobs */
-#define EXEC_QUEUE_FLAG_BIND_ENGINE_CHILD BIT(3)
+#define EXEC_QUEUE_FLAG_BIND_ENGINE_CHILD BIT(2)
/* kernel exec_queue only, set priority to highest level */
-#define EXEC_QUEUE_FLAG_HIGH_PRIORITY BIT(4)
+#define EXEC_QUEUE_FLAG_HIGH_PRIORITY BIT(3)
/* flag to indicate low latency hint to guc */
-#define EXEC_QUEUE_FLAG_LOW_LATENCY BIT(5)
+#define EXEC_QUEUE_FLAG_LOW_LATENCY BIT(4)
/* for migration (kernel copy, clear, bind) jobs */
-#define EXEC_QUEUE_FLAG_MIGRATE BIT(6)
+#define EXEC_QUEUE_FLAG_MIGRATE BIT(5)
/* for programming COMMON_SLICE_CHICKEN3 on first submission */
-#define EXEC_QUEUE_FLAG_DISABLE_STATE_CACHE_PERF_FIX BIT(7)
+#define EXEC_QUEUE_FLAG_DISABLE_STATE_CACHE_PERF_FIX BIT(6)
/**
* @flags: flags for this exec queue, should statically setup aside from ban
diff --git a/drivers/gpu/drm/xe/xe_gsc.c b/drivers/gpu/drm/xe/xe_gsc.c
index aab59dc647fb..524ac56bdcc7 100644
--- a/drivers/gpu/drm/xe/xe_gsc.c
+++ b/drivers/gpu/drm/xe/xe_gsc.c
@@ -478,8 +478,7 @@ int xe_gsc_init_post_hwconfig(struct xe_gsc *gsc)
q = xe_exec_queue_create(xe, NULL,
BIT(hwe->logical_instance), 1, hwe,
- EXEC_QUEUE_FLAG_KERNEL |
- EXEC_QUEUE_FLAG_PERMANENT, 0);
+ EXEC_QUEUE_FLAG_KERNEL, 0);
if (IS_ERR(q)) {
xe_gt_err(gt, "Failed to create queue for GSC submission\n");
return PTR_ERR(q);
diff --git a/drivers/gpu/drm/xe/xe_gt.c b/drivers/gpu/drm/xe/xe_gt.c
index dfdacc0f6de9..6805e0d3bf21 100644
--- a/drivers/gpu/drm/xe/xe_gt.c
+++ b/drivers/gpu/drm/xe/xe_gt.c
@@ -48,6 +48,7 @@
#include "xe_hw_engine_class_sysfs.h"
#include "xe_irq.h"
#include "xe_lmtt.h"
+#include "xe_log.h"
#include "xe_lrc.h"
#include "xe_map.h"
#include "xe_migrate.h"
@@ -925,7 +926,7 @@ static void gt_reset_worker(struct work_struct *w)
if (!xe_device_uc_enabled(gt_to_xe(gt)))
goto err_pm_put;
- xe_gt_info(gt, "reset started\n");
+ xe_log_info(gt, GT, "reset started\n");
if (xe_fault_gt_reset()) {
err = -ECANCELED;
@@ -964,7 +965,7 @@ static void gt_reset_worker(struct work_struct *w)
/* Pair with get while enqueueing the work in xe_gt_reset_async() */
xe_pm_runtime_put(gt_to_xe(gt));
- xe_gt_info(gt, "reset done\n");
+ xe_log_info(gt, GT, "reset done\n");
return;
@@ -973,7 +974,7 @@ err_out:
XE_WARN_ON(xe_uc_start(&gt->uc));
err_fail:
- xe_gt_err(gt, "reset failed (%pe)\n", ERR_PTR(err));
+ xe_log_err_fatal(gt, GT, err, "reset failed\n");
xe_device_declare_wedged(gt_to_xe(gt));
err_pm_put:
xe_pm_runtime_put(gt_to_xe(gt));
diff --git a/drivers/gpu/drm/xe/xe_gt_debugfs.c b/drivers/gpu/drm/xe/xe_gt_debugfs.c
index c38bcacb27e4..ea78b57b1c31 100644
--- a/drivers/gpu/drm/xe/xe_gt_debugfs.c
+++ b/drivers/gpu/drm/xe/xe_gt_debugfs.c
@@ -9,7 +9,9 @@
#include <drm/drm_debugfs.h>
#include <drm/drm_managed.h>
+#include <linux/math.h>
+#include "regs/xe_gt_regs.h"
#include "xe_device.h"
#include "xe_force_wake.h"
#include "xe_gt.h"
@@ -22,6 +24,7 @@
#include "xe_guc_hwconfig.h"
#include "xe_hw_engine.h"
#include "xe_lrc.h"
+#include "xe_mmio.h"
#include "xe_mocs.h"
#include "xe_pat.h"
#include "xe_pm.h"
@@ -336,6 +339,80 @@ static int force_reset_sync_show(struct seq_file *s, void *unused)
}
DEFINE_SHOW_STORE_ATTRIBUTE(force_reset_sync);
+#define U1_15_ONE 0x8000
+#define U1_15_INT_BITS GENMASK(15, 15)
+#define U1_15_FRACTION_BITS GENMASK(14, 0)
+
+static void u1_15_decode(u16 num, u16 *i, u32 *frac)
+{
+ /*
+ * In U1.15 format, uppermost bit is integer value and the
+ * rest 15 are the fraction.
+ */
+
+ *i = FIELD_GET(U1_15_INT_BITS, num);
+ *frac = FIELD_GET(U1_15_FRACTION_BITS, num);
+}
+
+static void u1_15_decode_decimal(u16 value, u16 *i, u32 *frac, int digits)
+{
+ u1_15_decode(value, i, frac);
+ *frac = (*frac * int_pow(10, digits)) / (FIELD_MAX(U1_15_FRACTION_BITS) + 1);
+}
+
+static int gt_ia_bias_show(struct seq_file *s, void *unused)
+{
+ struct xe_gt *gt = s->private;
+ struct xe_device *xe = gt_to_xe(gt);
+ u32 val;
+ u32 ia_frac, gt_frac;
+ u16 ia_raw, gt_raw;
+ u16 ia_int, gt_int;
+
+ guard(xe_pm_runtime)(xe);
+ val = xe_mmio_read32(&gt->mmio, GT_IA_PERF_BIAS_REG);
+
+ ia_raw = REG_FIELD_GET(IA_BIAS, val);
+ gt_raw = REG_FIELD_GET(GT_BIAS, val);
+
+ u1_15_decode_decimal(ia_raw, &ia_int, &ia_frac, 4);
+ u1_15_decode_decimal(gt_raw, &gt_int, &gt_frac, 4);
+
+ seq_printf(s, "0x%x (GT: %u.%04u, IA: %u.%04u)\n",
+ val, gt_int, gt_frac, ia_int, ia_frac);
+
+ return 0;
+}
+
+static ssize_t gt_ia_bias_write(struct file *file,
+ const char __user *userbuf,
+ size_t count, loff_t *ppos)
+{
+ struct seq_file *s = file->private_data;
+ struct xe_gt *gt = s->private;
+ struct xe_device *xe = gt_to_xe(gt);
+ u32 val;
+ int ret;
+
+ ret = kstrtou32_from_user(userbuf, count, 0, &val);
+ if (ret)
+ return ret;
+
+ if (REG_FIELD_GET(IA_BIAS, val) > U1_15_ONE ||
+ REG_FIELD_GET(GT_BIAS, val) > U1_15_ONE)
+ return -EINVAL;
+
+ if (REG_FIELD_GET(IA_BIAS, val) < IA_BIAS_DEFAULT ||
+ REG_FIELD_GET(GT_BIAS, val) < GT_BIAS_DEFAULT)
+ return -EINVAL;
+
+ guard(xe_pm_runtime)(xe);
+ xe_mmio_write32(&gt->mmio, GT_IA_PERF_BIAS_REG, val);
+
+ return count;
+}
+DEFINE_SHOW_STORE_ATTRIBUTE(gt_ia_bias);
+
void xe_gt_debugfs_register(struct xe_gt *gt)
{
struct xe_device *xe = gt_to_xe(gt);
@@ -378,6 +455,9 @@ void xe_gt_debugfs_register(struct xe_gt *gt)
ARRAY_SIZE(pf_only_debugfs_list),
root, minor);
+ if (xe_gt_is_main_type(gt) && !IS_DGFX(xe) && !IS_SRIOV_VF(xe))
+ debugfs_create_file("gt_ia_bias", 0600, root, gt, &gt_ia_bias_fops);
+
xe_uc_debugfs_register(&gt->uc, root);
if (IS_SRIOV_PF(xe))
diff --git a/drivers/gpu/drm/xe/xe_gt_idle.c b/drivers/gpu/drm/xe/xe_gt_idle.c
index 04b24e1c8b78..7dc9873aef54 100644
--- a/drivers/gpu/drm/xe/xe_gt_idle.c
+++ b/drivers/gpu/drm/xe/xe_gt_idle.c
@@ -248,7 +248,8 @@ int xe_gt_idle_pg_print(struct xe_gt *gt, struct drm_printer *p)
pg_status = xe_mmio_read32(&gt->mmio, POWERGATE_DOMAIN_STATUS);
}
- if (gt->info.engine_mask & XE_HW_ENGINE_RCS_MASK) {
+ if (gt->info.engine_mask &
+ (XE_HW_ENGINE_RCS_MASK | XE_HW_ENGINE_CCS_MASK)) {
drm_printf(p, "Render Power Gating Enabled: %s\n",
str_yes_no(pg_enabled & RENDER_POWERGATE_ENABLE));
diff --git a/drivers/gpu/drm/xe/xe_gt_printk.h b/drivers/gpu/drm/xe/xe_gt_printk.h
index 1313d32862db..1598098d2987 100644
--- a/drivers/gpu/drm/xe/xe_gt_printk.h
+++ b/drivers/gpu/drm/xe/xe_gt_printk.h
@@ -35,6 +35,9 @@
#define xe_gt_dbg(_gt, _fmt, ...) \
xe_gt_printk((_gt), dbg, _fmt, ##__VA_ARGS__)
+#define xe_gt_dbg_ratelimited(_gt, _fmt, ...) \
+ xe_gt_printk((_gt), dbg_ratelimited, _fmt, ##__VA_ARGS__)
+
#define xe_gt_WARN_type(_gt, _type, _condition, _fmt, ...) \
xe_tile_WARN##_type((_gt)->tile, _condition, _fmt, ## __VA_ARGS__)
diff --git a/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c b/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c
index 2c9b85b84b1b..be0a413ee17c 100644
--- a/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c
+++ b/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c
@@ -668,7 +668,7 @@ static int pf_config_bulk_set_u64_done(struct xe_gt *gt, unsigned int first, uns
}
/**
- * xe_gt_sriov_pf_config_bulk_set_ggtt - Provision many VFs with GGTT.
+ * xe_gt_sriov_pf_config_bulk_set_ggtt_locked() - Provision many VFs with GGTT.
* @gt: the &xe_gt (can't be media)
* @vfid: starting VF identifier (can't be 0)
* @num_vfs: number of VFs to provision
@@ -678,31 +678,49 @@ static int pf_config_bulk_set_u64_done(struct xe_gt *gt, unsigned int first, uns
*
* Return: 0 on success or a negative error code on failure.
*/
-int xe_gt_sriov_pf_config_bulk_set_ggtt(struct xe_gt *gt, unsigned int vfid,
- unsigned int num_vfs, u64 size)
+int xe_gt_sriov_pf_config_bulk_set_ggtt_locked(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u64 size)
{
unsigned int n;
int err = 0;
xe_gt_assert(gt, vfid);
xe_gt_assert(gt, xe_gt_is_main_type(gt));
+ lockdep_assert_held(xe_gt_sriov_pf_master_mutex(gt));
if (!num_vfs)
return 0;
- mutex_lock(xe_gt_sriov_pf_master_mutex(gt));
for (n = vfid; n < vfid + num_vfs; n++) {
err = pf_provision_vf_ggtt(gt, n, size);
if (err)
break;
}
- mutex_unlock(xe_gt_sriov_pf_master_mutex(gt));
return pf_config_bulk_set_u64_done(gt, vfid, num_vfs, size,
- xe_gt_sriov_pf_config_get_ggtt,
+ pf_get_vf_config_ggtt,
"GGTT", n, err);
}
+/**
+ * xe_gt_sriov_pf_config_bulk_set_ggtt() - Provision many VFs with GGTT.
+ * @gt: the &xe_gt (can't be media)
+ * @vfid: starting VF identifier (can't be 0)
+ * @num_vfs: number of VFs to provision
+ * @size: requested GGTT size
+ *
+ * This function can only be called on PF.
+ *
+ * Return: 0 on success or a negative error code on failure.
+ */
+int xe_gt_sriov_pf_config_bulk_set_ggtt(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u64 size)
+{
+ guard(mutex)(xe_gt_sriov_pf_master_mutex(gt));
+
+ return xe_gt_sriov_pf_config_bulk_set_ggtt_locked(gt, vfid, num_vfs, size);
+}
+
/* Return: size of the largest continuous GGTT region */
static u64 pf_get_max_ggtt(struct xe_gt *gt)
{
@@ -775,10 +793,9 @@ int xe_gt_sriov_pf_config_set_fair_ggtt(struct xe_gt *gt, unsigned int vfid,
xe_gt_assert(gt, num_vfs);
xe_gt_assert(gt, xe_gt_is_main_type(gt));
- mutex_lock(xe_gt_sriov_pf_master_mutex(gt));
- fair = pf_estimate_fair_ggtt(gt, num_vfs);
- mutex_unlock(xe_gt_sriov_pf_master_mutex(gt));
+ guard(mutex)(xe_gt_sriov_pf_master_mutex(gt));
+ fair = pf_estimate_fair_ggtt(gt, num_vfs);
if (!fair)
return -ENOSPC;
@@ -787,7 +804,7 @@ int xe_gt_sriov_pf_config_set_fair_ggtt(struct xe_gt *gt, unsigned int vfid,
xe_gt_sriov_info(gt, "Using non-profile provisioning (%s %llu vs %llu)\n",
"GGTT", fair, profile);
- return xe_gt_sriov_pf_config_bulk_set_ggtt(gt, vfid, num_vfs, fair);
+ return xe_gt_sriov_pf_config_bulk_set_ggtt_locked(gt, vfid, num_vfs, fair);
}
/**
@@ -1099,7 +1116,7 @@ static int pf_config_bulk_set_u32_done(struct xe_gt *gt, unsigned int first, uns
}
/**
- * xe_gt_sriov_pf_config_bulk_set_ctxs - Provision many VFs with GuC context IDs.
+ * xe_gt_sriov_pf_config_bulk_set_ctxs_locked() - Provision many VFs with GuC context IDs.
* @gt: the &xe_gt
* @vfid: starting VF identifier
* @num_vfs: number of VFs to provision
@@ -1109,30 +1126,48 @@ static int pf_config_bulk_set_u32_done(struct xe_gt *gt, unsigned int first, uns
*
* Return: 0 on success or a negative error code on failure.
*/
-int xe_gt_sriov_pf_config_bulk_set_ctxs(struct xe_gt *gt, unsigned int vfid,
- unsigned int num_vfs, u32 num_ctxs)
+int xe_gt_sriov_pf_config_bulk_set_ctxs_locked(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u32 num_ctxs)
{
unsigned int n;
int err = 0;
xe_gt_assert(gt, vfid);
+ lockdep_assert_held(xe_gt_sriov_pf_master_mutex(gt));
if (!num_vfs)
return 0;
- mutex_lock(xe_gt_sriov_pf_master_mutex(gt));
for (n = vfid; n < vfid + num_vfs; n++) {
err = pf_provision_vf_ctxs(gt, n, num_ctxs);
if (err)
break;
}
- mutex_unlock(xe_gt_sriov_pf_master_mutex(gt));
return pf_config_bulk_set_u32_done(gt, vfid, num_vfs, num_ctxs,
- xe_gt_sriov_pf_config_get_ctxs,
+ pf_get_vf_config_ctxs,
"GuC context IDs", no_unit, n, err);
}
+/**
+ * xe_gt_sriov_pf_config_bulk_set_ctxs() - Provision many VFs with GuC context IDs.
+ * @gt: the &xe_gt
+ * @vfid: starting VF identifier
+ * @num_vfs: number of VFs to provision
+ * @num_ctxs: requested number of GuC contexts IDs (0 to release)
+ *
+ * This function can only be called on PF.
+ *
+ * Return: 0 on success or a negative error code on failure.
+ */
+int xe_gt_sriov_pf_config_bulk_set_ctxs(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u32 num_ctxs)
+{
+ guard(mutex)(xe_gt_sriov_pf_master_mutex(gt));
+
+ return xe_gt_sriov_pf_config_bulk_set_ctxs_locked(gt, vfid, num_vfs, num_ctxs);
+}
+
static u32 pf_profile_fair_ctxs(struct xe_gt *gt, unsigned int num_vfs)
{
bool admin_only_pf = xe_sriov_pf_admin_only(gt_to_xe(gt));
@@ -1181,10 +1216,9 @@ int xe_gt_sriov_pf_config_set_fair_ctxs(struct xe_gt *gt, unsigned int vfid,
xe_gt_assert(gt, vfid);
xe_gt_assert(gt, num_vfs);
- mutex_lock(xe_gt_sriov_pf_master_mutex(gt));
- fair = pf_estimate_fair_ctxs(gt, num_vfs);
- mutex_unlock(xe_gt_sriov_pf_master_mutex(gt));
+ guard(mutex)(xe_gt_sriov_pf_master_mutex(gt));
+ fair = pf_estimate_fair_ctxs(gt, num_vfs);
if (!fair)
return -ENOSPC;
@@ -1193,7 +1227,7 @@ int xe_gt_sriov_pf_config_set_fair_ctxs(struct xe_gt *gt, unsigned int vfid,
xe_gt_sriov_info(gt, "Using non-profile provisioning (%s %u vs %u)\n",
"GuC context IDs", fair, profile);
- return xe_gt_sriov_pf_config_bulk_set_ctxs(gt, vfid, num_vfs, fair);
+ return xe_gt_sriov_pf_config_bulk_set_ctxs_locked(gt, vfid, num_vfs, fair);
}
static u32 pf_get_min_spare_dbs(struct xe_gt *gt)
@@ -1363,7 +1397,7 @@ int xe_gt_sriov_pf_config_set_dbs(struct xe_gt *gt, unsigned int vfid, u32 num_d
}
/**
- * xe_gt_sriov_pf_config_bulk_set_dbs - Provision many VFs with GuC context IDs.
+ * xe_gt_sriov_pf_config_bulk_set_dbs_locked() - Provision many VFs with GuC doorbells.
* @gt: the &xe_gt
* @vfid: starting VF identifier (can't be 0)
* @num_vfs: number of VFs to provision
@@ -1373,30 +1407,48 @@ int xe_gt_sriov_pf_config_set_dbs(struct xe_gt *gt, unsigned int vfid, u32 num_d
*
* Return: 0 on success or a negative error code on failure.
*/
-int xe_gt_sriov_pf_config_bulk_set_dbs(struct xe_gt *gt, unsigned int vfid,
- unsigned int num_vfs, u32 num_dbs)
+int xe_gt_sriov_pf_config_bulk_set_dbs_locked(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u32 num_dbs)
{
unsigned int n;
int err = 0;
xe_gt_assert(gt, vfid);
+ lockdep_assert_held(xe_gt_sriov_pf_master_mutex(gt));
if (!num_vfs)
return 0;
- mutex_lock(xe_gt_sriov_pf_master_mutex(gt));
for (n = vfid; n < vfid + num_vfs; n++) {
err = pf_provision_vf_dbs(gt, n, num_dbs);
if (err)
break;
}
- mutex_unlock(xe_gt_sriov_pf_master_mutex(gt));
return pf_config_bulk_set_u32_done(gt, vfid, num_vfs, num_dbs,
- xe_gt_sriov_pf_config_get_dbs,
+ pf_get_vf_config_dbs,
"GuC doorbell IDs", no_unit, n, err);
}
+/**
+ * xe_gt_sriov_pf_config_bulk_set_dbs() - Provision many VFs with GuC doorbells.
+ * @gt: the &xe_gt
+ * @vfid: starting VF identifier (can't be 0)
+ * @num_vfs: number of VFs to provision
+ * @num_dbs: requested number of GuC doorbell IDs (0 to release)
+ *
+ * This function can only be called on PF.
+ *
+ * Return: 0 on success or a negative error code on failure.
+ */
+int xe_gt_sriov_pf_config_bulk_set_dbs(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u32 num_dbs)
+{
+ guard(mutex)(xe_gt_sriov_pf_master_mutex(gt));
+
+ return xe_gt_sriov_pf_config_bulk_set_dbs_locked(gt, vfid, num_vfs, num_dbs);
+}
+
static u32 pf_profile_fair_dbs(struct xe_gt *gt, unsigned int num_vfs)
{
bool admin_only_pf = xe_sriov_pf_admin_only(gt_to_xe(gt));
@@ -1446,10 +1498,9 @@ int xe_gt_sriov_pf_config_set_fair_dbs(struct xe_gt *gt, unsigned int vfid,
xe_gt_assert(gt, vfid);
xe_gt_assert(gt, num_vfs);
- mutex_lock(xe_gt_sriov_pf_master_mutex(gt));
- fair = pf_estimate_fair_dbs(gt, num_vfs);
- mutex_unlock(xe_gt_sriov_pf_master_mutex(gt));
+ guard(mutex)(xe_gt_sriov_pf_master_mutex(gt));
+ fair = pf_estimate_fair_dbs(gt, num_vfs);
if (!fair)
return -ENOSPC;
@@ -1458,7 +1509,7 @@ int xe_gt_sriov_pf_config_set_fair_dbs(struct xe_gt *gt, unsigned int vfid,
xe_gt_sriov_info(gt, "Using non-profile provisioning (%s %u vs %u)\n",
"GuC doorbell IDs", fair, profile);
- return xe_gt_sriov_pf_config_bulk_set_dbs(gt, vfid, num_vfs, fair);
+ return xe_gt_sriov_pf_config_bulk_set_dbs_locked(gt, vfid, num_vfs, fair);
}
static u64 pf_get_lmem_alignment(struct xe_gt *gt)
diff --git a/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h b/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h
index 2ec62c12ad5c..a56e63f3660a 100644
--- a/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h
+++ b/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h
@@ -18,18 +18,24 @@ int xe_gt_sriov_pf_config_set_fair_ggtt(struct xe_gt *gt,
unsigned int vfid, unsigned int num_vfs);
int xe_gt_sriov_pf_config_bulk_set_ggtt(struct xe_gt *gt,
unsigned int vfid, unsigned int num_vfs, u64 size);
+int xe_gt_sriov_pf_config_bulk_set_ggtt_locked(struct xe_gt *gt,
+ unsigned int vfid, unsigned int num_vfs, u64 size);
u32 xe_gt_sriov_pf_config_get_ctxs(struct xe_gt *gt, unsigned int vfid);
int xe_gt_sriov_pf_config_set_ctxs(struct xe_gt *gt, unsigned int vfid, u32 num_ctxs);
int xe_gt_sriov_pf_config_set_fair_ctxs(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs);
int xe_gt_sriov_pf_config_bulk_set_ctxs(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs,
u32 num_ctxs);
+int xe_gt_sriov_pf_config_bulk_set_ctxs_locked(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u32 num_ctxs);
u32 xe_gt_sriov_pf_config_get_dbs(struct xe_gt *gt, unsigned int vfid);
int xe_gt_sriov_pf_config_set_dbs(struct xe_gt *gt, unsigned int vfid, u32 num_dbs);
int xe_gt_sriov_pf_config_set_fair_dbs(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs);
int xe_gt_sriov_pf_config_bulk_set_dbs(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs,
u32 num_dbs);
+int xe_gt_sriov_pf_config_bulk_set_dbs_locked(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u32 num_dbs);
u64 xe_gt_sriov_pf_config_get_lmem(struct xe_gt *gt, unsigned int vfid);
int xe_gt_sriov_pf_config_set_lmem(struct xe_gt *gt, unsigned int vfid, u64 size);
diff --git a/drivers/gpu/drm/xe/xe_gt_sriov_printk.h b/drivers/gpu/drm/xe/xe_gt_sriov_printk.h
index d3457d608db8..ada0ee9f45e5 100644
--- a/drivers/gpu/drm/xe/xe_gt_sriov_printk.h
+++ b/drivers/gpu/drm/xe/xe_gt_sriov_printk.h
@@ -27,6 +27,9 @@
#define xe_gt_sriov_dbg(_gt, _fmt, ...) \
__xe_gt_sriov_printk(_gt, dbg, _fmt, ##__VA_ARGS__)
+#define xe_gt_sriov_dbg_ratelimited(_gt, _fmt, ...) \
+ __xe_gt_sriov_printk(_gt, dbg_ratelimited, _fmt, ##__VA_ARGS__)
+
/* for low level noisy debug messages */
#ifdef CONFIG_DRM_XE_DEBUG_SRIOV
#define xe_gt_sriov_dbg_verbose(_gt, _fmt, ...) xe_gt_sriov_dbg(_gt, _fmt, ##__VA_ARGS__)
diff --git a/drivers/gpu/drm/xe/xe_gt_stats.c b/drivers/gpu/drm/xe/xe_gt_stats.c
index 789397514f3e..2a40924660ee 100644
--- a/drivers/gpu/drm/xe/xe_gt_stats.c
+++ b/drivers/gpu/drm/xe/xe_gt_stats.c
@@ -95,12 +95,19 @@ void xe_gt_stats_incr(struct xe_gt *gt, const enum xe_gt_stats_id id, int incr)
#define DEF_STAT_STR(ID, name) [XE_GT_STATS_ID_##ID] = name
static const char *const stat_description[__XE_GT_STATS_NUM_IDS] = {
+ DEF_STAT_STR(CHAIN_PAGEFAULT_COUNT, "chain_pagefault_count"),
+ DEF_STAT_STR(CHAIN_IRQ_PAGEFAULT_COUNT, "chain_irq_pagefault_count"),
+ DEF_STAT_STR(CHAIN_DRAIN_IRQ_PAGEFAULT_COUNT, "chain_drain_irq_pagefault_count"),
+ DEF_STAT_STR(CHAIN_MISMATCH_PAGEFAULT_COUNT, "chain_mismatch_pagefault_count"),
+ DEF_STAT_STR(PARALLEL_PAGEFAULT_COUNT, "parallel_pagefault_count"),
+ DEF_STAT_STR(LAST_PAGEFAULT_COUNT, "last_pagefault_count"),
DEF_STAT_STR(SVM_PAGEFAULT_COUNT, "svm_pagefault_count"),
DEF_STAT_STR(TLB_INVAL, "tlb_inval_count"),
DEF_STAT_STR(SVM_TLB_INVAL_COUNT, "svm_tlb_inval_count"),
DEF_STAT_STR(SVM_TLB_INVAL_US, "svm_tlb_inval_us"),
DEF_STAT_STR(VMA_PAGEFAULT_COUNT, "vma_pagefault_count"),
DEF_STAT_STR(VMA_PAGEFAULT_KB, "vma_pagefault_kb"),
+ DEF_STAT_STR(PAGEFAULT_US, "pagefault_us"),
DEF_STAT_STR(INVALID_PREFETCH_PAGEFAULT_COUNT, "invalid_prefetch_pagefault_count"),
DEF_STAT_STR(SVM_4K_PAGEFAULT_COUNT, "svm_4K_pagefault_count"),
DEF_STAT_STR(SVM_64K_PAGEFAULT_COUNT, "svm_64K_pagefault_count"),
diff --git a/drivers/gpu/drm/xe/xe_gt_stats_types.h b/drivers/gpu/drm/xe/xe_gt_stats_types.h
index 425491bed6c4..24abeb9f2137 100644
--- a/drivers/gpu/drm/xe/xe_gt_stats_types.h
+++ b/drivers/gpu/drm/xe/xe_gt_stats_types.h
@@ -10,6 +10,19 @@
/**
* enum xe_gt_stats_id - GT statistics identifiers
+ * @XE_GT_STATS_ID_CHAIN_PAGEFAULT_COUNT: Total page faults chained onto an
+ * in-flight fault instead of being queued separately.
+ * @XE_GT_STATS_ID_CHAIN_IRQ_PAGEFAULT_COUNT: Page faults chained directly from
+ * the IRQ handler.
+ * @XE_GT_STATS_ID_CHAIN_DRAIN_IRQ_PAGEFAULT_COUNT: IRQ-handler chained faults
+ * that also drained the fault queue.
+ * @XE_GT_STATS_ID_CHAIN_MISMATCH_PAGEFAULT_COUNT: Chained faults requeued
+ * because their fault range did not match the fault they were chained onto.
+ * @XE_GT_STATS_ID_PARALLEL_PAGEFAULT_COUNT: Faults dequeued while another page
+ * fault worker was already handling a fault concurrently.
+ * @XE_GT_STATS_ID_LAST_PAGEFAULT_COUNT: Faults whose range matched the last
+ * serviced range, allowing an immediate ack.
+ *
* @XE_GT_STATS_ID_SVM_PAGEFAULT_COUNT: Total SVM page faults handled.
* @XE_GT_STATS_ID_TLB_INVAL: Total GPU Translation Lookaside Buffer (TLB)
* invalidations issued.
@@ -22,6 +35,8 @@
* handled.
* @XE_GT_STATS_ID_VMA_PAGEFAULT_KB: Size (KiB) of VMAs involved in
* buffer-object page fault handling.
+ * @XE_GT_STATS_ID_PAGEFAULT_US: Cumulative time (µs) handling of all page
+ * faults.
* @XE_GT_STATS_ID_INVALID_PREFETCH_PAGEFAULT_COUNT: GPU prefetch faults for
* addresses with no valid backing.
*
@@ -127,12 +142,19 @@
* See Documentation/gpu/xe/xe_gt_stats.rst.
*/
enum xe_gt_stats_id {
+ XE_GT_STATS_ID_CHAIN_PAGEFAULT_COUNT,
+ XE_GT_STATS_ID_CHAIN_IRQ_PAGEFAULT_COUNT,
+ XE_GT_STATS_ID_CHAIN_DRAIN_IRQ_PAGEFAULT_COUNT,
+ XE_GT_STATS_ID_CHAIN_MISMATCH_PAGEFAULT_COUNT,
+ XE_GT_STATS_ID_PARALLEL_PAGEFAULT_COUNT,
+ XE_GT_STATS_ID_LAST_PAGEFAULT_COUNT,
XE_GT_STATS_ID_SVM_PAGEFAULT_COUNT,
XE_GT_STATS_ID_TLB_INVAL,
XE_GT_STATS_ID_SVM_TLB_INVAL_COUNT,
XE_GT_STATS_ID_SVM_TLB_INVAL_US,
XE_GT_STATS_ID_VMA_PAGEFAULT_COUNT,
XE_GT_STATS_ID_VMA_PAGEFAULT_KB,
+ XE_GT_STATS_ID_PAGEFAULT_US,
XE_GT_STATS_ID_INVALID_PREFETCH_PAGEFAULT_COUNT,
XE_GT_STATS_ID_SVM_4K_PAGEFAULT_COUNT,
XE_GT_STATS_ID_SVM_64K_PAGEFAULT_COUNT,
diff --git a/drivers/gpu/drm/xe/xe_guc.c b/drivers/gpu/drm/xe/xe_guc.c
index 4286bd05c686..c7f8bbd4cb92 100644
--- a/drivers/gpu/drm/xe/xe_guc.c
+++ b/drivers/gpu/drm/xe/xe_guc.c
@@ -39,6 +39,7 @@
#include "xe_guc_rc.h"
#include "xe_guc_relay.h"
#include "xe_guc_submit.h"
+#include "xe_log.h"
#include "xe_memirq.h"
#include "xe_mmio.h"
#include "xe_platform_types.h"
@@ -1542,8 +1543,9 @@ retry:
/* scratch registers might be cleared during FLR, try once more */
if (!header) {
if (++lost > MAX_RETRIES_ON_FLR) {
- xe_gt_err(gt, "GuC mmio request %#x: lost, too many retries %u\n",
- request[0], lost);
+ xe_log_err(gt, GUC, -ENOLINK,
+ "MMIO request %#x: lost, too many retries %u\n",
+ request[0], lost);
return -ENOLINK;
}
xe_gt_dbg(gt, "GuC mmio request %#x: lost, trying again\n", request[0]);
@@ -1551,8 +1553,8 @@ retry:
goto retry;
}
timeout:
- xe_gt_err(gt, "GuC mmio request %#x: no reply %#x\n",
- request[0], header);
+ xe_log_err(gt, GUC, ret, "MMIO request %#x: no reply %#x\n",
+ request[0], header);
return ret;
}
@@ -1607,16 +1609,16 @@ timeout:
return -EREMCHG;
}
- xe_gt_err(gt, "GuC mmio request %#x: failure %#x hint %#x\n",
- request[0], error, hint);
+ xe_log_err(gt, GUC, -ENXIO, "MMIO request %#x: failure %#x hint %#x\n",
+ request[0], error, hint);
return -ENXIO;
}
if (FIELD_GET(GUC_HXG_MSG_0_TYPE, header) !=
GUC_HXG_TYPE_RESPONSE_SUCCESS) {
proto:
- xe_gt_err(gt, "GuC mmio request %#x: unexpected reply %#x\n",
- request[0], header);
+ xe_log_err(gt, GUC, -EPROTO, "MMIO request %#x: unexpected reply %#x\n",
+ request[0], header);
return -EPROTO;
}
diff --git a/drivers/gpu/drm/xe/xe_guc_ct.c b/drivers/gpu/drm/xe/xe_guc_ct.c
index fe70c0fd85c5..5c4733da385c 100644
--- a/drivers/gpu/drm/xe/xe_guc_ct.c
+++ b/drivers/gpu/drm/xe/xe_guc_ct.c
@@ -265,7 +265,7 @@ static bool g2h_fence_needs_alloc(struct g2h_fence *g2h_fence)
#define CTB_DESC_SIZE ALIGN(sizeof(struct guc_ct_buffer_desc), SZ_2K)
#define CTB_H2G_BUFFER_OFFSET (CTB_DESC_SIZE * 2)
#define CTB_G2H_BUFFER_OFFSET (CTB_DESC_SIZE * 2)
-#define CTB_H2G_BUFFER_SIZE (SZ_4K)
+#define CTB_H2G_BUFFER_SIZE (SZ_16K)
#define CTB_H2G_BUFFER_DWORDS (CTB_H2G_BUFFER_SIZE / sizeof(u32))
#define CTB_G2H_BUFFER_SIZE (SZ_128K)
#define CTB_G2H_BUFFER_DWORDS (CTB_G2H_BUFFER_SIZE / sizeof(u32))
@@ -939,7 +939,7 @@ static bool vf_action_can_safely_fail(struct xe_device *xe, u32 action)
#define H2G_CT_HEADERS (GUC_CTB_HDR_LEN + 1) /* one DW CTB header and one DW HxG header */
static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
- u32 ct_fence_value, bool want_response)
+ u32 ct_fence_value, bool want_response, bool defer_flush)
{
struct xe_device *xe = ct_to_xe(ct);
struct xe_gt *gt = ct_to_gt(ct);
@@ -956,7 +956,6 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
xe_gt_assert(gt, full_len <= GUC_CTB_MSG_MAX_LEN);
if (IS_ENABLED(CONFIG_DRM_XE_DEBUG)) {
- u32 desc_tail = desc_read(xe, h2g, tail);
u32 desc_head = desc_read(xe, h2g, head);
u32 desc_status;
@@ -966,12 +965,6 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
goto corrupted;
}
- if (tail != desc_tail) {
- desc_write(xe, h2g, status, desc_status | GUC_CTB_STATUS_MISMATCH);
- xe_gt_err(gt, "CT write: tail was modified %u != %u\n", desc_tail, tail);
- goto corrupted;
- }
-
if (tail > h2g->info.size) {
desc_write(xe, h2g, status, desc_status | GUC_CTB_STATUS_OVERFLOW);
xe_gt_err(gt, "CT write: tail out of range: %u vs %u\n",
@@ -993,7 +986,10 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
(h2g->info.size - tail) * sizeof(u32));
h2g_reserve_space(ct, (h2g->info.size - tail));
h2g->info.tail = 0;
- desc_write(xe, h2g, tail, h2g->info.tail);
+ if (!defer_flush) {
+ xe_device_wmb(xe);
+ desc_write(xe, h2g, tail, h2g->info.tail);
+ }
return -EAGAIN;
}
@@ -1024,14 +1020,15 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
/* Write H2G ensuring visible before descriptor update */
xe_map_memcpy_to(xe, &map, 0, cmd, H2G_CT_HEADERS * sizeof(u32));
xe_map_memcpy_to(xe, &map, H2G_CT_HEADERS * sizeof(u32), action, len * sizeof(u32));
- xe_device_wmb(xe);
-
/* Update local copies */
h2g->info.tail = (tail + full_len) % h2g->info.size;
h2g_reserve_space(ct, full_len);
/* Update descriptor */
- desc_write(xe, h2g, tail, h2g->info.tail);
+ if (!defer_flush) {
+ xe_device_wmb(xe);
+ desc_write(xe, h2g, tail, h2g->info.tail);
+ }
/*
* desc_read() performs an VRAM read which serializes the CPU and drains
@@ -1052,7 +1049,7 @@ corrupted:
static int __guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action,
u32 len, u32 g2h_len, u32 num_g2h,
- struct g2h_fence *g2h_fence)
+ struct g2h_fence *g2h_fence, bool defer_flush)
{
struct xe_gt *gt = ct_to_gt(ct);
u16 seqno;
@@ -1112,7 +1109,7 @@ retry:
if (unlikely(ret))
goto out_unlock;
- ret = h2g_write(ct, action, len, seqno, !!g2h_fence);
+ ret = h2g_write(ct, action, len, seqno, !!g2h_fence, defer_flush);
if (unlikely(ret)) {
if (ret == -EAGAIN)
goto retry;
@@ -1120,7 +1117,8 @@ retry:
}
__g2h_reserve_space(ct, g2h_len, num_g2h);
- xe_guc_notify(ct_to_guc(ct));
+ if (!defer_flush)
+ xe_guc_notify(ct_to_guc(ct));
out_unlock:
if (g2h_len)
spin_unlock_irq(&ct->fast_lock);
@@ -1196,7 +1194,7 @@ static bool guc_ct_send_wait_for_retry(struct xe_guc_ct *ct, u32 len,
static int guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len,
u32 g2h_len, u32 num_g2h,
- struct g2h_fence *g2h_fence)
+ struct g2h_fence *g2h_fence, bool defer_flush)
{
struct xe_gt *gt = ct_to_gt(ct);
unsigned int sleep_period_ms = 1;
@@ -1209,9 +1207,10 @@ static int guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len,
try_again:
ret = __guc_ct_send_locked(ct, action, len, g2h_len, num_g2h,
- g2h_fence);
+ g2h_fence, defer_flush);
if (unlikely(ret == -EBUSY)) {
+ xe_guc_ct_send_flush(ct);
if (!guc_ct_send_wait_for_retry(ct, len, g2h_len, g2h_fence,
&sleep_period_ms, &sleep_total_ms))
goto broken;
@@ -1235,7 +1234,8 @@ static int guc_ct_send(struct xe_guc_ct *ct, const u32 *action, u32 len,
xe_gt_assert(ct_to_gt(ct), !g2h_len || !g2h_fence);
mutex_lock(&ct->lock);
- ret = guc_ct_send_locked(ct, action, len, g2h_len, num_g2h, g2h_fence);
+ ret = guc_ct_send_locked(ct, action, len, g2h_len, num_g2h, g2h_fence,
+ false);
mutex_unlock(&ct->lock);
return ret;
@@ -1283,25 +1283,76 @@ int xe_guc_ct_send(struct xe_guc_ct *ct, const u32 *action, u32 len,
return ret;
}
+/**
+ * xe_guc_ct_send_locked() - submit a GuC CT H2G message with CT lock held
+ * @ct: GuC CT object
+ * @action: payload dwords (HxG header dword is expected at @action[-1])
+ * @len: number of payload dwords in @action
+ * @defer_flush: defer publishing/doorbell for batching
+ *
+ * Sends a single H2G message to the GuC CT buffer while the caller already
+ * holds @ct->lock.
+ *
+ * If @defer_flush is false, the function completes the submission immediately:
+ * it makes the payload visible to the device, updates the H2G descriptor and
+ * rings the GuC doorbell.
+ *
+ * If @defer_flush is true, the message payload is copied into the H2G ring and
+ * the software tail is advanced, but the descriptor update and doorbell are
+ * deferred so multiple messages can be batched. In this mode, the caller must
+ * eventually call xe_guc_ct_send_flush() (still holding @ct->lock) to publish
+ * the descriptor and notify the GuC. On internal retry paths (-EBUSY), the
+ * implementation may force a flush to ensure forward progress.
+ *
+ * Return: 0 on success, negative errno on failure.
+ *
+ * Locking:
+ * Must be called with @ct->lock held.
+ */
int xe_guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len,
- u32 g2h_len, u32 num_g2h)
+ bool defer_flush)
{
int ret;
- ret = guc_ct_send_locked(ct, action, len, g2h_len, num_g2h, NULL);
+ ret = guc_ct_send_locked(ct, action, len, 0, 0, NULL, defer_flush);
if (ret == -EDEADLK)
kick_reset(ct);
return ret;
}
+/**
+ * xe_guc_ct_send_flush() - flush pending GuC CT H2G writes
+ * @ct: GuC CT instance
+ *
+ * Some callers batch multiple H2G writes using xe_guc_ct_send_locked() in
+ * "write-only" mode (i.e., queue the message payloads but defer ringing the
+ * doorbell / updating the CT descriptor). This helper completes the submission
+ * by ensuring the payload writes are visible to the device, updating the H2G
+ * descriptor, and ringing the GuC CT doorbell.
+ *
+ * Locking:
+ * Must be called with @ct->lock held.
+ */
+void xe_guc_ct_send_flush(struct xe_guc_ct *ct)
+{
+ struct xe_device *xe = ct_to_xe(ct);
+ struct guc_ctb *h2g = &ct->ctbs.h2g;
+
+ lockdep_assert_held(&ct->lock);
+
+ xe_device_wmb(xe);
+ desc_write(xe, h2g, tail, h2g->info.tail);
+ xe_guc_notify(ct_to_guc(ct));
+}
+
int xe_guc_ct_send_g2h_handler(struct xe_guc_ct *ct, const u32 *action, u32 len)
{
int ret;
lockdep_assert_held(&ct->lock);
- ret = guc_ct_send_locked(ct, action, len, 0, 0, NULL);
+ ret = guc_ct_send_locked(ct, action, len, 0, 0, NULL, false);
if (ret == -EDEADLK)
kick_reset(ct);
diff --git a/drivers/gpu/drm/xe/xe_guc_ct.h b/drivers/gpu/drm/xe/xe_guc_ct.h
index 767365a33dee..3ddc665ab84a 100644
--- a/drivers/gpu/drm/xe/xe_guc_ct.h
+++ b/drivers/gpu/drm/xe/xe_guc_ct.h
@@ -6,6 +6,8 @@
#ifndef _XE_GUC_CT_H_
#define _XE_GUC_CT_H_
+#include <linux/mutex.h>
+
#include "xe_guc_ct_types.h"
struct drm_printer;
@@ -54,7 +56,7 @@ static inline void xe_guc_ct_irq_handler(struct xe_guc_ct *ct)
int xe_guc_ct_send(struct xe_guc_ct *ct, const u32 *action, u32 len,
u32 g2h_len, u32 num_g2h);
int xe_guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len,
- u32 g2h_len, u32 num_g2h);
+ bool defer_flush);
int xe_guc_ct_send_recv(struct xe_guc_ct *ct, const u32 *action, u32 len,
u32 *response_buffer);
static inline int
@@ -63,6 +65,8 @@ xe_guc_ct_send_block(struct xe_guc_ct *ct, const u32 *action, u32 len)
return xe_guc_ct_send_recv(ct, action, len, NULL);
}
+void xe_guc_ct_send_flush(struct xe_guc_ct *ct);
+
/* This is only version of the send CT you can call from a G2H handler */
int xe_guc_ct_send_g2h_handler(struct xe_guc_ct *ct, const u32 *action,
u32 len);
@@ -87,4 +91,36 @@ static inline void xe_guc_ct_wake_waiters(struct xe_guc_ct *ct)
wake_up_all(&ct->wq);
}
+/**
+ * xe_guc_ct_lock() - take the GuC CT mutex
+ * @ct: GuC CT object
+ *
+ * Wrapper around mutex_lock(&ct->lock) for cases where CT operations need to be
+ * performed from contexts that want an explicit "CT locked" pair without
+ * exporting the lock itself.
+ *
+ * Return/Locking:
+ * Acquires @ct->lock.
+ */
+static inline void xe_guc_ct_lock(struct xe_guc_ct *ct)
+__acquires(&ct->lock)
+{
+ mutex_lock(&ct->lock);
+}
+
+/**
+ * xe_guc_ct_unlock() - release the GuC CT mutex
+ * @ct: GuC CT object
+ *
+ * Counterpart to xe_guc_ct_lock().
+ *
+ * Locking:
+ * Releases @ct->lock.
+ */
+static inline void xe_guc_ct_unlock(struct xe_guc_ct *ct)
+__releases(&ct->lock)
+{
+ mutex_unlock(&ct->lock);
+}
+
#endif
diff --git a/drivers/gpu/drm/xe/xe_guc_exec_queue_types.h b/drivers/gpu/drm/xe/xe_guc_exec_queue_types.h
index acdc24d1a6bd..d27826b36649 100644
--- a/drivers/gpu/drm/xe/xe_guc_exec_queue_types.h
+++ b/drivers/gpu/drm/xe/xe_guc_exec_queue_types.h
@@ -36,7 +36,7 @@ struct xe_guc_exec_queue {
* a message needs to sent through the GPU scheduler but memory
* allocations are not allowed.
*/
-#define MAX_STATIC_MSG_TYPE 3
+#define MAX_STATIC_MSG_TYPE 4
struct xe_sched_msg static_msgs[MAX_STATIC_MSG_TYPE];
/** @destroy_async: do final destroy async from this worker */
struct work_struct destroy_async;
@@ -76,6 +76,36 @@ struct xe_guc_exec_queue {
* recovery.
*/
bool needs_resume;
+ /** @multi_queue: multi-queue group CGP state for VF post migration recovery */
+ struct {
+ /**
+ * @multi_queue.needs_cgp_sync: Needs a CGP_SYNC (dynamic CGP
+ * update) message replayed during recovery.
+ */
+ u8 needs_cgp_sync:1;
+ /**
+ * @multi_queue.re_register: A registration-time CGP update was
+ * interrupted by recovery; the queue must be re-registered.
+ */
+ u8 re_register:1;
+ /**
+ * @multi_queue.re_update: A dynamic CGP update was interrupted
+ * by recovery; the CGP update must be replayed.
+ */
+ u8 re_update:1;
+ /**
+ * @multi_queue.registering_cgp: This queue's currently
+ * outstanding CGP_SYNC is a registration (matched against
+ * group->cgp_update_q in revert).
+ */
+ u8 registering_cgp:1;
+ /**
+ * @multi_queue.updating_cgp: This queue's currently outstanding
+ * CGP_SYNC is a dynamic update (matched against
+ * group->cgp_update_q in revert).
+ */
+ u8 updating_cgp:1;
+ } multi_queue;
};
#endif
diff --git a/drivers/gpu/drm/xe/xe_guc_pagefault.c b/drivers/gpu/drm/xe/xe_guc_pagefault.c
index 607e32392f46..8f8210a732e9 100644
--- a/drivers/gpu/drm/xe/xe_guc_pagefault.c
+++ b/drivers/gpu/drm/xe/xe_guc_pagefault.c
@@ -10,6 +10,22 @@
#include "xe_pagefault.h"
#include "xe_pagefault_types.h"
+#define XE_GUC_PAGEFAULT_FLUSH_PERIOD BIT(4) /* Sixteen */
+
+static void guc_ack_fault_begin(void *private)
+{
+ struct xe_guc *guc = private;
+
+ xe_guc_ct_lock(&guc->ct);
+
+ BUILD_BUG_ON(((XE_GUC_PAGEFAULT_FLUSH_PERIOD - 1) &
+ XE_GUC_PAGEFAULT_FLUSH_PERIOD) != 0);
+
+ /* Ack the 2nd, then 18th, etc... */
+ guc->pagefault_ack_counter =
+ XE_GUC_PAGEFAULT_FLUSH_PERIOD - 1;
+}
+
static void guc_ack_fault(struct xe_pagefault *pf, int err)
{
u32 vfid = FIELD_GET(PFD_VFID, pf->producer.msg[2]);
@@ -36,12 +52,26 @@ static void guc_ack_fault(struct xe_pagefault *pf, int err)
FIELD_PREP(PFR_PDATA, pdata),
};
struct xe_guc *guc = pf->producer.private;
+ bool write_only = guc->pagefault_ack_counter++ &
+ (XE_GUC_PAGEFAULT_FLUSH_PERIOD - 1);
+
+ xe_guc_ct_send_locked(&guc->ct, action, ARRAY_SIZE(action),
+ write_only);
+}
+
+static void guc_ack_fault_end(void *private)
+{
+ struct xe_guc *guc = private;
- xe_guc_ct_send(&guc->ct, action, ARRAY_SIZE(action), 0, 0);
+ if ((guc->pagefault_ack_counter & (XE_GUC_PAGEFAULT_FLUSH_PERIOD - 1)) != 1)
+ xe_guc_ct_send_flush(&guc->ct);
+ xe_guc_ct_unlock(&guc->ct);
}
static const struct xe_pagefault_ops guc_pagefault_ops = {
+ .ack_fault_begin = guc_ack_fault_begin,
.ack_fault = guc_ack_fault,
+ .ack_fault_end = guc_ack_fault_end,
};
/**
@@ -89,8 +119,11 @@ int xe_guc_pagefault_handler(struct xe_guc *guc, u32 *msg, u32 len)
FIELD_GET(PFD_FAULT_LEVEL, msg[0])) |
FIELD_PREP(XE_PAGEFAULT_TYPE_MASK,
FIELD_GET(PFD_FAULT_TYPE, msg[2]));
- pf.consumer.engine_class = FIELD_GET(PFD_ENG_CLASS, msg[0]);
- pf.consumer.engine_instance = FIELD_GET(PFD_ENG_INSTANCE, msg[0]);
+ pf.consumer.engine_class_instance =
+ FIELD_PREP(XE_PAGEFAULT_ENGINE_CLASS_MASK,
+ FIELD_GET(PFD_ENG_CLASS, msg[0])) |
+ FIELD_PREP(XE_PAGEFAULT_ENGINE_INSTANCE_MASK,
+ FIELD_GET(PFD_ENG_INSTANCE, msg[0]));
pf.producer.private = guc;
pf.producer.ops = &guc_pagefault_ops;
diff --git a/drivers/gpu/drm/xe/xe_guc_pc.c b/drivers/gpu/drm/xe/xe_guc_pc.c
index 7cf8f4858598..097b075bd89a 100644
--- a/drivers/gpu/drm/xe/xe_guc_pc.c
+++ b/drivers/gpu/drm/xe/xe_guc_pc.c
@@ -1241,8 +1241,21 @@ static int pc_modify_defaults(struct xe_guc_pc *pc)
if (xe->info.platform == XE_PANTHERLAKE) {
ret = pc_action_set_dcc(pc, false);
- if (unlikely(ret))
+ if (unlikely(ret)) {
xe_gt_err(gt, "Failed to modify DCC default: %pe\n", ERR_PTR(ret));
+ return ret;
+ }
+
+ if (xe_gt_is_main_type(gt) &&
+ GUC_FIRMWARE_VER_AT_LEAST(&gt->uc.guc, 70, 48, 0)) {
+ ret = pc_action_set_param(pc,
+ SLPC_PARAM_SET_IBC_VERSION,
+ 3);
+ if (unlikely(ret)) {
+ xe_gt_err(gt, "Failed to modify IBC version: %pe\n", ERR_PTR(ret));
+ return ret;
+ }
+ }
}
return ret;
diff --git a/drivers/gpu/drm/xe/xe_guc_submit.c b/drivers/gpu/drm/xe/xe_guc_submit.c
index 8aaed4fd13ea..99d8c807ff05 100644
--- a/drivers/gpu/drm/xe/xe_guc_submit.c
+++ b/drivers/gpu/drm/xe/xe_guc_submit.c
@@ -800,9 +800,12 @@ static void xe_guc_exec_queue_group_cgp_update(struct xe_device *xe,
}
}
+#define CGP_SYNC_REGISTRATION BIT(0)
+
static void xe_guc_exec_queue_group_cgp_sync(struct xe_guc *guc,
struct xe_exec_queue *q,
- const u32 *action, u32 len)
+ const u32 *action, u32 len,
+ unsigned int flags)
{
struct xe_exec_queue_group *group = q->multi_queue.group;
struct xe_device *xe = guc_to_xe(guc);
@@ -815,13 +818,12 @@ static void xe_guc_exec_queue_group_cgp_sync(struct xe_guc *guc,
* Hence, no locking is required here.
* Wait for any pending CGP_SYNC_DONE response before updating the
* CGP page and sending CGP_SYNC message.
- *
- * FIXME: Support VF migration
*/
ret = wait_event_timeout(guc->ct.wq,
!READ_ONCE(group->sync_pending) ||
- xe_guc_read_stopped(guc), HZ);
- if (!ret || xe_guc_read_stopped(guc)) {
+ xe_guc_read_stopped(guc) || vf_recovery(guc),
+ HZ);
+ if ((!ret && !vf_recovery(guc)) || xe_guc_read_stopped(guc)) {
/* CGP_SYNC failed. Reset gt, cleanup the group */
xe_gt_warn(guc_to_gt(guc), "Wait for CGP_SYNC_DONE response failed!\n");
set_exec_queue_group_banned(q);
@@ -830,17 +832,45 @@ static void xe_guc_exec_queue_group_cgp_sync(struct xe_guc *guc,
return;
}
+ /*
+ * If woken by VF migration recovery, do not touch the CGP or send: the
+ * message would be lost and, for a registration, GuC must (re-)register
+ * the context before its CGP entry may be read. Flag the queue so revert
+ * replays it - a registration by re-registration, a dynamic update by a
+ * replayed CGP_SYNC - and bail.
+ */
+ if (vf_recovery(guc)) {
+ if (flags & CGP_SYNC_REGISTRATION)
+ q->guc->multi_queue.re_register = true;
+ else
+ q->guc->multi_queue.re_update = true;
+ return;
+ }
+
scoped_guard(spinlock, &q->multi_queue.lock)
priority = q->multi_queue.priority;
xe_lrc_set_multi_queue_priority(q->lrc[0], priority);
xe_guc_exec_queue_group_cgp_update(xe, q);
+ /*
+ * Record the nature of this outstanding sync so revert can replay it if
+ * its CGP_SYNC_DONE is lost across a migration: a registration is
+ * recovered by re-registration, a dynamic update by a replayed CGP_SYNC.
+ */
+ if (flags & CGP_SYNC_REGISTRATION) {
+ q->guc->multi_queue.registering_cgp = true;
+ q->guc->multi_queue.updating_cgp = false;
+ } else {
+ q->guc->multi_queue.updating_cgp = true;
+ q->guc->multi_queue.registering_cgp = false;
+ }
+ WRITE_ONCE(group->cgp_update_q, q);
WRITE_ONCE(group->sync_pending, true);
xe_guc_ct_send(&guc->ct, action, len, G2H_LEN_DW_MULTI_QUEUE_CONTEXT, 1);
}
-static void guc_exec_queue_send_cgp_sync(struct xe_exec_queue *q)
+static void guc_exec_queue_send_cgp_sync(struct xe_exec_queue *q, unsigned int flags)
{
#define MAX_MULTI_QUEUE_CGP_SYNC_SIZE (2)
struct xe_guc *guc = exec_queue_to_guc(q);
@@ -854,7 +884,7 @@ static void guc_exec_queue_send_cgp_sync(struct xe_exec_queue *q)
xe_gt_assert(guc_to_gt(guc), len <= MAX_MULTI_QUEUE_CGP_SYNC_SIZE);
#undef MAX_MULTI_QUEUE_CGP_SYNC_SIZE
- xe_guc_exec_queue_group_cgp_sync(guc, q, action, len);
+ xe_guc_exec_queue_group_cgp_sync(guc, q, action, len, flags);
}
static void __register_exec_queue_group(struct xe_exec_queue *q,
@@ -882,7 +912,8 @@ static void __register_exec_queue_group(struct xe_exec_queue *q,
* XE_GUC_ACTION_NOTIFY_MULTI_QUEUE_CONTEXT_CGP_SYNC_DONE response
* from guc.
*/
- xe_guc_exec_queue_group_cgp_sync(guc, q, action, len);
+ xe_guc_exec_queue_group_cgp_sync(guc, q, action, len,
+ CGP_SYNC_REGISTRATION);
}
static void __register_mlrc_exec_queue(struct xe_guc *guc,
@@ -1042,7 +1073,7 @@ static void register_exec_queue(struct xe_exec_queue *q, int ctx_type)
init_policies(guc, q);
if (xe_exec_queue_is_multi_queue_secondary(q))
- guc_exec_queue_send_cgp_sync(q);
+ guc_exec_queue_send_cgp_sync(q, CGP_SYNC_REGISTRATION);
}
static u32 wq_space_until_wrap(struct xe_exec_queue *q)
@@ -1559,8 +1590,14 @@ guc_exec_queue_timedout_job(struct drm_sched_job *drm_job)
if (!skip_timeout_check && !check_timeout(q, job))
goto rearm;
+ /*
+ * Killed queues must not newly wedge the device, but preserve an
+ * already-wedged state to avoid warning on teardown timeouts.
+ */
if (!exec_queue_killed(q))
wedged = guc_submit_hint_wedged(exec_queue_to_guc(q));
+ else
+ wedged = xe_device_wedged(xe);
set_exec_queue_banned(q);
@@ -1781,32 +1818,15 @@ static void __guc_exec_queue_destroy_async(struct work_struct *w)
static void guc_exec_queue_destroy_async(struct xe_exec_queue *q)
{
INIT_WORK(&q->guc->destroy_async, __guc_exec_queue_destroy_async);
-
- /* We must block on kernel engines so slabs are empty on driver unload */
- if (q->flags & EXEC_QUEUE_FLAG_PERMANENT || exec_queue_wedged(q))
- guc_exec_queue_do_destroy(q);
- else
- xe_destroy_wq_queue(&q->guc->destroy_async);
+ xe_destroy_wq_queue(&q->guc->destroy_async);
}
-static void __guc_exec_queue_destroy(struct xe_guc *guc, struct xe_exec_queue *q)
-{
- /*
- * Might be done from within the GPU scheduler, need to do async as we
- * fini the scheduler when the engine is fini'd, the scheduler can't
- * complete fini within itself (circular dependency). Async resolves
- * this we and don't really care when everything is fini'd, just that it
- * is.
- */
- guc_exec_queue_destroy_async(q);
-}
-
-static void __guc_exec_queue_process_msg_cleanup(struct xe_sched_msg *msg)
+static void __guc_exec_queue_process_msg_cleanup(struct xe_sched_msg *msg,
+ bool bound)
{
struct xe_exec_queue *q = msg->private_data;
struct xe_guc *guc = exec_queue_to_guc(q);
- xe_gt_assert(guc_to_gt(guc), !(q->flags & EXEC_QUEUE_FLAG_PERMANENT));
trace_xe_exec_queue_cleanup_entity(q);
/*
@@ -1819,10 +1839,12 @@ static void __guc_exec_queue_process_msg_cleanup(struct xe_sched_msg *msg)
* it is safe to directly destroy the exec queue on driver side, as the GuC
* will not process further requests and all resources must be cleaned up locally.
*/
- if (exec_queue_registered(q) && xe_uc_fw_is_running(&guc->fw))
+ /* A wedged GuC won't answer the H2G, so tear down on the driver side. */
+ if (bound && !exec_queue_wedged(q) && exec_queue_registered(q) &&
+ xe_uc_fw_is_running(&guc->fw))
disable_scheduling_deregister(guc, q);
else
- __guc_exec_queue_destroy(guc, q);
+ guc_exec_queue_destroy_async(q);
}
static bool guc_exec_queue_allowed_to_change_state(struct xe_exec_queue *q)
@@ -1830,12 +1852,13 @@ static bool guc_exec_queue_allowed_to_change_state(struct xe_exec_queue *q)
return !exec_queue_killed_or_banned_or_wedged(q) && exec_queue_registered(q);
}
-static void __guc_exec_queue_process_msg_set_sched_props(struct xe_sched_msg *msg)
+static void __guc_exec_queue_process_msg_set_sched_props(struct xe_sched_msg *msg,
+ bool bound)
{
struct xe_exec_queue *q = msg->private_data;
struct xe_guc *guc = exec_queue_to_guc(q);
- if (guc_exec_queue_allowed_to_change_state(q))
+ if (guc_exec_queue_allowed_to_change_state(q) && bound)
init_policies(guc, q);
kfree(msg);
}
@@ -1873,13 +1896,14 @@ static void suspend_fence_signal(struct xe_exec_queue *q)
__suspend_fence_signal(q);
}
-static void __guc_exec_queue_process_msg_suspend(struct xe_sched_msg *msg)
+static void __guc_exec_queue_process_msg_suspend(struct xe_sched_msg *msg,
+ bool bound)
{
struct xe_exec_queue *q = msg->private_data;
struct xe_guc *guc = exec_queue_to_guc(q);
if (guc_exec_queue_allowed_to_change_state(q) && !exec_queue_suspended(q) &&
- exec_queue_enabled(q)) {
+ exec_queue_enabled(q) && bound) {
wait_event(guc->ct.wq, vf_recovery(guc) ||
((q->guc->resume_time != RESUME_PENDING ||
xe_guc_read_stopped(guc)) && !exec_queue_pending_disable(q)));
@@ -1903,11 +1927,12 @@ static void __guc_exec_queue_process_msg_suspend(struct xe_sched_msg *msg)
}
}
-static void __guc_exec_queue_process_msg_resume(struct xe_sched_msg *msg)
+static void __guc_exec_queue_process_msg_resume(struct xe_sched_msg *msg,
+ bool bound)
{
struct xe_exec_queue *q = msg->private_data;
- if (guc_exec_queue_allowed_to_change_state(q)) {
+ if (guc_exec_queue_allowed_to_change_state(q) && bound) {
clear_exec_queue_suspended(q);
if (!exec_queue_enabled(q)) {
q->guc->resume_time = RESUME_PENDING;
@@ -1919,52 +1944,79 @@ static void __guc_exec_queue_process_msg_resume(struct xe_sched_msg *msg)
}
}
-static void __guc_exec_queue_process_msg_set_multi_queue_priority(struct xe_sched_msg *msg)
+static void __guc_exec_queue_process_msg_set_multi_queue_priority(struct xe_sched_msg *msg,
+ bool bound)
{
struct xe_exec_queue *q = msg->private_data;
- if (guc_exec_queue_allowed_to_change_state(q))
- guc_exec_queue_send_cgp_sync(q);
+ if (guc_exec_queue_allowed_to_change_state(q) && bound)
+ guc_exec_queue_send_cgp_sync(q, 0);
kfree(msg);
}
+static void __guc_exec_queue_process_msg_cgp_sync(struct xe_sched_msg *msg,
+ bool bound)
+{
+ struct xe_exec_queue *q = msg->private_data;
+
+ /*
+ * Replay a dynamic CGP update lost across VF migration by re-issuing the
+ * CGP update + CGP_SYNC (re-applies the current priority from
+ * q->multi_queue.priority).
+ */
+ if (guc_exec_queue_allowed_to_change_state(q) && bound)
+ guc_exec_queue_send_cgp_sync(q, 0);
+}
+
#define CLEANUP 1 /* Non-zero values to catch uninitialized msg */
#define SET_SCHED_PROPS 2
#define SUSPEND 3
#define RESUME 4
#define SET_MULTI_QUEUE_PRIORITY 5
+#define CGP_SYNC_MSG 6
#define OPCODE_MASK 0xf
#define MSG_LOCKED BIT(8)
#define MSG_HEAD BIT(9)
+#define MSG_PM_REF BIT(10)
static void guc_exec_queue_process_msg(struct xe_sched_msg *msg)
{
struct xe_device *xe = guc_to_xe(exec_queue_to_guc(msg->private_data));
+ int idx;
+ bool pm_ref = !!(msg->opcode & MSG_PM_REF);
+ bool bound = drm_dev_enter(&xe->drm, &idx);
trace_xe_sched_msg_recv(msg);
- switch (msg->opcode) {
+ switch (msg->opcode & OPCODE_MASK) {
case CLEANUP:
- __guc_exec_queue_process_msg_cleanup(msg);
+ __guc_exec_queue_process_msg_cleanup(msg, bound);
break;
case SET_SCHED_PROPS:
- __guc_exec_queue_process_msg_set_sched_props(msg);
+ __guc_exec_queue_process_msg_set_sched_props(msg, bound);
break;
case SUSPEND:
- __guc_exec_queue_process_msg_suspend(msg);
+ __guc_exec_queue_process_msg_suspend(msg, bound);
break;
case RESUME:
- __guc_exec_queue_process_msg_resume(msg);
+ __guc_exec_queue_process_msg_resume(msg, bound);
break;
case SET_MULTI_QUEUE_PRIORITY:
- __guc_exec_queue_process_msg_set_multi_queue_priority(msg);
+ __guc_exec_queue_process_msg_set_multi_queue_priority(msg, bound);
+ break;
+ case CGP_SYNC_MSG:
+ __guc_exec_queue_process_msg_cgp_sync(msg, bound);
break;
default:
XE_WARN_ON("Unknown message type");
}
- xe_pm_runtime_put(xe);
+ if (pm_ref)
+ xe_pm_runtime_put(xe);
+
+ if (bound)
+ drm_dev_exit(idx);
}
static const struct drm_sched_backend_ops drm_sched_ops = {
@@ -2089,10 +2141,16 @@ static void guc_exec_queue_kill(struct xe_exec_queue *q)
static void guc_exec_queue_add_msg(struct xe_exec_queue *q, struct xe_sched_msg *msg,
u32 opcode)
{
- xe_pm_runtime_get_noresume(guc_to_xe(exec_queue_to_guc(q)));
+ struct xe_device *xe = guc_to_xe(exec_queue_to_guc(q));
+ int idx;
+ bool bound = drm_dev_enter(&xe->drm, &idx);
INIT_LIST_HEAD(&msg->link);
msg->opcode = opcode & OPCODE_MASK;
+ if (bound) {
+ xe_pm_runtime_get_noresume(xe);
+ msg->opcode |= MSG_PM_REF;
+ }
msg->private_data = q;
trace_xe_sched_msg_add(msg);
@@ -2102,6 +2160,9 @@ static void guc_exec_queue_add_msg(struct xe_exec_queue *q, struct xe_sched_msg
xe_sched_add_msg_locked(&q->guc->sched, msg);
else
xe_sched_add_msg(&q->guc->sched, msg);
+
+ if (bound)
+ drm_dev_exit(idx);
}
static void guc_exec_queue_try_add_msg_head(struct xe_exec_queue *q,
@@ -2129,14 +2190,12 @@ static bool guc_exec_queue_try_add_msg(struct xe_exec_queue *q,
#define STATIC_MSG_CLEANUP 0
#define STATIC_MSG_SUSPEND 1
#define STATIC_MSG_RESUME 2
+#define STATIC_MSG_CGP_SYNC 3
static void guc_exec_queue_destroy(struct xe_exec_queue *q)
{
struct xe_sched_msg *msg = q->guc->static_msgs + STATIC_MSG_CLEANUP;
- if (!(q->flags & EXEC_QUEUE_FLAG_PERMANENT) && !exec_queue_wedged(q))
- guc_exec_queue_add_msg(q, msg, CLEANUP);
- else
- __guc_exec_queue_destroy(exec_queue_to_guc(q), q);
+ guc_exec_queue_add_msg(q, msg, CLEANUP);
}
static int guc_exec_queue_set_priority(struct xe_exec_queue *q,
@@ -2601,7 +2660,7 @@ static void guc_exec_queue_stop(struct xe_guc *guc, struct xe_exec_queue *q)
}
if (do_destroy)
- __guc_exec_queue_destroy(guc, q);
+ guc_exec_queue_destroy_async(q);
}
static int guc_submit_reset_prepare(struct xe_guc *guc)
@@ -2717,6 +2776,57 @@ static void guc_exec_queue_revert_pending_state_change(struct xe_guc *guc,
q->guc->id);
}
+ /*
+ * A registration time CGP update that bailed when woken by VF recovery.
+ * Re-register the queue.
+ */
+ if (q->guc->multi_queue.re_register) {
+ clear_exec_queue_registered(q);
+ q->guc->multi_queue.re_register = false;
+ xe_gt_dbg(guc_to_gt(guc), "Replay REGISTER (cgp) - guc_id=%d",
+ q->guc->id);
+ }
+
+ /*
+ * If a CGP update gets dropped during migration, CGP_SYNC_DONE will not
+ * be received (sync_pending still set and this queue owns it). Recover
+ * it the same way and clear the stuck sync_pending.
+ */
+ if (xe_exec_queue_is_multi_queue(q)) {
+ struct xe_exec_queue_group *group = q->multi_queue.group;
+
+ if (q == READ_ONCE(group->cgp_update_q) &&
+ READ_ONCE(group->sync_pending)) {
+ if (q->guc->multi_queue.registering_cgp) {
+ clear_exec_queue_registered(q);
+ xe_gt_dbg(guc_to_gt(guc), "Replay REGISTER (cgp sync) - guc_id=%d",
+ q->guc->id);
+ } else if (q->guc->multi_queue.updating_cgp) {
+ q->guc->multi_queue.needs_cgp_sync = true;
+ xe_gt_dbg(guc_to_gt(guc), "Replay CGP_SYNC - guc_id=%d",
+ q->guc->id);
+ }
+ q->guc->multi_queue.registering_cgp = false;
+ q->guc->multi_queue.updating_cgp = false;
+ WRITE_ONCE(group->cgp_update_q, NULL);
+ WRITE_ONCE(group->sync_pending, false);
+ }
+ }
+
+ /*
+ * A dynamic-time CGP update that bailed when woken by VF recovery.
+ * Replay the dynamic CGP update unless the queue is registered or being
+ * re-registered, which re-does the CGP anyway.
+ */
+ if (q->guc->multi_queue.re_update) {
+ q->guc->multi_queue.re_update = false;
+ if (exec_queue_registered(q)) {
+ q->guc->multi_queue.needs_cgp_sync = true;
+ xe_gt_dbg(guc_to_gt(guc), "Replay CGP_SYNC (re-update) - guc_id=%d",
+ q->guc->id);
+ }
+ }
+
q->guc->resume_time = 0;
}
@@ -2737,17 +2847,24 @@ static void lrc_parallel_clear(struct xe_lrc *lrc)
* during VF resume flows. The function scans the queue state, make adjustments
* as needed, and queues jobs / messages which replayed upon unpause.
*/
-static void guc_exec_queue_pause(struct xe_guc *guc, struct xe_exec_queue *q)
+static void guc_exec_queue_pause_prepare(struct xe_guc *guc, struct xe_exec_queue *q)
{
struct xe_gpu_scheduler *sched = &q->guc->sched;
- struct xe_sched_job *job;
- int i;
lockdep_assert_held(&guc->submission_state.lock);
/* Stop scheduling + flush any DRM scheduler operations */
xe_sched_submission_stop(sched);
cancel_delayed_work_sync(&sched->base.work_tdr);
+}
+
+static void guc_exec_queue_pause(struct xe_guc *guc, struct xe_exec_queue *q)
+{
+ struct xe_gpu_scheduler *sched = &q->guc->sched;
+ struct xe_sched_job *job;
+ int i;
+
+ lockdep_assert_held(&guc->submission_state.lock);
guc_exec_queue_revert_pending_state_change(guc, q);
@@ -2806,6 +2923,19 @@ void xe_guc_submit_pause_vf(struct xe_guc *guc)
xe_gt_assert(guc_to_gt(guc), vf_recovery(guc));
mutex_lock(&guc->submission_state.lock);
+ /*
+ * Stop all schedulers before reverting any queue: in a multi-queue
+ * group a secondary's run_job() can register the primary, which must
+ * not race an in-progress revert.
+ */
+ xa_for_each(&guc->submission_state.exec_queue_lookup, index, q) {
+ /* Prevent redundant attempts to stop parallel queues */
+ if (q->guc->id != index)
+ continue;
+
+ guc_exec_queue_pause_prepare(guc, q);
+ }
+
xa_for_each(&guc->submission_state.exec_queue_lookup, index, q) {
/* Prevent redundant attempts to stop parallel queues */
if (q->guc->id != index)
@@ -2922,6 +3052,16 @@ static void guc_exec_queue_replay_pending_state_change(struct xe_exec_queue *q)
struct xe_gpu_scheduler *sched = &q->guc->sched;
struct xe_sched_msg *msg;
+ if (q->guc->multi_queue.needs_cgp_sync) {
+ msg = q->guc->static_msgs + STATIC_MSG_CGP_SYNC;
+
+ xe_sched_msg_lock(sched);
+ guc_exec_queue_try_add_msg_head(q, msg, CGP_SYNC_MSG);
+ xe_sched_msg_unlock(sched);
+
+ q->guc->multi_queue.needs_cgp_sync = false;
+ }
+
if (q->guc->needs_cleanup) {
msg = q->guc->static_msgs + STATIC_MSG_CLEANUP;
@@ -3166,7 +3306,7 @@ static void handle_deregister_done(struct xe_guc *guc, struct xe_exec_queue *q)
trace_xe_exec_queue_deregister_done(q);
clear_exec_queue_registered(q);
- __guc_exec_queue_destroy(guc, q);
+ guc_exec_queue_destroy_async(q);
}
int xe_guc_deregister_done_handler(struct xe_guc *guc, u32 *msg, u32 len)
@@ -3405,7 +3545,8 @@ int xe_guc_exec_queue_cgp_context_error_handler(struct xe_guc *guc, u32 *msg,
int xe_guc_exec_queue_cgp_sync_done_handler(struct xe_guc *guc, u32 *msg, u32 len)
{
struct xe_device *xe = guc_to_xe(guc);
- struct xe_exec_queue *q;
+ struct xe_exec_queue_group *group;
+ struct xe_exec_queue *q, *upd_q;
u32 guc_id = msg[0];
if (unlikely(len < 1)) {
@@ -3422,8 +3563,20 @@ int xe_guc_exec_queue_cgp_sync_done_handler(struct xe_guc *guc, u32 *msg, u32 le
return -EPROTO;
}
+ /*
+ * The outstanding CGP update is now confirmed; clear the owning queue's
+ * tracking so a later migration does not needlessly replay it.
+ */
+ group = q->multi_queue.group;
+ upd_q = READ_ONCE(group->cgp_update_q);
+ if (upd_q) {
+ upd_q->guc->multi_queue.registering_cgp = false;
+ upd_q->guc->multi_queue.updating_cgp = false;
+ WRITE_ONCE(group->cgp_update_q, NULL);
+ }
+
/* Wakeup the serialized cgp update wait */
- WRITE_ONCE(q->multi_queue.group->sync_pending, false);
+ WRITE_ONCE(group->sync_pending, false);
xe_guc_ct_wake_waiters(&guc->ct);
return 0;
diff --git a/drivers/gpu/drm/xe/xe_guc_types.h b/drivers/gpu/drm/xe/xe_guc_types.h
index 31a2acb63ac3..3dbcb1331690 100644
--- a/drivers/gpu/drm/xe/xe_guc_types.h
+++ b/drivers/gpu/drm/xe/xe_guc_types.h
@@ -122,6 +122,12 @@ struct xe_guc {
struct xe_reg notify_reg;
/** @params: Control params for fw initialization */
u32 params[GUC_CTL_MAX_DWORDS];
+
+ /**
+ * @pagefault_ack_counter: Counter to determine when periodically ack
+ * pagefaults in a batch.
+ */
+ u32 pagefault_ack_counter;
};
#endif
diff --git a/drivers/gpu/drm/xe/xe_hwmon.c b/drivers/gpu/drm/xe/xe_hwmon.c
index de3f2aeffc3f..5284cab6703d 100644
--- a/drivers/gpu/drm/xe/xe_hwmon.c
+++ b/drivers/gpu/drm/xe/xe_hwmon.c
@@ -266,7 +266,7 @@ static struct xe_reg xe_hwmon_get_reg(struct xe_hwmon *hwmon, enum xe_hwmon_reg
switch (hwmon_reg) {
case REG_TEMP:
- if (xe->info.platform == XE_BATTLEMAGE) {
+ if (xe->info.platform == XE_BATTLEMAGE || xe->info.platform == XE_CRESCENTISLAND) {
if (channel == CHANNEL_PKG)
return BMG_PACKAGE_TEMPERATURE;
else if (channel == CHANNEL_VRAM)
@@ -575,23 +575,36 @@ xe_hwmon_power_max_interval_show(struct device *dev, struct device_attribute *at
mutex_unlock(&hwmon->hwmon_lock);
- x = REG_FIELD_GET(PWR_LIM_TIME_X, reg_val);
- y = REG_FIELD_GET(PWR_LIM_TIME_Y, reg_val);
+ if (hwmon->xe->info.platform >= XE_CRESCENTISLAND) {
+ /**
+ * On CRI and newer platforms, the interval encoding changed.
+ * The value is now stored directly in milliseconds as U5.2,
+ * replacing the older 1.x * 2^y representation.
+ * Bits [6:2] hold the integer part and bits [1:0] the fraction,
+ * so convert to milliseconds by extracting integer/fractional parts.
+ * Round fraction value to 1 when it is >= 0.5.
+ */
+ reg_val = REG_FIELD_GET(PWR_LIM_TIME, reg_val);
+ out = (u64)((reg_val >> 2) + ((reg_val & 0x3) >= 2));
+ } else {
+ x = REG_FIELD_GET(PWR_LIM_TIME_X, reg_val);
+ y = REG_FIELD_GET(PWR_LIM_TIME_Y, reg_val);
- /*
- * tau = (1 + (x / 4)) * power(2,y), x = bits(23:22), y = bits(21:17)
- * = (4 | x) << (y - 2)
- *
- * Here (y - 2) ensures a 1.x fixed point representation of 1.x
- * As x is 2 bits so 1.x can be 1.0, 1.25, 1.50, 1.75
- *
- * As y can be < 2, we compute tau4 = (4 | x) << y
- * and then add 2 when doing the final right shift to account for units
- */
- tau4 = (u64)((1 << x_w) | x) << y;
+ /*
+ * tau = (1 + (x / 4)) * power(2,y), x = bits(23:22), y = bits(21:17)
+ * = (4 | x) << (y - 2)
+ *
+ * Here (y - 2) ensures a 1.x fixed point representation of 1.x
+ * As x is 2 bits so 1.x can be 1.0, 1.25, 1.50, 1.75
+ *
+ * As y can be < 2, we compute tau4 = (4 | x) << y
+ * and then add 2 when doing the final right shift to account for units
+ */
+ tau4 = (u64)((1 << x_w) | x) << y;
- /* val in hwmon interface units (millisec) */
- out = mul_u64_u32_shr(tau4, SF_TIME, hwmon->scl_shift_time + x_w);
+ /* val in hwmon interface units (millisec) */
+ out = mul_u64_u32_shr(tau4, SF_TIME, hwmon->scl_shift_time + x_w);
+ }
return sysfs_emit(buf, "%llu\n", out);
}
@@ -637,24 +650,35 @@ xe_hwmon_power_max_interval_store(struct device *dev, struct device_attribute *a
if (val > max_win)
return -EINVAL;
- /* val in hw units */
- val = DIV_ROUND_CLOSEST_ULL((u64)val << hwmon->scl_shift_time, SF_TIME) + 1;
-
- /*
- * Convert val to 1.x * power(2,y)
- * y = ilog2(val)
- * x = (val - (1 << y)) >> (y - 2)
- */
- if (!val) {
- y = 0;
- x = 0;
+ if (hwmon->xe->info.platform >= XE_CRESCENTISLAND) {
+ /**
+ * On CRI and newer platforms, the interval encoding changed.
+ * The value is now stored directly in milliseconds as U5.2,
+ * replacing the older 1.x * 2^y representation.
+ * Bits [6:2] hold the integer part and bits [1:0] the fraction,so
+ * convert from milliseconds by shifting the value left by 2 to fit into the field.
+ */
+ rxy = REG_FIELD_PREP(PWR_LIM_TIME, (val << 2));
} else {
- y = ilog2(val);
- x = (val - (1ul << y)) << x_w >> y;
- }
+ /* val in hw units */
+ val = DIV_ROUND_CLOSEST_ULL((u64)val << hwmon->scl_shift_time, SF_TIME) + 1;
+
+ /*
+ * Convert val to 1.x * power(2,y)
+ * y = ilog2(val)
+ * x = (val - (1 << y)) >> (y - 2)
+ */
+ if (!val) {
+ y = 0;
+ x = 0;
+ } else {
+ y = ilog2(val);
+ x = (val - (1ul << y)) << x_w >> y;
+ }
- rxy = REG_FIELD_PREP(PWR_LIM_TIME_X, x) |
- REG_FIELD_PREP(PWR_LIM_TIME_Y, y);
+ rxy = REG_FIELD_PREP(PWR_LIM_TIME_X, x) |
+ REG_FIELD_PREP(PWR_LIM_TIME_Y, y);
+ }
guard(xe_pm_runtime)(hwmon->xe);
diff --git a/drivers/gpu/drm/xe/xe_log.c b/drivers/gpu/drm/xe/xe_log.c
new file mode 100644
index 000000000000..5549ef6966fd
--- /dev/null
+++ b/drivers/gpu/drm/xe/xe_log.c
@@ -0,0 +1,235 @@
+// SPDX-License-Identifier: MIT
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#include <kunit/static_stub.h>
+#include <kunit/visibility.h>
+
+#include "abi/xe_log_abi.h"
+
+#include "xe_device.h"
+#include "xe_log.h"
+#include "xe_printk.h"
+
+static void log_emit_cper(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid,
+ u32 component, u32 location, const void *data, size_t len,
+ struct va_format *vaf)
+{
+ KUNIT_STATIC_STUB_REDIRECT(log_emit_cper, pdev, cper_sev, sigid,
+ component, location, data, len, vaf);
+ /* TODO */
+}
+
+static const char *log_unknown_component_prefix(u32 component)
+{
+ u32 class = FIELD_GET(XE_LOG_COMPONENT_CLASS_MASK, component);
+ u32 type = FIELD_GET(XE_LOG_COMPONENT_TYPE_MASK, component);
+
+ WARN(IS_ENABLED(CONFIG_DRM_XE_DEBUG), "LOG: unrecognized component %u.%u\n", class, type);
+ switch (class) {
+#define MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(_CLASS) \
+ case XE_LOG_COMPONENT_CLASS_##_CLASS: return #_CLASS "? ";
+ MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(SYSTEM)
+ MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(DRIVER)
+ MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(FEATURE)
+ MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(FIRMWARE)
+ MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(HARDWARE)
+#undef MAKE_XE_LOG_COMPONENT_CLASS_PREFIX
+ }
+ return "COMP? ";
+}
+
+static const char *log_component_prefix(u32 component)
+{
+ switch (component) {
+#define MAKE_XE_LOG_COMPONENT_CASE_PREFIX(_CLASS, _ID, _TAG, _SIG, _NAME) \
+ case XE_LOG_COMPONENT_##_TAG: return #_TAG ": ";
+ DEFINE_XE_LOG_COMPONENTS(MAKE_XE_LOG_COMPONENT_CASE_PREFIX)
+#undef MAKE_XE_LOG_COMPONENT_CASE_PREFIX
+ }
+
+ return component ? log_unknown_component_prefix(component) : "";
+}
+
+static struct xe_gt *get_gt_safe(struct pci_dev *pdev, u8 id)
+{
+ struct xe_device *xe = pdev_to_xe_device(pdev);
+
+ return xe ? xe_device_get_gt(xe, id) : NULL;
+}
+
+static struct xe_tile *get_tile_safe(struct pci_dev *pdev, u8 id)
+{
+ struct xe_device *xe = pdev_to_xe_device(pdev);
+
+ return xe && id < xe->info.tile_count ? &xe->tiles[id] : NULL;
+}
+
+static const char *log_location_prefix(struct pci_dev *pdev, u32 location, char *buf, size_t size)
+{
+ u32 type = FIELD_GET(XE_LOG_LOCATION_TYPE_MASK, location);
+ u32 id = FIELD_GET(XE_LOG_LOCATION_ID_MASK, location);
+
+ if (!location || type == XE_LOG_LOCATION_TYPE_DEVICE) {
+ if (id)
+ goto unrecognized;
+ strscpy(buf, "", size);
+ } else if (type == XE_LOG_LOCATION_TYPE_TILE) {
+ struct xe_tile *tile = get_tile_safe(pdev, id);
+
+ if (!tile)
+ goto unrecognized;
+ snprintf(buf, size, "Tile%u: ", id);
+ } else if (type == XE_LOG_LOCATION_TYPE_GT) {
+ struct xe_gt *gt = get_gt_safe(pdev, id);
+
+ if (!gt)
+ goto unrecognized;
+ snprintf(buf, size, "Tile%u: GT%u: ", gt->tile->id, id);
+ } else {
+ goto unrecognized;
+ }
+
+ return buf;
+
+unrecognized:
+ pci_WARN(pdev, IS_ENABLED(CONFIG_DRM_XE_DEBUG),
+ "LOG: unrecognized location %u.%u\n", type, id);
+ snprintf(buf, size, "LOC%u.%u? ", type, id);
+ return buf;
+}
+
+static bool is_hw_sigid(enum xe_sigid sigid)
+{
+ return (int)sigid >= INTEL_SIGID_GPU_XE_HARDWARE_START;
+}
+
+static bool is_sev_error(int cper_sev)
+{
+ return cper_sev != CPER_SEV_INFORMATIONAL;
+}
+
+static const char *log_hwe_prefix(int cper_sev, enum xe_sigid sigid)
+{
+ return is_sev_error(cper_sev) && is_hw_sigid(sigid) ? HW_ERR : "";
+}
+
+static const char *log_sev_prefix(int cper_sev)
+{
+ switch (cper_sev) {
+ case CPER_SEV_FATAL:
+ return "FATAL ";
+ case CPER_SEV_RECOVERABLE:
+ return "";
+ case CPER_SEV_CORRECTED:
+ return "CORRECTED ";
+ case CPER_SEV_INFORMATIONAL:
+ return "";
+ default:
+ WARN(IS_ENABLED(CONFIG_DRM_XE_DEBUG), "LOG: unknown severity %d\n", cper_sev);
+ return "";
+ }
+}
+
+#define __LOG_DRM_PRINTK_FMT(fmt, args...) "[drm] " fmt, ##args
+#define __LOG_DRM_PRINTK_ERR_FMT(fmt, args...) __LOG_DRM_PRINTK_FMT("*ERROR* " fmt, args)
+
+static void log_dmesg_vprintk(struct pci_dev *pdev, int cper_sev, struct va_format *vaf)
+{
+ KUNIT_STATIC_STUB_REDIRECT(log_dmesg_vprintk, pdev, cper_sev, vaf);
+
+ if (cper_sev == CPER_SEV_INFORMATIONAL)
+ pci_info(pdev, __LOG_DRM_PRINTK_FMT("%pV", vaf));
+ else
+ pci_err(pdev, __LOG_DRM_PRINTK_ERR_FMT("%pV", vaf));
+}
+
+static void log_dmesg_printf(struct pci_dev *pdev, int cper_sev, const char *fmt, ...)
+{
+ struct va_format vaf;
+ va_list args;
+
+ va_start(args, fmt);
+ vaf.fmt = fmt;
+ vaf.va = &args;
+
+ log_dmesg_vprintk(pdev, cper_sev, &vaf);
+
+ va_end(args);
+}
+
+static void log_emit_dmesg(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid,
+ u32 component, u32 location, const void *data, size_t len,
+ struct va_format *vaf)
+{
+ char buf[32];
+ const char *loc_prefix = log_location_prefix(pdev, location, buf, sizeof(buf));
+ const char *comp_prefix = log_component_prefix(component);
+ const char *hwe_prefix = log_hwe_prefix(cper_sev, sigid);
+ const char *sev_prefix = log_sev_prefix(cper_sev);
+
+ if (IS_ERR(data))
+ log_dmesg_printf(pdev, cper_sev, "SIGID=%u %s(%pe) %s%s%s%pV",
+ sigid, sev_prefix, data, hwe_prefix,
+ loc_prefix, comp_prefix, vaf);
+ else if (data && len)
+ log_dmesg_printf(pdev, cper_sev, "SIGID=%u %s(%*phN) %s%s%s%pV",
+ sigid, sev_prefix, (int)len, data, hwe_prefix,
+ loc_prefix, comp_prefix, vaf);
+ else
+ log_dmesg_printf(pdev, cper_sev, "SIGID=%u %s%s%s%s%pV",
+ sigid, sev_prefix, hwe_prefix,
+ loc_prefix, comp_prefix, vaf);
+}
+
+/**
+ * __xe_log_emit() - Emit a structured SIGID log entry
+ * @pdev: the &pci_dev device
+ * @cper_sev: CPER severity (CPER_SEV_FATAL, CPER_SEV_RECOVERABLE, ...)
+ * @sigid: signature identifier, see &enum xe_sigid
+ * @component: component identifier
+ * @location: location details of the @component
+ * @data: pointer to the additional details, or ERR_PTR, or NULL if not applicable
+ * @len: length of the @data in bytes, or 0 if not applicable
+ * @fmt: printf-style format string
+ * @...: format arguments
+ *
+ * Emits a dmesg line that includes a single stable, machine-matchable token
+ * ``SIGID=<n>`` followed by the optional severity token (like ``FATAL``) and,
+ * when @data pointer is set, either the error printed with %pe or a packed hex
+ * dump of the @data binary blob. The dmesg line will also include printf-style
+ * text message.
+ *
+ * Note that the full dmesg line, with the free text message, is only a debugging
+ * aid, not an interface! Only the ``SIGID=<n>`` token is stable there.
+ * The durable machine record is the CPER carrying the same SIGID.
+ *
+ * Note: generation of the CPER record is a planned follow-up.
+ *
+ * Examples::
+ *
+ * <3> xe 0000:03:00.0: [drm] *ERROR* SIGID=104 FATAL (-EPROTO) Invalid GuC reply
+ * <3> xe 0000:03:00.0: [drm] *ERROR* SIGID=106 (-ETIMEDOUT) Engine 'rcs0' hung
+ * <6> xe 0000:03:00.0: [drm] SIGID=103 In survivability mode
+ */
+void __xe_log_emit(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid,
+ u32 component, u32 location, const void *data, size_t len,
+ const char *fmt, ...)
+{
+ struct va_format vaf;
+ va_list args;
+
+ va_start(args, fmt);
+ vaf.fmt = fmt;
+ vaf.va = &args;
+
+ log_emit_dmesg(pdev, cper_sev, sigid, component, location, data, len, &vaf);
+ log_emit_cper(pdev, cper_sev, sigid, component, location, data, len, &vaf);
+
+ va_end(args);
+}
+
+#if IS_BUILTIN(CONFIG_DRM_XE_KUNIT_TEST)
+#include "tests/xe_log_kunit.c"
+#endif
diff --git a/drivers/gpu/drm/xe/xe_log.h b/drivers/gpu/drm/xe/xe_log.h
new file mode 100644
index 000000000000..bb1185143589
--- /dev/null
+++ b/drivers/gpu/drm/xe/xe_log.h
@@ -0,0 +1,194 @@
+/* SPDX-License-Identifier: MIT */
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#ifndef _XE_LOG_H_
+#define _XE_LOG_H_
+
+#include <linux/cper.h>
+#include <linux/err.h>
+
+#include "abi/xe_log_abi.h"
+#include "abi/xe_sigid_abi.h"
+#include "xe_any.h"
+
+struct pci_dev;
+
+__printf(8, 9)
+void __xe_log_emit(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid,
+ u32 component, u32 location, const void *data, size_t len,
+ const char *fmt, ...);
+
+#define __xe_log_const_sev_to_level(sev) \
+ (__builtin_constant_p(sev) ? (sev) == CPER_SEV_INFORMATIONAL ? KERN_INFO : KERN_ERR : NULL)
+
+#define __xe_log_emit_printk_index(level, fmt) \
+ dev_printk_index_emit(level, "%s SIGID=%u %s" fmt);
+
+#define xe_log_emit(pdev, sev, sig, comp, loc, data, len, fmt, args...) do { \
+ __xe_log_emit_printk_index(__xe_log_const_sev_to_level(sev), fmt); \
+ __xe_log_emit((pdev), (sev), (sig), (comp), (loc), (data), (len), fmt, ##args); \
+} while (0)
+
+#define xe_log_emit_fatal(pdev, sig, comp, loc, data, len, fmt, args...) \
+ xe_log_emit((pdev), CPER_SEV_FATAL, (sig), (comp), (loc), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_emit_recoverable(pdev, sig, comp, loc, data, len, fmt, args...) \
+ xe_log_emit((pdev), CPER_SEV_RECOVERABLE, (sig), (comp), (loc), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_emit_corrected(pdev, sig, comp, loc, data, len, fmt, args...) \
+ xe_log_emit((pdev), CPER_SEV_CORRECTED, (sig), (comp), (loc), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_emit_info(pdev, sig, comp, loc, data, len, fmt, args...) \
+ xe_log_emit((pdev), CPER_SEV_INFORMATIONAL, (sig), (comp), (loc), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_location_type(any) \
+ _Generic((any), \
+ struct xe_gt * : XE_LOG_LOCATION_TYPE_GT, \
+ const struct xe_gt * : XE_LOG_LOCATION_TYPE_GT, \
+ struct xe_tile * : XE_LOG_LOCATION_TYPE_TILE, \
+ const struct xe_tile * : XE_LOG_LOCATION_TYPE_TILE, \
+ struct xe_device * : XE_LOG_LOCATION_TYPE_DEVICE, \
+ const struct xe_device * : XE_LOG_LOCATION_TYPE_DEVICE, \
+ struct pci_dev * : XE_LOG_LOCATION_TYPE_DEVICE, \
+ struct device * : XE_LOG_LOCATION_TYPE_DEVICE)
+
+#define xe_log_location(any) \
+ PREP_XE_LOG_LOCATION(xe_log_location_type(any), xe_any_id(any))
+
+/**
+ * xe_log_from() - Emit a structured SIGID log entry using @any pointer as location.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @cper_sev: CPER severity (CPER_SEV_FATAL, CPER_SEV_RECOVERABLE, ...)
+ * @sigid: signature identifier, see &enum xe_sigid
+ * @component: component identifer
+ * @data: pointer to the additional details, or ERR_PTR, or NULL if not applicable
+ * @len: length of the @data in bytes, or 0 if not applicable
+ * @fmt: printf-style format string
+ * @args: arguments for the @fmt format string
+ *
+ * The location used to emit SIGID entry will be based on the @any pointer type.
+ * See xe_log_emit() for more details.
+ */
+#define xe_log_from(any, cper_sev, sigid, component, data, len, fmt, args...) do { \
+ typeof(any) ___any = (any); \
+ xe_log_emit(xe_any_to_pdev(___any), (cper_sev), (sigid), (component), \
+ xe_log_location(___any), (data), (len), fmt, ##args); \
+} while (0)
+
+#define xe_log_from_fatal(any, sig, comp, data, len, fmt, args...) \
+ xe_log_from((any), CPER_SEV_FATAL, (sig), (comp), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_from_recoverable(any, sig, comp, data, len, fmt, args...) \
+ xe_log_from((any), CPER_SEV_RECOVERABLE, (sig), (comp), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_from_corrected(any, sig, comp, data, len, fmt, args...) \
+ xe_log_from((any), CPER_SEV_CORRECTED, (sig), (comp), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_from_info(any, sig, comp, data, len, fmt, args...) \
+ xe_log_from((any), CPER_SEV_INFORMATIONAL, (sig), (comp), \
+ (data), (len), fmt, ##args)
+
+/**
+ * xe_log_comp() - Emit a structured SIGID log entry on the component behalf.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @cper_sev: CPER severity (CPER_SEV_FATAL, CPER_SEV_RECOVERABLE, ...)
+ * @TAG: the component tag to use
+ * @data: pointer to the additional details, or ERR_PTR, or NULL if not applicable
+ * @len: length of the @data in bytes, or 0 if not applicable
+ * @fmt: printf-style free text format string (not a stable interface)
+ * @args: arguments for the @fmt format string
+ *
+ * The SIGID will be determined from the component's @TAG.
+ * The component identifier will be determined from the component's @TAG.
+ * The location used to emit SIGID entry will be based on the @any pointer type.
+ */
+#define xe_log_comp(any, cper_sev, TAG, data, len, fmt, args...) \
+ xe_log_from((any), (cper_sev), (int)XE_LOG_COMPONENT_##TAG##_SIGID, \
+ XE_LOG_COMPONENT_##TAG, (data), (len), fmt, ##args)
+
+#define xe_log_comp_fatal(any, TAG, data, len, fmt, args...) \
+ xe_log_comp((any), CPER_SEV_FATAL, TAG, (data), (len), fmt, ##args)
+
+#define xe_log_comp_recoverable(any, TAG, data, len, fmt, args...) \
+ xe_log_comp((any), CPER_SEV_RECOVERABLE, TAG, (data), (len), fmt, ##args)
+
+#define xe_log_comp_corrected(any, TAG, data, len, fmt, args...) \
+ xe_log_comp((any), CPER_SEV_CORRECTED, TAG, (data), (len), fmt, ##args)
+
+#define xe_log_comp_info(any, TAG, data, len, fmt, args...) \
+ xe_log_comp((any), CPER_SEV_INFORMATIONAL, TAG, (data), (len), fmt, ##args)
+
+/**
+ * xe_log_err() - Emit a structured SIGID error log entry on the component behalf.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @TAG: the component tag to use
+ * @err: negative errno for the failing operation, or 0 if not applicable
+ * @fmt: printf-style free text format string (not a stable interface)
+ * @args: arguments for the @fmt format string
+ *
+ * The log entry will be emitted with @CPER_SEV_RECOVERABLE severity.
+ */
+#define xe_log_err(any, TAG, err, fmt, args...) \
+ xe_log_comp_recoverable((any), TAG, ERR_PTR(err), 0, fmt, ##args)
+
+/**
+ * xe_log_err_fatal() - Emit a structured SIGID error log entry on the component behalf.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @TAG: the component tag to use
+ * @err: negative errno for the failing operation, or 0 if not applicable
+ * @fmt: printf-style free text format string (not a stable interface)
+ * @args: arguments for the @fmt format string
+ *
+ * The log entry will be emitted with @CPER_SEV_FATAL severity.
+ */
+#define xe_log_err_fatal(any, TAG, err, fmt, args...) \
+ xe_log_comp_fatal((any), TAG, ERR_PTR(err), 0, fmt, ##args)
+
+/**
+ * xe_log_err_corrected() - Emit a structured SIGID error log entry on the component behalf.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @TAG: the component tag to use
+ * @err: negative errno for the failing operation, or 0 if not applicable
+ * @fmt: printf-style free text format string (not a stable interface)
+ * @args: arguments for the @fmt format string
+ *
+ * The log entry will be emitted with @CPER_SEV_CORRECTED severity.
+ */
+#define xe_log_err_corrected(any, TAG, err, fmt, args...) \
+ xe_log_comp_corrected((any), TAG, ERR_PTR(err), 0, fmt, ##args)
+
+/**
+ * xe_log_err_info() - Emit a structured SIGID error log entry on the component behalf.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @TAG: the component tag to use
+ * @err: negative errno for the failing operation, or 0 if not applicable
+ * @fmt: printf-style free text format string (not a stable interface)
+ * @args: arguments for the @fmt format string
+ *
+ * The log entry will be emitted with @CPER_SEV_INFORMATIONAL severity.
+ */
+#define xe_log_err_info(any, TAG, err, fmt, args...) \
+ xe_log_comp_info((any), TAG, ERR_PTR(err), 0, fmt, ##args)
+
+/**
+ * xe_log_info() - Emit a structured SIGID information log entry on the component behalf.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @TAG: the component tag to use
+ * @fmt: printf-style free text format string (not a stable interface)
+ * @args: arguments for the @fmt format string
+ *
+ * The log entry will be emitted with @CPER_SEV_INFORMATIONAL severity.
+ */
+#define xe_log_info(any, TAG, fmt, args...) \
+ xe_log_err_info((any), TAG, 0, fmt, ##args)
+
+#endif
diff --git a/drivers/gpu/drm/xe/xe_mert.c b/drivers/gpu/drm/xe/xe_mert.c
index f637df95418b..32700c19a1df 100644
--- a/drivers/gpu/drm/xe/xe_mert.c
+++ b/drivers/gpu/drm/xe/xe_mert.c
@@ -77,7 +77,7 @@ static void mert_handle_cat_error(struct xe_device *xe)
xe_device_declare_wedged(xe);
break;
case CATERR_LMTT_FAULT:
- xe_sriov_dbg(xe, "MERT: CAT_ERR: VF%u LMTT fault!\n", vfid);
+ xe_sriov_dbg_ratelimited(xe, "MERT: CAT_ERR: VF%u LMTT fault!\n", vfid);
/* XXX: track/report malicious VF activity */
break;
default:
diff --git a/drivers/gpu/drm/xe/xe_migrate.c b/drivers/gpu/drm/xe/xe_migrate.c
index f79d0047bec6..75b83687f1b5 100644
--- a/drivers/gpu/drm/xe/xe_migrate.c
+++ b/drivers/gpu/drm/xe/xe_migrate.c
@@ -493,7 +493,6 @@ int xe_migrate_init(struct xe_migrate *m)
*/
m->q = xe_exec_queue_create(xe, vm, logical_mask, 1, hwe0,
EXEC_QUEUE_FLAG_KERNEL |
- EXEC_QUEUE_FLAG_PERMANENT |
EXEC_QUEUE_FLAG_HIGH_PRIORITY |
EXEC_QUEUE_FLAG_MIGRATE |
EXEC_QUEUE_FLAG_LOW_LATENCY, 0);
@@ -501,7 +500,6 @@ int xe_migrate_init(struct xe_migrate *m)
m->q = xe_exec_queue_create_class(xe, primary_gt, vm,
XE_ENGINE_CLASS_COPY,
EXEC_QUEUE_FLAG_KERNEL |
- EXEC_QUEUE_FLAG_PERMANENT |
EXEC_QUEUE_FLAG_MIGRATE, 0);
}
if (IS_ERR(m->q)) {
diff --git a/drivers/gpu/drm/xe/xe_module.c b/drivers/gpu/drm/xe/xe_module.c
index 848d65265443..4bc28dfc1992 100644
--- a/drivers/gpu/drm/xe/xe_module.c
+++ b/drivers/gpu/drm/xe/xe_module.c
@@ -29,6 +29,7 @@ struct xe_modparam xe_modparam = {
.max_vfs = XE_DEFAULT_MAX_VFS,
#endif
.wedged_mode = XE_DEFAULT_WEDGED_MODE,
+ .num_pf_work = XE_DEFAULT_NUM_PF_WORK,
.svm_notifier_size = XE_DEFAULT_SVM_NOTIFIER_SIZE,
/* the rest are 0 by default */
};
@@ -81,6 +82,9 @@ MODULE_PARM_DESC(wedged_mode,
"Module's default policy for the wedged mode (0=never, 1=upon-critical-error, 2=upon-any-hang-no-reset "
"[default=" XE_DEFAULT_WEDGED_MODE_STR "])");
+module_param_named(num_pf_work, xe_modparam.num_pf_work, uint, 0600);
+MODULE_PARM_DESC(num_pf_work, "Number of page fault work threads, default=2, min=1, max=8");
+
static int xe_check_nomodeset(void)
{
if (drm_firmware_drivers_only())
diff --git a/drivers/gpu/drm/xe/xe_module.h b/drivers/gpu/drm/xe/xe_module.h
index a0eb7db07770..6272d9e41207 100644
--- a/drivers/gpu/drm/xe/xe_module.h
+++ b/drivers/gpu/drm/xe/xe_module.h
@@ -23,6 +23,7 @@ struct xe_modparam {
unsigned int max_vfs;
#endif
unsigned int wedged_mode;
+ unsigned int num_pf_work;
u32 svm_notifier_size;
};
diff --git a/drivers/gpu/drm/xe/xe_oa.c b/drivers/gpu/drm/xe/xe_oa.c
index 9c5384b95c63..b460fcdfca15 100644
--- a/drivers/gpu/drm/xe/xe_oa.c
+++ b/drivers/gpu/drm/xe/xe_oa.c
@@ -2581,10 +2581,11 @@ static u32 __hwe_oam_unit(struct xe_hw_engine *hwe)
return XE_OA_UNIT_INVALID;
else if (!IS_DGFX(gt_to_xe(hwe->gt)))
return XE_OAM_UNIT_SCMI_0;
- else if (hwe->class == XE_ENGINE_CLASS_VIDEO_DECODE)
- return (hwe->instance / 2 & 0x1) + 1;
- else if (hwe->class == XE_ENGINE_CLASS_VIDEO_ENHANCE)
+ else if (hwe->class == XE_ENGINE_CLASS_VIDEO_ENHANCE &&
+ MEDIA_VERx100(gt_to_xe(hwe->gt)) < 3500)
return (hwe->instance & 0x1) + 1;
+ else
+ return (hwe->instance / 2 & 0x1) + 1;
return XE_OA_UNIT_INVALID;
}
diff --git a/drivers/gpu/drm/xe/xe_pagefault.c b/drivers/gpu/drm/xe/xe_pagefault.c
index dd3c068e1a39..9e89d7b37395 100644
--- a/drivers/gpu/drm/xe/xe_pagefault.c
+++ b/drivers/gpu/drm/xe/xe_pagefault.c
@@ -14,6 +14,7 @@
#include "xe_gt_types.h"
#include "xe_gt_stats.h"
#include "xe_hw_engine.h"
+#include "xe_log.h"
#include "xe_pagefault.h"
#include "xe_pagefault_types.h"
#include "xe_svm.h"
@@ -35,6 +36,73 @@
* xe_pagefault.c implements the consumer layer.
*/
+/**
+ * DOC: Xe page fault cache
+ *
+ * Some Xe hardware can trigger “fault storms,” which are many page faults to
+ * the same address within a short period of time. An example is many EU threads
+ * faulting on the same page simultaneously. With the current page fault locking
+ * structure, only one page fault for a given address range can be processed at
+ * a time. This causes head-of-queue blocking across workers, killing
+ * parallelism. If the page fault handler must repeatedly look up resources
+ * (VMAs, ranges) to determine that the pages are valid for each fault in the
+ * storm, the time complexity grows rapidly.
+ *
+ * To address this, each page fault worker maintains a cache of the active fault
+ * being processed. Subsequent faults that hit in the cache are chained to the
+ * pending fault, and all chained faults are acknowledged once the initial fault
+ * completes. This alleviates head-of-queue blocking and quickly chains faults
+ * in the upper layers, avoiding expensive lookups in the main fault-handling
+ * path.
+ *
+ * Faults are buffered in the page fault queue in a way that provides stable
+ * storage for outstanding faults. In particular, faults may be chained directly
+ * while still resident in the queue storage (i.e., outside the worker’s current
+ * head/tail dequeue position). This allows the IRQ handler to match newly
+ * arrived faults against the per-worker cache and immediately chain cache hits
+ * onto the active fault under the queue lock, without allocating memory or
+ * waiting for the worker to pop the fault first.
+ *
+ * A per-fault state field is used to assert correctness of these invariants.
+ * The state tracks whether an entry is free, queued, chained, or currently
+ * active. Transitions are performed under the page fault queue lock, and the
+ * worker acknowledges faults by walking the chain and returning entries to the
+ * free state once they are complete.
+ */
+
+/**
+ * enum xe_pagefault_alloc_state - lifetime state for a page fault queue entry
+ * @XE_PAGEFAULT_ALLOC_STATE_FREE:
+ * Entry is unused and may be overwritten by the producer, consumer retry
+ * or requeue..
+ * @XE_PAGEFAULT_ALLOC_STATE_QUEUED:
+ * Entry has been enqueued and may be dequeued by a worker.
+ * @XE_PAGEFAULT_ALLOC_STATE_ACTIVE:
+ * Entry has been dequeued and is the worker's currently serviced fault.
+ * The worker may attach additional faults to it via consumer.next.
+ * @XE_PAGEFAULT_ALLOC_STATE_CHAINED:
+ * Entry is not independently serviced; it has been chained onto an
+ * ACTIVE entry via consumer.next and will be acknowledged when the
+ * leading fault completes.
+ * @XE_PAGEFAULT_ALLOC_STATE_COUNT:
+ * Count of allocation states.
+ *
+ * The page fault queue provides stable storage for outstanding faults so the
+ * IRQ handler can chain new cache hits directly onto a worker's active fault.
+ * Because entries may remain referenced outside the consumer dequeue window,
+ * the producer must only write into entries in the FREE state.
+ *
+ * State transitions are protected by the page fault queue lock. Workers return
+ * entries to FREE after acknowledging the fault (either as ACTIVE or CHAINED).
+ */
+enum xe_pagefault_alloc_state {
+ XE_PAGEFAULT_ALLOC_STATE_FREE = 0,
+ XE_PAGEFAULT_ALLOC_STATE_QUEUED = 1,
+ XE_PAGEFAULT_ALLOC_STATE_CHAINED = 2,
+ XE_PAGEFAULT_ALLOC_STATE_ACTIVE = 3,
+ XE_PAGEFAULT_ALLOC_STATE_COUNT = 4,
+};
+
static int xe_pagefault_entry_size(void)
{
/*
@@ -77,16 +145,16 @@ static int xe_pagefault_begin(struct drm_exec *exec, struct xe_vma *vma,
}
static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma,
- bool atomic)
+ struct xe_pagefault *pf, bool atomic)
{
struct xe_vm *vm = xe_vma_vm(vma);
struct xe_tile *tile = gt_to_tile(gt);
struct xe_validation_ctx ctx;
struct drm_exec exec;
struct dma_fence *fence;
- int err, needs_vram;
+ int err = 0, needs_vram;
- lockdep_assert_held_write(&vm->lock);
+ lockdep_assert_held(&vm->lock);
needs_vram = xe_vma_need_vram_for_atomic(vm->xe, vma, atomic);
if (needs_vram < 0 || (needs_vram && xe_vma_is_userptr(vma)))
@@ -98,50 +166,59 @@ static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma,
trace_xe_vma_pagefault(vma);
+ guard(mutex)(&vma->fault_lock);
+
/* Check if VMA is valid, opportunistic check only */
if (xe_vm_has_valid_gpu_mapping(tile, vma->tile_present,
- vma->tile_invalidated) && !atomic)
+ vma->tile_invalidated) && !atomic) {
+ xe_pagefault_set_start_addr(pf, xe_vma_start(vma));
+ xe_pagefault_set_end_addr(pf, xe_vma_end(vma));
return 0;
+ }
-retry_userptr:
- if (xe_vma_is_userptr(vma) &&
- xe_vma_userptr_check_repin(to_userptr_vma(vma))) {
- struct xe_userptr_vma *uvma = to_userptr_vma(vma);
+ do {
+ if (xe_vma_is_userptr(vma) &&
+ xe_vma_userptr_check_repin(to_userptr_vma(vma))) {
+ struct xe_userptr_vma *uvma = to_userptr_vma(vma);
- err = xe_vma_userptr_pin_pages(uvma);
- if (err)
- return err;
- }
+ err = xe_vma_userptr_pin_pages(uvma);
+ if (err)
+ return err;
+ }
- /* Lock VM and BOs dma-resv */
- xe_validation_ctx_init(&ctx, &vm->xe->val, &exec, (struct xe_val_flags) {});
- drm_exec_until_all_locked(&exec) {
- err = xe_pagefault_begin(&exec, vma, tile->mem.vram,
- needs_vram == 1);
- drm_exec_retry_on_contention(&exec);
- xe_validation_retry_on_oom(&ctx, &err);
- if (err)
- goto unlock_dma_resv;
-
- /* Bind VMA only to the GT that has faulted */
- trace_xe_vma_pf_bind(vma);
- xe_vm_set_validation_exec(vm, &exec);
- fence = xe_vma_rebind(vm, vma, BIT(tile->id));
- xe_vm_set_validation_exec(vm, NULL);
- if (IS_ERR(fence)) {
- err = PTR_ERR(fence);
+ /* Lock VM and BOs dma-resv */
+ xe_validation_ctx_init(&ctx, &vm->xe->val, &exec,
+ (struct xe_val_flags) {});
+ drm_exec_until_all_locked(&exec) {
+ err = xe_pagefault_begin(&exec, vma, tile->mem.vram,
+ needs_vram == 1);
+ drm_exec_retry_on_contention(&exec);
xe_validation_retry_on_oom(&ctx, &err);
- goto unlock_dma_resv;
+ if (err)
+ break;
+
+ /* Bind VMA only to the GT that has faulted */
+ trace_xe_vma_pf_bind(vma);
+ xe_vm_set_validation_exec(vm, &exec);
+ fence = xe_vma_rebind(vm, vma, BIT(tile->id));
+ xe_vm_set_validation_exec(vm, NULL);
+ if (IS_ERR(fence)) {
+ err = PTR_ERR(fence);
+ xe_validation_retry_on_oom(&ctx, &err);
+ break;
+ }
}
- }
+ xe_validation_ctx_fini(&ctx);
+ } while (err == -EAGAIN);
- dma_fence_wait(fence, false);
- dma_fence_put(fence);
+ if (!err) {
+ /* Give hint to immediately ack faults */
+ xe_pagefault_set_start_addr(pf, xe_vma_start(vma));
+ xe_pagefault_set_end_addr(pf, xe_vma_end(vma));
-unlock_dma_resv:
- xe_validation_ctx_fini(&ctx);
- if (err == -EAGAIN)
- goto retry_userptr;
+ dma_fence_wait(fence, false);
+ dma_fence_put(fence);
+ }
return err;
}
@@ -184,10 +261,7 @@ static int xe_pagefault_service(struct xe_pagefault *pf)
if (IS_ERR(vm))
return PTR_ERR(vm);
- /*
- * TODO: Change to read lock? Using write lock for simplicity.
- */
- down_write(&vm->lock);
+ down_read(&vm->lock);
if (xe_vm_is_closed(vm)) {
err = -ENOENT;
@@ -209,46 +283,275 @@ static int xe_pagefault_service(struct xe_pagefault *pf)
atomic = xe_pagefault_access_is_atomic(pf->consumer.access_type);
if (xe_vma_is_cpu_addr_mirror(vma))
- err = xe_svm_handle_pagefault(vm, vma, gt,
+ err = xe_svm_handle_pagefault(vm, vma, pf, gt,
pf->consumer.page_addr, atomic);
else
- err = xe_pagefault_handle_vma(gt, vma, atomic);
+ err = xe_pagefault_handle_vma(gt, vma, pf, atomic);
unlock_vm:
- if (!err)
- vm->usm.last_fault_vma = vma;
- up_write(&vm->lock);
+ up_read(&vm->lock);
xe_vm_put(vm);
return err;
}
-static bool xe_pagefault_queue_pop(struct xe_pagefault_queue *pf_queue,
- struct xe_pagefault *pf)
+#define XE_PAGEFAULT_CACHE_START_INVALID U64_MAX
+#define xe_pagefault_cache_start_invalidate(val) \
+ (val = XE_PAGEFAULT_CACHE_START_INVALID)
+
+static void
+xe_pagefault_cache_invalidate(struct xe_pagefault_queue *pf_queue,
+ struct xe_pagefault_work *pf_work)
+{
+ lockdep_assert_held(&pf_queue->lock);
+
+ xe_pagefault_cache_start_invalidate(pf_work->cache.start);
+}
+
+static bool xe_pagefault_queue_full(struct xe_pagefault_queue *pf_queue)
+{
+ lockdep_assert_held(&pf_queue->lock);
+
+ return CIRC_SPACE(pf_queue->head, pf_queue->tail,
+ pf_queue->size) <= xe_pagefault_entry_size();
+}
+
+static struct xe_pagefault *
+xe_pagefault_queue_add(struct xe_pagefault_queue *pf_queue,
+ struct xe_pagefault *pf)
{
- bool found_fault = false;
+ struct xe_device *xe = container_of(pf_queue, typeof(*xe),
+ usm.pf_queue);
+ struct xe_pagefault *lpf;
+
+ lockdep_assert_held(&pf_queue->lock);
- spin_lock_irq(&pf_queue->lock);
- if (pf_queue->tail != pf_queue->head) {
- memcpy(pf, pf_queue->data + pf_queue->tail, sizeof(*pf));
- pf_queue->tail = (pf_queue->tail + xe_pagefault_entry_size()) %
+ do {
+ /* Not possible, warn on and drop page fault */
+ if (WARN_ON_ONCE(xe_pagefault_queue_full(pf_queue))) {
+ xe_log_err(xe, PAGEFAULT, -ENOSPC, "Queue full!\n");
+ return NULL;
+ }
+
+ lpf = (pf_queue->data + pf_queue->head);
+ pf_queue->head = (pf_queue->head + xe_pagefault_entry_size()) %
pf_queue->size;
- found_fault = true;
+ } while (lpf->consumer.alloc_state != XE_PAGEFAULT_ALLOC_STATE_FREE);
+
+ xe_assert(xe, lpf != pf);
+ memcpy(lpf, pf, sizeof(*pf));
+ lpf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_QUEUED;
+
+ return lpf;
+}
+
+static struct xe_pagefault *
+xe_pagefault_queue_unchain_requeue(struct xe_pagefault_queue *pf_queue,
+ struct xe_pagefault *pf, struct xe_gt *gt)
+{
+ struct xe_device *xe = container_of(pf_queue, typeof(*xe),
+ usm.pf_queue);
+ struct xe_pagefault *next = pf->consumer.next, *lpf;
+
+ lockdep_assert_held(&pf_queue->lock);
+ xe_assert(xe, pf->consumer.alloc_state ==
+ XE_PAGEFAULT_ALLOC_STATE_CHAINED);
+
+ xe_gt_stats_incr(gt, XE_GT_STATS_ID_CHAIN_MISMATCH_PAGEFAULT_COUNT, 1);
+
+ pf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_FREE;
+ lpf = xe_pagefault_queue_add(pf_queue, pf);
+ if (lpf) {
+ lpf->consumer.next = NULL;
+ lpf->consumer.fault_type_level |= XE_PAGEFAULT_REQUEUE_MASK;
+ }
+
+ return next;
+}
+
+static bool xe_pagefault_match(struct xe_pagefault *pf, u64 start,
+ u64 end, u64 cache_asid)
+{
+ struct xe_device *xe = gt_to_xe(pf->gt);
+ u64 page_addr = pf->consumer.page_addr;
+ u32 pf_asid = pf->consumer.asid;
+
+ xe_assert(xe, pf->consumer.alloc_state !=
+ XE_PAGEFAULT_ALLOC_STATE_FREE);
+
+ return page_addr >= start && page_addr < end &&
+ pf_asid == cache_asid;
+}
+
+static bool xe_pagefault_try_chain(struct xe_pagefault_queue *pf_queue,
+ struct xe_pagefault *pf)
+{
+ struct xe_device *xe = container_of(pf_queue, typeof(*xe),
+ usm.pf_queue);
+ struct xe_pagefault_work *pf_work;
+ bool requeue = FIELD_GET(XE_PAGEFAULT_REQUEUE_MASK,
+ pf->consumer.fault_type_level);
+ int i;
+
+ lockdep_assert_held(&pf_queue->lock);
+ xe_assert(xe, pf->consumer.alloc_state ==
+ XE_PAGEFAULT_ALLOC_STATE_QUEUED);
+
+ /*
+ * If this is a retry, we may already have a chain attached. In that
+ * case, we cannot hit in the cache because chains cannot easily be
+ * combined.
+ */
+ if (pf->consumer.next)
+ return false;
+
+ for (i = 0, pf_work = xe->usm.pf_workers;
+ i < xe->info.num_pf_work; ++i, ++pf_work) {
+ u64 start = pf_work->cache.start;
+ u64 end = requeue ? start + SZ_4K : pf_work->cache.end;
+ u32 asid = pf_work->cache.asid;
+
+ if (xe_pagefault_match(pf, start, end, asid)) {
+ xe_assert(xe, pf_work->cache.pf->consumer.alloc_state ==
+ XE_PAGEFAULT_ALLOC_STATE_ACTIVE);
+
+ if (pf->producer.private !=
+ pf_work->cache.pf->producer.private)
+ continue;
+
+ xe_gt_stats_incr(pf->gt,
+ XE_GT_STATS_ID_CHAIN_PAGEFAULT_COUNT,
+ 1);
+
+ pf->consumer.alloc_state =
+ XE_PAGEFAULT_ALLOC_STATE_CHAINED;
+ pf->consumer.next = pf_work->cache.pf->consumer.next;
+ pf_work->cache.pf->consumer.next = pf;
+
+ return true;
+ }
+ }
+
+ return false;
+}
+
+static void xe_pagefault_queue_advance(struct xe_pagefault_queue *pf_queue)
+{
+ lockdep_assert_held(&pf_queue->lock);
+
+ pf_queue->tail = (pf_queue->tail + xe_pagefault_entry_size()) %
+ pf_queue->size;
+}
+
+static struct xe_pagefault *
+xe_pagefault_queue_tail_fault(struct xe_pagefault_queue *pf_queue)
+{
+ lockdep_assert_held(&pf_queue->lock);
+
+ return pf_queue->data + pf_queue->tail;
+}
+
+static bool xe_pagefault_queue_empty(struct xe_pagefault_queue *pf_queue)
+{
+ lockdep_assert_held(&pf_queue->lock);
+
+ return pf_queue->head == pf_queue->tail;
+}
+
+static bool xe_pagefault_queue_pop(struct xe_pagefault_queue *pf_queue,
+ struct xe_pagefault **pf, int id)
+{
+ struct xe_device *xe = container_of(pf_queue, typeof(*xe),
+ usm.pf_queue);
+ struct xe_pagefault_work *pf_work, *__pf_work;
+ struct xe_pagefault *lpf;
+ size_t align = SZ_2M;
+ int i;
+
+ guard(spinlock_irq)(&pf_queue->lock);
+
+ for (*pf = NULL; !*pf;) {
+ if (xe_pagefault_queue_empty(pf_queue))
+ return false;
+
+ lpf = xe_pagefault_queue_tail_fault(pf_queue);
+ xe_pagefault_queue_advance(pf_queue);
+
+ if (lpf->consumer.alloc_state !=
+ XE_PAGEFAULT_ALLOC_STATE_QUEUED)
+ continue;
+
+ if (xe_pagefault_try_chain(pf_queue, lpf))
+ continue;
+
+ *pf = lpf; /* Hand back page fault for processing */
}
- spin_unlock_irq(&pf_queue->lock);
- return found_fault;
+ /*
+ * No cache hit; allocate a new cache entry. We assume most faults
+ * within a 2M range will hit the same pages. If this assumption proves
+ * false, the mismatched fault is requeued after the initial fault is
+ * acknowledged.
+ */
+ pf_work = xe->usm.pf_workers + id;
+ if (FIELD_GET(XE_PAGEFAULT_REQUEUE_MASK,
+ lpf->consumer.fault_type_level))
+ align = SZ_4K;
+ pf_work->cache.start = ALIGN_DOWN(lpf->consumer.page_addr, align);
+ pf_work->cache.end = pf_work->cache.start + align;
+ pf_work->cache.asid = lpf->consumer.asid;
+ pf_work->cache.pf = lpf;
+ lpf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_ACTIVE;
+
+ for (i = 0, __pf_work = xe->usm.pf_workers;
+ i < xe->info.num_pf_work; ++i, ++__pf_work) {
+ u64 cache_start = __pf_work->cache.start;
+
+ if (__pf_work == pf_work)
+ continue;
+
+ if (cache_start != XE_PAGEFAULT_CACHE_START_INVALID) {
+ xe_gt_stats_incr(xe_root_mmio_gt(xe),
+ XE_GT_STATS_ID_PARALLEL_PAGEFAULT_COUNT,
+ 1);
+ break;
+ }
+ }
+
+ /* Drain queue until empty or new fault found */
+ while (1) {
+ if (xe_pagefault_queue_empty(pf_queue))
+ break;
+
+ lpf = xe_pagefault_queue_tail_fault(pf_queue);
+
+ if (lpf->consumer.alloc_state !=
+ XE_PAGEFAULT_ALLOC_STATE_QUEUED) {
+ xe_pagefault_queue_advance(pf_queue);
+ continue;
+ }
+
+ if (!xe_pagefault_try_chain(pf_queue, lpf))
+ break;
+
+ xe_pagefault_queue_advance(pf_queue);
+ }
+
+ return true;
}
static void xe_pagefault_print(struct xe_pagefault *pf)
{
+ u8 engine_class = FIELD_GET(XE_PAGEFAULT_ENGINE_CLASS_MASK,
+ pf->consumer.engine_class_instance);
+
xe_gt_info(pf->gt, "\n\tASID: %d\n"
"\tFaulted Address: 0x%08x%08x\n"
"\tFaultType: %lu\n"
"\tAccessType: %lu\n"
"\tFaultLevel: %lu\n"
"\tEngineClass: %d %s\n"
- "\tEngineInstance: %d\n",
+ "\tEngineInstance: %lu\n",
pf->consumer.asid,
upper_32_bits(pf->consumer.page_addr),
lower_32_bits(pf->consumer.page_addr),
@@ -258,9 +561,10 @@ static void xe_pagefault_print(struct xe_pagefault *pf)
pf->consumer.access_type),
FIELD_GET(XE_PAGEFAULT_LEVEL_MASK,
pf->consumer.fault_type_level),
- pf->consumer.engine_class,
- xe_hw_engine_class_to_str(pf->consumer.engine_class),
- pf->consumer.engine_instance);
+ engine_class,
+ xe_hw_engine_class_to_str(engine_class),
+ FIELD_GET(XE_PAGEFAULT_ENGINE_INSTANCE_MASK,
+ pf->consumer.engine_class_instance));
}
static void xe_pagefault_save_to_vm(struct xe_device *xe, struct xe_pagefault *pf)
@@ -290,42 +594,104 @@ static void xe_pagefault_save_to_vm(struct xe_device *xe, struct xe_pagefault *p
static void xe_pagefault_queue_work(struct work_struct *w)
{
- struct xe_pagefault_queue *pf_queue =
- container_of(w, typeof(*pf_queue), worker);
- struct xe_pagefault pf;
+ struct xe_pagefault_work *pf_work =
+ container_of(w, typeof(*pf_work), work);
+ struct xe_device *xe = pf_work->xe;
+ struct xe_pagefault_queue *pf_queue = &xe->usm.pf_queue;
+ struct xe_pagefault *pf;
+ ktime_t start = xe_gt_stats_ktime_get();
unsigned long threshold;
+ u64 cache_start = XE_PAGEFAULT_CACHE_START_INVALID, cache_end = 0;
+ u32 cache_asid = 0;
#define USM_QUEUE_MAX_RUNTIME_MS 20
threshold = jiffies + msecs_to_jiffies(USM_QUEUE_MAX_RUNTIME_MS);
- while (xe_pagefault_queue_pop(pf_queue, &pf)) {
- int err;
+ while (xe_pagefault_queue_pop(pf_queue, &pf, pf_work->id)) {
+ const struct xe_pagefault_ops *ops = pf->producer.ops;
+ void *private = pf->producer.private;
+ struct xe_gt *gt = pf->gt;
+ u32 asid = pf->consumer.asid;
+ int err = 0;
+ bool invalidated = false;
+
+ /* Last fault same address, ack immediately */
+ if (xe_pagefault_match(pf, cache_start, cache_end, cache_asid)) {
+ xe_gt_stats_incr(gt, XE_GT_STATS_ID_LAST_PAGEFAULT_COUNT, 1);
+ goto ack_fault;
+ }
- if (!pf.gt) /* Fault squashed during reset */
- continue;
+ err = xe_pagefault_service(pf);
- err = xe_pagefault_service(&pf);
if (err) {
- xe_pagefault_save_to_vm(gt_to_xe(pf.gt), &pf);
- if (!(pf.consumer.access_type & XE_PAGEFAULT_ACCESS_PREFETCH)) {
- xe_pagefault_print(&pf);
- xe_gt_info(pf.gt, "Fault response: Unsuccessful %pe\n",
- ERR_PTR(err));
+ if (!(pf->consumer.access_type & XE_PAGEFAULT_ACCESS_PREFETCH)) {
+ xe_pagefault_save_to_vm(gt_to_xe(gt), pf);
+ xe_pagefault_cache_start_invalidate(cache_start);
+ xe_pagefault_print(pf);
+ xe_log_err_info(pf->gt, PAGEFAULT, err, "Unsuccessful response\n");
} else {
- xe_gt_stats_incr(pf.gt, XE_GT_STATS_ID_INVALID_PREFETCH_PAGEFAULT_COUNT, 1);
- xe_gt_dbg(pf.gt, "Prefetch Fault response: Unsuccessful %pe\n",
+ xe_gt_stats_incr(pf->gt, XE_GT_STATS_ID_INVALID_PREFETCH_PAGEFAULT_COUNT, 1);
+ xe_gt_dbg(pf->gt, "Prefetch Fault response: Unsuccessful %pe\n",
ERR_PTR(err));
}
+ } else {
+ /* Cache valid fault locally */
+ cache_start = xe_pagefault_start_addr(pf);
+ cache_end = xe_pagefault_end_addr(pf);
+ cache_asid = asid;
}
- pf.producer.ops->ack_fault(&pf, err);
+ack_fault:
+ xe_assert(xe, pf->consumer.alloc_state ==
+ XE_PAGEFAULT_ALLOC_STATE_ACTIVE);
+ xe_assert(xe, pf == pf_work->cache.pf);
+
+ ops->ack_fault_begin(private);
+ while (pf) {
+ xe_assert(xe, pf->consumer.alloc_state ==
+ XE_PAGEFAULT_ALLOC_STATE_ACTIVE);
+ xe_assert(xe, ops == pf->producer.ops);
+ xe_assert(xe, gt == pf->gt);
+
+ ops->ack_fault(pf, err);
+
+ spin_lock_irq(&pf_queue->lock);
+
+ if (!invalidated) {
+ invalidated = true;
+ xe_pagefault_cache_invalidate(pf_queue,
+ pf_work);
+ }
+
+ pf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_FREE;
+ pf = pf->consumer.next;
+
+ /*
+ * Requeue chained faults which do not match the last
+ * fault processed
+ */
+ while (pf && !xe_pagefault_match(pf, cache_start,
+ cache_end, cache_asid))
+ pf = xe_pagefault_queue_unchain_requeue(pf_queue, pf, gt);
+
+ /* Ensure resets are safe */
+ if (pf)
+ pf->consumer.alloc_state =
+ XE_PAGEFAULT_ALLOC_STATE_ACTIVE;
+
+ spin_unlock_irq(&pf_queue->lock);
+ }
+ ops->ack_fault_end(private);
if (time_after(jiffies, threshold)) {
- queue_work(gt_to_xe(pf.gt)->usm.pf_wq, w);
+ queue_work(xe->usm.pagefault_wq, w);
break;
}
}
#undef USM_QUEUE_MAX_RUNTIME_MS
+
+ xe_gt_stats_incr(xe_root_mmio_gt(xe), XE_GT_STATS_ID_PAGEFAULT_US,
+ xe_gt_stats_ktime_us_delta(start));
}
static int xe_pagefault_queue_init(struct xe_device *xe,
@@ -366,7 +732,6 @@ static int xe_pagefault_queue_init(struct xe_device *xe,
xe_pagefault_entry_size(), total_num_eus, pf_queue->size);
spin_lock_init(&pf_queue->lock);
- INIT_WORK(&pf_queue->worker, xe_pagefault_queue_work);
pf_queue->data = drmm_kzalloc(&xe->drm, pf_queue->size, GFP_KERNEL);
if (!pf_queue->data)
@@ -379,7 +744,8 @@ static void xe_pagefault_fini(void *arg)
{
struct xe_device *xe = arg;
- destroy_workqueue(xe->usm.pf_wq);
+ destroy_workqueue(xe->usm.prefetch_wq);
+ destroy_workqueue(xe->usm.pagefault_wq);
}
/**
@@ -397,22 +763,39 @@ int xe_pagefault_init(struct xe_device *xe)
if (!xe->info.has_usm)
return 0;
- xe->usm.pf_wq = alloc_workqueue("xe_page_fault_work_queue",
- WQ_UNBOUND | WQ_HIGHPRI,
- XE_PAGEFAULT_QUEUE_COUNT);
- if (!xe->usm.pf_wq)
+ xe->usm.pagefault_wq = alloc_workqueue("xe_page_fault_work_queue",
+ WQ_UNBOUND | WQ_HIGHPRI,
+ xe->info.num_pf_work);
+ if (!xe->usm.pagefault_wq)
return -ENOMEM;
- for (i = 0; i < XE_PAGEFAULT_QUEUE_COUNT; ++i) {
- err = xe_pagefault_queue_init(xe, xe->usm.pf_queue + i);
- if (err)
- goto err_out;
+ xe->usm.prefetch_wq = alloc_workqueue("xe_prefetch_work_queue",
+ WQ_UNBOUND,
+ xe->info.num_pf_work);
+ if (!xe->usm.prefetch_wq) {
+ err = -ENOMEM;
+ goto err_pagefault_wq;
+ }
+
+ err = xe_pagefault_queue_init(xe, &xe->usm.pf_queue);
+ if (err)
+ goto err_out;
+
+ for (i = 0; i < xe->info.num_pf_work; ++i) {
+ struct xe_pagefault_work *pf_work = xe->usm.pf_workers + i;
+
+ pf_work->xe = xe;
+ pf_work->id = i;
+ xe_pagefault_cache_start_invalidate(pf_work->cache.start);
+ INIT_WORK(&pf_work->work, xe_pagefault_queue_work);
}
return devm_add_action_or_reset(xe->drm.dev, xe_pagefault_fini, xe);
err_out:
- destroy_workqueue(xe->usm.pf_wq);
+ destroy_workqueue(xe->usm.prefetch_wq);
+err_pagefault_wq:
+ destroy_workqueue(xe->usm.pagefault_wq);
return err;
}
@@ -427,15 +810,23 @@ static void xe_pagefault_queue_reset(struct xe_device *xe, struct xe_gt *gt,
/* Squash all pending faults on the GT */
- spin_lock_irq(&pf_queue->lock);
- for (i = pf_queue->tail; i != pf_queue->head;
- i = (i + xe_pagefault_entry_size()) % pf_queue->size) {
+ guard(spinlock_irq)(&pf_queue->lock);
+
+ for (i = 0; i < pf_queue->size; i += xe_pagefault_entry_size()) {
struct xe_pagefault *pf = pf_queue->data + i;
+ bool active = pf->consumer.alloc_state ==
+ XE_PAGEFAULT_ALLOC_STATE_ACTIVE;
- if (pf->gt == gt)
- pf->gt = NULL;
+ if (pf->gt != gt || active) {
+ if (active)
+ pf->consumer.next = NULL;
+ continue;
+ }
+
+ pf->consumer.alloc_state =
+ XE_PAGEFAULT_ALLOC_STATE_FREE;
+ pf->consumer.next = NULL;
}
- spin_unlock_irq(&pf_queue->lock);
}
/**
@@ -448,18 +839,18 @@ static void xe_pagefault_queue_reset(struct xe_device *xe, struct xe_gt *gt,
*/
void xe_pagefault_reset(struct xe_device *xe, struct xe_gt *gt)
{
- int i;
-
- for (i = 0; i < XE_PAGEFAULT_QUEUE_COUNT; ++i)
- xe_pagefault_queue_reset(xe, gt, xe->usm.pf_queue + i);
+ xe_pagefault_queue_reset(xe, gt, &xe->usm.pf_queue);
}
-static bool xe_pagefault_queue_full(struct xe_pagefault_queue *pf_queue)
+/*
+ * This function can race with multiple page fault producers, but worst case we
+ * stick a page fault on the same queue for consumption.
+ */
+static int xe_pagefault_work_index(struct xe_device *xe)
{
- lockdep_assert_held(&pf_queue->lock);
+ lockdep_assert_held(&xe->usm.pf_queue.lock);
- return CIRC_SPACE(pf_queue->head, pf_queue->tail, pf_queue->size) <=
- xe_pagefault_entry_size();
+ return xe->usm.current_pf_work++ % xe->info.num_pf_work;
}
/**
@@ -474,24 +865,95 @@ static bool xe_pagefault_queue_full(struct xe_pagefault_queue *pf_queue)
*/
int xe_pagefault_handler(struct xe_device *xe, struct xe_pagefault *pf)
{
- struct xe_pagefault_queue *pf_queue = xe->usm.pf_queue +
- (pf->consumer.asid % XE_PAGEFAULT_QUEUE_COUNT);
- unsigned long flags;
- bool full;
-
- spin_lock_irqsave(&pf_queue->lock, flags);
- full = xe_pagefault_queue_full(pf_queue);
- if (!full) {
- memcpy(pf_queue->data + pf_queue->head, pf, sizeof(*pf));
- pf_queue->head = (pf_queue->head + xe_pagefault_entry_size()) %
- pf_queue->size;
- queue_work(xe->usm.pf_wq, &pf_queue->worker);
+ struct xe_pagefault_queue *pf_queue = &xe->usm.pf_queue;
+ struct xe_pagefault *lpf;
+ bool empty;
+
+ guard(spinlock_irqsave)(&pf_queue->lock);
+
+ empty = xe_pagefault_queue_empty(pf_queue);
+ lpf = xe_pagefault_queue_add(pf_queue, pf);
+ if (!lpf)
+ return -ENOSPC;
+
+ lpf->consumer.next = NULL;
+ if (xe_pagefault_try_chain(pf_queue, lpf)) {
+ xe_gt_stats_incr(pf->gt, XE_GT_STATS_ID_CHAIN_IRQ_PAGEFAULT_COUNT, 1);
+ if (empty) {
+ xe_gt_stats_incr(pf->gt,
+ XE_GT_STATS_ID_CHAIN_DRAIN_IRQ_PAGEFAULT_COUNT, 1);
+ xe_pagefault_queue_advance(pf_queue);
+ }
} else {
- drm_warn(&xe->drm,
- "PageFault Queue (%d) full, shouldn't be possible\n",
- pf->consumer.asid % XE_PAGEFAULT_QUEUE_COUNT);
+ int work_index = xe_pagefault_work_index(xe);
+
+ queue_work(xe->usm.pagefault_wq,
+ &xe->usm.pf_workers[work_index].work);
+ }
+
+ return 0;
+}
+
+/**
+ * xe_pagefault_print_info() - dump page fault queue/cache debug information
+ * @xe: Xe device
+ * @p: DRM printer to emit output to
+ *
+ * Print a snapshot of the page fault queue state for debugging. The output
+ * includes queue parameters (entry size, total size, head/tail), a histogram
+ * of per-entry allocation state values, and the validity of each per-worker
+ * page fault cache.
+ *
+ * This function is intended for debugfs and similar diagnostics. It acquires
+ * the page fault queue spinlock internally to serialize against IRQ-side
+ * producers and the worker consumer path, so callers must not hold the queue
+ * lock.
+ */
+void xe_pagefault_print_info(struct xe_device *xe, struct drm_printer *p)
+{
+ struct xe_pagefault_queue *pf_queue = &xe->usm.pf_queue;
+ struct xe_pagefault_work *pf_work;
+ static const char * const alloc_state_names[] = {
+ [XE_PAGEFAULT_ALLOC_STATE_FREE] = "free",
+ [XE_PAGEFAULT_ALLOC_STATE_QUEUED] = "queued",
+ [XE_PAGEFAULT_ALLOC_STATE_CHAINED] = "chained",
+ [XE_PAGEFAULT_ALLOC_STATE_ACTIVE] = "active",
+ };
+ u32 i, counts[XE_PAGEFAULT_ALLOC_STATE_COUNT] = {};
+
+ /* Driver load failure guard / USM not enabled guard */
+ if (!pf_queue->data)
+ return;
+
+ guard(spinlock_irq)(&pf_queue->lock);
+
+ drm_printf(p, "pagefault size: %u\n", xe_pagefault_entry_size());
+ drm_printf(p, "pagefault queue size: %u\n", pf_queue->size);
+ drm_printf(p, "pagefault queue head: %u\n", pf_queue->head);
+ drm_printf(p, "pagefault queue tail: %u\n", pf_queue->tail);
+
+ for (i = 0; i < pf_queue->size; i += xe_pagefault_entry_size()) {
+ struct xe_pagefault *pf = pf_queue->data + i;
+
+ if (pf->consumer.alloc_state >=
+ XE_PAGEFAULT_ALLOC_STATE_COUNT) {
+ drm_printf(p, "pagefault[%u] corrupted alloc_state=%u\n",
+ i, pf->consumer.alloc_state);
+ continue;
+ }
+
+ counts[pf->consumer.alloc_state]++;
}
- spin_unlock_irqrestore(&pf_queue->lock, flags);
- return full ? -ENOSPC : 0;
+ for (i = 0; i < XE_PAGEFAULT_ALLOC_STATE_COUNT; ++i)
+ drm_printf(p, "pagefault queue %s count: %u\n",
+ alloc_state_names[i], counts[i]);
+
+ for (i = 0, pf_work = xe->usm.pf_workers;
+ i < xe->info.num_pf_work; ++i, ++pf_work) {
+ if (pf_work->cache.start == XE_PAGEFAULT_CACHE_START_INVALID)
+ drm_printf(p, "pagefault work[%u] cache invalid\n", i);
+ else
+ drm_printf(p, "pagefault work[%u] cache valid\n", i);
+ }
}
diff --git a/drivers/gpu/drm/xe/xe_pagefault.h b/drivers/gpu/drm/xe/xe_pagefault.h
index bd0cdf9ed37f..e9c5d1f03760 100644
--- a/drivers/gpu/drm/xe/xe_pagefault.h
+++ b/drivers/gpu/drm/xe/xe_pagefault.h
@@ -6,6 +6,9 @@
#ifndef _XE_PAGEFAULT_H_
#define _XE_PAGEFAULT_H_
+#include "xe_pagefault_types.h"
+
+struct drm_printer;
struct xe_device;
struct xe_gt;
struct xe_pagefault;
@@ -16,4 +19,75 @@ void xe_pagefault_reset(struct xe_device *xe, struct xe_gt *gt);
int xe_pagefault_handler(struct xe_device *xe, struct xe_pagefault *pf);
+void xe_pagefault_print_info(struct xe_device *xe, struct drm_printer *p);
+
+#define XE_PAGEFAULT_END_ADDR_MASK (~0xfffull)
+
+/**
+ * xe_pagefault_set_end_addr() - store serviced range end for a pagefault
+ * @pf: Pagefault entry
+ * @end_addr: Inclusive end address of the serviced fault range
+ *
+ * The pagefault consumer stores the resolved fault range so subsequent faults
+ * hitting the same range can be immediately acknowledged without re-running
+ * the full fault handling path.
+ *
+ * The end address shares storage with other consumer metadata and therefore
+ * must be masked with %XE_PAGEFAULT_END_ADDR_MASK before storing. Bits outside
+ * the mask are reserved for internal state tracking and must be preserved.
+ */
+static inline void
+xe_pagefault_set_end_addr(struct xe_pagefault *pf, u64 end_addr)
+{
+ pf->consumer.end_addr &= ~XE_PAGEFAULT_END_ADDR_MASK;
+ pf->consumer.end_addr |= end_addr;
+}
+
+/**
+ * xe_pagefault_end_addr() - read serviced range end for a pagefault
+ * @pf: Pagefault entry
+ *
+ * Returns the inclusive end address of the range previously recorded by
+ * xe_pagefault_set_end_addr(). Only the bits covered by
+ * %XE_PAGEFAULT_END_ADDR_MASK are returned; other bits in the storage are
+ * reserved for internal state.
+ *
+ * Return: End address of the serviced fault range.
+ */
+static inline u64 xe_pagefault_end_addr(struct xe_pagefault *pf)
+{
+ return pf->consumer.end_addr & XE_PAGEFAULT_END_ADDR_MASK;
+}
+
+#undef XE_PAGEFAULT_END_ADDR_MASK
+
+/**
+ * xe_pagefault_set_start_addr() - store serviced range start for a pagefault
+ * @pf: Pagefault entry
+ * @start_addr: Start address of the serviced fault range
+ *
+ * The pagefault consumer stores the resolved fault range so subsequent faults
+ * hitting the same range can be immediately acknowledged without re-running
+ * the full fault handling path.
+ */
+static inline void
+xe_pagefault_set_start_addr(struct xe_pagefault *pf, u64 start_addr)
+{
+ pf->consumer.page_addr = start_addr;
+}
+
+/**
+ * xe_pagefault_start_addr() - read serviced range start for a pagefault
+ * @pf: Pagefault entry
+ *
+ * Returns the inclusive start address of the range previously recorded by
+ * xe_pagefault_set_start_addr().
+ *
+ * Return: Start address of the serviced fault range.
+ */
+static inline u64 xe_pagefault_start_addr(struct xe_pagefault *pf)
+{
+ return pf->consumer.page_addr;
+}
+
#endif
diff --git a/drivers/gpu/drm/xe/xe_pagefault_types.h b/drivers/gpu/drm/xe/xe_pagefault_types.h
index c4ee625b93dd..efeba5c3a58b 100644
--- a/drivers/gpu/drm/xe/xe_pagefault_types.h
+++ b/drivers/gpu/drm/xe/xe_pagefault_types.h
@@ -34,6 +34,13 @@ enum xe_pagefault_type {
/** struct xe_pagefault_ops - Xe pagefault ops (producer) */
struct xe_pagefault_ops {
/**
+ * @ack_fault_begin: Ack fault begin
+ * @private: producer private data
+ *
+ * Page fault producer begins acknowledgment from the consumer.
+ */
+ void (*ack_fault_begin)(void *private);
+ /**
* @ack_fault: Ack fault
* @pf: Page fault
* @err: Error state of fault
@@ -42,6 +49,13 @@ struct xe_pagefault_ops {
* sends the result to the HW/FW interface.
*/
void (*ack_fault)(struct xe_pagefault *pf, int err);
+ /**
+ * @ack_fault_end: Ack fault end
+ * @private: producer private data
+ *
+ * Page fault producer ends acknowledgment from the consumer.
+ */
+ void (*ack_fault_end)(void *private);
};
/**
@@ -60,34 +74,58 @@ struct xe_pagefault {
/**
* @consumer: State for the software handling the fault. Populated by
* the producer and may be modified by the consumer to communicate
- * information back to the producer upon fault acknowledgment.
+ * information back to the producer upon fault acknowledgment. After
+ * fault acknowledgment, the producer should only access consumer fields
+ * via well defined helpers.
*/
struct {
- /** @consumer.page_addr: address of page fault */
- u64 page_addr;
- /** @consumer.asid: address space ID */
- u32 asid;
/**
- * @consumer.access_type: access type and prefetch flag packed
- * into a u8.
+ * @consumer.page_addr: address of page fault, populated by
+ * consumer after fault completion
*/
- u8 access_type;
+ u64 page_addr;
+ union {
+ struct {
+ /**
+ * @consumer.alloc_state: page fault allocation
+ * state
+ */
+ u8 alloc_state;
+ /**
+ * @consumer.access_type: access type, u8 rather
+ * than enum to keep size compact
+ */
+ u8 access_type;
#define XE_PAGEFAULT_ACCESS_TYPE_MASK GENMASK(1, 0)
#define XE_PAGEFAULT_ACCESS_PREFETCH BIT(7)
+ /**
+ * @consumer.fault_type_level: fault type and
+ * level, u8 rather than enum to keep size
+ * compact
+ */
+ u8 fault_type_level;
+#define XE_PAGEFAULT_TYPE_LEVEL_NACK 0xff /* Producer indicates nack fault */
+#define XE_PAGEFAULT_LEVEL_MASK GENMASK(2, 0)
+#define XE_PAGEFAULT_TYPE_MASK GENMASK(6, 3)
+#define XE_PAGEFAULT_REQUEUE_MASK BIT(7)
+ /** @consumer.engine_class_instance: engine class and instance */
+ u8 engine_class_instance;
+#define XE_PAGEFAULT_ENGINE_CLASS_MASK GENMASK(3, 0)
+#define XE_PAGEFAULT_ENGINE_INSTANCE_MASK GENMASK(7, 4)
+ /** @consumer.asid: address space ID */
+ u32 asid;
+ };
+ /**
+ * @consumer.end_addr: end address of page fault,
+ * populated by consumer after fault completion
+ */
+ u64 end_addr;
+ };
/**
- * @consumer.fault_type_level: fault type and level, u8 rather
- * than enum to keep size compact
+ * @consumer.next: next pagefault chained to this fault,
+ * protected by pf_queue lock
*/
- u8 fault_type_level;
-#define XE_PAGEFAULT_TYPE_LEVEL_NACK 0xff /* Producer indicates nack fault */
-#define XE_PAGEFAULT_LEVEL_MASK GENMASK(3, 0)
-#define XE_PAGEFAULT_TYPE_MASK GENMASK(7, 4)
- /** @consumer.engine_class: engine class */
- u8 engine_class;
- /** @consumer.engine_instance: engine instance */
- u8 engine_instance;
- /** @consumer.reserved: reserved bits for future expansion */
- u64 reserved;
+ struct xe_pagefault *next;
} consumer;
/**
* @producer: State for the producer (i.e., HW/FW interface). Populated
@@ -129,10 +167,38 @@ struct xe_pagefault_queue {
u32 head;
/** @tail: Tail pointer in bytes, moved by consumer, protected by @lock */
u32 tail;
- /** @lock: protects page fault queue */
+ /** @lock: protects page fault queue, workers caches */
spinlock_t lock;
- /** @worker: to process page faults */
- struct work_struct worker;
+};
+
+/**
+ * struct xe_pagefault_work - Xe page fault work item (consumer)
+ *
+ * Represents a worker that pops a &struct xe_pagefault from the page fault
+ * queue and processes it.
+ */
+struct xe_pagefault_work {
+ /** @xe: Back-pointer to the Xe device */
+ struct xe_device *xe;
+ /** @id: Identifier for this work item */
+ int id;
+ /**
+ * @cache: Page fault cache for the currently processed fault
+ *
+ * Protected by the page fault queue lock.
+ */
+ struct {
+ /** @cache.start: Start address of the current page fault */
+ u64 start;
+ /** @cache.end: End address of the current page fault */
+ u64 end;
+ /** @cache.asid: Address space ID of the current page fault */
+ u32 asid;
+ /** @cache.pf: Pointer to the current page fault */
+ struct xe_pagefault *pf;
+ } cache;
+ /** @work: Work item used to process the page fault */
+ struct work_struct work;
};
#endif
diff --git a/drivers/gpu/drm/xe/xe_pci.c b/drivers/gpu/drm/xe/xe_pci.c
index 5f2a0b19839d..1e04e8ef2611 100644
--- a/drivers/gpu/drm/xe/xe_pci.c
+++ b/drivers/gpu/drm/xe/xe_pci.c
@@ -25,6 +25,7 @@
#include "xe_gt_printk.h"
#include "xe_gt_sriov_vf.h"
#include "xe_guc.h"
+#include "xe_log.h"
#include "xe_mmio.h"
#include "xe_module.h"
#include "xe_pci_error.h"
@@ -1145,17 +1146,12 @@ static void xe_pci_remove(struct pci_dev *pdev)
* caller. Therefore there is no consequence on those specific callers when
* function error injection skips the whole function.
*/
+static int __xe_pci_probe(struct pci_dev *pdev, const struct xe_device_desc *desc);
static int xe_pci_probe(struct pci_dev *pdev, const struct pci_device_id *ent)
{
- struct xe_probed_info probed_info = {};
const struct xe_device_desc *desc = (const void *)ent->driver_data;
- const struct xe_subplatform_desc *subplatform_desc;
- struct xe_device *xe;
- void *group;
int err;
- subplatform_desc = find_subplatform(desc, pdev->device);
-
xe_configfs_check_device(pdev);
if (desc->require_force_probe && !id_forced(pdev->device)) {
@@ -1171,14 +1167,34 @@ static int xe_pci_probe(struct pci_dev *pdev, const struct pci_device_id *ent)
}
if (id_blocked(pdev->device)) {
- dev_info(&pdev->dev, "Probe blocked for device [%04x:%04x].\n",
- pdev->vendor, pdev->device);
+ xe_log_info(pdev, PROBE, "driver loading blocked for device '%04x'\n",
+ pdev->device);
return -ENODEV;
}
if (xe_display_driver_probe_defer(pdev))
return -EPROBE_DEFER;
+ err = __xe_pci_probe(pdev, desc);
+ if (err) {
+ xe_log_err_fatal(pdev, PROBE, err, "driver loading failed for device '%04x'\n",
+ pdev->device);
+ return err;
+ }
+
+ return 0;
+}
+
+static int __xe_pci_probe(struct pci_dev *pdev, const struct xe_device_desc *desc)
+{
+ const struct xe_subplatform_desc *subplatform_desc;
+ struct xe_probed_info probed_info = {};
+ struct xe_device *xe;
+ void *group;
+ int err;
+
+ subplatform_desc = find_subplatform(desc, pdev->device);
+
/* Group all devres so xe_pci_error_slot_reset() can release them as a unit. */
group = devres_open_group(&pdev->dev, NULL, GFP_KERNEL);
if (!group)
diff --git a/drivers/gpu/drm/xe/xe_pci_error.c b/drivers/gpu/drm/xe/xe_pci_error.c
index e41af2ac7f23..79ce0c671549 100644
--- a/drivers/gpu/drm/xe/xe_pci_error.c
+++ b/drivers/gpu/drm/xe/xe_pci_error.c
@@ -7,6 +7,7 @@
#include "xe_device.h"
#include "xe_gt.h"
+#include "xe_log.h"
#include "xe_pci.h"
#include "xe_pm.h"
#include "xe_printk.h"
@@ -83,6 +84,12 @@ static pci_ers_result_t xe_pci_error_mmio_enabled(struct pci_dev *pdev)
xe_info(xe, "PCI error: MMIO enabled\n");
action = xe_ras_process_errors(xe);
+ /* User wants to debug the error, prevent reset */
+ if (xe->wedged.mode == XE_WEDGED_MODE_UPON_ANY_HANG_NO_RESET) {
+ xe_device_declare_wedged(xe);
+ return PCI_ERS_RESULT_DISCONNECT;
+ }
+
return ras_action_to_pci_result(pdev, action);
}
@@ -90,13 +97,15 @@ static pci_ers_result_t xe_pci_error_slot_reset(struct pci_dev *pdev)
{
const struct pci_device_id *ent = pci_match_id(pdev->driver->id_table, pdev);
struct xe_device *xe = pdev_to_xe_device(pdev);
+ int err;
xe_info(xe, "PCI error: slot reset\n");
pci_restore_state(pdev);
- if (pci_enable_device(pdev)) {
- xe_err(xe, "Cannot re-enable PCI device after reset\n");
+ err = pci_enable_device(pdev);
+ if (err) {
+ xe_log_err_fatal(xe, PCI, err, "Cannot re-enable PCI device after reset\n");
return PCI_ERS_RESULT_DISCONNECT;
}
diff --git a/drivers/gpu/drm/xe/xe_pcode.c b/drivers/gpu/drm/xe/xe_pcode.c
index ccc3bdeed6bb..e1b8062541a9 100644
--- a/drivers/gpu/drm/xe/xe_pcode.c
+++ b/drivers/gpu/drm/xe/xe_pcode.c
@@ -14,6 +14,7 @@
#include "regs/xe_pmt.h"
#include "xe_assert.h"
#include "xe_device.h"
+#include "xe_log.h"
#include "xe_mmio.h"
#include "xe_pcode_api.h"
#include "xe_pm.h"
@@ -61,9 +62,7 @@ static int pcode_mailbox_status(struct xe_tile *tile)
}
if (err) {
- drm_err(&tile_to_xe(tile)->drm, "PCODE Mailbox failed: %d %s",
- err_decode, err_str);
-
+ xe_log_err(tile, PCODE, err_decode, "Mailbox failed: %s\n", err_str);
return err_decode;
}
@@ -219,8 +218,7 @@ int xe_pcode_request(struct xe_tile *tile, u32 mbox, u32 request,
* requests, and for any quirks of the PCODE firmware that delays
* the request completion.
*/
- drm_err(&tile_to_xe(tile)->drm,
- "PCODE timeout, retrying with preemption disabled\n");
+ xe_log_err(tile, PCODE, ret, "timeout, retrying with preemption disabled\n");
preempt_disable();
ret = pcode_try_request(tile, mbox, request, reply_mask, reply, &status,
true, 50 * 1000, true);
@@ -299,7 +297,7 @@ int xe_pcode_ready(struct xe_device *xe, bool locked)
{
u32 status, request = DGFX_GET_INIT_STATUS;
struct xe_tile *tile = xe_device_get_root_tile(xe);
- int timeout_us = 180000000; /* 3 min */
+ long timeout_us = 3 * 60 * USEC_PER_SEC; /* 3 min */
int ret;
if (xe->info.skip_pcode)
@@ -320,8 +318,8 @@ int xe_pcode_ready(struct xe_device *xe, bool locked)
mutex_unlock(&tile->pcode.lock);
if (ret)
- drm_err(&xe->drm,
- "PCODE initialization timedout after: 3 min\n");
+ xe_log_err(tile, PCODE, ret, "initialization timedout after %ld seconds\n",
+ timeout_us / USEC_PER_SEC);
return ret;
}
diff --git a/drivers/gpu/drm/xe/xe_pm.c b/drivers/gpu/drm/xe/xe_pm.c
index a5289a9df8d2..f517bf453b54 100644
--- a/drivers/gpu/drm/xe/xe_pm.c
+++ b/drivers/gpu/drm/xe/xe_pm.c
@@ -905,6 +905,11 @@ static bool xe_pm_suspending_or_resuming(struct xe_device *xe)
* break scope-based handling, or when the lifetime of the runtime PM reference
* does not match a specific scope (e.g., runtime PM obtained in one function
* and released in a different one).
+ *
+ * This helper assumes the caller already holds a runtime PM reference and
+ * only warns when it cannot see one. After hot-unplug runtime PM is disabled
+ * and the check fails even when a reference is held, so callers that may run
+ * after unplug must guard it with drm_dev_enter()/drm_dev_exit() instead.
*/
void xe_pm_runtime_get_noresume(struct xe_device *xe)
{
diff --git a/drivers/gpu/drm/xe/xe_printk.h b/drivers/gpu/drm/xe/xe_printk.h
index c5be2385aa95..87f51d7edfa8 100644
--- a/drivers/gpu/drm/xe/xe_printk.h
+++ b/drivers/gpu/drm/xe/xe_printk.h
@@ -36,6 +36,9 @@
#define xe_dbg(_xe, _fmt, ...) \
xe_printk((_xe), dbg, _fmt, ##__VA_ARGS__)
+#define xe_dbg_ratelimited(_xe, _fmt, ...) \
+ xe_printk((_xe), dbg_ratelimited, _fmt, ##__VA_ARGS__)
+
#define xe_WARN_type(_xe, _type, _condition, _fmt, ...) \
drm_WARN##_type(&(_xe)->drm, _condition, _fmt, ## __VA_ARGS__)
diff --git a/drivers/gpu/drm/xe/xe_pt.c b/drivers/gpu/drm/xe/xe_pt.c
index a07316a45d79..5d990c1c3740 100644
--- a/drivers/gpu/drm/xe/xe_pt.c
+++ b/drivers/gpu/drm/xe/xe_pt.c
@@ -2414,6 +2414,12 @@ static int op_prepare(struct xe_vm *vm,
xa_for_each(&op->prefetch_range.range, i, range) {
err = bind_range_prepare(vm, tile, pt_update_ops,
vma, range);
+ /*
+ * Don't tell user space to retry, rather let
+ * page faults fixup the pages.
+ */
+ if (err == -EAGAIN)
+ err = -ENODATA;
if (err)
return err;
}
diff --git a/drivers/gpu/drm/xe/xe_pxp_submit.c b/drivers/gpu/drm/xe/xe_pxp_submit.c
index e60526e30030..5de86a8cf27d 100644
--- a/drivers/gpu/drm/xe/xe_pxp_submit.c
+++ b/drivers/gpu/drm/xe/xe_pxp_submit.c
@@ -46,7 +46,7 @@ static int allocate_vcs_execution_resources(struct xe_pxp *pxp)
return -ENODEV;
q = xe_exec_queue_create(xe, NULL, BIT(hwe->logical_instance), 1, hwe,
- EXEC_QUEUE_FLAG_KERNEL | EXEC_QUEUE_FLAG_PERMANENT, 0);
+ EXEC_QUEUE_FLAG_KERNEL, 0);
if (IS_ERR(q))
return PTR_ERR(q);
@@ -144,8 +144,7 @@ static int allocate_gsc_client_resources(struct xe_gt *gt,
}
q = xe_exec_queue_create(xe, vm, BIT(hwe->logical_instance), 1, hwe,
- EXEC_QUEUE_FLAG_KERNEL |
- EXEC_QUEUE_FLAG_PERMANENT, 0);
+ EXEC_QUEUE_FLAG_KERNEL, 0);
if (IS_ERR(q)) {
err = PTR_ERR(q);
goto bo_out;
diff --git a/drivers/gpu/drm/xe/xe_ras.c b/drivers/gpu/drm/xe/xe_ras.c
index d98ff9453f60..d25d25f77531 100644
--- a/drivers/gpu/drm/xe/xe_ras.c
+++ b/drivers/gpu/drm/xe/xe_ras.c
@@ -3,8 +3,10 @@
* Copyright © 2026 Intel Corporation
*/
+#include "xe_debugfs.h"
#include "xe_device.h"
#include "xe_drm_ras.h"
+#include "xe_log.h"
#include "xe_pm.h"
#include "xe_printk.h"
#include "xe_ras.h"
@@ -45,6 +47,16 @@ enum xe_ras_component {
XE_RAS_COMP_MAX
};
+#define CHECK_COMPONENT(RAS_COMP, LOG_COMP) \
+ static_assert(MAKE_XE_LOG_COMPONENT(HARDWARE, (RAS_COMP)) == (LOG_COMP))
+ /* make sure components definitions maintain stable relation */
+ CHECK_COMPONENT(XE_RAS_COMP_DEVICE_MEMORY, XE_LOG_COMPONENT_DEVICE_MEMORY);
+ CHECK_COMPONENT(XE_RAS_COMP_CORE_COMPUTE, XE_LOG_COMPONENT_CORE_COMPUTE);
+ CHECK_COMPONENT(XE_RAS_COMP_PCIE, XE_LOG_COMPONENT_PCIE);
+ CHECK_COMPONENT(XE_RAS_COMP_FABRIC, XE_LOG_COMPONENT_FABRIC);
+ CHECK_COMPONENT(XE_RAS_COMP_SOC_INTERNAL, XE_LOG_COMPONENT_SOC_INTERNAL);
+#undef CHECK_COMPONENT
+
/* RAS response status codes */
enum xe_ras_response_status {
XE_RAS_STATUS_SUCCESS = 0,
@@ -90,6 +102,8 @@ static const char * const gpu_health_states[] = {
};
static_assert(ARRAY_SIZE(gpu_health_states) == XE_RAS_HEALTH_MAX);
+static int get_counter(struct xe_device *xe, struct xe_ras_error_class *counter, u32 *value);
+
static u8 drm_to_xe_ras_severity(u8 severity)
{
switch (severity) {
@@ -102,6 +116,18 @@ static u8 drm_to_xe_ras_severity(u8 severity)
}
}
+static u8 xe_to_drm_ras_severity(u8 severity)
+{
+ switch (severity) {
+ case XE_RAS_SEV_CORRECTABLE:
+ return DRM_XE_RAS_ERR_SEV_CORRECTABLE;
+ case XE_RAS_SEV_UNCORRECTABLE:
+ return DRM_XE_RAS_ERR_SEV_UNCORRECTABLE;
+ default:
+ return DRM_XE_RAS_ERR_SEV_MAX;
+ }
+}
+
static u8 drm_to_xe_ras_component(u8 component)
{
switch (component) {
@@ -120,6 +146,24 @@ static u8 drm_to_xe_ras_component(u8 component)
}
}
+static u8 xe_to_drm_ras_component(u8 component)
+{
+ switch (component) {
+ case XE_RAS_COMP_DEVICE_MEMORY:
+ return DRM_XE_RAS_ERR_COMP_DEVICE_MEMORY;
+ case XE_RAS_COMP_CORE_COMPUTE:
+ return DRM_XE_RAS_ERR_COMP_CORE_COMPUTE;
+ case XE_RAS_COMP_PCIE:
+ return DRM_XE_RAS_ERR_COMP_PCIE;
+ case XE_RAS_COMP_FABRIC:
+ return DRM_XE_RAS_ERR_COMP_FABRIC;
+ case XE_RAS_COMP_SOC_INTERNAL:
+ return DRM_XE_RAS_ERR_COMP_SOC_INTERNAL;
+ default:
+ return DRM_XE_RAS_ERR_COMP_MAX;
+ }
+}
+
static int ras_status_to_errno(u32 status)
{
switch (status) {
@@ -156,6 +200,24 @@ static inline const char *comp_to_str(u8 component)
return xe_ras_components[component];
}
+static bool ras_counter_is_valid(struct xe_device *xe, struct xe_ras_error_class *counter)
+{
+ u8 severity = counter->common.severity;
+ u8 component = counter->common.component;
+
+ if (!in_range(severity, XE_RAS_SEV_NOT_SUPPORTED + 1, XE_RAS_SEV_MAX - 1)) {
+ xe_err(xe, "sysctrl: unexpected severity %u\n", severity);
+ return false;
+ }
+
+ if (!in_range(component, XE_RAS_COMP_NOT_SUPPORTED + 1, XE_RAS_COMP_MAX - 1)) {
+ xe_err(xe, "sysctrl: unexpected component %u\n", component);
+ return false;
+ }
+
+ return true;
+}
+
static struct pci_dev *find_usp_dev(struct pci_dev *pdev)
{
struct pci_dev *vsp;
@@ -218,6 +280,26 @@ static void ras_usp_aer_init(struct xe_device *xe)
dev_dbg(&usp->dev, "Uncorrectable Internal Errors downgraded and unmasked\n");
}
+static void ras_send_error_event(struct xe_device *xe, u8 severity, u8 component)
+{
+ struct xe_ras_error_class counter = {0};
+ u8 drm_severity, drm_component;
+ u32 value;
+ int ret;
+
+ counter.common.severity = severity;
+ counter.common.component = component;
+
+ ret = get_counter(xe, &counter, &value);
+ if (ret)
+ return;
+
+ drm_severity = xe_to_drm_ras_severity(severity);
+ drm_component = xe_to_drm_ras_component(component);
+
+ xe_drm_ras_event(xe, drm_component, drm_severity, value);
+}
+
static u8 handle_core_compute_errors(struct xe_ras_error_array *arr)
{
struct xe_ras_compute_error *error_info = (void *)arr->details;
@@ -236,6 +318,12 @@ static u8 handle_core_compute_errors(struct xe_ras_error_array *arr)
return XE_RAS_RECOVERY_ACTION_RECOVERED;
}
+static void punit_error_handler(struct xe_device *xe)
+{
+ xe_device_set_wedged_method(xe, DRM_WEDGE_RECOVERY_COLD_RESET);
+ xe_device_declare_wedged(xe);
+}
+
static u8 handle_soc_internal_errors(struct xe_device *xe, struct xe_ras_error_array *arr)
{
struct xe_ras_soc_error *info = (void *)arr->details;
@@ -267,7 +355,7 @@ static u8 handle_soc_internal_errors(struct xe_device *xe, struct xe_ras_error_a
xe_err(xe, "[RAS]: PUNIT %s detected: 0x%x\n",
sev_to_str(counter->common.severity),
ieh_error->global_error_status);
- /* TODO: Add PUNIT error handling */
+ punit_error_handler(xe);
return XE_RAS_RECOVERY_ACTION_DISCONNECT;
}
}
@@ -312,8 +400,10 @@ void xe_ras_counter_threshold_crossed(struct xe_device *xe,
struct xe_ras_threshold_crossed *pending = (void *)&response->data;
struct xe_ras_error_class *errors = pending->counters;
u32 id, ncounters = pending->ncounters;
+ u8 sent = 0;
BUILD_BUG_ON(sizeof(response->data) < sizeof(*pending));
+ BUILD_BUG_ON(BITS_PER_TYPE(sent) < XE_RAS_COMP_MAX);
xe_device_assert_mem_access(xe);
if (!ncounters || ncounters > XE_RAS_NUM_COUNTERS)
@@ -327,8 +417,18 @@ void xe_ras_counter_threshold_crossed(struct xe_device *xe,
severity = errors[id].common.severity;
component = errors[id].common.component;
+ if (!ras_counter_is_valid(xe, &errors[id]))
+ continue;
+
xe_warn(xe, "[RAS]: %s %s detected\n",
comp_to_str(component), sev_to_str(severity));
+
+ /* Send event once per component */
+ if (sent & BIT(component))
+ continue;
+ sent |= BIT(component);
+
+ ras_send_error_event(xe, severity, component);
}
}
@@ -358,6 +458,9 @@ static int get_counter(struct xe_device *xe, struct xe_ras_error_class *counter,
return -EIO;
}
+ if (!ras_counter_is_valid(xe, &response.counter))
+ return -EBADMSG;
+
common = &response.counter.common;
*value = response.value;
@@ -382,12 +485,20 @@ enum xe_ras_recovery_action xe_ras_process_errors(struct xe_device *xe)
enum xe_ras_recovery_action final_action;
u32 remaining = XE_SYSCTRL_FLOOD_LIMIT;
struct xe_ras_get_soc_error response;
+ u8 sent = 0;
size_t rlen;
int ret;
+ if (xe_fault_wedge_cold_reset()) {
+ xe_err(xe, "[RAS]: cold-reset wedge injected\n");
+ punit_error_handler(xe);
+ return XE_RAS_RECOVERY_ACTION_DISCONNECT;
+ }
+
if (!xe->info.has_sysctrl)
return XE_RAS_RECOVERY_ACTION_RESET;
+ BUILD_BUG_ON(BITS_PER_TYPE(sent) < XE_RAS_COMP_MAX);
/* Default action */
final_action = XE_RAS_RECOVERY_ACTION_RECOVERED;
@@ -422,9 +533,18 @@ enum xe_ras_recovery_action xe_ras_process_errors(struct xe_device *xe)
component = arr->counter.common.component;
severity = arr->counter.common.severity;
+ if (!ras_counter_is_valid(xe, &arr->counter))
+ continue;
+
xe_info(xe, "[RAS]: %s %s detected\n", comp_to_str(component),
sev_to_str(severity));
+ /* Send event once per component */
+ if (!(sent & BIT(component))) {
+ sent |= BIT(component);
+ ras_send_error_event(xe, severity, component);
+ }
+
switch (component) {
case XE_RAS_COMP_CORE_COMPUTE:
action = handle_core_compute_errors(arr);
@@ -532,6 +652,9 @@ int xe_ras_clear_counter(struct xe_device *xe, u8 severity, u8 component)
counter = &response.counter;
+ if (!ras_counter_is_valid(xe, counter))
+ return -EBADMSG;
+
xe_dbg(xe, "[RAS]: clear counter for %s %s\n", comp_to_str(counter->common.component),
sev_to_str(counter->common.severity));
diff --git a/drivers/gpu/drm/xe/xe_sriov_printk.h b/drivers/gpu/drm/xe/xe_sriov_printk.h
index 4c6b5c3d2190..5931c918655f 100644
--- a/drivers/gpu/drm/xe/xe_sriov_printk.h
+++ b/drivers/gpu/drm/xe/xe_sriov_printk.h
@@ -36,6 +36,9 @@
#define xe_sriov_dbg(xe, fmt, ...) \
xe_sriov_printk((xe), dbg, fmt, ##__VA_ARGS__)
+#define xe_sriov_dbg_ratelimited(xe, fmt, ...) \
+ xe_sriov_printk((xe), dbg_ratelimited, fmt, ##__VA_ARGS__)
+
/* for low level noisy debug messages */
#ifdef CONFIG_DRM_XE_DEBUG_SRIOV
#define xe_sriov_dbg_verbose(xe, fmt, ...) xe_sriov_dbg(xe, fmt, ##__VA_ARGS__)
diff --git a/drivers/gpu/drm/xe/xe_sriov_vf_ccs.c b/drivers/gpu/drm/xe/xe_sriov_vf_ccs.c
index a8c831fbee3b..a54138461f44 100644
--- a/drivers/gpu/drm/xe/xe_sriov_vf_ccs.c
+++ b/drivers/gpu/drm/xe/xe_sriov_vf_ccs.c
@@ -354,7 +354,6 @@ int xe_sriov_vf_ccs_init(struct xe_device *xe)
ctx->ctx_id = ctx_id;
flags = EXEC_QUEUE_FLAG_KERNEL |
- EXEC_QUEUE_FLAG_PERMANENT |
EXEC_QUEUE_FLAG_MIGRATE;
q = xe_exec_queue_create_bind(xe, tile, NULL, flags, 0);
if (IS_ERR(q)) {
diff --git a/drivers/gpu/drm/xe/xe_survivability_mode.c b/drivers/gpu/drm/xe/xe_survivability_mode.c
index 4c506027fa94..45d44ebe288b 100644
--- a/drivers/gpu/drm/xe/xe_survivability_mode.c
+++ b/drivers/gpu/drm/xe/xe_survivability_mode.c
@@ -14,9 +14,11 @@
#include "xe_device.h"
#include "xe_heci_gsc.h"
#include "xe_i2c.h"
+#include "xe_log.h"
#include "xe_mmio.h"
#include "xe_nvm.h"
#include "xe_pcode_api.h"
+#include "xe_printk.h"
#include "xe_vsec.h"
/**
@@ -172,18 +174,32 @@ static void populate_survivability_info(struct xe_device *xe)
}
}
-static void log_survivability_info(struct pci_dev *pdev)
+static const char *boot_status_str(u8 boot_status)
+{
+ switch (boot_status) {
+ case CRITICAL_FAILURE:
+ return "Critical Failure";
+ case NON_CRITICAL_FAILURE:
+ return "Non Critical Failure";
+ default:
+ return "Other";
+ }
+}
+
+static void log_survivability_info(struct xe_device *xe)
{
- struct xe_device *xe = pdev_to_xe_device(pdev);
struct xe_survivability *survivability = &xe->survivability;
u32 *info = survivability->info;
int id;
- dev_info(&pdev->dev, "Survivability Boot Status : Critical Failure (%d)\n",
- survivability->boot_status);
+ xe_log_info(xe, SURVIVABILITY, "Boot Status: %#x (%s)\n",
+ survivability->boot_status,
+ boot_status_str(survivability->boot_status));
+
for (id = 0; id < MAX_SCRATCH_REG; id++) {
- if (info[id])
- dev_info(&pdev->dev, "%s: 0x%x\n", reg_map[id], info[id]);
+ if (!info[id])
+ continue;
+ xe_log_info(xe, SURVIVABILITY, "%s: %#x\n", reg_map[id], info[id]);
}
}
@@ -289,41 +305,46 @@ static const struct attribute_group survivability_info_group = {
static int create_survivability_sysfs(struct pci_dev *pdev)
{
- struct device *dev = &pdev->dev;
+ /* Survivability info is required if not enabled via configfs */
+ bool needs_info = !xe_configfs_get_survivability_mode(pdev);
struct xe_device *xe = pdev_to_xe_device(pdev);
+ struct device *dev = &pdev->dev;
int ret;
ret = device_create_file(dev, &dev_attr_survivability_mode);
- if (ret) {
- dev_warn(dev, "Failed to create survivability sysfs files\n");
- return ret;
- }
+ if (ret)
+ goto failed;
ret = devm_add_action_or_reset(xe->drm.dev,
xe_survivability_mode_fini, xe);
if (ret)
- return ret;
+ goto failed;
- /* Survivability info is not required if enabled via configfs */
- if (!xe_configfs_get_survivability_mode(pdev)) {
+ if (needs_info) {
ret = devm_device_add_group(dev, &survivability_info_group);
if (ret)
- return ret;
+ goto failed;
}
return 0;
+
+failed:
+ xe_err(xe, "Failed to create survivability sysfs files: %pe\n", ERR_PTR(ret));
+ /* no sysfs, dump Survivability info to dmesg instead */
+ if (needs_info)
+ log_survivability_info(xe);
+ return ret;
}
static int enable_boot_survivability_mode(struct pci_dev *pdev)
{
- struct device *dev = &pdev->dev;
struct xe_device *xe = pdev_to_xe_device(pdev);
struct xe_survivability *survivability = &xe->survivability;
- int ret = 0;
+ int ret;
ret = create_survivability_sysfs(pdev);
if (ret)
- return ret;
+ goto failed;
/* Make sure xe_heci_gsc_init() and xe_i2c_probe() are aware of survivability */
survivability->mode = true;
@@ -335,19 +356,22 @@ static int enable_boot_survivability_mode(struct pci_dev *pdev)
if (survivability->fdo_mode) {
ret = xe_nvm_init(xe);
if (ret)
- goto err;
+ goto failed;
}
ret = xe_i2c_probe(xe);
if (ret)
- goto err;
+ goto failed;
- dev_err(dev, "In Survivability Mode\n");
+ if (check_boot_failure(xe))
+ xe_log_err_fatal(pdev, SURVIVABILITY, 0, "Boot Mode enabled!\n");
+ else
+ xe_log_info(pdev, SURVIVABILITY, "Boot Mode enabled!\n");
return 0;
-err:
- dev_err(dev, "Failed to enable Survivability Mode\n");
+failed:
+ xe_log_err_fatal(pdev, SURVIVABILITY, ret, "Failed to enable Boot Mode!\n");
survivability->mode = false;
return ret;
}
@@ -412,21 +436,22 @@ void xe_survivability_mode_runtime_enable(struct xe_device *xe)
struct pci_dev *pdev = to_pci_dev(xe->drm.dev);
if (!IS_DGFX(xe) || IS_SRIOV_VF(xe) || xe->info.platform < XE_BATTLEMAGE) {
- dev_err(&pdev->dev, "Runtime Survivability Mode not supported\n");
+ xe_log_err(xe, SURVIVABILITY, -EOPNOTSUPP, "Runtime Mode not supported!\n");
return;
}
populate_survivability_info(xe);
-
- if (create_survivability_sysfs(pdev))
- dev_err(&pdev->dev, "Failed to create survivability sysfs\n");
+ create_survivability_sysfs(pdev);
survivability->type = XE_SURVIVABILITY_TYPE_RUNTIME;
- dev_err(&pdev->dev, "Runtime Survivability mode enabled\n");
+ xe_log_err(xe, SURVIVABILITY, 0, "Runtime Mode enabled!\n");
xe_device_set_wedged_method(xe, DRM_WEDGE_RECOVERY_VENDOR);
xe_device_declare_wedged(xe);
- dev_err(&pdev->dev, "Firmware flash required, Please refer to the userspace documentation for more details!\n");
+
+ xe_log_err(xe, SURVIVABILITY, 0, "Firmware flash required!\n");
+ xe_info(xe, "Please refer to the userspace documentation for more details how to flash the firmware on %s!\n",
+ xe->info.platform_name);
}
/**
@@ -452,7 +477,7 @@ int xe_survivability_mode_boot_enable(struct xe_device *xe)
* v2 supports survivability mode for critical errors
*/
if (survivability->version < 2 && survivability->boot_status == CRITICAL_FAILURE) {
- log_survivability_info(pdev);
+ log_survivability_info(xe);
return -ENXIO;
}
diff --git a/drivers/gpu/drm/xe/xe_svm.c b/drivers/gpu/drm/xe/xe_svm.c
index b228a737cfd6..627a741293d5 100644
--- a/drivers/gpu/drm/xe/xe_svm.c
+++ b/drivers/gpu/drm/xe/xe_svm.c
@@ -15,6 +15,7 @@
#include "xe_gt_stats.h"
#include "xe_migrate.h"
#include "xe_module.h"
+#include "xe_pagefault.h"
#include "xe_pm.h"
#include "xe_pt.h"
#include "xe_svm.h"
@@ -115,6 +116,7 @@ xe_svm_range_alloc(struct drm_gpusvm *gpusvm)
return NULL;
INIT_LIST_HEAD(&range->garbage_collector_link);
+ mutex_init(&range->lock);
drm_gpusvm_init_pages(&range->pages, &gpusvm_to_vm(gpusvm)->xe->drm);
xe_vm_get(gpusvm_to_vm(gpusvm));
@@ -125,6 +127,7 @@ static void xe_svm_range_free(struct drm_gpusvm_range *range)
{
drm_gpusvm_free_pages(range->gpusvm, &(to_xe_range(range)->pages),
drm_gpusvm_range_size(range) >> PAGE_SHIFT);
+ mutex_destroy(&to_xe_range(range)->lock);
xe_vm_put(range_to_vm(range));
kfree(to_xe_range(range));
}
@@ -140,13 +143,13 @@ xe_svm_garbage_collector_add_range(struct xe_vm *vm, struct xe_svm_range *range,
drm_gpusvm_range_set_unmapped(&range->base, &range->pages, 1,
mmu_range);
- spin_lock(&vm->svm.garbage_collector.lock);
+ spin_lock(&vm->svm.garbage_collector.list_lock);
if (list_empty(&range->garbage_collector_link))
list_add_tail(&range->garbage_collector_link,
&vm->svm.garbage_collector.range_list);
- spin_unlock(&vm->svm.garbage_collector.lock);
+ spin_unlock(&vm->svm.garbage_collector.list_lock);
- queue_work(xe->usm.pf_wq, &vm->svm.garbage_collector.work);
+ queue_work(xe->usm.pagefault_wq, &vm->svm.garbage_collector.work);
}
static void xe_svm_tlb_inval_count_stats_incr(struct xe_gt *gt)
@@ -309,18 +312,30 @@ static int __xe_svm_garbage_collector(struct xe_vm *vm,
range_debug(range, "GARBAGE COLLECTOR");
- xe_vm_lock(vm, false);
- fence = xe_vm_range_unbind(vm, range);
- xe_vm_unlock(vm);
- if (IS_ERR(fence))
- return PTR_ERR(fence);
- dma_fence_put(fence);
+ scoped_guard(mutex, &range->lock) {
+ drm_gpusvm_range_get(&range->base);
+ range->removed = true;
+
+ range_debug(range, "GARBAGE COLLECTOR");
+
+ xe_vm_lock(vm, false);
+ fence = xe_vm_range_unbind(vm, range);
+ xe_vm_unlock(vm);
+ if (IS_ERR(fence)) {
+ drm_gpusvm_range_put(&range->base);
+ return PTR_ERR(fence);
+ }
+ dma_fence_put(fence);
+
+ drm_gpusvm_unmap_pages(&vm->svm.gpusvm, &range->pages,
+ drm_gpusvm_range_size(&range->base) >> PAGE_SHIFT,
+ &ctx);
- drm_gpusvm_unmap_pages(&vm->svm.gpusvm, &range->pages,
- drm_gpusvm_range_size(&range->base) >> PAGE_SHIFT,
- &ctx);
+ scoped_guard(mutex, &vm->svm.range_lock)
+ drm_gpusvm_range_remove(&vm->svm.gpusvm, &range->base);
+ }
- drm_gpusvm_range_remove(&vm->svm.gpusvm, &range->base);
+ drm_gpusvm_range_put(&range->base);
return 0;
}
@@ -393,13 +408,15 @@ static int xe_svm_garbage_collector(struct xe_vm *vm)
u64 range_end;
int err, ret = 0;
- lockdep_assert_held_write(&vm->lock);
+ lockdep_assert_held(&vm->lock);
if (xe_vm_is_closed_or_banned(vm))
return -ENOENT;
+ guard(mutex)(&vm->svm.garbage_collector.lock);
+
for (;;) {
- spin_lock(&vm->svm.garbage_collector.lock);
+ spin_lock(&vm->svm.garbage_collector.list_lock);
range = list_first_entry_or_null(&vm->svm.garbage_collector.range_list,
typeof(*range),
garbage_collector_link);
@@ -410,7 +427,7 @@ static int xe_svm_garbage_collector(struct xe_vm *vm)
range_end = xe_svm_range_end(range);
list_del(&range->garbage_collector_link);
- spin_unlock(&vm->svm.garbage_collector.lock);
+ spin_unlock(&vm->svm.garbage_collector.list_lock);
err = __xe_svm_garbage_collector(vm, range);
if (err) {
@@ -429,7 +446,7 @@ static int xe_svm_garbage_collector(struct xe_vm *vm)
return err;
}
}
- spin_unlock(&vm->svm.garbage_collector.lock);
+ spin_unlock(&vm->svm.garbage_collector.list_lock);
return ret;
}
@@ -439,9 +456,8 @@ static void xe_svm_garbage_collector_work_func(struct work_struct *w)
struct xe_vm *vm = container_of(w, struct xe_vm,
svm.garbage_collector.work);
- down_write(&vm->lock);
+ guard(rwsem_read)(&vm->lock);
xe_svm_garbage_collector(vm);
- up_write(&vm->lock);
}
#if IS_ENABLED(CONFIG_DRM_XE_PAGEMAP)
@@ -893,8 +909,11 @@ int xe_svm_init(struct xe_vm *vm)
{
int err;
+ mutex_init(&vm->svm.range_lock);
+ mutex_init(&vm->svm.garbage_collector.lock);
+
if (vm->flags & XE_VM_FLAG_FAULT_MODE) {
- spin_lock_init(&vm->svm.garbage_collector.lock);
+ spin_lock_init(&vm->svm.garbage_collector.list_lock);
INIT_LIST_HEAD(&vm->svm.garbage_collector.range_list);
INIT_WORK(&vm->svm.garbage_collector.work,
xe_svm_garbage_collector_work_func);
@@ -903,12 +922,12 @@ int xe_svm_init(struct xe_vm *vm)
err = drm_pagemap_acquire_owner(&vm->svm.peer, &xe_owner_list,
xe_has_interconnect);
if (err)
- return err;
+ goto out_err;
err = xe_svm_get_pagemaps(vm);
if (err) {
drm_pagemap_release_owner(&vm->svm.peer);
- return err;
+ goto out_err;
}
err = drm_gpusvm_init(&vm->svm.gpusvm, "Xe SVM",
@@ -916,19 +935,27 @@ int xe_svm_init(struct xe_vm *vm)
xe_modparam.svm_notifier_size * SZ_1M,
&gpusvm_ops, fault_chunk_sizes,
ARRAY_SIZE(fault_chunk_sizes));
- drm_gpusvm_driver_set_lock(&vm->svm.gpusvm, &vm->lock);
+ drm_gpusvm_driver_set_lock(&vm->svm.gpusvm, &vm->svm.range_lock);
if (err) {
xe_svm_put_pagemaps(vm);
drm_pagemap_release_owner(&vm->svm.peer);
- return err;
+ goto out_err;
}
} else {
err = drm_gpusvm_init(&vm->svm.gpusvm, "Xe SVM (simple)",
NULL, 0, 0, 0, NULL,
NULL, 0);
+ if (err)
+ goto out_err;
}
+ return 0;
+
+out_err:
+ mutex_destroy(&vm->svm.range_lock);
+ mutex_destroy(&vm->svm.garbage_collector.lock);
+
return err;
}
@@ -969,7 +996,10 @@ void xe_svm_fini(struct xe_vm *vm)
&ctx);
}
- drm_gpusvm_fini(&vm->svm.gpusvm);
+ scoped_guard(mutex, &vm->svm.range_lock)
+ drm_gpusvm_fini(&vm->svm.gpusvm);
+ mutex_destroy(&vm->svm.range_lock);
+ mutex_destroy(&vm->svm.garbage_collector.lock);
}
static bool xe_svm_range_has_pagemap_locked(const struct xe_svm_range *range,
@@ -1022,6 +1052,7 @@ void xe_svm_range_migrate_to_smem(struct xe_vm *vm, struct xe_svm_range *range)
* @tile_mask: Mask representing the tiles to be checked
* @dpagemap: if !%NULL, the range is expected to be present
* in device memory identified by this parameter.
+ * @valid_pages: Pages are valid, result written back to caller
*
* The xe_svm_range_validate() function checks if a range is
* valid and located in the desired memory region.
@@ -1030,7 +1061,8 @@ void xe_svm_range_migrate_to_smem(struct xe_vm *vm, struct xe_svm_range *range)
*/
bool xe_svm_range_validate(struct xe_vm *vm,
struct xe_svm_range *range,
- u8 tile_mask, const struct drm_pagemap *dpagemap)
+ u8 tile_mask, const struct drm_pagemap *dpagemap,
+ bool *valid_pages)
{
bool ret;
@@ -1042,6 +1074,8 @@ bool xe_svm_range_validate(struct xe_vm *vm,
else
ret = ret && !range->pages.dpagemap;
+ *valid_pages = xe_svm_range_pages_valid(range);
+
xe_svm_notifier_unlock(vm);
return ret;
@@ -1231,8 +1265,8 @@ DECL_SVM_RANGE_US_STATS(bind, BIND)
DECL_SVM_RANGE_US_STATS(fault, PAGEFAULT)
static int __xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
- struct xe_gt *gt, u64 fault_addr,
- bool need_vram)
+ struct xe_pagefault *pf, struct xe_gt *gt,
+ u64 fault_addr, bool need_vram)
{
int devmem_possible = IS_DGFX(vm->xe) &&
IS_ENABLED(CONFIG_DRM_XE_PAGEMAP);
@@ -1246,21 +1280,27 @@ static int __xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
};
struct xe_validation_ctx vctx;
struct drm_exec exec;
- struct xe_svm_range *range;
+ struct xe_svm_range *range = NULL;
struct drm_gpusvm_range_flags range_flags;
struct dma_fence *fence;
struct drm_pagemap *dpagemap;
struct xe_tile *tile = gt_to_tile(gt);
int migrate_try_count = ctx.devmem_only ? 3 : 1;
ktime_t start = xe_gt_stats_ktime_get(), bind_start, get_pages_start;
- int err;
+ int err = 0;
- lockdep_assert_held_write(&vm->lock);
+ lockdep_assert_held(&vm->lock);
xe_assert(vm->xe, xe_vma_is_cpu_addr_mirror(vma));
xe_gt_stats_incr(gt, XE_GT_STATS_ID_SVM_PAGEFAULT_COUNT, 1);
retry:
+ /* Release old range */
+ if (range) {
+ mutex_unlock(&range->lock);
+ drm_gpusvm_range_put(&range->base);
+ }
+
/* Always process UNMAPs first so view SVM ranges is current */
err = xe_svm_garbage_collector(vm);
if (err)
@@ -1276,10 +1316,17 @@ retry:
xe_svm_range_fault_count_stats_incr(gt, range);
+ mutex_lock(&range->lock);
+
+ if (xe_svm_range_is_removed(range))
+ goto retry;
+
/* READ_ONCE pairs with WRITE_ONCE in drm_gpusvm_range_set_unmapped() */
range_flags.__flags = READ_ONCE(range->base.flags.__flags);
- if (ctx.devmem_only && !range_flags.migrate_devmem)
- return -EACCES;
+ if (ctx.devmem_only && !range_flags.migrate_devmem) {
+ err = -EACCES;
+ goto err_out;
+ }
if (xe_svm_range_is_valid(range, tile, ctx.devmem_only, dpagemap)) {
xe_svm_range_valid_fault_count_stats_incr(gt, range);
@@ -1317,7 +1364,7 @@ retry:
drm_err(&vm->xe->drm,
"VRAM allocation failed, retry count exceeded, asid=%u, errno=%pe\n",
vm->usm.asid, ERR_PTR(err));
- return err;
+ goto err_out;
}
}
}
@@ -1344,7 +1391,7 @@ get_pages:
}
if (err) {
range_debug(range, "PAGE FAULT - FAIL PAGE COLLECT");
- goto out;
+ goto err_out;
} else if (IS_ENABLED(CONFIG_DRM_XE_DEBUG_VM)) {
drm_dbg(&vm->xe->drm, "After page collect data location is %sin \"%s\".\n",
xe_svm_range_has_pagemap(range, dpagemap) ? "" : "NOT ",
@@ -1378,7 +1425,13 @@ get_pages:
xe_svm_range_bind_us_stats_incr(gt, range, bind_start);
out:
+ /* Give hint to immediately ack faults */
+ xe_pagefault_set_start_addr(pf, xe_svm_range_start(range));
+ xe_pagefault_set_end_addr(pf, xe_svm_range_end(range));
+
xe_svm_range_fault_us_stats_incr(gt, range, start);
+ mutex_unlock(&range->lock);
+ drm_gpusvm_range_put(&range->base);
return 0;
err_out:
@@ -1388,6 +1441,9 @@ err_out:
goto retry;
}
+ mutex_unlock(&range->lock);
+ drm_gpusvm_range_put(&range->base);
+
return err;
}
@@ -1395,6 +1451,7 @@ err_out:
* xe_svm_handle_pagefault() - SVM handle page fault
* @vm: The VM.
* @vma: The CPU address mirror VMA.
+ * @pf: Pagefault structure
* @gt: The gt upon the fault occurred.
* @fault_addr: The GPU fault address.
* @atomic: The fault atomic access bit.
@@ -1405,8 +1462,8 @@ err_out:
* Return: 0 on success, negative error code on error.
*/
int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
- struct xe_gt *gt, u64 fault_addr,
- bool atomic)
+ struct xe_pagefault *pf, struct xe_gt *gt,
+ u64 fault_addr, bool atomic)
{
int need_vram, ret;
retry:
@@ -1414,7 +1471,7 @@ retry:
if (need_vram < 0)
return need_vram;
- ret = __xe_svm_handle_pagefault(vm, vma, gt, fault_addr,
+ ret = __xe_svm_handle_pagefault(vm, vma, pf, gt, fault_addr,
need_vram ? true : false);
if (ret == -EAGAIN) {
/*
@@ -1470,9 +1527,9 @@ void xe_svm_unmap_address_range(struct xe_vm *vm, u64 start, u64 end)
drm_gpusvm_range_get(range);
__xe_svm_garbage_collector(vm, to_xe_range(range));
if (!list_empty(&to_xe_range(range)->garbage_collector_link)) {
- spin_lock(&vm->svm.garbage_collector.lock);
+ spin_lock(&vm->svm.garbage_collector.list_lock);
list_del(&to_xe_range(range)->garbage_collector_link);
- spin_unlock(&vm->svm.garbage_collector.lock);
+ spin_unlock(&vm->svm.garbage_collector.list_lock);
}
drm_gpusvm_range_put(range);
}
@@ -1502,7 +1559,7 @@ int xe_svm_bo_evict(struct xe_bo *bo)
* @ctx: GPU SVM context
*
* This function finds or inserts a newly allocated a SVM range based on the
- * address.
+ * address. Take a reference to SVM range on success.
*
* Return: Pointer to the SVM range on success, ERR_PTR() on failure.
*/
@@ -1511,11 +1568,15 @@ struct xe_svm_range *xe_svm_range_find_or_insert(struct xe_vm *vm, u64 addr,
{
struct drm_gpusvm_range *r;
+ guard(mutex)(&vm->svm.range_lock);
+
r = drm_gpusvm_range_find_or_insert(&vm->svm.gpusvm, max(addr, xe_vma_start(vma)),
xe_vma_start(vma), xe_vma_end(vma), ctx);
if (IS_ERR(r))
return ERR_CAST(r);
+ drm_gpusvm_range_get(r);
+
return to_xe_range(r);
}
@@ -1535,6 +1596,8 @@ int xe_svm_range_get_pages(struct xe_vm *vm, struct xe_svm_range *range,
{
int err = 0;
+ lockdep_assert_held(&range->lock);
+
err = drm_gpusvm_get_pages(&vm->svm.gpusvm, &range->pages,
vm->svm.gpusvm.mm,
&range->base.notifier->notifier,
@@ -1659,6 +1722,7 @@ int xe_svm_alloc_vram(struct xe_svm_range *range, const struct drm_gpusvm_ctx *c
.__flags = READ_ONCE(range->base.flags.__flags),
};
+ lockdep_assert_held(&range->lock);
xe_assert(range_to_vm(&range->base)->xe, flags.migrate_devmem);
range_debug(range, "ALLOCATE VRAM");
diff --git a/drivers/gpu/drm/xe/xe_svm.h b/drivers/gpu/drm/xe/xe_svm.h
index a921556d3466..2a0dc0d125c9 100644
--- a/drivers/gpu/drm/xe/xe_svm.h
+++ b/drivers/gpu/drm/xe/xe_svm.h
@@ -21,6 +21,7 @@ struct drm_file;
struct xe_bo;
struct xe_gt;
struct xe_device;
+struct xe_pagefault;
struct xe_vram_region;
struct xe_tile;
struct xe_vm;
@@ -39,6 +40,13 @@ struct xe_svm_range {
*/
struct list_head garbage_collector_link;
/**
+ * @lock: Protects fault handler, garbage collector, and prefetch
+ * critical sections, ensuring only one thread operates on a range at a
+ * time. Locking order: inside vm->lock and garbage collector, outside
+ * dma-resv locks, vm->svm.range_lock.
+ */
+ struct mutex lock;
+ /**
* @tile_present: Tile mask of binding is present for this range.
* Protected by GPU SVM notifier lock.
*/
@@ -48,9 +56,23 @@ struct xe_svm_range {
* range. Protected by GPU SVM notifier lock.
*/
u8 tile_invalidated;
+ /**
+ * @removed: Range has been removed from GPU SVM tree, protected by
+ * @lock.
+ */
+ bool removed;
};
/**
+ * xe_svm_range_put() - SVM range put
+ * @range: SVM range
+ */
+static inline void xe_svm_range_put(struct xe_svm_range *range)
+{
+ drm_gpusvm_range_put(&range->base);
+}
+
+/**
* struct xe_pagemap - Manages xe device_private memory for SVM.
* @pagemap: The struct dev_pagemap providing the struct pages.
* @dpagemap: The drm_pagemap managing allocation and migration.
@@ -88,8 +110,8 @@ void xe_svm_fini(struct xe_vm *vm);
void xe_svm_close(struct xe_vm *vm);
int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
- struct xe_gt *gt, u64 fault_addr,
- bool atomic);
+ struct xe_pagefault *pf, struct xe_gt *gt,
+ u64 fault_addr, bool atomic);
bool xe_svm_has_mapping(struct xe_vm *vm, u64 start, u64 end);
@@ -113,7 +135,8 @@ void xe_svm_range_migrate_to_smem(struct xe_vm *vm, struct xe_svm_range *range);
bool xe_svm_range_validate(struct xe_vm *vm,
struct xe_svm_range *range,
- u8 tile_mask, const struct drm_pagemap *dpagemap);
+ u8 tile_mask, const struct drm_pagemap *dpagemap,
+ bool *valid_pages);
u64 xe_svm_find_vma_start(struct xe_vm *vm, u64 addr, u64 end, struct xe_vma *vma);
@@ -138,6 +161,19 @@ static inline bool xe_svm_range_has_dma_mapping(struct xe_svm_range *range)
}
/**
+ * xe_svm_range_is_removed() - SVM range is removed from GPU SVM tree
+ * @range: SVM range
+ *
+ * Return: True if SVM range is removed from GPU SVM tree, False otherwise
+ */
+static inline bool xe_svm_range_is_removed(struct xe_svm_range *range)
+{
+ lockdep_assert_held(&range->lock);
+
+ return range->removed;
+}
+
+/**
* to_xe_range - Convert a drm_gpusvm_range pointer to a xe_svm_range
* @r: Pointer to the drm_gpusvm_range structure
*
@@ -216,10 +252,15 @@ struct xe_svm_range {
struct {
const struct drm_pagemap_addr *dma_addr;
} pages;
+ struct mutex lock;
u32 tile_present;
u32 tile_invalidated;
};
+static inline void xe_svm_range_put(struct xe_svm_range *range)
+{
+}
+
static inline bool xe_svm_range_pages_valid(struct xe_svm_range *range)
{
return false;
@@ -258,8 +299,8 @@ void xe_svm_close(struct xe_vm *vm)
static inline
int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
- struct xe_gt *gt, u64 fault_addr,
- bool atomic)
+ struct xe_pagefault *pf, struct xe_gt *gt,
+ u64 fault_addr, bool atomic)
{
return 0;
}
@@ -337,7 +378,8 @@ void xe_svm_range_migrate_to_smem(struct xe_vm *vm, struct xe_svm_range *range)
static inline
bool xe_svm_range_validate(struct xe_vm *vm,
struct xe_svm_range *range,
- u8 tile_mask, bool devmem_preferred)
+ u8 tile_mask, const struct drm_pagemap *dpagemap,
+ bool *valid_pages)
{
return false;
}
@@ -389,6 +431,11 @@ static inline struct drm_pagemap *xe_drm_pagemap_from_fd(int fd, u32 region_inst
return ERR_PTR(-ENOENT);
}
+static inline bool xe_svm_range_is_removed(struct xe_svm_range *range)
+{
+ return false;
+}
+
#define xe_svm_range_has_dma_mapping(...) false
#endif /* CONFIG_DRM_XE_GPUSVM */
diff --git a/drivers/gpu/drm/xe/xe_tile_printk.h b/drivers/gpu/drm/xe/xe_tile_printk.h
index 738433a764bd..c87c07e14fd1 100644
--- a/drivers/gpu/drm/xe/xe_tile_printk.h
+++ b/drivers/gpu/drm/xe/xe_tile_printk.h
@@ -34,6 +34,9 @@
#define xe_tile_dbg(_tile, _fmt, ...) \
xe_tile_printk((_tile), dbg, _fmt, ##__VA_ARGS__)
+#define xe_tile_dbg_ratelimited(_tile, _fmt, ...) \
+ xe_tile_printk((_tile), dbg_ratelimited, _fmt, ##__VA_ARGS__)
+
#define xe_tile_WARN_type(_tile, _type, _condition, _fmt, ...) \
xe_WARN##_type((_tile)->xe, _condition, _fmt, ## __VA_ARGS__)
diff --git a/drivers/gpu/drm/xe/xe_tile_sriov_printk.h b/drivers/gpu/drm/xe/xe_tile_sriov_printk.h
index 68323512872c..c23e698bc8db 100644
--- a/drivers/gpu/drm/xe/xe_tile_sriov_printk.h
+++ b/drivers/gpu/drm/xe/xe_tile_sriov_printk.h
@@ -27,6 +27,9 @@
#define xe_tile_sriov_dbg(_tile, _fmt, ...) \
xe_tile_sriov_printk(_tile, dbg, _fmt, ##__VA_ARGS__)
+#define xe_tile_sriov_dbg_ratelimited(_tile, _fmt, ...) \
+ xe_tile_sriov_printk(_tile, dbg_ratelimited, _fmt, ##__VA_ARGS__)
+
#define xe_tile_sriov_dbg_verbose(_tile, _fmt, ...) \
xe_tile_sriov_printk(_tile, dbg_verbose, _fmt, ##__VA_ARGS__)
diff --git a/drivers/gpu/drm/xe/xe_userptr.c b/drivers/gpu/drm/xe/xe_userptr.c
index 8b2d461ea0b2..90ac141fc12d 100644
--- a/drivers/gpu/drm/xe/xe_userptr.c
+++ b/drivers/gpu/drm/xe/xe_userptr.c
@@ -57,6 +57,23 @@ int __xe_vm_userptr_needs_repin(struct xe_vm *vm)
list_empty(&vm->userptr.invalidated)) ? 0 : -EAGAIN;
}
+#if IS_ENABLED(CONFIG_PROVE_LOCKING)
+static bool __xe_vma_userptr_lockdep(struct xe_userptr_vma *uvma)
+{
+ struct xe_vma *vma = &uvma->vma;
+ struct xe_vm *vm = xe_vma_vm(vma);
+
+ return lockdep_is_held_type(&vm->lock, 0) ||
+ (lockdep_is_held_type(&vm->lock, 1) &&
+ lockdep_is_held_type(&vma->fault_lock, 0));
+}
+
+#define xe_vma_userptr_lockdep(uvma) \
+ lockdep_assert(__xe_vma_userptr_lockdep(uvma))
+#else
+#define xe_vma_userptr_lockdep(uvma)
+#endif
+
int xe_vma_userptr_pin_pages(struct xe_userptr_vma *uvma)
{
struct xe_vma *vma = &uvma->vma;
@@ -68,7 +85,7 @@ int xe_vma_userptr_pin_pages(struct xe_userptr_vma *uvma)
.allow_mixed = true,
};
- lockdep_assert_held(&vm->lock);
+ xe_vma_userptr_lockdep(uvma);
xe_assert(xe, xe_vma_is_userptr(vma));
if (vma->gpuva.flags & XE_VMA_DESTROYED)
diff --git a/drivers/gpu/drm/xe/xe_vm.c b/drivers/gpu/drm/xe/xe_vm.c
index 9e0176861cb6..19b3d0be7928 100644
--- a/drivers/gpu/drm/xe/xe_vm.c
+++ b/drivers/gpu/drm/xe/xe_vm.c
@@ -615,10 +615,13 @@ void xe_vm_add_fault_entry_pf(struct xe_vm *vm, struct xe_pagefault *pf)
{
struct xe_vm_fault_entry *e;
struct xe_hw_engine *hwe;
+ u8 engine_class = FIELD_GET(XE_PAGEFAULT_ENGINE_CLASS_MASK,
+ pf->consumer.engine_class_instance);
+ u8 engine_instance = FIELD_GET(XE_PAGEFAULT_ENGINE_INSTANCE_MASK,
+ pf->consumer.engine_class_instance);
/* Do not report faults on reserved engines */
- hwe = xe_gt_hw_engine(pf->gt, pf->consumer.engine_class,
- pf->consumer.engine_instance, false);
+ hwe = xe_gt_hw_engine(pf->gt, engine_class, engine_instance, false);
if (!hwe || xe_hw_engine_is_reserved(hwe))
return;
@@ -689,6 +692,17 @@ static int xe_vma_ops_alloc(struct xe_vma_ops *vops, bool array_of_binds)
}
ALLOW_ERROR_INJECTION(xe_vma_ops_alloc, ERRNO);
+static void xe_vma_svm_prefetch_ranges_fini(struct xe_vma_op *op)
+{
+ struct xe_svm_range *svm_range;
+ unsigned long i;
+
+ xa_for_each(&op->prefetch_range.range, i, svm_range)
+ xe_svm_range_put(svm_range);
+
+ xa_destroy(&op->prefetch_range.range);
+}
+
static void xe_vma_svm_prefetch_op_fini(struct xe_vma_op *op)
{
struct xe_vma *vma;
@@ -696,7 +710,7 @@ static void xe_vma_svm_prefetch_op_fini(struct xe_vma_op *op)
vma = gpuva_to_vma(op->base.prefetch.va);
if (op->base.op == DRM_GPUVA_OP_PREFETCH && xe_vma_is_cpu_addr_mirror(vma))
- xa_destroy(&op->prefetch_range.range);
+ xe_vma_svm_prefetch_ranges_fini(op);
}
static void xe_vma_svm_prefetch_ops_fini(struct xe_vma_ops *vops)
@@ -930,6 +944,7 @@ struct dma_fence *xe_vm_range_rebind(struct xe_vm *vm,
u8 id;
int err;
+ lockdep_assert_held(&range->lock);
lockdep_assert_held(&vm->lock);
xe_vm_assert_held(vm);
xe_assert(vm->xe, xe_vm_in_fault_mode(vm));
@@ -1012,6 +1027,7 @@ struct dma_fence *xe_vm_range_unbind(struct xe_vm *vm,
u8 id;
int err;
+ lockdep_assert_held(&range->lock);
lockdep_assert_held(&vm->lock);
xe_vm_assert_held(vm);
xe_assert(vm->xe, xe_vm_in_fault_mode(vm));
@@ -1187,6 +1203,8 @@ static struct xe_vma *xe_vma_create(struct xe_vm *vm,
xe_vm_get(vm);
}
+ mutex_init(&vma->fault_lock);
+
return vma;
}
@@ -1211,6 +1229,7 @@ static void xe_vma_destroy_late(struct xe_vma *vma)
xe_bo_put(bo);
}
+ mutex_destroy(&vma->fault_lock);
xe_vma_free(vma);
}
@@ -1231,12 +1250,19 @@ static void vma_destroy_cb(struct dma_fence *fence,
queue_work(system_dfl_wq, &vma->destroy_work);
}
+static void xe_vm_assert_write_mode_or_garbage_collector(struct xe_vm *vm)
+{
+ lockdep_assert(lockdep_is_held_type(&vm->lock, 0) ||
+ (lockdep_is_held_type(&vm->lock, 1) &&
+ lockdep_is_held_type(&vm->svm.garbage_collector.lock, 0)));
+}
+
static void xe_vma_destroy(struct xe_vma *vma, struct dma_fence *fence)
{
struct xe_vm *vm = xe_vma_vm(vma);
struct xe_bo *bo = xe_vma_bo(vma);
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
xe_assert(vm->xe, list_empty(&vma->combined_links.destroy));
if (xe_vma_is_userptr(vma)) {
@@ -1320,7 +1346,9 @@ xe_vm_find_overlapping_vma(struct xe_vm *vm, u64 start, u64 range)
xe_assert(vm->xe, start + range <= vm->size);
+ mutex_lock(&vm->snap_mutex);
gpuva = drm_gpuva_find_first(&vm->gpuvm, start, range);
+ mutex_unlock(&vm->snap_mutex);
return gpuva ? gpuva_to_vma(gpuva) : NULL;
}
@@ -1330,7 +1358,7 @@ static int xe_vm_insert_vma(struct xe_vm *vm, struct xe_vma *vma)
int err;
xe_assert(vm->xe, xe_vma_vm(vma) == vm);
- lockdep_assert_held(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
mutex_lock(&vm->snap_mutex);
err = drm_gpuva_insert(&vm->gpuvm, &vma->gpuva);
@@ -1343,13 +1371,11 @@ static int xe_vm_insert_vma(struct xe_vm *vm, struct xe_vma *vma)
static void xe_vm_remove_vma(struct xe_vm *vm, struct xe_vma *vma)
{
xe_assert(vm->xe, xe_vma_vm(vma) == vm);
- lockdep_assert_held(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
mutex_lock(&vm->snap_mutex);
drm_gpuva_remove(&vma->gpuva);
mutex_unlock(&vm->snap_mutex);
- if (vm->usm.last_fault_vma == vma)
- vm->usm.last_fault_vma = NULL;
}
static struct drm_gpuva_op *xe_vm_op_alloc(void)
@@ -2185,7 +2211,7 @@ static int xe_vm_query_vmas(struct xe_vm *vm, u64 start, u64 end)
struct drm_gpuva *gpuva;
u32 num_vmas = 0;
- lockdep_assert_held(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
drm_gpuvm_for_each_va_range(gpuva, &vm->gpuvm, start, end)
num_vmas++;
@@ -2198,7 +2224,7 @@ static int get_mem_attrs(struct xe_vm *vm, u32 *num_vmas, u64 start,
struct drm_gpuva *gpuva;
int i = 0;
- lockdep_assert_held(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
drm_gpuvm_for_each_va_range(gpuva, &vm->gpuvm, start, end) {
struct xe_vma *vma = gpuva_to_vma(gpuva);
@@ -2244,7 +2270,7 @@ int xe_vm_query_vmas_attrs_ioctl(struct drm_device *dev, void *data, struct drm_
if (XE_IOCTL_DBG(xe, !vm))
return -EINVAL;
- err = down_read_interruptible(&vm->lock);
+ err = down_write_killable(&vm->lock);
if (err)
goto put_vm;
@@ -2278,21 +2304,12 @@ int xe_vm_query_vmas_attrs_ioctl(struct drm_device *dev, void *data, struct drm_
free_mem_attrs:
kvfree(mem_attrs);
unlock_vm:
- up_read(&vm->lock);
+ up_write(&vm->lock);
put_vm:
xe_vm_put(vm);
return err;
}
-static bool vma_matches(struct xe_vma *vma, u64 page_addr)
-{
- if (page_addr > xe_vma_end(vma) - 1 ||
- page_addr + SZ_4K - 1 < xe_vma_start(vma))
- return false;
-
- return true;
-}
-
/**
* xe_vm_find_vma_by_addr() - Find a VMA by its address
*
@@ -2301,16 +2318,7 @@ static bool vma_matches(struct xe_vma *vma, u64 page_addr)
*/
struct xe_vma *xe_vm_find_vma_by_addr(struct xe_vm *vm, u64 page_addr)
{
- struct xe_vma *vma = NULL;
-
- if (vm->usm.last_fault_vma) { /* Fast lookup */
- if (vma_matches(vm->usm.last_fault_vma, page_addr))
- vma = vm->usm.last_fault_vma;
- }
- if (!vma)
- vma = xe_vm_find_overlapping_vma(vm, page_addr, SZ_4K);
-
- return vma;
+ return xe_vm_find_overlapping_vma(vm, page_addr, SZ_4K);
}
static const u32 region_to_mem_type[] = {
@@ -2423,7 +2431,7 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
u64 range_end = addr + range;
int err;
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
vm_dbg(&vm->xe->drm,
"op=%d, addr=0x%016llx, range=0x%016llx, bo_offset_or_userptr=0x%016llx",
@@ -2446,10 +2454,12 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
.map.gem.offset = bo_offset_or_userptr,
};
+ vops->flags |= XE_VMA_OPS_FLAG_MODIFIES_GPUVA;
ops = drm_gpuvm_sm_map_ops_create(&vm->gpuvm, &map_req);
break;
}
case DRM_XE_VM_BIND_OP_UNMAP:
+ vops->flags |= XE_VMA_OPS_FLAG_MODIFIES_GPUVA;
ops = drm_gpuvm_sm_unmap_ops_create(&vm->gpuvm, addr, range);
break;
case DRM_XE_VM_BIND_OP_PREFETCH:
@@ -2458,6 +2468,7 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
case DRM_XE_VM_BIND_OP_UNMAP_ALL:
xe_assert(vm->xe, bo);
+ vops->flags |= XE_VMA_OPS_FLAG_MODIFIES_GPUVA;
err = xe_bo_lock(bo, true);
if (err)
return ERR_PTR(err);
@@ -2479,6 +2490,16 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
if (IS_ERR(ops))
return ops;
+ /* Setup safe unwind */
+ drm_gpuva_for_each_op(__op, ops) {
+ struct xe_vma_op *op = gpuva_op_to_vma_op(__op);
+
+ if (__op->op == DRM_GPUVA_OP_PREFETCH) {
+ xa_init_flags(&op->prefetch_range.range, XA_FLAGS_ALLOC);
+ op->prefetch_range.ranges_count = 0;
+ }
+ }
+
drm_gpuva_for_each_op(__op, ops) {
struct xe_vma_op *op = gpuva_op_to_vma_op(__op);
@@ -2507,10 +2528,14 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
struct drm_pagemap *dpagemap = NULL;
u8 id, tile_mask = 0;
u32 i;
+ bool need_put, valid_pages;
+
+ if (xe_vma_is_userptr(vma))
+ vops->flags |= XE_VMA_OPS_FLAG_MODIFIES_GPUVA;
if (!xe_vma_is_cpu_addr_mirror(vma)) {
op->prefetch.region = prefetch_region;
- break;
+ continue;
}
ctx.read_only = xe_vma_read_only(vma);
@@ -2520,9 +2545,6 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
for_each_tile(tile, vm->xe, id)
tile_mask |= 0x1 << id;
- xa_init_flags(&op->prefetch_range.range, XA_FLAGS_ALLOC);
- op->prefetch_range.ranges_count = 0;
-
if (prefetch_region == DRM_XE_CONSULT_MEM_ADVISE_PREF_LOC) {
dpagemap = xe_vma_resolve_pagemap(vma,
xe_device_get_root_tile(vm->xe));
@@ -2534,6 +2556,7 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
op->prefetch_range.dpagemap = dpagemap;
alloc_next_range:
+ need_put = false;
svm_range = xe_svm_range_find_or_insert(vm, addr, vma, &ctx);
if (PTR_ERR(svm_range) == -ENOENT) {
@@ -2551,8 +2574,11 @@ alloc_next_range:
goto unwind_prefetch_ops;
}
- if (xe_svm_range_validate(vm, svm_range, tile_mask, dpagemap)) {
+ if (xe_svm_range_validate(vm, svm_range, tile_mask,
+ dpagemap, &valid_pages)) {
xe_svm_range_debug(svm_range, "PREFETCH - RANGE IS VALID");
+ xe_assert(vm->xe, valid_pages);
+ need_put = true;
goto check_next_range;
}
@@ -2560,18 +2586,27 @@ alloc_next_range:
&i, svm_range, xa_limit_32b,
GFP_KERNEL);
- if (err)
+ if (err) {
+ xe_svm_range_put(svm_range);
goto unwind_prefetch_ops;
+ }
op->prefetch_range.ranges_count++;
vops->flags |= XE_VMA_OPS_FLAG_HAS_SVM_PREFETCH;
+ if (valid_pages)
+ vops->flags |= XE_VMA_OPS_FLAG_HAS_SVM_VALID_RANGE;
xe_svm_range_debug(svm_range, "PREFETCH - RANGE CREATED");
check_next_range:
if (range_end > xe_svm_range_end(svm_range) &&
xe_svm_range_end(svm_range) < xe_vma_end(vma)) {
addr = xe_svm_range_end(svm_range);
+ if (need_put)
+ xe_svm_range_put(svm_range);
goto alloc_next_range;
}
+ if (need_put)
+ xe_svm_range_put(svm_range);
+
}
print_op_label:
print_op(vm->xe, __op);
@@ -2596,7 +2631,7 @@ static struct xe_vma *new_vma(struct xe_vm *vm, struct drm_gpuva_op_map *op,
struct xe_vma *vma;
int err = 0;
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
if (bo) {
err = 0;
@@ -2693,10 +2728,12 @@ static int xe_vma_op_commit(struct xe_vm *vm, struct xe_vma_op *op)
{
int err = 0;
- lockdep_assert_held_write(&vm->lock);
+ lockdep_assert_held(&vm->lock);
switch (op->base.op) {
case DRM_GPUVA_OP_MAP:
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
+
err |= xe_vm_insert_vma(vm, op->map.vma);
if (!err)
op->flags |= XE_VMA_OP_COMMITTED;
@@ -2706,6 +2743,8 @@ static int xe_vma_op_commit(struct xe_vm *vm, struct xe_vma_op *op)
u8 tile_present =
gpuva_to_vma(op->base.remap.unmap->va)->tile_present;
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
+
prep_vma_destroy(vm, gpuva_to_vma(op->base.remap.unmap->va),
true);
op->flags |= XE_VMA_OP_COMMITTED;
@@ -2740,6 +2779,8 @@ static int xe_vma_op_commit(struct xe_vm *vm, struct xe_vma_op *op)
break;
}
case DRM_GPUVA_OP_UNMAP:
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
+
prep_vma_destroy(vm, gpuva_to_vma(op->base.unmap.va), true);
op->flags |= XE_VMA_OP_COMMITTED;
break;
@@ -2785,7 +2826,7 @@ static int vm_bind_ioctl_ops_parse(struct xe_vm *vm, struct drm_gpuva_ops *ops,
u8 id, tile_mask = 0;
int err = 0;
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
for_each_tile(tile, vm->xe, id)
tile_mask |= 0x1 << id;
@@ -2964,10 +3005,12 @@ static void xe_vma_op_unwind(struct xe_vm *vm, struct xe_vma_op *op,
bool post_commit, bool prev_post_commit,
bool next_post_commit)
{
- lockdep_assert_held_write(&vm->lock);
+ lockdep_assert_held(&vm->lock);
switch (op->base.op) {
case DRM_GPUVA_OP_MAP:
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
+
if (op->map.vma) {
prep_vma_destroy(vm, op->map.vma, post_commit);
xe_vma_destroy_unlocked(op->map.vma);
@@ -2977,6 +3020,8 @@ static void xe_vma_op_unwind(struct xe_vm *vm, struct xe_vma_op *op,
{
struct xe_vma *vma = gpuva_to_vma(op->base.unmap.va);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
+
if (vma) {
xe_svm_notifier_lock(vm);
vma->gpuva.flags &= ~XE_VMA_DESTROYED;
@@ -2990,6 +3035,8 @@ static void xe_vma_op_unwind(struct xe_vm *vm, struct xe_vma_op *op,
{
struct xe_vma *vma = gpuva_to_vma(op->base.remap.unmap->va);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
+
if (op->remap.prev) {
prep_vma_destroy(vm, op->remap.prev, prev_post_commit);
xe_vma_destroy_unlocked(op->remap.prev);
@@ -3118,16 +3165,87 @@ static int check_ufence(struct xe_vma *vma)
return 0;
}
-static int prefetch_ranges(struct xe_vm *vm, struct xe_vma_op *op)
+struct prefetch_thread {
+ struct work_struct work;
+ struct drm_gpusvm_ctx *ctx;
+ struct xe_vma *vma;
+ struct xe_svm_range *svm_range;
+ struct drm_pagemap *dpagemap;
+ int err;
+};
+
+static void prefetch_thread_func(struct prefetch_thread *thread)
+{
+ struct xe_vma *vma = thread->vma;
+ struct xe_vm *vm = xe_vma_vm(vma);
+ struct xe_svm_range *svm_range = thread->svm_range;
+ struct drm_pagemap *dpagemap = thread->dpagemap;
+ int err = 0;
+
+ guard(mutex)(&svm_range->lock);
+
+ if (xe_svm_range_is_removed(svm_range))
+ return;
+
+ if (!dpagemap)
+ xe_svm_range_migrate_to_smem(vm, svm_range);
+
+ if (IS_ENABLED(CONFIG_DRM_XE_DEBUG_VM)) {
+ drm_dbg(&vm->xe->drm,
+ "Prefetch pagemap is %s start 0x%016lx end 0x%016lx\n",
+ dpagemap ? dpagemap->drm->unique : "system",
+ xe_svm_range_start(svm_range), xe_svm_range_end(svm_range));
+ }
+
+ if (xe_svm_range_needs_migrate_to_vram(svm_range, vma, dpagemap)) {
+ err = xe_svm_alloc_vram(svm_range, thread->ctx, dpagemap);
+ if (err) {
+ drm_dbg(&vm->xe->drm, "VRAM allocation failed, retry from userspace, asid=%u, gpusvm=%p, errno=%pe\n",
+ vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err));
+ /*
+ * We intentionally return -ENODATA on any races to
+ * commit any VMA updates from other ops without
+ * updating any page tables deferring to page faults to
+ * page updates skipped in the IOCTL.
+ */
+ thread->err = -ENODATA;
+ return;
+ }
+ xe_svm_range_debug(svm_range, "PREFETCH - RANGE MIGRATED TO VRAM");
+ }
+
+ err = xe_svm_range_get_pages(vm, svm_range, thread->ctx);
+ if (err) {
+ drm_dbg(&vm->xe->drm, "Get pages failed, asid=%u, gpusvm=%p, errno=%pe\n",
+ vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err));
+ if (err == -EOPNOTSUPP || err == -EFAULT || err == -EPERM)
+ err = -ENODATA;
+ thread->err = err;
+ return;
+ }
+ xe_svm_range_debug(svm_range, "PREFETCH - RANGE GET PAGES DONE");
+}
+
+static void prefetch_work_func(struct work_struct *w)
+{
+ struct prefetch_thread *thread =
+ container_of(w, struct prefetch_thread, work);
+
+ prefetch_thread_func(thread);
+}
+
+static int prefetch_ranges(struct xe_vm *vm, struct xe_vma_ops *vops,
+ struct xe_vma_op *op)
{
bool devmem_possible = IS_DGFX(vm->xe) && IS_ENABLED(CONFIG_DRM_XE_PAGEMAP);
struct xe_vma *vma = gpuva_to_vma(op->base.prefetch.va);
struct drm_pagemap *dpagemap = op->prefetch_range.dpagemap;
- int err = 0;
-
struct xe_svm_range *svm_range;
struct drm_gpusvm_ctx ctx = {};
+ struct prefetch_thread stack_thread, *thread, *prefetches;
unsigned long i;
+ int err = 0, idx = 0;
+ bool skip_threads;
if (!xe_vma_is_cpu_addr_mirror(vma))
return 0;
@@ -3137,37 +3255,50 @@ static int prefetch_ranges(struct xe_vm *vm, struct xe_vma_op *op)
ctx.check_pages_threshold = devmem_possible ? SZ_64K : 0;
ctx.device_private_page_owner = xe_svm_private_page_owner(vm, !dpagemap);
- /* TODO: Threading the migration */
+ skip_threads = op->prefetch_range.ranges_count == 1 ||
+ (!dpagemap && !(vops->flags &
+ XE_VMA_OPS_FLAG_HAS_SVM_VALID_RANGE)) ||
+ !(vops->flags & XE_VMA_OPS_FLAG_DOWNGRADE_LOCK) ||
+ vm->xe->info.num_pf_work == 1;
+ thread = skip_threads ? &stack_thread : NULL;
+
+ if (!skip_threads) {
+ prefetches = kvmalloc_array(op->prefetch_range.ranges_count,
+ sizeof(*prefetches), GFP_KERNEL);
+ if (!prefetches)
+ return -ENOMEM;
+ }
+
xa_for_each(&op->prefetch_range.range, i, svm_range) {
- if (!dpagemap)
- xe_svm_range_migrate_to_smem(vm, svm_range);
-
- if (IS_ENABLED(CONFIG_DRM_XE_DEBUG_VM)) {
- drm_dbg(&vm->xe->drm,
- "Prefetch pagemap is %s start 0x%016lx end 0x%016lx\n",
- dpagemap ? dpagemap->drm->unique : "system",
- xe_svm_range_start(svm_range), xe_svm_range_end(svm_range));
+ if (!skip_threads) {
+ thread = prefetches + idx++;
+ INIT_WORK(&thread->work, prefetch_work_func);
}
- if (xe_svm_range_needs_migrate_to_vram(svm_range, vma, dpagemap)) {
- err = xe_svm_alloc_vram(svm_range, &ctx, dpagemap);
- if (err) {
- drm_dbg(&vm->xe->drm, "VRAM allocation failed, retry from userspace, asid=%u, gpusvm=%p, errno=%pe\n",
- vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err));
- return -ENODATA;
- }
- xe_svm_range_debug(svm_range, "PREFETCH - RANGE MIGRATED TO VRAM");
+ thread->ctx = &ctx;
+ thread->vma = vma;
+ thread->svm_range = svm_range;
+ thread->dpagemap = dpagemap;
+ thread->err = 0;
+
+ if (skip_threads) {
+ prefetch_thread_func(thread);
+ if (thread->err)
+ return thread->err;
+ } else {
+ queue_work(vm->xe->usm.prefetch_wq, &thread->work);
}
+ }
- err = xe_svm_range_get_pages(vm, svm_range, &ctx);
- if (err) {
- drm_dbg(&vm->xe->drm, "Get pages failed, asid=%u, gpusvm=%p, errno=%pe\n",
- vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err));
- if (err == -EOPNOTSUPP || err == -EFAULT || err == -EPERM)
- err = -ENODATA;
- return err;
+ if (!skip_threads) {
+ for (i = 0; i < idx; ++i) {
+ thread = prefetches + i;
+
+ flush_work(&thread->work);
+ if (thread->err && !err)
+ err = thread->err;
}
- xe_svm_range_debug(svm_range, "PREFETCH - RANGE GET PAGES DONE");
+ kvfree(prefetches);
}
return err;
@@ -3298,7 +3429,8 @@ static int op_lock_and_prep(struct drm_exec *exec, struct xe_vm *vm,
return err;
}
-static int vm_bind_ioctl_ops_prefetch_ranges(struct xe_vm *vm, struct xe_vma_ops *vops)
+static int vm_bind_ioctl_ops_prefetch_ranges(struct xe_vm *vm,
+ struct xe_vma_ops *vops)
{
struct xe_vma_op *op;
int err;
@@ -3308,7 +3440,7 @@ static int vm_bind_ioctl_ops_prefetch_ranges(struct xe_vm *vm, struct xe_vma_ops
list_for_each_entry(op, &vops->list, link) {
if (op->base.op == DRM_GPUVA_OP_PREFETCH) {
- err = prefetch_ranges(vm, op);
+ err = prefetch_ranges(vm, vops, op);
if (err)
return err;
}
@@ -3569,7 +3701,7 @@ static struct dma_fence *vm_bind_ioctl_ops_execute(struct xe_vm *vm,
struct dma_fence *fence;
int err = 0;
- lockdep_assert_held_write(&vm->lock);
+ lockdep_assert_held(&vm->lock);
xe_validation_guard(&ctx, &vm->xe->val, &exec,
((struct xe_val_flags) {
@@ -3892,7 +4024,7 @@ int xe_vm_bind_ioctl(struct drm_device *dev, void *data, struct drm_file *file)
u32 num_syncs, num_ufence = 0;
struct xe_sync_entry *syncs = NULL;
struct drm_xe_vm_bind_op *bind_ops = NULL;
- struct xe_vma_ops vops;
+ struct xe_vma_ops vops = { .flags = 0, };
struct dma_fence *fence;
int err;
int i;
@@ -4067,6 +4199,11 @@ int xe_vm_bind_ioctl(struct drm_device *dev, void *data, struct drm_file *file)
goto unwind_ops;
}
+ if (!(vops.flags & XE_VMA_OPS_FLAG_MODIFIES_GPUVA)) {
+ vops.flags |= XE_VMA_OPS_FLAG_DOWNGRADE_LOCK;
+ downgrade_write(&vm->lock);
+ }
+
err = xe_vma_ops_alloc(&vops, args->num_binds > 1);
if (err)
goto unwind_ops;
@@ -4103,7 +4240,10 @@ put_obj:
free_bos:
kvfree(bos);
release_vm_lock:
- up_write(&vm->lock);
+ if (vops.flags & XE_VMA_OPS_FLAG_DOWNGRADE_LOCK)
+ up_read(&vm->lock);
+ else
+ up_write(&vm->lock);
put_exec_queue:
if (q)
xe_exec_queue_put(q);
@@ -4735,7 +4875,7 @@ static int xe_vm_alloc_vma(struct xe_vm *vm,
u16 default_pat;
int err;
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
if (is_madvise)
ops = drm_gpuvm_madvise_ops_create(&vm->gpuvm, map_req);
@@ -4869,7 +5009,7 @@ int xe_vm_alloc_madvise_vma(struct xe_vm *vm, uint64_t start, uint64_t range)
.map.va.range = range,
};
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
vm_dbg(&vm->xe->drm, "MADVISE_OPS_CREATE: addr=0x%016llx, size=0x%016llx", start, range);
@@ -4901,7 +5041,7 @@ void xe_vm_find_cpu_addr_mirror_vma_range(struct xe_vm *vm, u64 *start, u64 *end
{
struct xe_vma *prev, *next;
- lockdep_assert_held(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
if (*start >= SZ_4K) {
prev = xe_vm_find_vma_by_addr(vm, *start - SZ_4K);
@@ -4933,7 +5073,7 @@ int xe_vm_alloc_cpu_addr_mirror_vma(struct xe_vm *vm, uint64_t start, uint64_t r
.map.va.range = range,
};
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
vm_dbg(&vm->xe->drm, "CPU_ADDR_MIRROR_VMA_OPS_CREATE: addr=0x%016llx, size=0x%016llx",
start, range);
@@ -4955,7 +5095,6 @@ void xe_vm_add_exec_queue(struct xe_vm *vm, struct xe_exec_queue *q)
/* User VMs and queues only */
xe_assert(xe, !(q->flags & EXEC_QUEUE_FLAG_KERNEL));
- xe_assert(xe, !(q->flags & EXEC_QUEUE_FLAG_PERMANENT));
xe_assert(xe, !(q->flags & EXEC_QUEUE_FLAG_VM));
xe_assert(xe, !(q->flags & EXEC_QUEUE_FLAG_MIGRATE));
xe_assert(xe, vm->xef);
diff --git a/drivers/gpu/drm/xe/xe_vm_types.h b/drivers/gpu/drm/xe/xe_vm_types.h
index 635ed29b9a69..68588b624212 100644
--- a/drivers/gpu/drm/xe/xe_vm_types.h
+++ b/drivers/gpu/drm/xe/xe_vm_types.h
@@ -133,6 +133,12 @@ struct xe_vma {
};
/**
+ * @fault_lock: Synchronizes fault processing. Locking order: inside
+ * vm->lock, outside dma-resv.
+ */
+ struct mutex fault_lock;
+
+ /**
* @tile_invalidated: Tile mask of binding are invalidated for this VMA.
* protected by BO's resv and for userptrs, vm->svm.gpusvm.notifier_lock in
* write mode for writing or vm->svm.gpusvm.notifier_lock in read mode and
@@ -215,12 +221,26 @@ struct xe_vm {
/** @svm.gpusvm: base GPUSVM used to track fault allocations */
struct drm_gpusvm gpusvm;
/**
+ * @svm.range_lock: Protects insertion and removal of ranges
+ * from GPU SVM tree.
+ */
+ struct mutex range_lock;
+ /**
* @svm.garbage_collector: Garbage collector which is used unmap
* SVM range's GPU bindings and destroy the ranges.
*/
struct {
- /** @svm.garbage_collector.lock: Protect's range list */
- spinlock_t lock;
+ /**
+ * @svm.garbage_collector.lock: Ensures only one thread
+ * runs the garbage collector at a time. Locking order:
+ * inside vm->lock, outside range->lock and dma-resv.
+ */
+ struct mutex lock;
+ /**
+ * @svm.garbage_collector.list_lock: Protect's range
+ * list
+ */
+ spinlock_t list_lock;
/**
* @svm.garbage_collector.range_list: List of SVM ranges
* in the garbage collector.
@@ -350,11 +370,6 @@ struct xe_vm {
struct {
/** @asid: address space ID, unique to each VM */
u32 asid;
- /**
- * @last_fault_vma: Last fault VMA, used for fast lookup when we
- * get a flood of faults to the same VMA
- */
- struct xe_vma *last_fault_vma;
} usm;
/** @error_capture: allow to track errors */
@@ -541,11 +556,14 @@ struct xe_vma_ops {
/** @pt_update_ops: page table update operations */
struct xe_vm_pgtable_update_ops pt_update_ops[XE_MAX_TILES_PER_DEVICE];
/** @flag: signify the properties within xe_vma_ops*/
-#define XE_VMA_OPS_FLAG_HAS_SVM_PREFETCH BIT(0)
-#define XE_VMA_OPS_FLAG_MADVISE BIT(1)
-#define XE_VMA_OPS_ARRAY_OF_BINDS BIT(2)
-#define XE_VMA_OPS_FLAG_SKIP_TLB_WAIT BIT(3)
-#define XE_VMA_OPS_FLAG_ALLOW_SVM_UNMAP BIT(4)
+#define XE_VMA_OPS_FLAG_HAS_SVM_PREFETCH BIT(0)
+#define XE_VMA_OPS_FLAG_MADVISE BIT(1)
+#define XE_VMA_OPS_ARRAY_OF_BINDS BIT(2)
+#define XE_VMA_OPS_FLAG_SKIP_TLB_WAIT BIT(3)
+#define XE_VMA_OPS_FLAG_ALLOW_SVM_UNMAP BIT(4)
+#define XE_VMA_OPS_FLAG_MODIFIES_GPUVA BIT(5)
+#define XE_VMA_OPS_FLAG_DOWNGRADE_LOCK BIT(6)
+#define XE_VMA_OPS_FLAG_HAS_SVM_VALID_RANGE BIT(7)
u32 flags;
#ifdef TEST_VM_OPS_ERROR
/** @inject_error: inject error to test error handling */
diff --git a/drivers/gpu/drm/xe/xe_wa_oob.rules b/drivers/gpu/drm/xe/xe_wa_oob.rules
index f02ac9bf7424..dd69ad07f7a9 100644
--- a/drivers/gpu/drm/xe/xe_wa_oob.rules
+++ b/drivers/gpu/drm/xe/xe_wa_oob.rules
@@ -63,7 +63,7 @@
16026007364 MEDIA_VERSION(3000)
14020316580 MEDIA_VERSION(1301)
-14025883347 MEDIA_VERSION_RANGE(1301, 3503)
+14025883347 MEDIA_VERSION_RANGE(1301, 3500)
GRAPHICS_VERSION_RANGE(2004, 3005)
16029380221 MEDIA_VERSION(3500)
22022079272 MEDIA_VERSION(3503)