summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorMark Brown <broonie@kernel.org>2026-08-21 13:57:56 +0100
committerMark Brown <broonie@kernel.org>2026-08-21 13:57:56 +0100
commit07f6b48454b65cd7bb098356a99d30329369c18a (patch)
tree3290d6fdc23b2b4c336abd77795027f834104b5d
parent136c6deceac3769a9e4fd9cd1192aabe7ed761b9 (diff)
parent8899e413c5ab85443ec9bbc50cffe924c6b596de (diff)
downloadlinux-next-07f6b48454b65cd7bb098356a99d30329369c18a.tar.gz
linux-next-07f6b48454b65cd7bb098356a99d30329369c18a.zip
Merge branch 'drm-xe-next' of https://gitlab.freedesktop.org/drm/xe/kernel.git
-rw-r--r--Documentation/gpu/drm-ras.rst21
-rw-r--r--Documentation/gpu/drm-uapi.rst93
-rw-r--r--Documentation/gpu/xe/index.rst1
-rw-r--r--Documentation/gpu/xe/xe_sigid.rst14
-rw-r--r--Documentation/netlink/specs/drm_ras.yaml48
-rw-r--r--MAINTAINERS13
-rw-r--r--drivers/gpu/drm/drm_drv.c2
-rw-r--r--drivers/gpu/drm/drm_ras.c101
-rw-r--r--drivers/gpu/drm/drm_ras_nl.c6
-rw-r--r--drivers/gpu/drm/drm_ras_nl.h4
-rw-r--r--drivers/gpu/drm/xe/Makefile1
-rw-r--r--drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h1
-rw-r--r--drivers/gpu/drm/xe/abi/xe_log_abi.h199
-rw-r--r--drivers/gpu/drm/xe/abi/xe_sigid_abi.h172
-rw-r--r--drivers/gpu/drm/xe/regs/xe_gt_regs.h6
-rw-r--r--drivers/gpu/drm/xe/tests/Makefile1
-rw-r--r--drivers/gpu/drm/xe/tests/xe_any_kunit.c213
-rw-r--r--drivers/gpu/drm/xe/tests/xe_kunit_helpers.c4
-rw-r--r--drivers/gpu/drm/xe/tests/xe_log_kunit.c553
-rw-r--r--drivers/gpu/drm/xe/tests/xe_migrate.c3
-rw-r--r--drivers/gpu/drm/xe/xe_any.h137
-rw-r--r--drivers/gpu/drm/xe/xe_debugfs.c15
-rw-r--r--drivers/gpu/drm/xe/xe_debugfs.h2
-rw-r--r--drivers/gpu/drm/xe/xe_defaults.h1
-rw-r--r--drivers/gpu/drm/xe/xe_device.c30
-rw-r--r--drivers/gpu/drm/xe/xe_device_types.h23
-rw-r--r--drivers/gpu/drm/xe/xe_drm_ras.c31
-rw-r--r--drivers/gpu/drm/xe/xe_drm_ras.h3
-rw-r--r--drivers/gpu/drm/xe/xe_exec_queue.c6
-rw-r--r--drivers/gpu/drm/xe/xe_exec_queue_types.h21
-rw-r--r--drivers/gpu/drm/xe/xe_gsc.c3
-rw-r--r--drivers/gpu/drm/xe/xe_gt.c7
-rw-r--r--drivers/gpu/drm/xe/xe_gt_debugfs.c80
-rw-r--r--drivers/gpu/drm/xe/xe_gt_idle.c3
-rw-r--r--drivers/gpu/drm/xe/xe_gt_printk.h3
-rw-r--r--drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c111
-rw-r--r--drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h6
-rw-r--r--drivers/gpu/drm/xe/xe_gt_sriov_printk.h3
-rw-r--r--drivers/gpu/drm/xe/xe_gt_stats.c7
-rw-r--r--drivers/gpu/drm/xe/xe_gt_stats_types.h22
-rw-r--r--drivers/gpu/drm/xe/xe_guc.c18
-rw-r--r--drivers/gpu/drm/xe/xe_guc_ct.c95
-rw-r--r--drivers/gpu/drm/xe/xe_guc_ct.h38
-rw-r--r--drivers/gpu/drm/xe/xe_guc_exec_queue_types.h32
-rw-r--r--drivers/gpu/drm/xe/xe_guc_pagefault.c39
-rw-r--r--drivers/gpu/drm/xe/xe_guc_pc.c15
-rw-r--r--drivers/gpu/drm/xe/xe_guc_submit.c271
-rw-r--r--drivers/gpu/drm/xe/xe_guc_types.h6
-rw-r--r--drivers/gpu/drm/xe/xe_hwmon.c88
-rw-r--r--drivers/gpu/drm/xe/xe_log.c235
-rw-r--r--drivers/gpu/drm/xe/xe_log.h194
-rw-r--r--drivers/gpu/drm/xe/xe_mert.c2
-rw-r--r--drivers/gpu/drm/xe/xe_migrate.c2
-rw-r--r--drivers/gpu/drm/xe/xe_module.c4
-rw-r--r--drivers/gpu/drm/xe/xe_module.h1
-rw-r--r--drivers/gpu/drm/xe/xe_oa.c7
-rw-r--r--drivers/gpu/drm/xe/xe_pagefault.c700
-rw-r--r--drivers/gpu/drm/xe/xe_pagefault.h74
-rw-r--r--drivers/gpu/drm/xe/xe_pagefault_types.h112
-rw-r--r--drivers/gpu/drm/xe/xe_pci.c32
-rw-r--r--drivers/gpu/drm/xe/xe_pci_error.c13
-rw-r--r--drivers/gpu/drm/xe/xe_pcode.c14
-rw-r--r--drivers/gpu/drm/xe/xe_pm.c5
-rw-r--r--drivers/gpu/drm/xe/xe_printk.h3
-rw-r--r--drivers/gpu/drm/xe/xe_pt.c6
-rw-r--r--drivers/gpu/drm/xe/xe_pxp_submit.c5
-rw-r--r--drivers/gpu/drm/xe/xe_ras.c125
-rw-r--r--drivers/gpu/drm/xe/xe_sriov_printk.h3
-rw-r--r--drivers/gpu/drm/xe/xe_sriov_vf_ccs.c1
-rw-r--r--drivers/gpu/drm/xe/xe_survivability_mode.c85
-rw-r--r--drivers/gpu/drm/xe/xe_svm.c146
-rw-r--r--drivers/gpu/drm/xe/xe_svm.h59
-rw-r--r--drivers/gpu/drm/xe/xe_tile_printk.h3
-rw-r--r--drivers/gpu/drm/xe/xe_tile_sriov_printk.h3
-rw-r--r--drivers/gpu/drm/xe/xe_userptr.c19
-rw-r--r--drivers/gpu/drm/xe/xe_vm.c299
-rw-r--r--drivers/gpu/drm/xe/xe_vm_types.h42
-rw-r--r--drivers/gpu/drm/xe/xe_wa_oob.rules2
-rw-r--r--include/drm/drm_device.h1
-rw-r--r--include/drm/drm_ras.h5
-rw-r--r--include/uapi/drm/drm_ras.h15
81 files changed, 4248 insertions, 541 deletions
diff --git a/Documentation/gpu/drm-ras.rst b/Documentation/gpu/drm-ras.rst
index 83c21853b74b..406e4c49bac1 100644
--- a/Documentation/gpu/drm-ras.rst
+++ b/Documentation/gpu/drm-ras.rst
@@ -56,6 +56,7 @@ User space tools can:
``node-id`` and ``error-id`` as parameters.
* Clear specific error counters with the ``clear-error-counter`` command, using both
``node-id`` and ``error-id`` as parameters.
+* Subscribe to the ``error-report`` multicast group to receive ``error-event``.
YAML-based Interface
--------------------
@@ -111,3 +112,23 @@ Example: Clear an error counter for a given node
sudo ynl --family drm_ras --do clear-error-counter --json '{"node-id":0, "error-id":1}'
None
+
+Example: Subscribe to ``error-report`` multicast group
+
+.. code-block:: bash
+
+ sudo ynl --family drm_ras --output-json --subscribe error-report
+
+.. code-block:: json
+
+ {
+ "name": "error-event",
+ "msg": {
+ "device-name": "0000:03:00.0",
+ "node-id": 1,
+ "node-name": "uncorrectable-errors",
+ "error-id": 1,
+ "error-name": "error_name1",
+ "error-value": 1
+ }
+ }
diff --git a/Documentation/gpu/drm-uapi.rst b/Documentation/gpu/drm-uapi.rst
index 93df92c4ac8c..52255247a6db 100644
--- a/Documentation/gpu/drm-uapi.rst
+++ b/Documentation/gpu/drm-uapi.rst
@@ -424,7 +424,7 @@ needed.
Recovery
--------
-Current implementation defines four recovery methods, out of which, drivers
+Current implementation defines several recovery methods, out of which, drivers
can use any one, multiple or none. Method(s) of choice will be sent in the
uevent environment as ``WEDGED=<method1>[,..,<methodN>]`` in order of less to
more side-effects. See the section `Vendor Specific Recovery`_
@@ -434,15 +434,16 @@ method is unknown, ``WEDGED=unknown`` will be sent instead.
Userspace consumers can parse this event and attempt recovery as per the
following expectations.
- =============== ========================================
+ =============== =========================================
Recovery method Consumer expectations
- =============== ========================================
+ =============== =========================================
none optional telemetry collection
rebind unbind + bind driver
bus-reset unbind + bus reset/re-enumeration + bind
vendor-specific vendor specific recovery method
+ cold-reset remove device + slot power cycle + rescan
unknown consumer policy
- =============== ========================================
+ =============== =========================================
No Recovery
-----------
@@ -453,6 +454,17 @@ debug purpose in order to root cause the hang. This is useful because the first
hang is usually the most critical one which can result in consequential hangs
or complete wedging.
+Cold Reset Recovery
+-------------------
+
+When ``WEDGED=cold-reset`` is sent, it indicates that the device has
+encountered an error condition that cannot be resolved through other
+recovery methods such as driver rebind or bus reset, and requires a complete
+device power cycle to restore functionality.
+
+This method is used by devices that are plugged directly into the PCIe slot
+which supports removing the power.
+
Vendor Specific Recovery
------------------------
@@ -516,7 +528,7 @@ Example - rebind
Udev rule::
- SUBSYSTEM=="drm", ENV{WEDGED}=="rebind", DEVPATH=="*/drm/card[0-9]",
+ SUBSYSTEM=="drm", ENV{WEDGED}=="rebind", DEVPATH=="*/drm/card[0-9]", \
RUN+="/path/to/rebind.sh $env{DEVPATH}"
Recovery script::
@@ -530,6 +542,77 @@ Recovery script::
echo -n $DEVICE > $DRIVER/unbind
echo -n $DEVICE > $DRIVER/bind
+Example - cold-reset
+--------------------
+
+Udev rule::
+
+ SUBSYSTEM=="drm", ENV{WEDGED}=="cold-reset", DEVPATH=="*/drm/card[0-9]", \
+ RUN+="/path/to/cold-reset.sh $env{DEVPATH}"
+
+Recovery script::
+
+ #!/bin/sh
+ die() { echo "ERROR: $*" >&2; exit 1; }
+
+ [ -n "$1" ] || die "Usage: $0 <device-path>"
+
+ PCI_DEVS=/sys/bus/pci/devices
+ PCI_SLOTS=/sys/bus/pci/slots
+
+ syspath=$(readlink -f "/sys/$1/device" 2>/dev/null || readlink -f "/sys/$1" 2>/dev/null)
+ [ -n "$syspath" ] || die "cannot resolve sysfs path for: $1"
+
+ dev=$(basename "$syspath")
+ [ -e "$PCI_DEVS/$dev" ] || die "not a PCI device: $dev"
+ echo "device : $dev"
+
+ slot=""
+ walk=$(dirname "$(readlink -f "$PCI_DEVS/$dev")")
+
+ while true; do
+ ancestor=$(basename "$walk")
+ case "$ancestor" in pci*) break ;; esac # reached the virtual bus root
+
+ ancestor_nofn=${ancestor%.*} # strip function: 0000:03:01.0 -> 0000:03:01
+
+ for f in "$PCI_SLOTS"/*/address; do
+ [ -f "$f" ] || continue
+ addr=$(cat "$f")
+ case "$ancestor_nofn" in
+ *"$addr") slot=$(basename "$(dirname "$f")"); break ;;
+ esac
+ done
+
+ if [ -n "$slot" ] && [ -e "$PCI_SLOTS/$slot/power" ]; then
+ echo "slot : $slot (port $ancestor)"
+ break
+ fi
+ slot=""
+ walk=$(dirname "$walk")
+ done
+
+ [ -n "$slot" ] || die "no hotplug slot with power control found in PCIe topology"
+
+ # Cold reset: remove the device, cut slot power, restore power, rescan.
+ echo "Removing $dev..."
+ [ -e "$PCI_DEVS/$dev" ] && echo 1 > "$PCI_DEVS/$dev/remove"
+
+ echo "Powering off slot $slot..."
+ echo 0 > "$PCI_SLOTS/$slot/power"
+
+ echo "Powering on slot $slot..."
+ echo 1 > "$PCI_SLOTS/$slot/power"
+
+ echo "Rescanning PCI bus..."
+ echo 1 > /sys/bus/pci/rescan
+
+ if [ -e "$PCI_DEVS/$dev" ]; then
+ echo "Done: $dev is back online."
+ else
+ echo "WARNING: $dev did not re-appear after rescan."
+ fi
+
Customization
-------------
diff --git a/Documentation/gpu/xe/index.rst b/Documentation/gpu/xe/index.rst
index 665c0e93601c..0247a255f7e6 100644
--- a/Documentation/gpu/xe/index.rst
+++ b/Documentation/gpu/xe/index.rst
@@ -35,3 +35,4 @@ The display, or :ref:`drm-kms`, support for drm/xe is provided by
xe-drm-usage-stats.rst
xe_configfs
xe_gt_stats
+ xe_sigid
diff --git a/Documentation/gpu/xe/xe_sigid.rst b/Documentation/gpu/xe/xe_sigid.rst
new file mode 100644
index 000000000000..45d84a62f185
--- /dev/null
+++ b/Documentation/gpu/xe/xe_sigid.rst
@@ -0,0 +1,14 @@
+.. SPDX-License-Identifier: (GPL-2.0+ OR MIT)
+
+========
+Xe SIGID
+========
+
+.. kernel-doc:: drivers/gpu/drm/xe/abi/xe_sigid_abi.h
+ :doc: Xe Error Signatures (SIGID)
+
+Signature Identifiers
+=====================
+
+.. kernel-doc:: drivers/gpu/drm/xe/abi/xe_sigid_abi.h
+ :internal:
diff --git a/Documentation/netlink/specs/drm_ras.yaml b/Documentation/netlink/specs/drm_ras.yaml
index e113056f8c01..8aed3d4515e5 100644
--- a/Documentation/netlink/specs/drm_ras.yaml
+++ b/Documentation/netlink/specs/drm_ras.yaml
@@ -69,6 +69,33 @@ attribute-sets:
name: error-value
type: u32
doc: Current value of the requested error counter.
+ -
+ name: error-event-attrs
+ attributes:
+ -
+ name: device-name
+ type: string
+ doc: Device (PCI BDF, UUID) that reported the error.
+ -
+ name: node-id
+ type: u32
+ doc: ID of the node that reported the error.
+ -
+ name: node-name
+ type: string
+ doc: Name of the node that reported the error.
+ -
+ name: error-id
+ type: u32
+ doc: ID of the error counter.
+ -
+ name: error-name
+ type: string
+ doc: Name of the error.
+ -
+ name: error-value
+ type: u32
+ doc: Current value of the error counter.
operations:
list:
@@ -124,3 +151,24 @@ operations:
do:
request:
attributes: *id-attrs
+ -
+ name: error-event
+ doc: >-
+ Report an error event to userspace.
+ The event includes the device, node and error information
+ of the error that triggered the event.
+ attribute-set: error-event-attrs
+ mcgrp: error-report
+ event:
+ attributes:
+ - device-name
+ - node-id
+ - node-name
+ - error-id
+ - error-name
+ - error-value
+
+mcast-groups:
+ list:
+ -
+ name: error-report
diff --git a/MAINTAINERS b/MAINTAINERS
index 218ef1717388..6e93f637ba30 100644
--- a/MAINTAINERS
+++ b/MAINTAINERS
@@ -9080,6 +9080,19 @@ T: git https://gitlab.freedesktop.org/drm/misc/kernel.git
F: drivers/gpu/drm/drm_privacy_screen*
F: include/drm/drm_privacy_screen*
+DRM RAS
+M: Riana Tauro <riana.tauro@intel.com>
+M: Rodrigo Vivi <rodrigo.vivi@intel.com>
+R: Aravind Iddamsetty <aravind.iddamsetty@linux.intel.com>
+R: Raag Jadav <raag.jadav@intel.com>
+L: dri-devel@lists.freedesktop.org
+S: Supported
+T: git https://gitlab.freedesktop.org/drm/misc/kernel.git
+F: Documentation/netlink/specs/drm_ras.yaml
+F: Documentation/gpu/drm-ras.rst
+F: drivers/gpu/drm/drm_ras*
+F: include/drm/drm_ras*
+
DRM TTM SUBSYSTEM
M: Christian Koenig <christian.koenig@amd.com>
M: Huang Rui <ray.huang@amd.com>
diff --git a/drivers/gpu/drm/drm_drv.c b/drivers/gpu/drm/drm_drv.c
index c808958a2188..cb53baa70995 100644
--- a/drivers/gpu/drm/drm_drv.c
+++ b/drivers/gpu/drm/drm_drv.c
@@ -545,6 +545,8 @@ static const char *drm_get_wedge_recovery(unsigned int opt)
return "bus-reset";
case DRM_WEDGE_RECOVERY_VENDOR:
return "vendor-specific";
+ case DRM_WEDGE_RECOVERY_COLD_RESET:
+ return "cold-reset";
default:
return NULL;
}
diff --git a/drivers/gpu/drm/drm_ras.c b/drivers/gpu/drm/drm_ras.c
index d6eab29a1394..39155fb514de 100644
--- a/drivers/gpu/drm/drm_ras.c
+++ b/drivers/gpu/drm/drm_ras.c
@@ -41,6 +41,11 @@
* Userspace must provide Node ID, Error ID.
* Clears specific error counter of a node if supported.
*
+ * 4. ERROR_REPORT: Subscribe to this multicast group to receive error events
+ *
+ * 5. ERROR_EVENT: Report an error event to userspace. The event contains device, node
+ * and error information that triggered the event.
+ *
* Node registration:
*
* - drm_ras_node_register(): Registers a new node and assigns
@@ -186,6 +191,34 @@ static int msg_reply_value(struct sk_buff *msg, u32 error_id,
value);
}
+static int msg_put_error_event_attrs(struct sk_buff *msg, struct drm_ras_node *node,
+ u32 error_id, const char *error_name, u32 value)
+{
+ int ret;
+
+ ret = nla_put_string(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_DEVICE_NAME, node->device_name);
+ if (ret)
+ return ret;
+
+ ret = nla_put_u32(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_NODE_ID, node->id);
+ if (ret)
+ return ret;
+
+ ret = nla_put_string(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_NODE_NAME, node->node_name);
+ if (ret)
+ return ret;
+
+ ret = nla_put_u32(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_ERROR_ID, error_id);
+ if (ret)
+ return ret;
+
+ ret = nla_put_string(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_ERROR_NAME, error_name);
+ if (ret)
+ return ret;
+
+ return nla_put_u32(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_ERROR_VALUE, value);
+}
+
static int doit_reply_value(struct genl_info *info, u32 node_id,
u32 error_id)
{
@@ -223,6 +256,74 @@ static int doit_reply_value(struct genl_info *info, u32 node_id,
}
/**
+ * drm_ras_nl_error_event() - Report an error event
+ * @node: Node structure
+ * @error_id: ID of the error
+ * @error_name: Name of the error
+ * @value: Value of the error counter
+ *
+ * Report an error-event to userspace using the error-report multicast group.
+ *
+ * Context: Process context only. Uses %GFP_KERNEL and multicasts to all
+ * netns, which is unbounded work; callers in interrupt handlers
+ * or other atomic context must defer event.
+ *
+ * Return: 0 on success, or negative errno on failure.
+ */
+int drm_ras_nl_error_event(struct drm_ras_node *node, u32 error_id, const char *error_name,
+ u32 value)
+{
+ struct genl_info info;
+ struct sk_buff *msg;
+ struct nlattr *hdr;
+ int ret;
+
+ if (!node || !error_name)
+ return -EINVAL;
+
+ /* Check the node is currently registered */
+ if (xa_load(&drm_ras_xa, node->id) != node)
+ return -ENOENT;
+
+ /* Currently only Error Counter events are supported */
+ if (node->type != DRM_RAS_NODE_TYPE_ERROR_COUNTER)
+ return -EOPNOTSUPP;
+
+ /* Check the error ID is within the valid range */
+ if (error_id < node->error_counter_range.first ||
+ error_id > node->error_counter_range.last)
+ return -EINVAL;
+
+ genl_info_init_ntf(&info, &drm_ras_nl_family, DRM_RAS_CMD_ERROR_EVENT);
+
+ msg = genlmsg_new(NLMSG_GOODSIZE, GFP_KERNEL);
+ if (!msg)
+ return -ENOMEM;
+
+ hdr = genlmsg_iput(msg, &info);
+ if (!hdr) {
+ ret = -EMSGSIZE;
+ goto free_msg;
+ }
+
+ ret = msg_put_error_event_attrs(msg, node, error_id, error_name, value);
+ if (ret)
+ goto cancel_msg;
+
+ genlmsg_end(msg, hdr);
+ genlmsg_multicast_allns(&drm_ras_nl_family, msg, 0, DRM_RAS_NLGRP_ERROR_REPORT);
+
+ return 0;
+
+cancel_msg:
+ genlmsg_cancel(msg, hdr);
+free_msg:
+ nlmsg_free(msg);
+ return ret;
+}
+EXPORT_SYMBOL(drm_ras_nl_error_event);
+
+/**
* drm_ras_nl_get_error_counter_dumpit() - Dump all Error Counters
* @skb: Netlink message buffer
* @cb: Callback context for multi-part dumps
diff --git a/drivers/gpu/drm/drm_ras_nl.c b/drivers/gpu/drm/drm_ras_nl.c
index dea1c1b2494e..9d3123cc9f9c 100644
--- a/drivers/gpu/drm/drm_ras_nl.c
+++ b/drivers/gpu/drm/drm_ras_nl.c
@@ -58,6 +58,10 @@ static const struct genl_split_ops drm_ras_nl_ops[] = {
},
};
+static const struct genl_multicast_group drm_ras_nl_mcgrps[] = {
+ [DRM_RAS_NLGRP_ERROR_REPORT] = { "error-report", },
+};
+
struct genl_family drm_ras_nl_family __ro_after_init = {
.name = DRM_RAS_FAMILY_NAME,
.version = DRM_RAS_FAMILY_VERSION,
@@ -66,4 +70,6 @@ struct genl_family drm_ras_nl_family __ro_after_init = {
.module = THIS_MODULE,
.split_ops = drm_ras_nl_ops,
.n_split_ops = ARRAY_SIZE(drm_ras_nl_ops),
+ .mcgrps = drm_ras_nl_mcgrps,
+ .n_mcgrps = ARRAY_SIZE(drm_ras_nl_mcgrps),
};
diff --git a/drivers/gpu/drm/drm_ras_nl.h b/drivers/gpu/drm/drm_ras_nl.h
index a398643572a5..03ec275aca92 100644
--- a/drivers/gpu/drm/drm_ras_nl.h
+++ b/drivers/gpu/drm/drm_ras_nl.h
@@ -21,6 +21,10 @@ int drm_ras_nl_get_error_counter_dumpit(struct sk_buff *skb,
int drm_ras_nl_clear_error_counter_doit(struct sk_buff *skb,
struct genl_info *info);
+enum {
+ DRM_RAS_NLGRP_ERROR_REPORT,
+};
+
extern struct genl_family drm_ras_nl_family;
#endif /* _LINUX_DRM_RAS_GEN_H */
diff --git a/drivers/gpu/drm/xe/Makefile b/drivers/gpu/drm/xe/Makefile
index 67ada1d6c2fb..7ac3954737f9 100644
--- a/drivers/gpu/drm/xe/Makefile
+++ b/drivers/gpu/drm/xe/Makefile
@@ -87,6 +87,7 @@ xe-y += xe_bb.o \
xe_hw_fence.o \
xe_irq.o \
xe_late_bind_fw.o \
+ xe_log.o \
xe_lrc.o \
xe_mem_pool.o \
xe_migrate.o \
diff --git a/drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h b/drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h
index ce5c59517528..57baa68bcba4 100644
--- a/drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h
+++ b/drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h
@@ -119,6 +119,7 @@ enum slpc_param_id {
SLPC_PARAM_STRATEGIES = 26,
SLPC_PARAM_POWER_PROFILE = 27,
SLPC_PARAM_IGNORE_EFFICIENT_FREQUENCY = 28,
+ SLPC_PARAM_SET_IBC_VERSION = 29,
SLPC_MAX_PARAM = 32,
};
diff --git a/drivers/gpu/drm/xe/abi/xe_log_abi.h b/drivers/gpu/drm/xe/abi/xe_log_abi.h
new file mode 100644
index 000000000000..d6105520173e
--- /dev/null
+++ b/drivers/gpu/drm/xe/abi/xe_log_abi.h
@@ -0,0 +1,199 @@
+/* SPDX-License-Identifier: MIT */
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#ifndef _ABI_XE_LOG_ABI_H_
+#define _ABI_XE_LOG_ABI_H_
+
+#include <linux/bits.h>
+#include <linux/bitfield.h>
+
+#include "abi/xe_sigid_abi.h"
+
+/**
+ * enum xe_log_component_bits - bits for components structure definitions
+ *
+ * Component identifiers are structured based on::
+ *
+ * COMPONENT = CLASS(8b).TYPE(8b)
+ *
+ * and the structure looks like this::
+ *
+ * ├── SYSTEM(0)
+ * │ └── ...
+ * ├── DRIVER(1)
+ * │ └── ...
+ * ├── FEATURE(2)
+ * │ └── ...
+ * ├── FIRMWARE(4)
+ * │ └── ...
+ * └── HARDWARE(8)
+ * └── ...
+ *
+ * Examples::
+ *
+ * COMPONENT(0.type) = SYSTEM.type = system component
+ * COMPONENT(1.type) = DRIVER.type = driver core component
+ * COMPONENT(3.type) = DRIVER_FEATURE.type = driver feature
+ * COMPONENT(5.type) = DRIVER_FIRMWARE.type = firmware driver component
+ * COMPONENT(9.type) = DRIVER_HARDWARE.type = hardware driver component
+ *
+ */
+enum xe_log_component_bits {
+ /* private: */
+ XE_LOG_COMPONENT_CLASS_MASK = GENMASK_U16(7, 0),
+ XE_LOG_COMPONENT_TYPE_MASK = GENMASK_U16(15, 8),
+ /* private: component classes */
+ XE_LOG_COMPONENT_CLASS_SYSTEM = 0u,
+ XE_LOG_COMPONENT_CLASS_DRIVER = 1u,
+ XE_LOG_COMPONENT_CLASS_FEATURE = 2u,
+ XE_LOG_COMPONENT_CLASS_FIRMWARE = 4u,
+ XE_LOG_COMPONENT_CLASS_HARDWARE = 8u,
+ XE_LOG_COMPONENT_CLASS_DRIVER_FEATURE = XE_LOG_COMPONENT_CLASS_DRIVER |
+ XE_LOG_COMPONENT_CLASS_FEATURE,
+ XE_LOG_COMPONENT_CLASS_DRIVER_FIRMWARE = XE_LOG_COMPONENT_CLASS_DRIVER |
+ XE_LOG_COMPONENT_CLASS_FIRMWARE,
+ XE_LOG_COMPONENT_CLASS_DRIVER_HARDWARE = XE_LOG_COMPONENT_CLASS_DRIVER |
+ XE_LOG_COMPONENT_CLASS_HARDWARE,
+ /* private: reserved identifiers */
+ XE_LOG_COMPONENT_NONE = 0u,
+};
+
+#define MAKE_XE_LOG_COMPONENT(_CLASS, type) \
+ (FIELD_PREP_CONST(XE_LOG_COMPONENT_CLASS_MASK, \
+ XE_LOG_COMPONENT_CLASS_##_CLASS) | \
+ FIELD_PREP_CONST(XE_LOG_COMPONENT_TYPE_MASK, (type)))
+
+/**
+ * enum xe_log_location_bits - bits for location structure definitions
+ *
+ * Location identifiers are structured based on::
+ *
+ * LOCATION = TYPE(8b).ID(8b)
+ *
+ * and the structure looks like this::
+ *
+ * ├── DEVICE(0)
+ * │ └── MBZ(0)
+ * ├── TILE(1)
+ * │ ├── Tile0(0)
+ * │ ├── ...
+ * │ └── TileN(n)
+ * ├── GT(1)
+ * │ ├── GT0(0)
+ * │ ├── ...
+ * │ └── GTn(n)
+ * └── ...
+ *
+ * Examples::
+ *
+ * LOCATION(0.0) = NONE
+ * LOCATION(1.0) = DEVICE.0 = "Device"
+ * LOCATION(2.1) = TILE.1 = "Tile1"
+ * LOCATION(3.2) = GT.2 = "GT2"
+ *
+ */
+enum xe_log_location_bits {
+ /* private: */
+ XE_LOG_LOCATION_TYPE_MASK = GENMASK_U16(7, 0),
+ XE_LOG_LOCATION_ID_MASK = GENMASK_U16(15, 8),
+ /* private: location types */
+ XE_LOG_LOCATION_TYPE_DEVICE = 1u,
+ XE_LOG_LOCATION_TYPE_TILE = 2u,
+ XE_LOG_LOCATION_TYPE_GT = 3u,
+ /* private: reserved identifiers */
+ XE_LOG_LOCATION_NONE = 0u,
+};
+
+#define PREP_XE_LOG_LOCATION(type, id) \
+ (FIELD_PREP(XE_LOG_LOCATION_TYPE_MASK, (type)) | \
+ FIELD_PREP(XE_LOG_LOCATION_ID_MASK, (id)))
+
+#define MAKE_XE_LOG_LOCATION(_TYPE, id) \
+ PREP_XE_LOG_LOCATION(XE_LOG_LOCATION_TYPE_##_TYPE, (id))
+
+/**
+ * DEFINE_XE_LOG_COMPONENTS() - Define log components.
+ * @define: name of the inner macro to expand.
+ *
+ * Use this super macro to define custom code for the log components.
+ * The following parameters are available for each component::
+ *
+ * define(CLASS, ID, TAG, SIGID, NAME)
+ *
+ * where:
+ *
+ * @CLASS is the component class name (without the XE_LOG_COMPONENT_CLASS_ prefix)
+ * @ID is the unique component identifier within @CLASS
+ * @TAG is unique component tag (across all components)
+ * @SIGID is the default xe_sigid for the component (without the XE_SIGID_ prefix)
+ */
+#define DEFINE_XE_LOG_COMPONENTS(define) \
+ DEFINE_XE_LOG_SOFTWARE_COMPONENTS(define) \
+ DEFINE_XE_LOG_HARDWARE_COMPONENTS(define)
+
+#define DEFINE_XE_LOG_SOFTWARE_COMPONENTS(define) \
+ /* */ \
+ define(SYSTEM, 1, PCI, IO_BUS, "Linux PCI Subsystem") \
+ define(SYSTEM, 2, DRM, SW, "DRM") \
+ /* */ \
+ define(DRIVER, 1, XE, SW, "Xe Driver") \
+ define(DRIVER, 2, PROBE, PROBE, "Driver Initialization") \
+ define(DRIVER, 3, WEDGED, WEDGED, "Device Malfunction") \
+ define(DRIVER, 4, RTP, SW, "Register Table Processing") \
+ define(DRIVER, 5, WA, SW, "Workarounds") \
+ define(DRIVER, 6, PAGEFAULT, MEM_FAULT, "Page Fault") \
+ /* */ \
+ define(DRIVER_HARDWARE, 1, REGS, IO_BUS, "Registers") \
+ define(DRIVER_HARDWARE, 2, GGTT, IO_BUS, "Global GTT") \
+ define(DRIVER_HARDWARE, 3, GT, GT_TDR, "Graphics Technology") \
+ define(DRIVER_HARDWARE, 4, LMTT, IO_BUS, "LMEM Translation Table") \
+ define(DRIVER_HARDWARE, 5, MEMIRQ, IO_BUS, "Memory Based IRQ") \
+ /* */ \
+ define(DRIVER_FEATURE, 1, PF, SW, "SR-IOV Physical Function") \
+ define(DRIVER_FEATURE, 2, VF, SW, "SR-IOV Virtual Function") \
+ define(DRIVER_FEATURE, 3, SURVIVABILITY, SURVIVABILITY, "Survivability") \
+ define(DRIVER_FEATURE, 4, RAS, SW, "Reliability, Accessibility, Serviceability") \
+ /* */ \
+ define(DRIVER_FIRMWARE, 1, GUC, RUNTIME_FW, "GuC") \
+ define(DRIVER_FIRMWARE, 2, HUC, RUNTIME_FW, "HuC") \
+ define(DRIVER_FIRMWARE, 3, GSC, RUNTIME_FW, "GSC") \
+ define(DRIVER_FIRMWARE, 16, PCODE, DEVICE_FW, "PCode") \
+ define(DRIVER_FIRMWARE, 17, SYSCTRL, DEVICE_FW, "System Controller") \
+
+#define DEFINE_XE_LOG_HARDWARE_COMPONENTS(define) \
+ define(HARDWARE, 1, DEVICE_MEMORY, DEVICE_MEMORY, "Device Memory") \
+ define(HARDWARE, 2, CORE_COMPUTE, CORE_COMPUTE, "Core Compute") \
+ /* HARDWARE, 3, RESERVED */ \
+ define(HARDWARE, 4, PCIE, PCIE, "PCIe Interface") \
+ define(HARDWARE, 5, FABRIC, FABRIC, "Fabric") \
+ define(HARDWARE, 6, SOC_INTERNAL, SOC_INTERNAL, "SoC Internal") \
+ /* eod */
+
+/**
+ * enum xe_log_component_tags - TAGs of all supported components
+ */
+enum xe_log_component_tags {
+ /* private: */
+#define MAKE_XE_LOG_COMPONENT_ENUM(_CLASS, _ID, _TAG, _SIG, _NAME) \
+ XE_LOG_COMPONENT_##_TAG = MAKE_XE_LOG_COMPONENT(_CLASS, (_ID)), \
+ XE_LOG_COMPONENT_##_CLASS##_##_ID = XE_LOG_COMPONENT_##_TAG, \
+ /* eod */
+ DEFINE_XE_LOG_COMPONENTS(MAKE_XE_LOG_COMPONENT_ENUM)
+#undef MAKE_XE_LOG_COMPONENT_ENUM
+};
+
+/**
+ * enum xe_log_component_sigids - SIGIDs of all supported components
+ */
+enum xe_log_component_sigids {
+ /* private: */
+#define MAKE_XE_LOG_COMPONENT_SIGID(_CLASS, _ID, _TAG, _SIG, _NAME) \
+ XE_LOG_COMPONENT_##_TAG##_SIGID = XE_SIGID_##_SIG, \
+ /* eod */
+ DEFINE_XE_LOG_COMPONENTS(MAKE_XE_LOG_COMPONENT_SIGID)
+#undef MAKE_XE_LOG_COMPONENT_SIGID
+};
+
+#endif
diff --git a/drivers/gpu/drm/xe/abi/xe_sigid_abi.h b/drivers/gpu/drm/xe/abi/xe_sigid_abi.h
new file mode 100644
index 000000000000..2f3440505f69
--- /dev/null
+++ b/drivers/gpu/drm/xe/abi/xe_sigid_abi.h
@@ -0,0 +1,172 @@
+/* SPDX-License-Identifier: MIT */
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#ifndef _ABI_XE_SIGID_ABI_H_
+#define _ABI_XE_SIGID_ABI_H_
+
+/**
+ * DOC: Xe Error Signatures (SIGID)
+ *
+ * What SIGID stands for
+ * ---------------------
+ *
+ * SIGID is short for *Signature Identifier*. It is a small, stable integer
+ * that names one of the *recognised fault sites* -- nothing more. It is the
+ * primary handle used for triage and maps directly to a specific report site.
+ *
+ * Numbering
+ * ---------
+ *
+ * SIGIDs are a single flat list numbered sequentially within the assigned range,
+ * in the order the fault sites were introduced. Values are stable: once assigned
+ * they are only ever appended, never renumbered or reused. A retired fault site
+ * SIGID value is deprecated in place, never re-purposed.
+ *
+ * Why this exists
+ * ---------------
+ *
+ * Today the driver reports faults with ad-hoc ``xe_err()`` / ``xe_gt_err()``
+ * strings that have no stable shape. That is fine for a human reading dmesg,
+ * but it gives fleet tooling nothing durable to match on: the wording changes
+ * between releases, lines can be rate-limited or dropped under an error storm,
+ * and there is no consistent way to ask "which recognised fault just happened?"
+ *
+ * A SIGID answers exactly that one question, identically across driver and
+ * firmware versions, and (eventually) across other Intel devices in a node.
+ *
+ * What a SIGID is not
+ * -------------------
+ *
+ * SIGID deliberately does not encode the detailed reason or the outcome. Those
+ * are carried alongside it::
+ *
+ * SIGID -> which recognised fault site is being reported
+ * severity -> how serious this instance is
+ * errno -> the failing operation's error, if available, shown with %pe
+ * message -> free-form human-readable context
+ *
+ * Severity is independent of the SIGID. The same SIGID can be reported at
+ * different severities depending on the instance and the recovery taken.
+ *
+ * When to use SIGID logging
+ * -------------------------
+ *
+ * The xe_log_*() helpers are for these recognised fault sites only --
+ * important, operator-relevant faults and events. The driver's only job is to
+ * emit the right SIGID next to the usual human-readable text.
+
+ * They are not a replacement for ``xe_info()`` / ``xe_dbg()`` / tracing, nor
+ * for one-off diagnostics; using them for ordinary logging would dilute the
+ * fault stream. Not every ``xe_err()`` needs to become a SIGID report -- only
+ * those that correspond to a published fault site.
+ *
+ * SIGID log output (dmesg vs. the machine record)
+ * -----------------------------------------------
+ *
+ * The dmesg line stays close to a normal xe error message so it remains
+ * readable for admins; the only stable, machine-matchable token on it is
+ * ``SIGID=<n>`` (``dmesg | grep SIGID=``).
+ *
+ * The full dmesg line is not an ABI: the surrounding text may change freely,
+ * and lines may be dropped. The durable record for tooling is the CPER record
+ * carrying the same SIGID (generation is a planned follow-up).
+ *
+ * How to pick a SIGID (the uniqueness rule)
+ * -----------------------------------------
+ *
+ * Pick per *report site*, not per incident. Each site emits the single most
+ * specific recognised SIGID *for that site* -- so the question is never
+ * "classify this whole failure", it is "what does this site detect?", which has
+ * one answer. A single underlying failure therefore legitimately produces a
+ * *chain* of reports from different layers, each with its own SIGID -- e.g. a
+ * GuC communication failure is reported as %XE_SIGID_RUNTIME_FW by the firmware
+ * path, the failed recovery as %XE_SIGID_GT_TDR by the reset path, and an
+ * aborted bind as %XE_SIGID_PROBE by the probe path. That chain lets triage
+ * follow a fault from origin to final effect; it is not a duplicate.
+ *
+ * If a site does not match any defined SIGID, keep using the ordinary
+ * ``xe_err()`` / ``xe_gt_err()`` logging rather than forcing a SIGID: a wrong
+ * or over-broad classification is harder to retire than a missing one. When a
+ * new report site is genuinely worth triaging, add it to the list below.
+ *
+ * Usage of the existing SIGID reports must reevaluated according to this section
+ * after making significant changes to the site that emits this SIGID.
+ *
+ * Scope: software vs hardware emitted signatures
+ * ----------------------------------------------
+ *
+ * Some SIGID represents fault sites that the *driver itself* detects and
+ * reports from the software POV: probe abort, wedged, survivability, driver-
+ * detected firmware failures, engine TDR, memory faults and IO/bus faults.
+ * These are the only values the driver assigns on its own.
+ *
+ * Signatures that *originate* in firmware or hardware are a different thing:
+ * they are produced and identified by the firmware or the hardware itself
+ * (e.g. via their own records or error counters), and the driver merely logs
+ * them as they are given to us. They are deliberately enumerated separately.
+ *
+ * The two driver-detected firmware report sites below (%XE_SIGID_RUNTIME_FW,
+ * %XE_SIGID_DEVICE_FW) are software signatures: they mark that *the driver*
+ * observed a firmware problem, not a signature reported by the firmware.
+ */
+
+/*
+ * Top level Intel Error Signature Identifiers.
+ */
+#define INTEL_SIGID_INVALID 0
+#define INTEL_SIGID_BATCH 100
+#define INTEL_SIGID_RANGE_START(n) ((n) * INTEL_SIGID_BATCH)
+#define INTEL_SIGID_RANGE_END(n) (INTEL_SIGID_RANGE_START((n) + 1) - 1)
+
+/* SIGIDs 1xx are reserved for Xe GPU software and 2xx for Xe GPU hardware */
+#define INTEL_SIGID_GPU_XE_SOFTWARE_START INTEL_SIGID_RANGE_START(1)
+#define INTEL_SIGID_GPU_XE_SOFTWARE_END INTEL_SIGID_RANGE_END(1)
+#define INTEL_SIGID_GPU_XE_HARDWARE_START INTEL_SIGID_RANGE_START(2)
+#define INTEL_SIGID_GPU_XE_HARDWARE_END INTEL_SIGID_RANGE_END(2)
+
+/**
+ * enum xe_sigid - Stable Xe Error Signature Identifiers (SIGID).
+ * @XE_SIGID_SW: Software component failure.
+ * @XE_SIGID_PROBE: Device probe/bind was aborted.
+ * @XE_SIGID_WEDGED: Device was declared wedged and is no longer usable.
+ * @XE_SIGID_SURVIVABILITY: Device entered survivability mode.
+ * @XE_SIGID_RUNTIME_FW: Driver-detected runtime firmware failure, GuC/HuC/GSC.
+ * @XE_SIGID_DEVICE_FW: Driver-detected device firmware failure, PCODE/sysctrl.
+ * @XE_SIGID_GT_TDR: Engine hang / timeout detection and recovery (reset).
+ * @XE_SIGID_MEM_FAULT: VM bind, page fault or GTT fault.
+ * @XE_SIGID_IO_BUS: Runtime PCIe / IOMMU / MMIO access fault.
+ * @XE_SIGID_HW: Generic hardware failure.
+ * @XE_SIGID_PCIE: PCIe interface errors.
+ * @XE_SIGID_DEVICE_MEMORY: Device memory errors.
+ * @XE_SIGID_CORE_COMPUTE: Compute/shader core errors.
+ * @XE_SIGID_FABRIC: Fabric errors.
+ * @XE_SIGID_SOC_INTERNAL: SoC-internal errors.
+ *
+ * Each SIGID represents the report sites the driver detects and reports.
+ * Values are numbered sequentially, are only ever appended, and are never
+ * renumbered or reused.
+ *
+ * Firmware- and hardware-originated signatures are numbered separately.
+ */
+enum xe_sigid {
+ XE_SIGID_SW = INTEL_SIGID_GPU_XE_SOFTWARE_START,
+ XE_SIGID_PROBE = INTEL_SIGID_GPU_XE_SOFTWARE_START + 1,
+ XE_SIGID_WEDGED = INTEL_SIGID_GPU_XE_SOFTWARE_START + 2,
+ XE_SIGID_SURVIVABILITY = INTEL_SIGID_GPU_XE_SOFTWARE_START + 3,
+ XE_SIGID_RUNTIME_FW = INTEL_SIGID_GPU_XE_SOFTWARE_START + 4,
+ XE_SIGID_DEVICE_FW = INTEL_SIGID_GPU_XE_SOFTWARE_START + 5,
+ XE_SIGID_GT_TDR = INTEL_SIGID_GPU_XE_SOFTWARE_START + 6,
+ XE_SIGID_MEM_FAULT = INTEL_SIGID_GPU_XE_SOFTWARE_START + 7,
+ XE_SIGID_IO_BUS = INTEL_SIGID_GPU_XE_SOFTWARE_START + 8,
+
+ XE_SIGID_HW = INTEL_SIGID_GPU_XE_HARDWARE_START,
+ XE_SIGID_PCIE = INTEL_SIGID_GPU_XE_HARDWARE_START + 1,
+ XE_SIGID_DEVICE_MEMORY = INTEL_SIGID_GPU_XE_HARDWARE_START + 2,
+ XE_SIGID_CORE_COMPUTE = INTEL_SIGID_GPU_XE_HARDWARE_START + 3,
+ XE_SIGID_FABRIC = INTEL_SIGID_GPU_XE_HARDWARE_START + 4,
+ XE_SIGID_SOC_INTERNAL = INTEL_SIGID_GPU_XE_HARDWARE_START + 5,
+};
+
+#endif
diff --git a/drivers/gpu/drm/xe/regs/xe_gt_regs.h b/drivers/gpu/drm/xe/regs/xe_gt_regs.h
index 08251c7a1a4b..48c515d91882 100644
--- a/drivers/gpu/drm/xe/regs/xe_gt_regs.h
+++ b/drivers/gpu/drm/xe/regs/xe_gt_regs.h
@@ -634,6 +634,12 @@
#define GT_GFX_RC6_LOCKED XE_REG(0x138104)
#define GT_GFX_RC6 XE_REG(0x138108)
+#define GT_IA_PERF_BIAS_REG XE_REG(0x138158)
+#define GT_BIAS REG_GENMASK(31, 16)
+#define IA_BIAS REG_GENMASK(15, 0)
+#define GT_BIAS_DEFAULT 0x10
+#define IA_BIAS_DEFAULT 0x10
+
#define GT0_PERF_LIMIT_REASONS XE_REG(0x1381a8)
/* Common performance limit reason bits - available on all platforms */
#define GT0_PERF_LIMIT_REASONS_MASK 0xde3
diff --git a/drivers/gpu/drm/xe/tests/Makefile b/drivers/gpu/drm/xe/tests/Makefile
index f7aa47f11a36..0b809a252cfa 100644
--- a/drivers/gpu/drm/xe/tests/Makefile
+++ b/drivers/gpu/drm/xe/tests/Makefile
@@ -7,6 +7,7 @@ xe_live_test-y = xe_live_test_mod.o
# Normal kunit tests
obj-$(CONFIG_DRM_XE_KUNIT_TEST) += xe_test.o
xe_test-y = xe_test_mod.o \
+ xe_any_kunit.o \
xe_args_test.o \
xe_pci_test.o \
xe_rtp_tables_test.o \
diff --git a/drivers/gpu/drm/xe/tests/xe_any_kunit.c b/drivers/gpu/drm/xe/tests/xe_any_kunit.c
new file mode 100644
index 000000000000..0a5f28894cd2
--- /dev/null
+++ b/drivers/gpu/drm/xe/tests/xe_any_kunit.c
@@ -0,0 +1,213 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#include <kunit/test.h>
+
+#include "tests/xe_kunit_helpers.h"
+#include "tests/xe_pci_test.h"
+#include "xe_any.h"
+#include "xe_device.h"
+
+static void test_to_xe(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+ const struct xe_device *cxe = xe;
+ const struct xe_tile *ctile = tile;
+ const struct xe_gt *cgt = gt;
+
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(xe));
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(tile));
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(gt));
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(drm));
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(dev));
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(pdev));
+
+ KUNIT_EXPECT_PTR_EQ(test, cxe, xe_any_to_xe(cxe));
+ KUNIT_EXPECT_PTR_EQ(test, cxe, xe_any_to_xe(ctile));
+ KUNIT_EXPECT_PTR_EQ(test, cxe, xe_any_to_xe(cgt));
+}
+
+static void test_to_pdev(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(xe));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(tile));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(gt));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(drm));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(dev));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(pdev));
+
+ /* mimic early probe stage */
+ dev_set_drvdata(xe->drm.dev, NULL);
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(dev));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(pdev));
+}
+
+static void test_to_dev(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(xe));
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(tile));
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(gt));
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(drm));
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(dev));
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(pdev));
+
+ /* mimic early probe stage */
+ dev_set_drvdata(xe->drm.dev, NULL);
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(dev));
+ KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(pdev));
+}
+
+static void test_to_drm(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(xe));
+ KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(tile));
+ KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(gt));
+ KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(drm));
+ KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(dev));
+ KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(pdev));
+}
+
+static void test_if_pdev(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_TRUE(test, xe && drm && tile && gt && dev && pdev);
+ KUNIT_EXPECT_NULL(test, xe_any_if_pdev(xe));
+ KUNIT_EXPECT_NULL(test, xe_any_if_pdev(tile));
+ KUNIT_EXPECT_NULL(test, xe_any_if_pdev(gt));
+ KUNIT_EXPECT_NULL(test, xe_any_if_pdev(drm));
+ KUNIT_EXPECT_NULL(test, xe_any_if_pdev(dev));
+ KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_if_pdev(pdev));
+}
+
+static void test_if_xe(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_TRUE(test, xe && drm && tile && gt && dev && pdev);
+ KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_if_xe(xe));
+ KUNIT_EXPECT_NULL(test, xe_any_if_xe(tile));
+ KUNIT_EXPECT_NULL(test, xe_any_if_xe(gt));
+ KUNIT_EXPECT_NULL(test, xe_any_if_xe(drm));
+ KUNIT_EXPECT_NULL(test, xe_any_if_xe(dev));
+ KUNIT_EXPECT_NULL(test, xe_any_if_xe(pdev));
+}
+
+static void test_if_tile(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_TRUE(test, xe && drm && tile && gt && dev && pdev);
+ KUNIT_EXPECT_NULL(test, xe_any_if_tile(xe));
+ KUNIT_EXPECT_PTR_EQ(test, tile, xe_any_if_tile(tile));
+ KUNIT_EXPECT_NULL(test, xe_any_if_tile(gt));
+ KUNIT_EXPECT_NULL(test, xe_any_if_tile(drm));
+ KUNIT_EXPECT_NULL(test, xe_any_if_tile(dev));
+ KUNIT_EXPECT_NULL(test, xe_any_if_tile(pdev));
+}
+
+static void test_if_gt(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct drm_device *drm = &xe->drm;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+
+ KUNIT_EXPECT_TRUE(test, xe && drm && tile && gt && dev && pdev);
+ KUNIT_EXPECT_NULL(test, xe_any_if_gt(xe));
+ KUNIT_EXPECT_NULL(test, xe_any_if_gt(tile));
+ KUNIT_EXPECT_PTR_EQ(test, gt, xe_any_if_gt(gt));
+ KUNIT_EXPECT_NULL(test, xe_any_if_gt(drm));
+ KUNIT_EXPECT_NULL(test, xe_any_if_gt(dev));
+ KUNIT_EXPECT_NULL(test, xe_any_if_gt(pdev));
+}
+
+static void test_to_id(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ struct xe_tile *tile = xe_device_get_root_tile(xe);
+ struct xe_gt *gt = tile->primary_gt;
+ struct device *dev = xe->drm.dev;
+ struct pci_dev *pdev = to_pci_dev(dev);
+ const struct xe_device *cxe = xe;
+ const struct xe_tile *ctile = tile;
+ const struct xe_gt *cgt = gt;
+
+ tile->id = 1;
+ gt->info.id = 2;
+
+ KUNIT_EXPECT_EQ(test, 0, xe_any_id(xe));
+ KUNIT_EXPECT_EQ(test, 1, xe_any_id(tile));
+ KUNIT_EXPECT_EQ(test, 2, xe_any_id(gt));
+ KUNIT_EXPECT_EQ(test, 0, xe_any_id(dev));
+ KUNIT_EXPECT_EQ(test, 0, xe_any_id(pdev));
+ KUNIT_EXPECT_EQ(test, 0, xe_any_id(cxe));
+ KUNIT_EXPECT_EQ(test, 1, xe_any_id(ctile));
+ KUNIT_EXPECT_EQ(test, 2, xe_any_id(cgt));
+}
+
+static struct kunit_case xe_any_tests[] = {
+ KUNIT_CASE(test_to_xe),
+ KUNIT_CASE(test_to_dev),
+ KUNIT_CASE(test_to_pdev),
+ KUNIT_CASE(test_to_drm),
+ KUNIT_CASE(test_if_pdev),
+ KUNIT_CASE(test_if_xe),
+ KUNIT_CASE(test_if_tile),
+ KUNIT_CASE(test_if_gt),
+ KUNIT_CASE(test_to_id),
+ {}
+};
+
+static struct kunit_suite xe_any_test_suite = {
+ .name = "xe_any",
+ .test_cases = xe_any_tests,
+ .init = xe_kunit_helper_xe_device_test_init,
+};
+
+kunit_test_suite(xe_any_test_suite);
diff --git a/drivers/gpu/drm/xe/tests/xe_kunit_helpers.c b/drivers/gpu/drm/xe/tests/xe_kunit_helpers.c
index bc5156966ce9..27740b40c8ae 100644
--- a/drivers/gpu/drm/xe/tests/xe_kunit_helpers.c
+++ b/drivers/gpu/drm/xe/tests/xe_kunit_helpers.c
@@ -39,6 +39,10 @@ struct xe_device *xe_kunit_helper_alloc_xe_device(struct kunit *test,
struct xe_device,
drm, DRIVER_GEM);
KUNIT_ASSERT_NOT_ERR_OR_NULL(test, xe);
+
+ dev_set_drvdata(xe->drm.dev, &xe->drm);
+ KUNIT_ASSERT_PTR_EQ(test, xe, kdev_to_xe_device(dev));
+
return xe;
}
EXPORT_SYMBOL_IF_KUNIT(xe_kunit_helper_alloc_xe_device);
diff --git a/drivers/gpu/drm/xe/tests/xe_log_kunit.c b/drivers/gpu/drm/xe/tests/xe_log_kunit.c
new file mode 100644
index 000000000000..56c36c8a08e1
--- /dev/null
+++ b/drivers/gpu/drm/xe/tests/xe_log_kunit.c
@@ -0,0 +1,553 @@
+// SPDX-License-Identifier: GPL-2.0 AND MIT
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#include <kunit/static_stub.h>
+#include <kunit/test.h>
+#include <kunit/test-bug.h>
+
+#include "tests/xe_kunit_helpers.h"
+#include "tests/xe_pci_test.h"
+#include "xe_device.h"
+#include "xe_log.h"
+
+static void nop_dmesg_vprintk(struct pci_dev *pdev, int cper_sev, struct va_format *vaf)
+{
+}
+
+static void nop_emit_cper(struct pci_dev *pdev, int cper_sev,
+ enum xe_sigid sigid, u32 component, u32 location,
+ const void *data, size_t len, struct va_format *vaf)
+{
+}
+
+static const char *component_name(u32 component)
+{
+ switch (component) {
+#define make_component_tag_case(_CLASS, _ID, _TAG, _SIG, _NAME) \
+ case XE_LOG_COMPONENT_##_TAG: return _NAME;
+ DEFINE_XE_LOG_COMPONENTS(make_component_tag_case)
+#undef make_component_tag_case
+ }
+ return component ? "???" : "";
+}
+
+static const char *location_type(u32 location)
+{
+ u32 type = FIELD_GET(XE_LOG_LOCATION_TYPE_MASK, location);
+
+ return type == XE_LOG_LOCATION_TYPE_DEVICE ? "DEVICE" :
+ type == XE_LOG_LOCATION_TYPE_TILE ? "TILE" :
+ type == XE_LOG_LOCATION_TYPE_GT ? "GT" :
+ location ? "?" : "";
+}
+
+static void fake_emit_cper(struct pci_dev *pdev, int cper_sev,
+ enum xe_sigid sigid, u32 component, u32 location,
+ const void *data, size_t len, struct va_format *vaf)
+{
+ char msg[64];
+ int n;
+
+ pr_info("\n");
+ pr_info("CPER SEV=%u SIGID=%u\n", cper_sev, sigid);
+ pr_info("CPER DEVICE=%s\n", dev_name(&pdev->dev));
+ if (location)
+ pr_info("CPER LOCATION=%#x \t# %s.%u\n",
+ location, location_type(location),
+ FIELD_GET(XE_LOG_LOCATION_ID_MASK, location));
+ if (component)
+ pr_info("CPER COMPONENT=%#x \t# %s\n",
+ component, component_name(component));
+ if (IS_ERR(data))
+ pr_info("CPER ERR=%ld \t\t# %pe\n", PTR_ERR(data), data);
+ else if (len)
+ print_hex_dump(KERN_INFO, "CPER BIN=", DUMP_PREFIX_OFFSET,
+ 16, 1, data, len, false);
+
+ n = vscnprintf(msg, sizeof(msg), vaf->fmt, *vaf->va);
+ print_hex_dump(KERN_INFO, "CPER MSG=", DUMP_PREFIX_OFFSET, 16, 1, msg, n, true);
+ pr_info("CPER END\n");
+}
+
+static const u8 blob[] = { 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12 };
+static const u8 dead[] = { 0xde, 0xad, 0xbe, 0xef };
+
+static struct xe_tile *to_tile_safe(struct xe_device *xe)
+{
+ return xe ? &xe->tiles[1] : NULL;
+}
+
+static struct xe_gt *to_gt_safe(struct xe_device *xe)
+{
+ return xe ? to_tile_safe(xe)->primary_gt : NULL;
+}
+
+static struct pci_dev *to_pdev_safe(struct xe_device *xe)
+{
+ return xe ? xe_any_to_pdev(xe) : NULL;
+}
+
+static void demo(struct xe_device *xe)
+{
+ struct pci_dev *pdev = xe_any_to_pdev(xe);
+ struct xe_tile *tile = to_tile_safe(xe);
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ /* SW errno */
+ xe_log_emit(pdev, CPER_SEV_FATAL, XE_SIGID_PROBE,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ ERR_PTR(-ENODEV), 0, "testing %s signature\n", "software");
+
+ xe_log_err(tile, PROBE, -ENODEV, "testing %s signature\n", "software");
+ xe_log_err_corrected(gt, PROBE, -ENODEV, "testing %s signature\n", "software");
+ xe_log_info(gt, PROBE, "testing %s signature\n", "software");
+
+ /* HW data */
+ xe_log_emit(pdev, CPER_SEV_FATAL, XE_SIGID_PCIE,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ blob, sizeof(blob), "testing %s signature\n", "HARDWARE");
+ xe_log_emit_recoverable(pdev, XE_SIGID_FABRIC,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ blob, sizeof(blob), "testing %s signature\n", "HARDWARE");
+ xe_log_from_corrected(tile, XE_SIGID_DEVICE_MEMORY, XE_LOG_COMPONENT_NONE,
+ blob, sizeof(blob), "testing %s signature\n", "HARDWARE");
+ xe_log_from_info(gt, XE_SIGID_CORE_COMPUTE, XE_LOG_COMPONENT_NONE,
+ blob, sizeof(blob), "testing %s signature\n", "HARDWARE");
+}
+
+static void demo_dmesg(struct kunit *test)
+{
+ kunit_activate_static_stub(test, log_emit_cper, nop_emit_cper);
+ demo(test->priv);
+}
+
+static void demo_cper(struct kunit *test)
+{
+ kunit_activate_static_stub(test, log_emit_cper, fake_emit_cper);
+ kunit_activate_static_stub(test, log_dmesg_vprintk, nop_dmesg_vprintk);
+ demo(test->priv);
+}
+
+static const char *test_fatal(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit_fatal(pdev, XE_SIGID_PROBE,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ NULL, 0, "testing %d\n", 123);
+ return "SIGID=101 FATAL testing 123\n";
+}
+
+static const char *test_fatal_tile(struct xe_device *xe)
+{
+ struct xe_tile *tile = to_tile_safe(xe);
+
+ if (tile)
+ xe_log_from_fatal(tile, XE_SIGID_PROBE, XE_LOG_COMPONENT_NONE,
+ ERR_PTR(-ENODEV), 0, "testing %d\n", 123);
+ return "SIGID=101 FATAL (-ENODEV) Tile1: testing 123\n";
+}
+
+static const char *test_fatal_gt(struct xe_device *xe)
+{
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (gt)
+ xe_log_from_fatal(gt, XE_SIGID_PROBE, XE_LOG_COMPONENT_NONE,
+ ERR_PTR(-ENODEV), 0, "testing %d\n", 123);
+ return "SIGID=101 FATAL (-ENODEV) Tile1: GT1: testing 123\n";
+}
+
+static const char *test_fatal_comp(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_err_fatal(pdev, PROBE, -ENODEV, "testing %d\n", 123);
+ return "SIGID=101 FATAL (-ENODEV) PROBE: testing 123\n";
+}
+
+static const char *test_fatal_comp_tile(struct xe_device *xe)
+{
+ struct xe_tile *tile = to_tile_safe(xe);
+
+ if (tile)
+ xe_log_err_fatal(tile, PROBE, -ENODEV, "testing %d\n", 123);
+ return "SIGID=101 FATAL (-ENODEV) Tile1: PROBE: testing 123\n";
+}
+
+static const char *test_fatal_comp_gt(struct xe_device *xe)
+{
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (gt)
+ xe_log_err_fatal(gt, PROBE, -ENODEV, "testing %d\n", 123);
+ return "SIGID=101 FATAL (-ENODEV) Tile1: GT1: PROBE: testing 123\n";
+}
+
+static const char *test_fatal_all(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (pdev)
+ xe_log_emit(pdev, CPER_SEV_FATAL, XE_SIGID_PROBE,
+ XE_LOG_COMPONENT_PROBE, MAKE_XE_LOG_LOCATION(GT, gt->info.id),
+ ERR_PTR(-ENODEV), 0, "testing %d\n", 123);
+ return "SIGID=101 FATAL (-ENODEV) Tile1: GT1: PROBE: testing 123\n";
+}
+
+static const char *test_recoverable(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit_recoverable(pdev, XE_SIGID_RUNTIME_FW,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ ERR_PTR(-EIO), 0, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) testing 123\n";
+}
+
+static const char *test_recoverable_tile(struct xe_device *xe)
+{
+ struct xe_tile *tile = to_tile_safe(xe);
+
+ if (tile)
+ xe_log_from_recoverable(tile, XE_SIGID_RUNTIME_FW, XE_LOG_COMPONENT_NONE,
+ ERR_PTR(-EIO), 0, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) Tile1: testing 123\n";
+}
+
+static const char *test_recoverable_gt(struct xe_device *xe)
+{
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (gt)
+ xe_log_from_recoverable(gt, XE_SIGID_RUNTIME_FW, XE_LOG_COMPONENT_NONE,
+ ERR_PTR(-EIO), 0, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) Tile1: GT1: testing 123\n";
+}
+
+static const char *test_recoverable_comp(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_err(pdev, GUC, -EIO, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) GUC: testing 123\n";
+}
+
+static const char *test_recoverable_comp_tile(struct xe_device *xe)
+{
+ struct xe_tile *tile = to_tile_safe(xe);
+
+ if (tile)
+ xe_log_err(tile, GUC, -EIO, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) Tile1: GUC: testing 123\n";
+}
+
+static const char *test_recoverable_comp_gt(struct xe_device *xe)
+{
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (gt)
+ xe_log_err(gt, GUC, -EIO, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) Tile1: GT1: GUC: testing 123\n";
+}
+
+static const char *test_recoverable_all(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (pdev)
+ xe_log_emit(pdev, CPER_SEV_RECOVERABLE, XE_SIGID_RUNTIME_FW,
+ XE_LOG_COMPONENT_GUC, MAKE_XE_LOG_LOCATION(GT, gt->info.id),
+ ERR_PTR(-EIO), 0, "testing %d\n", 123);
+ return "SIGID=104 (-EIO) Tile1: GT1: GUC: testing 123\n";
+}
+
+static const char *test_info(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit(pdev, CPER_SEV_INFORMATIONAL, XE_SIGID_DEVICE_FW,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ NULL, 0, "testing %d\n", 123);
+ return "SIGID=105 testing 123\n";
+}
+
+static const char *test_info_tile(struct xe_device *xe)
+{
+ struct xe_tile *tile = to_tile_safe(xe);
+
+ if (tile)
+ xe_log_from_info(tile, XE_SIGID_DEVICE_FW, XE_LOG_COMPONENT_NONE,
+ NULL, 0, "testing %d\n", 123);
+ return "SIGID=105 Tile1: testing 123\n";
+}
+
+static const char *test_info_gt(struct xe_device *xe)
+{
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (gt)
+ xe_log_from_info(gt, XE_SIGID_DEVICE_FW, XE_LOG_COMPONENT_NONE,
+ NULL, 0, "testing %d\n", 123);
+ return "SIGID=105 Tile1: GT1: testing 123\n";
+}
+
+static const char *test_info_err(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_err_info(pdev, PCODE, -EPROTO, "testing %d\n", 123);
+ return "SIGID=105 (-EPROTO) PCODE: testing 123\n";
+}
+
+static const char *test_info_comp(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_info(pdev, PCODE, "testing %d\n", 123);
+ return "SIGID=105 PCODE: testing 123\n";
+}
+
+static const char *test_info_comp_tile(struct xe_device *xe)
+{
+ struct xe_tile *tile = to_tile_safe(xe);
+
+ if (tile)
+ xe_log_info(tile, PCODE, "testing %d\n", 123);
+ return "SIGID=105 Tile1: PCODE: testing 123\n";
+}
+
+static const char *test_info_comp_gt(struct xe_device *xe)
+{
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (gt)
+ xe_log_info(gt, PCODE, "testing %d\n", 123);
+ return "SIGID=105 Tile1: GT1: PCODE: testing 123\n";
+}
+
+static const char *test_info_all(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+ struct xe_gt *gt = to_gt_safe(xe);
+
+ if (pdev)
+ xe_log_emit(pdev, CPER_SEV_INFORMATIONAL, XE_SIGID_DEVICE_FW,
+ XE_LOG_COMPONENT_PCODE, MAKE_XE_LOG_LOCATION(GT, gt->info.id),
+ NULL, 0, "testing %d\n", 123);
+ return "SIGID=105 Tile1: GT1: PCODE: testing 123\n";
+}
+
+static const char *test_hw_fatal(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit_fatal(pdev, XE_SIGID_PCIE,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ dead, sizeof(dead), "testing %d\n", 123);
+ return "SIGID=201 FATAL (deadbeef) " HW_ERR "testing 123\n";
+}
+
+static const char *test_hw_recoverable(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit_recoverable(pdev, XE_SIGID_DEVICE_MEMORY,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ dead, sizeof(dead), "testing %d\n", 123);
+ return "SIGID=202 (deadbeef) " HW_ERR "testing 123\n";
+}
+
+static const char *test_hw_corrected(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit_corrected(pdev, XE_SIGID_DEVICE_MEMORY,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ dead, sizeof(dead), "testing %d\n", 123);
+ return "SIGID=202 CORRECTED (deadbeef) " HW_ERR "testing 123\n";
+}
+
+static const char *test_hw_informational(struct xe_device *xe)
+{
+ struct pci_dev *pdev = to_pdev_safe(xe);
+
+ if (pdev)
+ xe_log_emit_info(pdev, XE_SIGID_DEVICE_MEMORY,
+ XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE,
+ dead, sizeof(dead), "testing %d\n", 123);
+ return "SIGID=202 (deadbeef) testing 123\n";
+}
+
+static const struct log_test_param {
+ const char *(*func)(struct xe_device *xe);
+} log_test_params[] = {
+ { .func = test_fatal },
+ { .func = test_fatal_tile },
+ { .func = test_fatal_gt },
+ { .func = test_fatal_comp },
+ { .func = test_fatal_comp_tile },
+ { .func = test_fatal_comp_gt },
+ { .func = test_fatal_all },
+ { .func = test_recoverable },
+ { .func = test_recoverable_tile },
+ { .func = test_recoverable_gt },
+ { .func = test_recoverable_comp },
+ { .func = test_recoverable_comp_tile },
+ { .func = test_recoverable_comp_gt },
+ { .func = test_recoverable_all },
+ { .func = test_info },
+ { .func = test_info_tile },
+ { .func = test_info_gt },
+ { .func = test_info_err },
+ { .func = test_info_comp },
+ { .func = test_info_comp_tile },
+ { .func = test_info_comp_gt },
+ { .func = test_info_all },
+ { .func = test_hw_fatal },
+ { .func = test_hw_recoverable },
+ { .func = test_hw_corrected },
+ { .func = test_hw_informational },
+};
+
+static void log_param_get_desc(const struct log_test_param *p, char *desc)
+{
+ snprintf(desc, KUNIT_PARAM_DESC_SIZE, "%ps", p->func);
+}
+
+KUNIT_ARRAY_PARAM(log, log_test_params, log_param_get_desc);
+
+static void check_dmesg_vprintk(struct pci_dev *pdev, int cper_sev, struct va_format *vaf)
+{
+ struct kunit *test = kunit_get_current_test();
+ const struct log_test_param *param = test->param_value;
+ const char *exp = param->func(NULL);
+ char msg[64];
+ int n;
+
+ n = vsnprintf(msg, sizeof(msg), vaf->fmt, *vaf->va);
+ KUNIT_EXPECT_LT(test, n, sizeof(msg));
+ KUNIT_EXPECT_STREQ(test, msg, exp);
+}
+
+static void test_dmesg(struct kunit *test)
+{
+ const struct log_test_param *param = test->param_value;
+
+ kunit_activate_static_stub(test, log_emit_cper, nop_emit_cper);
+ kunit_activate_static_stub(test, log_dmesg_vprintk, check_dmesg_vprintk);
+
+ param->func(test->priv);
+}
+
+#define INVALID_TILEID (XE_MAX_TILES_PER_DEVICE + 1)
+#define INVALID_GTID (XE_MAX_TILES_PER_DEVICE * XE_MAX_GT_PER_TILE + 1)
+
+#define PREP_TEST_LOCATION(type, id) \
+ (FIELD_PREP_CONST(XE_LOG_LOCATION_TYPE_MASK, (type)) | \
+ FIELD_PREP_CONST(XE_LOG_LOCATION_ID_MASK, (id)))
+#define PREP_TEST_COMPONENT(class, type) \
+ (FIELD_PREP_CONST(XE_LOG_COMPONENT_CLASS_MASK, (class)) | \
+ FIELD_PREP_CONST(XE_LOG_COMPONENT_TYPE_MASK, (type)))
+
+static const struct {
+ u32 comp;
+ u32 loc;
+ const char *name;
+} invalid_params[] = {
+ { .name = "no-component no-location no-warn" },
+ { .loc = PREP_TEST_LOCATION(0, 1),
+ .name = "reserved location" },
+ { .loc = PREP_TEST_LOCATION(255, 0),
+ .name = "unknown location" },
+ { .loc = PREP_TEST_LOCATION(XE_LOG_LOCATION_TYPE_DEVICE, 1),
+ .name = "nonzero-device-id location" },
+ { .loc = PREP_TEST_LOCATION(XE_LOG_LOCATION_TYPE_TILE, INVALID_TILEID),
+ .name = "invalid-tile-id location" },
+ { .loc = PREP_TEST_LOCATION(XE_LOG_LOCATION_TYPE_GT, INVALID_GTID),
+ .name = "invalid-gt-id location" },
+ { .comp = PREP_TEST_COMPONENT(255, 0),
+ .name = "unknown component class" },
+ { .comp = PREP_TEST_COMPONENT(XE_LOG_COMPONENT_CLASS_SYSTEM, 255),
+ .name = "unknown system component" },
+ { .comp = PREP_TEST_COMPONENT(XE_LOG_COMPONENT_CLASS_HARDWARE, 255),
+ .name = "unknown hardware component" },
+ { .comp = PREP_TEST_COMPONENT(255, 1),
+ .loc = PREP_TEST_LOCATION(255, 1),
+ .name = "unknown component and location" },
+};
+
+KUNIT_ARRAY_PARAM_DESC(invalid_param, invalid_params, name);
+
+static void test_invalid(struct kunit *test)
+{
+ struct xe_device *xe = test->priv;
+ typeof(invalid_params[0]) *param = test->param_value;
+
+ struct pci_dev *pdev = xe_any_to_pdev(xe);
+
+ if (!IS_ENABLED(CONFIG_DRM_XE_DEBUG))
+ kunit_skip(test, "requires CONFIG_DRM_XE_DEBUG\n");
+
+ kunit_activate_static_stub(test, log_emit_cper, nop_emit_cper);
+ kunit_activate_static_stub(test, log_dmesg_vprintk, nop_dmesg_vprintk);
+
+ kunit_warning_suppress(test) {
+ xe_log_emit(pdev, CPER_SEV_FATAL, XE_SIGID_PROBE,
+ param->comp, param->loc,
+ NULL, 0, "testing %s\n", param->name);
+ KUNIT_EXPECT_SUPPRESSED_WARNING_COUNT(test, !!param->comp + !!param->loc);
+ }
+}
+
+static int xe_log_test_init(struct kunit *test)
+{
+ struct xe_pci_fake_data fake = {
+ .platform = XE_PVC, /* with max_remote_tiles != 0 */
+ .subplatform = XE_SUBPLATFORM_NONE,
+ .graphics_verx100 = 2001,
+ .media_verx100 = 2001,
+ };
+ struct xe_device *xe;
+
+ test->priv = &fake;
+ xe_kunit_helper_xe_device_test_init(test);
+ xe = test->priv;
+
+ KUNIT_ASSERT_NOT_NULL(test, to_tile_safe(xe));
+ KUNIT_ASSERT_NOT_NULL(test, to_gt_safe(xe));
+ KUNIT_EXPECT_EQ(test, 1, xe_any_id(to_tile_safe(xe)));
+ KUNIT_EXPECT_EQ(test, 1, xe_any_id(to_gt_safe(xe)));
+
+ return 0;
+}
+
+static struct kunit_case xe_log_test_cases[] = {
+ KUNIT_CASE(demo_cper),
+ KUNIT_CASE(demo_dmesg),
+ KUNIT_CASE_PARAM(test_dmesg, log_gen_params),
+ KUNIT_CASE_PARAM(test_invalid, invalid_param_gen_params),
+ {}
+};
+
+static struct kunit_suite xe_log_suite = {
+ .name = "xe_log",
+ .test_cases = xe_log_test_cases,
+ .init = xe_log_test_init,
+};
+
+kunit_test_suites(&xe_log_suite);
diff --git a/drivers/gpu/drm/xe/tests/xe_migrate.c b/drivers/gpu/drm/xe/tests/xe_migrate.c
index 3c1be809be82..f10d9513747b 100644
--- a/drivers/gpu/drm/xe/tests/xe_migrate.c
+++ b/drivers/gpu/drm/xe/tests/xe_migrate.c
@@ -198,8 +198,7 @@ static void xe_migrate_sanity_test(struct xe_migrate *m, struct kunit *test,
err = xe_bo_vmap(bo);
if (err) {
- KUNIT_FAIL(test, "Failed to vmap our pagetables: %li\n",
- PTR_ERR(bo));
+ KUNIT_FAIL(test, "Failed to vmap our pagetables: %d\n", err);
return;
}
diff --git a/drivers/gpu/drm/xe/xe_any.h b/drivers/gpu/drm/xe/xe_any.h
new file mode 100644
index 000000000000..5d97afa76915
--- /dev/null
+++ b/drivers/gpu/drm/xe/xe_any.h
@@ -0,0 +1,137 @@
+/* SPDX-License-Identifier: MIT */
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#ifndef _XE_ANY_H_
+#define _XE_ANY_H_
+
+#include "xe_device.h"
+
+#define __xe_any_to_self_assoc(type, any) \
+ const type * : (any), \
+ type * : (any)
+
+/**
+ * xe_any_if_type() - Get the pointer only if it is @type pointer.
+ * @any: any pointer
+ * @type: data type to look for
+ *
+ * Return: the @type pointer or NULL.
+ */
+#define xe_any_if_type(any, type) \
+ _Generic((any), \
+ __xe_any_to_self_assoc(type, (any)), \
+ default : NULL)
+
+/**
+ * xe_any_if_gt() - Get the pointer only if it is &xe_gt.
+ * @any: any pointer
+ *
+ * Return: the @xe_gt pointer or NULL.
+ */
+#define xe_any_if_gt(any) xe_any_if_type((any), struct xe_gt)
+
+/**
+ * xe_any_if_tile() - Get the pointer only if it is &xe_tile.
+ * @any: any pointer
+ *
+ * Return: the @xe_tile pointer or NULL.
+ */
+#define xe_any_if_tile(any) xe_any_if_type((any), struct xe_tile)
+
+/**
+ * xe_any_if_xe() - Get the pointer only if it is &xe_device.
+ * @any: any pointer
+ *
+ * Return: the @xe_device pointer or NULL.
+ */
+#define xe_any_if_xe(any) xe_any_if_type((any), struct xe_device)
+
+/**
+ * xe_any_if_pdev() - Get the pointer only if it is &pci_dev.
+ * @any: any pointer
+ *
+ * Return: the @pci_dev pointer or NULL.
+ */
+#define xe_any_if_pdev(any) xe_any_if_type((any), struct pci_dev)
+
+#define __xe_any_to_other_assoc(const, from, other, p) \
+ const struct from * : __##from##_to_##other((const struct from *)(p))
+
+#define __xe_tile_to_xe_device(p) tile_to_xe(p)
+#define __xe_gt_to_xe_device(p) gt_to_xe(p)
+#define __pci_dev_to_xe_device(p) pdev_to_xe_device(p)
+#define __device_to_xe_device(p) kdev_to_xe_device(p)
+#define __drm_device_to_xe_device(p) to_xe_device(p)
+#define __pci_dev_to_device(p) (&(p)->dev)
+
+/**
+ * xe_any_to_xe() - Obtain the &xe_device pointer.
+ * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer
+ *
+ * Return: the @xe_device pointer or backpointer.
+ */
+#define xe_any_to_xe(any) \
+ _Generic((any), \
+ __xe_any_to_self_assoc(struct xe_device, (any)), \
+ __xe_any_to_other_assoc(/* */, xe_tile, xe_device, (any)), \
+ __xe_any_to_other_assoc(const, xe_tile, xe_device, (any)), \
+ __xe_any_to_other_assoc(/* */, xe_gt, xe_device, (any)), \
+ __xe_any_to_other_assoc(const, xe_gt, xe_device, (any)), \
+ __xe_any_to_other_assoc(, drm_device, xe_device, (any)), \
+ __xe_any_to_other_assoc(, pci_dev, xe_device, (any)), \
+ __xe_any_to_other_assoc(, device, xe_device, (any)))
+
+/**
+ * xe_any_to_drm() - Obtain the &drm_device pointer.
+ * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer
+ *
+ * Return: the @drm_device pointer or backpointer.
+ */
+#define xe_any_to_drm(any) \
+ _Generic((any), \
+ __xe_any_to_self_assoc(struct drm_device, (any)), \
+ default : &xe_any_to_xe(any)->drm)
+
+/**
+ * xe_any_to_dev() - Obtain the &device pointer.
+ * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer
+ *
+ * Return: the @device pointer or backpointer.
+ */
+#define xe_any_to_dev(any) \
+ _Generic((any), \
+ __xe_any_to_self_assoc(struct device, (any)), \
+ __xe_any_to_other_assoc(, pci_dev, device, (any)), \
+ default : xe_any_to_drm(any)->dev)
+
+/**
+ * xe_any_to_pdev() - Obtain the &pci_dev pointer.
+ * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer
+ *
+ * Return: the @pci_dev pointer or backpointer.
+ */
+#define xe_any_to_pdev(any) \
+ _Generic((any), \
+ __xe_any_to_self_assoc(struct pci_dev, (any)), \
+ default : to_pci_dev(xe_any_to_dev(any)))
+
+#define __xe_tile_to_id(p) ((p)->id)
+#define __xe_gt_to_id(p) ((p)->info.id)
+
+/**
+ * xe_any_id() - Get the identifier of the underlying object.
+ * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer
+ *
+ * Return: the identifier of the object, or 0 if not applicable/available.
+ */
+#define xe_any_id(any) \
+ _Generic((any), \
+ __xe_any_to_other_assoc(/* */, xe_tile, id, (any)), \
+ __xe_any_to_other_assoc(const, xe_tile, id, (any)), \
+ __xe_any_to_other_assoc(/* */, xe_gt, id, (any)), \
+ __xe_any_to_other_assoc(const, xe_gt, id, (any)), \
+ default : 0)
+
+#endif
diff --git a/drivers/gpu/drm/xe/xe_debugfs.c b/drivers/gpu/drm/xe/xe_debugfs.c
index 8de78cd0aa03..28135f84e286 100644
--- a/drivers/gpu/drm/xe/xe_debugfs.c
+++ b/drivers/gpu/drm/xe/xe_debugfs.c
@@ -22,6 +22,7 @@
#include "xe_guc_ads.h"
#include "xe_hw_engine.h"
#include "xe_mmio.h"
+#include "xe_pagefault.h"
#include "xe_pcode.h"
#include "xe_pm.h"
#include "xe_psmi.h"
@@ -42,6 +43,7 @@
DECLARE_FAULT_ATTR(gt_reset_failure);
DECLARE_FAULT_ATTR(inject_csc_hw_error);
+DECLARE_FAULT_ATTR(wedge_cold_reset);
static bool csc_hw_error_available(struct xe_device *xe)
{
@@ -62,6 +64,8 @@ static struct {
{ .name = "inject_csc_hw_error",
.attr = &inject_csc_hw_error,
.is_visible = csc_hw_error_available },
+ { .name = "wedge_cold_reset",
+ .attr = &wedge_cold_reset },
};
/*
@@ -76,6 +80,7 @@ bool xe_fault_##name(void) \
FAULT_ACTION(gt_reset, gt_reset_failure)
FAULT_ACTION(csc_hw_error, inject_csc_hw_error)
+FAULT_ACTION(wedge_cold_reset, wedge_cold_reset)
static void xe_fault_inject_debugfs_register(struct xe_device *xe,
struct dentry *root)
@@ -194,6 +199,15 @@ static int sriov_info(struct seq_file *m, void *data)
return 0;
}
+static int pagefault_info(struct seq_file *m, void *data)
+{
+ struct xe_device *xe = node_to_xe(m->private);
+ struct drm_printer p = drm_seq_file_printer(m);
+
+ xe_pagefault_print_info(xe, &p);
+ return 0;
+}
+
static int workarounds(struct xe_device *xe, struct drm_printer *p)
{
guard(xe_pm_runtime)(xe);
@@ -285,6 +299,7 @@ static const struct drm_info_list debugfs_list[] = {
{"info", info, 0},
{ .name = "sriov_info", .show = sriov_info, },
{ .name = "workarounds", .show = workaround_info, },
+ { .name = "pagefault_info", .show = pagefault_info, },
};
static const struct drm_info_list pcode_info_debugfs[] = {
diff --git a/drivers/gpu/drm/xe/xe_debugfs.h b/drivers/gpu/drm/xe/xe_debugfs.h
index cd56f7442b99..0dcd28fd7dc0 100644
--- a/drivers/gpu/drm/xe/xe_debugfs.h
+++ b/drivers/gpu/drm/xe/xe_debugfs.h
@@ -13,10 +13,12 @@ struct xe_device;
#ifdef CONFIG_DEBUG_FS
bool xe_fault_gt_reset(void);
bool xe_fault_csc_hw_error(void);
+bool xe_fault_wedge_cold_reset(void);
void xe_debugfs_register(struct xe_device *xe);
#else
static inline bool xe_fault_gt_reset(void) { return false; }
static inline bool xe_fault_csc_hw_error(void) { return false; }
+static inline bool xe_fault_wedge_cold_reset(void) { return false; }
static inline void xe_debugfs_register(struct xe_device *xe) { }
#endif
diff --git a/drivers/gpu/drm/xe/xe_defaults.h b/drivers/gpu/drm/xe/xe_defaults.h
index c8ae1d5f3d60..0884224ef7c7 100644
--- a/drivers/gpu/drm/xe/xe_defaults.h
+++ b/drivers/gpu/drm/xe/xe_defaults.h
@@ -22,5 +22,6 @@
#define XE_DEFAULT_WEDGED_MODE XE_WEDGED_MODE_UPON_CRITICAL_ERROR
#define XE_DEFAULT_WEDGED_MODE_STR "upon-critical-error"
#define XE_DEFAULT_SVM_NOTIFIER_SIZE 512
+#define XE_DEFAULT_NUM_PF_WORK 2
#endif
diff --git a/drivers/gpu/drm/xe/xe_device.c b/drivers/gpu/drm/xe/xe_device.c
index ee732e5495f7..396d02eb2af8 100644
--- a/drivers/gpu/drm/xe/xe_device.c
+++ b/drivers/gpu/drm/xe/xe_device.c
@@ -48,6 +48,7 @@
#include "xe_i2c.h"
#include "xe_irq.h"
#include "xe_late_bind_fw.h"
+#include "xe_log.h"
#include "xe_mmio.h"
#include "xe_module.h"
#include "xe_nvm.h"
@@ -513,6 +514,17 @@ struct xe_device *xe_device_create(struct pci_dev *pdev)
}
ALLOW_ERROR_INJECTION(xe_device_create, ERRNO); /* See xe_pci_probe() */
+static void xe_device_parse_modparam(struct xe_device *xe)
+{
+ xe->atomic_svm_timeslice_ms = 5;
+ xe->min_run_period_lr_ms = 5;
+ xe->info.num_pf_work = xe_modparam.num_pf_work;
+ if (xe->info.num_pf_work < 1)
+ xe->info.num_pf_work = 1;
+ else if (xe->info.num_pf_work > XE_PAGEFAULT_WORK_MAX)
+ xe->info.num_pf_work = XE_PAGEFAULT_WORK_MAX;
+}
+
/**
* xe_device_init_early() - Initialize a new &xe_device instance
* @xe: the &xe_device to initialize
@@ -539,8 +551,7 @@ int xe_device_init_early(struct xe_device *xe)
if (err)
return err;
- xe->atomic_svm_timeslice_ms = 5;
- xe->min_run_period_lr_ms = 5;
+ xe_device_parse_modparam(xe);
err = xe_irq_init(xe);
if (err)
@@ -1432,6 +1443,9 @@ void xe_device_set_wedged_method(struct xe_device *xe, unsigned long method)
xe->wedged.method = method;
}
+#define WEDGED_URL "https://docs.kernel.org/gpu/drm-uapi.html#device-wedging"
+#define XE_BUG_URL "https://gitlab.freedesktop.org/drm/xe/kernel/issues/new"
+
/**
* xe_device_declare_wedged - Declare device wedged
* @xe: xe device instance
@@ -1463,12 +1477,12 @@ void xe_device_declare_wedged(struct xe_device *xe)
if (!atomic_xchg(&xe->wedged.flag, 1)) {
xe->needs_flr_on_fini = true;
xe_pm_runtime_get_noresume(xe);
- drm_err(&xe->drm,
- "CRITICAL: Xe has declared device %s as wedged.\n"
- "IOCTLs and executions are blocked.\n"
- "For recovery procedure, refer to https://docs.kernel.org/gpu/drm-uapi.html#device-wedging\n"
- "Please file a _new_ bug report at https://gitlab.freedesktop.org/drm/xe/kernel/issues/new\n",
- dev_name(xe->drm.dev));
+
+ xe_log_err_fatal(xe, WEDGED, -EIO, "Device declared wedged!\n");
+ xe_err_once(xe, "IOCTLs and executions are now blocked!\n"
+ "For recovery procedure, refer to %s\n"
+ "Please file a _new_ bug report at %s\n",
+ WEDGED_URL, XE_BUG_URL);
}
for_each_gt(gt, xe, id)
diff --git a/drivers/gpu/drm/xe/xe_device_types.h b/drivers/gpu/drm/xe/xe_device_types.h
index 03a7bb08adf7..180d450a6deb 100644
--- a/drivers/gpu/drm/xe/xe_device_types.h
+++ b/drivers/gpu/drm/xe/xe_device_types.h
@@ -140,6 +140,8 @@ struct xe_device {
u8 revid;
/** @info.step: stepping information for each IP */
struct xe_step_info step;
+ /** @info.num_pf_work: Number of page fault work thread */
+ int num_pf_work;
/** @info.dma_mask_size: DMA address bits */
u8 dma_mask_size;
/** @info.vram_flags: Vram flags */
@@ -320,18 +322,19 @@ struct xe_device {
struct xarray asid_to_vm;
/** @usm.next_asid: next ASID, used to cyclical alloc asids */
u32 next_asid;
+ /** @usm.current_pf_work: current page fault work item */
+ u32 current_pf_work;
/** @usm.lock: protects UM state */
struct rw_semaphore lock;
- /** @usm.pf_wq: page fault work queue, unbound, high priority */
- struct workqueue_struct *pf_wq;
- /*
- * We pick 4 here because, in the current implementation, it
- * yields the best bandwidth utilization of the kernel paging
- * engine.
- */
-#define XE_PAGEFAULT_QUEUE_COUNT 4
- /** @usm.pf_queue: Page fault queues */
- struct xe_pagefault_queue pf_queue[XE_PAGEFAULT_QUEUE_COUNT];
+ /** @usm.pagefault_wq: page fault work queue, unbound, high priority */
+ struct workqueue_struct *pagefault_wq;
+ /** @usm.prefetch_wq: threaded prefetch work queue, unbound */
+ struct workqueue_struct *prefetch_wq;
+#define XE_PAGEFAULT_WORK_MAX 8
+ /** @usm.pf_workers: Page fault workers */
+ struct xe_pagefault_work pf_workers[XE_PAGEFAULT_WORK_MAX];
+ /** @usm.pf_queue: Page fault queue */
+ struct xe_pagefault_queue pf_queue;
#if IS_ENABLED(CONFIG_DRM_XE_PAGEMAP)
/** @usm.dpagemap_shrinker: Shrinker for unused pagemaps */
struct drm_pagemap_shrinker *dpagemap_shrinker;
diff --git a/drivers/gpu/drm/xe/xe_drm_ras.c b/drivers/gpu/drm/xe/xe_drm_ras.c
index 984fa67b9015..78184b6ea7d4 100644
--- a/drivers/gpu/drm/xe/xe_drm_ras.c
+++ b/drivers/gpu/drm/xe/xe_drm_ras.c
@@ -186,6 +186,37 @@ null_info:
}
/**
+ * xe_drm_ras_event() - Report drm-ras error event to userspace
+ * @xe: xe device structure
+ * @component: error component (see &enum drm_xe_ras_error_component)
+ * @severity: error severity (see &enum drm_xe_ras_error_severity)
+ * @value: value of error counter
+ *
+ * Report an error-event to userspace.
+ */
+void xe_drm_ras_event(struct xe_device *xe, u32 component, u32 severity, u32 value)
+{
+ struct xe_drm_ras *ras = &xe->ras;
+ struct xe_drm_ras_counter *info = ras->info[severity];
+ struct drm_ras_node *node;
+ int ret;
+
+ /* Event is supported only if drm-ras is enabled */
+ if (!xe->info.has_drm_ras)
+ return;
+
+ node = &ras->node[severity];
+
+ if (!info || !info[component].name)
+ return;
+
+ ret = drm_ras_nl_error_event(node, component, info[component].name, value);
+ if (ret)
+ drm_err_ratelimited(&xe->drm, "drm-ras error-event failed: %d for %s %s\n", ret,
+ info[component].name, error_severity[severity]);
+}
+
+/**
* xe_drm_ras_init() - Initialize DRM RAS
* @xe: xe device instance
*
diff --git a/drivers/gpu/drm/xe/xe_drm_ras.h b/drivers/gpu/drm/xe/xe_drm_ras.h
index 365c70e93e82..723229b2cffb 100644
--- a/drivers/gpu/drm/xe/xe_drm_ras.h
+++ b/drivers/gpu/drm/xe/xe_drm_ras.h
@@ -5,11 +5,14 @@
#ifndef _XE_DRM_RAS_H_
#define _XE_DRM_RAS_H_
+#include <linux/types.h>
+
struct xe_device;
#define for_each_error_severity(i) \
for (i = 0; i < DRM_XE_RAS_ERR_SEV_MAX; i++)
int xe_drm_ras_init(struct xe_device *xe);
+void xe_drm_ras_event(struct xe_device *xe, u32 component, u32 severity, u32 value);
#endif
diff --git a/drivers/gpu/drm/xe/xe_exec_queue.c b/drivers/gpu/drm/xe/xe_exec_queue.c
index cd053de3c6b5..c4213bb9c137 100644
--- a/drivers/gpu/drm/xe/xe_exec_queue.c
+++ b/drivers/gpu/drm/xe/xe_exec_queue.c
@@ -207,9 +207,6 @@ static struct xe_exec_queue *__xe_exec_queue_alloc(struct xe_device *xe,
struct xe_gt *gt = hwe->gt;
int err;
- /* only kernel queues can be permanent */
- XE_WARN_ON((flags & EXEC_QUEUE_FLAG_PERMANENT) && !(flags & EXEC_QUEUE_FLAG_KERNEL));
-
q = kzalloc_flex(*q, lrc, width);
if (!q)
return ERR_PTR(-ENOMEM);
@@ -1121,6 +1118,9 @@ static int exec_queue_user_ext_set_property(struct xe_device *xe,
if (!exec_queue_set_property_funcs[idx])
return -EINVAL;
+ if (XE_IOCTL_DBG(xe, *properties & BIT_ULL(idx)))
+ return -EINVAL;
+
*properties |= BIT_ULL(idx);
err = exec_queue_user_ext_check(q, *properties);
if (XE_IOCTL_DBG(xe, err))
diff --git a/drivers/gpu/drm/xe/xe_exec_queue_types.h b/drivers/gpu/drm/xe/xe_exec_queue_types.h
index 53b6c0bf4849..95f75d61a647 100644
--- a/drivers/gpu/drm/xe/xe_exec_queue_types.h
+++ b/drivers/gpu/drm/xe/xe_exec_queue_types.h
@@ -70,6 +70,13 @@ struct xe_exec_queue_group {
spinlock_t suspend_lock;
/** @sync_pending: CGP_SYNC_DONE g2h response pending */
bool sync_pending;
+ /**
+ * @cgp_update_q: Queue that issued the currently outstanding (sent)
+ * CGP_SYNC or REGISTER_CONTEXT_MULTI_QUEUE; NULL when none is
+ * outstanding. Used during VF recovery to identify and replay the
+ * message whose CGP_SYNC_DONE was not received.
+ */
+ struct xe_exec_queue *cgp_update_q;
/** @banned: Group banned */
bool banned;
/** @stopped: Group is stopped, protected by list_lock */
@@ -128,20 +135,18 @@ struct xe_exec_queue {
/* queue used for kernel submission only */
#define EXEC_QUEUE_FLAG_KERNEL BIT(0)
-/* kernel engine only destroyed at driver unload */
-#define EXEC_QUEUE_FLAG_PERMANENT BIT(1)
/* for VM jobs. Caller needs to hold rpm ref when creating queue with this flag */
-#define EXEC_QUEUE_FLAG_VM BIT(2)
+#define EXEC_QUEUE_FLAG_VM BIT(1)
/* child of VM queue for multi-tile VM jobs */
-#define EXEC_QUEUE_FLAG_BIND_ENGINE_CHILD BIT(3)
+#define EXEC_QUEUE_FLAG_BIND_ENGINE_CHILD BIT(2)
/* kernel exec_queue only, set priority to highest level */
-#define EXEC_QUEUE_FLAG_HIGH_PRIORITY BIT(4)
+#define EXEC_QUEUE_FLAG_HIGH_PRIORITY BIT(3)
/* flag to indicate low latency hint to guc */
-#define EXEC_QUEUE_FLAG_LOW_LATENCY BIT(5)
+#define EXEC_QUEUE_FLAG_LOW_LATENCY BIT(4)
/* for migration (kernel copy, clear, bind) jobs */
-#define EXEC_QUEUE_FLAG_MIGRATE BIT(6)
+#define EXEC_QUEUE_FLAG_MIGRATE BIT(5)
/* for programming COMMON_SLICE_CHICKEN3 on first submission */
-#define EXEC_QUEUE_FLAG_DISABLE_STATE_CACHE_PERF_FIX BIT(7)
+#define EXEC_QUEUE_FLAG_DISABLE_STATE_CACHE_PERF_FIX BIT(6)
/**
* @flags: flags for this exec queue, should statically setup aside from ban
diff --git a/drivers/gpu/drm/xe/xe_gsc.c b/drivers/gpu/drm/xe/xe_gsc.c
index aab59dc647fb..524ac56bdcc7 100644
--- a/drivers/gpu/drm/xe/xe_gsc.c
+++ b/drivers/gpu/drm/xe/xe_gsc.c
@@ -478,8 +478,7 @@ int xe_gsc_init_post_hwconfig(struct xe_gsc *gsc)
q = xe_exec_queue_create(xe, NULL,
BIT(hwe->logical_instance), 1, hwe,
- EXEC_QUEUE_FLAG_KERNEL |
- EXEC_QUEUE_FLAG_PERMANENT, 0);
+ EXEC_QUEUE_FLAG_KERNEL, 0);
if (IS_ERR(q)) {
xe_gt_err(gt, "Failed to create queue for GSC submission\n");
return PTR_ERR(q);
diff --git a/drivers/gpu/drm/xe/xe_gt.c b/drivers/gpu/drm/xe/xe_gt.c
index dfdacc0f6de9..6805e0d3bf21 100644
--- a/drivers/gpu/drm/xe/xe_gt.c
+++ b/drivers/gpu/drm/xe/xe_gt.c
@@ -48,6 +48,7 @@
#include "xe_hw_engine_class_sysfs.h"
#include "xe_irq.h"
#include "xe_lmtt.h"
+#include "xe_log.h"
#include "xe_lrc.h"
#include "xe_map.h"
#include "xe_migrate.h"
@@ -925,7 +926,7 @@ static void gt_reset_worker(struct work_struct *w)
if (!xe_device_uc_enabled(gt_to_xe(gt)))
goto err_pm_put;
- xe_gt_info(gt, "reset started\n");
+ xe_log_info(gt, GT, "reset started\n");
if (xe_fault_gt_reset()) {
err = -ECANCELED;
@@ -964,7 +965,7 @@ static void gt_reset_worker(struct work_struct *w)
/* Pair with get while enqueueing the work in xe_gt_reset_async() */
xe_pm_runtime_put(gt_to_xe(gt));
- xe_gt_info(gt, "reset done\n");
+ xe_log_info(gt, GT, "reset done\n");
return;
@@ -973,7 +974,7 @@ err_out:
XE_WARN_ON(xe_uc_start(&gt->uc));
err_fail:
- xe_gt_err(gt, "reset failed (%pe)\n", ERR_PTR(err));
+ xe_log_err_fatal(gt, GT, err, "reset failed\n");
xe_device_declare_wedged(gt_to_xe(gt));
err_pm_put:
xe_pm_runtime_put(gt_to_xe(gt));
diff --git a/drivers/gpu/drm/xe/xe_gt_debugfs.c b/drivers/gpu/drm/xe/xe_gt_debugfs.c
index c38bcacb27e4..ea78b57b1c31 100644
--- a/drivers/gpu/drm/xe/xe_gt_debugfs.c
+++ b/drivers/gpu/drm/xe/xe_gt_debugfs.c
@@ -9,7 +9,9 @@
#include <drm/drm_debugfs.h>
#include <drm/drm_managed.h>
+#include <linux/math.h>
+#include "regs/xe_gt_regs.h"
#include "xe_device.h"
#include "xe_force_wake.h"
#include "xe_gt.h"
@@ -22,6 +24,7 @@
#include "xe_guc_hwconfig.h"
#include "xe_hw_engine.h"
#include "xe_lrc.h"
+#include "xe_mmio.h"
#include "xe_mocs.h"
#include "xe_pat.h"
#include "xe_pm.h"
@@ -336,6 +339,80 @@ static int force_reset_sync_show(struct seq_file *s, void *unused)
}
DEFINE_SHOW_STORE_ATTRIBUTE(force_reset_sync);
+#define U1_15_ONE 0x8000
+#define U1_15_INT_BITS GENMASK(15, 15)
+#define U1_15_FRACTION_BITS GENMASK(14, 0)
+
+static void u1_15_decode(u16 num, u16 *i, u32 *frac)
+{
+ /*
+ * In U1.15 format, uppermost bit is integer value and the
+ * rest 15 are the fraction.
+ */
+
+ *i = FIELD_GET(U1_15_INT_BITS, num);
+ *frac = FIELD_GET(U1_15_FRACTION_BITS, num);
+}
+
+static void u1_15_decode_decimal(u16 value, u16 *i, u32 *frac, int digits)
+{
+ u1_15_decode(value, i, frac);
+ *frac = (*frac * int_pow(10, digits)) / (FIELD_MAX(U1_15_FRACTION_BITS) + 1);
+}
+
+static int gt_ia_bias_show(struct seq_file *s, void *unused)
+{
+ struct xe_gt *gt = s->private;
+ struct xe_device *xe = gt_to_xe(gt);
+ u32 val;
+ u32 ia_frac, gt_frac;
+ u16 ia_raw, gt_raw;
+ u16 ia_int, gt_int;
+
+ guard(xe_pm_runtime)(xe);
+ val = xe_mmio_read32(&gt->mmio, GT_IA_PERF_BIAS_REG);
+
+ ia_raw = REG_FIELD_GET(IA_BIAS, val);
+ gt_raw = REG_FIELD_GET(GT_BIAS, val);
+
+ u1_15_decode_decimal(ia_raw, &ia_int, &ia_frac, 4);
+ u1_15_decode_decimal(gt_raw, &gt_int, &gt_frac, 4);
+
+ seq_printf(s, "0x%x (GT: %u.%04u, IA: %u.%04u)\n",
+ val, gt_int, gt_frac, ia_int, ia_frac);
+
+ return 0;
+}
+
+static ssize_t gt_ia_bias_write(struct file *file,
+ const char __user *userbuf,
+ size_t count, loff_t *ppos)
+{
+ struct seq_file *s = file->private_data;
+ struct xe_gt *gt = s->private;
+ struct xe_device *xe = gt_to_xe(gt);
+ u32 val;
+ int ret;
+
+ ret = kstrtou32_from_user(userbuf, count, 0, &val);
+ if (ret)
+ return ret;
+
+ if (REG_FIELD_GET(IA_BIAS, val) > U1_15_ONE ||
+ REG_FIELD_GET(GT_BIAS, val) > U1_15_ONE)
+ return -EINVAL;
+
+ if (REG_FIELD_GET(IA_BIAS, val) < IA_BIAS_DEFAULT ||
+ REG_FIELD_GET(GT_BIAS, val) < GT_BIAS_DEFAULT)
+ return -EINVAL;
+
+ guard(xe_pm_runtime)(xe);
+ xe_mmio_write32(&gt->mmio, GT_IA_PERF_BIAS_REG, val);
+
+ return count;
+}
+DEFINE_SHOW_STORE_ATTRIBUTE(gt_ia_bias);
+
void xe_gt_debugfs_register(struct xe_gt *gt)
{
struct xe_device *xe = gt_to_xe(gt);
@@ -378,6 +455,9 @@ void xe_gt_debugfs_register(struct xe_gt *gt)
ARRAY_SIZE(pf_only_debugfs_list),
root, minor);
+ if (xe_gt_is_main_type(gt) && !IS_DGFX(xe) && !IS_SRIOV_VF(xe))
+ debugfs_create_file("gt_ia_bias", 0600, root, gt, &gt_ia_bias_fops);
+
xe_uc_debugfs_register(&gt->uc, root);
if (IS_SRIOV_PF(xe))
diff --git a/drivers/gpu/drm/xe/xe_gt_idle.c b/drivers/gpu/drm/xe/xe_gt_idle.c
index 04b24e1c8b78..7dc9873aef54 100644
--- a/drivers/gpu/drm/xe/xe_gt_idle.c
+++ b/drivers/gpu/drm/xe/xe_gt_idle.c
@@ -248,7 +248,8 @@ int xe_gt_idle_pg_print(struct xe_gt *gt, struct drm_printer *p)
pg_status = xe_mmio_read32(&gt->mmio, POWERGATE_DOMAIN_STATUS);
}
- if (gt->info.engine_mask & XE_HW_ENGINE_RCS_MASK) {
+ if (gt->info.engine_mask &
+ (XE_HW_ENGINE_RCS_MASK | XE_HW_ENGINE_CCS_MASK)) {
drm_printf(p, "Render Power Gating Enabled: %s\n",
str_yes_no(pg_enabled & RENDER_POWERGATE_ENABLE));
diff --git a/drivers/gpu/drm/xe/xe_gt_printk.h b/drivers/gpu/drm/xe/xe_gt_printk.h
index 1313d32862db..1598098d2987 100644
--- a/drivers/gpu/drm/xe/xe_gt_printk.h
+++ b/drivers/gpu/drm/xe/xe_gt_printk.h
@@ -35,6 +35,9 @@
#define xe_gt_dbg(_gt, _fmt, ...) \
xe_gt_printk((_gt), dbg, _fmt, ##__VA_ARGS__)
+#define xe_gt_dbg_ratelimited(_gt, _fmt, ...) \
+ xe_gt_printk((_gt), dbg_ratelimited, _fmt, ##__VA_ARGS__)
+
#define xe_gt_WARN_type(_gt, _type, _condition, _fmt, ...) \
xe_tile_WARN##_type((_gt)->tile, _condition, _fmt, ## __VA_ARGS__)
diff --git a/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c b/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c
index 2c9b85b84b1b..be0a413ee17c 100644
--- a/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c
+++ b/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c
@@ -668,7 +668,7 @@ static int pf_config_bulk_set_u64_done(struct xe_gt *gt, unsigned int first, uns
}
/**
- * xe_gt_sriov_pf_config_bulk_set_ggtt - Provision many VFs with GGTT.
+ * xe_gt_sriov_pf_config_bulk_set_ggtt_locked() - Provision many VFs with GGTT.
* @gt: the &xe_gt (can't be media)
* @vfid: starting VF identifier (can't be 0)
* @num_vfs: number of VFs to provision
@@ -678,31 +678,49 @@ static int pf_config_bulk_set_u64_done(struct xe_gt *gt, unsigned int first, uns
*
* Return: 0 on success or a negative error code on failure.
*/
-int xe_gt_sriov_pf_config_bulk_set_ggtt(struct xe_gt *gt, unsigned int vfid,
- unsigned int num_vfs, u64 size)
+int xe_gt_sriov_pf_config_bulk_set_ggtt_locked(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u64 size)
{
unsigned int n;
int err = 0;
xe_gt_assert(gt, vfid);
xe_gt_assert(gt, xe_gt_is_main_type(gt));
+ lockdep_assert_held(xe_gt_sriov_pf_master_mutex(gt));
if (!num_vfs)
return 0;
- mutex_lock(xe_gt_sriov_pf_master_mutex(gt));
for (n = vfid; n < vfid + num_vfs; n++) {
err = pf_provision_vf_ggtt(gt, n, size);
if (err)
break;
}
- mutex_unlock(xe_gt_sriov_pf_master_mutex(gt));
return pf_config_bulk_set_u64_done(gt, vfid, num_vfs, size,
- xe_gt_sriov_pf_config_get_ggtt,
+ pf_get_vf_config_ggtt,
"GGTT", n, err);
}
+/**
+ * xe_gt_sriov_pf_config_bulk_set_ggtt() - Provision many VFs with GGTT.
+ * @gt: the &xe_gt (can't be media)
+ * @vfid: starting VF identifier (can't be 0)
+ * @num_vfs: number of VFs to provision
+ * @size: requested GGTT size
+ *
+ * This function can only be called on PF.
+ *
+ * Return: 0 on success or a negative error code on failure.
+ */
+int xe_gt_sriov_pf_config_bulk_set_ggtt(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u64 size)
+{
+ guard(mutex)(xe_gt_sriov_pf_master_mutex(gt));
+
+ return xe_gt_sriov_pf_config_bulk_set_ggtt_locked(gt, vfid, num_vfs, size);
+}
+
/* Return: size of the largest continuous GGTT region */
static u64 pf_get_max_ggtt(struct xe_gt *gt)
{
@@ -775,10 +793,9 @@ int xe_gt_sriov_pf_config_set_fair_ggtt(struct xe_gt *gt, unsigned int vfid,
xe_gt_assert(gt, num_vfs);
xe_gt_assert(gt, xe_gt_is_main_type(gt));
- mutex_lock(xe_gt_sriov_pf_master_mutex(gt));
- fair = pf_estimate_fair_ggtt(gt, num_vfs);
- mutex_unlock(xe_gt_sriov_pf_master_mutex(gt));
+ guard(mutex)(xe_gt_sriov_pf_master_mutex(gt));
+ fair = pf_estimate_fair_ggtt(gt, num_vfs);
if (!fair)
return -ENOSPC;
@@ -787,7 +804,7 @@ int xe_gt_sriov_pf_config_set_fair_ggtt(struct xe_gt *gt, unsigned int vfid,
xe_gt_sriov_info(gt, "Using non-profile provisioning (%s %llu vs %llu)\n",
"GGTT", fair, profile);
- return xe_gt_sriov_pf_config_bulk_set_ggtt(gt, vfid, num_vfs, fair);
+ return xe_gt_sriov_pf_config_bulk_set_ggtt_locked(gt, vfid, num_vfs, fair);
}
/**
@@ -1099,7 +1116,7 @@ static int pf_config_bulk_set_u32_done(struct xe_gt *gt, unsigned int first, uns
}
/**
- * xe_gt_sriov_pf_config_bulk_set_ctxs - Provision many VFs with GuC context IDs.
+ * xe_gt_sriov_pf_config_bulk_set_ctxs_locked() - Provision many VFs with GuC context IDs.
* @gt: the &xe_gt
* @vfid: starting VF identifier
* @num_vfs: number of VFs to provision
@@ -1109,30 +1126,48 @@ static int pf_config_bulk_set_u32_done(struct xe_gt *gt, unsigned int first, uns
*
* Return: 0 on success or a negative error code on failure.
*/
-int xe_gt_sriov_pf_config_bulk_set_ctxs(struct xe_gt *gt, unsigned int vfid,
- unsigned int num_vfs, u32 num_ctxs)
+int xe_gt_sriov_pf_config_bulk_set_ctxs_locked(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u32 num_ctxs)
{
unsigned int n;
int err = 0;
xe_gt_assert(gt, vfid);
+ lockdep_assert_held(xe_gt_sriov_pf_master_mutex(gt));
if (!num_vfs)
return 0;
- mutex_lock(xe_gt_sriov_pf_master_mutex(gt));
for (n = vfid; n < vfid + num_vfs; n++) {
err = pf_provision_vf_ctxs(gt, n, num_ctxs);
if (err)
break;
}
- mutex_unlock(xe_gt_sriov_pf_master_mutex(gt));
return pf_config_bulk_set_u32_done(gt, vfid, num_vfs, num_ctxs,
- xe_gt_sriov_pf_config_get_ctxs,
+ pf_get_vf_config_ctxs,
"GuC context IDs", no_unit, n, err);
}
+/**
+ * xe_gt_sriov_pf_config_bulk_set_ctxs() - Provision many VFs with GuC context IDs.
+ * @gt: the &xe_gt
+ * @vfid: starting VF identifier
+ * @num_vfs: number of VFs to provision
+ * @num_ctxs: requested number of GuC contexts IDs (0 to release)
+ *
+ * This function can only be called on PF.
+ *
+ * Return: 0 on success or a negative error code on failure.
+ */
+int xe_gt_sriov_pf_config_bulk_set_ctxs(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u32 num_ctxs)
+{
+ guard(mutex)(xe_gt_sriov_pf_master_mutex(gt));
+
+ return xe_gt_sriov_pf_config_bulk_set_ctxs_locked(gt, vfid, num_vfs, num_ctxs);
+}
+
static u32 pf_profile_fair_ctxs(struct xe_gt *gt, unsigned int num_vfs)
{
bool admin_only_pf = xe_sriov_pf_admin_only(gt_to_xe(gt));
@@ -1181,10 +1216,9 @@ int xe_gt_sriov_pf_config_set_fair_ctxs(struct xe_gt *gt, unsigned int vfid,
xe_gt_assert(gt, vfid);
xe_gt_assert(gt, num_vfs);
- mutex_lock(xe_gt_sriov_pf_master_mutex(gt));
- fair = pf_estimate_fair_ctxs(gt, num_vfs);
- mutex_unlock(xe_gt_sriov_pf_master_mutex(gt));
+ guard(mutex)(xe_gt_sriov_pf_master_mutex(gt));
+ fair = pf_estimate_fair_ctxs(gt, num_vfs);
if (!fair)
return -ENOSPC;
@@ -1193,7 +1227,7 @@ int xe_gt_sriov_pf_config_set_fair_ctxs(struct xe_gt *gt, unsigned int vfid,
xe_gt_sriov_info(gt, "Using non-profile provisioning (%s %u vs %u)\n",
"GuC context IDs", fair, profile);
- return xe_gt_sriov_pf_config_bulk_set_ctxs(gt, vfid, num_vfs, fair);
+ return xe_gt_sriov_pf_config_bulk_set_ctxs_locked(gt, vfid, num_vfs, fair);
}
static u32 pf_get_min_spare_dbs(struct xe_gt *gt)
@@ -1363,7 +1397,7 @@ int xe_gt_sriov_pf_config_set_dbs(struct xe_gt *gt, unsigned int vfid, u32 num_d
}
/**
- * xe_gt_sriov_pf_config_bulk_set_dbs - Provision many VFs with GuC context IDs.
+ * xe_gt_sriov_pf_config_bulk_set_dbs_locked() - Provision many VFs with GuC doorbells.
* @gt: the &xe_gt
* @vfid: starting VF identifier (can't be 0)
* @num_vfs: number of VFs to provision
@@ -1373,30 +1407,48 @@ int xe_gt_sriov_pf_config_set_dbs(struct xe_gt *gt, unsigned int vfid, u32 num_d
*
* Return: 0 on success or a negative error code on failure.
*/
-int xe_gt_sriov_pf_config_bulk_set_dbs(struct xe_gt *gt, unsigned int vfid,
- unsigned int num_vfs, u32 num_dbs)
+int xe_gt_sriov_pf_config_bulk_set_dbs_locked(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u32 num_dbs)
{
unsigned int n;
int err = 0;
xe_gt_assert(gt, vfid);
+ lockdep_assert_held(xe_gt_sriov_pf_master_mutex(gt));
if (!num_vfs)
return 0;
- mutex_lock(xe_gt_sriov_pf_master_mutex(gt));
for (n = vfid; n < vfid + num_vfs; n++) {
err = pf_provision_vf_dbs(gt, n, num_dbs);
if (err)
break;
}
- mutex_unlock(xe_gt_sriov_pf_master_mutex(gt));
return pf_config_bulk_set_u32_done(gt, vfid, num_vfs, num_dbs,
- xe_gt_sriov_pf_config_get_dbs,
+ pf_get_vf_config_dbs,
"GuC doorbell IDs", no_unit, n, err);
}
+/**
+ * xe_gt_sriov_pf_config_bulk_set_dbs() - Provision many VFs with GuC doorbells.
+ * @gt: the &xe_gt
+ * @vfid: starting VF identifier (can't be 0)
+ * @num_vfs: number of VFs to provision
+ * @num_dbs: requested number of GuC doorbell IDs (0 to release)
+ *
+ * This function can only be called on PF.
+ *
+ * Return: 0 on success or a negative error code on failure.
+ */
+int xe_gt_sriov_pf_config_bulk_set_dbs(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u32 num_dbs)
+{
+ guard(mutex)(xe_gt_sriov_pf_master_mutex(gt));
+
+ return xe_gt_sriov_pf_config_bulk_set_dbs_locked(gt, vfid, num_vfs, num_dbs);
+}
+
static u32 pf_profile_fair_dbs(struct xe_gt *gt, unsigned int num_vfs)
{
bool admin_only_pf = xe_sriov_pf_admin_only(gt_to_xe(gt));
@@ -1446,10 +1498,9 @@ int xe_gt_sriov_pf_config_set_fair_dbs(struct xe_gt *gt, unsigned int vfid,
xe_gt_assert(gt, vfid);
xe_gt_assert(gt, num_vfs);
- mutex_lock(xe_gt_sriov_pf_master_mutex(gt));
- fair = pf_estimate_fair_dbs(gt, num_vfs);
- mutex_unlock(xe_gt_sriov_pf_master_mutex(gt));
+ guard(mutex)(xe_gt_sriov_pf_master_mutex(gt));
+ fair = pf_estimate_fair_dbs(gt, num_vfs);
if (!fair)
return -ENOSPC;
@@ -1458,7 +1509,7 @@ int xe_gt_sriov_pf_config_set_fair_dbs(struct xe_gt *gt, unsigned int vfid,
xe_gt_sriov_info(gt, "Using non-profile provisioning (%s %u vs %u)\n",
"GuC doorbell IDs", fair, profile);
- return xe_gt_sriov_pf_config_bulk_set_dbs(gt, vfid, num_vfs, fair);
+ return xe_gt_sriov_pf_config_bulk_set_dbs_locked(gt, vfid, num_vfs, fair);
}
static u64 pf_get_lmem_alignment(struct xe_gt *gt)
diff --git a/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h b/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h
index 2ec62c12ad5c..a56e63f3660a 100644
--- a/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h
+++ b/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h
@@ -18,18 +18,24 @@ int xe_gt_sriov_pf_config_set_fair_ggtt(struct xe_gt *gt,
unsigned int vfid, unsigned int num_vfs);
int xe_gt_sriov_pf_config_bulk_set_ggtt(struct xe_gt *gt,
unsigned int vfid, unsigned int num_vfs, u64 size);
+int xe_gt_sriov_pf_config_bulk_set_ggtt_locked(struct xe_gt *gt,
+ unsigned int vfid, unsigned int num_vfs, u64 size);
u32 xe_gt_sriov_pf_config_get_ctxs(struct xe_gt *gt, unsigned int vfid);
int xe_gt_sriov_pf_config_set_ctxs(struct xe_gt *gt, unsigned int vfid, u32 num_ctxs);
int xe_gt_sriov_pf_config_set_fair_ctxs(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs);
int xe_gt_sriov_pf_config_bulk_set_ctxs(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs,
u32 num_ctxs);
+int xe_gt_sriov_pf_config_bulk_set_ctxs_locked(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u32 num_ctxs);
u32 xe_gt_sriov_pf_config_get_dbs(struct xe_gt *gt, unsigned int vfid);
int xe_gt_sriov_pf_config_set_dbs(struct xe_gt *gt, unsigned int vfid, u32 num_dbs);
int xe_gt_sriov_pf_config_set_fair_dbs(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs);
int xe_gt_sriov_pf_config_bulk_set_dbs(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs,
u32 num_dbs);
+int xe_gt_sriov_pf_config_bulk_set_dbs_locked(struct xe_gt *gt, unsigned int vfid,
+ unsigned int num_vfs, u32 num_dbs);
u64 xe_gt_sriov_pf_config_get_lmem(struct xe_gt *gt, unsigned int vfid);
int xe_gt_sriov_pf_config_set_lmem(struct xe_gt *gt, unsigned int vfid, u64 size);
diff --git a/drivers/gpu/drm/xe/xe_gt_sriov_printk.h b/drivers/gpu/drm/xe/xe_gt_sriov_printk.h
index d3457d608db8..ada0ee9f45e5 100644
--- a/drivers/gpu/drm/xe/xe_gt_sriov_printk.h
+++ b/drivers/gpu/drm/xe/xe_gt_sriov_printk.h
@@ -27,6 +27,9 @@
#define xe_gt_sriov_dbg(_gt, _fmt, ...) \
__xe_gt_sriov_printk(_gt, dbg, _fmt, ##__VA_ARGS__)
+#define xe_gt_sriov_dbg_ratelimited(_gt, _fmt, ...) \
+ __xe_gt_sriov_printk(_gt, dbg_ratelimited, _fmt, ##__VA_ARGS__)
+
/* for low level noisy debug messages */
#ifdef CONFIG_DRM_XE_DEBUG_SRIOV
#define xe_gt_sriov_dbg_verbose(_gt, _fmt, ...) xe_gt_sriov_dbg(_gt, _fmt, ##__VA_ARGS__)
diff --git a/drivers/gpu/drm/xe/xe_gt_stats.c b/drivers/gpu/drm/xe/xe_gt_stats.c
index 789397514f3e..2a40924660ee 100644
--- a/drivers/gpu/drm/xe/xe_gt_stats.c
+++ b/drivers/gpu/drm/xe/xe_gt_stats.c
@@ -95,12 +95,19 @@ void xe_gt_stats_incr(struct xe_gt *gt, const enum xe_gt_stats_id id, int incr)
#define DEF_STAT_STR(ID, name) [XE_GT_STATS_ID_##ID] = name
static const char *const stat_description[__XE_GT_STATS_NUM_IDS] = {
+ DEF_STAT_STR(CHAIN_PAGEFAULT_COUNT, "chain_pagefault_count"),
+ DEF_STAT_STR(CHAIN_IRQ_PAGEFAULT_COUNT, "chain_irq_pagefault_count"),
+ DEF_STAT_STR(CHAIN_DRAIN_IRQ_PAGEFAULT_COUNT, "chain_drain_irq_pagefault_count"),
+ DEF_STAT_STR(CHAIN_MISMATCH_PAGEFAULT_COUNT, "chain_mismatch_pagefault_count"),
+ DEF_STAT_STR(PARALLEL_PAGEFAULT_COUNT, "parallel_pagefault_count"),
+ DEF_STAT_STR(LAST_PAGEFAULT_COUNT, "last_pagefault_count"),
DEF_STAT_STR(SVM_PAGEFAULT_COUNT, "svm_pagefault_count"),
DEF_STAT_STR(TLB_INVAL, "tlb_inval_count"),
DEF_STAT_STR(SVM_TLB_INVAL_COUNT, "svm_tlb_inval_count"),
DEF_STAT_STR(SVM_TLB_INVAL_US, "svm_tlb_inval_us"),
DEF_STAT_STR(VMA_PAGEFAULT_COUNT, "vma_pagefault_count"),
DEF_STAT_STR(VMA_PAGEFAULT_KB, "vma_pagefault_kb"),
+ DEF_STAT_STR(PAGEFAULT_US, "pagefault_us"),
DEF_STAT_STR(INVALID_PREFETCH_PAGEFAULT_COUNT, "invalid_prefetch_pagefault_count"),
DEF_STAT_STR(SVM_4K_PAGEFAULT_COUNT, "svm_4K_pagefault_count"),
DEF_STAT_STR(SVM_64K_PAGEFAULT_COUNT, "svm_64K_pagefault_count"),
diff --git a/drivers/gpu/drm/xe/xe_gt_stats_types.h b/drivers/gpu/drm/xe/xe_gt_stats_types.h
index 425491bed6c4..24abeb9f2137 100644
--- a/drivers/gpu/drm/xe/xe_gt_stats_types.h
+++ b/drivers/gpu/drm/xe/xe_gt_stats_types.h
@@ -10,6 +10,19 @@
/**
* enum xe_gt_stats_id - GT statistics identifiers
+ * @XE_GT_STATS_ID_CHAIN_PAGEFAULT_COUNT: Total page faults chained onto an
+ * in-flight fault instead of being queued separately.
+ * @XE_GT_STATS_ID_CHAIN_IRQ_PAGEFAULT_COUNT: Page faults chained directly from
+ * the IRQ handler.
+ * @XE_GT_STATS_ID_CHAIN_DRAIN_IRQ_PAGEFAULT_COUNT: IRQ-handler chained faults
+ * that also drained the fault queue.
+ * @XE_GT_STATS_ID_CHAIN_MISMATCH_PAGEFAULT_COUNT: Chained faults requeued
+ * because their fault range did not match the fault they were chained onto.
+ * @XE_GT_STATS_ID_PARALLEL_PAGEFAULT_COUNT: Faults dequeued while another page
+ * fault worker was already handling a fault concurrently.
+ * @XE_GT_STATS_ID_LAST_PAGEFAULT_COUNT: Faults whose range matched the last
+ * serviced range, allowing an immediate ack.
+ *
* @XE_GT_STATS_ID_SVM_PAGEFAULT_COUNT: Total SVM page faults handled.
* @XE_GT_STATS_ID_TLB_INVAL: Total GPU Translation Lookaside Buffer (TLB)
* invalidations issued.
@@ -22,6 +35,8 @@
* handled.
* @XE_GT_STATS_ID_VMA_PAGEFAULT_KB: Size (KiB) of VMAs involved in
* buffer-object page fault handling.
+ * @XE_GT_STATS_ID_PAGEFAULT_US: Cumulative time (µs) handling of all page
+ * faults.
* @XE_GT_STATS_ID_INVALID_PREFETCH_PAGEFAULT_COUNT: GPU prefetch faults for
* addresses with no valid backing.
*
@@ -127,12 +142,19 @@
* See Documentation/gpu/xe/xe_gt_stats.rst.
*/
enum xe_gt_stats_id {
+ XE_GT_STATS_ID_CHAIN_PAGEFAULT_COUNT,
+ XE_GT_STATS_ID_CHAIN_IRQ_PAGEFAULT_COUNT,
+ XE_GT_STATS_ID_CHAIN_DRAIN_IRQ_PAGEFAULT_COUNT,
+ XE_GT_STATS_ID_CHAIN_MISMATCH_PAGEFAULT_COUNT,
+ XE_GT_STATS_ID_PARALLEL_PAGEFAULT_COUNT,
+ XE_GT_STATS_ID_LAST_PAGEFAULT_COUNT,
XE_GT_STATS_ID_SVM_PAGEFAULT_COUNT,
XE_GT_STATS_ID_TLB_INVAL,
XE_GT_STATS_ID_SVM_TLB_INVAL_COUNT,
XE_GT_STATS_ID_SVM_TLB_INVAL_US,
XE_GT_STATS_ID_VMA_PAGEFAULT_COUNT,
XE_GT_STATS_ID_VMA_PAGEFAULT_KB,
+ XE_GT_STATS_ID_PAGEFAULT_US,
XE_GT_STATS_ID_INVALID_PREFETCH_PAGEFAULT_COUNT,
XE_GT_STATS_ID_SVM_4K_PAGEFAULT_COUNT,
XE_GT_STATS_ID_SVM_64K_PAGEFAULT_COUNT,
diff --git a/drivers/gpu/drm/xe/xe_guc.c b/drivers/gpu/drm/xe/xe_guc.c
index 4286bd05c686..c7f8bbd4cb92 100644
--- a/drivers/gpu/drm/xe/xe_guc.c
+++ b/drivers/gpu/drm/xe/xe_guc.c
@@ -39,6 +39,7 @@
#include "xe_guc_rc.h"
#include "xe_guc_relay.h"
#include "xe_guc_submit.h"
+#include "xe_log.h"
#include "xe_memirq.h"
#include "xe_mmio.h"
#include "xe_platform_types.h"
@@ -1542,8 +1543,9 @@ retry:
/* scratch registers might be cleared during FLR, try once more */
if (!header) {
if (++lost > MAX_RETRIES_ON_FLR) {
- xe_gt_err(gt, "GuC mmio request %#x: lost, too many retries %u\n",
- request[0], lost);
+ xe_log_err(gt, GUC, -ENOLINK,
+ "MMIO request %#x: lost, too many retries %u\n",
+ request[0], lost);
return -ENOLINK;
}
xe_gt_dbg(gt, "GuC mmio request %#x: lost, trying again\n", request[0]);
@@ -1551,8 +1553,8 @@ retry:
goto retry;
}
timeout:
- xe_gt_err(gt, "GuC mmio request %#x: no reply %#x\n",
- request[0], header);
+ xe_log_err(gt, GUC, ret, "MMIO request %#x: no reply %#x\n",
+ request[0], header);
return ret;
}
@@ -1607,16 +1609,16 @@ timeout:
return -EREMCHG;
}
- xe_gt_err(gt, "GuC mmio request %#x: failure %#x hint %#x\n",
- request[0], error, hint);
+ xe_log_err(gt, GUC, -ENXIO, "MMIO request %#x: failure %#x hint %#x\n",
+ request[0], error, hint);
return -ENXIO;
}
if (FIELD_GET(GUC_HXG_MSG_0_TYPE, header) !=
GUC_HXG_TYPE_RESPONSE_SUCCESS) {
proto:
- xe_gt_err(gt, "GuC mmio request %#x: unexpected reply %#x\n",
- request[0], header);
+ xe_log_err(gt, GUC, -EPROTO, "MMIO request %#x: unexpected reply %#x\n",
+ request[0], header);
return -EPROTO;
}
diff --git a/drivers/gpu/drm/xe/xe_guc_ct.c b/drivers/gpu/drm/xe/xe_guc_ct.c
index fe70c0fd85c5..5c4733da385c 100644
--- a/drivers/gpu/drm/xe/xe_guc_ct.c
+++ b/drivers/gpu/drm/xe/xe_guc_ct.c
@@ -265,7 +265,7 @@ static bool g2h_fence_needs_alloc(struct g2h_fence *g2h_fence)
#define CTB_DESC_SIZE ALIGN(sizeof(struct guc_ct_buffer_desc), SZ_2K)
#define CTB_H2G_BUFFER_OFFSET (CTB_DESC_SIZE * 2)
#define CTB_G2H_BUFFER_OFFSET (CTB_DESC_SIZE * 2)
-#define CTB_H2G_BUFFER_SIZE (SZ_4K)
+#define CTB_H2G_BUFFER_SIZE (SZ_16K)
#define CTB_H2G_BUFFER_DWORDS (CTB_H2G_BUFFER_SIZE / sizeof(u32))
#define CTB_G2H_BUFFER_SIZE (SZ_128K)
#define CTB_G2H_BUFFER_DWORDS (CTB_G2H_BUFFER_SIZE / sizeof(u32))
@@ -939,7 +939,7 @@ static bool vf_action_can_safely_fail(struct xe_device *xe, u32 action)
#define H2G_CT_HEADERS (GUC_CTB_HDR_LEN + 1) /* one DW CTB header and one DW HxG header */
static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
- u32 ct_fence_value, bool want_response)
+ u32 ct_fence_value, bool want_response, bool defer_flush)
{
struct xe_device *xe = ct_to_xe(ct);
struct xe_gt *gt = ct_to_gt(ct);
@@ -956,7 +956,6 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
xe_gt_assert(gt, full_len <= GUC_CTB_MSG_MAX_LEN);
if (IS_ENABLED(CONFIG_DRM_XE_DEBUG)) {
- u32 desc_tail = desc_read(xe, h2g, tail);
u32 desc_head = desc_read(xe, h2g, head);
u32 desc_status;
@@ -966,12 +965,6 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
goto corrupted;
}
- if (tail != desc_tail) {
- desc_write(xe, h2g, status, desc_status | GUC_CTB_STATUS_MISMATCH);
- xe_gt_err(gt, "CT write: tail was modified %u != %u\n", desc_tail, tail);
- goto corrupted;
- }
-
if (tail > h2g->info.size) {
desc_write(xe, h2g, status, desc_status | GUC_CTB_STATUS_OVERFLOW);
xe_gt_err(gt, "CT write: tail out of range: %u vs %u\n",
@@ -993,7 +986,10 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
(h2g->info.size - tail) * sizeof(u32));
h2g_reserve_space(ct, (h2g->info.size - tail));
h2g->info.tail = 0;
- desc_write(xe, h2g, tail, h2g->info.tail);
+ if (!defer_flush) {
+ xe_device_wmb(xe);
+ desc_write(xe, h2g, tail, h2g->info.tail);
+ }
return -EAGAIN;
}
@@ -1024,14 +1020,15 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
/* Write H2G ensuring visible before descriptor update */
xe_map_memcpy_to(xe, &map, 0, cmd, H2G_CT_HEADERS * sizeof(u32));
xe_map_memcpy_to(xe, &map, H2G_CT_HEADERS * sizeof(u32), action, len * sizeof(u32));
- xe_device_wmb(xe);
-
/* Update local copies */
h2g->info.tail = (tail + full_len) % h2g->info.size;
h2g_reserve_space(ct, full_len);
/* Update descriptor */
- desc_write(xe, h2g, tail, h2g->info.tail);
+ if (!defer_flush) {
+ xe_device_wmb(xe);
+ desc_write(xe, h2g, tail, h2g->info.tail);
+ }
/*
* desc_read() performs an VRAM read which serializes the CPU and drains
@@ -1052,7 +1049,7 @@ corrupted:
static int __guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action,
u32 len, u32 g2h_len, u32 num_g2h,
- struct g2h_fence *g2h_fence)
+ struct g2h_fence *g2h_fence, bool defer_flush)
{
struct xe_gt *gt = ct_to_gt(ct);
u16 seqno;
@@ -1112,7 +1109,7 @@ retry:
if (unlikely(ret))
goto out_unlock;
- ret = h2g_write(ct, action, len, seqno, !!g2h_fence);
+ ret = h2g_write(ct, action, len, seqno, !!g2h_fence, defer_flush);
if (unlikely(ret)) {
if (ret == -EAGAIN)
goto retry;
@@ -1120,7 +1117,8 @@ retry:
}
__g2h_reserve_space(ct, g2h_len, num_g2h);
- xe_guc_notify(ct_to_guc(ct));
+ if (!defer_flush)
+ xe_guc_notify(ct_to_guc(ct));
out_unlock:
if (g2h_len)
spin_unlock_irq(&ct->fast_lock);
@@ -1196,7 +1194,7 @@ static bool guc_ct_send_wait_for_retry(struct xe_guc_ct *ct, u32 len,
static int guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len,
u32 g2h_len, u32 num_g2h,
- struct g2h_fence *g2h_fence)
+ struct g2h_fence *g2h_fence, bool defer_flush)
{
struct xe_gt *gt = ct_to_gt(ct);
unsigned int sleep_period_ms = 1;
@@ -1209,9 +1207,10 @@ static int guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len,
try_again:
ret = __guc_ct_send_locked(ct, action, len, g2h_len, num_g2h,
- g2h_fence);
+ g2h_fence, defer_flush);
if (unlikely(ret == -EBUSY)) {
+ xe_guc_ct_send_flush(ct);
if (!guc_ct_send_wait_for_retry(ct, len, g2h_len, g2h_fence,
&sleep_period_ms, &sleep_total_ms))
goto broken;
@@ -1235,7 +1234,8 @@ static int guc_ct_send(struct xe_guc_ct *ct, const u32 *action, u32 len,
xe_gt_assert(ct_to_gt(ct), !g2h_len || !g2h_fence);
mutex_lock(&ct->lock);
- ret = guc_ct_send_locked(ct, action, len, g2h_len, num_g2h, g2h_fence);
+ ret = guc_ct_send_locked(ct, action, len, g2h_len, num_g2h, g2h_fence,
+ false);
mutex_unlock(&ct->lock);
return ret;
@@ -1283,25 +1283,76 @@ int xe_guc_ct_send(struct xe_guc_ct *ct, const u32 *action, u32 len,
return ret;
}
+/**
+ * xe_guc_ct_send_locked() - submit a GuC CT H2G message with CT lock held
+ * @ct: GuC CT object
+ * @action: payload dwords (HxG header dword is expected at @action[-1])
+ * @len: number of payload dwords in @action
+ * @defer_flush: defer publishing/doorbell for batching
+ *
+ * Sends a single H2G message to the GuC CT buffer while the caller already
+ * holds @ct->lock.
+ *
+ * If @defer_flush is false, the function completes the submission immediately:
+ * it makes the payload visible to the device, updates the H2G descriptor and
+ * rings the GuC doorbell.
+ *
+ * If @defer_flush is true, the message payload is copied into the H2G ring and
+ * the software tail is advanced, but the descriptor update and doorbell are
+ * deferred so multiple messages can be batched. In this mode, the caller must
+ * eventually call xe_guc_ct_send_flush() (still holding @ct->lock) to publish
+ * the descriptor and notify the GuC. On internal retry paths (-EBUSY), the
+ * implementation may force a flush to ensure forward progress.
+ *
+ * Return: 0 on success, negative errno on failure.
+ *
+ * Locking:
+ * Must be called with @ct->lock held.
+ */
int xe_guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len,
- u32 g2h_len, u32 num_g2h)
+ bool defer_flush)
{
int ret;
- ret = guc_ct_send_locked(ct, action, len, g2h_len, num_g2h, NULL);
+ ret = guc_ct_send_locked(ct, action, len, 0, 0, NULL, defer_flush);
if (ret == -EDEADLK)
kick_reset(ct);
return ret;
}
+/**
+ * xe_guc_ct_send_flush() - flush pending GuC CT H2G writes
+ * @ct: GuC CT instance
+ *
+ * Some callers batch multiple H2G writes using xe_guc_ct_send_locked() in
+ * "write-only" mode (i.e., queue the message payloads but defer ringing the
+ * doorbell / updating the CT descriptor). This helper completes the submission
+ * by ensuring the payload writes are visible to the device, updating the H2G
+ * descriptor, and ringing the GuC CT doorbell.
+ *
+ * Locking:
+ * Must be called with @ct->lock held.
+ */
+void xe_guc_ct_send_flush(struct xe_guc_ct *ct)
+{
+ struct xe_device *xe = ct_to_xe(ct);
+ struct guc_ctb *h2g = &ct->ctbs.h2g;
+
+ lockdep_assert_held(&ct->lock);
+
+ xe_device_wmb(xe);
+ desc_write(xe, h2g, tail, h2g->info.tail);
+ xe_guc_notify(ct_to_guc(ct));
+}
+
int xe_guc_ct_send_g2h_handler(struct xe_guc_ct *ct, const u32 *action, u32 len)
{
int ret;
lockdep_assert_held(&ct->lock);
- ret = guc_ct_send_locked(ct, action, len, 0, 0, NULL);
+ ret = guc_ct_send_locked(ct, action, len, 0, 0, NULL, false);
if (ret == -EDEADLK)
kick_reset(ct);
diff --git a/drivers/gpu/drm/xe/xe_guc_ct.h b/drivers/gpu/drm/xe/xe_guc_ct.h
index 767365a33dee..3ddc665ab84a 100644
--- a/drivers/gpu/drm/xe/xe_guc_ct.h
+++ b/drivers/gpu/drm/xe/xe_guc_ct.h
@@ -6,6 +6,8 @@
#ifndef _XE_GUC_CT_H_
#define _XE_GUC_CT_H_
+#include <linux/mutex.h>
+
#include "xe_guc_ct_types.h"
struct drm_printer;
@@ -54,7 +56,7 @@ static inline void xe_guc_ct_irq_handler(struct xe_guc_ct *ct)
int xe_guc_ct_send(struct xe_guc_ct *ct, const u32 *action, u32 len,
u32 g2h_len, u32 num_g2h);
int xe_guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len,
- u32 g2h_len, u32 num_g2h);
+ bool defer_flush);
int xe_guc_ct_send_recv(struct xe_guc_ct *ct, const u32 *action, u32 len,
u32 *response_buffer);
static inline int
@@ -63,6 +65,8 @@ xe_guc_ct_send_block(struct xe_guc_ct *ct, const u32 *action, u32 len)
return xe_guc_ct_send_recv(ct, action, len, NULL);
}
+void xe_guc_ct_send_flush(struct xe_guc_ct *ct);
+
/* This is only version of the send CT you can call from a G2H handler */
int xe_guc_ct_send_g2h_handler(struct xe_guc_ct *ct, const u32 *action,
u32 len);
@@ -87,4 +91,36 @@ static inline void xe_guc_ct_wake_waiters(struct xe_guc_ct *ct)
wake_up_all(&ct->wq);
}
+/**
+ * xe_guc_ct_lock() - take the GuC CT mutex
+ * @ct: GuC CT object
+ *
+ * Wrapper around mutex_lock(&ct->lock) for cases where CT operations need to be
+ * performed from contexts that want an explicit "CT locked" pair without
+ * exporting the lock itself.
+ *
+ * Return/Locking:
+ * Acquires @ct->lock.
+ */
+static inline void xe_guc_ct_lock(struct xe_guc_ct *ct)
+__acquires(&ct->lock)
+{
+ mutex_lock(&ct->lock);
+}
+
+/**
+ * xe_guc_ct_unlock() - release the GuC CT mutex
+ * @ct: GuC CT object
+ *
+ * Counterpart to xe_guc_ct_lock().
+ *
+ * Locking:
+ * Releases @ct->lock.
+ */
+static inline void xe_guc_ct_unlock(struct xe_guc_ct *ct)
+__releases(&ct->lock)
+{
+ mutex_unlock(&ct->lock);
+}
+
#endif
diff --git a/drivers/gpu/drm/xe/xe_guc_exec_queue_types.h b/drivers/gpu/drm/xe/xe_guc_exec_queue_types.h
index acdc24d1a6bd..d27826b36649 100644
--- a/drivers/gpu/drm/xe/xe_guc_exec_queue_types.h
+++ b/drivers/gpu/drm/xe/xe_guc_exec_queue_types.h
@@ -36,7 +36,7 @@ struct xe_guc_exec_queue {
* a message needs to sent through the GPU scheduler but memory
* allocations are not allowed.
*/
-#define MAX_STATIC_MSG_TYPE 3
+#define MAX_STATIC_MSG_TYPE 4
struct xe_sched_msg static_msgs[MAX_STATIC_MSG_TYPE];
/** @destroy_async: do final destroy async from this worker */
struct work_struct destroy_async;
@@ -76,6 +76,36 @@ struct xe_guc_exec_queue {
* recovery.
*/
bool needs_resume;
+ /** @multi_queue: multi-queue group CGP state for VF post migration recovery */
+ struct {
+ /**
+ * @multi_queue.needs_cgp_sync: Needs a CGP_SYNC (dynamic CGP
+ * update) message replayed during recovery.
+ */
+ u8 needs_cgp_sync:1;
+ /**
+ * @multi_queue.re_register: A registration-time CGP update was
+ * interrupted by recovery; the queue must be re-registered.
+ */
+ u8 re_register:1;
+ /**
+ * @multi_queue.re_update: A dynamic CGP update was interrupted
+ * by recovery; the CGP update must be replayed.
+ */
+ u8 re_update:1;
+ /**
+ * @multi_queue.registering_cgp: This queue's currently
+ * outstanding CGP_SYNC is a registration (matched against
+ * group->cgp_update_q in revert).
+ */
+ u8 registering_cgp:1;
+ /**
+ * @multi_queue.updating_cgp: This queue's currently outstanding
+ * CGP_SYNC is a dynamic update (matched against
+ * group->cgp_update_q in revert).
+ */
+ u8 updating_cgp:1;
+ } multi_queue;
};
#endif
diff --git a/drivers/gpu/drm/xe/xe_guc_pagefault.c b/drivers/gpu/drm/xe/xe_guc_pagefault.c
index 607e32392f46..8f8210a732e9 100644
--- a/drivers/gpu/drm/xe/xe_guc_pagefault.c
+++ b/drivers/gpu/drm/xe/xe_guc_pagefault.c
@@ -10,6 +10,22 @@
#include "xe_pagefault.h"
#include "xe_pagefault_types.h"
+#define XE_GUC_PAGEFAULT_FLUSH_PERIOD BIT(4) /* Sixteen */
+
+static void guc_ack_fault_begin(void *private)
+{
+ struct xe_guc *guc = private;
+
+ xe_guc_ct_lock(&guc->ct);
+
+ BUILD_BUG_ON(((XE_GUC_PAGEFAULT_FLUSH_PERIOD - 1) &
+ XE_GUC_PAGEFAULT_FLUSH_PERIOD) != 0);
+
+ /* Ack the 2nd, then 18th, etc... */
+ guc->pagefault_ack_counter =
+ XE_GUC_PAGEFAULT_FLUSH_PERIOD - 1;
+}
+
static void guc_ack_fault(struct xe_pagefault *pf, int err)
{
u32 vfid = FIELD_GET(PFD_VFID, pf->producer.msg[2]);
@@ -36,12 +52,26 @@ static void guc_ack_fault(struct xe_pagefault *pf, int err)
FIELD_PREP(PFR_PDATA, pdata),
};
struct xe_guc *guc = pf->producer.private;
+ bool write_only = guc->pagefault_ack_counter++ &
+ (XE_GUC_PAGEFAULT_FLUSH_PERIOD - 1);
+
+ xe_guc_ct_send_locked(&guc->ct, action, ARRAY_SIZE(action),
+ write_only);
+}
+
+static void guc_ack_fault_end(void *private)
+{
+ struct xe_guc *guc = private;
- xe_guc_ct_send(&guc->ct, action, ARRAY_SIZE(action), 0, 0);
+ if ((guc->pagefault_ack_counter & (XE_GUC_PAGEFAULT_FLUSH_PERIOD - 1)) != 1)
+ xe_guc_ct_send_flush(&guc->ct);
+ xe_guc_ct_unlock(&guc->ct);
}
static const struct xe_pagefault_ops guc_pagefault_ops = {
+ .ack_fault_begin = guc_ack_fault_begin,
.ack_fault = guc_ack_fault,
+ .ack_fault_end = guc_ack_fault_end,
};
/**
@@ -89,8 +119,11 @@ int xe_guc_pagefault_handler(struct xe_guc *guc, u32 *msg, u32 len)
FIELD_GET(PFD_FAULT_LEVEL, msg[0])) |
FIELD_PREP(XE_PAGEFAULT_TYPE_MASK,
FIELD_GET(PFD_FAULT_TYPE, msg[2]));
- pf.consumer.engine_class = FIELD_GET(PFD_ENG_CLASS, msg[0]);
- pf.consumer.engine_instance = FIELD_GET(PFD_ENG_INSTANCE, msg[0]);
+ pf.consumer.engine_class_instance =
+ FIELD_PREP(XE_PAGEFAULT_ENGINE_CLASS_MASK,
+ FIELD_GET(PFD_ENG_CLASS, msg[0])) |
+ FIELD_PREP(XE_PAGEFAULT_ENGINE_INSTANCE_MASK,
+ FIELD_GET(PFD_ENG_INSTANCE, msg[0]));
pf.producer.private = guc;
pf.producer.ops = &guc_pagefault_ops;
diff --git a/drivers/gpu/drm/xe/xe_guc_pc.c b/drivers/gpu/drm/xe/xe_guc_pc.c
index 7cf8f4858598..097b075bd89a 100644
--- a/drivers/gpu/drm/xe/xe_guc_pc.c
+++ b/drivers/gpu/drm/xe/xe_guc_pc.c
@@ -1241,8 +1241,21 @@ static int pc_modify_defaults(struct xe_guc_pc *pc)
if (xe->info.platform == XE_PANTHERLAKE) {
ret = pc_action_set_dcc(pc, false);
- if (unlikely(ret))
+ if (unlikely(ret)) {
xe_gt_err(gt, "Failed to modify DCC default: %pe\n", ERR_PTR(ret));
+ return ret;
+ }
+
+ if (xe_gt_is_main_type(gt) &&
+ GUC_FIRMWARE_VER_AT_LEAST(&gt->uc.guc, 70, 48, 0)) {
+ ret = pc_action_set_param(pc,
+ SLPC_PARAM_SET_IBC_VERSION,
+ 3);
+ if (unlikely(ret)) {
+ xe_gt_err(gt, "Failed to modify IBC version: %pe\n", ERR_PTR(ret));
+ return ret;
+ }
+ }
}
return ret;
diff --git a/drivers/gpu/drm/xe/xe_guc_submit.c b/drivers/gpu/drm/xe/xe_guc_submit.c
index 8aaed4fd13ea..99d8c807ff05 100644
--- a/drivers/gpu/drm/xe/xe_guc_submit.c
+++ b/drivers/gpu/drm/xe/xe_guc_submit.c
@@ -800,9 +800,12 @@ static void xe_guc_exec_queue_group_cgp_update(struct xe_device *xe,
}
}
+#define CGP_SYNC_REGISTRATION BIT(0)
+
static void xe_guc_exec_queue_group_cgp_sync(struct xe_guc *guc,
struct xe_exec_queue *q,
- const u32 *action, u32 len)
+ const u32 *action, u32 len,
+ unsigned int flags)
{
struct xe_exec_queue_group *group = q->multi_queue.group;
struct xe_device *xe = guc_to_xe(guc);
@@ -815,13 +818,12 @@ static void xe_guc_exec_queue_group_cgp_sync(struct xe_guc *guc,
* Hence, no locking is required here.
* Wait for any pending CGP_SYNC_DONE response before updating the
* CGP page and sending CGP_SYNC message.
- *
- * FIXME: Support VF migration
*/
ret = wait_event_timeout(guc->ct.wq,
!READ_ONCE(group->sync_pending) ||
- xe_guc_read_stopped(guc), HZ);
- if (!ret || xe_guc_read_stopped(guc)) {
+ xe_guc_read_stopped(guc) || vf_recovery(guc),
+ HZ);
+ if ((!ret && !vf_recovery(guc)) || xe_guc_read_stopped(guc)) {
/* CGP_SYNC failed. Reset gt, cleanup the group */
xe_gt_warn(guc_to_gt(guc), "Wait for CGP_SYNC_DONE response failed!\n");
set_exec_queue_group_banned(q);
@@ -830,17 +832,45 @@ static void xe_guc_exec_queue_group_cgp_sync(struct xe_guc *guc,
return;
}
+ /*
+ * If woken by VF migration recovery, do not touch the CGP or send: the
+ * message would be lost and, for a registration, GuC must (re-)register
+ * the context before its CGP entry may be read. Flag the queue so revert
+ * replays it - a registration by re-registration, a dynamic update by a
+ * replayed CGP_SYNC - and bail.
+ */
+ if (vf_recovery(guc)) {
+ if (flags & CGP_SYNC_REGISTRATION)
+ q->guc->multi_queue.re_register = true;
+ else
+ q->guc->multi_queue.re_update = true;
+ return;
+ }
+
scoped_guard(spinlock, &q->multi_queue.lock)
priority = q->multi_queue.priority;
xe_lrc_set_multi_queue_priority(q->lrc[0], priority);
xe_guc_exec_queue_group_cgp_update(xe, q);
+ /*
+ * Record the nature of this outstanding sync so revert can replay it if
+ * its CGP_SYNC_DONE is lost across a migration: a registration is
+ * recovered by re-registration, a dynamic update by a replayed CGP_SYNC.
+ */
+ if (flags & CGP_SYNC_REGISTRATION) {
+ q->guc->multi_queue.registering_cgp = true;
+ q->guc->multi_queue.updating_cgp = false;
+ } else {
+ q->guc->multi_queue.updating_cgp = true;
+ q->guc->multi_queue.registering_cgp = false;
+ }
+ WRITE_ONCE(group->cgp_update_q, q);
WRITE_ONCE(group->sync_pending, true);
xe_guc_ct_send(&guc->ct, action, len, G2H_LEN_DW_MULTI_QUEUE_CONTEXT, 1);
}
-static void guc_exec_queue_send_cgp_sync(struct xe_exec_queue *q)
+static void guc_exec_queue_send_cgp_sync(struct xe_exec_queue *q, unsigned int flags)
{
#define MAX_MULTI_QUEUE_CGP_SYNC_SIZE (2)
struct xe_guc *guc = exec_queue_to_guc(q);
@@ -854,7 +884,7 @@ static void guc_exec_queue_send_cgp_sync(struct xe_exec_queue *q)
xe_gt_assert(guc_to_gt(guc), len <= MAX_MULTI_QUEUE_CGP_SYNC_SIZE);
#undef MAX_MULTI_QUEUE_CGP_SYNC_SIZE
- xe_guc_exec_queue_group_cgp_sync(guc, q, action, len);
+ xe_guc_exec_queue_group_cgp_sync(guc, q, action, len, flags);
}
static void __register_exec_queue_group(struct xe_exec_queue *q,
@@ -882,7 +912,8 @@ static void __register_exec_queue_group(struct xe_exec_queue *q,
* XE_GUC_ACTION_NOTIFY_MULTI_QUEUE_CONTEXT_CGP_SYNC_DONE response
* from guc.
*/
- xe_guc_exec_queue_group_cgp_sync(guc, q, action, len);
+ xe_guc_exec_queue_group_cgp_sync(guc, q, action, len,
+ CGP_SYNC_REGISTRATION);
}
static void __register_mlrc_exec_queue(struct xe_guc *guc,
@@ -1042,7 +1073,7 @@ static void register_exec_queue(struct xe_exec_queue *q, int ctx_type)
init_policies(guc, q);
if (xe_exec_queue_is_multi_queue_secondary(q))
- guc_exec_queue_send_cgp_sync(q);
+ guc_exec_queue_send_cgp_sync(q, CGP_SYNC_REGISTRATION);
}
static u32 wq_space_until_wrap(struct xe_exec_queue *q)
@@ -1559,8 +1590,14 @@ guc_exec_queue_timedout_job(struct drm_sched_job *drm_job)
if (!skip_timeout_check && !check_timeout(q, job))
goto rearm;
+ /*
+ * Killed queues must not newly wedge the device, but preserve an
+ * already-wedged state to avoid warning on teardown timeouts.
+ */
if (!exec_queue_killed(q))
wedged = guc_submit_hint_wedged(exec_queue_to_guc(q));
+ else
+ wedged = xe_device_wedged(xe);
set_exec_queue_banned(q);
@@ -1781,32 +1818,15 @@ static void __guc_exec_queue_destroy_async(struct work_struct *w)
static void guc_exec_queue_destroy_async(struct xe_exec_queue *q)
{
INIT_WORK(&q->guc->destroy_async, __guc_exec_queue_destroy_async);
-
- /* We must block on kernel engines so slabs are empty on driver unload */
- if (q->flags & EXEC_QUEUE_FLAG_PERMANENT || exec_queue_wedged(q))
- guc_exec_queue_do_destroy(q);
- else
- xe_destroy_wq_queue(&q->guc->destroy_async);
+ xe_destroy_wq_queue(&q->guc->destroy_async);
}
-static void __guc_exec_queue_destroy(struct xe_guc *guc, struct xe_exec_queue *q)
-{
- /*
- * Might be done from within the GPU scheduler, need to do async as we
- * fini the scheduler when the engine is fini'd, the scheduler can't
- * complete fini within itself (circular dependency). Async resolves
- * this we and don't really care when everything is fini'd, just that it
- * is.
- */
- guc_exec_queue_destroy_async(q);
-}
-
-static void __guc_exec_queue_process_msg_cleanup(struct xe_sched_msg *msg)
+static void __guc_exec_queue_process_msg_cleanup(struct xe_sched_msg *msg,
+ bool bound)
{
struct xe_exec_queue *q = msg->private_data;
struct xe_guc *guc = exec_queue_to_guc(q);
- xe_gt_assert(guc_to_gt(guc), !(q->flags & EXEC_QUEUE_FLAG_PERMANENT));
trace_xe_exec_queue_cleanup_entity(q);
/*
@@ -1819,10 +1839,12 @@ static void __guc_exec_queue_process_msg_cleanup(struct xe_sched_msg *msg)
* it is safe to directly destroy the exec queue on driver side, as the GuC
* will not process further requests and all resources must be cleaned up locally.
*/
- if (exec_queue_registered(q) && xe_uc_fw_is_running(&guc->fw))
+ /* A wedged GuC won't answer the H2G, so tear down on the driver side. */
+ if (bound && !exec_queue_wedged(q) && exec_queue_registered(q) &&
+ xe_uc_fw_is_running(&guc->fw))
disable_scheduling_deregister(guc, q);
else
- __guc_exec_queue_destroy(guc, q);
+ guc_exec_queue_destroy_async(q);
}
static bool guc_exec_queue_allowed_to_change_state(struct xe_exec_queue *q)
@@ -1830,12 +1852,13 @@ static bool guc_exec_queue_allowed_to_change_state(struct xe_exec_queue *q)
return !exec_queue_killed_or_banned_or_wedged(q) && exec_queue_registered(q);
}
-static void __guc_exec_queue_process_msg_set_sched_props(struct xe_sched_msg *msg)
+static void __guc_exec_queue_process_msg_set_sched_props(struct xe_sched_msg *msg,
+ bool bound)
{
struct xe_exec_queue *q = msg->private_data;
struct xe_guc *guc = exec_queue_to_guc(q);
- if (guc_exec_queue_allowed_to_change_state(q))
+ if (guc_exec_queue_allowed_to_change_state(q) && bound)
init_policies(guc, q);
kfree(msg);
}
@@ -1873,13 +1896,14 @@ static void suspend_fence_signal(struct xe_exec_queue *q)
__suspend_fence_signal(q);
}
-static void __guc_exec_queue_process_msg_suspend(struct xe_sched_msg *msg)
+static void __guc_exec_queue_process_msg_suspend(struct xe_sched_msg *msg,
+ bool bound)
{
struct xe_exec_queue *q = msg->private_data;
struct xe_guc *guc = exec_queue_to_guc(q);
if (guc_exec_queue_allowed_to_change_state(q) && !exec_queue_suspended(q) &&
- exec_queue_enabled(q)) {
+ exec_queue_enabled(q) && bound) {
wait_event(guc->ct.wq, vf_recovery(guc) ||
((q->guc->resume_time != RESUME_PENDING ||
xe_guc_read_stopped(guc)) && !exec_queue_pending_disable(q)));
@@ -1903,11 +1927,12 @@ static void __guc_exec_queue_process_msg_suspend(struct xe_sched_msg *msg)
}
}
-static void __guc_exec_queue_process_msg_resume(struct xe_sched_msg *msg)
+static void __guc_exec_queue_process_msg_resume(struct xe_sched_msg *msg,
+ bool bound)
{
struct xe_exec_queue *q = msg->private_data;
- if (guc_exec_queue_allowed_to_change_state(q)) {
+ if (guc_exec_queue_allowed_to_change_state(q) && bound) {
clear_exec_queue_suspended(q);
if (!exec_queue_enabled(q)) {
q->guc->resume_time = RESUME_PENDING;
@@ -1919,52 +1944,79 @@ static void __guc_exec_queue_process_msg_resume(struct xe_sched_msg *msg)
}
}
-static void __guc_exec_queue_process_msg_set_multi_queue_priority(struct xe_sched_msg *msg)
+static void __guc_exec_queue_process_msg_set_multi_queue_priority(struct xe_sched_msg *msg,
+ bool bound)
{
struct xe_exec_queue *q = msg->private_data;
- if (guc_exec_queue_allowed_to_change_state(q))
- guc_exec_queue_send_cgp_sync(q);
+ if (guc_exec_queue_allowed_to_change_state(q) && bound)
+ guc_exec_queue_send_cgp_sync(q, 0);
kfree(msg);
}
+static void __guc_exec_queue_process_msg_cgp_sync(struct xe_sched_msg *msg,
+ bool bound)
+{
+ struct xe_exec_queue *q = msg->private_data;
+
+ /*
+ * Replay a dynamic CGP update lost across VF migration by re-issuing the
+ * CGP update + CGP_SYNC (re-applies the current priority from
+ * q->multi_queue.priority).
+ */
+ if (guc_exec_queue_allowed_to_change_state(q) && bound)
+ guc_exec_queue_send_cgp_sync(q, 0);
+}
+
#define CLEANUP 1 /* Non-zero values to catch uninitialized msg */
#define SET_SCHED_PROPS 2
#define SUSPEND 3
#define RESUME 4
#define SET_MULTI_QUEUE_PRIORITY 5
+#define CGP_SYNC_MSG 6
#define OPCODE_MASK 0xf
#define MSG_LOCKED BIT(8)
#define MSG_HEAD BIT(9)
+#define MSG_PM_REF BIT(10)
static void guc_exec_queue_process_msg(struct xe_sched_msg *msg)
{
struct xe_device *xe = guc_to_xe(exec_queue_to_guc(msg->private_data));
+ int idx;
+ bool pm_ref = !!(msg->opcode & MSG_PM_REF);
+ bool bound = drm_dev_enter(&xe->drm, &idx);
trace_xe_sched_msg_recv(msg);
- switch (msg->opcode) {
+ switch (msg->opcode & OPCODE_MASK) {
case CLEANUP:
- __guc_exec_queue_process_msg_cleanup(msg);
+ __guc_exec_queue_process_msg_cleanup(msg, bound);
break;
case SET_SCHED_PROPS:
- __guc_exec_queue_process_msg_set_sched_props(msg);
+ __guc_exec_queue_process_msg_set_sched_props(msg, bound);
break;
case SUSPEND:
- __guc_exec_queue_process_msg_suspend(msg);
+ __guc_exec_queue_process_msg_suspend(msg, bound);
break;
case RESUME:
- __guc_exec_queue_process_msg_resume(msg);
+ __guc_exec_queue_process_msg_resume(msg, bound);
break;
case SET_MULTI_QUEUE_PRIORITY:
- __guc_exec_queue_process_msg_set_multi_queue_priority(msg);
+ __guc_exec_queue_process_msg_set_multi_queue_priority(msg, bound);
+ break;
+ case CGP_SYNC_MSG:
+ __guc_exec_queue_process_msg_cgp_sync(msg, bound);
break;
default:
XE_WARN_ON("Unknown message type");
}
- xe_pm_runtime_put(xe);
+ if (pm_ref)
+ xe_pm_runtime_put(xe);
+
+ if (bound)
+ drm_dev_exit(idx);
}
static const struct drm_sched_backend_ops drm_sched_ops = {
@@ -2089,10 +2141,16 @@ static void guc_exec_queue_kill(struct xe_exec_queue *q)
static void guc_exec_queue_add_msg(struct xe_exec_queue *q, struct xe_sched_msg *msg,
u32 opcode)
{
- xe_pm_runtime_get_noresume(guc_to_xe(exec_queue_to_guc(q)));
+ struct xe_device *xe = guc_to_xe(exec_queue_to_guc(q));
+ int idx;
+ bool bound = drm_dev_enter(&xe->drm, &idx);
INIT_LIST_HEAD(&msg->link);
msg->opcode = opcode & OPCODE_MASK;
+ if (bound) {
+ xe_pm_runtime_get_noresume(xe);
+ msg->opcode |= MSG_PM_REF;
+ }
msg->private_data = q;
trace_xe_sched_msg_add(msg);
@@ -2102,6 +2160,9 @@ static void guc_exec_queue_add_msg(struct xe_exec_queue *q, struct xe_sched_msg
xe_sched_add_msg_locked(&q->guc->sched, msg);
else
xe_sched_add_msg(&q->guc->sched, msg);
+
+ if (bound)
+ drm_dev_exit(idx);
}
static void guc_exec_queue_try_add_msg_head(struct xe_exec_queue *q,
@@ -2129,14 +2190,12 @@ static bool guc_exec_queue_try_add_msg(struct xe_exec_queue *q,
#define STATIC_MSG_CLEANUP 0
#define STATIC_MSG_SUSPEND 1
#define STATIC_MSG_RESUME 2
+#define STATIC_MSG_CGP_SYNC 3
static void guc_exec_queue_destroy(struct xe_exec_queue *q)
{
struct xe_sched_msg *msg = q->guc->static_msgs + STATIC_MSG_CLEANUP;
- if (!(q->flags & EXEC_QUEUE_FLAG_PERMANENT) && !exec_queue_wedged(q))
- guc_exec_queue_add_msg(q, msg, CLEANUP);
- else
- __guc_exec_queue_destroy(exec_queue_to_guc(q), q);
+ guc_exec_queue_add_msg(q, msg, CLEANUP);
}
static int guc_exec_queue_set_priority(struct xe_exec_queue *q,
@@ -2601,7 +2660,7 @@ static void guc_exec_queue_stop(struct xe_guc *guc, struct xe_exec_queue *q)
}
if (do_destroy)
- __guc_exec_queue_destroy(guc, q);
+ guc_exec_queue_destroy_async(q);
}
static int guc_submit_reset_prepare(struct xe_guc *guc)
@@ -2717,6 +2776,57 @@ static void guc_exec_queue_revert_pending_state_change(struct xe_guc *guc,
q->guc->id);
}
+ /*
+ * A registration time CGP update that bailed when woken by VF recovery.
+ * Re-register the queue.
+ */
+ if (q->guc->multi_queue.re_register) {
+ clear_exec_queue_registered(q);
+ q->guc->multi_queue.re_register = false;
+ xe_gt_dbg(guc_to_gt(guc), "Replay REGISTER (cgp) - guc_id=%d",
+ q->guc->id);
+ }
+
+ /*
+ * If a CGP update gets dropped during migration, CGP_SYNC_DONE will not
+ * be received (sync_pending still set and this queue owns it). Recover
+ * it the same way and clear the stuck sync_pending.
+ */
+ if (xe_exec_queue_is_multi_queue(q)) {
+ struct xe_exec_queue_group *group = q->multi_queue.group;
+
+ if (q == READ_ONCE(group->cgp_update_q) &&
+ READ_ONCE(group->sync_pending)) {
+ if (q->guc->multi_queue.registering_cgp) {
+ clear_exec_queue_registered(q);
+ xe_gt_dbg(guc_to_gt(guc), "Replay REGISTER (cgp sync) - guc_id=%d",
+ q->guc->id);
+ } else if (q->guc->multi_queue.updating_cgp) {
+ q->guc->multi_queue.needs_cgp_sync = true;
+ xe_gt_dbg(guc_to_gt(guc), "Replay CGP_SYNC - guc_id=%d",
+ q->guc->id);
+ }
+ q->guc->multi_queue.registering_cgp = false;
+ q->guc->multi_queue.updating_cgp = false;
+ WRITE_ONCE(group->cgp_update_q, NULL);
+ WRITE_ONCE(group->sync_pending, false);
+ }
+ }
+
+ /*
+ * A dynamic-time CGP update that bailed when woken by VF recovery.
+ * Replay the dynamic CGP update unless the queue is registered or being
+ * re-registered, which re-does the CGP anyway.
+ */
+ if (q->guc->multi_queue.re_update) {
+ q->guc->multi_queue.re_update = false;
+ if (exec_queue_registered(q)) {
+ q->guc->multi_queue.needs_cgp_sync = true;
+ xe_gt_dbg(guc_to_gt(guc), "Replay CGP_SYNC (re-update) - guc_id=%d",
+ q->guc->id);
+ }
+ }
+
q->guc->resume_time = 0;
}
@@ -2737,17 +2847,24 @@ static void lrc_parallel_clear(struct xe_lrc *lrc)
* during VF resume flows. The function scans the queue state, make adjustments
* as needed, and queues jobs / messages which replayed upon unpause.
*/
-static void guc_exec_queue_pause(struct xe_guc *guc, struct xe_exec_queue *q)
+static void guc_exec_queue_pause_prepare(struct xe_guc *guc, struct xe_exec_queue *q)
{
struct xe_gpu_scheduler *sched = &q->guc->sched;
- struct xe_sched_job *job;
- int i;
lockdep_assert_held(&guc->submission_state.lock);
/* Stop scheduling + flush any DRM scheduler operations */
xe_sched_submission_stop(sched);
cancel_delayed_work_sync(&sched->base.work_tdr);
+}
+
+static void guc_exec_queue_pause(struct xe_guc *guc, struct xe_exec_queue *q)
+{
+ struct xe_gpu_scheduler *sched = &q->guc->sched;
+ struct xe_sched_job *job;
+ int i;
+
+ lockdep_assert_held(&guc->submission_state.lock);
guc_exec_queue_revert_pending_state_change(guc, q);
@@ -2806,6 +2923,19 @@ void xe_guc_submit_pause_vf(struct xe_guc *guc)
xe_gt_assert(guc_to_gt(guc), vf_recovery(guc));
mutex_lock(&guc->submission_state.lock);
+ /*
+ * Stop all schedulers before reverting any queue: in a multi-queue
+ * group a secondary's run_job() can register the primary, which must
+ * not race an in-progress revert.
+ */
+ xa_for_each(&guc->submission_state.exec_queue_lookup, index, q) {
+ /* Prevent redundant attempts to stop parallel queues */
+ if (q->guc->id != index)
+ continue;
+
+ guc_exec_queue_pause_prepare(guc, q);
+ }
+
xa_for_each(&guc->submission_state.exec_queue_lookup, index, q) {
/* Prevent redundant attempts to stop parallel queues */
if (q->guc->id != index)
@@ -2922,6 +3052,16 @@ static void guc_exec_queue_replay_pending_state_change(struct xe_exec_queue *q)
struct xe_gpu_scheduler *sched = &q->guc->sched;
struct xe_sched_msg *msg;
+ if (q->guc->multi_queue.needs_cgp_sync) {
+ msg = q->guc->static_msgs + STATIC_MSG_CGP_SYNC;
+
+ xe_sched_msg_lock(sched);
+ guc_exec_queue_try_add_msg_head(q, msg, CGP_SYNC_MSG);
+ xe_sched_msg_unlock(sched);
+
+ q->guc->multi_queue.needs_cgp_sync = false;
+ }
+
if (q->guc->needs_cleanup) {
msg = q->guc->static_msgs + STATIC_MSG_CLEANUP;
@@ -3166,7 +3306,7 @@ static void handle_deregister_done(struct xe_guc *guc, struct xe_exec_queue *q)
trace_xe_exec_queue_deregister_done(q);
clear_exec_queue_registered(q);
- __guc_exec_queue_destroy(guc, q);
+ guc_exec_queue_destroy_async(q);
}
int xe_guc_deregister_done_handler(struct xe_guc *guc, u32 *msg, u32 len)
@@ -3405,7 +3545,8 @@ int xe_guc_exec_queue_cgp_context_error_handler(struct xe_guc *guc, u32 *msg,
int xe_guc_exec_queue_cgp_sync_done_handler(struct xe_guc *guc, u32 *msg, u32 len)
{
struct xe_device *xe = guc_to_xe(guc);
- struct xe_exec_queue *q;
+ struct xe_exec_queue_group *group;
+ struct xe_exec_queue *q, *upd_q;
u32 guc_id = msg[0];
if (unlikely(len < 1)) {
@@ -3422,8 +3563,20 @@ int xe_guc_exec_queue_cgp_sync_done_handler(struct xe_guc *guc, u32 *msg, u32 le
return -EPROTO;
}
+ /*
+ * The outstanding CGP update is now confirmed; clear the owning queue's
+ * tracking so a later migration does not needlessly replay it.
+ */
+ group = q->multi_queue.group;
+ upd_q = READ_ONCE(group->cgp_update_q);
+ if (upd_q) {
+ upd_q->guc->multi_queue.registering_cgp = false;
+ upd_q->guc->multi_queue.updating_cgp = false;
+ WRITE_ONCE(group->cgp_update_q, NULL);
+ }
+
/* Wakeup the serialized cgp update wait */
- WRITE_ONCE(q->multi_queue.group->sync_pending, false);
+ WRITE_ONCE(group->sync_pending, false);
xe_guc_ct_wake_waiters(&guc->ct);
return 0;
diff --git a/drivers/gpu/drm/xe/xe_guc_types.h b/drivers/gpu/drm/xe/xe_guc_types.h
index 31a2acb63ac3..3dbcb1331690 100644
--- a/drivers/gpu/drm/xe/xe_guc_types.h
+++ b/drivers/gpu/drm/xe/xe_guc_types.h
@@ -122,6 +122,12 @@ struct xe_guc {
struct xe_reg notify_reg;
/** @params: Control params for fw initialization */
u32 params[GUC_CTL_MAX_DWORDS];
+
+ /**
+ * @pagefault_ack_counter: Counter to determine when periodically ack
+ * pagefaults in a batch.
+ */
+ u32 pagefault_ack_counter;
};
#endif
diff --git a/drivers/gpu/drm/xe/xe_hwmon.c b/drivers/gpu/drm/xe/xe_hwmon.c
index de3f2aeffc3f..5284cab6703d 100644
--- a/drivers/gpu/drm/xe/xe_hwmon.c
+++ b/drivers/gpu/drm/xe/xe_hwmon.c
@@ -266,7 +266,7 @@ static struct xe_reg xe_hwmon_get_reg(struct xe_hwmon *hwmon, enum xe_hwmon_reg
switch (hwmon_reg) {
case REG_TEMP:
- if (xe->info.platform == XE_BATTLEMAGE) {
+ if (xe->info.platform == XE_BATTLEMAGE || xe->info.platform == XE_CRESCENTISLAND) {
if (channel == CHANNEL_PKG)
return BMG_PACKAGE_TEMPERATURE;
else if (channel == CHANNEL_VRAM)
@@ -575,23 +575,36 @@ xe_hwmon_power_max_interval_show(struct device *dev, struct device_attribute *at
mutex_unlock(&hwmon->hwmon_lock);
- x = REG_FIELD_GET(PWR_LIM_TIME_X, reg_val);
- y = REG_FIELD_GET(PWR_LIM_TIME_Y, reg_val);
+ if (hwmon->xe->info.platform >= XE_CRESCENTISLAND) {
+ /**
+ * On CRI and newer platforms, the interval encoding changed.
+ * The value is now stored directly in milliseconds as U5.2,
+ * replacing the older 1.x * 2^y representation.
+ * Bits [6:2] hold the integer part and bits [1:0] the fraction,
+ * so convert to milliseconds by extracting integer/fractional parts.
+ * Round fraction value to 1 when it is >= 0.5.
+ */
+ reg_val = REG_FIELD_GET(PWR_LIM_TIME, reg_val);
+ out = (u64)((reg_val >> 2) + ((reg_val & 0x3) >= 2));
+ } else {
+ x = REG_FIELD_GET(PWR_LIM_TIME_X, reg_val);
+ y = REG_FIELD_GET(PWR_LIM_TIME_Y, reg_val);
- /*
- * tau = (1 + (x / 4)) * power(2,y), x = bits(23:22), y = bits(21:17)
- * = (4 | x) << (y - 2)
- *
- * Here (y - 2) ensures a 1.x fixed point representation of 1.x
- * As x is 2 bits so 1.x can be 1.0, 1.25, 1.50, 1.75
- *
- * As y can be < 2, we compute tau4 = (4 | x) << y
- * and then add 2 when doing the final right shift to account for units
- */
- tau4 = (u64)((1 << x_w) | x) << y;
+ /*
+ * tau = (1 + (x / 4)) * power(2,y), x = bits(23:22), y = bits(21:17)
+ * = (4 | x) << (y - 2)
+ *
+ * Here (y - 2) ensures a 1.x fixed point representation of 1.x
+ * As x is 2 bits so 1.x can be 1.0, 1.25, 1.50, 1.75
+ *
+ * As y can be < 2, we compute tau4 = (4 | x) << y
+ * and then add 2 when doing the final right shift to account for units
+ */
+ tau4 = (u64)((1 << x_w) | x) << y;
- /* val in hwmon interface units (millisec) */
- out = mul_u64_u32_shr(tau4, SF_TIME, hwmon->scl_shift_time + x_w);
+ /* val in hwmon interface units (millisec) */
+ out = mul_u64_u32_shr(tau4, SF_TIME, hwmon->scl_shift_time + x_w);
+ }
return sysfs_emit(buf, "%llu\n", out);
}
@@ -637,24 +650,35 @@ xe_hwmon_power_max_interval_store(struct device *dev, struct device_attribute *a
if (val > max_win)
return -EINVAL;
- /* val in hw units */
- val = DIV_ROUND_CLOSEST_ULL((u64)val << hwmon->scl_shift_time, SF_TIME) + 1;
-
- /*
- * Convert val to 1.x * power(2,y)
- * y = ilog2(val)
- * x = (val - (1 << y)) >> (y - 2)
- */
- if (!val) {
- y = 0;
- x = 0;
+ if (hwmon->xe->info.platform >= XE_CRESCENTISLAND) {
+ /**
+ * On CRI and newer platforms, the interval encoding changed.
+ * The value is now stored directly in milliseconds as U5.2,
+ * replacing the older 1.x * 2^y representation.
+ * Bits [6:2] hold the integer part and bits [1:0] the fraction,so
+ * convert from milliseconds by shifting the value left by 2 to fit into the field.
+ */
+ rxy = REG_FIELD_PREP(PWR_LIM_TIME, (val << 2));
} else {
- y = ilog2(val);
- x = (val - (1ul << y)) << x_w >> y;
- }
+ /* val in hw units */
+ val = DIV_ROUND_CLOSEST_ULL((u64)val << hwmon->scl_shift_time, SF_TIME) + 1;
+
+ /*
+ * Convert val to 1.x * power(2,y)
+ * y = ilog2(val)
+ * x = (val - (1 << y)) >> (y - 2)
+ */
+ if (!val) {
+ y = 0;
+ x = 0;
+ } else {
+ y = ilog2(val);
+ x = (val - (1ul << y)) << x_w >> y;
+ }
- rxy = REG_FIELD_PREP(PWR_LIM_TIME_X, x) |
- REG_FIELD_PREP(PWR_LIM_TIME_Y, y);
+ rxy = REG_FIELD_PREP(PWR_LIM_TIME_X, x) |
+ REG_FIELD_PREP(PWR_LIM_TIME_Y, y);
+ }
guard(xe_pm_runtime)(hwmon->xe);
diff --git a/drivers/gpu/drm/xe/xe_log.c b/drivers/gpu/drm/xe/xe_log.c
new file mode 100644
index 000000000000..5549ef6966fd
--- /dev/null
+++ b/drivers/gpu/drm/xe/xe_log.c
@@ -0,0 +1,235 @@
+// SPDX-License-Identifier: MIT
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#include <kunit/static_stub.h>
+#include <kunit/visibility.h>
+
+#include "abi/xe_log_abi.h"
+
+#include "xe_device.h"
+#include "xe_log.h"
+#include "xe_printk.h"
+
+static void log_emit_cper(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid,
+ u32 component, u32 location, const void *data, size_t len,
+ struct va_format *vaf)
+{
+ KUNIT_STATIC_STUB_REDIRECT(log_emit_cper, pdev, cper_sev, sigid,
+ component, location, data, len, vaf);
+ /* TODO */
+}
+
+static const char *log_unknown_component_prefix(u32 component)
+{
+ u32 class = FIELD_GET(XE_LOG_COMPONENT_CLASS_MASK, component);
+ u32 type = FIELD_GET(XE_LOG_COMPONENT_TYPE_MASK, component);
+
+ WARN(IS_ENABLED(CONFIG_DRM_XE_DEBUG), "LOG: unrecognized component %u.%u\n", class, type);
+ switch (class) {
+#define MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(_CLASS) \
+ case XE_LOG_COMPONENT_CLASS_##_CLASS: return #_CLASS "? ";
+ MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(SYSTEM)
+ MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(DRIVER)
+ MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(FEATURE)
+ MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(FIRMWARE)
+ MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(HARDWARE)
+#undef MAKE_XE_LOG_COMPONENT_CLASS_PREFIX
+ }
+ return "COMP? ";
+}
+
+static const char *log_component_prefix(u32 component)
+{
+ switch (component) {
+#define MAKE_XE_LOG_COMPONENT_CASE_PREFIX(_CLASS, _ID, _TAG, _SIG, _NAME) \
+ case XE_LOG_COMPONENT_##_TAG: return #_TAG ": ";
+ DEFINE_XE_LOG_COMPONENTS(MAKE_XE_LOG_COMPONENT_CASE_PREFIX)
+#undef MAKE_XE_LOG_COMPONENT_CASE_PREFIX
+ }
+
+ return component ? log_unknown_component_prefix(component) : "";
+}
+
+static struct xe_gt *get_gt_safe(struct pci_dev *pdev, u8 id)
+{
+ struct xe_device *xe = pdev_to_xe_device(pdev);
+
+ return xe ? xe_device_get_gt(xe, id) : NULL;
+}
+
+static struct xe_tile *get_tile_safe(struct pci_dev *pdev, u8 id)
+{
+ struct xe_device *xe = pdev_to_xe_device(pdev);
+
+ return xe && id < xe->info.tile_count ? &xe->tiles[id] : NULL;
+}
+
+static const char *log_location_prefix(struct pci_dev *pdev, u32 location, char *buf, size_t size)
+{
+ u32 type = FIELD_GET(XE_LOG_LOCATION_TYPE_MASK, location);
+ u32 id = FIELD_GET(XE_LOG_LOCATION_ID_MASK, location);
+
+ if (!location || type == XE_LOG_LOCATION_TYPE_DEVICE) {
+ if (id)
+ goto unrecognized;
+ strscpy(buf, "", size);
+ } else if (type == XE_LOG_LOCATION_TYPE_TILE) {
+ struct xe_tile *tile = get_tile_safe(pdev, id);
+
+ if (!tile)
+ goto unrecognized;
+ snprintf(buf, size, "Tile%u: ", id);
+ } else if (type == XE_LOG_LOCATION_TYPE_GT) {
+ struct xe_gt *gt = get_gt_safe(pdev, id);
+
+ if (!gt)
+ goto unrecognized;
+ snprintf(buf, size, "Tile%u: GT%u: ", gt->tile->id, id);
+ } else {
+ goto unrecognized;
+ }
+
+ return buf;
+
+unrecognized:
+ pci_WARN(pdev, IS_ENABLED(CONFIG_DRM_XE_DEBUG),
+ "LOG: unrecognized location %u.%u\n", type, id);
+ snprintf(buf, size, "LOC%u.%u? ", type, id);
+ return buf;
+}
+
+static bool is_hw_sigid(enum xe_sigid sigid)
+{
+ return (int)sigid >= INTEL_SIGID_GPU_XE_HARDWARE_START;
+}
+
+static bool is_sev_error(int cper_sev)
+{
+ return cper_sev != CPER_SEV_INFORMATIONAL;
+}
+
+static const char *log_hwe_prefix(int cper_sev, enum xe_sigid sigid)
+{
+ return is_sev_error(cper_sev) && is_hw_sigid(sigid) ? HW_ERR : "";
+}
+
+static const char *log_sev_prefix(int cper_sev)
+{
+ switch (cper_sev) {
+ case CPER_SEV_FATAL:
+ return "FATAL ";
+ case CPER_SEV_RECOVERABLE:
+ return "";
+ case CPER_SEV_CORRECTED:
+ return "CORRECTED ";
+ case CPER_SEV_INFORMATIONAL:
+ return "";
+ default:
+ WARN(IS_ENABLED(CONFIG_DRM_XE_DEBUG), "LOG: unknown severity %d\n", cper_sev);
+ return "";
+ }
+}
+
+#define __LOG_DRM_PRINTK_FMT(fmt, args...) "[drm] " fmt, ##args
+#define __LOG_DRM_PRINTK_ERR_FMT(fmt, args...) __LOG_DRM_PRINTK_FMT("*ERROR* " fmt, args)
+
+static void log_dmesg_vprintk(struct pci_dev *pdev, int cper_sev, struct va_format *vaf)
+{
+ KUNIT_STATIC_STUB_REDIRECT(log_dmesg_vprintk, pdev, cper_sev, vaf);
+
+ if (cper_sev == CPER_SEV_INFORMATIONAL)
+ pci_info(pdev, __LOG_DRM_PRINTK_FMT("%pV", vaf));
+ else
+ pci_err(pdev, __LOG_DRM_PRINTK_ERR_FMT("%pV", vaf));
+}
+
+static void log_dmesg_printf(struct pci_dev *pdev, int cper_sev, const char *fmt, ...)
+{
+ struct va_format vaf;
+ va_list args;
+
+ va_start(args, fmt);
+ vaf.fmt = fmt;
+ vaf.va = &args;
+
+ log_dmesg_vprintk(pdev, cper_sev, &vaf);
+
+ va_end(args);
+}
+
+static void log_emit_dmesg(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid,
+ u32 component, u32 location, const void *data, size_t len,
+ struct va_format *vaf)
+{
+ char buf[32];
+ const char *loc_prefix = log_location_prefix(pdev, location, buf, sizeof(buf));
+ const char *comp_prefix = log_component_prefix(component);
+ const char *hwe_prefix = log_hwe_prefix(cper_sev, sigid);
+ const char *sev_prefix = log_sev_prefix(cper_sev);
+
+ if (IS_ERR(data))
+ log_dmesg_printf(pdev, cper_sev, "SIGID=%u %s(%pe) %s%s%s%pV",
+ sigid, sev_prefix, data, hwe_prefix,
+ loc_prefix, comp_prefix, vaf);
+ else if (data && len)
+ log_dmesg_printf(pdev, cper_sev, "SIGID=%u %s(%*phN) %s%s%s%pV",
+ sigid, sev_prefix, (int)len, data, hwe_prefix,
+ loc_prefix, comp_prefix, vaf);
+ else
+ log_dmesg_printf(pdev, cper_sev, "SIGID=%u %s%s%s%s%pV",
+ sigid, sev_prefix, hwe_prefix,
+ loc_prefix, comp_prefix, vaf);
+}
+
+/**
+ * __xe_log_emit() - Emit a structured SIGID log entry
+ * @pdev: the &pci_dev device
+ * @cper_sev: CPER severity (CPER_SEV_FATAL, CPER_SEV_RECOVERABLE, ...)
+ * @sigid: signature identifier, see &enum xe_sigid
+ * @component: component identifier
+ * @location: location details of the @component
+ * @data: pointer to the additional details, or ERR_PTR, or NULL if not applicable
+ * @len: length of the @data in bytes, or 0 if not applicable
+ * @fmt: printf-style format string
+ * @...: format arguments
+ *
+ * Emits a dmesg line that includes a single stable, machine-matchable token
+ * ``SIGID=<n>`` followed by the optional severity token (like ``FATAL``) and,
+ * when @data pointer is set, either the error printed with %pe or a packed hex
+ * dump of the @data binary blob. The dmesg line will also include printf-style
+ * text message.
+ *
+ * Note that the full dmesg line, with the free text message, is only a debugging
+ * aid, not an interface! Only the ``SIGID=<n>`` token is stable there.
+ * The durable machine record is the CPER carrying the same SIGID.
+ *
+ * Note: generation of the CPER record is a planned follow-up.
+ *
+ * Examples::
+ *
+ * <3> xe 0000:03:00.0: [drm] *ERROR* SIGID=104 FATAL (-EPROTO) Invalid GuC reply
+ * <3> xe 0000:03:00.0: [drm] *ERROR* SIGID=106 (-ETIMEDOUT) Engine 'rcs0' hung
+ * <6> xe 0000:03:00.0: [drm] SIGID=103 In survivability mode
+ */
+void __xe_log_emit(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid,
+ u32 component, u32 location, const void *data, size_t len,
+ const char *fmt, ...)
+{
+ struct va_format vaf;
+ va_list args;
+
+ va_start(args, fmt);
+ vaf.fmt = fmt;
+ vaf.va = &args;
+
+ log_emit_dmesg(pdev, cper_sev, sigid, component, location, data, len, &vaf);
+ log_emit_cper(pdev, cper_sev, sigid, component, location, data, len, &vaf);
+
+ va_end(args);
+}
+
+#if IS_BUILTIN(CONFIG_DRM_XE_KUNIT_TEST)
+#include "tests/xe_log_kunit.c"
+#endif
diff --git a/drivers/gpu/drm/xe/xe_log.h b/drivers/gpu/drm/xe/xe_log.h
new file mode 100644
index 000000000000..bb1185143589
--- /dev/null
+++ b/drivers/gpu/drm/xe/xe_log.h
@@ -0,0 +1,194 @@
+/* SPDX-License-Identifier: MIT */
+/*
+ * Copyright © 2026 Intel Corporation
+ */
+
+#ifndef _XE_LOG_H_
+#define _XE_LOG_H_
+
+#include <linux/cper.h>
+#include <linux/err.h>
+
+#include "abi/xe_log_abi.h"
+#include "abi/xe_sigid_abi.h"
+#include "xe_any.h"
+
+struct pci_dev;
+
+__printf(8, 9)
+void __xe_log_emit(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid,
+ u32 component, u32 location, const void *data, size_t len,
+ const char *fmt, ...);
+
+#define __xe_log_const_sev_to_level(sev) \
+ (__builtin_constant_p(sev) ? (sev) == CPER_SEV_INFORMATIONAL ? KERN_INFO : KERN_ERR : NULL)
+
+#define __xe_log_emit_printk_index(level, fmt) \
+ dev_printk_index_emit(level, "%s SIGID=%u %s" fmt);
+
+#define xe_log_emit(pdev, sev, sig, comp, loc, data, len, fmt, args...) do { \
+ __xe_log_emit_printk_index(__xe_log_const_sev_to_level(sev), fmt); \
+ __xe_log_emit((pdev), (sev), (sig), (comp), (loc), (data), (len), fmt, ##args); \
+} while (0)
+
+#define xe_log_emit_fatal(pdev, sig, comp, loc, data, len, fmt, args...) \
+ xe_log_emit((pdev), CPER_SEV_FATAL, (sig), (comp), (loc), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_emit_recoverable(pdev, sig, comp, loc, data, len, fmt, args...) \
+ xe_log_emit((pdev), CPER_SEV_RECOVERABLE, (sig), (comp), (loc), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_emit_corrected(pdev, sig, comp, loc, data, len, fmt, args...) \
+ xe_log_emit((pdev), CPER_SEV_CORRECTED, (sig), (comp), (loc), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_emit_info(pdev, sig, comp, loc, data, len, fmt, args...) \
+ xe_log_emit((pdev), CPER_SEV_INFORMATIONAL, (sig), (comp), (loc), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_location_type(any) \
+ _Generic((any), \
+ struct xe_gt * : XE_LOG_LOCATION_TYPE_GT, \
+ const struct xe_gt * : XE_LOG_LOCATION_TYPE_GT, \
+ struct xe_tile * : XE_LOG_LOCATION_TYPE_TILE, \
+ const struct xe_tile * : XE_LOG_LOCATION_TYPE_TILE, \
+ struct xe_device * : XE_LOG_LOCATION_TYPE_DEVICE, \
+ const struct xe_device * : XE_LOG_LOCATION_TYPE_DEVICE, \
+ struct pci_dev * : XE_LOG_LOCATION_TYPE_DEVICE, \
+ struct device * : XE_LOG_LOCATION_TYPE_DEVICE)
+
+#define xe_log_location(any) \
+ PREP_XE_LOG_LOCATION(xe_log_location_type(any), xe_any_id(any))
+
+/**
+ * xe_log_from() - Emit a structured SIGID log entry using @any pointer as location.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @cper_sev: CPER severity (CPER_SEV_FATAL, CPER_SEV_RECOVERABLE, ...)
+ * @sigid: signature identifier, see &enum xe_sigid
+ * @component: component identifer
+ * @data: pointer to the additional details, or ERR_PTR, or NULL if not applicable
+ * @len: length of the @data in bytes, or 0 if not applicable
+ * @fmt: printf-style format string
+ * @args: arguments for the @fmt format string
+ *
+ * The location used to emit SIGID entry will be based on the @any pointer type.
+ * See xe_log_emit() for more details.
+ */
+#define xe_log_from(any, cper_sev, sigid, component, data, len, fmt, args...) do { \
+ typeof(any) ___any = (any); \
+ xe_log_emit(xe_any_to_pdev(___any), (cper_sev), (sigid), (component), \
+ xe_log_location(___any), (data), (len), fmt, ##args); \
+} while (0)
+
+#define xe_log_from_fatal(any, sig, comp, data, len, fmt, args...) \
+ xe_log_from((any), CPER_SEV_FATAL, (sig), (comp), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_from_recoverable(any, sig, comp, data, len, fmt, args...) \
+ xe_log_from((any), CPER_SEV_RECOVERABLE, (sig), (comp), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_from_corrected(any, sig, comp, data, len, fmt, args...) \
+ xe_log_from((any), CPER_SEV_CORRECTED, (sig), (comp), \
+ (data), (len), fmt, ##args)
+
+#define xe_log_from_info(any, sig, comp, data, len, fmt, args...) \
+ xe_log_from((any), CPER_SEV_INFORMATIONAL, (sig), (comp), \
+ (data), (len), fmt, ##args)
+
+/**
+ * xe_log_comp() - Emit a structured SIGID log entry on the component behalf.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @cper_sev: CPER severity (CPER_SEV_FATAL, CPER_SEV_RECOVERABLE, ...)
+ * @TAG: the component tag to use
+ * @data: pointer to the additional details, or ERR_PTR, or NULL if not applicable
+ * @len: length of the @data in bytes, or 0 if not applicable
+ * @fmt: printf-style free text format string (not a stable interface)
+ * @args: arguments for the @fmt format string
+ *
+ * The SIGID will be determined from the component's @TAG.
+ * The component identifier will be determined from the component's @TAG.
+ * The location used to emit SIGID entry will be based on the @any pointer type.
+ */
+#define xe_log_comp(any, cper_sev, TAG, data, len, fmt, args...) \
+ xe_log_from((any), (cper_sev), (int)XE_LOG_COMPONENT_##TAG##_SIGID, \
+ XE_LOG_COMPONENT_##TAG, (data), (len), fmt, ##args)
+
+#define xe_log_comp_fatal(any, TAG, data, len, fmt, args...) \
+ xe_log_comp((any), CPER_SEV_FATAL, TAG, (data), (len), fmt, ##args)
+
+#define xe_log_comp_recoverable(any, TAG, data, len, fmt, args...) \
+ xe_log_comp((any), CPER_SEV_RECOVERABLE, TAG, (data), (len), fmt, ##args)
+
+#define xe_log_comp_corrected(any, TAG, data, len, fmt, args...) \
+ xe_log_comp((any), CPER_SEV_CORRECTED, TAG, (data), (len), fmt, ##args)
+
+#define xe_log_comp_info(any, TAG, data, len, fmt, args...) \
+ xe_log_comp((any), CPER_SEV_INFORMATIONAL, TAG, (data), (len), fmt, ##args)
+
+/**
+ * xe_log_err() - Emit a structured SIGID error log entry on the component behalf.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @TAG: the component tag to use
+ * @err: negative errno for the failing operation, or 0 if not applicable
+ * @fmt: printf-style free text format string (not a stable interface)
+ * @args: arguments for the @fmt format string
+ *
+ * The log entry will be emitted with @CPER_SEV_RECOVERABLE severity.
+ */
+#define xe_log_err(any, TAG, err, fmt, args...) \
+ xe_log_comp_recoverable((any), TAG, ERR_PTR(err), 0, fmt, ##args)
+
+/**
+ * xe_log_err_fatal() - Emit a structured SIGID error log entry on the component behalf.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @TAG: the component tag to use
+ * @err: negative errno for the failing operation, or 0 if not applicable
+ * @fmt: printf-style free text format string (not a stable interface)
+ * @args: arguments for the @fmt format string
+ *
+ * The log entry will be emitted with @CPER_SEV_FATAL severity.
+ */
+#define xe_log_err_fatal(any, TAG, err, fmt, args...) \
+ xe_log_comp_fatal((any), TAG, ERR_PTR(err), 0, fmt, ##args)
+
+/**
+ * xe_log_err_corrected() - Emit a structured SIGID error log entry on the component behalf.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @TAG: the component tag to use
+ * @err: negative errno for the failing operation, or 0 if not applicable
+ * @fmt: printf-style free text format string (not a stable interface)
+ * @args: arguments for the @fmt format string
+ *
+ * The log entry will be emitted with @CPER_SEV_CORRECTED severity.
+ */
+#define xe_log_err_corrected(any, TAG, err, fmt, args...) \
+ xe_log_comp_corrected((any), TAG, ERR_PTR(err), 0, fmt, ##args)
+
+/**
+ * xe_log_err_info() - Emit a structured SIGID error log entry on the component behalf.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @TAG: the component tag to use
+ * @err: negative errno for the failing operation, or 0 if not applicable
+ * @fmt: printf-style free text format string (not a stable interface)
+ * @args: arguments for the @fmt format string
+ *
+ * The log entry will be emitted with @CPER_SEV_INFORMATIONAL severity.
+ */
+#define xe_log_err_info(any, TAG, err, fmt, args...) \
+ xe_log_comp_info((any), TAG, ERR_PTR(err), 0, fmt, ##args)
+
+/**
+ * xe_log_info() - Emit a structured SIGID information log entry on the component behalf.
+ * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to
+ * @TAG: the component tag to use
+ * @fmt: printf-style free text format string (not a stable interface)
+ * @args: arguments for the @fmt format string
+ *
+ * The log entry will be emitted with @CPER_SEV_INFORMATIONAL severity.
+ */
+#define xe_log_info(any, TAG, fmt, args...) \
+ xe_log_err_info((any), TAG, 0, fmt, ##args)
+
+#endif
diff --git a/drivers/gpu/drm/xe/xe_mert.c b/drivers/gpu/drm/xe/xe_mert.c
index f637df95418b..32700c19a1df 100644
--- a/drivers/gpu/drm/xe/xe_mert.c
+++ b/drivers/gpu/drm/xe/xe_mert.c
@@ -77,7 +77,7 @@ static void mert_handle_cat_error(struct xe_device *xe)
xe_device_declare_wedged(xe);
break;
case CATERR_LMTT_FAULT:
- xe_sriov_dbg(xe, "MERT: CAT_ERR: VF%u LMTT fault!\n", vfid);
+ xe_sriov_dbg_ratelimited(xe, "MERT: CAT_ERR: VF%u LMTT fault!\n", vfid);
/* XXX: track/report malicious VF activity */
break;
default:
diff --git a/drivers/gpu/drm/xe/xe_migrate.c b/drivers/gpu/drm/xe/xe_migrate.c
index f79d0047bec6..75b83687f1b5 100644
--- a/drivers/gpu/drm/xe/xe_migrate.c
+++ b/drivers/gpu/drm/xe/xe_migrate.c
@@ -493,7 +493,6 @@ int xe_migrate_init(struct xe_migrate *m)
*/
m->q = xe_exec_queue_create(xe, vm, logical_mask, 1, hwe0,
EXEC_QUEUE_FLAG_KERNEL |
- EXEC_QUEUE_FLAG_PERMANENT |
EXEC_QUEUE_FLAG_HIGH_PRIORITY |
EXEC_QUEUE_FLAG_MIGRATE |
EXEC_QUEUE_FLAG_LOW_LATENCY, 0);
@@ -501,7 +500,6 @@ int xe_migrate_init(struct xe_migrate *m)
m->q = xe_exec_queue_create_class(xe, primary_gt, vm,
XE_ENGINE_CLASS_COPY,
EXEC_QUEUE_FLAG_KERNEL |
- EXEC_QUEUE_FLAG_PERMANENT |
EXEC_QUEUE_FLAG_MIGRATE, 0);
}
if (IS_ERR(m->q)) {
diff --git a/drivers/gpu/drm/xe/xe_module.c b/drivers/gpu/drm/xe/xe_module.c
index 848d65265443..4bc28dfc1992 100644
--- a/drivers/gpu/drm/xe/xe_module.c
+++ b/drivers/gpu/drm/xe/xe_module.c
@@ -29,6 +29,7 @@ struct xe_modparam xe_modparam = {
.max_vfs = XE_DEFAULT_MAX_VFS,
#endif
.wedged_mode = XE_DEFAULT_WEDGED_MODE,
+ .num_pf_work = XE_DEFAULT_NUM_PF_WORK,
.svm_notifier_size = XE_DEFAULT_SVM_NOTIFIER_SIZE,
/* the rest are 0 by default */
};
@@ -81,6 +82,9 @@ MODULE_PARM_DESC(wedged_mode,
"Module's default policy for the wedged mode (0=never, 1=upon-critical-error, 2=upon-any-hang-no-reset "
"[default=" XE_DEFAULT_WEDGED_MODE_STR "])");
+module_param_named(num_pf_work, xe_modparam.num_pf_work, uint, 0600);
+MODULE_PARM_DESC(num_pf_work, "Number of page fault work threads, default=2, min=1, max=8");
+
static int xe_check_nomodeset(void)
{
if (drm_firmware_drivers_only())
diff --git a/drivers/gpu/drm/xe/xe_module.h b/drivers/gpu/drm/xe/xe_module.h
index a0eb7db07770..6272d9e41207 100644
--- a/drivers/gpu/drm/xe/xe_module.h
+++ b/drivers/gpu/drm/xe/xe_module.h
@@ -23,6 +23,7 @@ struct xe_modparam {
unsigned int max_vfs;
#endif
unsigned int wedged_mode;
+ unsigned int num_pf_work;
u32 svm_notifier_size;
};
diff --git a/drivers/gpu/drm/xe/xe_oa.c b/drivers/gpu/drm/xe/xe_oa.c
index 9c5384b95c63..b460fcdfca15 100644
--- a/drivers/gpu/drm/xe/xe_oa.c
+++ b/drivers/gpu/drm/xe/xe_oa.c
@@ -2581,10 +2581,11 @@ static u32 __hwe_oam_unit(struct xe_hw_engine *hwe)
return XE_OA_UNIT_INVALID;
else if (!IS_DGFX(gt_to_xe(hwe->gt)))
return XE_OAM_UNIT_SCMI_0;
- else if (hwe->class == XE_ENGINE_CLASS_VIDEO_DECODE)
- return (hwe->instance / 2 & 0x1) + 1;
- else if (hwe->class == XE_ENGINE_CLASS_VIDEO_ENHANCE)
+ else if (hwe->class == XE_ENGINE_CLASS_VIDEO_ENHANCE &&
+ MEDIA_VERx100(gt_to_xe(hwe->gt)) < 3500)
return (hwe->instance & 0x1) + 1;
+ else
+ return (hwe->instance / 2 & 0x1) + 1;
return XE_OA_UNIT_INVALID;
}
diff --git a/drivers/gpu/drm/xe/xe_pagefault.c b/drivers/gpu/drm/xe/xe_pagefault.c
index dd3c068e1a39..9e89d7b37395 100644
--- a/drivers/gpu/drm/xe/xe_pagefault.c
+++ b/drivers/gpu/drm/xe/xe_pagefault.c
@@ -14,6 +14,7 @@
#include "xe_gt_types.h"
#include "xe_gt_stats.h"
#include "xe_hw_engine.h"
+#include "xe_log.h"
#include "xe_pagefault.h"
#include "xe_pagefault_types.h"
#include "xe_svm.h"
@@ -35,6 +36,73 @@
* xe_pagefault.c implements the consumer layer.
*/
+/**
+ * DOC: Xe page fault cache
+ *
+ * Some Xe hardware can trigger “fault storms,” which are many page faults to
+ * the same address within a short period of time. An example is many EU threads
+ * faulting on the same page simultaneously. With the current page fault locking
+ * structure, only one page fault for a given address range can be processed at
+ * a time. This causes head-of-queue blocking across workers, killing
+ * parallelism. If the page fault handler must repeatedly look up resources
+ * (VMAs, ranges) to determine that the pages are valid for each fault in the
+ * storm, the time complexity grows rapidly.
+ *
+ * To address this, each page fault worker maintains a cache of the active fault
+ * being processed. Subsequent faults that hit in the cache are chained to the
+ * pending fault, and all chained faults are acknowledged once the initial fault
+ * completes. This alleviates head-of-queue blocking and quickly chains faults
+ * in the upper layers, avoiding expensive lookups in the main fault-handling
+ * path.
+ *
+ * Faults are buffered in the page fault queue in a way that provides stable
+ * storage for outstanding faults. In particular, faults may be chained directly
+ * while still resident in the queue storage (i.e., outside the worker’s current
+ * head/tail dequeue position). This allows the IRQ handler to match newly
+ * arrived faults against the per-worker cache and immediately chain cache hits
+ * onto the active fault under the queue lock, without allocating memory or
+ * waiting for the worker to pop the fault first.
+ *
+ * A per-fault state field is used to assert correctness of these invariants.
+ * The state tracks whether an entry is free, queued, chained, or currently
+ * active. Transitions are performed under the page fault queue lock, and the
+ * worker acknowledges faults by walking the chain and returning entries to the
+ * free state once they are complete.
+ */
+
+/**
+ * enum xe_pagefault_alloc_state - lifetime state for a page fault queue entry
+ * @XE_PAGEFAULT_ALLOC_STATE_FREE:
+ * Entry is unused and may be overwritten by the producer, consumer retry
+ * or requeue..
+ * @XE_PAGEFAULT_ALLOC_STATE_QUEUED:
+ * Entry has been enqueued and may be dequeued by a worker.
+ * @XE_PAGEFAULT_ALLOC_STATE_ACTIVE:
+ * Entry has been dequeued and is the worker's currently serviced fault.
+ * The worker may attach additional faults to it via consumer.next.
+ * @XE_PAGEFAULT_ALLOC_STATE_CHAINED:
+ * Entry is not independently serviced; it has been chained onto an
+ * ACTIVE entry via consumer.next and will be acknowledged when the
+ * leading fault completes.
+ * @XE_PAGEFAULT_ALLOC_STATE_COUNT:
+ * Count of allocation states.
+ *
+ * The page fault queue provides stable storage for outstanding faults so the
+ * IRQ handler can chain new cache hits directly onto a worker's active fault.
+ * Because entries may remain referenced outside the consumer dequeue window,
+ * the producer must only write into entries in the FREE state.
+ *
+ * State transitions are protected by the page fault queue lock. Workers return
+ * entries to FREE after acknowledging the fault (either as ACTIVE or CHAINED).
+ */
+enum xe_pagefault_alloc_state {
+ XE_PAGEFAULT_ALLOC_STATE_FREE = 0,
+ XE_PAGEFAULT_ALLOC_STATE_QUEUED = 1,
+ XE_PAGEFAULT_ALLOC_STATE_CHAINED = 2,
+ XE_PAGEFAULT_ALLOC_STATE_ACTIVE = 3,
+ XE_PAGEFAULT_ALLOC_STATE_COUNT = 4,
+};
+
static int xe_pagefault_entry_size(void)
{
/*
@@ -77,16 +145,16 @@ static int xe_pagefault_begin(struct drm_exec *exec, struct xe_vma *vma,
}
static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma,
- bool atomic)
+ struct xe_pagefault *pf, bool atomic)
{
struct xe_vm *vm = xe_vma_vm(vma);
struct xe_tile *tile = gt_to_tile(gt);
struct xe_validation_ctx ctx;
struct drm_exec exec;
struct dma_fence *fence;
- int err, needs_vram;
+ int err = 0, needs_vram;
- lockdep_assert_held_write(&vm->lock);
+ lockdep_assert_held(&vm->lock);
needs_vram = xe_vma_need_vram_for_atomic(vm->xe, vma, atomic);
if (needs_vram < 0 || (needs_vram && xe_vma_is_userptr(vma)))
@@ -98,50 +166,59 @@ static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma,
trace_xe_vma_pagefault(vma);
+ guard(mutex)(&vma->fault_lock);
+
/* Check if VMA is valid, opportunistic check only */
if (xe_vm_has_valid_gpu_mapping(tile, vma->tile_present,
- vma->tile_invalidated) && !atomic)
+ vma->tile_invalidated) && !atomic) {
+ xe_pagefault_set_start_addr(pf, xe_vma_start(vma));
+ xe_pagefault_set_end_addr(pf, xe_vma_end(vma));
return 0;
+ }
-retry_userptr:
- if (xe_vma_is_userptr(vma) &&
- xe_vma_userptr_check_repin(to_userptr_vma(vma))) {
- struct xe_userptr_vma *uvma = to_userptr_vma(vma);
+ do {
+ if (xe_vma_is_userptr(vma) &&
+ xe_vma_userptr_check_repin(to_userptr_vma(vma))) {
+ struct xe_userptr_vma *uvma = to_userptr_vma(vma);
- err = xe_vma_userptr_pin_pages(uvma);
- if (err)
- return err;
- }
+ err = xe_vma_userptr_pin_pages(uvma);
+ if (err)
+ return err;
+ }
- /* Lock VM and BOs dma-resv */
- xe_validation_ctx_init(&ctx, &vm->xe->val, &exec, (struct xe_val_flags) {});
- drm_exec_until_all_locked(&exec) {
- err = xe_pagefault_begin(&exec, vma, tile->mem.vram,
- needs_vram == 1);
- drm_exec_retry_on_contention(&exec);
- xe_validation_retry_on_oom(&ctx, &err);
- if (err)
- goto unlock_dma_resv;
-
- /* Bind VMA only to the GT that has faulted */
- trace_xe_vma_pf_bind(vma);
- xe_vm_set_validation_exec(vm, &exec);
- fence = xe_vma_rebind(vm, vma, BIT(tile->id));
- xe_vm_set_validation_exec(vm, NULL);
- if (IS_ERR(fence)) {
- err = PTR_ERR(fence);
+ /* Lock VM and BOs dma-resv */
+ xe_validation_ctx_init(&ctx, &vm->xe->val, &exec,
+ (struct xe_val_flags) {});
+ drm_exec_until_all_locked(&exec) {
+ err = xe_pagefault_begin(&exec, vma, tile->mem.vram,
+ needs_vram == 1);
+ drm_exec_retry_on_contention(&exec);
xe_validation_retry_on_oom(&ctx, &err);
- goto unlock_dma_resv;
+ if (err)
+ break;
+
+ /* Bind VMA only to the GT that has faulted */
+ trace_xe_vma_pf_bind(vma);
+ xe_vm_set_validation_exec(vm, &exec);
+ fence = xe_vma_rebind(vm, vma, BIT(tile->id));
+ xe_vm_set_validation_exec(vm, NULL);
+ if (IS_ERR(fence)) {
+ err = PTR_ERR(fence);
+ xe_validation_retry_on_oom(&ctx, &err);
+ break;
+ }
}
- }
+ xe_validation_ctx_fini(&ctx);
+ } while (err == -EAGAIN);
- dma_fence_wait(fence, false);
- dma_fence_put(fence);
+ if (!err) {
+ /* Give hint to immediately ack faults */
+ xe_pagefault_set_start_addr(pf, xe_vma_start(vma));
+ xe_pagefault_set_end_addr(pf, xe_vma_end(vma));
-unlock_dma_resv:
- xe_validation_ctx_fini(&ctx);
- if (err == -EAGAIN)
- goto retry_userptr;
+ dma_fence_wait(fence, false);
+ dma_fence_put(fence);
+ }
return err;
}
@@ -184,10 +261,7 @@ static int xe_pagefault_service(struct xe_pagefault *pf)
if (IS_ERR(vm))
return PTR_ERR(vm);
- /*
- * TODO: Change to read lock? Using write lock for simplicity.
- */
- down_write(&vm->lock);
+ down_read(&vm->lock);
if (xe_vm_is_closed(vm)) {
err = -ENOENT;
@@ -209,46 +283,275 @@ static int xe_pagefault_service(struct xe_pagefault *pf)
atomic = xe_pagefault_access_is_atomic(pf->consumer.access_type);
if (xe_vma_is_cpu_addr_mirror(vma))
- err = xe_svm_handle_pagefault(vm, vma, gt,
+ err = xe_svm_handle_pagefault(vm, vma, pf, gt,
pf->consumer.page_addr, atomic);
else
- err = xe_pagefault_handle_vma(gt, vma, atomic);
+ err = xe_pagefault_handle_vma(gt, vma, pf, atomic);
unlock_vm:
- if (!err)
- vm->usm.last_fault_vma = vma;
- up_write(&vm->lock);
+ up_read(&vm->lock);
xe_vm_put(vm);
return err;
}
-static bool xe_pagefault_queue_pop(struct xe_pagefault_queue *pf_queue,
- struct xe_pagefault *pf)
+#define XE_PAGEFAULT_CACHE_START_INVALID U64_MAX
+#define xe_pagefault_cache_start_invalidate(val) \
+ (val = XE_PAGEFAULT_CACHE_START_INVALID)
+
+static void
+xe_pagefault_cache_invalidate(struct xe_pagefault_queue *pf_queue,
+ struct xe_pagefault_work *pf_work)
+{
+ lockdep_assert_held(&pf_queue->lock);
+
+ xe_pagefault_cache_start_invalidate(pf_work->cache.start);
+}
+
+static bool xe_pagefault_queue_full(struct xe_pagefault_queue *pf_queue)
+{
+ lockdep_assert_held(&pf_queue->lock);
+
+ return CIRC_SPACE(pf_queue->head, pf_queue->tail,
+ pf_queue->size) <= xe_pagefault_entry_size();
+}
+
+static struct xe_pagefault *
+xe_pagefault_queue_add(struct xe_pagefault_queue *pf_queue,
+ struct xe_pagefault *pf)
{
- bool found_fault = false;
+ struct xe_device *xe = container_of(pf_queue, typeof(*xe),
+ usm.pf_queue);
+ struct xe_pagefault *lpf;
+
+ lockdep_assert_held(&pf_queue->lock);
- spin_lock_irq(&pf_queue->lock);
- if (pf_queue->tail != pf_queue->head) {
- memcpy(pf, pf_queue->data + pf_queue->tail, sizeof(*pf));
- pf_queue->tail = (pf_queue->tail + xe_pagefault_entry_size()) %
+ do {
+ /* Not possible, warn on and drop page fault */
+ if (WARN_ON_ONCE(xe_pagefault_queue_full(pf_queue))) {
+ xe_log_err(xe, PAGEFAULT, -ENOSPC, "Queue full!\n");
+ return NULL;
+ }
+
+ lpf = (pf_queue->data + pf_queue->head);
+ pf_queue->head = (pf_queue->head + xe_pagefault_entry_size()) %
pf_queue->size;
- found_fault = true;
+ } while (lpf->consumer.alloc_state != XE_PAGEFAULT_ALLOC_STATE_FREE);
+
+ xe_assert(xe, lpf != pf);
+ memcpy(lpf, pf, sizeof(*pf));
+ lpf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_QUEUED;
+
+ return lpf;
+}
+
+static struct xe_pagefault *
+xe_pagefault_queue_unchain_requeue(struct xe_pagefault_queue *pf_queue,
+ struct xe_pagefault *pf, struct xe_gt *gt)
+{
+ struct xe_device *xe = container_of(pf_queue, typeof(*xe),
+ usm.pf_queue);
+ struct xe_pagefault *next = pf->consumer.next, *lpf;
+
+ lockdep_assert_held(&pf_queue->lock);
+ xe_assert(xe, pf->consumer.alloc_state ==
+ XE_PAGEFAULT_ALLOC_STATE_CHAINED);
+
+ xe_gt_stats_incr(gt, XE_GT_STATS_ID_CHAIN_MISMATCH_PAGEFAULT_COUNT, 1);
+
+ pf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_FREE;
+ lpf = xe_pagefault_queue_add(pf_queue, pf);
+ if (lpf) {
+ lpf->consumer.next = NULL;
+ lpf->consumer.fault_type_level |= XE_PAGEFAULT_REQUEUE_MASK;
+ }
+
+ return next;
+}
+
+static bool xe_pagefault_match(struct xe_pagefault *pf, u64 start,
+ u64 end, u64 cache_asid)
+{
+ struct xe_device *xe = gt_to_xe(pf->gt);
+ u64 page_addr = pf->consumer.page_addr;
+ u32 pf_asid = pf->consumer.asid;
+
+ xe_assert(xe, pf->consumer.alloc_state !=
+ XE_PAGEFAULT_ALLOC_STATE_FREE);
+
+ return page_addr >= start && page_addr < end &&
+ pf_asid == cache_asid;
+}
+
+static bool xe_pagefault_try_chain(struct xe_pagefault_queue *pf_queue,
+ struct xe_pagefault *pf)
+{
+ struct xe_device *xe = container_of(pf_queue, typeof(*xe),
+ usm.pf_queue);
+ struct xe_pagefault_work *pf_work;
+ bool requeue = FIELD_GET(XE_PAGEFAULT_REQUEUE_MASK,
+ pf->consumer.fault_type_level);
+ int i;
+
+ lockdep_assert_held(&pf_queue->lock);
+ xe_assert(xe, pf->consumer.alloc_state ==
+ XE_PAGEFAULT_ALLOC_STATE_QUEUED);
+
+ /*
+ * If this is a retry, we may already have a chain attached. In that
+ * case, we cannot hit in the cache because chains cannot easily be
+ * combined.
+ */
+ if (pf->consumer.next)
+ return false;
+
+ for (i = 0, pf_work = xe->usm.pf_workers;
+ i < xe->info.num_pf_work; ++i, ++pf_work) {
+ u64 start = pf_work->cache.start;
+ u64 end = requeue ? start + SZ_4K : pf_work->cache.end;
+ u32 asid = pf_work->cache.asid;
+
+ if (xe_pagefault_match(pf, start, end, asid)) {
+ xe_assert(xe, pf_work->cache.pf->consumer.alloc_state ==
+ XE_PAGEFAULT_ALLOC_STATE_ACTIVE);
+
+ if (pf->producer.private !=
+ pf_work->cache.pf->producer.private)
+ continue;
+
+ xe_gt_stats_incr(pf->gt,
+ XE_GT_STATS_ID_CHAIN_PAGEFAULT_COUNT,
+ 1);
+
+ pf->consumer.alloc_state =
+ XE_PAGEFAULT_ALLOC_STATE_CHAINED;
+ pf->consumer.next = pf_work->cache.pf->consumer.next;
+ pf_work->cache.pf->consumer.next = pf;
+
+ return true;
+ }
+ }
+
+ return false;
+}
+
+static void xe_pagefault_queue_advance(struct xe_pagefault_queue *pf_queue)
+{
+ lockdep_assert_held(&pf_queue->lock);
+
+ pf_queue->tail = (pf_queue->tail + xe_pagefault_entry_size()) %
+ pf_queue->size;
+}
+
+static struct xe_pagefault *
+xe_pagefault_queue_tail_fault(struct xe_pagefault_queue *pf_queue)
+{
+ lockdep_assert_held(&pf_queue->lock);
+
+ return pf_queue->data + pf_queue->tail;
+}
+
+static bool xe_pagefault_queue_empty(struct xe_pagefault_queue *pf_queue)
+{
+ lockdep_assert_held(&pf_queue->lock);
+
+ return pf_queue->head == pf_queue->tail;
+}
+
+static bool xe_pagefault_queue_pop(struct xe_pagefault_queue *pf_queue,
+ struct xe_pagefault **pf, int id)
+{
+ struct xe_device *xe = container_of(pf_queue, typeof(*xe),
+ usm.pf_queue);
+ struct xe_pagefault_work *pf_work, *__pf_work;
+ struct xe_pagefault *lpf;
+ size_t align = SZ_2M;
+ int i;
+
+ guard(spinlock_irq)(&pf_queue->lock);
+
+ for (*pf = NULL; !*pf;) {
+ if (xe_pagefault_queue_empty(pf_queue))
+ return false;
+
+ lpf = xe_pagefault_queue_tail_fault(pf_queue);
+ xe_pagefault_queue_advance(pf_queue);
+
+ if (lpf->consumer.alloc_state !=
+ XE_PAGEFAULT_ALLOC_STATE_QUEUED)
+ continue;
+
+ if (xe_pagefault_try_chain(pf_queue, lpf))
+ continue;
+
+ *pf = lpf; /* Hand back page fault for processing */
}
- spin_unlock_irq(&pf_queue->lock);
- return found_fault;
+ /*
+ * No cache hit; allocate a new cache entry. We assume most faults
+ * within a 2M range will hit the same pages. If this assumption proves
+ * false, the mismatched fault is requeued after the initial fault is
+ * acknowledged.
+ */
+ pf_work = xe->usm.pf_workers + id;
+ if (FIELD_GET(XE_PAGEFAULT_REQUEUE_MASK,
+ lpf->consumer.fault_type_level))
+ align = SZ_4K;
+ pf_work->cache.start = ALIGN_DOWN(lpf->consumer.page_addr, align);
+ pf_work->cache.end = pf_work->cache.start + align;
+ pf_work->cache.asid = lpf->consumer.asid;
+ pf_work->cache.pf = lpf;
+ lpf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_ACTIVE;
+
+ for (i = 0, __pf_work = xe->usm.pf_workers;
+ i < xe->info.num_pf_work; ++i, ++__pf_work) {
+ u64 cache_start = __pf_work->cache.start;
+
+ if (__pf_work == pf_work)
+ continue;
+
+ if (cache_start != XE_PAGEFAULT_CACHE_START_INVALID) {
+ xe_gt_stats_incr(xe_root_mmio_gt(xe),
+ XE_GT_STATS_ID_PARALLEL_PAGEFAULT_COUNT,
+ 1);
+ break;
+ }
+ }
+
+ /* Drain queue until empty or new fault found */
+ while (1) {
+ if (xe_pagefault_queue_empty(pf_queue))
+ break;
+
+ lpf = xe_pagefault_queue_tail_fault(pf_queue);
+
+ if (lpf->consumer.alloc_state !=
+ XE_PAGEFAULT_ALLOC_STATE_QUEUED) {
+ xe_pagefault_queue_advance(pf_queue);
+ continue;
+ }
+
+ if (!xe_pagefault_try_chain(pf_queue, lpf))
+ break;
+
+ xe_pagefault_queue_advance(pf_queue);
+ }
+
+ return true;
}
static void xe_pagefault_print(struct xe_pagefault *pf)
{
+ u8 engine_class = FIELD_GET(XE_PAGEFAULT_ENGINE_CLASS_MASK,
+ pf->consumer.engine_class_instance);
+
xe_gt_info(pf->gt, "\n\tASID: %d\n"
"\tFaulted Address: 0x%08x%08x\n"
"\tFaultType: %lu\n"
"\tAccessType: %lu\n"
"\tFaultLevel: %lu\n"
"\tEngineClass: %d %s\n"
- "\tEngineInstance: %d\n",
+ "\tEngineInstance: %lu\n",
pf->consumer.asid,
upper_32_bits(pf->consumer.page_addr),
lower_32_bits(pf->consumer.page_addr),
@@ -258,9 +561,10 @@ static void xe_pagefault_print(struct xe_pagefault *pf)
pf->consumer.access_type),
FIELD_GET(XE_PAGEFAULT_LEVEL_MASK,
pf->consumer.fault_type_level),
- pf->consumer.engine_class,
- xe_hw_engine_class_to_str(pf->consumer.engine_class),
- pf->consumer.engine_instance);
+ engine_class,
+ xe_hw_engine_class_to_str(engine_class),
+ FIELD_GET(XE_PAGEFAULT_ENGINE_INSTANCE_MASK,
+ pf->consumer.engine_class_instance));
}
static void xe_pagefault_save_to_vm(struct xe_device *xe, struct xe_pagefault *pf)
@@ -290,42 +594,104 @@ static void xe_pagefault_save_to_vm(struct xe_device *xe, struct xe_pagefault *p
static void xe_pagefault_queue_work(struct work_struct *w)
{
- struct xe_pagefault_queue *pf_queue =
- container_of(w, typeof(*pf_queue), worker);
- struct xe_pagefault pf;
+ struct xe_pagefault_work *pf_work =
+ container_of(w, typeof(*pf_work), work);
+ struct xe_device *xe = pf_work->xe;
+ struct xe_pagefault_queue *pf_queue = &xe->usm.pf_queue;
+ struct xe_pagefault *pf;
+ ktime_t start = xe_gt_stats_ktime_get();
unsigned long threshold;
+ u64 cache_start = XE_PAGEFAULT_CACHE_START_INVALID, cache_end = 0;
+ u32 cache_asid = 0;
#define USM_QUEUE_MAX_RUNTIME_MS 20
threshold = jiffies + msecs_to_jiffies(USM_QUEUE_MAX_RUNTIME_MS);
- while (xe_pagefault_queue_pop(pf_queue, &pf)) {
- int err;
+ while (xe_pagefault_queue_pop(pf_queue, &pf, pf_work->id)) {
+ const struct xe_pagefault_ops *ops = pf->producer.ops;
+ void *private = pf->producer.private;
+ struct xe_gt *gt = pf->gt;
+ u32 asid = pf->consumer.asid;
+ int err = 0;
+ bool invalidated = false;
+
+ /* Last fault same address, ack immediately */
+ if (xe_pagefault_match(pf, cache_start, cache_end, cache_asid)) {
+ xe_gt_stats_incr(gt, XE_GT_STATS_ID_LAST_PAGEFAULT_COUNT, 1);
+ goto ack_fault;
+ }
- if (!pf.gt) /* Fault squashed during reset */
- continue;
+ err = xe_pagefault_service(pf);
- err = xe_pagefault_service(&pf);
if (err) {
- xe_pagefault_save_to_vm(gt_to_xe(pf.gt), &pf);
- if (!(pf.consumer.access_type & XE_PAGEFAULT_ACCESS_PREFETCH)) {
- xe_pagefault_print(&pf);
- xe_gt_info(pf.gt, "Fault response: Unsuccessful %pe\n",
- ERR_PTR(err));
+ if (!(pf->consumer.access_type & XE_PAGEFAULT_ACCESS_PREFETCH)) {
+ xe_pagefault_save_to_vm(gt_to_xe(gt), pf);
+ xe_pagefault_cache_start_invalidate(cache_start);
+ xe_pagefault_print(pf);
+ xe_log_err_info(pf->gt, PAGEFAULT, err, "Unsuccessful response\n");
} else {
- xe_gt_stats_incr(pf.gt, XE_GT_STATS_ID_INVALID_PREFETCH_PAGEFAULT_COUNT, 1);
- xe_gt_dbg(pf.gt, "Prefetch Fault response: Unsuccessful %pe\n",
+ xe_gt_stats_incr(pf->gt, XE_GT_STATS_ID_INVALID_PREFETCH_PAGEFAULT_COUNT, 1);
+ xe_gt_dbg(pf->gt, "Prefetch Fault response: Unsuccessful %pe\n",
ERR_PTR(err));
}
+ } else {
+ /* Cache valid fault locally */
+ cache_start = xe_pagefault_start_addr(pf);
+ cache_end = xe_pagefault_end_addr(pf);
+ cache_asid = asid;
}
- pf.producer.ops->ack_fault(&pf, err);
+ack_fault:
+ xe_assert(xe, pf->consumer.alloc_state ==
+ XE_PAGEFAULT_ALLOC_STATE_ACTIVE);
+ xe_assert(xe, pf == pf_work->cache.pf);
+
+ ops->ack_fault_begin(private);
+ while (pf) {
+ xe_assert(xe, pf->consumer.alloc_state ==
+ XE_PAGEFAULT_ALLOC_STATE_ACTIVE);
+ xe_assert(xe, ops == pf->producer.ops);
+ xe_assert(xe, gt == pf->gt);
+
+ ops->ack_fault(pf, err);
+
+ spin_lock_irq(&pf_queue->lock);
+
+ if (!invalidated) {
+ invalidated = true;
+ xe_pagefault_cache_invalidate(pf_queue,
+ pf_work);
+ }
+
+ pf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_FREE;
+ pf = pf->consumer.next;
+
+ /*
+ * Requeue chained faults which do not match the last
+ * fault processed
+ */
+ while (pf && !xe_pagefault_match(pf, cache_start,
+ cache_end, cache_asid))
+ pf = xe_pagefault_queue_unchain_requeue(pf_queue, pf, gt);
+
+ /* Ensure resets are safe */
+ if (pf)
+ pf->consumer.alloc_state =
+ XE_PAGEFAULT_ALLOC_STATE_ACTIVE;
+
+ spin_unlock_irq(&pf_queue->lock);
+ }
+ ops->ack_fault_end(private);
if (time_after(jiffies, threshold)) {
- queue_work(gt_to_xe(pf.gt)->usm.pf_wq, w);
+ queue_work(xe->usm.pagefault_wq, w);
break;
}
}
#undef USM_QUEUE_MAX_RUNTIME_MS
+
+ xe_gt_stats_incr(xe_root_mmio_gt(xe), XE_GT_STATS_ID_PAGEFAULT_US,
+ xe_gt_stats_ktime_us_delta(start));
}
static int xe_pagefault_queue_init(struct xe_device *xe,
@@ -366,7 +732,6 @@ static int xe_pagefault_queue_init(struct xe_device *xe,
xe_pagefault_entry_size(), total_num_eus, pf_queue->size);
spin_lock_init(&pf_queue->lock);
- INIT_WORK(&pf_queue->worker, xe_pagefault_queue_work);
pf_queue->data = drmm_kzalloc(&xe->drm, pf_queue->size, GFP_KERNEL);
if (!pf_queue->data)
@@ -379,7 +744,8 @@ static void xe_pagefault_fini(void *arg)
{
struct xe_device *xe = arg;
- destroy_workqueue(xe->usm.pf_wq);
+ destroy_workqueue(xe->usm.prefetch_wq);
+ destroy_workqueue(xe->usm.pagefault_wq);
}
/**
@@ -397,22 +763,39 @@ int xe_pagefault_init(struct xe_device *xe)
if (!xe->info.has_usm)
return 0;
- xe->usm.pf_wq = alloc_workqueue("xe_page_fault_work_queue",
- WQ_UNBOUND | WQ_HIGHPRI,
- XE_PAGEFAULT_QUEUE_COUNT);
- if (!xe->usm.pf_wq)
+ xe->usm.pagefault_wq = alloc_workqueue("xe_page_fault_work_queue",
+ WQ_UNBOUND | WQ_HIGHPRI,
+ xe->info.num_pf_work);
+ if (!xe->usm.pagefault_wq)
return -ENOMEM;
- for (i = 0; i < XE_PAGEFAULT_QUEUE_COUNT; ++i) {
- err = xe_pagefault_queue_init(xe, xe->usm.pf_queue + i);
- if (err)
- goto err_out;
+ xe->usm.prefetch_wq = alloc_workqueue("xe_prefetch_work_queue",
+ WQ_UNBOUND,
+ xe->info.num_pf_work);
+ if (!xe->usm.prefetch_wq) {
+ err = -ENOMEM;
+ goto err_pagefault_wq;
+ }
+
+ err = xe_pagefault_queue_init(xe, &xe->usm.pf_queue);
+ if (err)
+ goto err_out;
+
+ for (i = 0; i < xe->info.num_pf_work; ++i) {
+ struct xe_pagefault_work *pf_work = xe->usm.pf_workers + i;
+
+ pf_work->xe = xe;
+ pf_work->id = i;
+ xe_pagefault_cache_start_invalidate(pf_work->cache.start);
+ INIT_WORK(&pf_work->work, xe_pagefault_queue_work);
}
return devm_add_action_or_reset(xe->drm.dev, xe_pagefault_fini, xe);
err_out:
- destroy_workqueue(xe->usm.pf_wq);
+ destroy_workqueue(xe->usm.prefetch_wq);
+err_pagefault_wq:
+ destroy_workqueue(xe->usm.pagefault_wq);
return err;
}
@@ -427,15 +810,23 @@ static void xe_pagefault_queue_reset(struct xe_device *xe, struct xe_gt *gt,
/* Squash all pending faults on the GT */
- spin_lock_irq(&pf_queue->lock);
- for (i = pf_queue->tail; i != pf_queue->head;
- i = (i + xe_pagefault_entry_size()) % pf_queue->size) {
+ guard(spinlock_irq)(&pf_queue->lock);
+
+ for (i = 0; i < pf_queue->size; i += xe_pagefault_entry_size()) {
struct xe_pagefault *pf = pf_queue->data + i;
+ bool active = pf->consumer.alloc_state ==
+ XE_PAGEFAULT_ALLOC_STATE_ACTIVE;
- if (pf->gt == gt)
- pf->gt = NULL;
+ if (pf->gt != gt || active) {
+ if (active)
+ pf->consumer.next = NULL;
+ continue;
+ }
+
+ pf->consumer.alloc_state =
+ XE_PAGEFAULT_ALLOC_STATE_FREE;
+ pf->consumer.next = NULL;
}
- spin_unlock_irq(&pf_queue->lock);
}
/**
@@ -448,18 +839,18 @@ static void xe_pagefault_queue_reset(struct xe_device *xe, struct xe_gt *gt,
*/
void xe_pagefault_reset(struct xe_device *xe, struct xe_gt *gt)
{
- int i;
-
- for (i = 0; i < XE_PAGEFAULT_QUEUE_COUNT; ++i)
- xe_pagefault_queue_reset(xe, gt, xe->usm.pf_queue + i);
+ xe_pagefault_queue_reset(xe, gt, &xe->usm.pf_queue);
}
-static bool xe_pagefault_queue_full(struct xe_pagefault_queue *pf_queue)
+/*
+ * This function can race with multiple page fault producers, but worst case we
+ * stick a page fault on the same queue for consumption.
+ */
+static int xe_pagefault_work_index(struct xe_device *xe)
{
- lockdep_assert_held(&pf_queue->lock);
+ lockdep_assert_held(&xe->usm.pf_queue.lock);
- return CIRC_SPACE(pf_queue->head, pf_queue->tail, pf_queue->size) <=
- xe_pagefault_entry_size();
+ return xe->usm.current_pf_work++ % xe->info.num_pf_work;
}
/**
@@ -474,24 +865,95 @@ static bool xe_pagefault_queue_full(struct xe_pagefault_queue *pf_queue)
*/
int xe_pagefault_handler(struct xe_device *xe, struct xe_pagefault *pf)
{
- struct xe_pagefault_queue *pf_queue = xe->usm.pf_queue +
- (pf->consumer.asid % XE_PAGEFAULT_QUEUE_COUNT);
- unsigned long flags;
- bool full;
-
- spin_lock_irqsave(&pf_queue->lock, flags);
- full = xe_pagefault_queue_full(pf_queue);
- if (!full) {
- memcpy(pf_queue->data + pf_queue->head, pf, sizeof(*pf));
- pf_queue->head = (pf_queue->head + xe_pagefault_entry_size()) %
- pf_queue->size;
- queue_work(xe->usm.pf_wq, &pf_queue->worker);
+ struct xe_pagefault_queue *pf_queue = &xe->usm.pf_queue;
+ struct xe_pagefault *lpf;
+ bool empty;
+
+ guard(spinlock_irqsave)(&pf_queue->lock);
+
+ empty = xe_pagefault_queue_empty(pf_queue);
+ lpf = xe_pagefault_queue_add(pf_queue, pf);
+ if (!lpf)
+ return -ENOSPC;
+
+ lpf->consumer.next = NULL;
+ if (xe_pagefault_try_chain(pf_queue, lpf)) {
+ xe_gt_stats_incr(pf->gt, XE_GT_STATS_ID_CHAIN_IRQ_PAGEFAULT_COUNT, 1);
+ if (empty) {
+ xe_gt_stats_incr(pf->gt,
+ XE_GT_STATS_ID_CHAIN_DRAIN_IRQ_PAGEFAULT_COUNT, 1);
+ xe_pagefault_queue_advance(pf_queue);
+ }
} else {
- drm_warn(&xe->drm,
- "PageFault Queue (%d) full, shouldn't be possible\n",
- pf->consumer.asid % XE_PAGEFAULT_QUEUE_COUNT);
+ int work_index = xe_pagefault_work_index(xe);
+
+ queue_work(xe->usm.pagefault_wq,
+ &xe->usm.pf_workers[work_index].work);
+ }
+
+ return 0;
+}
+
+/**
+ * xe_pagefault_print_info() - dump page fault queue/cache debug information
+ * @xe: Xe device
+ * @p: DRM printer to emit output to
+ *
+ * Print a snapshot of the page fault queue state for debugging. The output
+ * includes queue parameters (entry size, total size, head/tail), a histogram
+ * of per-entry allocation state values, and the validity of each per-worker
+ * page fault cache.
+ *
+ * This function is intended for debugfs and similar diagnostics. It acquires
+ * the page fault queue spinlock internally to serialize against IRQ-side
+ * producers and the worker consumer path, so callers must not hold the queue
+ * lock.
+ */
+void xe_pagefault_print_info(struct xe_device *xe, struct drm_printer *p)
+{
+ struct xe_pagefault_queue *pf_queue = &xe->usm.pf_queue;
+ struct xe_pagefault_work *pf_work;
+ static const char * const alloc_state_names[] = {
+ [XE_PAGEFAULT_ALLOC_STATE_FREE] = "free",
+ [XE_PAGEFAULT_ALLOC_STATE_QUEUED] = "queued",
+ [XE_PAGEFAULT_ALLOC_STATE_CHAINED] = "chained",
+ [XE_PAGEFAULT_ALLOC_STATE_ACTIVE] = "active",
+ };
+ u32 i, counts[XE_PAGEFAULT_ALLOC_STATE_COUNT] = {};
+
+ /* Driver load failure guard / USM not enabled guard */
+ if (!pf_queue->data)
+ return;
+
+ guard(spinlock_irq)(&pf_queue->lock);
+
+ drm_printf(p, "pagefault size: %u\n", xe_pagefault_entry_size());
+ drm_printf(p, "pagefault queue size: %u\n", pf_queue->size);
+ drm_printf(p, "pagefault queue head: %u\n", pf_queue->head);
+ drm_printf(p, "pagefault queue tail: %u\n", pf_queue->tail);
+
+ for (i = 0; i < pf_queue->size; i += xe_pagefault_entry_size()) {
+ struct xe_pagefault *pf = pf_queue->data + i;
+
+ if (pf->consumer.alloc_state >=
+ XE_PAGEFAULT_ALLOC_STATE_COUNT) {
+ drm_printf(p, "pagefault[%u] corrupted alloc_state=%u\n",
+ i, pf->consumer.alloc_state);
+ continue;
+ }
+
+ counts[pf->consumer.alloc_state]++;
}
- spin_unlock_irqrestore(&pf_queue->lock, flags);
- return full ? -ENOSPC : 0;
+ for (i = 0; i < XE_PAGEFAULT_ALLOC_STATE_COUNT; ++i)
+ drm_printf(p, "pagefault queue %s count: %u\n",
+ alloc_state_names[i], counts[i]);
+
+ for (i = 0, pf_work = xe->usm.pf_workers;
+ i < xe->info.num_pf_work; ++i, ++pf_work) {
+ if (pf_work->cache.start == XE_PAGEFAULT_CACHE_START_INVALID)
+ drm_printf(p, "pagefault work[%u] cache invalid\n", i);
+ else
+ drm_printf(p, "pagefault work[%u] cache valid\n", i);
+ }
}
diff --git a/drivers/gpu/drm/xe/xe_pagefault.h b/drivers/gpu/drm/xe/xe_pagefault.h
index bd0cdf9ed37f..e9c5d1f03760 100644
--- a/drivers/gpu/drm/xe/xe_pagefault.h
+++ b/drivers/gpu/drm/xe/xe_pagefault.h
@@ -6,6 +6,9 @@
#ifndef _XE_PAGEFAULT_H_
#define _XE_PAGEFAULT_H_
+#include "xe_pagefault_types.h"
+
+struct drm_printer;
struct xe_device;
struct xe_gt;
struct xe_pagefault;
@@ -16,4 +19,75 @@ void xe_pagefault_reset(struct xe_device *xe, struct xe_gt *gt);
int xe_pagefault_handler(struct xe_device *xe, struct xe_pagefault *pf);
+void xe_pagefault_print_info(struct xe_device *xe, struct drm_printer *p);
+
+#define XE_PAGEFAULT_END_ADDR_MASK (~0xfffull)
+
+/**
+ * xe_pagefault_set_end_addr() - store serviced range end for a pagefault
+ * @pf: Pagefault entry
+ * @end_addr: Inclusive end address of the serviced fault range
+ *
+ * The pagefault consumer stores the resolved fault range so subsequent faults
+ * hitting the same range can be immediately acknowledged without re-running
+ * the full fault handling path.
+ *
+ * The end address shares storage with other consumer metadata and therefore
+ * must be masked with %XE_PAGEFAULT_END_ADDR_MASK before storing. Bits outside
+ * the mask are reserved for internal state tracking and must be preserved.
+ */
+static inline void
+xe_pagefault_set_end_addr(struct xe_pagefault *pf, u64 end_addr)
+{
+ pf->consumer.end_addr &= ~XE_PAGEFAULT_END_ADDR_MASK;
+ pf->consumer.end_addr |= end_addr;
+}
+
+/**
+ * xe_pagefault_end_addr() - read serviced range end for a pagefault
+ * @pf: Pagefault entry
+ *
+ * Returns the inclusive end address of the range previously recorded by
+ * xe_pagefault_set_end_addr(). Only the bits covered by
+ * %XE_PAGEFAULT_END_ADDR_MASK are returned; other bits in the storage are
+ * reserved for internal state.
+ *
+ * Return: End address of the serviced fault range.
+ */
+static inline u64 xe_pagefault_end_addr(struct xe_pagefault *pf)
+{
+ return pf->consumer.end_addr & XE_PAGEFAULT_END_ADDR_MASK;
+}
+
+#undef XE_PAGEFAULT_END_ADDR_MASK
+
+/**
+ * xe_pagefault_set_start_addr() - store serviced range start for a pagefault
+ * @pf: Pagefault entry
+ * @start_addr: Start address of the serviced fault range
+ *
+ * The pagefault consumer stores the resolved fault range so subsequent faults
+ * hitting the same range can be immediately acknowledged without re-running
+ * the full fault handling path.
+ */
+static inline void
+xe_pagefault_set_start_addr(struct xe_pagefault *pf, u64 start_addr)
+{
+ pf->consumer.page_addr = start_addr;
+}
+
+/**
+ * xe_pagefault_start_addr() - read serviced range start for a pagefault
+ * @pf: Pagefault entry
+ *
+ * Returns the inclusive start address of the range previously recorded by
+ * xe_pagefault_set_start_addr().
+ *
+ * Return: Start address of the serviced fault range.
+ */
+static inline u64 xe_pagefault_start_addr(struct xe_pagefault *pf)
+{
+ return pf->consumer.page_addr;
+}
+
#endif
diff --git a/drivers/gpu/drm/xe/xe_pagefault_types.h b/drivers/gpu/drm/xe/xe_pagefault_types.h
index c4ee625b93dd..efeba5c3a58b 100644
--- a/drivers/gpu/drm/xe/xe_pagefault_types.h
+++ b/drivers/gpu/drm/xe/xe_pagefault_types.h
@@ -34,6 +34,13 @@ enum xe_pagefault_type {
/** struct xe_pagefault_ops - Xe pagefault ops (producer) */
struct xe_pagefault_ops {
/**
+ * @ack_fault_begin: Ack fault begin
+ * @private: producer private data
+ *
+ * Page fault producer begins acknowledgment from the consumer.
+ */
+ void (*ack_fault_begin)(void *private);
+ /**
* @ack_fault: Ack fault
* @pf: Page fault
* @err: Error state of fault
@@ -42,6 +49,13 @@ struct xe_pagefault_ops {
* sends the result to the HW/FW interface.
*/
void (*ack_fault)(struct xe_pagefault *pf, int err);
+ /**
+ * @ack_fault_end: Ack fault end
+ * @private: producer private data
+ *
+ * Page fault producer ends acknowledgment from the consumer.
+ */
+ void (*ack_fault_end)(void *private);
};
/**
@@ -60,34 +74,58 @@ struct xe_pagefault {
/**
* @consumer: State for the software handling the fault. Populated by
* the producer and may be modified by the consumer to communicate
- * information back to the producer upon fault acknowledgment.
+ * information back to the producer upon fault acknowledgment. After
+ * fault acknowledgment, the producer should only access consumer fields
+ * via well defined helpers.
*/
struct {
- /** @consumer.page_addr: address of page fault */
- u64 page_addr;
- /** @consumer.asid: address space ID */
- u32 asid;
/**
- * @consumer.access_type: access type and prefetch flag packed
- * into a u8.
+ * @consumer.page_addr: address of page fault, populated by
+ * consumer after fault completion
*/
- u8 access_type;
+ u64 page_addr;
+ union {
+ struct {
+ /**
+ * @consumer.alloc_state: page fault allocation
+ * state
+ */
+ u8 alloc_state;
+ /**
+ * @consumer.access_type: access type, u8 rather
+ * than enum to keep size compact
+ */
+ u8 access_type;
#define XE_PAGEFAULT_ACCESS_TYPE_MASK GENMASK(1, 0)
#define XE_PAGEFAULT_ACCESS_PREFETCH BIT(7)
+ /**
+ * @consumer.fault_type_level: fault type and
+ * level, u8 rather than enum to keep size
+ * compact
+ */
+ u8 fault_type_level;
+#define XE_PAGEFAULT_TYPE_LEVEL_NACK 0xff /* Producer indicates nack fault */
+#define XE_PAGEFAULT_LEVEL_MASK GENMASK(2, 0)
+#define XE_PAGEFAULT_TYPE_MASK GENMASK(6, 3)
+#define XE_PAGEFAULT_REQUEUE_MASK BIT(7)
+ /** @consumer.engine_class_instance: engine class and instance */
+ u8 engine_class_instance;
+#define XE_PAGEFAULT_ENGINE_CLASS_MASK GENMASK(3, 0)
+#define XE_PAGEFAULT_ENGINE_INSTANCE_MASK GENMASK(7, 4)
+ /** @consumer.asid: address space ID */
+ u32 asid;
+ };
+ /**
+ * @consumer.end_addr: end address of page fault,
+ * populated by consumer after fault completion
+ */
+ u64 end_addr;
+ };
/**
- * @consumer.fault_type_level: fault type and level, u8 rather
- * than enum to keep size compact
+ * @consumer.next: next pagefault chained to this fault,
+ * protected by pf_queue lock
*/
- u8 fault_type_level;
-#define XE_PAGEFAULT_TYPE_LEVEL_NACK 0xff /* Producer indicates nack fault */
-#define XE_PAGEFAULT_LEVEL_MASK GENMASK(3, 0)
-#define XE_PAGEFAULT_TYPE_MASK GENMASK(7, 4)
- /** @consumer.engine_class: engine class */
- u8 engine_class;
- /** @consumer.engine_instance: engine instance */
- u8 engine_instance;
- /** @consumer.reserved: reserved bits for future expansion */
- u64 reserved;
+ struct xe_pagefault *next;
} consumer;
/**
* @producer: State for the producer (i.e., HW/FW interface). Populated
@@ -129,10 +167,38 @@ struct xe_pagefault_queue {
u32 head;
/** @tail: Tail pointer in bytes, moved by consumer, protected by @lock */
u32 tail;
- /** @lock: protects page fault queue */
+ /** @lock: protects page fault queue, workers caches */
spinlock_t lock;
- /** @worker: to process page faults */
- struct work_struct worker;
+};
+
+/**
+ * struct xe_pagefault_work - Xe page fault work item (consumer)
+ *
+ * Represents a worker that pops a &struct xe_pagefault from the page fault
+ * queue and processes it.
+ */
+struct xe_pagefault_work {
+ /** @xe: Back-pointer to the Xe device */
+ struct xe_device *xe;
+ /** @id: Identifier for this work item */
+ int id;
+ /**
+ * @cache: Page fault cache for the currently processed fault
+ *
+ * Protected by the page fault queue lock.
+ */
+ struct {
+ /** @cache.start: Start address of the current page fault */
+ u64 start;
+ /** @cache.end: End address of the current page fault */
+ u64 end;
+ /** @cache.asid: Address space ID of the current page fault */
+ u32 asid;
+ /** @cache.pf: Pointer to the current page fault */
+ struct xe_pagefault *pf;
+ } cache;
+ /** @work: Work item used to process the page fault */
+ struct work_struct work;
};
#endif
diff --git a/drivers/gpu/drm/xe/xe_pci.c b/drivers/gpu/drm/xe/xe_pci.c
index 5f2a0b19839d..1e04e8ef2611 100644
--- a/drivers/gpu/drm/xe/xe_pci.c
+++ b/drivers/gpu/drm/xe/xe_pci.c
@@ -25,6 +25,7 @@
#include "xe_gt_printk.h"
#include "xe_gt_sriov_vf.h"
#include "xe_guc.h"
+#include "xe_log.h"
#include "xe_mmio.h"
#include "xe_module.h"
#include "xe_pci_error.h"
@@ -1145,17 +1146,12 @@ static void xe_pci_remove(struct pci_dev *pdev)
* caller. Therefore there is no consequence on those specific callers when
* function error injection skips the whole function.
*/
+static int __xe_pci_probe(struct pci_dev *pdev, const struct xe_device_desc *desc);
static int xe_pci_probe(struct pci_dev *pdev, const struct pci_device_id *ent)
{
- struct xe_probed_info probed_info = {};
const struct xe_device_desc *desc = (const void *)ent->driver_data;
- const struct xe_subplatform_desc *subplatform_desc;
- struct xe_device *xe;
- void *group;
int err;
- subplatform_desc = find_subplatform(desc, pdev->device);
-
xe_configfs_check_device(pdev);
if (desc->require_force_probe && !id_forced(pdev->device)) {
@@ -1171,14 +1167,34 @@ static int xe_pci_probe(struct pci_dev *pdev, const struct pci_device_id *ent)
}
if (id_blocked(pdev->device)) {
- dev_info(&pdev->dev, "Probe blocked for device [%04x:%04x].\n",
- pdev->vendor, pdev->device);
+ xe_log_info(pdev, PROBE, "driver loading blocked for device '%04x'\n",
+ pdev->device);
return -ENODEV;
}
if (xe_display_driver_probe_defer(pdev))
return -EPROBE_DEFER;
+ err = __xe_pci_probe(pdev, desc);
+ if (err) {
+ xe_log_err_fatal(pdev, PROBE, err, "driver loading failed for device '%04x'\n",
+ pdev->device);
+ return err;
+ }
+
+ return 0;
+}
+
+static int __xe_pci_probe(struct pci_dev *pdev, const struct xe_device_desc *desc)
+{
+ const struct xe_subplatform_desc *subplatform_desc;
+ struct xe_probed_info probed_info = {};
+ struct xe_device *xe;
+ void *group;
+ int err;
+
+ subplatform_desc = find_subplatform(desc, pdev->device);
+
/* Group all devres so xe_pci_error_slot_reset() can release them as a unit. */
group = devres_open_group(&pdev->dev, NULL, GFP_KERNEL);
if (!group)
diff --git a/drivers/gpu/drm/xe/xe_pci_error.c b/drivers/gpu/drm/xe/xe_pci_error.c
index e41af2ac7f23..79ce0c671549 100644
--- a/drivers/gpu/drm/xe/xe_pci_error.c
+++ b/drivers/gpu/drm/xe/xe_pci_error.c
@@ -7,6 +7,7 @@
#include "xe_device.h"
#include "xe_gt.h"
+#include "xe_log.h"
#include "xe_pci.h"
#include "xe_pm.h"
#include "xe_printk.h"
@@ -83,6 +84,12 @@ static pci_ers_result_t xe_pci_error_mmio_enabled(struct pci_dev *pdev)
xe_info(xe, "PCI error: MMIO enabled\n");
action = xe_ras_process_errors(xe);
+ /* User wants to debug the error, prevent reset */
+ if (xe->wedged.mode == XE_WEDGED_MODE_UPON_ANY_HANG_NO_RESET) {
+ xe_device_declare_wedged(xe);
+ return PCI_ERS_RESULT_DISCONNECT;
+ }
+
return ras_action_to_pci_result(pdev, action);
}
@@ -90,13 +97,15 @@ static pci_ers_result_t xe_pci_error_slot_reset(struct pci_dev *pdev)
{
const struct pci_device_id *ent = pci_match_id(pdev->driver->id_table, pdev);
struct xe_device *xe = pdev_to_xe_device(pdev);
+ int err;
xe_info(xe, "PCI error: slot reset\n");
pci_restore_state(pdev);
- if (pci_enable_device(pdev)) {
- xe_err(xe, "Cannot re-enable PCI device after reset\n");
+ err = pci_enable_device(pdev);
+ if (err) {
+ xe_log_err_fatal(xe, PCI, err, "Cannot re-enable PCI device after reset\n");
return PCI_ERS_RESULT_DISCONNECT;
}
diff --git a/drivers/gpu/drm/xe/xe_pcode.c b/drivers/gpu/drm/xe/xe_pcode.c
index ccc3bdeed6bb..e1b8062541a9 100644
--- a/drivers/gpu/drm/xe/xe_pcode.c
+++ b/drivers/gpu/drm/xe/xe_pcode.c
@@ -14,6 +14,7 @@
#include "regs/xe_pmt.h"
#include "xe_assert.h"
#include "xe_device.h"
+#include "xe_log.h"
#include "xe_mmio.h"
#include "xe_pcode_api.h"
#include "xe_pm.h"
@@ -61,9 +62,7 @@ static int pcode_mailbox_status(struct xe_tile *tile)
}
if (err) {
- drm_err(&tile_to_xe(tile)->drm, "PCODE Mailbox failed: %d %s",
- err_decode, err_str);
-
+ xe_log_err(tile, PCODE, err_decode, "Mailbox failed: %s\n", err_str);
return err_decode;
}
@@ -219,8 +218,7 @@ int xe_pcode_request(struct xe_tile *tile, u32 mbox, u32 request,
* requests, and for any quirks of the PCODE firmware that delays
* the request completion.
*/
- drm_err(&tile_to_xe(tile)->drm,
- "PCODE timeout, retrying with preemption disabled\n");
+ xe_log_err(tile, PCODE, ret, "timeout, retrying with preemption disabled\n");
preempt_disable();
ret = pcode_try_request(tile, mbox, request, reply_mask, reply, &status,
true, 50 * 1000, true);
@@ -299,7 +297,7 @@ int xe_pcode_ready(struct xe_device *xe, bool locked)
{
u32 status, request = DGFX_GET_INIT_STATUS;
struct xe_tile *tile = xe_device_get_root_tile(xe);
- int timeout_us = 180000000; /* 3 min */
+ long timeout_us = 3 * 60 * USEC_PER_SEC; /* 3 min */
int ret;
if (xe->info.skip_pcode)
@@ -320,8 +318,8 @@ int xe_pcode_ready(struct xe_device *xe, bool locked)
mutex_unlock(&tile->pcode.lock);
if (ret)
- drm_err(&xe->drm,
- "PCODE initialization timedout after: 3 min\n");
+ xe_log_err(tile, PCODE, ret, "initialization timedout after %ld seconds\n",
+ timeout_us / USEC_PER_SEC);
return ret;
}
diff --git a/drivers/gpu/drm/xe/xe_pm.c b/drivers/gpu/drm/xe/xe_pm.c
index a5289a9df8d2..f517bf453b54 100644
--- a/drivers/gpu/drm/xe/xe_pm.c
+++ b/drivers/gpu/drm/xe/xe_pm.c
@@ -905,6 +905,11 @@ static bool xe_pm_suspending_or_resuming(struct xe_device *xe)
* break scope-based handling, or when the lifetime of the runtime PM reference
* does not match a specific scope (e.g., runtime PM obtained in one function
* and released in a different one).
+ *
+ * This helper assumes the caller already holds a runtime PM reference and
+ * only warns when it cannot see one. After hot-unplug runtime PM is disabled
+ * and the check fails even when a reference is held, so callers that may run
+ * after unplug must guard it with drm_dev_enter()/drm_dev_exit() instead.
*/
void xe_pm_runtime_get_noresume(struct xe_device *xe)
{
diff --git a/drivers/gpu/drm/xe/xe_printk.h b/drivers/gpu/drm/xe/xe_printk.h
index c5be2385aa95..87f51d7edfa8 100644
--- a/drivers/gpu/drm/xe/xe_printk.h
+++ b/drivers/gpu/drm/xe/xe_printk.h
@@ -36,6 +36,9 @@
#define xe_dbg(_xe, _fmt, ...) \
xe_printk((_xe), dbg, _fmt, ##__VA_ARGS__)
+#define xe_dbg_ratelimited(_xe, _fmt, ...) \
+ xe_printk((_xe), dbg_ratelimited, _fmt, ##__VA_ARGS__)
+
#define xe_WARN_type(_xe, _type, _condition, _fmt, ...) \
drm_WARN##_type(&(_xe)->drm, _condition, _fmt, ## __VA_ARGS__)
diff --git a/drivers/gpu/drm/xe/xe_pt.c b/drivers/gpu/drm/xe/xe_pt.c
index a07316a45d79..5d990c1c3740 100644
--- a/drivers/gpu/drm/xe/xe_pt.c
+++ b/drivers/gpu/drm/xe/xe_pt.c
@@ -2414,6 +2414,12 @@ static int op_prepare(struct xe_vm *vm,
xa_for_each(&op->prefetch_range.range, i, range) {
err = bind_range_prepare(vm, tile, pt_update_ops,
vma, range);
+ /*
+ * Don't tell user space to retry, rather let
+ * page faults fixup the pages.
+ */
+ if (err == -EAGAIN)
+ err = -ENODATA;
if (err)
return err;
}
diff --git a/drivers/gpu/drm/xe/xe_pxp_submit.c b/drivers/gpu/drm/xe/xe_pxp_submit.c
index e60526e30030..5de86a8cf27d 100644
--- a/drivers/gpu/drm/xe/xe_pxp_submit.c
+++ b/drivers/gpu/drm/xe/xe_pxp_submit.c
@@ -46,7 +46,7 @@ static int allocate_vcs_execution_resources(struct xe_pxp *pxp)
return -ENODEV;
q = xe_exec_queue_create(xe, NULL, BIT(hwe->logical_instance), 1, hwe,
- EXEC_QUEUE_FLAG_KERNEL | EXEC_QUEUE_FLAG_PERMANENT, 0);
+ EXEC_QUEUE_FLAG_KERNEL, 0);
if (IS_ERR(q))
return PTR_ERR(q);
@@ -144,8 +144,7 @@ static int allocate_gsc_client_resources(struct xe_gt *gt,
}
q = xe_exec_queue_create(xe, vm, BIT(hwe->logical_instance), 1, hwe,
- EXEC_QUEUE_FLAG_KERNEL |
- EXEC_QUEUE_FLAG_PERMANENT, 0);
+ EXEC_QUEUE_FLAG_KERNEL, 0);
if (IS_ERR(q)) {
err = PTR_ERR(q);
goto bo_out;
diff --git a/drivers/gpu/drm/xe/xe_ras.c b/drivers/gpu/drm/xe/xe_ras.c
index d98ff9453f60..d25d25f77531 100644
--- a/drivers/gpu/drm/xe/xe_ras.c
+++ b/drivers/gpu/drm/xe/xe_ras.c
@@ -3,8 +3,10 @@
* Copyright © 2026 Intel Corporation
*/
+#include "xe_debugfs.h"
#include "xe_device.h"
#include "xe_drm_ras.h"
+#include "xe_log.h"
#include "xe_pm.h"
#include "xe_printk.h"
#include "xe_ras.h"
@@ -45,6 +47,16 @@ enum xe_ras_component {
XE_RAS_COMP_MAX
};
+#define CHECK_COMPONENT(RAS_COMP, LOG_COMP) \
+ static_assert(MAKE_XE_LOG_COMPONENT(HARDWARE, (RAS_COMP)) == (LOG_COMP))
+ /* make sure components definitions maintain stable relation */
+ CHECK_COMPONENT(XE_RAS_COMP_DEVICE_MEMORY, XE_LOG_COMPONENT_DEVICE_MEMORY);
+ CHECK_COMPONENT(XE_RAS_COMP_CORE_COMPUTE, XE_LOG_COMPONENT_CORE_COMPUTE);
+ CHECK_COMPONENT(XE_RAS_COMP_PCIE, XE_LOG_COMPONENT_PCIE);
+ CHECK_COMPONENT(XE_RAS_COMP_FABRIC, XE_LOG_COMPONENT_FABRIC);
+ CHECK_COMPONENT(XE_RAS_COMP_SOC_INTERNAL, XE_LOG_COMPONENT_SOC_INTERNAL);
+#undef CHECK_COMPONENT
+
/* RAS response status codes */
enum xe_ras_response_status {
XE_RAS_STATUS_SUCCESS = 0,
@@ -90,6 +102,8 @@ static const char * const gpu_health_states[] = {
};
static_assert(ARRAY_SIZE(gpu_health_states) == XE_RAS_HEALTH_MAX);
+static int get_counter(struct xe_device *xe, struct xe_ras_error_class *counter, u32 *value);
+
static u8 drm_to_xe_ras_severity(u8 severity)
{
switch (severity) {
@@ -102,6 +116,18 @@ static u8 drm_to_xe_ras_severity(u8 severity)
}
}
+static u8 xe_to_drm_ras_severity(u8 severity)
+{
+ switch (severity) {
+ case XE_RAS_SEV_CORRECTABLE:
+ return DRM_XE_RAS_ERR_SEV_CORRECTABLE;
+ case XE_RAS_SEV_UNCORRECTABLE:
+ return DRM_XE_RAS_ERR_SEV_UNCORRECTABLE;
+ default:
+ return DRM_XE_RAS_ERR_SEV_MAX;
+ }
+}
+
static u8 drm_to_xe_ras_component(u8 component)
{
switch (component) {
@@ -120,6 +146,24 @@ static u8 drm_to_xe_ras_component(u8 component)
}
}
+static u8 xe_to_drm_ras_component(u8 component)
+{
+ switch (component) {
+ case XE_RAS_COMP_DEVICE_MEMORY:
+ return DRM_XE_RAS_ERR_COMP_DEVICE_MEMORY;
+ case XE_RAS_COMP_CORE_COMPUTE:
+ return DRM_XE_RAS_ERR_COMP_CORE_COMPUTE;
+ case XE_RAS_COMP_PCIE:
+ return DRM_XE_RAS_ERR_COMP_PCIE;
+ case XE_RAS_COMP_FABRIC:
+ return DRM_XE_RAS_ERR_COMP_FABRIC;
+ case XE_RAS_COMP_SOC_INTERNAL:
+ return DRM_XE_RAS_ERR_COMP_SOC_INTERNAL;
+ default:
+ return DRM_XE_RAS_ERR_COMP_MAX;
+ }
+}
+
static int ras_status_to_errno(u32 status)
{
switch (status) {
@@ -156,6 +200,24 @@ static inline const char *comp_to_str(u8 component)
return xe_ras_components[component];
}
+static bool ras_counter_is_valid(struct xe_device *xe, struct xe_ras_error_class *counter)
+{
+ u8 severity = counter->common.severity;
+ u8 component = counter->common.component;
+
+ if (!in_range(severity, XE_RAS_SEV_NOT_SUPPORTED + 1, XE_RAS_SEV_MAX - 1)) {
+ xe_err(xe, "sysctrl: unexpected severity %u\n", severity);
+ return false;
+ }
+
+ if (!in_range(component, XE_RAS_COMP_NOT_SUPPORTED + 1, XE_RAS_COMP_MAX - 1)) {
+ xe_err(xe, "sysctrl: unexpected component %u\n", component);
+ return false;
+ }
+
+ return true;
+}
+
static struct pci_dev *find_usp_dev(struct pci_dev *pdev)
{
struct pci_dev *vsp;
@@ -218,6 +280,26 @@ static void ras_usp_aer_init(struct xe_device *xe)
dev_dbg(&usp->dev, "Uncorrectable Internal Errors downgraded and unmasked\n");
}
+static void ras_send_error_event(struct xe_device *xe, u8 severity, u8 component)
+{
+ struct xe_ras_error_class counter = {0};
+ u8 drm_severity, drm_component;
+ u32 value;
+ int ret;
+
+ counter.common.severity = severity;
+ counter.common.component = component;
+
+ ret = get_counter(xe, &counter, &value);
+ if (ret)
+ return;
+
+ drm_severity = xe_to_drm_ras_severity(severity);
+ drm_component = xe_to_drm_ras_component(component);
+
+ xe_drm_ras_event(xe, drm_component, drm_severity, value);
+}
+
static u8 handle_core_compute_errors(struct xe_ras_error_array *arr)
{
struct xe_ras_compute_error *error_info = (void *)arr->details;
@@ -236,6 +318,12 @@ static u8 handle_core_compute_errors(struct xe_ras_error_array *arr)
return XE_RAS_RECOVERY_ACTION_RECOVERED;
}
+static void punit_error_handler(struct xe_device *xe)
+{
+ xe_device_set_wedged_method(xe, DRM_WEDGE_RECOVERY_COLD_RESET);
+ xe_device_declare_wedged(xe);
+}
+
static u8 handle_soc_internal_errors(struct xe_device *xe, struct xe_ras_error_array *arr)
{
struct xe_ras_soc_error *info = (void *)arr->details;
@@ -267,7 +355,7 @@ static u8 handle_soc_internal_errors(struct xe_device *xe, struct xe_ras_error_a
xe_err(xe, "[RAS]: PUNIT %s detected: 0x%x\n",
sev_to_str(counter->common.severity),
ieh_error->global_error_status);
- /* TODO: Add PUNIT error handling */
+ punit_error_handler(xe);
return XE_RAS_RECOVERY_ACTION_DISCONNECT;
}
}
@@ -312,8 +400,10 @@ void xe_ras_counter_threshold_crossed(struct xe_device *xe,
struct xe_ras_threshold_crossed *pending = (void *)&response->data;
struct xe_ras_error_class *errors = pending->counters;
u32 id, ncounters = pending->ncounters;
+ u8 sent = 0;
BUILD_BUG_ON(sizeof(response->data) < sizeof(*pending));
+ BUILD_BUG_ON(BITS_PER_TYPE(sent) < XE_RAS_COMP_MAX);
xe_device_assert_mem_access(xe);
if (!ncounters || ncounters > XE_RAS_NUM_COUNTERS)
@@ -327,8 +417,18 @@ void xe_ras_counter_threshold_crossed(struct xe_device *xe,
severity = errors[id].common.severity;
component = errors[id].common.component;
+ if (!ras_counter_is_valid(xe, &errors[id]))
+ continue;
+
xe_warn(xe, "[RAS]: %s %s detected\n",
comp_to_str(component), sev_to_str(severity));
+
+ /* Send event once per component */
+ if (sent & BIT(component))
+ continue;
+ sent |= BIT(component);
+
+ ras_send_error_event(xe, severity, component);
}
}
@@ -358,6 +458,9 @@ static int get_counter(struct xe_device *xe, struct xe_ras_error_class *counter,
return -EIO;
}
+ if (!ras_counter_is_valid(xe, &response.counter))
+ return -EBADMSG;
+
common = &response.counter.common;
*value = response.value;
@@ -382,12 +485,20 @@ enum xe_ras_recovery_action xe_ras_process_errors(struct xe_device *xe)
enum xe_ras_recovery_action final_action;
u32 remaining = XE_SYSCTRL_FLOOD_LIMIT;
struct xe_ras_get_soc_error response;
+ u8 sent = 0;
size_t rlen;
int ret;
+ if (xe_fault_wedge_cold_reset()) {
+ xe_err(xe, "[RAS]: cold-reset wedge injected\n");
+ punit_error_handler(xe);
+ return XE_RAS_RECOVERY_ACTION_DISCONNECT;
+ }
+
if (!xe->info.has_sysctrl)
return XE_RAS_RECOVERY_ACTION_RESET;
+ BUILD_BUG_ON(BITS_PER_TYPE(sent) < XE_RAS_COMP_MAX);
/* Default action */
final_action = XE_RAS_RECOVERY_ACTION_RECOVERED;
@@ -422,9 +533,18 @@ enum xe_ras_recovery_action xe_ras_process_errors(struct xe_device *xe)
component = arr->counter.common.component;
severity = arr->counter.common.severity;
+ if (!ras_counter_is_valid(xe, &arr->counter))
+ continue;
+
xe_info(xe, "[RAS]: %s %s detected\n", comp_to_str(component),
sev_to_str(severity));
+ /* Send event once per component */
+ if (!(sent & BIT(component))) {
+ sent |= BIT(component);
+ ras_send_error_event(xe, severity, component);
+ }
+
switch (component) {
case XE_RAS_COMP_CORE_COMPUTE:
action = handle_core_compute_errors(arr);
@@ -532,6 +652,9 @@ int xe_ras_clear_counter(struct xe_device *xe, u8 severity, u8 component)
counter = &response.counter;
+ if (!ras_counter_is_valid(xe, counter))
+ return -EBADMSG;
+
xe_dbg(xe, "[RAS]: clear counter for %s %s\n", comp_to_str(counter->common.component),
sev_to_str(counter->common.severity));
diff --git a/drivers/gpu/drm/xe/xe_sriov_printk.h b/drivers/gpu/drm/xe/xe_sriov_printk.h
index 4c6b5c3d2190..5931c918655f 100644
--- a/drivers/gpu/drm/xe/xe_sriov_printk.h
+++ b/drivers/gpu/drm/xe/xe_sriov_printk.h
@@ -36,6 +36,9 @@
#define xe_sriov_dbg(xe, fmt, ...) \
xe_sriov_printk((xe), dbg, fmt, ##__VA_ARGS__)
+#define xe_sriov_dbg_ratelimited(xe, fmt, ...) \
+ xe_sriov_printk((xe), dbg_ratelimited, fmt, ##__VA_ARGS__)
+
/* for low level noisy debug messages */
#ifdef CONFIG_DRM_XE_DEBUG_SRIOV
#define xe_sriov_dbg_verbose(xe, fmt, ...) xe_sriov_dbg(xe, fmt, ##__VA_ARGS__)
diff --git a/drivers/gpu/drm/xe/xe_sriov_vf_ccs.c b/drivers/gpu/drm/xe/xe_sriov_vf_ccs.c
index a8c831fbee3b..a54138461f44 100644
--- a/drivers/gpu/drm/xe/xe_sriov_vf_ccs.c
+++ b/drivers/gpu/drm/xe/xe_sriov_vf_ccs.c
@@ -354,7 +354,6 @@ int xe_sriov_vf_ccs_init(struct xe_device *xe)
ctx->ctx_id = ctx_id;
flags = EXEC_QUEUE_FLAG_KERNEL |
- EXEC_QUEUE_FLAG_PERMANENT |
EXEC_QUEUE_FLAG_MIGRATE;
q = xe_exec_queue_create_bind(xe, tile, NULL, flags, 0);
if (IS_ERR(q)) {
diff --git a/drivers/gpu/drm/xe/xe_survivability_mode.c b/drivers/gpu/drm/xe/xe_survivability_mode.c
index 4c506027fa94..45d44ebe288b 100644
--- a/drivers/gpu/drm/xe/xe_survivability_mode.c
+++ b/drivers/gpu/drm/xe/xe_survivability_mode.c
@@ -14,9 +14,11 @@
#include "xe_device.h"
#include "xe_heci_gsc.h"
#include "xe_i2c.h"
+#include "xe_log.h"
#include "xe_mmio.h"
#include "xe_nvm.h"
#include "xe_pcode_api.h"
+#include "xe_printk.h"
#include "xe_vsec.h"
/**
@@ -172,18 +174,32 @@ static void populate_survivability_info(struct xe_device *xe)
}
}
-static void log_survivability_info(struct pci_dev *pdev)
+static const char *boot_status_str(u8 boot_status)
+{
+ switch (boot_status) {
+ case CRITICAL_FAILURE:
+ return "Critical Failure";
+ case NON_CRITICAL_FAILURE:
+ return "Non Critical Failure";
+ default:
+ return "Other";
+ }
+}
+
+static void log_survivability_info(struct xe_device *xe)
{
- struct xe_device *xe = pdev_to_xe_device(pdev);
struct xe_survivability *survivability = &xe->survivability;
u32 *info = survivability->info;
int id;
- dev_info(&pdev->dev, "Survivability Boot Status : Critical Failure (%d)\n",
- survivability->boot_status);
+ xe_log_info(xe, SURVIVABILITY, "Boot Status: %#x (%s)\n",
+ survivability->boot_status,
+ boot_status_str(survivability->boot_status));
+
for (id = 0; id < MAX_SCRATCH_REG; id++) {
- if (info[id])
- dev_info(&pdev->dev, "%s: 0x%x\n", reg_map[id], info[id]);
+ if (!info[id])
+ continue;
+ xe_log_info(xe, SURVIVABILITY, "%s: %#x\n", reg_map[id], info[id]);
}
}
@@ -289,41 +305,46 @@ static const struct attribute_group survivability_info_group = {
static int create_survivability_sysfs(struct pci_dev *pdev)
{
- struct device *dev = &pdev->dev;
+ /* Survivability info is required if not enabled via configfs */
+ bool needs_info = !xe_configfs_get_survivability_mode(pdev);
struct xe_device *xe = pdev_to_xe_device(pdev);
+ struct device *dev = &pdev->dev;
int ret;
ret = device_create_file(dev, &dev_attr_survivability_mode);
- if (ret) {
- dev_warn(dev, "Failed to create survivability sysfs files\n");
- return ret;
- }
+ if (ret)
+ goto failed;
ret = devm_add_action_or_reset(xe->drm.dev,
xe_survivability_mode_fini, xe);
if (ret)
- return ret;
+ goto failed;
- /* Survivability info is not required if enabled via configfs */
- if (!xe_configfs_get_survivability_mode(pdev)) {
+ if (needs_info) {
ret = devm_device_add_group(dev, &survivability_info_group);
if (ret)
- return ret;
+ goto failed;
}
return 0;
+
+failed:
+ xe_err(xe, "Failed to create survivability sysfs files: %pe\n", ERR_PTR(ret));
+ /* no sysfs, dump Survivability info to dmesg instead */
+ if (needs_info)
+ log_survivability_info(xe);
+ return ret;
}
static int enable_boot_survivability_mode(struct pci_dev *pdev)
{
- struct device *dev = &pdev->dev;
struct xe_device *xe = pdev_to_xe_device(pdev);
struct xe_survivability *survivability = &xe->survivability;
- int ret = 0;
+ int ret;
ret = create_survivability_sysfs(pdev);
if (ret)
- return ret;
+ goto failed;
/* Make sure xe_heci_gsc_init() and xe_i2c_probe() are aware of survivability */
survivability->mode = true;
@@ -335,19 +356,22 @@ static int enable_boot_survivability_mode(struct pci_dev *pdev)
if (survivability->fdo_mode) {
ret = xe_nvm_init(xe);
if (ret)
- goto err;
+ goto failed;
}
ret = xe_i2c_probe(xe);
if (ret)
- goto err;
+ goto failed;
- dev_err(dev, "In Survivability Mode\n");
+ if (check_boot_failure(xe))
+ xe_log_err_fatal(pdev, SURVIVABILITY, 0, "Boot Mode enabled!\n");
+ else
+ xe_log_info(pdev, SURVIVABILITY, "Boot Mode enabled!\n");
return 0;
-err:
- dev_err(dev, "Failed to enable Survivability Mode\n");
+failed:
+ xe_log_err_fatal(pdev, SURVIVABILITY, ret, "Failed to enable Boot Mode!\n");
survivability->mode = false;
return ret;
}
@@ -412,21 +436,22 @@ void xe_survivability_mode_runtime_enable(struct xe_device *xe)
struct pci_dev *pdev = to_pci_dev(xe->drm.dev);
if (!IS_DGFX(xe) || IS_SRIOV_VF(xe) || xe->info.platform < XE_BATTLEMAGE) {
- dev_err(&pdev->dev, "Runtime Survivability Mode not supported\n");
+ xe_log_err(xe, SURVIVABILITY, -EOPNOTSUPP, "Runtime Mode not supported!\n");
return;
}
populate_survivability_info(xe);
-
- if (create_survivability_sysfs(pdev))
- dev_err(&pdev->dev, "Failed to create survivability sysfs\n");
+ create_survivability_sysfs(pdev);
survivability->type = XE_SURVIVABILITY_TYPE_RUNTIME;
- dev_err(&pdev->dev, "Runtime Survivability mode enabled\n");
+ xe_log_err(xe, SURVIVABILITY, 0, "Runtime Mode enabled!\n");
xe_device_set_wedged_method(xe, DRM_WEDGE_RECOVERY_VENDOR);
xe_device_declare_wedged(xe);
- dev_err(&pdev->dev, "Firmware flash required, Please refer to the userspace documentation for more details!\n");
+
+ xe_log_err(xe, SURVIVABILITY, 0, "Firmware flash required!\n");
+ xe_info(xe, "Please refer to the userspace documentation for more details how to flash the firmware on %s!\n",
+ xe->info.platform_name);
}
/**
@@ -452,7 +477,7 @@ int xe_survivability_mode_boot_enable(struct xe_device *xe)
* v2 supports survivability mode for critical errors
*/
if (survivability->version < 2 && survivability->boot_status == CRITICAL_FAILURE) {
- log_survivability_info(pdev);
+ log_survivability_info(xe);
return -ENXIO;
}
diff --git a/drivers/gpu/drm/xe/xe_svm.c b/drivers/gpu/drm/xe/xe_svm.c
index b228a737cfd6..627a741293d5 100644
--- a/drivers/gpu/drm/xe/xe_svm.c
+++ b/drivers/gpu/drm/xe/xe_svm.c
@@ -15,6 +15,7 @@
#include "xe_gt_stats.h"
#include "xe_migrate.h"
#include "xe_module.h"
+#include "xe_pagefault.h"
#include "xe_pm.h"
#include "xe_pt.h"
#include "xe_svm.h"
@@ -115,6 +116,7 @@ xe_svm_range_alloc(struct drm_gpusvm *gpusvm)
return NULL;
INIT_LIST_HEAD(&range->garbage_collector_link);
+ mutex_init(&range->lock);
drm_gpusvm_init_pages(&range->pages, &gpusvm_to_vm(gpusvm)->xe->drm);
xe_vm_get(gpusvm_to_vm(gpusvm));
@@ -125,6 +127,7 @@ static void xe_svm_range_free(struct drm_gpusvm_range *range)
{
drm_gpusvm_free_pages(range->gpusvm, &(to_xe_range(range)->pages),
drm_gpusvm_range_size(range) >> PAGE_SHIFT);
+ mutex_destroy(&to_xe_range(range)->lock);
xe_vm_put(range_to_vm(range));
kfree(to_xe_range(range));
}
@@ -140,13 +143,13 @@ xe_svm_garbage_collector_add_range(struct xe_vm *vm, struct xe_svm_range *range,
drm_gpusvm_range_set_unmapped(&range->base, &range->pages, 1,
mmu_range);
- spin_lock(&vm->svm.garbage_collector.lock);
+ spin_lock(&vm->svm.garbage_collector.list_lock);
if (list_empty(&range->garbage_collector_link))
list_add_tail(&range->garbage_collector_link,
&vm->svm.garbage_collector.range_list);
- spin_unlock(&vm->svm.garbage_collector.lock);
+ spin_unlock(&vm->svm.garbage_collector.list_lock);
- queue_work(xe->usm.pf_wq, &vm->svm.garbage_collector.work);
+ queue_work(xe->usm.pagefault_wq, &vm->svm.garbage_collector.work);
}
static void xe_svm_tlb_inval_count_stats_incr(struct xe_gt *gt)
@@ -309,18 +312,30 @@ static int __xe_svm_garbage_collector(struct xe_vm *vm,
range_debug(range, "GARBAGE COLLECTOR");
- xe_vm_lock(vm, false);
- fence = xe_vm_range_unbind(vm, range);
- xe_vm_unlock(vm);
- if (IS_ERR(fence))
- return PTR_ERR(fence);
- dma_fence_put(fence);
+ scoped_guard(mutex, &range->lock) {
+ drm_gpusvm_range_get(&range->base);
+ range->removed = true;
+
+ range_debug(range, "GARBAGE COLLECTOR");
+
+ xe_vm_lock(vm, false);
+ fence = xe_vm_range_unbind(vm, range);
+ xe_vm_unlock(vm);
+ if (IS_ERR(fence)) {
+ drm_gpusvm_range_put(&range->base);
+ return PTR_ERR(fence);
+ }
+ dma_fence_put(fence);
+
+ drm_gpusvm_unmap_pages(&vm->svm.gpusvm, &range->pages,
+ drm_gpusvm_range_size(&range->base) >> PAGE_SHIFT,
+ &ctx);
- drm_gpusvm_unmap_pages(&vm->svm.gpusvm, &range->pages,
- drm_gpusvm_range_size(&range->base) >> PAGE_SHIFT,
- &ctx);
+ scoped_guard(mutex, &vm->svm.range_lock)
+ drm_gpusvm_range_remove(&vm->svm.gpusvm, &range->base);
+ }
- drm_gpusvm_range_remove(&vm->svm.gpusvm, &range->base);
+ drm_gpusvm_range_put(&range->base);
return 0;
}
@@ -393,13 +408,15 @@ static int xe_svm_garbage_collector(struct xe_vm *vm)
u64 range_end;
int err, ret = 0;
- lockdep_assert_held_write(&vm->lock);
+ lockdep_assert_held(&vm->lock);
if (xe_vm_is_closed_or_banned(vm))
return -ENOENT;
+ guard(mutex)(&vm->svm.garbage_collector.lock);
+
for (;;) {
- spin_lock(&vm->svm.garbage_collector.lock);
+ spin_lock(&vm->svm.garbage_collector.list_lock);
range = list_first_entry_or_null(&vm->svm.garbage_collector.range_list,
typeof(*range),
garbage_collector_link);
@@ -410,7 +427,7 @@ static int xe_svm_garbage_collector(struct xe_vm *vm)
range_end = xe_svm_range_end(range);
list_del(&range->garbage_collector_link);
- spin_unlock(&vm->svm.garbage_collector.lock);
+ spin_unlock(&vm->svm.garbage_collector.list_lock);
err = __xe_svm_garbage_collector(vm, range);
if (err) {
@@ -429,7 +446,7 @@ static int xe_svm_garbage_collector(struct xe_vm *vm)
return err;
}
}
- spin_unlock(&vm->svm.garbage_collector.lock);
+ spin_unlock(&vm->svm.garbage_collector.list_lock);
return ret;
}
@@ -439,9 +456,8 @@ static void xe_svm_garbage_collector_work_func(struct work_struct *w)
struct xe_vm *vm = container_of(w, struct xe_vm,
svm.garbage_collector.work);
- down_write(&vm->lock);
+ guard(rwsem_read)(&vm->lock);
xe_svm_garbage_collector(vm);
- up_write(&vm->lock);
}
#if IS_ENABLED(CONFIG_DRM_XE_PAGEMAP)
@@ -893,8 +909,11 @@ int xe_svm_init(struct xe_vm *vm)
{
int err;
+ mutex_init(&vm->svm.range_lock);
+ mutex_init(&vm->svm.garbage_collector.lock);
+
if (vm->flags & XE_VM_FLAG_FAULT_MODE) {
- spin_lock_init(&vm->svm.garbage_collector.lock);
+ spin_lock_init(&vm->svm.garbage_collector.list_lock);
INIT_LIST_HEAD(&vm->svm.garbage_collector.range_list);
INIT_WORK(&vm->svm.garbage_collector.work,
xe_svm_garbage_collector_work_func);
@@ -903,12 +922,12 @@ int xe_svm_init(struct xe_vm *vm)
err = drm_pagemap_acquire_owner(&vm->svm.peer, &xe_owner_list,
xe_has_interconnect);
if (err)
- return err;
+ goto out_err;
err = xe_svm_get_pagemaps(vm);
if (err) {
drm_pagemap_release_owner(&vm->svm.peer);
- return err;
+ goto out_err;
}
err = drm_gpusvm_init(&vm->svm.gpusvm, "Xe SVM",
@@ -916,19 +935,27 @@ int xe_svm_init(struct xe_vm *vm)
xe_modparam.svm_notifier_size * SZ_1M,
&gpusvm_ops, fault_chunk_sizes,
ARRAY_SIZE(fault_chunk_sizes));
- drm_gpusvm_driver_set_lock(&vm->svm.gpusvm, &vm->lock);
+ drm_gpusvm_driver_set_lock(&vm->svm.gpusvm, &vm->svm.range_lock);
if (err) {
xe_svm_put_pagemaps(vm);
drm_pagemap_release_owner(&vm->svm.peer);
- return err;
+ goto out_err;
}
} else {
err = drm_gpusvm_init(&vm->svm.gpusvm, "Xe SVM (simple)",
NULL, 0, 0, 0, NULL,
NULL, 0);
+ if (err)
+ goto out_err;
}
+ return 0;
+
+out_err:
+ mutex_destroy(&vm->svm.range_lock);
+ mutex_destroy(&vm->svm.garbage_collector.lock);
+
return err;
}
@@ -969,7 +996,10 @@ void xe_svm_fini(struct xe_vm *vm)
&ctx);
}
- drm_gpusvm_fini(&vm->svm.gpusvm);
+ scoped_guard(mutex, &vm->svm.range_lock)
+ drm_gpusvm_fini(&vm->svm.gpusvm);
+ mutex_destroy(&vm->svm.range_lock);
+ mutex_destroy(&vm->svm.garbage_collector.lock);
}
static bool xe_svm_range_has_pagemap_locked(const struct xe_svm_range *range,
@@ -1022,6 +1052,7 @@ void xe_svm_range_migrate_to_smem(struct xe_vm *vm, struct xe_svm_range *range)
* @tile_mask: Mask representing the tiles to be checked
* @dpagemap: if !%NULL, the range is expected to be present
* in device memory identified by this parameter.
+ * @valid_pages: Pages are valid, result written back to caller
*
* The xe_svm_range_validate() function checks if a range is
* valid and located in the desired memory region.
@@ -1030,7 +1061,8 @@ void xe_svm_range_migrate_to_smem(struct xe_vm *vm, struct xe_svm_range *range)
*/
bool xe_svm_range_validate(struct xe_vm *vm,
struct xe_svm_range *range,
- u8 tile_mask, const struct drm_pagemap *dpagemap)
+ u8 tile_mask, const struct drm_pagemap *dpagemap,
+ bool *valid_pages)
{
bool ret;
@@ -1042,6 +1074,8 @@ bool xe_svm_range_validate(struct xe_vm *vm,
else
ret = ret && !range->pages.dpagemap;
+ *valid_pages = xe_svm_range_pages_valid(range);
+
xe_svm_notifier_unlock(vm);
return ret;
@@ -1231,8 +1265,8 @@ DECL_SVM_RANGE_US_STATS(bind, BIND)
DECL_SVM_RANGE_US_STATS(fault, PAGEFAULT)
static int __xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
- struct xe_gt *gt, u64 fault_addr,
- bool need_vram)
+ struct xe_pagefault *pf, struct xe_gt *gt,
+ u64 fault_addr, bool need_vram)
{
int devmem_possible = IS_DGFX(vm->xe) &&
IS_ENABLED(CONFIG_DRM_XE_PAGEMAP);
@@ -1246,21 +1280,27 @@ static int __xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
};
struct xe_validation_ctx vctx;
struct drm_exec exec;
- struct xe_svm_range *range;
+ struct xe_svm_range *range = NULL;
struct drm_gpusvm_range_flags range_flags;
struct dma_fence *fence;
struct drm_pagemap *dpagemap;
struct xe_tile *tile = gt_to_tile(gt);
int migrate_try_count = ctx.devmem_only ? 3 : 1;
ktime_t start = xe_gt_stats_ktime_get(), bind_start, get_pages_start;
- int err;
+ int err = 0;
- lockdep_assert_held_write(&vm->lock);
+ lockdep_assert_held(&vm->lock);
xe_assert(vm->xe, xe_vma_is_cpu_addr_mirror(vma));
xe_gt_stats_incr(gt, XE_GT_STATS_ID_SVM_PAGEFAULT_COUNT, 1);
retry:
+ /* Release old range */
+ if (range) {
+ mutex_unlock(&range->lock);
+ drm_gpusvm_range_put(&range->base);
+ }
+
/* Always process UNMAPs first so view SVM ranges is current */
err = xe_svm_garbage_collector(vm);
if (err)
@@ -1276,10 +1316,17 @@ retry:
xe_svm_range_fault_count_stats_incr(gt, range);
+ mutex_lock(&range->lock);
+
+ if (xe_svm_range_is_removed(range))
+ goto retry;
+
/* READ_ONCE pairs with WRITE_ONCE in drm_gpusvm_range_set_unmapped() */
range_flags.__flags = READ_ONCE(range->base.flags.__flags);
- if (ctx.devmem_only && !range_flags.migrate_devmem)
- return -EACCES;
+ if (ctx.devmem_only && !range_flags.migrate_devmem) {
+ err = -EACCES;
+ goto err_out;
+ }
if (xe_svm_range_is_valid(range, tile, ctx.devmem_only, dpagemap)) {
xe_svm_range_valid_fault_count_stats_incr(gt, range);
@@ -1317,7 +1364,7 @@ retry:
drm_err(&vm->xe->drm,
"VRAM allocation failed, retry count exceeded, asid=%u, errno=%pe\n",
vm->usm.asid, ERR_PTR(err));
- return err;
+ goto err_out;
}
}
}
@@ -1344,7 +1391,7 @@ get_pages:
}
if (err) {
range_debug(range, "PAGE FAULT - FAIL PAGE COLLECT");
- goto out;
+ goto err_out;
} else if (IS_ENABLED(CONFIG_DRM_XE_DEBUG_VM)) {
drm_dbg(&vm->xe->drm, "After page collect data location is %sin \"%s\".\n",
xe_svm_range_has_pagemap(range, dpagemap) ? "" : "NOT ",
@@ -1378,7 +1425,13 @@ get_pages:
xe_svm_range_bind_us_stats_incr(gt, range, bind_start);
out:
+ /* Give hint to immediately ack faults */
+ xe_pagefault_set_start_addr(pf, xe_svm_range_start(range));
+ xe_pagefault_set_end_addr(pf, xe_svm_range_end(range));
+
xe_svm_range_fault_us_stats_incr(gt, range, start);
+ mutex_unlock(&range->lock);
+ drm_gpusvm_range_put(&range->base);
return 0;
err_out:
@@ -1388,6 +1441,9 @@ err_out:
goto retry;
}
+ mutex_unlock(&range->lock);
+ drm_gpusvm_range_put(&range->base);
+
return err;
}
@@ -1395,6 +1451,7 @@ err_out:
* xe_svm_handle_pagefault() - SVM handle page fault
* @vm: The VM.
* @vma: The CPU address mirror VMA.
+ * @pf: Pagefault structure
* @gt: The gt upon the fault occurred.
* @fault_addr: The GPU fault address.
* @atomic: The fault atomic access bit.
@@ -1405,8 +1462,8 @@ err_out:
* Return: 0 on success, negative error code on error.
*/
int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
- struct xe_gt *gt, u64 fault_addr,
- bool atomic)
+ struct xe_pagefault *pf, struct xe_gt *gt,
+ u64 fault_addr, bool atomic)
{
int need_vram, ret;
retry:
@@ -1414,7 +1471,7 @@ retry:
if (need_vram < 0)
return need_vram;
- ret = __xe_svm_handle_pagefault(vm, vma, gt, fault_addr,
+ ret = __xe_svm_handle_pagefault(vm, vma, pf, gt, fault_addr,
need_vram ? true : false);
if (ret == -EAGAIN) {
/*
@@ -1470,9 +1527,9 @@ void xe_svm_unmap_address_range(struct xe_vm *vm, u64 start, u64 end)
drm_gpusvm_range_get(range);
__xe_svm_garbage_collector(vm, to_xe_range(range));
if (!list_empty(&to_xe_range(range)->garbage_collector_link)) {
- spin_lock(&vm->svm.garbage_collector.lock);
+ spin_lock(&vm->svm.garbage_collector.list_lock);
list_del(&to_xe_range(range)->garbage_collector_link);
- spin_unlock(&vm->svm.garbage_collector.lock);
+ spin_unlock(&vm->svm.garbage_collector.list_lock);
}
drm_gpusvm_range_put(range);
}
@@ -1502,7 +1559,7 @@ int xe_svm_bo_evict(struct xe_bo *bo)
* @ctx: GPU SVM context
*
* This function finds or inserts a newly allocated a SVM range based on the
- * address.
+ * address. Take a reference to SVM range on success.
*
* Return: Pointer to the SVM range on success, ERR_PTR() on failure.
*/
@@ -1511,11 +1568,15 @@ struct xe_svm_range *xe_svm_range_find_or_insert(struct xe_vm *vm, u64 addr,
{
struct drm_gpusvm_range *r;
+ guard(mutex)(&vm->svm.range_lock);
+
r = drm_gpusvm_range_find_or_insert(&vm->svm.gpusvm, max(addr, xe_vma_start(vma)),
xe_vma_start(vma), xe_vma_end(vma), ctx);
if (IS_ERR(r))
return ERR_CAST(r);
+ drm_gpusvm_range_get(r);
+
return to_xe_range(r);
}
@@ -1535,6 +1596,8 @@ int xe_svm_range_get_pages(struct xe_vm *vm, struct xe_svm_range *range,
{
int err = 0;
+ lockdep_assert_held(&range->lock);
+
err = drm_gpusvm_get_pages(&vm->svm.gpusvm, &range->pages,
vm->svm.gpusvm.mm,
&range->base.notifier->notifier,
@@ -1659,6 +1722,7 @@ int xe_svm_alloc_vram(struct xe_svm_range *range, const struct drm_gpusvm_ctx *c
.__flags = READ_ONCE(range->base.flags.__flags),
};
+ lockdep_assert_held(&range->lock);
xe_assert(range_to_vm(&range->base)->xe, flags.migrate_devmem);
range_debug(range, "ALLOCATE VRAM");
diff --git a/drivers/gpu/drm/xe/xe_svm.h b/drivers/gpu/drm/xe/xe_svm.h
index a921556d3466..2a0dc0d125c9 100644
--- a/drivers/gpu/drm/xe/xe_svm.h
+++ b/drivers/gpu/drm/xe/xe_svm.h
@@ -21,6 +21,7 @@ struct drm_file;
struct xe_bo;
struct xe_gt;
struct xe_device;
+struct xe_pagefault;
struct xe_vram_region;
struct xe_tile;
struct xe_vm;
@@ -39,6 +40,13 @@ struct xe_svm_range {
*/
struct list_head garbage_collector_link;
/**
+ * @lock: Protects fault handler, garbage collector, and prefetch
+ * critical sections, ensuring only one thread operates on a range at a
+ * time. Locking order: inside vm->lock and garbage collector, outside
+ * dma-resv locks, vm->svm.range_lock.
+ */
+ struct mutex lock;
+ /**
* @tile_present: Tile mask of binding is present for this range.
* Protected by GPU SVM notifier lock.
*/
@@ -48,9 +56,23 @@ struct xe_svm_range {
* range. Protected by GPU SVM notifier lock.
*/
u8 tile_invalidated;
+ /**
+ * @removed: Range has been removed from GPU SVM tree, protected by
+ * @lock.
+ */
+ bool removed;
};
/**
+ * xe_svm_range_put() - SVM range put
+ * @range: SVM range
+ */
+static inline void xe_svm_range_put(struct xe_svm_range *range)
+{
+ drm_gpusvm_range_put(&range->base);
+}
+
+/**
* struct xe_pagemap - Manages xe device_private memory for SVM.
* @pagemap: The struct dev_pagemap providing the struct pages.
* @dpagemap: The drm_pagemap managing allocation and migration.
@@ -88,8 +110,8 @@ void xe_svm_fini(struct xe_vm *vm);
void xe_svm_close(struct xe_vm *vm);
int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
- struct xe_gt *gt, u64 fault_addr,
- bool atomic);
+ struct xe_pagefault *pf, struct xe_gt *gt,
+ u64 fault_addr, bool atomic);
bool xe_svm_has_mapping(struct xe_vm *vm, u64 start, u64 end);
@@ -113,7 +135,8 @@ void xe_svm_range_migrate_to_smem(struct xe_vm *vm, struct xe_svm_range *range);
bool xe_svm_range_validate(struct xe_vm *vm,
struct xe_svm_range *range,
- u8 tile_mask, const struct drm_pagemap *dpagemap);
+ u8 tile_mask, const struct drm_pagemap *dpagemap,
+ bool *valid_pages);
u64 xe_svm_find_vma_start(struct xe_vm *vm, u64 addr, u64 end, struct xe_vma *vma);
@@ -138,6 +161,19 @@ static inline bool xe_svm_range_has_dma_mapping(struct xe_svm_range *range)
}
/**
+ * xe_svm_range_is_removed() - SVM range is removed from GPU SVM tree
+ * @range: SVM range
+ *
+ * Return: True if SVM range is removed from GPU SVM tree, False otherwise
+ */
+static inline bool xe_svm_range_is_removed(struct xe_svm_range *range)
+{
+ lockdep_assert_held(&range->lock);
+
+ return range->removed;
+}
+
+/**
* to_xe_range - Convert a drm_gpusvm_range pointer to a xe_svm_range
* @r: Pointer to the drm_gpusvm_range structure
*
@@ -216,10 +252,15 @@ struct xe_svm_range {
struct {
const struct drm_pagemap_addr *dma_addr;
} pages;
+ struct mutex lock;
u32 tile_present;
u32 tile_invalidated;
};
+static inline void xe_svm_range_put(struct xe_svm_range *range)
+{
+}
+
static inline bool xe_svm_range_pages_valid(struct xe_svm_range *range)
{
return false;
@@ -258,8 +299,8 @@ void xe_svm_close(struct xe_vm *vm)
static inline
int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
- struct xe_gt *gt, u64 fault_addr,
- bool atomic)
+ struct xe_pagefault *pf, struct xe_gt *gt,
+ u64 fault_addr, bool atomic)
{
return 0;
}
@@ -337,7 +378,8 @@ void xe_svm_range_migrate_to_smem(struct xe_vm *vm, struct xe_svm_range *range)
static inline
bool xe_svm_range_validate(struct xe_vm *vm,
struct xe_svm_range *range,
- u8 tile_mask, bool devmem_preferred)
+ u8 tile_mask, const struct drm_pagemap *dpagemap,
+ bool *valid_pages)
{
return false;
}
@@ -389,6 +431,11 @@ static inline struct drm_pagemap *xe_drm_pagemap_from_fd(int fd, u32 region_inst
return ERR_PTR(-ENOENT);
}
+static inline bool xe_svm_range_is_removed(struct xe_svm_range *range)
+{
+ return false;
+}
+
#define xe_svm_range_has_dma_mapping(...) false
#endif /* CONFIG_DRM_XE_GPUSVM */
diff --git a/drivers/gpu/drm/xe/xe_tile_printk.h b/drivers/gpu/drm/xe/xe_tile_printk.h
index 738433a764bd..c87c07e14fd1 100644
--- a/drivers/gpu/drm/xe/xe_tile_printk.h
+++ b/drivers/gpu/drm/xe/xe_tile_printk.h
@@ -34,6 +34,9 @@
#define xe_tile_dbg(_tile, _fmt, ...) \
xe_tile_printk((_tile), dbg, _fmt, ##__VA_ARGS__)
+#define xe_tile_dbg_ratelimited(_tile, _fmt, ...) \
+ xe_tile_printk((_tile), dbg_ratelimited, _fmt, ##__VA_ARGS__)
+
#define xe_tile_WARN_type(_tile, _type, _condition, _fmt, ...) \
xe_WARN##_type((_tile)->xe, _condition, _fmt, ## __VA_ARGS__)
diff --git a/drivers/gpu/drm/xe/xe_tile_sriov_printk.h b/drivers/gpu/drm/xe/xe_tile_sriov_printk.h
index 68323512872c..c23e698bc8db 100644
--- a/drivers/gpu/drm/xe/xe_tile_sriov_printk.h
+++ b/drivers/gpu/drm/xe/xe_tile_sriov_printk.h
@@ -27,6 +27,9 @@
#define xe_tile_sriov_dbg(_tile, _fmt, ...) \
xe_tile_sriov_printk(_tile, dbg, _fmt, ##__VA_ARGS__)
+#define xe_tile_sriov_dbg_ratelimited(_tile, _fmt, ...) \
+ xe_tile_sriov_printk(_tile, dbg_ratelimited, _fmt, ##__VA_ARGS__)
+
#define xe_tile_sriov_dbg_verbose(_tile, _fmt, ...) \
xe_tile_sriov_printk(_tile, dbg_verbose, _fmt, ##__VA_ARGS__)
diff --git a/drivers/gpu/drm/xe/xe_userptr.c b/drivers/gpu/drm/xe/xe_userptr.c
index 8b2d461ea0b2..90ac141fc12d 100644
--- a/drivers/gpu/drm/xe/xe_userptr.c
+++ b/drivers/gpu/drm/xe/xe_userptr.c
@@ -57,6 +57,23 @@ int __xe_vm_userptr_needs_repin(struct xe_vm *vm)
list_empty(&vm->userptr.invalidated)) ? 0 : -EAGAIN;
}
+#if IS_ENABLED(CONFIG_PROVE_LOCKING)
+static bool __xe_vma_userptr_lockdep(struct xe_userptr_vma *uvma)
+{
+ struct xe_vma *vma = &uvma->vma;
+ struct xe_vm *vm = xe_vma_vm(vma);
+
+ return lockdep_is_held_type(&vm->lock, 0) ||
+ (lockdep_is_held_type(&vm->lock, 1) &&
+ lockdep_is_held_type(&vma->fault_lock, 0));
+}
+
+#define xe_vma_userptr_lockdep(uvma) \
+ lockdep_assert(__xe_vma_userptr_lockdep(uvma))
+#else
+#define xe_vma_userptr_lockdep(uvma)
+#endif
+
int xe_vma_userptr_pin_pages(struct xe_userptr_vma *uvma)
{
struct xe_vma *vma = &uvma->vma;
@@ -68,7 +85,7 @@ int xe_vma_userptr_pin_pages(struct xe_userptr_vma *uvma)
.allow_mixed = true,
};
- lockdep_assert_held(&vm->lock);
+ xe_vma_userptr_lockdep(uvma);
xe_assert(xe, xe_vma_is_userptr(vma));
if (vma->gpuva.flags & XE_VMA_DESTROYED)
diff --git a/drivers/gpu/drm/xe/xe_vm.c b/drivers/gpu/drm/xe/xe_vm.c
index 9e0176861cb6..19b3d0be7928 100644
--- a/drivers/gpu/drm/xe/xe_vm.c
+++ b/drivers/gpu/drm/xe/xe_vm.c
@@ -615,10 +615,13 @@ void xe_vm_add_fault_entry_pf(struct xe_vm *vm, struct xe_pagefault *pf)
{
struct xe_vm_fault_entry *e;
struct xe_hw_engine *hwe;
+ u8 engine_class = FIELD_GET(XE_PAGEFAULT_ENGINE_CLASS_MASK,
+ pf->consumer.engine_class_instance);
+ u8 engine_instance = FIELD_GET(XE_PAGEFAULT_ENGINE_INSTANCE_MASK,
+ pf->consumer.engine_class_instance);
/* Do not report faults on reserved engines */
- hwe = xe_gt_hw_engine(pf->gt, pf->consumer.engine_class,
- pf->consumer.engine_instance, false);
+ hwe = xe_gt_hw_engine(pf->gt, engine_class, engine_instance, false);
if (!hwe || xe_hw_engine_is_reserved(hwe))
return;
@@ -689,6 +692,17 @@ static int xe_vma_ops_alloc(struct xe_vma_ops *vops, bool array_of_binds)
}
ALLOW_ERROR_INJECTION(xe_vma_ops_alloc, ERRNO);
+static void xe_vma_svm_prefetch_ranges_fini(struct xe_vma_op *op)
+{
+ struct xe_svm_range *svm_range;
+ unsigned long i;
+
+ xa_for_each(&op->prefetch_range.range, i, svm_range)
+ xe_svm_range_put(svm_range);
+
+ xa_destroy(&op->prefetch_range.range);
+}
+
static void xe_vma_svm_prefetch_op_fini(struct xe_vma_op *op)
{
struct xe_vma *vma;
@@ -696,7 +710,7 @@ static void xe_vma_svm_prefetch_op_fini(struct xe_vma_op *op)
vma = gpuva_to_vma(op->base.prefetch.va);
if (op->base.op == DRM_GPUVA_OP_PREFETCH && xe_vma_is_cpu_addr_mirror(vma))
- xa_destroy(&op->prefetch_range.range);
+ xe_vma_svm_prefetch_ranges_fini(op);
}
static void xe_vma_svm_prefetch_ops_fini(struct xe_vma_ops *vops)
@@ -930,6 +944,7 @@ struct dma_fence *xe_vm_range_rebind(struct xe_vm *vm,
u8 id;
int err;
+ lockdep_assert_held(&range->lock);
lockdep_assert_held(&vm->lock);
xe_vm_assert_held(vm);
xe_assert(vm->xe, xe_vm_in_fault_mode(vm));
@@ -1012,6 +1027,7 @@ struct dma_fence *xe_vm_range_unbind(struct xe_vm *vm,
u8 id;
int err;
+ lockdep_assert_held(&range->lock);
lockdep_assert_held(&vm->lock);
xe_vm_assert_held(vm);
xe_assert(vm->xe, xe_vm_in_fault_mode(vm));
@@ -1187,6 +1203,8 @@ static struct xe_vma *xe_vma_create(struct xe_vm *vm,
xe_vm_get(vm);
}
+ mutex_init(&vma->fault_lock);
+
return vma;
}
@@ -1211,6 +1229,7 @@ static void xe_vma_destroy_late(struct xe_vma *vma)
xe_bo_put(bo);
}
+ mutex_destroy(&vma->fault_lock);
xe_vma_free(vma);
}
@@ -1231,12 +1250,19 @@ static void vma_destroy_cb(struct dma_fence *fence,
queue_work(system_dfl_wq, &vma->destroy_work);
}
+static void xe_vm_assert_write_mode_or_garbage_collector(struct xe_vm *vm)
+{
+ lockdep_assert(lockdep_is_held_type(&vm->lock, 0) ||
+ (lockdep_is_held_type(&vm->lock, 1) &&
+ lockdep_is_held_type(&vm->svm.garbage_collector.lock, 0)));
+}
+
static void xe_vma_destroy(struct xe_vma *vma, struct dma_fence *fence)
{
struct xe_vm *vm = xe_vma_vm(vma);
struct xe_bo *bo = xe_vma_bo(vma);
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
xe_assert(vm->xe, list_empty(&vma->combined_links.destroy));
if (xe_vma_is_userptr(vma)) {
@@ -1320,7 +1346,9 @@ xe_vm_find_overlapping_vma(struct xe_vm *vm, u64 start, u64 range)
xe_assert(vm->xe, start + range <= vm->size);
+ mutex_lock(&vm->snap_mutex);
gpuva = drm_gpuva_find_first(&vm->gpuvm, start, range);
+ mutex_unlock(&vm->snap_mutex);
return gpuva ? gpuva_to_vma(gpuva) : NULL;
}
@@ -1330,7 +1358,7 @@ static int xe_vm_insert_vma(struct xe_vm *vm, struct xe_vma *vma)
int err;
xe_assert(vm->xe, xe_vma_vm(vma) == vm);
- lockdep_assert_held(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
mutex_lock(&vm->snap_mutex);
err = drm_gpuva_insert(&vm->gpuvm, &vma->gpuva);
@@ -1343,13 +1371,11 @@ static int xe_vm_insert_vma(struct xe_vm *vm, struct xe_vma *vma)
static void xe_vm_remove_vma(struct xe_vm *vm, struct xe_vma *vma)
{
xe_assert(vm->xe, xe_vma_vm(vma) == vm);
- lockdep_assert_held(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
mutex_lock(&vm->snap_mutex);
drm_gpuva_remove(&vma->gpuva);
mutex_unlock(&vm->snap_mutex);
- if (vm->usm.last_fault_vma == vma)
- vm->usm.last_fault_vma = NULL;
}
static struct drm_gpuva_op *xe_vm_op_alloc(void)
@@ -2185,7 +2211,7 @@ static int xe_vm_query_vmas(struct xe_vm *vm, u64 start, u64 end)
struct drm_gpuva *gpuva;
u32 num_vmas = 0;
- lockdep_assert_held(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
drm_gpuvm_for_each_va_range(gpuva, &vm->gpuvm, start, end)
num_vmas++;
@@ -2198,7 +2224,7 @@ static int get_mem_attrs(struct xe_vm *vm, u32 *num_vmas, u64 start,
struct drm_gpuva *gpuva;
int i = 0;
- lockdep_assert_held(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
drm_gpuvm_for_each_va_range(gpuva, &vm->gpuvm, start, end) {
struct xe_vma *vma = gpuva_to_vma(gpuva);
@@ -2244,7 +2270,7 @@ int xe_vm_query_vmas_attrs_ioctl(struct drm_device *dev, void *data, struct drm_
if (XE_IOCTL_DBG(xe, !vm))
return -EINVAL;
- err = down_read_interruptible(&vm->lock);
+ err = down_write_killable(&vm->lock);
if (err)
goto put_vm;
@@ -2278,21 +2304,12 @@ int xe_vm_query_vmas_attrs_ioctl(struct drm_device *dev, void *data, struct drm_
free_mem_attrs:
kvfree(mem_attrs);
unlock_vm:
- up_read(&vm->lock);
+ up_write(&vm->lock);
put_vm:
xe_vm_put(vm);
return err;
}
-static bool vma_matches(struct xe_vma *vma, u64 page_addr)
-{
- if (page_addr > xe_vma_end(vma) - 1 ||
- page_addr + SZ_4K - 1 < xe_vma_start(vma))
- return false;
-
- return true;
-}
-
/**
* xe_vm_find_vma_by_addr() - Find a VMA by its address
*
@@ -2301,16 +2318,7 @@ static bool vma_matches(struct xe_vma *vma, u64 page_addr)
*/
struct xe_vma *xe_vm_find_vma_by_addr(struct xe_vm *vm, u64 page_addr)
{
- struct xe_vma *vma = NULL;
-
- if (vm->usm.last_fault_vma) { /* Fast lookup */
- if (vma_matches(vm->usm.last_fault_vma, page_addr))
- vma = vm->usm.last_fault_vma;
- }
- if (!vma)
- vma = xe_vm_find_overlapping_vma(vm, page_addr, SZ_4K);
-
- return vma;
+ return xe_vm_find_overlapping_vma(vm, page_addr, SZ_4K);
}
static const u32 region_to_mem_type[] = {
@@ -2423,7 +2431,7 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
u64 range_end = addr + range;
int err;
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
vm_dbg(&vm->xe->drm,
"op=%d, addr=0x%016llx, range=0x%016llx, bo_offset_or_userptr=0x%016llx",
@@ -2446,10 +2454,12 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
.map.gem.offset = bo_offset_or_userptr,
};
+ vops->flags |= XE_VMA_OPS_FLAG_MODIFIES_GPUVA;
ops = drm_gpuvm_sm_map_ops_create(&vm->gpuvm, &map_req);
break;
}
case DRM_XE_VM_BIND_OP_UNMAP:
+ vops->flags |= XE_VMA_OPS_FLAG_MODIFIES_GPUVA;
ops = drm_gpuvm_sm_unmap_ops_create(&vm->gpuvm, addr, range);
break;
case DRM_XE_VM_BIND_OP_PREFETCH:
@@ -2458,6 +2468,7 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
case DRM_XE_VM_BIND_OP_UNMAP_ALL:
xe_assert(vm->xe, bo);
+ vops->flags |= XE_VMA_OPS_FLAG_MODIFIES_GPUVA;
err = xe_bo_lock(bo, true);
if (err)
return ERR_PTR(err);
@@ -2479,6 +2490,16 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
if (IS_ERR(ops))
return ops;
+ /* Setup safe unwind */
+ drm_gpuva_for_each_op(__op, ops) {
+ struct xe_vma_op *op = gpuva_op_to_vma_op(__op);
+
+ if (__op->op == DRM_GPUVA_OP_PREFETCH) {
+ xa_init_flags(&op->prefetch_range.range, XA_FLAGS_ALLOC);
+ op->prefetch_range.ranges_count = 0;
+ }
+ }
+
drm_gpuva_for_each_op(__op, ops) {
struct xe_vma_op *op = gpuva_op_to_vma_op(__op);
@@ -2507,10 +2528,14 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
struct drm_pagemap *dpagemap = NULL;
u8 id, tile_mask = 0;
u32 i;
+ bool need_put, valid_pages;
+
+ if (xe_vma_is_userptr(vma))
+ vops->flags |= XE_VMA_OPS_FLAG_MODIFIES_GPUVA;
if (!xe_vma_is_cpu_addr_mirror(vma)) {
op->prefetch.region = prefetch_region;
- break;
+ continue;
}
ctx.read_only = xe_vma_read_only(vma);
@@ -2520,9 +2545,6 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
for_each_tile(tile, vm->xe, id)
tile_mask |= 0x1 << id;
- xa_init_flags(&op->prefetch_range.range, XA_FLAGS_ALLOC);
- op->prefetch_range.ranges_count = 0;
-
if (prefetch_region == DRM_XE_CONSULT_MEM_ADVISE_PREF_LOC) {
dpagemap = xe_vma_resolve_pagemap(vma,
xe_device_get_root_tile(vm->xe));
@@ -2534,6 +2556,7 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops,
op->prefetch_range.dpagemap = dpagemap;
alloc_next_range:
+ need_put = false;
svm_range = xe_svm_range_find_or_insert(vm, addr, vma, &ctx);
if (PTR_ERR(svm_range) == -ENOENT) {
@@ -2551,8 +2574,11 @@ alloc_next_range:
goto unwind_prefetch_ops;
}
- if (xe_svm_range_validate(vm, svm_range, tile_mask, dpagemap)) {
+ if (xe_svm_range_validate(vm, svm_range, tile_mask,
+ dpagemap, &valid_pages)) {
xe_svm_range_debug(svm_range, "PREFETCH - RANGE IS VALID");
+ xe_assert(vm->xe, valid_pages);
+ need_put = true;
goto check_next_range;
}
@@ -2560,18 +2586,27 @@ alloc_next_range:
&i, svm_range, xa_limit_32b,
GFP_KERNEL);
- if (err)
+ if (err) {
+ xe_svm_range_put(svm_range);
goto unwind_prefetch_ops;
+ }
op->prefetch_range.ranges_count++;
vops->flags |= XE_VMA_OPS_FLAG_HAS_SVM_PREFETCH;
+ if (valid_pages)
+ vops->flags |= XE_VMA_OPS_FLAG_HAS_SVM_VALID_RANGE;
xe_svm_range_debug(svm_range, "PREFETCH - RANGE CREATED");
check_next_range:
if (range_end > xe_svm_range_end(svm_range) &&
xe_svm_range_end(svm_range) < xe_vma_end(vma)) {
addr = xe_svm_range_end(svm_range);
+ if (need_put)
+ xe_svm_range_put(svm_range);
goto alloc_next_range;
}
+ if (need_put)
+ xe_svm_range_put(svm_range);
+
}
print_op_label:
print_op(vm->xe, __op);
@@ -2596,7 +2631,7 @@ static struct xe_vma *new_vma(struct xe_vm *vm, struct drm_gpuva_op_map *op,
struct xe_vma *vma;
int err = 0;
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
if (bo) {
err = 0;
@@ -2693,10 +2728,12 @@ static int xe_vma_op_commit(struct xe_vm *vm, struct xe_vma_op *op)
{
int err = 0;
- lockdep_assert_held_write(&vm->lock);
+ lockdep_assert_held(&vm->lock);
switch (op->base.op) {
case DRM_GPUVA_OP_MAP:
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
+
err |= xe_vm_insert_vma(vm, op->map.vma);
if (!err)
op->flags |= XE_VMA_OP_COMMITTED;
@@ -2706,6 +2743,8 @@ static int xe_vma_op_commit(struct xe_vm *vm, struct xe_vma_op *op)
u8 tile_present =
gpuva_to_vma(op->base.remap.unmap->va)->tile_present;
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
+
prep_vma_destroy(vm, gpuva_to_vma(op->base.remap.unmap->va),
true);
op->flags |= XE_VMA_OP_COMMITTED;
@@ -2740,6 +2779,8 @@ static int xe_vma_op_commit(struct xe_vm *vm, struct xe_vma_op *op)
break;
}
case DRM_GPUVA_OP_UNMAP:
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
+
prep_vma_destroy(vm, gpuva_to_vma(op->base.unmap.va), true);
op->flags |= XE_VMA_OP_COMMITTED;
break;
@@ -2785,7 +2826,7 @@ static int vm_bind_ioctl_ops_parse(struct xe_vm *vm, struct drm_gpuva_ops *ops,
u8 id, tile_mask = 0;
int err = 0;
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
for_each_tile(tile, vm->xe, id)
tile_mask |= 0x1 << id;
@@ -2964,10 +3005,12 @@ static void xe_vma_op_unwind(struct xe_vm *vm, struct xe_vma_op *op,
bool post_commit, bool prev_post_commit,
bool next_post_commit)
{
- lockdep_assert_held_write(&vm->lock);
+ lockdep_assert_held(&vm->lock);
switch (op->base.op) {
case DRM_GPUVA_OP_MAP:
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
+
if (op->map.vma) {
prep_vma_destroy(vm, op->map.vma, post_commit);
xe_vma_destroy_unlocked(op->map.vma);
@@ -2977,6 +3020,8 @@ static void xe_vma_op_unwind(struct xe_vm *vm, struct xe_vma_op *op,
{
struct xe_vma *vma = gpuva_to_vma(op->base.unmap.va);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
+
if (vma) {
xe_svm_notifier_lock(vm);
vma->gpuva.flags &= ~XE_VMA_DESTROYED;
@@ -2990,6 +3035,8 @@ static void xe_vma_op_unwind(struct xe_vm *vm, struct xe_vma_op *op,
{
struct xe_vma *vma = gpuva_to_vma(op->base.remap.unmap->va);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
+
if (op->remap.prev) {
prep_vma_destroy(vm, op->remap.prev, prev_post_commit);
xe_vma_destroy_unlocked(op->remap.prev);
@@ -3118,16 +3165,87 @@ static int check_ufence(struct xe_vma *vma)
return 0;
}
-static int prefetch_ranges(struct xe_vm *vm, struct xe_vma_op *op)
+struct prefetch_thread {
+ struct work_struct work;
+ struct drm_gpusvm_ctx *ctx;
+ struct xe_vma *vma;
+ struct xe_svm_range *svm_range;
+ struct drm_pagemap *dpagemap;
+ int err;
+};
+
+static void prefetch_thread_func(struct prefetch_thread *thread)
+{
+ struct xe_vma *vma = thread->vma;
+ struct xe_vm *vm = xe_vma_vm(vma);
+ struct xe_svm_range *svm_range = thread->svm_range;
+ struct drm_pagemap *dpagemap = thread->dpagemap;
+ int err = 0;
+
+ guard(mutex)(&svm_range->lock);
+
+ if (xe_svm_range_is_removed(svm_range))
+ return;
+
+ if (!dpagemap)
+ xe_svm_range_migrate_to_smem(vm, svm_range);
+
+ if (IS_ENABLED(CONFIG_DRM_XE_DEBUG_VM)) {
+ drm_dbg(&vm->xe->drm,
+ "Prefetch pagemap is %s start 0x%016lx end 0x%016lx\n",
+ dpagemap ? dpagemap->drm->unique : "system",
+ xe_svm_range_start(svm_range), xe_svm_range_end(svm_range));
+ }
+
+ if (xe_svm_range_needs_migrate_to_vram(svm_range, vma, dpagemap)) {
+ err = xe_svm_alloc_vram(svm_range, thread->ctx, dpagemap);
+ if (err) {
+ drm_dbg(&vm->xe->drm, "VRAM allocation failed, retry from userspace, asid=%u, gpusvm=%p, errno=%pe\n",
+ vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err));
+ /*
+ * We intentionally return -ENODATA on any races to
+ * commit any VMA updates from other ops without
+ * updating any page tables deferring to page faults to
+ * page updates skipped in the IOCTL.
+ */
+ thread->err = -ENODATA;
+ return;
+ }
+ xe_svm_range_debug(svm_range, "PREFETCH - RANGE MIGRATED TO VRAM");
+ }
+
+ err = xe_svm_range_get_pages(vm, svm_range, thread->ctx);
+ if (err) {
+ drm_dbg(&vm->xe->drm, "Get pages failed, asid=%u, gpusvm=%p, errno=%pe\n",
+ vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err));
+ if (err == -EOPNOTSUPP || err == -EFAULT || err == -EPERM)
+ err = -ENODATA;
+ thread->err = err;
+ return;
+ }
+ xe_svm_range_debug(svm_range, "PREFETCH - RANGE GET PAGES DONE");
+}
+
+static void prefetch_work_func(struct work_struct *w)
+{
+ struct prefetch_thread *thread =
+ container_of(w, struct prefetch_thread, work);
+
+ prefetch_thread_func(thread);
+}
+
+static int prefetch_ranges(struct xe_vm *vm, struct xe_vma_ops *vops,
+ struct xe_vma_op *op)
{
bool devmem_possible = IS_DGFX(vm->xe) && IS_ENABLED(CONFIG_DRM_XE_PAGEMAP);
struct xe_vma *vma = gpuva_to_vma(op->base.prefetch.va);
struct drm_pagemap *dpagemap = op->prefetch_range.dpagemap;
- int err = 0;
-
struct xe_svm_range *svm_range;
struct drm_gpusvm_ctx ctx = {};
+ struct prefetch_thread stack_thread, *thread, *prefetches;
unsigned long i;
+ int err = 0, idx = 0;
+ bool skip_threads;
if (!xe_vma_is_cpu_addr_mirror(vma))
return 0;
@@ -3137,37 +3255,50 @@ static int prefetch_ranges(struct xe_vm *vm, struct xe_vma_op *op)
ctx.check_pages_threshold = devmem_possible ? SZ_64K : 0;
ctx.device_private_page_owner = xe_svm_private_page_owner(vm, !dpagemap);
- /* TODO: Threading the migration */
+ skip_threads = op->prefetch_range.ranges_count == 1 ||
+ (!dpagemap && !(vops->flags &
+ XE_VMA_OPS_FLAG_HAS_SVM_VALID_RANGE)) ||
+ !(vops->flags & XE_VMA_OPS_FLAG_DOWNGRADE_LOCK) ||
+ vm->xe->info.num_pf_work == 1;
+ thread = skip_threads ? &stack_thread : NULL;
+
+ if (!skip_threads) {
+ prefetches = kvmalloc_array(op->prefetch_range.ranges_count,
+ sizeof(*prefetches), GFP_KERNEL);
+ if (!prefetches)
+ return -ENOMEM;
+ }
+
xa_for_each(&op->prefetch_range.range, i, svm_range) {
- if (!dpagemap)
- xe_svm_range_migrate_to_smem(vm, svm_range);
-
- if (IS_ENABLED(CONFIG_DRM_XE_DEBUG_VM)) {
- drm_dbg(&vm->xe->drm,
- "Prefetch pagemap is %s start 0x%016lx end 0x%016lx\n",
- dpagemap ? dpagemap->drm->unique : "system",
- xe_svm_range_start(svm_range), xe_svm_range_end(svm_range));
+ if (!skip_threads) {
+ thread = prefetches + idx++;
+ INIT_WORK(&thread->work, prefetch_work_func);
}
- if (xe_svm_range_needs_migrate_to_vram(svm_range, vma, dpagemap)) {
- err = xe_svm_alloc_vram(svm_range, &ctx, dpagemap);
- if (err) {
- drm_dbg(&vm->xe->drm, "VRAM allocation failed, retry from userspace, asid=%u, gpusvm=%p, errno=%pe\n",
- vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err));
- return -ENODATA;
- }
- xe_svm_range_debug(svm_range, "PREFETCH - RANGE MIGRATED TO VRAM");
+ thread->ctx = &ctx;
+ thread->vma = vma;
+ thread->svm_range = svm_range;
+ thread->dpagemap = dpagemap;
+ thread->err = 0;
+
+ if (skip_threads) {
+ prefetch_thread_func(thread);
+ if (thread->err)
+ return thread->err;
+ } else {
+ queue_work(vm->xe->usm.prefetch_wq, &thread->work);
}
+ }
- err = xe_svm_range_get_pages(vm, svm_range, &ctx);
- if (err) {
- drm_dbg(&vm->xe->drm, "Get pages failed, asid=%u, gpusvm=%p, errno=%pe\n",
- vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err));
- if (err == -EOPNOTSUPP || err == -EFAULT || err == -EPERM)
- err = -ENODATA;
- return err;
+ if (!skip_threads) {
+ for (i = 0; i < idx; ++i) {
+ thread = prefetches + i;
+
+ flush_work(&thread->work);
+ if (thread->err && !err)
+ err = thread->err;
}
- xe_svm_range_debug(svm_range, "PREFETCH - RANGE GET PAGES DONE");
+ kvfree(prefetches);
}
return err;
@@ -3298,7 +3429,8 @@ static int op_lock_and_prep(struct drm_exec *exec, struct xe_vm *vm,
return err;
}
-static int vm_bind_ioctl_ops_prefetch_ranges(struct xe_vm *vm, struct xe_vma_ops *vops)
+static int vm_bind_ioctl_ops_prefetch_ranges(struct xe_vm *vm,
+ struct xe_vma_ops *vops)
{
struct xe_vma_op *op;
int err;
@@ -3308,7 +3440,7 @@ static int vm_bind_ioctl_ops_prefetch_ranges(struct xe_vm *vm, struct xe_vma_ops
list_for_each_entry(op, &vops->list, link) {
if (op->base.op == DRM_GPUVA_OP_PREFETCH) {
- err = prefetch_ranges(vm, op);
+ err = prefetch_ranges(vm, vops, op);
if (err)
return err;
}
@@ -3569,7 +3701,7 @@ static struct dma_fence *vm_bind_ioctl_ops_execute(struct xe_vm *vm,
struct dma_fence *fence;
int err = 0;
- lockdep_assert_held_write(&vm->lock);
+ lockdep_assert_held(&vm->lock);
xe_validation_guard(&ctx, &vm->xe->val, &exec,
((struct xe_val_flags) {
@@ -3892,7 +4024,7 @@ int xe_vm_bind_ioctl(struct drm_device *dev, void *data, struct drm_file *file)
u32 num_syncs, num_ufence = 0;
struct xe_sync_entry *syncs = NULL;
struct drm_xe_vm_bind_op *bind_ops = NULL;
- struct xe_vma_ops vops;
+ struct xe_vma_ops vops = { .flags = 0, };
struct dma_fence *fence;
int err;
int i;
@@ -4067,6 +4199,11 @@ int xe_vm_bind_ioctl(struct drm_device *dev, void *data, struct drm_file *file)
goto unwind_ops;
}
+ if (!(vops.flags & XE_VMA_OPS_FLAG_MODIFIES_GPUVA)) {
+ vops.flags |= XE_VMA_OPS_FLAG_DOWNGRADE_LOCK;
+ downgrade_write(&vm->lock);
+ }
+
err = xe_vma_ops_alloc(&vops, args->num_binds > 1);
if (err)
goto unwind_ops;
@@ -4103,7 +4240,10 @@ put_obj:
free_bos:
kvfree(bos);
release_vm_lock:
- up_write(&vm->lock);
+ if (vops.flags & XE_VMA_OPS_FLAG_DOWNGRADE_LOCK)
+ up_read(&vm->lock);
+ else
+ up_write(&vm->lock);
put_exec_queue:
if (q)
xe_exec_queue_put(q);
@@ -4735,7 +4875,7 @@ static int xe_vm_alloc_vma(struct xe_vm *vm,
u16 default_pat;
int err;
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
if (is_madvise)
ops = drm_gpuvm_madvise_ops_create(&vm->gpuvm, map_req);
@@ -4869,7 +5009,7 @@ int xe_vm_alloc_madvise_vma(struct xe_vm *vm, uint64_t start, uint64_t range)
.map.va.range = range,
};
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
vm_dbg(&vm->xe->drm, "MADVISE_OPS_CREATE: addr=0x%016llx, size=0x%016llx", start, range);
@@ -4901,7 +5041,7 @@ void xe_vm_find_cpu_addr_mirror_vma_range(struct xe_vm *vm, u64 *start, u64 *end
{
struct xe_vma *prev, *next;
- lockdep_assert_held(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
if (*start >= SZ_4K) {
prev = xe_vm_find_vma_by_addr(vm, *start - SZ_4K);
@@ -4933,7 +5073,7 @@ int xe_vm_alloc_cpu_addr_mirror_vma(struct xe_vm *vm, uint64_t start, uint64_t r
.map.va.range = range,
};
- lockdep_assert_held_write(&vm->lock);
+ xe_vm_assert_write_mode_or_garbage_collector(vm);
vm_dbg(&vm->xe->drm, "CPU_ADDR_MIRROR_VMA_OPS_CREATE: addr=0x%016llx, size=0x%016llx",
start, range);
@@ -4955,7 +5095,6 @@ void xe_vm_add_exec_queue(struct xe_vm *vm, struct xe_exec_queue *q)
/* User VMs and queues only */
xe_assert(xe, !(q->flags & EXEC_QUEUE_FLAG_KERNEL));
- xe_assert(xe, !(q->flags & EXEC_QUEUE_FLAG_PERMANENT));
xe_assert(xe, !(q->flags & EXEC_QUEUE_FLAG_VM));
xe_assert(xe, !(q->flags & EXEC_QUEUE_FLAG_MIGRATE));
xe_assert(xe, vm->xef);
diff --git a/drivers/gpu/drm/xe/xe_vm_types.h b/drivers/gpu/drm/xe/xe_vm_types.h
index 635ed29b9a69..68588b624212 100644
--- a/drivers/gpu/drm/xe/xe_vm_types.h
+++ b/drivers/gpu/drm/xe/xe_vm_types.h
@@ -133,6 +133,12 @@ struct xe_vma {
};
/**
+ * @fault_lock: Synchronizes fault processing. Locking order: inside
+ * vm->lock, outside dma-resv.
+ */
+ struct mutex fault_lock;
+
+ /**
* @tile_invalidated: Tile mask of binding are invalidated for this VMA.
* protected by BO's resv and for userptrs, vm->svm.gpusvm.notifier_lock in
* write mode for writing or vm->svm.gpusvm.notifier_lock in read mode and
@@ -215,12 +221,26 @@ struct xe_vm {
/** @svm.gpusvm: base GPUSVM used to track fault allocations */
struct drm_gpusvm gpusvm;
/**
+ * @svm.range_lock: Protects insertion and removal of ranges
+ * from GPU SVM tree.
+ */
+ struct mutex range_lock;
+ /**
* @svm.garbage_collector: Garbage collector which is used unmap
* SVM range's GPU bindings and destroy the ranges.
*/
struct {
- /** @svm.garbage_collector.lock: Protect's range list */
- spinlock_t lock;
+ /**
+ * @svm.garbage_collector.lock: Ensures only one thread
+ * runs the garbage collector at a time. Locking order:
+ * inside vm->lock, outside range->lock and dma-resv.
+ */
+ struct mutex lock;
+ /**
+ * @svm.garbage_collector.list_lock: Protect's range
+ * list
+ */
+ spinlock_t list_lock;
/**
* @svm.garbage_collector.range_list: List of SVM ranges
* in the garbage collector.
@@ -350,11 +370,6 @@ struct xe_vm {
struct {
/** @asid: address space ID, unique to each VM */
u32 asid;
- /**
- * @last_fault_vma: Last fault VMA, used for fast lookup when we
- * get a flood of faults to the same VMA
- */
- struct xe_vma *last_fault_vma;
} usm;
/** @error_capture: allow to track errors */
@@ -541,11 +556,14 @@ struct xe_vma_ops {
/** @pt_update_ops: page table update operations */
struct xe_vm_pgtable_update_ops pt_update_ops[XE_MAX_TILES_PER_DEVICE];
/** @flag: signify the properties within xe_vma_ops*/
-#define XE_VMA_OPS_FLAG_HAS_SVM_PREFETCH BIT(0)
-#define XE_VMA_OPS_FLAG_MADVISE BIT(1)
-#define XE_VMA_OPS_ARRAY_OF_BINDS BIT(2)
-#define XE_VMA_OPS_FLAG_SKIP_TLB_WAIT BIT(3)
-#define XE_VMA_OPS_FLAG_ALLOW_SVM_UNMAP BIT(4)
+#define XE_VMA_OPS_FLAG_HAS_SVM_PREFETCH BIT(0)
+#define XE_VMA_OPS_FLAG_MADVISE BIT(1)
+#define XE_VMA_OPS_ARRAY_OF_BINDS BIT(2)
+#define XE_VMA_OPS_FLAG_SKIP_TLB_WAIT BIT(3)
+#define XE_VMA_OPS_FLAG_ALLOW_SVM_UNMAP BIT(4)
+#define XE_VMA_OPS_FLAG_MODIFIES_GPUVA BIT(5)
+#define XE_VMA_OPS_FLAG_DOWNGRADE_LOCK BIT(6)
+#define XE_VMA_OPS_FLAG_HAS_SVM_VALID_RANGE BIT(7)
u32 flags;
#ifdef TEST_VM_OPS_ERROR
/** @inject_error: inject error to test error handling */
diff --git a/drivers/gpu/drm/xe/xe_wa_oob.rules b/drivers/gpu/drm/xe/xe_wa_oob.rules
index f02ac9bf7424..dd69ad07f7a9 100644
--- a/drivers/gpu/drm/xe/xe_wa_oob.rules
+++ b/drivers/gpu/drm/xe/xe_wa_oob.rules
@@ -63,7 +63,7 @@
16026007364 MEDIA_VERSION(3000)
14020316580 MEDIA_VERSION(1301)
-14025883347 MEDIA_VERSION_RANGE(1301, 3503)
+14025883347 MEDIA_VERSION_RANGE(1301, 3500)
GRAPHICS_VERSION_RANGE(2004, 3005)
16029380221 MEDIA_VERSION(3500)
22022079272 MEDIA_VERSION(3503)
diff --git a/include/drm/drm_device.h b/include/drm/drm_device.h
index 768a8dae83c5..75f030d027ee 100644
--- a/include/drm/drm_device.h
+++ b/include/drm/drm_device.h
@@ -37,6 +37,7 @@ struct pci_controller;
#define DRM_WEDGE_RECOVERY_REBIND BIT(1) /* unbind + bind driver */
#define DRM_WEDGE_RECOVERY_BUS_RESET BIT(2) /* unbind + reset bus device + bind */
#define DRM_WEDGE_RECOVERY_VENDOR BIT(3) /* vendor specific recovery method */
+#define DRM_WEDGE_RECOVERY_COLD_RESET BIT(4) /* remove device + slot power cycle + rescan */
/**
* struct drm_wedge_task_info - information about the guilty task of a wedge dev
diff --git a/include/drm/drm_ras.h b/include/drm/drm_ras.h
index 0beede3ddc4e..ee2caa0edc6f 100644
--- a/include/drm/drm_ras.h
+++ b/include/drm/drm_ras.h
@@ -80,9 +80,14 @@ struct drm_device;
#if IS_ENABLED(CONFIG_DRM_RAS)
int drm_ras_node_register(struct drm_ras_node *node);
void drm_ras_node_unregister(struct drm_ras_node *node);
+int drm_ras_nl_error_event(struct drm_ras_node *node, u32 error_id, const char *error_name,
+ u32 value);
#else
static inline int drm_ras_node_register(struct drm_ras_node *node) { return 0; }
static inline void drm_ras_node_unregister(struct drm_ras_node *node) { }
+static inline int drm_ras_nl_error_event(struct drm_ras_node *node, u32 error_id,
+ const char *error_name, u32 value)
+{ return 0; }
#endif
#endif
diff --git a/include/uapi/drm/drm_ras.h b/include/uapi/drm/drm_ras.h
index 218a3ee86805..eab8231aa87c 100644
--- a/include/uapi/drm/drm_ras.h
+++ b/include/uapi/drm/drm_ras.h
@@ -39,12 +39,27 @@ enum {
};
enum {
+ DRM_RAS_A_ERROR_EVENT_ATTRS_DEVICE_NAME = 1,
+ DRM_RAS_A_ERROR_EVENT_ATTRS_NODE_ID,
+ DRM_RAS_A_ERROR_EVENT_ATTRS_NODE_NAME,
+ DRM_RAS_A_ERROR_EVENT_ATTRS_ERROR_ID,
+ DRM_RAS_A_ERROR_EVENT_ATTRS_ERROR_NAME,
+ DRM_RAS_A_ERROR_EVENT_ATTRS_ERROR_VALUE,
+
+ __DRM_RAS_A_ERROR_EVENT_ATTRS_MAX,
+ DRM_RAS_A_ERROR_EVENT_ATTRS_MAX = (__DRM_RAS_A_ERROR_EVENT_ATTRS_MAX - 1)
+};
+
+enum {
DRM_RAS_CMD_LIST_NODES = 1,
DRM_RAS_CMD_GET_ERROR_COUNTER,
DRM_RAS_CMD_CLEAR_ERROR_COUNTER,
+ DRM_RAS_CMD_ERROR_EVENT,
__DRM_RAS_CMD_MAX,
DRM_RAS_CMD_MAX = (__DRM_RAS_CMD_MAX - 1)
};
+#define DRM_RAS_MCGRP_ERROR_REPORT "error-report"
+
#endif /* _UAPI_LINUX_DRM_RAS_H */