diff options
Diffstat (limited to 'drivers/gpu')
72 files changed, 4042 insertions, 536 deletions
diff --git a/drivers/gpu/drm/drm_drv.c b/drivers/gpu/drm/drm_drv.c index c808958a2188..cb53baa70995 100644 --- a/drivers/gpu/drm/drm_drv.c +++ b/drivers/gpu/drm/drm_drv.c @@ -545,6 +545,8 @@ static const char *drm_get_wedge_recovery(unsigned int opt) return "bus-reset"; case DRM_WEDGE_RECOVERY_VENDOR: return "vendor-specific"; + case DRM_WEDGE_RECOVERY_COLD_RESET: + return "cold-reset"; default: return NULL; } diff --git a/drivers/gpu/drm/drm_ras.c b/drivers/gpu/drm/drm_ras.c index d6eab29a1394..39155fb514de 100644 --- a/drivers/gpu/drm/drm_ras.c +++ b/drivers/gpu/drm/drm_ras.c @@ -41,6 +41,11 @@ * Userspace must provide Node ID, Error ID. * Clears specific error counter of a node if supported. * + * 4. ERROR_REPORT: Subscribe to this multicast group to receive error events + * + * 5. ERROR_EVENT: Report an error event to userspace. The event contains device, node + * and error information that triggered the event. + * * Node registration: * * - drm_ras_node_register(): Registers a new node and assigns @@ -186,6 +191,34 @@ static int msg_reply_value(struct sk_buff *msg, u32 error_id, value); } +static int msg_put_error_event_attrs(struct sk_buff *msg, struct drm_ras_node *node, + u32 error_id, const char *error_name, u32 value) +{ + int ret; + + ret = nla_put_string(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_DEVICE_NAME, node->device_name); + if (ret) + return ret; + + ret = nla_put_u32(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_NODE_ID, node->id); + if (ret) + return ret; + + ret = nla_put_string(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_NODE_NAME, node->node_name); + if (ret) + return ret; + + ret = nla_put_u32(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_ERROR_ID, error_id); + if (ret) + return ret; + + ret = nla_put_string(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_ERROR_NAME, error_name); + if (ret) + return ret; + + return nla_put_u32(msg, DRM_RAS_A_ERROR_EVENT_ATTRS_ERROR_VALUE, value); +} + static int doit_reply_value(struct genl_info *info, u32 node_id, u32 error_id) { @@ -223,6 +256,74 @@ static int doit_reply_value(struct genl_info *info, u32 node_id, } /** + * drm_ras_nl_error_event() - Report an error event + * @node: Node structure + * @error_id: ID of the error + * @error_name: Name of the error + * @value: Value of the error counter + * + * Report an error-event to userspace using the error-report multicast group. + * + * Context: Process context only. Uses %GFP_KERNEL and multicasts to all + * netns, which is unbounded work; callers in interrupt handlers + * or other atomic context must defer event. + * + * Return: 0 on success, or negative errno on failure. + */ +int drm_ras_nl_error_event(struct drm_ras_node *node, u32 error_id, const char *error_name, + u32 value) +{ + struct genl_info info; + struct sk_buff *msg; + struct nlattr *hdr; + int ret; + + if (!node || !error_name) + return -EINVAL; + + /* Check the node is currently registered */ + if (xa_load(&drm_ras_xa, node->id) != node) + return -ENOENT; + + /* Currently only Error Counter events are supported */ + if (node->type != DRM_RAS_NODE_TYPE_ERROR_COUNTER) + return -EOPNOTSUPP; + + /* Check the error ID is within the valid range */ + if (error_id < node->error_counter_range.first || + error_id > node->error_counter_range.last) + return -EINVAL; + + genl_info_init_ntf(&info, &drm_ras_nl_family, DRM_RAS_CMD_ERROR_EVENT); + + msg = genlmsg_new(NLMSG_GOODSIZE, GFP_KERNEL); + if (!msg) + return -ENOMEM; + + hdr = genlmsg_iput(msg, &info); + if (!hdr) { + ret = -EMSGSIZE; + goto free_msg; + } + + ret = msg_put_error_event_attrs(msg, node, error_id, error_name, value); + if (ret) + goto cancel_msg; + + genlmsg_end(msg, hdr); + genlmsg_multicast_allns(&drm_ras_nl_family, msg, 0, DRM_RAS_NLGRP_ERROR_REPORT); + + return 0; + +cancel_msg: + genlmsg_cancel(msg, hdr); +free_msg: + nlmsg_free(msg); + return ret; +} +EXPORT_SYMBOL(drm_ras_nl_error_event); + +/** * drm_ras_nl_get_error_counter_dumpit() - Dump all Error Counters * @skb: Netlink message buffer * @cb: Callback context for multi-part dumps diff --git a/drivers/gpu/drm/drm_ras_nl.c b/drivers/gpu/drm/drm_ras_nl.c index dea1c1b2494e..9d3123cc9f9c 100644 --- a/drivers/gpu/drm/drm_ras_nl.c +++ b/drivers/gpu/drm/drm_ras_nl.c @@ -58,6 +58,10 @@ static const struct genl_split_ops drm_ras_nl_ops[] = { }, }; +static const struct genl_multicast_group drm_ras_nl_mcgrps[] = { + [DRM_RAS_NLGRP_ERROR_REPORT] = { "error-report", }, +}; + struct genl_family drm_ras_nl_family __ro_after_init = { .name = DRM_RAS_FAMILY_NAME, .version = DRM_RAS_FAMILY_VERSION, @@ -66,4 +70,6 @@ struct genl_family drm_ras_nl_family __ro_after_init = { .module = THIS_MODULE, .split_ops = drm_ras_nl_ops, .n_split_ops = ARRAY_SIZE(drm_ras_nl_ops), + .mcgrps = drm_ras_nl_mcgrps, + .n_mcgrps = ARRAY_SIZE(drm_ras_nl_mcgrps), }; diff --git a/drivers/gpu/drm/drm_ras_nl.h b/drivers/gpu/drm/drm_ras_nl.h index a398643572a5..03ec275aca92 100644 --- a/drivers/gpu/drm/drm_ras_nl.h +++ b/drivers/gpu/drm/drm_ras_nl.h @@ -21,6 +21,10 @@ int drm_ras_nl_get_error_counter_dumpit(struct sk_buff *skb, int drm_ras_nl_clear_error_counter_doit(struct sk_buff *skb, struct genl_info *info); +enum { + DRM_RAS_NLGRP_ERROR_REPORT, +}; + extern struct genl_family drm_ras_nl_family; #endif /* _LINUX_DRM_RAS_GEN_H */ diff --git a/drivers/gpu/drm/xe/Makefile b/drivers/gpu/drm/xe/Makefile index 67ada1d6c2fb..7ac3954737f9 100644 --- a/drivers/gpu/drm/xe/Makefile +++ b/drivers/gpu/drm/xe/Makefile @@ -87,6 +87,7 @@ xe-y += xe_bb.o \ xe_hw_fence.o \ xe_irq.o \ xe_late_bind_fw.o \ + xe_log.o \ xe_lrc.o \ xe_mem_pool.o \ xe_migrate.o \ diff --git a/drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h b/drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h index ce5c59517528..57baa68bcba4 100644 --- a/drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h +++ b/drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h @@ -119,6 +119,7 @@ enum slpc_param_id { SLPC_PARAM_STRATEGIES = 26, SLPC_PARAM_POWER_PROFILE = 27, SLPC_PARAM_IGNORE_EFFICIENT_FREQUENCY = 28, + SLPC_PARAM_SET_IBC_VERSION = 29, SLPC_MAX_PARAM = 32, }; diff --git a/drivers/gpu/drm/xe/abi/xe_log_abi.h b/drivers/gpu/drm/xe/abi/xe_log_abi.h new file mode 100644 index 000000000000..d6105520173e --- /dev/null +++ b/drivers/gpu/drm/xe/abi/xe_log_abi.h @@ -0,0 +1,199 @@ +/* SPDX-License-Identifier: MIT */ +/* + * Copyright © 2026 Intel Corporation + */ + +#ifndef _ABI_XE_LOG_ABI_H_ +#define _ABI_XE_LOG_ABI_H_ + +#include <linux/bits.h> +#include <linux/bitfield.h> + +#include "abi/xe_sigid_abi.h" + +/** + * enum xe_log_component_bits - bits for components structure definitions + * + * Component identifiers are structured based on:: + * + * COMPONENT = CLASS(8b).TYPE(8b) + * + * and the structure looks like this:: + * + * ├── SYSTEM(0) + * │ └── ... + * ├── DRIVER(1) + * │ └── ... + * ├── FEATURE(2) + * │ └── ... + * ├── FIRMWARE(4) + * │ └── ... + * └── HARDWARE(8) + * └── ... + * + * Examples:: + * + * COMPONENT(0.type) = SYSTEM.type = system component + * COMPONENT(1.type) = DRIVER.type = driver core component + * COMPONENT(3.type) = DRIVER_FEATURE.type = driver feature + * COMPONENT(5.type) = DRIVER_FIRMWARE.type = firmware driver component + * COMPONENT(9.type) = DRIVER_HARDWARE.type = hardware driver component + * + */ +enum xe_log_component_bits { + /* private: */ + XE_LOG_COMPONENT_CLASS_MASK = GENMASK_U16(7, 0), + XE_LOG_COMPONENT_TYPE_MASK = GENMASK_U16(15, 8), + /* private: component classes */ + XE_LOG_COMPONENT_CLASS_SYSTEM = 0u, + XE_LOG_COMPONENT_CLASS_DRIVER = 1u, + XE_LOG_COMPONENT_CLASS_FEATURE = 2u, + XE_LOG_COMPONENT_CLASS_FIRMWARE = 4u, + XE_LOG_COMPONENT_CLASS_HARDWARE = 8u, + XE_LOG_COMPONENT_CLASS_DRIVER_FEATURE = XE_LOG_COMPONENT_CLASS_DRIVER | + XE_LOG_COMPONENT_CLASS_FEATURE, + XE_LOG_COMPONENT_CLASS_DRIVER_FIRMWARE = XE_LOG_COMPONENT_CLASS_DRIVER | + XE_LOG_COMPONENT_CLASS_FIRMWARE, + XE_LOG_COMPONENT_CLASS_DRIVER_HARDWARE = XE_LOG_COMPONENT_CLASS_DRIVER | + XE_LOG_COMPONENT_CLASS_HARDWARE, + /* private: reserved identifiers */ + XE_LOG_COMPONENT_NONE = 0u, +}; + +#define MAKE_XE_LOG_COMPONENT(_CLASS, type) \ + (FIELD_PREP_CONST(XE_LOG_COMPONENT_CLASS_MASK, \ + XE_LOG_COMPONENT_CLASS_##_CLASS) | \ + FIELD_PREP_CONST(XE_LOG_COMPONENT_TYPE_MASK, (type))) + +/** + * enum xe_log_location_bits - bits for location structure definitions + * + * Location identifiers are structured based on:: + * + * LOCATION = TYPE(8b).ID(8b) + * + * and the structure looks like this:: + * + * ├── DEVICE(0) + * │ └── MBZ(0) + * ├── TILE(1) + * │ ├── Tile0(0) + * │ ├── ... + * │ └── TileN(n) + * ├── GT(1) + * │ ├── GT0(0) + * │ ├── ... + * │ └── GTn(n) + * └── ... + * + * Examples:: + * + * LOCATION(0.0) = NONE + * LOCATION(1.0) = DEVICE.0 = "Device" + * LOCATION(2.1) = TILE.1 = "Tile1" + * LOCATION(3.2) = GT.2 = "GT2" + * + */ +enum xe_log_location_bits { + /* private: */ + XE_LOG_LOCATION_TYPE_MASK = GENMASK_U16(7, 0), + XE_LOG_LOCATION_ID_MASK = GENMASK_U16(15, 8), + /* private: location types */ + XE_LOG_LOCATION_TYPE_DEVICE = 1u, + XE_LOG_LOCATION_TYPE_TILE = 2u, + XE_LOG_LOCATION_TYPE_GT = 3u, + /* private: reserved identifiers */ + XE_LOG_LOCATION_NONE = 0u, +}; + +#define PREP_XE_LOG_LOCATION(type, id) \ + (FIELD_PREP(XE_LOG_LOCATION_TYPE_MASK, (type)) | \ + FIELD_PREP(XE_LOG_LOCATION_ID_MASK, (id))) + +#define MAKE_XE_LOG_LOCATION(_TYPE, id) \ + PREP_XE_LOG_LOCATION(XE_LOG_LOCATION_TYPE_##_TYPE, (id)) + +/** + * DEFINE_XE_LOG_COMPONENTS() - Define log components. + * @define: name of the inner macro to expand. + * + * Use this super macro to define custom code for the log components. + * The following parameters are available for each component:: + * + * define(CLASS, ID, TAG, SIGID, NAME) + * + * where: + * + * @CLASS is the component class name (without the XE_LOG_COMPONENT_CLASS_ prefix) + * @ID is the unique component identifier within @CLASS + * @TAG is unique component tag (across all components) + * @SIGID is the default xe_sigid for the component (without the XE_SIGID_ prefix) + */ +#define DEFINE_XE_LOG_COMPONENTS(define) \ + DEFINE_XE_LOG_SOFTWARE_COMPONENTS(define) \ + DEFINE_XE_LOG_HARDWARE_COMPONENTS(define) + +#define DEFINE_XE_LOG_SOFTWARE_COMPONENTS(define) \ + /* */ \ + define(SYSTEM, 1, PCI, IO_BUS, "Linux PCI Subsystem") \ + define(SYSTEM, 2, DRM, SW, "DRM") \ + /* */ \ + define(DRIVER, 1, XE, SW, "Xe Driver") \ + define(DRIVER, 2, PROBE, PROBE, "Driver Initialization") \ + define(DRIVER, 3, WEDGED, WEDGED, "Device Malfunction") \ + define(DRIVER, 4, RTP, SW, "Register Table Processing") \ + define(DRIVER, 5, WA, SW, "Workarounds") \ + define(DRIVER, 6, PAGEFAULT, MEM_FAULT, "Page Fault") \ + /* */ \ + define(DRIVER_HARDWARE, 1, REGS, IO_BUS, "Registers") \ + define(DRIVER_HARDWARE, 2, GGTT, IO_BUS, "Global GTT") \ + define(DRIVER_HARDWARE, 3, GT, GT_TDR, "Graphics Technology") \ + define(DRIVER_HARDWARE, 4, LMTT, IO_BUS, "LMEM Translation Table") \ + define(DRIVER_HARDWARE, 5, MEMIRQ, IO_BUS, "Memory Based IRQ") \ + /* */ \ + define(DRIVER_FEATURE, 1, PF, SW, "SR-IOV Physical Function") \ + define(DRIVER_FEATURE, 2, VF, SW, "SR-IOV Virtual Function") \ + define(DRIVER_FEATURE, 3, SURVIVABILITY, SURVIVABILITY, "Survivability") \ + define(DRIVER_FEATURE, 4, RAS, SW, "Reliability, Accessibility, Serviceability") \ + /* */ \ + define(DRIVER_FIRMWARE, 1, GUC, RUNTIME_FW, "GuC") \ + define(DRIVER_FIRMWARE, 2, HUC, RUNTIME_FW, "HuC") \ + define(DRIVER_FIRMWARE, 3, GSC, RUNTIME_FW, "GSC") \ + define(DRIVER_FIRMWARE, 16, PCODE, DEVICE_FW, "PCode") \ + define(DRIVER_FIRMWARE, 17, SYSCTRL, DEVICE_FW, "System Controller") \ + +#define DEFINE_XE_LOG_HARDWARE_COMPONENTS(define) \ + define(HARDWARE, 1, DEVICE_MEMORY, DEVICE_MEMORY, "Device Memory") \ + define(HARDWARE, 2, CORE_COMPUTE, CORE_COMPUTE, "Core Compute") \ + /* HARDWARE, 3, RESERVED */ \ + define(HARDWARE, 4, PCIE, PCIE, "PCIe Interface") \ + define(HARDWARE, 5, FABRIC, FABRIC, "Fabric") \ + define(HARDWARE, 6, SOC_INTERNAL, SOC_INTERNAL, "SoC Internal") \ + /* eod */ + +/** + * enum xe_log_component_tags - TAGs of all supported components + */ +enum xe_log_component_tags { + /* private: */ +#define MAKE_XE_LOG_COMPONENT_ENUM(_CLASS, _ID, _TAG, _SIG, _NAME) \ + XE_LOG_COMPONENT_##_TAG = MAKE_XE_LOG_COMPONENT(_CLASS, (_ID)), \ + XE_LOG_COMPONENT_##_CLASS##_##_ID = XE_LOG_COMPONENT_##_TAG, \ + /* eod */ + DEFINE_XE_LOG_COMPONENTS(MAKE_XE_LOG_COMPONENT_ENUM) +#undef MAKE_XE_LOG_COMPONENT_ENUM +}; + +/** + * enum xe_log_component_sigids - SIGIDs of all supported components + */ +enum xe_log_component_sigids { + /* private: */ +#define MAKE_XE_LOG_COMPONENT_SIGID(_CLASS, _ID, _TAG, _SIG, _NAME) \ + XE_LOG_COMPONENT_##_TAG##_SIGID = XE_SIGID_##_SIG, \ + /* eod */ + DEFINE_XE_LOG_COMPONENTS(MAKE_XE_LOG_COMPONENT_SIGID) +#undef MAKE_XE_LOG_COMPONENT_SIGID +}; + +#endif diff --git a/drivers/gpu/drm/xe/abi/xe_sigid_abi.h b/drivers/gpu/drm/xe/abi/xe_sigid_abi.h new file mode 100644 index 000000000000..2f3440505f69 --- /dev/null +++ b/drivers/gpu/drm/xe/abi/xe_sigid_abi.h @@ -0,0 +1,172 @@ +/* SPDX-License-Identifier: MIT */ +/* + * Copyright © 2026 Intel Corporation + */ + +#ifndef _ABI_XE_SIGID_ABI_H_ +#define _ABI_XE_SIGID_ABI_H_ + +/** + * DOC: Xe Error Signatures (SIGID) + * + * What SIGID stands for + * --------------------- + * + * SIGID is short for *Signature Identifier*. It is a small, stable integer + * that names one of the *recognised fault sites* -- nothing more. It is the + * primary handle used for triage and maps directly to a specific report site. + * + * Numbering + * --------- + * + * SIGIDs are a single flat list numbered sequentially within the assigned range, + * in the order the fault sites were introduced. Values are stable: once assigned + * they are only ever appended, never renumbered or reused. A retired fault site + * SIGID value is deprecated in place, never re-purposed. + * + * Why this exists + * --------------- + * + * Today the driver reports faults with ad-hoc ``xe_err()`` / ``xe_gt_err()`` + * strings that have no stable shape. That is fine for a human reading dmesg, + * but it gives fleet tooling nothing durable to match on: the wording changes + * between releases, lines can be rate-limited or dropped under an error storm, + * and there is no consistent way to ask "which recognised fault just happened?" + * + * A SIGID answers exactly that one question, identically across driver and + * firmware versions, and (eventually) across other Intel devices in a node. + * + * What a SIGID is not + * ------------------- + * + * SIGID deliberately does not encode the detailed reason or the outcome. Those + * are carried alongside it:: + * + * SIGID -> which recognised fault site is being reported + * severity -> how serious this instance is + * errno -> the failing operation's error, if available, shown with %pe + * message -> free-form human-readable context + * + * Severity is independent of the SIGID. The same SIGID can be reported at + * different severities depending on the instance and the recovery taken. + * + * When to use SIGID logging + * ------------------------- + * + * The xe_log_*() helpers are for these recognised fault sites only -- + * important, operator-relevant faults and events. The driver's only job is to + * emit the right SIGID next to the usual human-readable text. + + * They are not a replacement for ``xe_info()`` / ``xe_dbg()`` / tracing, nor + * for one-off diagnostics; using them for ordinary logging would dilute the + * fault stream. Not every ``xe_err()`` needs to become a SIGID report -- only + * those that correspond to a published fault site. + * + * SIGID log output (dmesg vs. the machine record) + * ----------------------------------------------- + * + * The dmesg line stays close to a normal xe error message so it remains + * readable for admins; the only stable, machine-matchable token on it is + * ``SIGID=<n>`` (``dmesg | grep SIGID=``). + * + * The full dmesg line is not an ABI: the surrounding text may change freely, + * and lines may be dropped. The durable record for tooling is the CPER record + * carrying the same SIGID (generation is a planned follow-up). + * + * How to pick a SIGID (the uniqueness rule) + * ----------------------------------------- + * + * Pick per *report site*, not per incident. Each site emits the single most + * specific recognised SIGID *for that site* -- so the question is never + * "classify this whole failure", it is "what does this site detect?", which has + * one answer. A single underlying failure therefore legitimately produces a + * *chain* of reports from different layers, each with its own SIGID -- e.g. a + * GuC communication failure is reported as %XE_SIGID_RUNTIME_FW by the firmware + * path, the failed recovery as %XE_SIGID_GT_TDR by the reset path, and an + * aborted bind as %XE_SIGID_PROBE by the probe path. That chain lets triage + * follow a fault from origin to final effect; it is not a duplicate. + * + * If a site does not match any defined SIGID, keep using the ordinary + * ``xe_err()`` / ``xe_gt_err()`` logging rather than forcing a SIGID: a wrong + * or over-broad classification is harder to retire than a missing one. When a + * new report site is genuinely worth triaging, add it to the list below. + * + * Usage of the existing SIGID reports must reevaluated according to this section + * after making significant changes to the site that emits this SIGID. + * + * Scope: software vs hardware emitted signatures + * ---------------------------------------------- + * + * Some SIGID represents fault sites that the *driver itself* detects and + * reports from the software POV: probe abort, wedged, survivability, driver- + * detected firmware failures, engine TDR, memory faults and IO/bus faults. + * These are the only values the driver assigns on its own. + * + * Signatures that *originate* in firmware or hardware are a different thing: + * they are produced and identified by the firmware or the hardware itself + * (e.g. via their own records or error counters), and the driver merely logs + * them as they are given to us. They are deliberately enumerated separately. + * + * The two driver-detected firmware report sites below (%XE_SIGID_RUNTIME_FW, + * %XE_SIGID_DEVICE_FW) are software signatures: they mark that *the driver* + * observed a firmware problem, not a signature reported by the firmware. + */ + +/* + * Top level Intel Error Signature Identifiers. + */ +#define INTEL_SIGID_INVALID 0 +#define INTEL_SIGID_BATCH 100 +#define INTEL_SIGID_RANGE_START(n) ((n) * INTEL_SIGID_BATCH) +#define INTEL_SIGID_RANGE_END(n) (INTEL_SIGID_RANGE_START((n) + 1) - 1) + +/* SIGIDs 1xx are reserved for Xe GPU software and 2xx for Xe GPU hardware */ +#define INTEL_SIGID_GPU_XE_SOFTWARE_START INTEL_SIGID_RANGE_START(1) +#define INTEL_SIGID_GPU_XE_SOFTWARE_END INTEL_SIGID_RANGE_END(1) +#define INTEL_SIGID_GPU_XE_HARDWARE_START INTEL_SIGID_RANGE_START(2) +#define INTEL_SIGID_GPU_XE_HARDWARE_END INTEL_SIGID_RANGE_END(2) + +/** + * enum xe_sigid - Stable Xe Error Signature Identifiers (SIGID). + * @XE_SIGID_SW: Software component failure. + * @XE_SIGID_PROBE: Device probe/bind was aborted. + * @XE_SIGID_WEDGED: Device was declared wedged and is no longer usable. + * @XE_SIGID_SURVIVABILITY: Device entered survivability mode. + * @XE_SIGID_RUNTIME_FW: Driver-detected runtime firmware failure, GuC/HuC/GSC. + * @XE_SIGID_DEVICE_FW: Driver-detected device firmware failure, PCODE/sysctrl. + * @XE_SIGID_GT_TDR: Engine hang / timeout detection and recovery (reset). + * @XE_SIGID_MEM_FAULT: VM bind, page fault or GTT fault. + * @XE_SIGID_IO_BUS: Runtime PCIe / IOMMU / MMIO access fault. + * @XE_SIGID_HW: Generic hardware failure. + * @XE_SIGID_PCIE: PCIe interface errors. + * @XE_SIGID_DEVICE_MEMORY: Device memory errors. + * @XE_SIGID_CORE_COMPUTE: Compute/shader core errors. + * @XE_SIGID_FABRIC: Fabric errors. + * @XE_SIGID_SOC_INTERNAL: SoC-internal errors. + * + * Each SIGID represents the report sites the driver detects and reports. + * Values are numbered sequentially, are only ever appended, and are never + * renumbered or reused. + * + * Firmware- and hardware-originated signatures are numbered separately. + */ +enum xe_sigid { + XE_SIGID_SW = INTEL_SIGID_GPU_XE_SOFTWARE_START, + XE_SIGID_PROBE = INTEL_SIGID_GPU_XE_SOFTWARE_START + 1, + XE_SIGID_WEDGED = INTEL_SIGID_GPU_XE_SOFTWARE_START + 2, + XE_SIGID_SURVIVABILITY = INTEL_SIGID_GPU_XE_SOFTWARE_START + 3, + XE_SIGID_RUNTIME_FW = INTEL_SIGID_GPU_XE_SOFTWARE_START + 4, + XE_SIGID_DEVICE_FW = INTEL_SIGID_GPU_XE_SOFTWARE_START + 5, + XE_SIGID_GT_TDR = INTEL_SIGID_GPU_XE_SOFTWARE_START + 6, + XE_SIGID_MEM_FAULT = INTEL_SIGID_GPU_XE_SOFTWARE_START + 7, + XE_SIGID_IO_BUS = INTEL_SIGID_GPU_XE_SOFTWARE_START + 8, + + XE_SIGID_HW = INTEL_SIGID_GPU_XE_HARDWARE_START, + XE_SIGID_PCIE = INTEL_SIGID_GPU_XE_HARDWARE_START + 1, + XE_SIGID_DEVICE_MEMORY = INTEL_SIGID_GPU_XE_HARDWARE_START + 2, + XE_SIGID_CORE_COMPUTE = INTEL_SIGID_GPU_XE_HARDWARE_START + 3, + XE_SIGID_FABRIC = INTEL_SIGID_GPU_XE_HARDWARE_START + 4, + XE_SIGID_SOC_INTERNAL = INTEL_SIGID_GPU_XE_HARDWARE_START + 5, +}; + +#endif diff --git a/drivers/gpu/drm/xe/regs/xe_gt_regs.h b/drivers/gpu/drm/xe/regs/xe_gt_regs.h index 08251c7a1a4b..48c515d91882 100644 --- a/drivers/gpu/drm/xe/regs/xe_gt_regs.h +++ b/drivers/gpu/drm/xe/regs/xe_gt_regs.h @@ -634,6 +634,12 @@ #define GT_GFX_RC6_LOCKED XE_REG(0x138104) #define GT_GFX_RC6 XE_REG(0x138108) +#define GT_IA_PERF_BIAS_REG XE_REG(0x138158) +#define GT_BIAS REG_GENMASK(31, 16) +#define IA_BIAS REG_GENMASK(15, 0) +#define GT_BIAS_DEFAULT 0x10 +#define IA_BIAS_DEFAULT 0x10 + #define GT0_PERF_LIMIT_REASONS XE_REG(0x1381a8) /* Common performance limit reason bits - available on all platforms */ #define GT0_PERF_LIMIT_REASONS_MASK 0xde3 diff --git a/drivers/gpu/drm/xe/tests/Makefile b/drivers/gpu/drm/xe/tests/Makefile index f7aa47f11a36..0b809a252cfa 100644 --- a/drivers/gpu/drm/xe/tests/Makefile +++ b/drivers/gpu/drm/xe/tests/Makefile @@ -7,6 +7,7 @@ xe_live_test-y = xe_live_test_mod.o # Normal kunit tests obj-$(CONFIG_DRM_XE_KUNIT_TEST) += xe_test.o xe_test-y = xe_test_mod.o \ + xe_any_kunit.o \ xe_args_test.o \ xe_pci_test.o \ xe_rtp_tables_test.o \ diff --git a/drivers/gpu/drm/xe/tests/xe_any_kunit.c b/drivers/gpu/drm/xe/tests/xe_any_kunit.c new file mode 100644 index 000000000000..0a5f28894cd2 --- /dev/null +++ b/drivers/gpu/drm/xe/tests/xe_any_kunit.c @@ -0,0 +1,213 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Copyright © 2026 Intel Corporation + */ + +#include <kunit/test.h> + +#include "tests/xe_kunit_helpers.h" +#include "tests/xe_pci_test.h" +#include "xe_any.h" +#include "xe_device.h" + +static void test_to_xe(struct kunit *test) +{ + struct xe_device *xe = test->priv; + struct xe_tile *tile = xe_device_get_root_tile(xe); + struct xe_gt *gt = tile->primary_gt; + struct drm_device *drm = &xe->drm; + struct device *dev = xe->drm.dev; + struct pci_dev *pdev = to_pci_dev(dev); + const struct xe_device *cxe = xe; + const struct xe_tile *ctile = tile; + const struct xe_gt *cgt = gt; + + KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(xe)); + KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(tile)); + KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(gt)); + KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(drm)); + KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(dev)); + KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_to_xe(pdev)); + + KUNIT_EXPECT_PTR_EQ(test, cxe, xe_any_to_xe(cxe)); + KUNIT_EXPECT_PTR_EQ(test, cxe, xe_any_to_xe(ctile)); + KUNIT_EXPECT_PTR_EQ(test, cxe, xe_any_to_xe(cgt)); +} + +static void test_to_pdev(struct kunit *test) +{ + struct xe_device *xe = test->priv; + struct xe_tile *tile = xe_device_get_root_tile(xe); + struct xe_gt *gt = tile->primary_gt; + struct drm_device *drm = &xe->drm; + struct device *dev = xe->drm.dev; + struct pci_dev *pdev = to_pci_dev(dev); + + KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(xe)); + KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(tile)); + KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(gt)); + KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(drm)); + KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(dev)); + KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(pdev)); + + /* mimic early probe stage */ + dev_set_drvdata(xe->drm.dev, NULL); + KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(dev)); + KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_to_pdev(pdev)); +} + +static void test_to_dev(struct kunit *test) +{ + struct xe_device *xe = test->priv; + struct xe_tile *tile = xe_device_get_root_tile(xe); + struct xe_gt *gt = tile->primary_gt; + struct drm_device *drm = &xe->drm; + struct device *dev = xe->drm.dev; + struct pci_dev *pdev = to_pci_dev(dev); + + KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(xe)); + KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(tile)); + KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(gt)); + KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(drm)); + KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(dev)); + KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(pdev)); + + /* mimic early probe stage */ + dev_set_drvdata(xe->drm.dev, NULL); + KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(dev)); + KUNIT_EXPECT_PTR_EQ(test, dev, xe_any_to_dev(pdev)); +} + +static void test_to_drm(struct kunit *test) +{ + struct xe_device *xe = test->priv; + struct xe_tile *tile = xe_device_get_root_tile(xe); + struct xe_gt *gt = tile->primary_gt; + struct drm_device *drm = &xe->drm; + struct device *dev = xe->drm.dev; + struct pci_dev *pdev = to_pci_dev(dev); + + KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(xe)); + KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(tile)); + KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(gt)); + KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(drm)); + KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(dev)); + KUNIT_EXPECT_PTR_EQ(test, drm, xe_any_to_drm(pdev)); +} + +static void test_if_pdev(struct kunit *test) +{ + struct xe_device *xe = test->priv; + struct xe_tile *tile = xe_device_get_root_tile(xe); + struct xe_gt *gt = tile->primary_gt; + struct drm_device *drm = &xe->drm; + struct device *dev = xe->drm.dev; + struct pci_dev *pdev = to_pci_dev(dev); + + KUNIT_EXPECT_TRUE(test, xe && drm && tile && gt && dev && pdev); + KUNIT_EXPECT_NULL(test, xe_any_if_pdev(xe)); + KUNIT_EXPECT_NULL(test, xe_any_if_pdev(tile)); + KUNIT_EXPECT_NULL(test, xe_any_if_pdev(gt)); + KUNIT_EXPECT_NULL(test, xe_any_if_pdev(drm)); + KUNIT_EXPECT_NULL(test, xe_any_if_pdev(dev)); + KUNIT_EXPECT_PTR_EQ(test, pdev, xe_any_if_pdev(pdev)); +} + +static void test_if_xe(struct kunit *test) +{ + struct xe_device *xe = test->priv; + struct xe_tile *tile = xe_device_get_root_tile(xe); + struct xe_gt *gt = tile->primary_gt; + struct drm_device *drm = &xe->drm; + struct device *dev = xe->drm.dev; + struct pci_dev *pdev = to_pci_dev(dev); + + KUNIT_EXPECT_TRUE(test, xe && drm && tile && gt && dev && pdev); + KUNIT_EXPECT_PTR_EQ(test, xe, xe_any_if_xe(xe)); + KUNIT_EXPECT_NULL(test, xe_any_if_xe(tile)); + KUNIT_EXPECT_NULL(test, xe_any_if_xe(gt)); + KUNIT_EXPECT_NULL(test, xe_any_if_xe(drm)); + KUNIT_EXPECT_NULL(test, xe_any_if_xe(dev)); + KUNIT_EXPECT_NULL(test, xe_any_if_xe(pdev)); +} + +static void test_if_tile(struct kunit *test) +{ + struct xe_device *xe = test->priv; + struct xe_tile *tile = xe_device_get_root_tile(xe); + struct xe_gt *gt = tile->primary_gt; + struct drm_device *drm = &xe->drm; + struct device *dev = xe->drm.dev; + struct pci_dev *pdev = to_pci_dev(dev); + + KUNIT_EXPECT_TRUE(test, xe && drm && tile && gt && dev && pdev); + KUNIT_EXPECT_NULL(test, xe_any_if_tile(xe)); + KUNIT_EXPECT_PTR_EQ(test, tile, xe_any_if_tile(tile)); + KUNIT_EXPECT_NULL(test, xe_any_if_tile(gt)); + KUNIT_EXPECT_NULL(test, xe_any_if_tile(drm)); + KUNIT_EXPECT_NULL(test, xe_any_if_tile(dev)); + KUNIT_EXPECT_NULL(test, xe_any_if_tile(pdev)); +} + +static void test_if_gt(struct kunit *test) +{ + struct xe_device *xe = test->priv; + struct xe_tile *tile = xe_device_get_root_tile(xe); + struct xe_gt *gt = tile->primary_gt; + struct drm_device *drm = &xe->drm; + struct device *dev = xe->drm.dev; + struct pci_dev *pdev = to_pci_dev(dev); + + KUNIT_EXPECT_TRUE(test, xe && drm && tile && gt && dev && pdev); + KUNIT_EXPECT_NULL(test, xe_any_if_gt(xe)); + KUNIT_EXPECT_NULL(test, xe_any_if_gt(tile)); + KUNIT_EXPECT_PTR_EQ(test, gt, xe_any_if_gt(gt)); + KUNIT_EXPECT_NULL(test, xe_any_if_gt(drm)); + KUNIT_EXPECT_NULL(test, xe_any_if_gt(dev)); + KUNIT_EXPECT_NULL(test, xe_any_if_gt(pdev)); +} + +static void test_to_id(struct kunit *test) +{ + struct xe_device *xe = test->priv; + struct xe_tile *tile = xe_device_get_root_tile(xe); + struct xe_gt *gt = tile->primary_gt; + struct device *dev = xe->drm.dev; + struct pci_dev *pdev = to_pci_dev(dev); + const struct xe_device *cxe = xe; + const struct xe_tile *ctile = tile; + const struct xe_gt *cgt = gt; + + tile->id = 1; + gt->info.id = 2; + + KUNIT_EXPECT_EQ(test, 0, xe_any_id(xe)); + KUNIT_EXPECT_EQ(test, 1, xe_any_id(tile)); + KUNIT_EXPECT_EQ(test, 2, xe_any_id(gt)); + KUNIT_EXPECT_EQ(test, 0, xe_any_id(dev)); + KUNIT_EXPECT_EQ(test, 0, xe_any_id(pdev)); + KUNIT_EXPECT_EQ(test, 0, xe_any_id(cxe)); + KUNIT_EXPECT_EQ(test, 1, xe_any_id(ctile)); + KUNIT_EXPECT_EQ(test, 2, xe_any_id(cgt)); +} + +static struct kunit_case xe_any_tests[] = { + KUNIT_CASE(test_to_xe), + KUNIT_CASE(test_to_dev), + KUNIT_CASE(test_to_pdev), + KUNIT_CASE(test_to_drm), + KUNIT_CASE(test_if_pdev), + KUNIT_CASE(test_if_xe), + KUNIT_CASE(test_if_tile), + KUNIT_CASE(test_if_gt), + KUNIT_CASE(test_to_id), + {} +}; + +static struct kunit_suite xe_any_test_suite = { + .name = "xe_any", + .test_cases = xe_any_tests, + .init = xe_kunit_helper_xe_device_test_init, +}; + +kunit_test_suite(xe_any_test_suite); diff --git a/drivers/gpu/drm/xe/tests/xe_kunit_helpers.c b/drivers/gpu/drm/xe/tests/xe_kunit_helpers.c index bc5156966ce9..27740b40c8ae 100644 --- a/drivers/gpu/drm/xe/tests/xe_kunit_helpers.c +++ b/drivers/gpu/drm/xe/tests/xe_kunit_helpers.c @@ -39,6 +39,10 @@ struct xe_device *xe_kunit_helper_alloc_xe_device(struct kunit *test, struct xe_device, drm, DRIVER_GEM); KUNIT_ASSERT_NOT_ERR_OR_NULL(test, xe); + + dev_set_drvdata(xe->drm.dev, &xe->drm); + KUNIT_ASSERT_PTR_EQ(test, xe, kdev_to_xe_device(dev)); + return xe; } EXPORT_SYMBOL_IF_KUNIT(xe_kunit_helper_alloc_xe_device); diff --git a/drivers/gpu/drm/xe/tests/xe_log_kunit.c b/drivers/gpu/drm/xe/tests/xe_log_kunit.c new file mode 100644 index 000000000000..56c36c8a08e1 --- /dev/null +++ b/drivers/gpu/drm/xe/tests/xe_log_kunit.c @@ -0,0 +1,553 @@ +// SPDX-License-Identifier: GPL-2.0 AND MIT +/* + * Copyright © 2026 Intel Corporation + */ + +#include <kunit/static_stub.h> +#include <kunit/test.h> +#include <kunit/test-bug.h> + +#include "tests/xe_kunit_helpers.h" +#include "tests/xe_pci_test.h" +#include "xe_device.h" +#include "xe_log.h" + +static void nop_dmesg_vprintk(struct pci_dev *pdev, int cper_sev, struct va_format *vaf) +{ +} + +static void nop_emit_cper(struct pci_dev *pdev, int cper_sev, + enum xe_sigid sigid, u32 component, u32 location, + const void *data, size_t len, struct va_format *vaf) +{ +} + +static const char *component_name(u32 component) +{ + switch (component) { +#define make_component_tag_case(_CLASS, _ID, _TAG, _SIG, _NAME) \ + case XE_LOG_COMPONENT_##_TAG: return _NAME; + DEFINE_XE_LOG_COMPONENTS(make_component_tag_case) +#undef make_component_tag_case + } + return component ? "???" : ""; +} + +static const char *location_type(u32 location) +{ + u32 type = FIELD_GET(XE_LOG_LOCATION_TYPE_MASK, location); + + return type == XE_LOG_LOCATION_TYPE_DEVICE ? "DEVICE" : + type == XE_LOG_LOCATION_TYPE_TILE ? "TILE" : + type == XE_LOG_LOCATION_TYPE_GT ? "GT" : + location ? "?" : ""; +} + +static void fake_emit_cper(struct pci_dev *pdev, int cper_sev, + enum xe_sigid sigid, u32 component, u32 location, + const void *data, size_t len, struct va_format *vaf) +{ + char msg[64]; + int n; + + pr_info("\n"); + pr_info("CPER SEV=%u SIGID=%u\n", cper_sev, sigid); + pr_info("CPER DEVICE=%s\n", dev_name(&pdev->dev)); + if (location) + pr_info("CPER LOCATION=%#x \t# %s.%u\n", + location, location_type(location), + FIELD_GET(XE_LOG_LOCATION_ID_MASK, location)); + if (component) + pr_info("CPER COMPONENT=%#x \t# %s\n", + component, component_name(component)); + if (IS_ERR(data)) + pr_info("CPER ERR=%ld \t\t# %pe\n", PTR_ERR(data), data); + else if (len) + print_hex_dump(KERN_INFO, "CPER BIN=", DUMP_PREFIX_OFFSET, + 16, 1, data, len, false); + + n = vscnprintf(msg, sizeof(msg), vaf->fmt, *vaf->va); + print_hex_dump(KERN_INFO, "CPER MSG=", DUMP_PREFIX_OFFSET, 16, 1, msg, n, true); + pr_info("CPER END\n"); +} + +static const u8 blob[] = { 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12 }; +static const u8 dead[] = { 0xde, 0xad, 0xbe, 0xef }; + +static struct xe_tile *to_tile_safe(struct xe_device *xe) +{ + return xe ? &xe->tiles[1] : NULL; +} + +static struct xe_gt *to_gt_safe(struct xe_device *xe) +{ + return xe ? to_tile_safe(xe)->primary_gt : NULL; +} + +static struct pci_dev *to_pdev_safe(struct xe_device *xe) +{ + return xe ? xe_any_to_pdev(xe) : NULL; +} + +static void demo(struct xe_device *xe) +{ + struct pci_dev *pdev = xe_any_to_pdev(xe); + struct xe_tile *tile = to_tile_safe(xe); + struct xe_gt *gt = to_gt_safe(xe); + + /* SW errno */ + xe_log_emit(pdev, CPER_SEV_FATAL, XE_SIGID_PROBE, + XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE, + ERR_PTR(-ENODEV), 0, "testing %s signature\n", "software"); + + xe_log_err(tile, PROBE, -ENODEV, "testing %s signature\n", "software"); + xe_log_err_corrected(gt, PROBE, -ENODEV, "testing %s signature\n", "software"); + xe_log_info(gt, PROBE, "testing %s signature\n", "software"); + + /* HW data */ + xe_log_emit(pdev, CPER_SEV_FATAL, XE_SIGID_PCIE, + XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE, + blob, sizeof(blob), "testing %s signature\n", "HARDWARE"); + xe_log_emit_recoverable(pdev, XE_SIGID_FABRIC, + XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE, + blob, sizeof(blob), "testing %s signature\n", "HARDWARE"); + xe_log_from_corrected(tile, XE_SIGID_DEVICE_MEMORY, XE_LOG_COMPONENT_NONE, + blob, sizeof(blob), "testing %s signature\n", "HARDWARE"); + xe_log_from_info(gt, XE_SIGID_CORE_COMPUTE, XE_LOG_COMPONENT_NONE, + blob, sizeof(blob), "testing %s signature\n", "HARDWARE"); +} + +static void demo_dmesg(struct kunit *test) +{ + kunit_activate_static_stub(test, log_emit_cper, nop_emit_cper); + demo(test->priv); +} + +static void demo_cper(struct kunit *test) +{ + kunit_activate_static_stub(test, log_emit_cper, fake_emit_cper); + kunit_activate_static_stub(test, log_dmesg_vprintk, nop_dmesg_vprintk); + demo(test->priv); +} + +static const char *test_fatal(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + + if (pdev) + xe_log_emit_fatal(pdev, XE_SIGID_PROBE, + XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE, + NULL, 0, "testing %d\n", 123); + return "SIGID=101 FATAL testing 123\n"; +} + +static const char *test_fatal_tile(struct xe_device *xe) +{ + struct xe_tile *tile = to_tile_safe(xe); + + if (tile) + xe_log_from_fatal(tile, XE_SIGID_PROBE, XE_LOG_COMPONENT_NONE, + ERR_PTR(-ENODEV), 0, "testing %d\n", 123); + return "SIGID=101 FATAL (-ENODEV) Tile1: testing 123\n"; +} + +static const char *test_fatal_gt(struct xe_device *xe) +{ + struct xe_gt *gt = to_gt_safe(xe); + + if (gt) + xe_log_from_fatal(gt, XE_SIGID_PROBE, XE_LOG_COMPONENT_NONE, + ERR_PTR(-ENODEV), 0, "testing %d\n", 123); + return "SIGID=101 FATAL (-ENODEV) Tile1: GT1: testing 123\n"; +} + +static const char *test_fatal_comp(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + + if (pdev) + xe_log_err_fatal(pdev, PROBE, -ENODEV, "testing %d\n", 123); + return "SIGID=101 FATAL (-ENODEV) PROBE: testing 123\n"; +} + +static const char *test_fatal_comp_tile(struct xe_device *xe) +{ + struct xe_tile *tile = to_tile_safe(xe); + + if (tile) + xe_log_err_fatal(tile, PROBE, -ENODEV, "testing %d\n", 123); + return "SIGID=101 FATAL (-ENODEV) Tile1: PROBE: testing 123\n"; +} + +static const char *test_fatal_comp_gt(struct xe_device *xe) +{ + struct xe_gt *gt = to_gt_safe(xe); + + if (gt) + xe_log_err_fatal(gt, PROBE, -ENODEV, "testing %d\n", 123); + return "SIGID=101 FATAL (-ENODEV) Tile1: GT1: PROBE: testing 123\n"; +} + +static const char *test_fatal_all(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + struct xe_gt *gt = to_gt_safe(xe); + + if (pdev) + xe_log_emit(pdev, CPER_SEV_FATAL, XE_SIGID_PROBE, + XE_LOG_COMPONENT_PROBE, MAKE_XE_LOG_LOCATION(GT, gt->info.id), + ERR_PTR(-ENODEV), 0, "testing %d\n", 123); + return "SIGID=101 FATAL (-ENODEV) Tile1: GT1: PROBE: testing 123\n"; +} + +static const char *test_recoverable(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + + if (pdev) + xe_log_emit_recoverable(pdev, XE_SIGID_RUNTIME_FW, + XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE, + ERR_PTR(-EIO), 0, "testing %d\n", 123); + return "SIGID=104 (-EIO) testing 123\n"; +} + +static const char *test_recoverable_tile(struct xe_device *xe) +{ + struct xe_tile *tile = to_tile_safe(xe); + + if (tile) + xe_log_from_recoverable(tile, XE_SIGID_RUNTIME_FW, XE_LOG_COMPONENT_NONE, + ERR_PTR(-EIO), 0, "testing %d\n", 123); + return "SIGID=104 (-EIO) Tile1: testing 123\n"; +} + +static const char *test_recoverable_gt(struct xe_device *xe) +{ + struct xe_gt *gt = to_gt_safe(xe); + + if (gt) + xe_log_from_recoverable(gt, XE_SIGID_RUNTIME_FW, XE_LOG_COMPONENT_NONE, + ERR_PTR(-EIO), 0, "testing %d\n", 123); + return "SIGID=104 (-EIO) Tile1: GT1: testing 123\n"; +} + +static const char *test_recoverable_comp(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + + if (pdev) + xe_log_err(pdev, GUC, -EIO, "testing %d\n", 123); + return "SIGID=104 (-EIO) GUC: testing 123\n"; +} + +static const char *test_recoverable_comp_tile(struct xe_device *xe) +{ + struct xe_tile *tile = to_tile_safe(xe); + + if (tile) + xe_log_err(tile, GUC, -EIO, "testing %d\n", 123); + return "SIGID=104 (-EIO) Tile1: GUC: testing 123\n"; +} + +static const char *test_recoverable_comp_gt(struct xe_device *xe) +{ + struct xe_gt *gt = to_gt_safe(xe); + + if (gt) + xe_log_err(gt, GUC, -EIO, "testing %d\n", 123); + return "SIGID=104 (-EIO) Tile1: GT1: GUC: testing 123\n"; +} + +static const char *test_recoverable_all(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + struct xe_gt *gt = to_gt_safe(xe); + + if (pdev) + xe_log_emit(pdev, CPER_SEV_RECOVERABLE, XE_SIGID_RUNTIME_FW, + XE_LOG_COMPONENT_GUC, MAKE_XE_LOG_LOCATION(GT, gt->info.id), + ERR_PTR(-EIO), 0, "testing %d\n", 123); + return "SIGID=104 (-EIO) Tile1: GT1: GUC: testing 123\n"; +} + +static const char *test_info(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + + if (pdev) + xe_log_emit(pdev, CPER_SEV_INFORMATIONAL, XE_SIGID_DEVICE_FW, + XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE, + NULL, 0, "testing %d\n", 123); + return "SIGID=105 testing 123\n"; +} + +static const char *test_info_tile(struct xe_device *xe) +{ + struct xe_tile *tile = to_tile_safe(xe); + + if (tile) + xe_log_from_info(tile, XE_SIGID_DEVICE_FW, XE_LOG_COMPONENT_NONE, + NULL, 0, "testing %d\n", 123); + return "SIGID=105 Tile1: testing 123\n"; +} + +static const char *test_info_gt(struct xe_device *xe) +{ + struct xe_gt *gt = to_gt_safe(xe); + + if (gt) + xe_log_from_info(gt, XE_SIGID_DEVICE_FW, XE_LOG_COMPONENT_NONE, + NULL, 0, "testing %d\n", 123); + return "SIGID=105 Tile1: GT1: testing 123\n"; +} + +static const char *test_info_err(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + + if (pdev) + xe_log_err_info(pdev, PCODE, -EPROTO, "testing %d\n", 123); + return "SIGID=105 (-EPROTO) PCODE: testing 123\n"; +} + +static const char *test_info_comp(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + + if (pdev) + xe_log_info(pdev, PCODE, "testing %d\n", 123); + return "SIGID=105 PCODE: testing 123\n"; +} + +static const char *test_info_comp_tile(struct xe_device *xe) +{ + struct xe_tile *tile = to_tile_safe(xe); + + if (tile) + xe_log_info(tile, PCODE, "testing %d\n", 123); + return "SIGID=105 Tile1: PCODE: testing 123\n"; +} + +static const char *test_info_comp_gt(struct xe_device *xe) +{ + struct xe_gt *gt = to_gt_safe(xe); + + if (gt) + xe_log_info(gt, PCODE, "testing %d\n", 123); + return "SIGID=105 Tile1: GT1: PCODE: testing 123\n"; +} + +static const char *test_info_all(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + struct xe_gt *gt = to_gt_safe(xe); + + if (pdev) + xe_log_emit(pdev, CPER_SEV_INFORMATIONAL, XE_SIGID_DEVICE_FW, + XE_LOG_COMPONENT_PCODE, MAKE_XE_LOG_LOCATION(GT, gt->info.id), + NULL, 0, "testing %d\n", 123); + return "SIGID=105 Tile1: GT1: PCODE: testing 123\n"; +} + +static const char *test_hw_fatal(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + + if (pdev) + xe_log_emit_fatal(pdev, XE_SIGID_PCIE, + XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE, + dead, sizeof(dead), "testing %d\n", 123); + return "SIGID=201 FATAL (deadbeef) " HW_ERR "testing 123\n"; +} + +static const char *test_hw_recoverable(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + + if (pdev) + xe_log_emit_recoverable(pdev, XE_SIGID_DEVICE_MEMORY, + XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE, + dead, sizeof(dead), "testing %d\n", 123); + return "SIGID=202 (deadbeef) " HW_ERR "testing 123\n"; +} + +static const char *test_hw_corrected(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + + if (pdev) + xe_log_emit_corrected(pdev, XE_SIGID_DEVICE_MEMORY, + XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE, + dead, sizeof(dead), "testing %d\n", 123); + return "SIGID=202 CORRECTED (deadbeef) " HW_ERR "testing 123\n"; +} + +static const char *test_hw_informational(struct xe_device *xe) +{ + struct pci_dev *pdev = to_pdev_safe(xe); + + if (pdev) + xe_log_emit_info(pdev, XE_SIGID_DEVICE_MEMORY, + XE_LOG_COMPONENT_NONE, XE_LOG_LOCATION_NONE, + dead, sizeof(dead), "testing %d\n", 123); + return "SIGID=202 (deadbeef) testing 123\n"; +} + +static const struct log_test_param { + const char *(*func)(struct xe_device *xe); +} log_test_params[] = { + { .func = test_fatal }, + { .func = test_fatal_tile }, + { .func = test_fatal_gt }, + { .func = test_fatal_comp }, + { .func = test_fatal_comp_tile }, + { .func = test_fatal_comp_gt }, + { .func = test_fatal_all }, + { .func = test_recoverable }, + { .func = test_recoverable_tile }, + { .func = test_recoverable_gt }, + { .func = test_recoverable_comp }, + { .func = test_recoverable_comp_tile }, + { .func = test_recoverable_comp_gt }, + { .func = test_recoverable_all }, + { .func = test_info }, + { .func = test_info_tile }, + { .func = test_info_gt }, + { .func = test_info_err }, + { .func = test_info_comp }, + { .func = test_info_comp_tile }, + { .func = test_info_comp_gt }, + { .func = test_info_all }, + { .func = test_hw_fatal }, + { .func = test_hw_recoverable }, + { .func = test_hw_corrected }, + { .func = test_hw_informational }, +}; + +static void log_param_get_desc(const struct log_test_param *p, char *desc) +{ + snprintf(desc, KUNIT_PARAM_DESC_SIZE, "%ps", p->func); +} + +KUNIT_ARRAY_PARAM(log, log_test_params, log_param_get_desc); + +static void check_dmesg_vprintk(struct pci_dev *pdev, int cper_sev, struct va_format *vaf) +{ + struct kunit *test = kunit_get_current_test(); + const struct log_test_param *param = test->param_value; + const char *exp = param->func(NULL); + char msg[64]; + int n; + + n = vsnprintf(msg, sizeof(msg), vaf->fmt, *vaf->va); + KUNIT_EXPECT_LT(test, n, sizeof(msg)); + KUNIT_EXPECT_STREQ(test, msg, exp); +} + +static void test_dmesg(struct kunit *test) +{ + const struct log_test_param *param = test->param_value; + + kunit_activate_static_stub(test, log_emit_cper, nop_emit_cper); + kunit_activate_static_stub(test, log_dmesg_vprintk, check_dmesg_vprintk); + + param->func(test->priv); +} + +#define INVALID_TILEID (XE_MAX_TILES_PER_DEVICE + 1) +#define INVALID_GTID (XE_MAX_TILES_PER_DEVICE * XE_MAX_GT_PER_TILE + 1) + +#define PREP_TEST_LOCATION(type, id) \ + (FIELD_PREP_CONST(XE_LOG_LOCATION_TYPE_MASK, (type)) | \ + FIELD_PREP_CONST(XE_LOG_LOCATION_ID_MASK, (id))) +#define PREP_TEST_COMPONENT(class, type) \ + (FIELD_PREP_CONST(XE_LOG_COMPONENT_CLASS_MASK, (class)) | \ + FIELD_PREP_CONST(XE_LOG_COMPONENT_TYPE_MASK, (type))) + +static const struct { + u32 comp; + u32 loc; + const char *name; +} invalid_params[] = { + { .name = "no-component no-location no-warn" }, + { .loc = PREP_TEST_LOCATION(0, 1), + .name = "reserved location" }, + { .loc = PREP_TEST_LOCATION(255, 0), + .name = "unknown location" }, + { .loc = PREP_TEST_LOCATION(XE_LOG_LOCATION_TYPE_DEVICE, 1), + .name = "nonzero-device-id location" }, + { .loc = PREP_TEST_LOCATION(XE_LOG_LOCATION_TYPE_TILE, INVALID_TILEID), + .name = "invalid-tile-id location" }, + { .loc = PREP_TEST_LOCATION(XE_LOG_LOCATION_TYPE_GT, INVALID_GTID), + .name = "invalid-gt-id location" }, + { .comp = PREP_TEST_COMPONENT(255, 0), + .name = "unknown component class" }, + { .comp = PREP_TEST_COMPONENT(XE_LOG_COMPONENT_CLASS_SYSTEM, 255), + .name = "unknown system component" }, + { .comp = PREP_TEST_COMPONENT(XE_LOG_COMPONENT_CLASS_HARDWARE, 255), + .name = "unknown hardware component" }, + { .comp = PREP_TEST_COMPONENT(255, 1), + .loc = PREP_TEST_LOCATION(255, 1), + .name = "unknown component and location" }, +}; + +KUNIT_ARRAY_PARAM_DESC(invalid_param, invalid_params, name); + +static void test_invalid(struct kunit *test) +{ + struct xe_device *xe = test->priv; + typeof(invalid_params[0]) *param = test->param_value; + + struct pci_dev *pdev = xe_any_to_pdev(xe); + + if (!IS_ENABLED(CONFIG_DRM_XE_DEBUG)) + kunit_skip(test, "requires CONFIG_DRM_XE_DEBUG\n"); + + kunit_activate_static_stub(test, log_emit_cper, nop_emit_cper); + kunit_activate_static_stub(test, log_dmesg_vprintk, nop_dmesg_vprintk); + + kunit_warning_suppress(test) { + xe_log_emit(pdev, CPER_SEV_FATAL, XE_SIGID_PROBE, + param->comp, param->loc, + NULL, 0, "testing %s\n", param->name); + KUNIT_EXPECT_SUPPRESSED_WARNING_COUNT(test, !!param->comp + !!param->loc); + } +} + +static int xe_log_test_init(struct kunit *test) +{ + struct xe_pci_fake_data fake = { + .platform = XE_PVC, /* with max_remote_tiles != 0 */ + .subplatform = XE_SUBPLATFORM_NONE, + .graphics_verx100 = 2001, + .media_verx100 = 2001, + }; + struct xe_device *xe; + + test->priv = &fake; + xe_kunit_helper_xe_device_test_init(test); + xe = test->priv; + + KUNIT_ASSERT_NOT_NULL(test, to_tile_safe(xe)); + KUNIT_ASSERT_NOT_NULL(test, to_gt_safe(xe)); + KUNIT_EXPECT_EQ(test, 1, xe_any_id(to_tile_safe(xe))); + KUNIT_EXPECT_EQ(test, 1, xe_any_id(to_gt_safe(xe))); + + return 0; +} + +static struct kunit_case xe_log_test_cases[] = { + KUNIT_CASE(demo_cper), + KUNIT_CASE(demo_dmesg), + KUNIT_CASE_PARAM(test_dmesg, log_gen_params), + KUNIT_CASE_PARAM(test_invalid, invalid_param_gen_params), + {} +}; + +static struct kunit_suite xe_log_suite = { + .name = "xe_log", + .test_cases = xe_log_test_cases, + .init = xe_log_test_init, +}; + +kunit_test_suites(&xe_log_suite); diff --git a/drivers/gpu/drm/xe/tests/xe_migrate.c b/drivers/gpu/drm/xe/tests/xe_migrate.c index 3c1be809be82..f10d9513747b 100644 --- a/drivers/gpu/drm/xe/tests/xe_migrate.c +++ b/drivers/gpu/drm/xe/tests/xe_migrate.c @@ -198,8 +198,7 @@ static void xe_migrate_sanity_test(struct xe_migrate *m, struct kunit *test, err = xe_bo_vmap(bo); if (err) { - KUNIT_FAIL(test, "Failed to vmap our pagetables: %li\n", - PTR_ERR(bo)); + KUNIT_FAIL(test, "Failed to vmap our pagetables: %d\n", err); return; } diff --git a/drivers/gpu/drm/xe/xe_any.h b/drivers/gpu/drm/xe/xe_any.h new file mode 100644 index 000000000000..5d97afa76915 --- /dev/null +++ b/drivers/gpu/drm/xe/xe_any.h @@ -0,0 +1,137 @@ +/* SPDX-License-Identifier: MIT */ +/* + * Copyright © 2026 Intel Corporation + */ + +#ifndef _XE_ANY_H_ +#define _XE_ANY_H_ + +#include "xe_device.h" + +#define __xe_any_to_self_assoc(type, any) \ + const type * : (any), \ + type * : (any) + +/** + * xe_any_if_type() - Get the pointer only if it is @type pointer. + * @any: any pointer + * @type: data type to look for + * + * Return: the @type pointer or NULL. + */ +#define xe_any_if_type(any, type) \ + _Generic((any), \ + __xe_any_to_self_assoc(type, (any)), \ + default : NULL) + +/** + * xe_any_if_gt() - Get the pointer only if it is &xe_gt. + * @any: any pointer + * + * Return: the @xe_gt pointer or NULL. + */ +#define xe_any_if_gt(any) xe_any_if_type((any), struct xe_gt) + +/** + * xe_any_if_tile() - Get the pointer only if it is &xe_tile. + * @any: any pointer + * + * Return: the @xe_tile pointer or NULL. + */ +#define xe_any_if_tile(any) xe_any_if_type((any), struct xe_tile) + +/** + * xe_any_if_xe() - Get the pointer only if it is &xe_device. + * @any: any pointer + * + * Return: the @xe_device pointer or NULL. + */ +#define xe_any_if_xe(any) xe_any_if_type((any), struct xe_device) + +/** + * xe_any_if_pdev() - Get the pointer only if it is &pci_dev. + * @any: any pointer + * + * Return: the @pci_dev pointer or NULL. + */ +#define xe_any_if_pdev(any) xe_any_if_type((any), struct pci_dev) + +#define __xe_any_to_other_assoc(const, from, other, p) \ + const struct from * : __##from##_to_##other((const struct from *)(p)) + +#define __xe_tile_to_xe_device(p) tile_to_xe(p) +#define __xe_gt_to_xe_device(p) gt_to_xe(p) +#define __pci_dev_to_xe_device(p) pdev_to_xe_device(p) +#define __device_to_xe_device(p) kdev_to_xe_device(p) +#define __drm_device_to_xe_device(p) to_xe_device(p) +#define __pci_dev_to_device(p) (&(p)->dev) + +/** + * xe_any_to_xe() - Obtain the &xe_device pointer. + * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer + * + * Return: the @xe_device pointer or backpointer. + */ +#define xe_any_to_xe(any) \ + _Generic((any), \ + __xe_any_to_self_assoc(struct xe_device, (any)), \ + __xe_any_to_other_assoc(/* */, xe_tile, xe_device, (any)), \ + __xe_any_to_other_assoc(const, xe_tile, xe_device, (any)), \ + __xe_any_to_other_assoc(/* */, xe_gt, xe_device, (any)), \ + __xe_any_to_other_assoc(const, xe_gt, xe_device, (any)), \ + __xe_any_to_other_assoc(, drm_device, xe_device, (any)), \ + __xe_any_to_other_assoc(, pci_dev, xe_device, (any)), \ + __xe_any_to_other_assoc(, device, xe_device, (any))) + +/** + * xe_any_to_drm() - Obtain the &drm_device pointer. + * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer + * + * Return: the @drm_device pointer or backpointer. + */ +#define xe_any_to_drm(any) \ + _Generic((any), \ + __xe_any_to_self_assoc(struct drm_device, (any)), \ + default : &xe_any_to_xe(any)->drm) + +/** + * xe_any_to_dev() - Obtain the &device pointer. + * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer + * + * Return: the @device pointer or backpointer. + */ +#define xe_any_to_dev(any) \ + _Generic((any), \ + __xe_any_to_self_assoc(struct device, (any)), \ + __xe_any_to_other_assoc(, pci_dev, device, (any)), \ + default : xe_any_to_drm(any)->dev) + +/** + * xe_any_to_pdev() - Obtain the &pci_dev pointer. + * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer + * + * Return: the @pci_dev pointer or backpointer. + */ +#define xe_any_to_pdev(any) \ + _Generic((any), \ + __xe_any_to_self_assoc(struct pci_dev, (any)), \ + default : to_pci_dev(xe_any_to_dev(any))) + +#define __xe_tile_to_id(p) ((p)->id) +#define __xe_gt_to_id(p) ((p)->info.id) + +/** + * xe_any_id() - Get the identifier of the underlying object. + * @any: the &pci_dev or the &xe_device or &xe_tile or &xe_gt pointer + * + * Return: the identifier of the object, or 0 if not applicable/available. + */ +#define xe_any_id(any) \ + _Generic((any), \ + __xe_any_to_other_assoc(/* */, xe_tile, id, (any)), \ + __xe_any_to_other_assoc(const, xe_tile, id, (any)), \ + __xe_any_to_other_assoc(/* */, xe_gt, id, (any)), \ + __xe_any_to_other_assoc(const, xe_gt, id, (any)), \ + default : 0) + +#endif diff --git a/drivers/gpu/drm/xe/xe_debugfs.c b/drivers/gpu/drm/xe/xe_debugfs.c index 8de78cd0aa03..28135f84e286 100644 --- a/drivers/gpu/drm/xe/xe_debugfs.c +++ b/drivers/gpu/drm/xe/xe_debugfs.c @@ -22,6 +22,7 @@ #include "xe_guc_ads.h" #include "xe_hw_engine.h" #include "xe_mmio.h" +#include "xe_pagefault.h" #include "xe_pcode.h" #include "xe_pm.h" #include "xe_psmi.h" @@ -42,6 +43,7 @@ DECLARE_FAULT_ATTR(gt_reset_failure); DECLARE_FAULT_ATTR(inject_csc_hw_error); +DECLARE_FAULT_ATTR(wedge_cold_reset); static bool csc_hw_error_available(struct xe_device *xe) { @@ -62,6 +64,8 @@ static struct { { .name = "inject_csc_hw_error", .attr = &inject_csc_hw_error, .is_visible = csc_hw_error_available }, + { .name = "wedge_cold_reset", + .attr = &wedge_cold_reset }, }; /* @@ -76,6 +80,7 @@ bool xe_fault_##name(void) \ FAULT_ACTION(gt_reset, gt_reset_failure) FAULT_ACTION(csc_hw_error, inject_csc_hw_error) +FAULT_ACTION(wedge_cold_reset, wedge_cold_reset) static void xe_fault_inject_debugfs_register(struct xe_device *xe, struct dentry *root) @@ -194,6 +199,15 @@ static int sriov_info(struct seq_file *m, void *data) return 0; } +static int pagefault_info(struct seq_file *m, void *data) +{ + struct xe_device *xe = node_to_xe(m->private); + struct drm_printer p = drm_seq_file_printer(m); + + xe_pagefault_print_info(xe, &p); + return 0; +} + static int workarounds(struct xe_device *xe, struct drm_printer *p) { guard(xe_pm_runtime)(xe); @@ -285,6 +299,7 @@ static const struct drm_info_list debugfs_list[] = { {"info", info, 0}, { .name = "sriov_info", .show = sriov_info, }, { .name = "workarounds", .show = workaround_info, }, + { .name = "pagefault_info", .show = pagefault_info, }, }; static const struct drm_info_list pcode_info_debugfs[] = { diff --git a/drivers/gpu/drm/xe/xe_debugfs.h b/drivers/gpu/drm/xe/xe_debugfs.h index cd56f7442b99..0dcd28fd7dc0 100644 --- a/drivers/gpu/drm/xe/xe_debugfs.h +++ b/drivers/gpu/drm/xe/xe_debugfs.h @@ -13,10 +13,12 @@ struct xe_device; #ifdef CONFIG_DEBUG_FS bool xe_fault_gt_reset(void); bool xe_fault_csc_hw_error(void); +bool xe_fault_wedge_cold_reset(void); void xe_debugfs_register(struct xe_device *xe); #else static inline bool xe_fault_gt_reset(void) { return false; } static inline bool xe_fault_csc_hw_error(void) { return false; } +static inline bool xe_fault_wedge_cold_reset(void) { return false; } static inline void xe_debugfs_register(struct xe_device *xe) { } #endif diff --git a/drivers/gpu/drm/xe/xe_defaults.h b/drivers/gpu/drm/xe/xe_defaults.h index c8ae1d5f3d60..0884224ef7c7 100644 --- a/drivers/gpu/drm/xe/xe_defaults.h +++ b/drivers/gpu/drm/xe/xe_defaults.h @@ -22,5 +22,6 @@ #define XE_DEFAULT_WEDGED_MODE XE_WEDGED_MODE_UPON_CRITICAL_ERROR #define XE_DEFAULT_WEDGED_MODE_STR "upon-critical-error" #define XE_DEFAULT_SVM_NOTIFIER_SIZE 512 +#define XE_DEFAULT_NUM_PF_WORK 2 #endif diff --git a/drivers/gpu/drm/xe/xe_device.c b/drivers/gpu/drm/xe/xe_device.c index ee732e5495f7..396d02eb2af8 100644 --- a/drivers/gpu/drm/xe/xe_device.c +++ b/drivers/gpu/drm/xe/xe_device.c @@ -48,6 +48,7 @@ #include "xe_i2c.h" #include "xe_irq.h" #include "xe_late_bind_fw.h" +#include "xe_log.h" #include "xe_mmio.h" #include "xe_module.h" #include "xe_nvm.h" @@ -513,6 +514,17 @@ struct xe_device *xe_device_create(struct pci_dev *pdev) } ALLOW_ERROR_INJECTION(xe_device_create, ERRNO); /* See xe_pci_probe() */ +static void xe_device_parse_modparam(struct xe_device *xe) +{ + xe->atomic_svm_timeslice_ms = 5; + xe->min_run_period_lr_ms = 5; + xe->info.num_pf_work = xe_modparam.num_pf_work; + if (xe->info.num_pf_work < 1) + xe->info.num_pf_work = 1; + else if (xe->info.num_pf_work > XE_PAGEFAULT_WORK_MAX) + xe->info.num_pf_work = XE_PAGEFAULT_WORK_MAX; +} + /** * xe_device_init_early() - Initialize a new &xe_device instance * @xe: the &xe_device to initialize @@ -539,8 +551,7 @@ int xe_device_init_early(struct xe_device *xe) if (err) return err; - xe->atomic_svm_timeslice_ms = 5; - xe->min_run_period_lr_ms = 5; + xe_device_parse_modparam(xe); err = xe_irq_init(xe); if (err) @@ -1432,6 +1443,9 @@ void xe_device_set_wedged_method(struct xe_device *xe, unsigned long method) xe->wedged.method = method; } +#define WEDGED_URL "https://docs.kernel.org/gpu/drm-uapi.html#device-wedging" +#define XE_BUG_URL "https://gitlab.freedesktop.org/drm/xe/kernel/issues/new" + /** * xe_device_declare_wedged - Declare device wedged * @xe: xe device instance @@ -1463,12 +1477,12 @@ void xe_device_declare_wedged(struct xe_device *xe) if (!atomic_xchg(&xe->wedged.flag, 1)) { xe->needs_flr_on_fini = true; xe_pm_runtime_get_noresume(xe); - drm_err(&xe->drm, - "CRITICAL: Xe has declared device %s as wedged.\n" - "IOCTLs and executions are blocked.\n" - "For recovery procedure, refer to https://docs.kernel.org/gpu/drm-uapi.html#device-wedging\n" - "Please file a _new_ bug report at https://gitlab.freedesktop.org/drm/xe/kernel/issues/new\n", - dev_name(xe->drm.dev)); + + xe_log_err_fatal(xe, WEDGED, -EIO, "Device declared wedged!\n"); + xe_err_once(xe, "IOCTLs and executions are now blocked!\n" + "For recovery procedure, refer to %s\n" + "Please file a _new_ bug report at %s\n", + WEDGED_URL, XE_BUG_URL); } for_each_gt(gt, xe, id) diff --git a/drivers/gpu/drm/xe/xe_device_types.h b/drivers/gpu/drm/xe/xe_device_types.h index 03a7bb08adf7..180d450a6deb 100644 --- a/drivers/gpu/drm/xe/xe_device_types.h +++ b/drivers/gpu/drm/xe/xe_device_types.h @@ -140,6 +140,8 @@ struct xe_device { u8 revid; /** @info.step: stepping information for each IP */ struct xe_step_info step; + /** @info.num_pf_work: Number of page fault work thread */ + int num_pf_work; /** @info.dma_mask_size: DMA address bits */ u8 dma_mask_size; /** @info.vram_flags: Vram flags */ @@ -320,18 +322,19 @@ struct xe_device { struct xarray asid_to_vm; /** @usm.next_asid: next ASID, used to cyclical alloc asids */ u32 next_asid; + /** @usm.current_pf_work: current page fault work item */ + u32 current_pf_work; /** @usm.lock: protects UM state */ struct rw_semaphore lock; - /** @usm.pf_wq: page fault work queue, unbound, high priority */ - struct workqueue_struct *pf_wq; - /* - * We pick 4 here because, in the current implementation, it - * yields the best bandwidth utilization of the kernel paging - * engine. - */ -#define XE_PAGEFAULT_QUEUE_COUNT 4 - /** @usm.pf_queue: Page fault queues */ - struct xe_pagefault_queue pf_queue[XE_PAGEFAULT_QUEUE_COUNT]; + /** @usm.pagefault_wq: page fault work queue, unbound, high priority */ + struct workqueue_struct *pagefault_wq; + /** @usm.prefetch_wq: threaded prefetch work queue, unbound */ + struct workqueue_struct *prefetch_wq; +#define XE_PAGEFAULT_WORK_MAX 8 + /** @usm.pf_workers: Page fault workers */ + struct xe_pagefault_work pf_workers[XE_PAGEFAULT_WORK_MAX]; + /** @usm.pf_queue: Page fault queue */ + struct xe_pagefault_queue pf_queue; #if IS_ENABLED(CONFIG_DRM_XE_PAGEMAP) /** @usm.dpagemap_shrinker: Shrinker for unused pagemaps */ struct drm_pagemap_shrinker *dpagemap_shrinker; diff --git a/drivers/gpu/drm/xe/xe_drm_ras.c b/drivers/gpu/drm/xe/xe_drm_ras.c index 984fa67b9015..78184b6ea7d4 100644 --- a/drivers/gpu/drm/xe/xe_drm_ras.c +++ b/drivers/gpu/drm/xe/xe_drm_ras.c @@ -186,6 +186,37 @@ null_info: } /** + * xe_drm_ras_event() - Report drm-ras error event to userspace + * @xe: xe device structure + * @component: error component (see &enum drm_xe_ras_error_component) + * @severity: error severity (see &enum drm_xe_ras_error_severity) + * @value: value of error counter + * + * Report an error-event to userspace. + */ +void xe_drm_ras_event(struct xe_device *xe, u32 component, u32 severity, u32 value) +{ + struct xe_drm_ras *ras = &xe->ras; + struct xe_drm_ras_counter *info = ras->info[severity]; + struct drm_ras_node *node; + int ret; + + /* Event is supported only if drm-ras is enabled */ + if (!xe->info.has_drm_ras) + return; + + node = &ras->node[severity]; + + if (!info || !info[component].name) + return; + + ret = drm_ras_nl_error_event(node, component, info[component].name, value); + if (ret) + drm_err_ratelimited(&xe->drm, "drm-ras error-event failed: %d for %s %s\n", ret, + info[component].name, error_severity[severity]); +} + +/** * xe_drm_ras_init() - Initialize DRM RAS * @xe: xe device instance * diff --git a/drivers/gpu/drm/xe/xe_drm_ras.h b/drivers/gpu/drm/xe/xe_drm_ras.h index 365c70e93e82..723229b2cffb 100644 --- a/drivers/gpu/drm/xe/xe_drm_ras.h +++ b/drivers/gpu/drm/xe/xe_drm_ras.h @@ -5,11 +5,14 @@ #ifndef _XE_DRM_RAS_H_ #define _XE_DRM_RAS_H_ +#include <linux/types.h> + struct xe_device; #define for_each_error_severity(i) \ for (i = 0; i < DRM_XE_RAS_ERR_SEV_MAX; i++) int xe_drm_ras_init(struct xe_device *xe); +void xe_drm_ras_event(struct xe_device *xe, u32 component, u32 severity, u32 value); #endif diff --git a/drivers/gpu/drm/xe/xe_exec_queue.c b/drivers/gpu/drm/xe/xe_exec_queue.c index cd053de3c6b5..c4213bb9c137 100644 --- a/drivers/gpu/drm/xe/xe_exec_queue.c +++ b/drivers/gpu/drm/xe/xe_exec_queue.c @@ -207,9 +207,6 @@ static struct xe_exec_queue *__xe_exec_queue_alloc(struct xe_device *xe, struct xe_gt *gt = hwe->gt; int err; - /* only kernel queues can be permanent */ - XE_WARN_ON((flags & EXEC_QUEUE_FLAG_PERMANENT) && !(flags & EXEC_QUEUE_FLAG_KERNEL)); - q = kzalloc_flex(*q, lrc, width); if (!q) return ERR_PTR(-ENOMEM); @@ -1121,6 +1118,9 @@ static int exec_queue_user_ext_set_property(struct xe_device *xe, if (!exec_queue_set_property_funcs[idx]) return -EINVAL; + if (XE_IOCTL_DBG(xe, *properties & BIT_ULL(idx))) + return -EINVAL; + *properties |= BIT_ULL(idx); err = exec_queue_user_ext_check(q, *properties); if (XE_IOCTL_DBG(xe, err)) diff --git a/drivers/gpu/drm/xe/xe_exec_queue_types.h b/drivers/gpu/drm/xe/xe_exec_queue_types.h index 53b6c0bf4849..95f75d61a647 100644 --- a/drivers/gpu/drm/xe/xe_exec_queue_types.h +++ b/drivers/gpu/drm/xe/xe_exec_queue_types.h @@ -70,6 +70,13 @@ struct xe_exec_queue_group { spinlock_t suspend_lock; /** @sync_pending: CGP_SYNC_DONE g2h response pending */ bool sync_pending; + /** + * @cgp_update_q: Queue that issued the currently outstanding (sent) + * CGP_SYNC or REGISTER_CONTEXT_MULTI_QUEUE; NULL when none is + * outstanding. Used during VF recovery to identify and replay the + * message whose CGP_SYNC_DONE was not received. + */ + struct xe_exec_queue *cgp_update_q; /** @banned: Group banned */ bool banned; /** @stopped: Group is stopped, protected by list_lock */ @@ -128,20 +135,18 @@ struct xe_exec_queue { /* queue used for kernel submission only */ #define EXEC_QUEUE_FLAG_KERNEL BIT(0) -/* kernel engine only destroyed at driver unload */ -#define EXEC_QUEUE_FLAG_PERMANENT BIT(1) /* for VM jobs. Caller needs to hold rpm ref when creating queue with this flag */ -#define EXEC_QUEUE_FLAG_VM BIT(2) +#define EXEC_QUEUE_FLAG_VM BIT(1) /* child of VM queue for multi-tile VM jobs */ -#define EXEC_QUEUE_FLAG_BIND_ENGINE_CHILD BIT(3) +#define EXEC_QUEUE_FLAG_BIND_ENGINE_CHILD BIT(2) /* kernel exec_queue only, set priority to highest level */ -#define EXEC_QUEUE_FLAG_HIGH_PRIORITY BIT(4) +#define EXEC_QUEUE_FLAG_HIGH_PRIORITY BIT(3) /* flag to indicate low latency hint to guc */ -#define EXEC_QUEUE_FLAG_LOW_LATENCY BIT(5) +#define EXEC_QUEUE_FLAG_LOW_LATENCY BIT(4) /* for migration (kernel copy, clear, bind) jobs */ -#define EXEC_QUEUE_FLAG_MIGRATE BIT(6) +#define EXEC_QUEUE_FLAG_MIGRATE BIT(5) /* for programming COMMON_SLICE_CHICKEN3 on first submission */ -#define EXEC_QUEUE_FLAG_DISABLE_STATE_CACHE_PERF_FIX BIT(7) +#define EXEC_QUEUE_FLAG_DISABLE_STATE_CACHE_PERF_FIX BIT(6) /** * @flags: flags for this exec queue, should statically setup aside from ban diff --git a/drivers/gpu/drm/xe/xe_gsc.c b/drivers/gpu/drm/xe/xe_gsc.c index aab59dc647fb..524ac56bdcc7 100644 --- a/drivers/gpu/drm/xe/xe_gsc.c +++ b/drivers/gpu/drm/xe/xe_gsc.c @@ -478,8 +478,7 @@ int xe_gsc_init_post_hwconfig(struct xe_gsc *gsc) q = xe_exec_queue_create(xe, NULL, BIT(hwe->logical_instance), 1, hwe, - EXEC_QUEUE_FLAG_KERNEL | - EXEC_QUEUE_FLAG_PERMANENT, 0); + EXEC_QUEUE_FLAG_KERNEL, 0); if (IS_ERR(q)) { xe_gt_err(gt, "Failed to create queue for GSC submission\n"); return PTR_ERR(q); diff --git a/drivers/gpu/drm/xe/xe_gt.c b/drivers/gpu/drm/xe/xe_gt.c index dfdacc0f6de9..6805e0d3bf21 100644 --- a/drivers/gpu/drm/xe/xe_gt.c +++ b/drivers/gpu/drm/xe/xe_gt.c @@ -48,6 +48,7 @@ #include "xe_hw_engine_class_sysfs.h" #include "xe_irq.h" #include "xe_lmtt.h" +#include "xe_log.h" #include "xe_lrc.h" #include "xe_map.h" #include "xe_migrate.h" @@ -925,7 +926,7 @@ static void gt_reset_worker(struct work_struct *w) if (!xe_device_uc_enabled(gt_to_xe(gt))) goto err_pm_put; - xe_gt_info(gt, "reset started\n"); + xe_log_info(gt, GT, "reset started\n"); if (xe_fault_gt_reset()) { err = -ECANCELED; @@ -964,7 +965,7 @@ static void gt_reset_worker(struct work_struct *w) /* Pair with get while enqueueing the work in xe_gt_reset_async() */ xe_pm_runtime_put(gt_to_xe(gt)); - xe_gt_info(gt, "reset done\n"); + xe_log_info(gt, GT, "reset done\n"); return; @@ -973,7 +974,7 @@ err_out: XE_WARN_ON(xe_uc_start(>->uc)); err_fail: - xe_gt_err(gt, "reset failed (%pe)\n", ERR_PTR(err)); + xe_log_err_fatal(gt, GT, err, "reset failed\n"); xe_device_declare_wedged(gt_to_xe(gt)); err_pm_put: xe_pm_runtime_put(gt_to_xe(gt)); diff --git a/drivers/gpu/drm/xe/xe_gt_debugfs.c b/drivers/gpu/drm/xe/xe_gt_debugfs.c index c38bcacb27e4..ea78b57b1c31 100644 --- a/drivers/gpu/drm/xe/xe_gt_debugfs.c +++ b/drivers/gpu/drm/xe/xe_gt_debugfs.c @@ -9,7 +9,9 @@ #include <drm/drm_debugfs.h> #include <drm/drm_managed.h> +#include <linux/math.h> +#include "regs/xe_gt_regs.h" #include "xe_device.h" #include "xe_force_wake.h" #include "xe_gt.h" @@ -22,6 +24,7 @@ #include "xe_guc_hwconfig.h" #include "xe_hw_engine.h" #include "xe_lrc.h" +#include "xe_mmio.h" #include "xe_mocs.h" #include "xe_pat.h" #include "xe_pm.h" @@ -336,6 +339,80 @@ static int force_reset_sync_show(struct seq_file *s, void *unused) } DEFINE_SHOW_STORE_ATTRIBUTE(force_reset_sync); +#define U1_15_ONE 0x8000 +#define U1_15_INT_BITS GENMASK(15, 15) +#define U1_15_FRACTION_BITS GENMASK(14, 0) + +static void u1_15_decode(u16 num, u16 *i, u32 *frac) +{ + /* + * In U1.15 format, uppermost bit is integer value and the + * rest 15 are the fraction. + */ + + *i = FIELD_GET(U1_15_INT_BITS, num); + *frac = FIELD_GET(U1_15_FRACTION_BITS, num); +} + +static void u1_15_decode_decimal(u16 value, u16 *i, u32 *frac, int digits) +{ + u1_15_decode(value, i, frac); + *frac = (*frac * int_pow(10, digits)) / (FIELD_MAX(U1_15_FRACTION_BITS) + 1); +} + +static int gt_ia_bias_show(struct seq_file *s, void *unused) +{ + struct xe_gt *gt = s->private; + struct xe_device *xe = gt_to_xe(gt); + u32 val; + u32 ia_frac, gt_frac; + u16 ia_raw, gt_raw; + u16 ia_int, gt_int; + + guard(xe_pm_runtime)(xe); + val = xe_mmio_read32(>->mmio, GT_IA_PERF_BIAS_REG); + + ia_raw = REG_FIELD_GET(IA_BIAS, val); + gt_raw = REG_FIELD_GET(GT_BIAS, val); + + u1_15_decode_decimal(ia_raw, &ia_int, &ia_frac, 4); + u1_15_decode_decimal(gt_raw, >_int, >_frac, 4); + + seq_printf(s, "0x%x (GT: %u.%04u, IA: %u.%04u)\n", + val, gt_int, gt_frac, ia_int, ia_frac); + + return 0; +} + +static ssize_t gt_ia_bias_write(struct file *file, + const char __user *userbuf, + size_t count, loff_t *ppos) +{ + struct seq_file *s = file->private_data; + struct xe_gt *gt = s->private; + struct xe_device *xe = gt_to_xe(gt); + u32 val; + int ret; + + ret = kstrtou32_from_user(userbuf, count, 0, &val); + if (ret) + return ret; + + if (REG_FIELD_GET(IA_BIAS, val) > U1_15_ONE || + REG_FIELD_GET(GT_BIAS, val) > U1_15_ONE) + return -EINVAL; + + if (REG_FIELD_GET(IA_BIAS, val) < IA_BIAS_DEFAULT || + REG_FIELD_GET(GT_BIAS, val) < GT_BIAS_DEFAULT) + return -EINVAL; + + guard(xe_pm_runtime)(xe); + xe_mmio_write32(>->mmio, GT_IA_PERF_BIAS_REG, val); + + return count; +} +DEFINE_SHOW_STORE_ATTRIBUTE(gt_ia_bias); + void xe_gt_debugfs_register(struct xe_gt *gt) { struct xe_device *xe = gt_to_xe(gt); @@ -378,6 +455,9 @@ void xe_gt_debugfs_register(struct xe_gt *gt) ARRAY_SIZE(pf_only_debugfs_list), root, minor); + if (xe_gt_is_main_type(gt) && !IS_DGFX(xe) && !IS_SRIOV_VF(xe)) + debugfs_create_file("gt_ia_bias", 0600, root, gt, >_ia_bias_fops); + xe_uc_debugfs_register(>->uc, root); if (IS_SRIOV_PF(xe)) diff --git a/drivers/gpu/drm/xe/xe_gt_idle.c b/drivers/gpu/drm/xe/xe_gt_idle.c index 04b24e1c8b78..7dc9873aef54 100644 --- a/drivers/gpu/drm/xe/xe_gt_idle.c +++ b/drivers/gpu/drm/xe/xe_gt_idle.c @@ -248,7 +248,8 @@ int xe_gt_idle_pg_print(struct xe_gt *gt, struct drm_printer *p) pg_status = xe_mmio_read32(>->mmio, POWERGATE_DOMAIN_STATUS); } - if (gt->info.engine_mask & XE_HW_ENGINE_RCS_MASK) { + if (gt->info.engine_mask & + (XE_HW_ENGINE_RCS_MASK | XE_HW_ENGINE_CCS_MASK)) { drm_printf(p, "Render Power Gating Enabled: %s\n", str_yes_no(pg_enabled & RENDER_POWERGATE_ENABLE)); diff --git a/drivers/gpu/drm/xe/xe_gt_printk.h b/drivers/gpu/drm/xe/xe_gt_printk.h index 1313d32862db..1598098d2987 100644 --- a/drivers/gpu/drm/xe/xe_gt_printk.h +++ b/drivers/gpu/drm/xe/xe_gt_printk.h @@ -35,6 +35,9 @@ #define xe_gt_dbg(_gt, _fmt, ...) \ xe_gt_printk((_gt), dbg, _fmt, ##__VA_ARGS__) +#define xe_gt_dbg_ratelimited(_gt, _fmt, ...) \ + xe_gt_printk((_gt), dbg_ratelimited, _fmt, ##__VA_ARGS__) + #define xe_gt_WARN_type(_gt, _type, _condition, _fmt, ...) \ xe_tile_WARN##_type((_gt)->tile, _condition, _fmt, ## __VA_ARGS__) diff --git a/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c b/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c index 2c9b85b84b1b..be0a413ee17c 100644 --- a/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c +++ b/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c @@ -668,7 +668,7 @@ static int pf_config_bulk_set_u64_done(struct xe_gt *gt, unsigned int first, uns } /** - * xe_gt_sriov_pf_config_bulk_set_ggtt - Provision many VFs with GGTT. + * xe_gt_sriov_pf_config_bulk_set_ggtt_locked() - Provision many VFs with GGTT. * @gt: the &xe_gt (can't be media) * @vfid: starting VF identifier (can't be 0) * @num_vfs: number of VFs to provision @@ -678,31 +678,49 @@ static int pf_config_bulk_set_u64_done(struct xe_gt *gt, unsigned int first, uns * * Return: 0 on success or a negative error code on failure. */ -int xe_gt_sriov_pf_config_bulk_set_ggtt(struct xe_gt *gt, unsigned int vfid, - unsigned int num_vfs, u64 size) +int xe_gt_sriov_pf_config_bulk_set_ggtt_locked(struct xe_gt *gt, unsigned int vfid, + unsigned int num_vfs, u64 size) { unsigned int n; int err = 0; xe_gt_assert(gt, vfid); xe_gt_assert(gt, xe_gt_is_main_type(gt)); + lockdep_assert_held(xe_gt_sriov_pf_master_mutex(gt)); if (!num_vfs) return 0; - mutex_lock(xe_gt_sriov_pf_master_mutex(gt)); for (n = vfid; n < vfid + num_vfs; n++) { err = pf_provision_vf_ggtt(gt, n, size); if (err) break; } - mutex_unlock(xe_gt_sriov_pf_master_mutex(gt)); return pf_config_bulk_set_u64_done(gt, vfid, num_vfs, size, - xe_gt_sriov_pf_config_get_ggtt, + pf_get_vf_config_ggtt, "GGTT", n, err); } +/** + * xe_gt_sriov_pf_config_bulk_set_ggtt() - Provision many VFs with GGTT. + * @gt: the &xe_gt (can't be media) + * @vfid: starting VF identifier (can't be 0) + * @num_vfs: number of VFs to provision + * @size: requested GGTT size + * + * This function can only be called on PF. + * + * Return: 0 on success or a negative error code on failure. + */ +int xe_gt_sriov_pf_config_bulk_set_ggtt(struct xe_gt *gt, unsigned int vfid, + unsigned int num_vfs, u64 size) +{ + guard(mutex)(xe_gt_sriov_pf_master_mutex(gt)); + + return xe_gt_sriov_pf_config_bulk_set_ggtt_locked(gt, vfid, num_vfs, size); +} + /* Return: size of the largest continuous GGTT region */ static u64 pf_get_max_ggtt(struct xe_gt *gt) { @@ -775,10 +793,9 @@ int xe_gt_sriov_pf_config_set_fair_ggtt(struct xe_gt *gt, unsigned int vfid, xe_gt_assert(gt, num_vfs); xe_gt_assert(gt, xe_gt_is_main_type(gt)); - mutex_lock(xe_gt_sriov_pf_master_mutex(gt)); - fair = pf_estimate_fair_ggtt(gt, num_vfs); - mutex_unlock(xe_gt_sriov_pf_master_mutex(gt)); + guard(mutex)(xe_gt_sriov_pf_master_mutex(gt)); + fair = pf_estimate_fair_ggtt(gt, num_vfs); if (!fair) return -ENOSPC; @@ -787,7 +804,7 @@ int xe_gt_sriov_pf_config_set_fair_ggtt(struct xe_gt *gt, unsigned int vfid, xe_gt_sriov_info(gt, "Using non-profile provisioning (%s %llu vs %llu)\n", "GGTT", fair, profile); - return xe_gt_sriov_pf_config_bulk_set_ggtt(gt, vfid, num_vfs, fair); + return xe_gt_sriov_pf_config_bulk_set_ggtt_locked(gt, vfid, num_vfs, fair); } /** @@ -1099,7 +1116,7 @@ static int pf_config_bulk_set_u32_done(struct xe_gt *gt, unsigned int first, uns } /** - * xe_gt_sriov_pf_config_bulk_set_ctxs - Provision many VFs with GuC context IDs. + * xe_gt_sriov_pf_config_bulk_set_ctxs_locked() - Provision many VFs with GuC context IDs. * @gt: the &xe_gt * @vfid: starting VF identifier * @num_vfs: number of VFs to provision @@ -1109,30 +1126,48 @@ static int pf_config_bulk_set_u32_done(struct xe_gt *gt, unsigned int first, uns * * Return: 0 on success or a negative error code on failure. */ -int xe_gt_sriov_pf_config_bulk_set_ctxs(struct xe_gt *gt, unsigned int vfid, - unsigned int num_vfs, u32 num_ctxs) +int xe_gt_sriov_pf_config_bulk_set_ctxs_locked(struct xe_gt *gt, unsigned int vfid, + unsigned int num_vfs, u32 num_ctxs) { unsigned int n; int err = 0; xe_gt_assert(gt, vfid); + lockdep_assert_held(xe_gt_sriov_pf_master_mutex(gt)); if (!num_vfs) return 0; - mutex_lock(xe_gt_sriov_pf_master_mutex(gt)); for (n = vfid; n < vfid + num_vfs; n++) { err = pf_provision_vf_ctxs(gt, n, num_ctxs); if (err) break; } - mutex_unlock(xe_gt_sriov_pf_master_mutex(gt)); return pf_config_bulk_set_u32_done(gt, vfid, num_vfs, num_ctxs, - xe_gt_sriov_pf_config_get_ctxs, + pf_get_vf_config_ctxs, "GuC context IDs", no_unit, n, err); } +/** + * xe_gt_sriov_pf_config_bulk_set_ctxs() - Provision many VFs with GuC context IDs. + * @gt: the &xe_gt + * @vfid: starting VF identifier + * @num_vfs: number of VFs to provision + * @num_ctxs: requested number of GuC contexts IDs (0 to release) + * + * This function can only be called on PF. + * + * Return: 0 on success or a negative error code on failure. + */ +int xe_gt_sriov_pf_config_bulk_set_ctxs(struct xe_gt *gt, unsigned int vfid, + unsigned int num_vfs, u32 num_ctxs) +{ + guard(mutex)(xe_gt_sriov_pf_master_mutex(gt)); + + return xe_gt_sriov_pf_config_bulk_set_ctxs_locked(gt, vfid, num_vfs, num_ctxs); +} + static u32 pf_profile_fair_ctxs(struct xe_gt *gt, unsigned int num_vfs) { bool admin_only_pf = xe_sriov_pf_admin_only(gt_to_xe(gt)); @@ -1181,10 +1216,9 @@ int xe_gt_sriov_pf_config_set_fair_ctxs(struct xe_gt *gt, unsigned int vfid, xe_gt_assert(gt, vfid); xe_gt_assert(gt, num_vfs); - mutex_lock(xe_gt_sriov_pf_master_mutex(gt)); - fair = pf_estimate_fair_ctxs(gt, num_vfs); - mutex_unlock(xe_gt_sriov_pf_master_mutex(gt)); + guard(mutex)(xe_gt_sriov_pf_master_mutex(gt)); + fair = pf_estimate_fair_ctxs(gt, num_vfs); if (!fair) return -ENOSPC; @@ -1193,7 +1227,7 @@ int xe_gt_sriov_pf_config_set_fair_ctxs(struct xe_gt *gt, unsigned int vfid, xe_gt_sriov_info(gt, "Using non-profile provisioning (%s %u vs %u)\n", "GuC context IDs", fair, profile); - return xe_gt_sriov_pf_config_bulk_set_ctxs(gt, vfid, num_vfs, fair); + return xe_gt_sriov_pf_config_bulk_set_ctxs_locked(gt, vfid, num_vfs, fair); } static u32 pf_get_min_spare_dbs(struct xe_gt *gt) @@ -1363,7 +1397,7 @@ int xe_gt_sriov_pf_config_set_dbs(struct xe_gt *gt, unsigned int vfid, u32 num_d } /** - * xe_gt_sriov_pf_config_bulk_set_dbs - Provision many VFs with GuC context IDs. + * xe_gt_sriov_pf_config_bulk_set_dbs_locked() - Provision many VFs with GuC doorbells. * @gt: the &xe_gt * @vfid: starting VF identifier (can't be 0) * @num_vfs: number of VFs to provision @@ -1373,30 +1407,48 @@ int xe_gt_sriov_pf_config_set_dbs(struct xe_gt *gt, unsigned int vfid, u32 num_d * * Return: 0 on success or a negative error code on failure. */ -int xe_gt_sriov_pf_config_bulk_set_dbs(struct xe_gt *gt, unsigned int vfid, - unsigned int num_vfs, u32 num_dbs) +int xe_gt_sriov_pf_config_bulk_set_dbs_locked(struct xe_gt *gt, unsigned int vfid, + unsigned int num_vfs, u32 num_dbs) { unsigned int n; int err = 0; xe_gt_assert(gt, vfid); + lockdep_assert_held(xe_gt_sriov_pf_master_mutex(gt)); if (!num_vfs) return 0; - mutex_lock(xe_gt_sriov_pf_master_mutex(gt)); for (n = vfid; n < vfid + num_vfs; n++) { err = pf_provision_vf_dbs(gt, n, num_dbs); if (err) break; } - mutex_unlock(xe_gt_sriov_pf_master_mutex(gt)); return pf_config_bulk_set_u32_done(gt, vfid, num_vfs, num_dbs, - xe_gt_sriov_pf_config_get_dbs, + pf_get_vf_config_dbs, "GuC doorbell IDs", no_unit, n, err); } +/** + * xe_gt_sriov_pf_config_bulk_set_dbs() - Provision many VFs with GuC doorbells. + * @gt: the &xe_gt + * @vfid: starting VF identifier (can't be 0) + * @num_vfs: number of VFs to provision + * @num_dbs: requested number of GuC doorbell IDs (0 to release) + * + * This function can only be called on PF. + * + * Return: 0 on success or a negative error code on failure. + */ +int xe_gt_sriov_pf_config_bulk_set_dbs(struct xe_gt *gt, unsigned int vfid, + unsigned int num_vfs, u32 num_dbs) +{ + guard(mutex)(xe_gt_sriov_pf_master_mutex(gt)); + + return xe_gt_sriov_pf_config_bulk_set_dbs_locked(gt, vfid, num_vfs, num_dbs); +} + static u32 pf_profile_fair_dbs(struct xe_gt *gt, unsigned int num_vfs) { bool admin_only_pf = xe_sriov_pf_admin_only(gt_to_xe(gt)); @@ -1446,10 +1498,9 @@ int xe_gt_sriov_pf_config_set_fair_dbs(struct xe_gt *gt, unsigned int vfid, xe_gt_assert(gt, vfid); xe_gt_assert(gt, num_vfs); - mutex_lock(xe_gt_sriov_pf_master_mutex(gt)); - fair = pf_estimate_fair_dbs(gt, num_vfs); - mutex_unlock(xe_gt_sriov_pf_master_mutex(gt)); + guard(mutex)(xe_gt_sriov_pf_master_mutex(gt)); + fair = pf_estimate_fair_dbs(gt, num_vfs); if (!fair) return -ENOSPC; @@ -1458,7 +1509,7 @@ int xe_gt_sriov_pf_config_set_fair_dbs(struct xe_gt *gt, unsigned int vfid, xe_gt_sriov_info(gt, "Using non-profile provisioning (%s %u vs %u)\n", "GuC doorbell IDs", fair, profile); - return xe_gt_sriov_pf_config_bulk_set_dbs(gt, vfid, num_vfs, fair); + return xe_gt_sriov_pf_config_bulk_set_dbs_locked(gt, vfid, num_vfs, fair); } static u64 pf_get_lmem_alignment(struct xe_gt *gt) diff --git a/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h b/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h index 2ec62c12ad5c..a56e63f3660a 100644 --- a/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h +++ b/drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h @@ -18,18 +18,24 @@ int xe_gt_sriov_pf_config_set_fair_ggtt(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs); int xe_gt_sriov_pf_config_bulk_set_ggtt(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs, u64 size); +int xe_gt_sriov_pf_config_bulk_set_ggtt_locked(struct xe_gt *gt, + unsigned int vfid, unsigned int num_vfs, u64 size); u32 xe_gt_sriov_pf_config_get_ctxs(struct xe_gt *gt, unsigned int vfid); int xe_gt_sriov_pf_config_set_ctxs(struct xe_gt *gt, unsigned int vfid, u32 num_ctxs); int xe_gt_sriov_pf_config_set_fair_ctxs(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs); int xe_gt_sriov_pf_config_bulk_set_ctxs(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs, u32 num_ctxs); +int xe_gt_sriov_pf_config_bulk_set_ctxs_locked(struct xe_gt *gt, unsigned int vfid, + unsigned int num_vfs, u32 num_ctxs); u32 xe_gt_sriov_pf_config_get_dbs(struct xe_gt *gt, unsigned int vfid); int xe_gt_sriov_pf_config_set_dbs(struct xe_gt *gt, unsigned int vfid, u32 num_dbs); int xe_gt_sriov_pf_config_set_fair_dbs(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs); int xe_gt_sriov_pf_config_bulk_set_dbs(struct xe_gt *gt, unsigned int vfid, unsigned int num_vfs, u32 num_dbs); +int xe_gt_sriov_pf_config_bulk_set_dbs_locked(struct xe_gt *gt, unsigned int vfid, + unsigned int num_vfs, u32 num_dbs); u64 xe_gt_sriov_pf_config_get_lmem(struct xe_gt *gt, unsigned int vfid); int xe_gt_sriov_pf_config_set_lmem(struct xe_gt *gt, unsigned int vfid, u64 size); diff --git a/drivers/gpu/drm/xe/xe_gt_sriov_printk.h b/drivers/gpu/drm/xe/xe_gt_sriov_printk.h index d3457d608db8..ada0ee9f45e5 100644 --- a/drivers/gpu/drm/xe/xe_gt_sriov_printk.h +++ b/drivers/gpu/drm/xe/xe_gt_sriov_printk.h @@ -27,6 +27,9 @@ #define xe_gt_sriov_dbg(_gt, _fmt, ...) \ __xe_gt_sriov_printk(_gt, dbg, _fmt, ##__VA_ARGS__) +#define xe_gt_sriov_dbg_ratelimited(_gt, _fmt, ...) \ + __xe_gt_sriov_printk(_gt, dbg_ratelimited, _fmt, ##__VA_ARGS__) + /* for low level noisy debug messages */ #ifdef CONFIG_DRM_XE_DEBUG_SRIOV #define xe_gt_sriov_dbg_verbose(_gt, _fmt, ...) xe_gt_sriov_dbg(_gt, _fmt, ##__VA_ARGS__) diff --git a/drivers/gpu/drm/xe/xe_gt_stats.c b/drivers/gpu/drm/xe/xe_gt_stats.c index 789397514f3e..2a40924660ee 100644 --- a/drivers/gpu/drm/xe/xe_gt_stats.c +++ b/drivers/gpu/drm/xe/xe_gt_stats.c @@ -95,12 +95,19 @@ void xe_gt_stats_incr(struct xe_gt *gt, const enum xe_gt_stats_id id, int incr) #define DEF_STAT_STR(ID, name) [XE_GT_STATS_ID_##ID] = name static const char *const stat_description[__XE_GT_STATS_NUM_IDS] = { + DEF_STAT_STR(CHAIN_PAGEFAULT_COUNT, "chain_pagefault_count"), + DEF_STAT_STR(CHAIN_IRQ_PAGEFAULT_COUNT, "chain_irq_pagefault_count"), + DEF_STAT_STR(CHAIN_DRAIN_IRQ_PAGEFAULT_COUNT, "chain_drain_irq_pagefault_count"), + DEF_STAT_STR(CHAIN_MISMATCH_PAGEFAULT_COUNT, "chain_mismatch_pagefault_count"), + DEF_STAT_STR(PARALLEL_PAGEFAULT_COUNT, "parallel_pagefault_count"), + DEF_STAT_STR(LAST_PAGEFAULT_COUNT, "last_pagefault_count"), DEF_STAT_STR(SVM_PAGEFAULT_COUNT, "svm_pagefault_count"), DEF_STAT_STR(TLB_INVAL, "tlb_inval_count"), DEF_STAT_STR(SVM_TLB_INVAL_COUNT, "svm_tlb_inval_count"), DEF_STAT_STR(SVM_TLB_INVAL_US, "svm_tlb_inval_us"), DEF_STAT_STR(VMA_PAGEFAULT_COUNT, "vma_pagefault_count"), DEF_STAT_STR(VMA_PAGEFAULT_KB, "vma_pagefault_kb"), + DEF_STAT_STR(PAGEFAULT_US, "pagefault_us"), DEF_STAT_STR(INVALID_PREFETCH_PAGEFAULT_COUNT, "invalid_prefetch_pagefault_count"), DEF_STAT_STR(SVM_4K_PAGEFAULT_COUNT, "svm_4K_pagefault_count"), DEF_STAT_STR(SVM_64K_PAGEFAULT_COUNT, "svm_64K_pagefault_count"), diff --git a/drivers/gpu/drm/xe/xe_gt_stats_types.h b/drivers/gpu/drm/xe/xe_gt_stats_types.h index 425491bed6c4..24abeb9f2137 100644 --- a/drivers/gpu/drm/xe/xe_gt_stats_types.h +++ b/drivers/gpu/drm/xe/xe_gt_stats_types.h @@ -10,6 +10,19 @@ /** * enum xe_gt_stats_id - GT statistics identifiers + * @XE_GT_STATS_ID_CHAIN_PAGEFAULT_COUNT: Total page faults chained onto an + * in-flight fault instead of being queued separately. + * @XE_GT_STATS_ID_CHAIN_IRQ_PAGEFAULT_COUNT: Page faults chained directly from + * the IRQ handler. + * @XE_GT_STATS_ID_CHAIN_DRAIN_IRQ_PAGEFAULT_COUNT: IRQ-handler chained faults + * that also drained the fault queue. + * @XE_GT_STATS_ID_CHAIN_MISMATCH_PAGEFAULT_COUNT: Chained faults requeued + * because their fault range did not match the fault they were chained onto. + * @XE_GT_STATS_ID_PARALLEL_PAGEFAULT_COUNT: Faults dequeued while another page + * fault worker was already handling a fault concurrently. + * @XE_GT_STATS_ID_LAST_PAGEFAULT_COUNT: Faults whose range matched the last + * serviced range, allowing an immediate ack. + * * @XE_GT_STATS_ID_SVM_PAGEFAULT_COUNT: Total SVM page faults handled. * @XE_GT_STATS_ID_TLB_INVAL: Total GPU Translation Lookaside Buffer (TLB) * invalidations issued. @@ -22,6 +35,8 @@ * handled. * @XE_GT_STATS_ID_VMA_PAGEFAULT_KB: Size (KiB) of VMAs involved in * buffer-object page fault handling. + * @XE_GT_STATS_ID_PAGEFAULT_US: Cumulative time (µs) handling of all page + * faults. * @XE_GT_STATS_ID_INVALID_PREFETCH_PAGEFAULT_COUNT: GPU prefetch faults for * addresses with no valid backing. * @@ -127,12 +142,19 @@ * See Documentation/gpu/xe/xe_gt_stats.rst. */ enum xe_gt_stats_id { + XE_GT_STATS_ID_CHAIN_PAGEFAULT_COUNT, + XE_GT_STATS_ID_CHAIN_IRQ_PAGEFAULT_COUNT, + XE_GT_STATS_ID_CHAIN_DRAIN_IRQ_PAGEFAULT_COUNT, + XE_GT_STATS_ID_CHAIN_MISMATCH_PAGEFAULT_COUNT, + XE_GT_STATS_ID_PARALLEL_PAGEFAULT_COUNT, + XE_GT_STATS_ID_LAST_PAGEFAULT_COUNT, XE_GT_STATS_ID_SVM_PAGEFAULT_COUNT, XE_GT_STATS_ID_TLB_INVAL, XE_GT_STATS_ID_SVM_TLB_INVAL_COUNT, XE_GT_STATS_ID_SVM_TLB_INVAL_US, XE_GT_STATS_ID_VMA_PAGEFAULT_COUNT, XE_GT_STATS_ID_VMA_PAGEFAULT_KB, + XE_GT_STATS_ID_PAGEFAULT_US, XE_GT_STATS_ID_INVALID_PREFETCH_PAGEFAULT_COUNT, XE_GT_STATS_ID_SVM_4K_PAGEFAULT_COUNT, XE_GT_STATS_ID_SVM_64K_PAGEFAULT_COUNT, diff --git a/drivers/gpu/drm/xe/xe_guc.c b/drivers/gpu/drm/xe/xe_guc.c index 4286bd05c686..c7f8bbd4cb92 100644 --- a/drivers/gpu/drm/xe/xe_guc.c +++ b/drivers/gpu/drm/xe/xe_guc.c @@ -39,6 +39,7 @@ #include "xe_guc_rc.h" #include "xe_guc_relay.h" #include "xe_guc_submit.h" +#include "xe_log.h" #include "xe_memirq.h" #include "xe_mmio.h" #include "xe_platform_types.h" @@ -1542,8 +1543,9 @@ retry: /* scratch registers might be cleared during FLR, try once more */ if (!header) { if (++lost > MAX_RETRIES_ON_FLR) { - xe_gt_err(gt, "GuC mmio request %#x: lost, too many retries %u\n", - request[0], lost); + xe_log_err(gt, GUC, -ENOLINK, + "MMIO request %#x: lost, too many retries %u\n", + request[0], lost); return -ENOLINK; } xe_gt_dbg(gt, "GuC mmio request %#x: lost, trying again\n", request[0]); @@ -1551,8 +1553,8 @@ retry: goto retry; } timeout: - xe_gt_err(gt, "GuC mmio request %#x: no reply %#x\n", - request[0], header); + xe_log_err(gt, GUC, ret, "MMIO request %#x: no reply %#x\n", + request[0], header); return ret; } @@ -1607,16 +1609,16 @@ timeout: return -EREMCHG; } - xe_gt_err(gt, "GuC mmio request %#x: failure %#x hint %#x\n", - request[0], error, hint); + xe_log_err(gt, GUC, -ENXIO, "MMIO request %#x: failure %#x hint %#x\n", + request[0], error, hint); return -ENXIO; } if (FIELD_GET(GUC_HXG_MSG_0_TYPE, header) != GUC_HXG_TYPE_RESPONSE_SUCCESS) { proto: - xe_gt_err(gt, "GuC mmio request %#x: unexpected reply %#x\n", - request[0], header); + xe_log_err(gt, GUC, -EPROTO, "MMIO request %#x: unexpected reply %#x\n", + request[0], header); return -EPROTO; } diff --git a/drivers/gpu/drm/xe/xe_guc_ct.c b/drivers/gpu/drm/xe/xe_guc_ct.c index fe70c0fd85c5..5c4733da385c 100644 --- a/drivers/gpu/drm/xe/xe_guc_ct.c +++ b/drivers/gpu/drm/xe/xe_guc_ct.c @@ -265,7 +265,7 @@ static bool g2h_fence_needs_alloc(struct g2h_fence *g2h_fence) #define CTB_DESC_SIZE ALIGN(sizeof(struct guc_ct_buffer_desc), SZ_2K) #define CTB_H2G_BUFFER_OFFSET (CTB_DESC_SIZE * 2) #define CTB_G2H_BUFFER_OFFSET (CTB_DESC_SIZE * 2) -#define CTB_H2G_BUFFER_SIZE (SZ_4K) +#define CTB_H2G_BUFFER_SIZE (SZ_16K) #define CTB_H2G_BUFFER_DWORDS (CTB_H2G_BUFFER_SIZE / sizeof(u32)) #define CTB_G2H_BUFFER_SIZE (SZ_128K) #define CTB_G2H_BUFFER_DWORDS (CTB_G2H_BUFFER_SIZE / sizeof(u32)) @@ -939,7 +939,7 @@ static bool vf_action_can_safely_fail(struct xe_device *xe, u32 action) #define H2G_CT_HEADERS (GUC_CTB_HDR_LEN + 1) /* one DW CTB header and one DW HxG header */ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len, - u32 ct_fence_value, bool want_response) + u32 ct_fence_value, bool want_response, bool defer_flush) { struct xe_device *xe = ct_to_xe(ct); struct xe_gt *gt = ct_to_gt(ct); @@ -956,7 +956,6 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len, xe_gt_assert(gt, full_len <= GUC_CTB_MSG_MAX_LEN); if (IS_ENABLED(CONFIG_DRM_XE_DEBUG)) { - u32 desc_tail = desc_read(xe, h2g, tail); u32 desc_head = desc_read(xe, h2g, head); u32 desc_status; @@ -966,12 +965,6 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len, goto corrupted; } - if (tail != desc_tail) { - desc_write(xe, h2g, status, desc_status | GUC_CTB_STATUS_MISMATCH); - xe_gt_err(gt, "CT write: tail was modified %u != %u\n", desc_tail, tail); - goto corrupted; - } - if (tail > h2g->info.size) { desc_write(xe, h2g, status, desc_status | GUC_CTB_STATUS_OVERFLOW); xe_gt_err(gt, "CT write: tail out of range: %u vs %u\n", @@ -993,7 +986,10 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len, (h2g->info.size - tail) * sizeof(u32)); h2g_reserve_space(ct, (h2g->info.size - tail)); h2g->info.tail = 0; - desc_write(xe, h2g, tail, h2g->info.tail); + if (!defer_flush) { + xe_device_wmb(xe); + desc_write(xe, h2g, tail, h2g->info.tail); + } return -EAGAIN; } @@ -1024,14 +1020,15 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len, /* Write H2G ensuring visible before descriptor update */ xe_map_memcpy_to(xe, &map, 0, cmd, H2G_CT_HEADERS * sizeof(u32)); xe_map_memcpy_to(xe, &map, H2G_CT_HEADERS * sizeof(u32), action, len * sizeof(u32)); - xe_device_wmb(xe); - /* Update local copies */ h2g->info.tail = (tail + full_len) % h2g->info.size; h2g_reserve_space(ct, full_len); /* Update descriptor */ - desc_write(xe, h2g, tail, h2g->info.tail); + if (!defer_flush) { + xe_device_wmb(xe); + desc_write(xe, h2g, tail, h2g->info.tail); + } /* * desc_read() performs an VRAM read which serializes the CPU and drains @@ -1052,7 +1049,7 @@ corrupted: static int __guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len, u32 g2h_len, u32 num_g2h, - struct g2h_fence *g2h_fence) + struct g2h_fence *g2h_fence, bool defer_flush) { struct xe_gt *gt = ct_to_gt(ct); u16 seqno; @@ -1112,7 +1109,7 @@ retry: if (unlikely(ret)) goto out_unlock; - ret = h2g_write(ct, action, len, seqno, !!g2h_fence); + ret = h2g_write(ct, action, len, seqno, !!g2h_fence, defer_flush); if (unlikely(ret)) { if (ret == -EAGAIN) goto retry; @@ -1120,7 +1117,8 @@ retry: } __g2h_reserve_space(ct, g2h_len, num_g2h); - xe_guc_notify(ct_to_guc(ct)); + if (!defer_flush) + xe_guc_notify(ct_to_guc(ct)); out_unlock: if (g2h_len) spin_unlock_irq(&ct->fast_lock); @@ -1196,7 +1194,7 @@ static bool guc_ct_send_wait_for_retry(struct xe_guc_ct *ct, u32 len, static int guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len, u32 g2h_len, u32 num_g2h, - struct g2h_fence *g2h_fence) + struct g2h_fence *g2h_fence, bool defer_flush) { struct xe_gt *gt = ct_to_gt(ct); unsigned int sleep_period_ms = 1; @@ -1209,9 +1207,10 @@ static int guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len, try_again: ret = __guc_ct_send_locked(ct, action, len, g2h_len, num_g2h, - g2h_fence); + g2h_fence, defer_flush); if (unlikely(ret == -EBUSY)) { + xe_guc_ct_send_flush(ct); if (!guc_ct_send_wait_for_retry(ct, len, g2h_len, g2h_fence, &sleep_period_ms, &sleep_total_ms)) goto broken; @@ -1235,7 +1234,8 @@ static int guc_ct_send(struct xe_guc_ct *ct, const u32 *action, u32 len, xe_gt_assert(ct_to_gt(ct), !g2h_len || !g2h_fence); mutex_lock(&ct->lock); - ret = guc_ct_send_locked(ct, action, len, g2h_len, num_g2h, g2h_fence); + ret = guc_ct_send_locked(ct, action, len, g2h_len, num_g2h, g2h_fence, + false); mutex_unlock(&ct->lock); return ret; @@ -1283,25 +1283,76 @@ int xe_guc_ct_send(struct xe_guc_ct *ct, const u32 *action, u32 len, return ret; } +/** + * xe_guc_ct_send_locked() - submit a GuC CT H2G message with CT lock held + * @ct: GuC CT object + * @action: payload dwords (HxG header dword is expected at @action[-1]) + * @len: number of payload dwords in @action + * @defer_flush: defer publishing/doorbell for batching + * + * Sends a single H2G message to the GuC CT buffer while the caller already + * holds @ct->lock. + * + * If @defer_flush is false, the function completes the submission immediately: + * it makes the payload visible to the device, updates the H2G descriptor and + * rings the GuC doorbell. + * + * If @defer_flush is true, the message payload is copied into the H2G ring and + * the software tail is advanced, but the descriptor update and doorbell are + * deferred so multiple messages can be batched. In this mode, the caller must + * eventually call xe_guc_ct_send_flush() (still holding @ct->lock) to publish + * the descriptor and notify the GuC. On internal retry paths (-EBUSY), the + * implementation may force a flush to ensure forward progress. + * + * Return: 0 on success, negative errno on failure. + * + * Locking: + * Must be called with @ct->lock held. + */ int xe_guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len, - u32 g2h_len, u32 num_g2h) + bool defer_flush) { int ret; - ret = guc_ct_send_locked(ct, action, len, g2h_len, num_g2h, NULL); + ret = guc_ct_send_locked(ct, action, len, 0, 0, NULL, defer_flush); if (ret == -EDEADLK) kick_reset(ct); return ret; } +/** + * xe_guc_ct_send_flush() - flush pending GuC CT H2G writes + * @ct: GuC CT instance + * + * Some callers batch multiple H2G writes using xe_guc_ct_send_locked() in + * "write-only" mode (i.e., queue the message payloads but defer ringing the + * doorbell / updating the CT descriptor). This helper completes the submission + * by ensuring the payload writes are visible to the device, updating the H2G + * descriptor, and ringing the GuC CT doorbell. + * + * Locking: + * Must be called with @ct->lock held. + */ +void xe_guc_ct_send_flush(struct xe_guc_ct *ct) +{ + struct xe_device *xe = ct_to_xe(ct); + struct guc_ctb *h2g = &ct->ctbs.h2g; + + lockdep_assert_held(&ct->lock); + + xe_device_wmb(xe); + desc_write(xe, h2g, tail, h2g->info.tail); + xe_guc_notify(ct_to_guc(ct)); +} + int xe_guc_ct_send_g2h_handler(struct xe_guc_ct *ct, const u32 *action, u32 len) { int ret; lockdep_assert_held(&ct->lock); - ret = guc_ct_send_locked(ct, action, len, 0, 0, NULL); + ret = guc_ct_send_locked(ct, action, len, 0, 0, NULL, false); if (ret == -EDEADLK) kick_reset(ct); diff --git a/drivers/gpu/drm/xe/xe_guc_ct.h b/drivers/gpu/drm/xe/xe_guc_ct.h index 767365a33dee..3ddc665ab84a 100644 --- a/drivers/gpu/drm/xe/xe_guc_ct.h +++ b/drivers/gpu/drm/xe/xe_guc_ct.h @@ -6,6 +6,8 @@ #ifndef _XE_GUC_CT_H_ #define _XE_GUC_CT_H_ +#include <linux/mutex.h> + #include "xe_guc_ct_types.h" struct drm_printer; @@ -54,7 +56,7 @@ static inline void xe_guc_ct_irq_handler(struct xe_guc_ct *ct) int xe_guc_ct_send(struct xe_guc_ct *ct, const u32 *action, u32 len, u32 g2h_len, u32 num_g2h); int xe_guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len, - u32 g2h_len, u32 num_g2h); + bool defer_flush); int xe_guc_ct_send_recv(struct xe_guc_ct *ct, const u32 *action, u32 len, u32 *response_buffer); static inline int @@ -63,6 +65,8 @@ xe_guc_ct_send_block(struct xe_guc_ct *ct, const u32 *action, u32 len) return xe_guc_ct_send_recv(ct, action, len, NULL); } +void xe_guc_ct_send_flush(struct xe_guc_ct *ct); + /* This is only version of the send CT you can call from a G2H handler */ int xe_guc_ct_send_g2h_handler(struct xe_guc_ct *ct, const u32 *action, u32 len); @@ -87,4 +91,36 @@ static inline void xe_guc_ct_wake_waiters(struct xe_guc_ct *ct) wake_up_all(&ct->wq); } +/** + * xe_guc_ct_lock() - take the GuC CT mutex + * @ct: GuC CT object + * + * Wrapper around mutex_lock(&ct->lock) for cases where CT operations need to be + * performed from contexts that want an explicit "CT locked" pair without + * exporting the lock itself. + * + * Return/Locking: + * Acquires @ct->lock. + */ +static inline void xe_guc_ct_lock(struct xe_guc_ct *ct) +__acquires(&ct->lock) +{ + mutex_lock(&ct->lock); +} + +/** + * xe_guc_ct_unlock() - release the GuC CT mutex + * @ct: GuC CT object + * + * Counterpart to xe_guc_ct_lock(). + * + * Locking: + * Releases @ct->lock. + */ +static inline void xe_guc_ct_unlock(struct xe_guc_ct *ct) +__releases(&ct->lock) +{ + mutex_unlock(&ct->lock); +} + #endif diff --git a/drivers/gpu/drm/xe/xe_guc_exec_queue_types.h b/drivers/gpu/drm/xe/xe_guc_exec_queue_types.h index acdc24d1a6bd..d27826b36649 100644 --- a/drivers/gpu/drm/xe/xe_guc_exec_queue_types.h +++ b/drivers/gpu/drm/xe/xe_guc_exec_queue_types.h @@ -36,7 +36,7 @@ struct xe_guc_exec_queue { * a message needs to sent through the GPU scheduler but memory * allocations are not allowed. */ -#define MAX_STATIC_MSG_TYPE 3 +#define MAX_STATIC_MSG_TYPE 4 struct xe_sched_msg static_msgs[MAX_STATIC_MSG_TYPE]; /** @destroy_async: do final destroy async from this worker */ struct work_struct destroy_async; @@ -76,6 +76,36 @@ struct xe_guc_exec_queue { * recovery. */ bool needs_resume; + /** @multi_queue: multi-queue group CGP state for VF post migration recovery */ + struct { + /** + * @multi_queue.needs_cgp_sync: Needs a CGP_SYNC (dynamic CGP + * update) message replayed during recovery. + */ + u8 needs_cgp_sync:1; + /** + * @multi_queue.re_register: A registration-time CGP update was + * interrupted by recovery; the queue must be re-registered. + */ + u8 re_register:1; + /** + * @multi_queue.re_update: A dynamic CGP update was interrupted + * by recovery; the CGP update must be replayed. + */ + u8 re_update:1; + /** + * @multi_queue.registering_cgp: This queue's currently + * outstanding CGP_SYNC is a registration (matched against + * group->cgp_update_q in revert). + */ + u8 registering_cgp:1; + /** + * @multi_queue.updating_cgp: This queue's currently outstanding + * CGP_SYNC is a dynamic update (matched against + * group->cgp_update_q in revert). + */ + u8 updating_cgp:1; + } multi_queue; }; #endif diff --git a/drivers/gpu/drm/xe/xe_guc_pagefault.c b/drivers/gpu/drm/xe/xe_guc_pagefault.c index 607e32392f46..8f8210a732e9 100644 --- a/drivers/gpu/drm/xe/xe_guc_pagefault.c +++ b/drivers/gpu/drm/xe/xe_guc_pagefault.c @@ -10,6 +10,22 @@ #include "xe_pagefault.h" #include "xe_pagefault_types.h" +#define XE_GUC_PAGEFAULT_FLUSH_PERIOD BIT(4) /* Sixteen */ + +static void guc_ack_fault_begin(void *private) +{ + struct xe_guc *guc = private; + + xe_guc_ct_lock(&guc->ct); + + BUILD_BUG_ON(((XE_GUC_PAGEFAULT_FLUSH_PERIOD - 1) & + XE_GUC_PAGEFAULT_FLUSH_PERIOD) != 0); + + /* Ack the 2nd, then 18th, etc... */ + guc->pagefault_ack_counter = + XE_GUC_PAGEFAULT_FLUSH_PERIOD - 1; +} + static void guc_ack_fault(struct xe_pagefault *pf, int err) { u32 vfid = FIELD_GET(PFD_VFID, pf->producer.msg[2]); @@ -36,12 +52,26 @@ static void guc_ack_fault(struct xe_pagefault *pf, int err) FIELD_PREP(PFR_PDATA, pdata), }; struct xe_guc *guc = pf->producer.private; + bool write_only = guc->pagefault_ack_counter++ & + (XE_GUC_PAGEFAULT_FLUSH_PERIOD - 1); + + xe_guc_ct_send_locked(&guc->ct, action, ARRAY_SIZE(action), + write_only); +} + +static void guc_ack_fault_end(void *private) +{ + struct xe_guc *guc = private; - xe_guc_ct_send(&guc->ct, action, ARRAY_SIZE(action), 0, 0); + if ((guc->pagefault_ack_counter & (XE_GUC_PAGEFAULT_FLUSH_PERIOD - 1)) != 1) + xe_guc_ct_send_flush(&guc->ct); + xe_guc_ct_unlock(&guc->ct); } static const struct xe_pagefault_ops guc_pagefault_ops = { + .ack_fault_begin = guc_ack_fault_begin, .ack_fault = guc_ack_fault, + .ack_fault_end = guc_ack_fault_end, }; /** @@ -89,8 +119,11 @@ int xe_guc_pagefault_handler(struct xe_guc *guc, u32 *msg, u32 len) FIELD_GET(PFD_FAULT_LEVEL, msg[0])) | FIELD_PREP(XE_PAGEFAULT_TYPE_MASK, FIELD_GET(PFD_FAULT_TYPE, msg[2])); - pf.consumer.engine_class = FIELD_GET(PFD_ENG_CLASS, msg[0]); - pf.consumer.engine_instance = FIELD_GET(PFD_ENG_INSTANCE, msg[0]); + pf.consumer.engine_class_instance = + FIELD_PREP(XE_PAGEFAULT_ENGINE_CLASS_MASK, + FIELD_GET(PFD_ENG_CLASS, msg[0])) | + FIELD_PREP(XE_PAGEFAULT_ENGINE_INSTANCE_MASK, + FIELD_GET(PFD_ENG_INSTANCE, msg[0])); pf.producer.private = guc; pf.producer.ops = &guc_pagefault_ops; diff --git a/drivers/gpu/drm/xe/xe_guc_pc.c b/drivers/gpu/drm/xe/xe_guc_pc.c index 7cf8f4858598..097b075bd89a 100644 --- a/drivers/gpu/drm/xe/xe_guc_pc.c +++ b/drivers/gpu/drm/xe/xe_guc_pc.c @@ -1241,8 +1241,21 @@ static int pc_modify_defaults(struct xe_guc_pc *pc) if (xe->info.platform == XE_PANTHERLAKE) { ret = pc_action_set_dcc(pc, false); - if (unlikely(ret)) + if (unlikely(ret)) { xe_gt_err(gt, "Failed to modify DCC default: %pe\n", ERR_PTR(ret)); + return ret; + } + + if (xe_gt_is_main_type(gt) && + GUC_FIRMWARE_VER_AT_LEAST(>->uc.guc, 70, 48, 0)) { + ret = pc_action_set_param(pc, + SLPC_PARAM_SET_IBC_VERSION, + 3); + if (unlikely(ret)) { + xe_gt_err(gt, "Failed to modify IBC version: %pe\n", ERR_PTR(ret)); + return ret; + } + } } return ret; diff --git a/drivers/gpu/drm/xe/xe_guc_submit.c b/drivers/gpu/drm/xe/xe_guc_submit.c index 8aaed4fd13ea..99d8c807ff05 100644 --- a/drivers/gpu/drm/xe/xe_guc_submit.c +++ b/drivers/gpu/drm/xe/xe_guc_submit.c @@ -800,9 +800,12 @@ static void xe_guc_exec_queue_group_cgp_update(struct xe_device *xe, } } +#define CGP_SYNC_REGISTRATION BIT(0) + static void xe_guc_exec_queue_group_cgp_sync(struct xe_guc *guc, struct xe_exec_queue *q, - const u32 *action, u32 len) + const u32 *action, u32 len, + unsigned int flags) { struct xe_exec_queue_group *group = q->multi_queue.group; struct xe_device *xe = guc_to_xe(guc); @@ -815,13 +818,12 @@ static void xe_guc_exec_queue_group_cgp_sync(struct xe_guc *guc, * Hence, no locking is required here. * Wait for any pending CGP_SYNC_DONE response before updating the * CGP page and sending CGP_SYNC message. - * - * FIXME: Support VF migration */ ret = wait_event_timeout(guc->ct.wq, !READ_ONCE(group->sync_pending) || - xe_guc_read_stopped(guc), HZ); - if (!ret || xe_guc_read_stopped(guc)) { + xe_guc_read_stopped(guc) || vf_recovery(guc), + HZ); + if ((!ret && !vf_recovery(guc)) || xe_guc_read_stopped(guc)) { /* CGP_SYNC failed. Reset gt, cleanup the group */ xe_gt_warn(guc_to_gt(guc), "Wait for CGP_SYNC_DONE response failed!\n"); set_exec_queue_group_banned(q); @@ -830,17 +832,45 @@ static void xe_guc_exec_queue_group_cgp_sync(struct xe_guc *guc, return; } + /* + * If woken by VF migration recovery, do not touch the CGP or send: the + * message would be lost and, for a registration, GuC must (re-)register + * the context before its CGP entry may be read. Flag the queue so revert + * replays it - a registration by re-registration, a dynamic update by a + * replayed CGP_SYNC - and bail. + */ + if (vf_recovery(guc)) { + if (flags & CGP_SYNC_REGISTRATION) + q->guc->multi_queue.re_register = true; + else + q->guc->multi_queue.re_update = true; + return; + } + scoped_guard(spinlock, &q->multi_queue.lock) priority = q->multi_queue.priority; xe_lrc_set_multi_queue_priority(q->lrc[0], priority); xe_guc_exec_queue_group_cgp_update(xe, q); + /* + * Record the nature of this outstanding sync so revert can replay it if + * its CGP_SYNC_DONE is lost across a migration: a registration is + * recovered by re-registration, a dynamic update by a replayed CGP_SYNC. + */ + if (flags & CGP_SYNC_REGISTRATION) { + q->guc->multi_queue.registering_cgp = true; + q->guc->multi_queue.updating_cgp = false; + } else { + q->guc->multi_queue.updating_cgp = true; + q->guc->multi_queue.registering_cgp = false; + } + WRITE_ONCE(group->cgp_update_q, q); WRITE_ONCE(group->sync_pending, true); xe_guc_ct_send(&guc->ct, action, len, G2H_LEN_DW_MULTI_QUEUE_CONTEXT, 1); } -static void guc_exec_queue_send_cgp_sync(struct xe_exec_queue *q) +static void guc_exec_queue_send_cgp_sync(struct xe_exec_queue *q, unsigned int flags) { #define MAX_MULTI_QUEUE_CGP_SYNC_SIZE (2) struct xe_guc *guc = exec_queue_to_guc(q); @@ -854,7 +884,7 @@ static void guc_exec_queue_send_cgp_sync(struct xe_exec_queue *q) xe_gt_assert(guc_to_gt(guc), len <= MAX_MULTI_QUEUE_CGP_SYNC_SIZE); #undef MAX_MULTI_QUEUE_CGP_SYNC_SIZE - xe_guc_exec_queue_group_cgp_sync(guc, q, action, len); + xe_guc_exec_queue_group_cgp_sync(guc, q, action, len, flags); } static void __register_exec_queue_group(struct xe_exec_queue *q, @@ -882,7 +912,8 @@ static void __register_exec_queue_group(struct xe_exec_queue *q, * XE_GUC_ACTION_NOTIFY_MULTI_QUEUE_CONTEXT_CGP_SYNC_DONE response * from guc. */ - xe_guc_exec_queue_group_cgp_sync(guc, q, action, len); + xe_guc_exec_queue_group_cgp_sync(guc, q, action, len, + CGP_SYNC_REGISTRATION); } static void __register_mlrc_exec_queue(struct xe_guc *guc, @@ -1042,7 +1073,7 @@ static void register_exec_queue(struct xe_exec_queue *q, int ctx_type) init_policies(guc, q); if (xe_exec_queue_is_multi_queue_secondary(q)) - guc_exec_queue_send_cgp_sync(q); + guc_exec_queue_send_cgp_sync(q, CGP_SYNC_REGISTRATION); } static u32 wq_space_until_wrap(struct xe_exec_queue *q) @@ -1559,8 +1590,14 @@ guc_exec_queue_timedout_job(struct drm_sched_job *drm_job) if (!skip_timeout_check && !check_timeout(q, job)) goto rearm; + /* + * Killed queues must not newly wedge the device, but preserve an + * already-wedged state to avoid warning on teardown timeouts. + */ if (!exec_queue_killed(q)) wedged = guc_submit_hint_wedged(exec_queue_to_guc(q)); + else + wedged = xe_device_wedged(xe); set_exec_queue_banned(q); @@ -1781,32 +1818,15 @@ static void __guc_exec_queue_destroy_async(struct work_struct *w) static void guc_exec_queue_destroy_async(struct xe_exec_queue *q) { INIT_WORK(&q->guc->destroy_async, __guc_exec_queue_destroy_async); - - /* We must block on kernel engines so slabs are empty on driver unload */ - if (q->flags & EXEC_QUEUE_FLAG_PERMANENT || exec_queue_wedged(q)) - guc_exec_queue_do_destroy(q); - else - xe_destroy_wq_queue(&q->guc->destroy_async); + xe_destroy_wq_queue(&q->guc->destroy_async); } -static void __guc_exec_queue_destroy(struct xe_guc *guc, struct xe_exec_queue *q) -{ - /* - * Might be done from within the GPU scheduler, need to do async as we - * fini the scheduler when the engine is fini'd, the scheduler can't - * complete fini within itself (circular dependency). Async resolves - * this we and don't really care when everything is fini'd, just that it - * is. - */ - guc_exec_queue_destroy_async(q); -} - -static void __guc_exec_queue_process_msg_cleanup(struct xe_sched_msg *msg) +static void __guc_exec_queue_process_msg_cleanup(struct xe_sched_msg *msg, + bool bound) { struct xe_exec_queue *q = msg->private_data; struct xe_guc *guc = exec_queue_to_guc(q); - xe_gt_assert(guc_to_gt(guc), !(q->flags & EXEC_QUEUE_FLAG_PERMANENT)); trace_xe_exec_queue_cleanup_entity(q); /* @@ -1819,10 +1839,12 @@ static void __guc_exec_queue_process_msg_cleanup(struct xe_sched_msg *msg) * it is safe to directly destroy the exec queue on driver side, as the GuC * will not process further requests and all resources must be cleaned up locally. */ - if (exec_queue_registered(q) && xe_uc_fw_is_running(&guc->fw)) + /* A wedged GuC won't answer the H2G, so tear down on the driver side. */ + if (bound && !exec_queue_wedged(q) && exec_queue_registered(q) && + xe_uc_fw_is_running(&guc->fw)) disable_scheduling_deregister(guc, q); else - __guc_exec_queue_destroy(guc, q); + guc_exec_queue_destroy_async(q); } static bool guc_exec_queue_allowed_to_change_state(struct xe_exec_queue *q) @@ -1830,12 +1852,13 @@ static bool guc_exec_queue_allowed_to_change_state(struct xe_exec_queue *q) return !exec_queue_killed_or_banned_or_wedged(q) && exec_queue_registered(q); } -static void __guc_exec_queue_process_msg_set_sched_props(struct xe_sched_msg *msg) +static void __guc_exec_queue_process_msg_set_sched_props(struct xe_sched_msg *msg, + bool bound) { struct xe_exec_queue *q = msg->private_data; struct xe_guc *guc = exec_queue_to_guc(q); - if (guc_exec_queue_allowed_to_change_state(q)) + if (guc_exec_queue_allowed_to_change_state(q) && bound) init_policies(guc, q); kfree(msg); } @@ -1873,13 +1896,14 @@ static void suspend_fence_signal(struct xe_exec_queue *q) __suspend_fence_signal(q); } -static void __guc_exec_queue_process_msg_suspend(struct xe_sched_msg *msg) +static void __guc_exec_queue_process_msg_suspend(struct xe_sched_msg *msg, + bool bound) { struct xe_exec_queue *q = msg->private_data; struct xe_guc *guc = exec_queue_to_guc(q); if (guc_exec_queue_allowed_to_change_state(q) && !exec_queue_suspended(q) && - exec_queue_enabled(q)) { + exec_queue_enabled(q) && bound) { wait_event(guc->ct.wq, vf_recovery(guc) || ((q->guc->resume_time != RESUME_PENDING || xe_guc_read_stopped(guc)) && !exec_queue_pending_disable(q))); @@ -1903,11 +1927,12 @@ static void __guc_exec_queue_process_msg_suspend(struct xe_sched_msg *msg) } } -static void __guc_exec_queue_process_msg_resume(struct xe_sched_msg *msg) +static void __guc_exec_queue_process_msg_resume(struct xe_sched_msg *msg, + bool bound) { struct xe_exec_queue *q = msg->private_data; - if (guc_exec_queue_allowed_to_change_state(q)) { + if (guc_exec_queue_allowed_to_change_state(q) && bound) { clear_exec_queue_suspended(q); if (!exec_queue_enabled(q)) { q->guc->resume_time = RESUME_PENDING; @@ -1919,52 +1944,79 @@ static void __guc_exec_queue_process_msg_resume(struct xe_sched_msg *msg) } } -static void __guc_exec_queue_process_msg_set_multi_queue_priority(struct xe_sched_msg *msg) +static void __guc_exec_queue_process_msg_set_multi_queue_priority(struct xe_sched_msg *msg, + bool bound) { struct xe_exec_queue *q = msg->private_data; - if (guc_exec_queue_allowed_to_change_state(q)) - guc_exec_queue_send_cgp_sync(q); + if (guc_exec_queue_allowed_to_change_state(q) && bound) + guc_exec_queue_send_cgp_sync(q, 0); kfree(msg); } +static void __guc_exec_queue_process_msg_cgp_sync(struct xe_sched_msg *msg, + bool bound) +{ + struct xe_exec_queue *q = msg->private_data; + + /* + * Replay a dynamic CGP update lost across VF migration by re-issuing the + * CGP update + CGP_SYNC (re-applies the current priority from + * q->multi_queue.priority). + */ + if (guc_exec_queue_allowed_to_change_state(q) && bound) + guc_exec_queue_send_cgp_sync(q, 0); +} + #define CLEANUP 1 /* Non-zero values to catch uninitialized msg */ #define SET_SCHED_PROPS 2 #define SUSPEND 3 #define RESUME 4 #define SET_MULTI_QUEUE_PRIORITY 5 +#define CGP_SYNC_MSG 6 #define OPCODE_MASK 0xf #define MSG_LOCKED BIT(8) #define MSG_HEAD BIT(9) +#define MSG_PM_REF BIT(10) static void guc_exec_queue_process_msg(struct xe_sched_msg *msg) { struct xe_device *xe = guc_to_xe(exec_queue_to_guc(msg->private_data)); + int idx; + bool pm_ref = !!(msg->opcode & MSG_PM_REF); + bool bound = drm_dev_enter(&xe->drm, &idx); trace_xe_sched_msg_recv(msg); - switch (msg->opcode) { + switch (msg->opcode & OPCODE_MASK) { case CLEANUP: - __guc_exec_queue_process_msg_cleanup(msg); + __guc_exec_queue_process_msg_cleanup(msg, bound); break; case SET_SCHED_PROPS: - __guc_exec_queue_process_msg_set_sched_props(msg); + __guc_exec_queue_process_msg_set_sched_props(msg, bound); break; case SUSPEND: - __guc_exec_queue_process_msg_suspend(msg); + __guc_exec_queue_process_msg_suspend(msg, bound); break; case RESUME: - __guc_exec_queue_process_msg_resume(msg); + __guc_exec_queue_process_msg_resume(msg, bound); break; case SET_MULTI_QUEUE_PRIORITY: - __guc_exec_queue_process_msg_set_multi_queue_priority(msg); + __guc_exec_queue_process_msg_set_multi_queue_priority(msg, bound); + break; + case CGP_SYNC_MSG: + __guc_exec_queue_process_msg_cgp_sync(msg, bound); break; default: XE_WARN_ON("Unknown message type"); } - xe_pm_runtime_put(xe); + if (pm_ref) + xe_pm_runtime_put(xe); + + if (bound) + drm_dev_exit(idx); } static const struct drm_sched_backend_ops drm_sched_ops = { @@ -2089,10 +2141,16 @@ static void guc_exec_queue_kill(struct xe_exec_queue *q) static void guc_exec_queue_add_msg(struct xe_exec_queue *q, struct xe_sched_msg *msg, u32 opcode) { - xe_pm_runtime_get_noresume(guc_to_xe(exec_queue_to_guc(q))); + struct xe_device *xe = guc_to_xe(exec_queue_to_guc(q)); + int idx; + bool bound = drm_dev_enter(&xe->drm, &idx); INIT_LIST_HEAD(&msg->link); msg->opcode = opcode & OPCODE_MASK; + if (bound) { + xe_pm_runtime_get_noresume(xe); + msg->opcode |= MSG_PM_REF; + } msg->private_data = q; trace_xe_sched_msg_add(msg); @@ -2102,6 +2160,9 @@ static void guc_exec_queue_add_msg(struct xe_exec_queue *q, struct xe_sched_msg xe_sched_add_msg_locked(&q->guc->sched, msg); else xe_sched_add_msg(&q->guc->sched, msg); + + if (bound) + drm_dev_exit(idx); } static void guc_exec_queue_try_add_msg_head(struct xe_exec_queue *q, @@ -2129,14 +2190,12 @@ static bool guc_exec_queue_try_add_msg(struct xe_exec_queue *q, #define STATIC_MSG_CLEANUP 0 #define STATIC_MSG_SUSPEND 1 #define STATIC_MSG_RESUME 2 +#define STATIC_MSG_CGP_SYNC 3 static void guc_exec_queue_destroy(struct xe_exec_queue *q) { struct xe_sched_msg *msg = q->guc->static_msgs + STATIC_MSG_CLEANUP; - if (!(q->flags & EXEC_QUEUE_FLAG_PERMANENT) && !exec_queue_wedged(q)) - guc_exec_queue_add_msg(q, msg, CLEANUP); - else - __guc_exec_queue_destroy(exec_queue_to_guc(q), q); + guc_exec_queue_add_msg(q, msg, CLEANUP); } static int guc_exec_queue_set_priority(struct xe_exec_queue *q, @@ -2601,7 +2660,7 @@ static void guc_exec_queue_stop(struct xe_guc *guc, struct xe_exec_queue *q) } if (do_destroy) - __guc_exec_queue_destroy(guc, q); + guc_exec_queue_destroy_async(q); } static int guc_submit_reset_prepare(struct xe_guc *guc) @@ -2717,6 +2776,57 @@ static void guc_exec_queue_revert_pending_state_change(struct xe_guc *guc, q->guc->id); } + /* + * A registration time CGP update that bailed when woken by VF recovery. + * Re-register the queue. + */ + if (q->guc->multi_queue.re_register) { + clear_exec_queue_registered(q); + q->guc->multi_queue.re_register = false; + xe_gt_dbg(guc_to_gt(guc), "Replay REGISTER (cgp) - guc_id=%d", + q->guc->id); + } + + /* + * If a CGP update gets dropped during migration, CGP_SYNC_DONE will not + * be received (sync_pending still set and this queue owns it). Recover + * it the same way and clear the stuck sync_pending. + */ + if (xe_exec_queue_is_multi_queue(q)) { + struct xe_exec_queue_group *group = q->multi_queue.group; + + if (q == READ_ONCE(group->cgp_update_q) && + READ_ONCE(group->sync_pending)) { + if (q->guc->multi_queue.registering_cgp) { + clear_exec_queue_registered(q); + xe_gt_dbg(guc_to_gt(guc), "Replay REGISTER (cgp sync) - guc_id=%d", + q->guc->id); + } else if (q->guc->multi_queue.updating_cgp) { + q->guc->multi_queue.needs_cgp_sync = true; + xe_gt_dbg(guc_to_gt(guc), "Replay CGP_SYNC - guc_id=%d", + q->guc->id); + } + q->guc->multi_queue.registering_cgp = false; + q->guc->multi_queue.updating_cgp = false; + WRITE_ONCE(group->cgp_update_q, NULL); + WRITE_ONCE(group->sync_pending, false); + } + } + + /* + * A dynamic-time CGP update that bailed when woken by VF recovery. + * Replay the dynamic CGP update unless the queue is registered or being + * re-registered, which re-does the CGP anyway. + */ + if (q->guc->multi_queue.re_update) { + q->guc->multi_queue.re_update = false; + if (exec_queue_registered(q)) { + q->guc->multi_queue.needs_cgp_sync = true; + xe_gt_dbg(guc_to_gt(guc), "Replay CGP_SYNC (re-update) - guc_id=%d", + q->guc->id); + } + } + q->guc->resume_time = 0; } @@ -2737,17 +2847,24 @@ static void lrc_parallel_clear(struct xe_lrc *lrc) * during VF resume flows. The function scans the queue state, make adjustments * as needed, and queues jobs / messages which replayed upon unpause. */ -static void guc_exec_queue_pause(struct xe_guc *guc, struct xe_exec_queue *q) +static void guc_exec_queue_pause_prepare(struct xe_guc *guc, struct xe_exec_queue *q) { struct xe_gpu_scheduler *sched = &q->guc->sched; - struct xe_sched_job *job; - int i; lockdep_assert_held(&guc->submission_state.lock); /* Stop scheduling + flush any DRM scheduler operations */ xe_sched_submission_stop(sched); cancel_delayed_work_sync(&sched->base.work_tdr); +} + +static void guc_exec_queue_pause(struct xe_guc *guc, struct xe_exec_queue *q) +{ + struct xe_gpu_scheduler *sched = &q->guc->sched; + struct xe_sched_job *job; + int i; + + lockdep_assert_held(&guc->submission_state.lock); guc_exec_queue_revert_pending_state_change(guc, q); @@ -2806,6 +2923,19 @@ void xe_guc_submit_pause_vf(struct xe_guc *guc) xe_gt_assert(guc_to_gt(guc), vf_recovery(guc)); mutex_lock(&guc->submission_state.lock); + /* + * Stop all schedulers before reverting any queue: in a multi-queue + * group a secondary's run_job() can register the primary, which must + * not race an in-progress revert. + */ + xa_for_each(&guc->submission_state.exec_queue_lookup, index, q) { + /* Prevent redundant attempts to stop parallel queues */ + if (q->guc->id != index) + continue; + + guc_exec_queue_pause_prepare(guc, q); + } + xa_for_each(&guc->submission_state.exec_queue_lookup, index, q) { /* Prevent redundant attempts to stop parallel queues */ if (q->guc->id != index) @@ -2922,6 +3052,16 @@ static void guc_exec_queue_replay_pending_state_change(struct xe_exec_queue *q) struct xe_gpu_scheduler *sched = &q->guc->sched; struct xe_sched_msg *msg; + if (q->guc->multi_queue.needs_cgp_sync) { + msg = q->guc->static_msgs + STATIC_MSG_CGP_SYNC; + + xe_sched_msg_lock(sched); + guc_exec_queue_try_add_msg_head(q, msg, CGP_SYNC_MSG); + xe_sched_msg_unlock(sched); + + q->guc->multi_queue.needs_cgp_sync = false; + } + if (q->guc->needs_cleanup) { msg = q->guc->static_msgs + STATIC_MSG_CLEANUP; @@ -3166,7 +3306,7 @@ static void handle_deregister_done(struct xe_guc *guc, struct xe_exec_queue *q) trace_xe_exec_queue_deregister_done(q); clear_exec_queue_registered(q); - __guc_exec_queue_destroy(guc, q); + guc_exec_queue_destroy_async(q); } int xe_guc_deregister_done_handler(struct xe_guc *guc, u32 *msg, u32 len) @@ -3405,7 +3545,8 @@ int xe_guc_exec_queue_cgp_context_error_handler(struct xe_guc *guc, u32 *msg, int xe_guc_exec_queue_cgp_sync_done_handler(struct xe_guc *guc, u32 *msg, u32 len) { struct xe_device *xe = guc_to_xe(guc); - struct xe_exec_queue *q; + struct xe_exec_queue_group *group; + struct xe_exec_queue *q, *upd_q; u32 guc_id = msg[0]; if (unlikely(len < 1)) { @@ -3422,8 +3563,20 @@ int xe_guc_exec_queue_cgp_sync_done_handler(struct xe_guc *guc, u32 *msg, u32 le return -EPROTO; } + /* + * The outstanding CGP update is now confirmed; clear the owning queue's + * tracking so a later migration does not needlessly replay it. + */ + group = q->multi_queue.group; + upd_q = READ_ONCE(group->cgp_update_q); + if (upd_q) { + upd_q->guc->multi_queue.registering_cgp = false; + upd_q->guc->multi_queue.updating_cgp = false; + WRITE_ONCE(group->cgp_update_q, NULL); + } + /* Wakeup the serialized cgp update wait */ - WRITE_ONCE(q->multi_queue.group->sync_pending, false); + WRITE_ONCE(group->sync_pending, false); xe_guc_ct_wake_waiters(&guc->ct); return 0; diff --git a/drivers/gpu/drm/xe/xe_guc_types.h b/drivers/gpu/drm/xe/xe_guc_types.h index 31a2acb63ac3..3dbcb1331690 100644 --- a/drivers/gpu/drm/xe/xe_guc_types.h +++ b/drivers/gpu/drm/xe/xe_guc_types.h @@ -122,6 +122,12 @@ struct xe_guc { struct xe_reg notify_reg; /** @params: Control params for fw initialization */ u32 params[GUC_CTL_MAX_DWORDS]; + + /** + * @pagefault_ack_counter: Counter to determine when periodically ack + * pagefaults in a batch. + */ + u32 pagefault_ack_counter; }; #endif diff --git a/drivers/gpu/drm/xe/xe_hwmon.c b/drivers/gpu/drm/xe/xe_hwmon.c index de3f2aeffc3f..5284cab6703d 100644 --- a/drivers/gpu/drm/xe/xe_hwmon.c +++ b/drivers/gpu/drm/xe/xe_hwmon.c @@ -266,7 +266,7 @@ static struct xe_reg xe_hwmon_get_reg(struct xe_hwmon *hwmon, enum xe_hwmon_reg switch (hwmon_reg) { case REG_TEMP: - if (xe->info.platform == XE_BATTLEMAGE) { + if (xe->info.platform == XE_BATTLEMAGE || xe->info.platform == XE_CRESCENTISLAND) { if (channel == CHANNEL_PKG) return BMG_PACKAGE_TEMPERATURE; else if (channel == CHANNEL_VRAM) @@ -575,23 +575,36 @@ xe_hwmon_power_max_interval_show(struct device *dev, struct device_attribute *at mutex_unlock(&hwmon->hwmon_lock); - x = REG_FIELD_GET(PWR_LIM_TIME_X, reg_val); - y = REG_FIELD_GET(PWR_LIM_TIME_Y, reg_val); + if (hwmon->xe->info.platform >= XE_CRESCENTISLAND) { + /** + * On CRI and newer platforms, the interval encoding changed. + * The value is now stored directly in milliseconds as U5.2, + * replacing the older 1.x * 2^y representation. + * Bits [6:2] hold the integer part and bits [1:0] the fraction, + * so convert to milliseconds by extracting integer/fractional parts. + * Round fraction value to 1 when it is >= 0.5. + */ + reg_val = REG_FIELD_GET(PWR_LIM_TIME, reg_val); + out = (u64)((reg_val >> 2) + ((reg_val & 0x3) >= 2)); + } else { + x = REG_FIELD_GET(PWR_LIM_TIME_X, reg_val); + y = REG_FIELD_GET(PWR_LIM_TIME_Y, reg_val); - /* - * tau = (1 + (x / 4)) * power(2,y), x = bits(23:22), y = bits(21:17) - * = (4 | x) << (y - 2) - * - * Here (y - 2) ensures a 1.x fixed point representation of 1.x - * As x is 2 bits so 1.x can be 1.0, 1.25, 1.50, 1.75 - * - * As y can be < 2, we compute tau4 = (4 | x) << y - * and then add 2 when doing the final right shift to account for units - */ - tau4 = (u64)((1 << x_w) | x) << y; + /* + * tau = (1 + (x / 4)) * power(2,y), x = bits(23:22), y = bits(21:17) + * = (4 | x) << (y - 2) + * + * Here (y - 2) ensures a 1.x fixed point representation of 1.x + * As x is 2 bits so 1.x can be 1.0, 1.25, 1.50, 1.75 + * + * As y can be < 2, we compute tau4 = (4 | x) << y + * and then add 2 when doing the final right shift to account for units + */ + tau4 = (u64)((1 << x_w) | x) << y; - /* val in hwmon interface units (millisec) */ - out = mul_u64_u32_shr(tau4, SF_TIME, hwmon->scl_shift_time + x_w); + /* val in hwmon interface units (millisec) */ + out = mul_u64_u32_shr(tau4, SF_TIME, hwmon->scl_shift_time + x_w); + } return sysfs_emit(buf, "%llu\n", out); } @@ -637,24 +650,35 @@ xe_hwmon_power_max_interval_store(struct device *dev, struct device_attribute *a if (val > max_win) return -EINVAL; - /* val in hw units */ - val = DIV_ROUND_CLOSEST_ULL((u64)val << hwmon->scl_shift_time, SF_TIME) + 1; - - /* - * Convert val to 1.x * power(2,y) - * y = ilog2(val) - * x = (val - (1 << y)) >> (y - 2) - */ - if (!val) { - y = 0; - x = 0; + if (hwmon->xe->info.platform >= XE_CRESCENTISLAND) { + /** + * On CRI and newer platforms, the interval encoding changed. + * The value is now stored directly in milliseconds as U5.2, + * replacing the older 1.x * 2^y representation. + * Bits [6:2] hold the integer part and bits [1:0] the fraction,so + * convert from milliseconds by shifting the value left by 2 to fit into the field. + */ + rxy = REG_FIELD_PREP(PWR_LIM_TIME, (val << 2)); } else { - y = ilog2(val); - x = (val - (1ul << y)) << x_w >> y; - } + /* val in hw units */ + val = DIV_ROUND_CLOSEST_ULL((u64)val << hwmon->scl_shift_time, SF_TIME) + 1; + + /* + * Convert val to 1.x * power(2,y) + * y = ilog2(val) + * x = (val - (1 << y)) >> (y - 2) + */ + if (!val) { + y = 0; + x = 0; + } else { + y = ilog2(val); + x = (val - (1ul << y)) << x_w >> y; + } - rxy = REG_FIELD_PREP(PWR_LIM_TIME_X, x) | - REG_FIELD_PREP(PWR_LIM_TIME_Y, y); + rxy = REG_FIELD_PREP(PWR_LIM_TIME_X, x) | + REG_FIELD_PREP(PWR_LIM_TIME_Y, y); + } guard(xe_pm_runtime)(hwmon->xe); diff --git a/drivers/gpu/drm/xe/xe_log.c b/drivers/gpu/drm/xe/xe_log.c new file mode 100644 index 000000000000..5549ef6966fd --- /dev/null +++ b/drivers/gpu/drm/xe/xe_log.c @@ -0,0 +1,235 @@ +// SPDX-License-Identifier: MIT +/* + * Copyright © 2026 Intel Corporation + */ + +#include <kunit/static_stub.h> +#include <kunit/visibility.h> + +#include "abi/xe_log_abi.h" + +#include "xe_device.h" +#include "xe_log.h" +#include "xe_printk.h" + +static void log_emit_cper(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid, + u32 component, u32 location, const void *data, size_t len, + struct va_format *vaf) +{ + KUNIT_STATIC_STUB_REDIRECT(log_emit_cper, pdev, cper_sev, sigid, + component, location, data, len, vaf); + /* TODO */ +} + +static const char *log_unknown_component_prefix(u32 component) +{ + u32 class = FIELD_GET(XE_LOG_COMPONENT_CLASS_MASK, component); + u32 type = FIELD_GET(XE_LOG_COMPONENT_TYPE_MASK, component); + + WARN(IS_ENABLED(CONFIG_DRM_XE_DEBUG), "LOG: unrecognized component %u.%u\n", class, type); + switch (class) { +#define MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(_CLASS) \ + case XE_LOG_COMPONENT_CLASS_##_CLASS: return #_CLASS "? "; + MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(SYSTEM) + MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(DRIVER) + MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(FEATURE) + MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(FIRMWARE) + MAKE_XE_LOG_COMPONENT_CLASS_PREFIX(HARDWARE) +#undef MAKE_XE_LOG_COMPONENT_CLASS_PREFIX + } + return "COMP? "; +} + +static const char *log_component_prefix(u32 component) +{ + switch (component) { +#define MAKE_XE_LOG_COMPONENT_CASE_PREFIX(_CLASS, _ID, _TAG, _SIG, _NAME) \ + case XE_LOG_COMPONENT_##_TAG: return #_TAG ": "; + DEFINE_XE_LOG_COMPONENTS(MAKE_XE_LOG_COMPONENT_CASE_PREFIX) +#undef MAKE_XE_LOG_COMPONENT_CASE_PREFIX + } + + return component ? log_unknown_component_prefix(component) : ""; +} + +static struct xe_gt *get_gt_safe(struct pci_dev *pdev, u8 id) +{ + struct xe_device *xe = pdev_to_xe_device(pdev); + + return xe ? xe_device_get_gt(xe, id) : NULL; +} + +static struct xe_tile *get_tile_safe(struct pci_dev *pdev, u8 id) +{ + struct xe_device *xe = pdev_to_xe_device(pdev); + + return xe && id < xe->info.tile_count ? &xe->tiles[id] : NULL; +} + +static const char *log_location_prefix(struct pci_dev *pdev, u32 location, char *buf, size_t size) +{ + u32 type = FIELD_GET(XE_LOG_LOCATION_TYPE_MASK, location); + u32 id = FIELD_GET(XE_LOG_LOCATION_ID_MASK, location); + + if (!location || type == XE_LOG_LOCATION_TYPE_DEVICE) { + if (id) + goto unrecognized; + strscpy(buf, "", size); + } else if (type == XE_LOG_LOCATION_TYPE_TILE) { + struct xe_tile *tile = get_tile_safe(pdev, id); + + if (!tile) + goto unrecognized; + snprintf(buf, size, "Tile%u: ", id); + } else if (type == XE_LOG_LOCATION_TYPE_GT) { + struct xe_gt *gt = get_gt_safe(pdev, id); + + if (!gt) + goto unrecognized; + snprintf(buf, size, "Tile%u: GT%u: ", gt->tile->id, id); + } else { + goto unrecognized; + } + + return buf; + +unrecognized: + pci_WARN(pdev, IS_ENABLED(CONFIG_DRM_XE_DEBUG), + "LOG: unrecognized location %u.%u\n", type, id); + snprintf(buf, size, "LOC%u.%u? ", type, id); + return buf; +} + +static bool is_hw_sigid(enum xe_sigid sigid) +{ + return (int)sigid >= INTEL_SIGID_GPU_XE_HARDWARE_START; +} + +static bool is_sev_error(int cper_sev) +{ + return cper_sev != CPER_SEV_INFORMATIONAL; +} + +static const char *log_hwe_prefix(int cper_sev, enum xe_sigid sigid) +{ + return is_sev_error(cper_sev) && is_hw_sigid(sigid) ? HW_ERR : ""; +} + +static const char *log_sev_prefix(int cper_sev) +{ + switch (cper_sev) { + case CPER_SEV_FATAL: + return "FATAL "; + case CPER_SEV_RECOVERABLE: + return ""; + case CPER_SEV_CORRECTED: + return "CORRECTED "; + case CPER_SEV_INFORMATIONAL: + return ""; + default: + WARN(IS_ENABLED(CONFIG_DRM_XE_DEBUG), "LOG: unknown severity %d\n", cper_sev); + return ""; + } +} + +#define __LOG_DRM_PRINTK_FMT(fmt, args...) "[drm] " fmt, ##args +#define __LOG_DRM_PRINTK_ERR_FMT(fmt, args...) __LOG_DRM_PRINTK_FMT("*ERROR* " fmt, args) + +static void log_dmesg_vprintk(struct pci_dev *pdev, int cper_sev, struct va_format *vaf) +{ + KUNIT_STATIC_STUB_REDIRECT(log_dmesg_vprintk, pdev, cper_sev, vaf); + + if (cper_sev == CPER_SEV_INFORMATIONAL) + pci_info(pdev, __LOG_DRM_PRINTK_FMT("%pV", vaf)); + else + pci_err(pdev, __LOG_DRM_PRINTK_ERR_FMT("%pV", vaf)); +} + +static void log_dmesg_printf(struct pci_dev *pdev, int cper_sev, const char *fmt, ...) +{ + struct va_format vaf; + va_list args; + + va_start(args, fmt); + vaf.fmt = fmt; + vaf.va = &args; + + log_dmesg_vprintk(pdev, cper_sev, &vaf); + + va_end(args); +} + +static void log_emit_dmesg(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid, + u32 component, u32 location, const void *data, size_t len, + struct va_format *vaf) +{ + char buf[32]; + const char *loc_prefix = log_location_prefix(pdev, location, buf, sizeof(buf)); + const char *comp_prefix = log_component_prefix(component); + const char *hwe_prefix = log_hwe_prefix(cper_sev, sigid); + const char *sev_prefix = log_sev_prefix(cper_sev); + + if (IS_ERR(data)) + log_dmesg_printf(pdev, cper_sev, "SIGID=%u %s(%pe) %s%s%s%pV", + sigid, sev_prefix, data, hwe_prefix, + loc_prefix, comp_prefix, vaf); + else if (data && len) + log_dmesg_printf(pdev, cper_sev, "SIGID=%u %s(%*phN) %s%s%s%pV", + sigid, sev_prefix, (int)len, data, hwe_prefix, + loc_prefix, comp_prefix, vaf); + else + log_dmesg_printf(pdev, cper_sev, "SIGID=%u %s%s%s%s%pV", + sigid, sev_prefix, hwe_prefix, + loc_prefix, comp_prefix, vaf); +} + +/** + * __xe_log_emit() - Emit a structured SIGID log entry + * @pdev: the &pci_dev device + * @cper_sev: CPER severity (CPER_SEV_FATAL, CPER_SEV_RECOVERABLE, ...) + * @sigid: signature identifier, see &enum xe_sigid + * @component: component identifier + * @location: location details of the @component + * @data: pointer to the additional details, or ERR_PTR, or NULL if not applicable + * @len: length of the @data in bytes, or 0 if not applicable + * @fmt: printf-style format string + * @...: format arguments + * + * Emits a dmesg line that includes a single stable, machine-matchable token + * ``SIGID=<n>`` followed by the optional severity token (like ``FATAL``) and, + * when @data pointer is set, either the error printed with %pe or a packed hex + * dump of the @data binary blob. The dmesg line will also include printf-style + * text message. + * + * Note that the full dmesg line, with the free text message, is only a debugging + * aid, not an interface! Only the ``SIGID=<n>`` token is stable there. + * The durable machine record is the CPER carrying the same SIGID. + * + * Note: generation of the CPER record is a planned follow-up. + * + * Examples:: + * + * <3> xe 0000:03:00.0: [drm] *ERROR* SIGID=104 FATAL (-EPROTO) Invalid GuC reply + * <3> xe 0000:03:00.0: [drm] *ERROR* SIGID=106 (-ETIMEDOUT) Engine 'rcs0' hung + * <6> xe 0000:03:00.0: [drm] SIGID=103 In survivability mode + */ +void __xe_log_emit(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid, + u32 component, u32 location, const void *data, size_t len, + const char *fmt, ...) +{ + struct va_format vaf; + va_list args; + + va_start(args, fmt); + vaf.fmt = fmt; + vaf.va = &args; + + log_emit_dmesg(pdev, cper_sev, sigid, component, location, data, len, &vaf); + log_emit_cper(pdev, cper_sev, sigid, component, location, data, len, &vaf); + + va_end(args); +} + +#if IS_BUILTIN(CONFIG_DRM_XE_KUNIT_TEST) +#include "tests/xe_log_kunit.c" +#endif diff --git a/drivers/gpu/drm/xe/xe_log.h b/drivers/gpu/drm/xe/xe_log.h new file mode 100644 index 000000000000..bb1185143589 --- /dev/null +++ b/drivers/gpu/drm/xe/xe_log.h @@ -0,0 +1,194 @@ +/* SPDX-License-Identifier: MIT */ +/* + * Copyright © 2026 Intel Corporation + */ + +#ifndef _XE_LOG_H_ +#define _XE_LOG_H_ + +#include <linux/cper.h> +#include <linux/err.h> + +#include "abi/xe_log_abi.h" +#include "abi/xe_sigid_abi.h" +#include "xe_any.h" + +struct pci_dev; + +__printf(8, 9) +void __xe_log_emit(struct pci_dev *pdev, int cper_sev, enum xe_sigid sigid, + u32 component, u32 location, const void *data, size_t len, + const char *fmt, ...); + +#define __xe_log_const_sev_to_level(sev) \ + (__builtin_constant_p(sev) ? (sev) == CPER_SEV_INFORMATIONAL ? KERN_INFO : KERN_ERR : NULL) + +#define __xe_log_emit_printk_index(level, fmt) \ + dev_printk_index_emit(level, "%s SIGID=%u %s" fmt); + +#define xe_log_emit(pdev, sev, sig, comp, loc, data, len, fmt, args...) do { \ + __xe_log_emit_printk_index(__xe_log_const_sev_to_level(sev), fmt); \ + __xe_log_emit((pdev), (sev), (sig), (comp), (loc), (data), (len), fmt, ##args); \ +} while (0) + +#define xe_log_emit_fatal(pdev, sig, comp, loc, data, len, fmt, args...) \ + xe_log_emit((pdev), CPER_SEV_FATAL, (sig), (comp), (loc), \ + (data), (len), fmt, ##args) + +#define xe_log_emit_recoverable(pdev, sig, comp, loc, data, len, fmt, args...) \ + xe_log_emit((pdev), CPER_SEV_RECOVERABLE, (sig), (comp), (loc), \ + (data), (len), fmt, ##args) + +#define xe_log_emit_corrected(pdev, sig, comp, loc, data, len, fmt, args...) \ + xe_log_emit((pdev), CPER_SEV_CORRECTED, (sig), (comp), (loc), \ + (data), (len), fmt, ##args) + +#define xe_log_emit_info(pdev, sig, comp, loc, data, len, fmt, args...) \ + xe_log_emit((pdev), CPER_SEV_INFORMATIONAL, (sig), (comp), (loc), \ + (data), (len), fmt, ##args) + +#define xe_log_location_type(any) \ + _Generic((any), \ + struct xe_gt * : XE_LOG_LOCATION_TYPE_GT, \ + const struct xe_gt * : XE_LOG_LOCATION_TYPE_GT, \ + struct xe_tile * : XE_LOG_LOCATION_TYPE_TILE, \ + const struct xe_tile * : XE_LOG_LOCATION_TYPE_TILE, \ + struct xe_device * : XE_LOG_LOCATION_TYPE_DEVICE, \ + const struct xe_device * : XE_LOG_LOCATION_TYPE_DEVICE, \ + struct pci_dev * : XE_LOG_LOCATION_TYPE_DEVICE, \ + struct device * : XE_LOG_LOCATION_TYPE_DEVICE) + +#define xe_log_location(any) \ + PREP_XE_LOG_LOCATION(xe_log_location_type(any), xe_any_id(any)) + +/** + * xe_log_from() - Emit a structured SIGID log entry using @any pointer as location. + * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to + * @cper_sev: CPER severity (CPER_SEV_FATAL, CPER_SEV_RECOVERABLE, ...) + * @sigid: signature identifier, see &enum xe_sigid + * @component: component identifer + * @data: pointer to the additional details, or ERR_PTR, or NULL if not applicable + * @len: length of the @data in bytes, or 0 if not applicable + * @fmt: printf-style format string + * @args: arguments for the @fmt format string + * + * The location used to emit SIGID entry will be based on the @any pointer type. + * See xe_log_emit() for more details. + */ +#define xe_log_from(any, cper_sev, sigid, component, data, len, fmt, args...) do { \ + typeof(any) ___any = (any); \ + xe_log_emit(xe_any_to_pdev(___any), (cper_sev), (sigid), (component), \ + xe_log_location(___any), (data), (len), fmt, ##args); \ +} while (0) + +#define xe_log_from_fatal(any, sig, comp, data, len, fmt, args...) \ + xe_log_from((any), CPER_SEV_FATAL, (sig), (comp), \ + (data), (len), fmt, ##args) + +#define xe_log_from_recoverable(any, sig, comp, data, len, fmt, args...) \ + xe_log_from((any), CPER_SEV_RECOVERABLE, (sig), (comp), \ + (data), (len), fmt, ##args) + +#define xe_log_from_corrected(any, sig, comp, data, len, fmt, args...) \ + xe_log_from((any), CPER_SEV_CORRECTED, (sig), (comp), \ + (data), (len), fmt, ##args) + +#define xe_log_from_info(any, sig, comp, data, len, fmt, args...) \ + xe_log_from((any), CPER_SEV_INFORMATIONAL, (sig), (comp), \ + (data), (len), fmt, ##args) + +/** + * xe_log_comp() - Emit a structured SIGID log entry on the component behalf. + * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to + * @cper_sev: CPER severity (CPER_SEV_FATAL, CPER_SEV_RECOVERABLE, ...) + * @TAG: the component tag to use + * @data: pointer to the additional details, or ERR_PTR, or NULL if not applicable + * @len: length of the @data in bytes, or 0 if not applicable + * @fmt: printf-style free text format string (not a stable interface) + * @args: arguments for the @fmt format string + * + * The SIGID will be determined from the component's @TAG. + * The component identifier will be determined from the component's @TAG. + * The location used to emit SIGID entry will be based on the @any pointer type. + */ +#define xe_log_comp(any, cper_sev, TAG, data, len, fmt, args...) \ + xe_log_from((any), (cper_sev), (int)XE_LOG_COMPONENT_##TAG##_SIGID, \ + XE_LOG_COMPONENT_##TAG, (data), (len), fmt, ##args) + +#define xe_log_comp_fatal(any, TAG, data, len, fmt, args...) \ + xe_log_comp((any), CPER_SEV_FATAL, TAG, (data), (len), fmt, ##args) + +#define xe_log_comp_recoverable(any, TAG, data, len, fmt, args...) \ + xe_log_comp((any), CPER_SEV_RECOVERABLE, TAG, (data), (len), fmt, ##args) + +#define xe_log_comp_corrected(any, TAG, data, len, fmt, args...) \ + xe_log_comp((any), CPER_SEV_CORRECTED, TAG, (data), (len), fmt, ##args) + +#define xe_log_comp_info(any, TAG, data, len, fmt, args...) \ + xe_log_comp((any), CPER_SEV_INFORMATIONAL, TAG, (data), (len), fmt, ##args) + +/** + * xe_log_err() - Emit a structured SIGID error log entry on the component behalf. + * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to + * @TAG: the component tag to use + * @err: negative errno for the failing operation, or 0 if not applicable + * @fmt: printf-style free text format string (not a stable interface) + * @args: arguments for the @fmt format string + * + * The log entry will be emitted with @CPER_SEV_RECOVERABLE severity. + */ +#define xe_log_err(any, TAG, err, fmt, args...) \ + xe_log_comp_recoverable((any), TAG, ERR_PTR(err), 0, fmt, ##args) + +/** + * xe_log_err_fatal() - Emit a structured SIGID error log entry on the component behalf. + * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to + * @TAG: the component tag to use + * @err: negative errno for the failing operation, or 0 if not applicable + * @fmt: printf-style free text format string (not a stable interface) + * @args: arguments for the @fmt format string + * + * The log entry will be emitted with @CPER_SEV_FATAL severity. + */ +#define xe_log_err_fatal(any, TAG, err, fmt, args...) \ + xe_log_comp_fatal((any), TAG, ERR_PTR(err), 0, fmt, ##args) + +/** + * xe_log_err_corrected() - Emit a structured SIGID error log entry on the component behalf. + * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to + * @TAG: the component tag to use + * @err: negative errno for the failing operation, or 0 if not applicable + * @fmt: printf-style free text format string (not a stable interface) + * @args: arguments for the @fmt format string + * + * The log entry will be emitted with @CPER_SEV_CORRECTED severity. + */ +#define xe_log_err_corrected(any, TAG, err, fmt, args...) \ + xe_log_comp_corrected((any), TAG, ERR_PTR(err), 0, fmt, ##args) + +/** + * xe_log_err_info() - Emit a structured SIGID error log entry on the component behalf. + * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to + * @TAG: the component tag to use + * @err: negative errno for the failing operation, or 0 if not applicable + * @fmt: printf-style free text format string (not a stable interface) + * @args: arguments for the @fmt format string + * + * The log entry will be emitted with @CPER_SEV_INFORMATIONAL severity. + */ +#define xe_log_err_info(any, TAG, err, fmt, args...) \ + xe_log_comp_info((any), TAG, ERR_PTR(err), 0, fmt, ##args) + +/** + * xe_log_info() - Emit a structured SIGID information log entry on the component behalf. + * @any: the &xe_device or &xe_tile or &xe_gt pointer this report relates to + * @TAG: the component tag to use + * @fmt: printf-style free text format string (not a stable interface) + * @args: arguments for the @fmt format string + * + * The log entry will be emitted with @CPER_SEV_INFORMATIONAL severity. + */ +#define xe_log_info(any, TAG, fmt, args...) \ + xe_log_err_info((any), TAG, 0, fmt, ##args) + +#endif diff --git a/drivers/gpu/drm/xe/xe_mert.c b/drivers/gpu/drm/xe/xe_mert.c index f637df95418b..32700c19a1df 100644 --- a/drivers/gpu/drm/xe/xe_mert.c +++ b/drivers/gpu/drm/xe/xe_mert.c @@ -77,7 +77,7 @@ static void mert_handle_cat_error(struct xe_device *xe) xe_device_declare_wedged(xe); break; case CATERR_LMTT_FAULT: - xe_sriov_dbg(xe, "MERT: CAT_ERR: VF%u LMTT fault!\n", vfid); + xe_sriov_dbg_ratelimited(xe, "MERT: CAT_ERR: VF%u LMTT fault!\n", vfid); /* XXX: track/report malicious VF activity */ break; default: diff --git a/drivers/gpu/drm/xe/xe_migrate.c b/drivers/gpu/drm/xe/xe_migrate.c index f79d0047bec6..75b83687f1b5 100644 --- a/drivers/gpu/drm/xe/xe_migrate.c +++ b/drivers/gpu/drm/xe/xe_migrate.c @@ -493,7 +493,6 @@ int xe_migrate_init(struct xe_migrate *m) */ m->q = xe_exec_queue_create(xe, vm, logical_mask, 1, hwe0, EXEC_QUEUE_FLAG_KERNEL | - EXEC_QUEUE_FLAG_PERMANENT | EXEC_QUEUE_FLAG_HIGH_PRIORITY | EXEC_QUEUE_FLAG_MIGRATE | EXEC_QUEUE_FLAG_LOW_LATENCY, 0); @@ -501,7 +500,6 @@ int xe_migrate_init(struct xe_migrate *m) m->q = xe_exec_queue_create_class(xe, primary_gt, vm, XE_ENGINE_CLASS_COPY, EXEC_QUEUE_FLAG_KERNEL | - EXEC_QUEUE_FLAG_PERMANENT | EXEC_QUEUE_FLAG_MIGRATE, 0); } if (IS_ERR(m->q)) { diff --git a/drivers/gpu/drm/xe/xe_module.c b/drivers/gpu/drm/xe/xe_module.c index 848d65265443..4bc28dfc1992 100644 --- a/drivers/gpu/drm/xe/xe_module.c +++ b/drivers/gpu/drm/xe/xe_module.c @@ -29,6 +29,7 @@ struct xe_modparam xe_modparam = { .max_vfs = XE_DEFAULT_MAX_VFS, #endif .wedged_mode = XE_DEFAULT_WEDGED_MODE, + .num_pf_work = XE_DEFAULT_NUM_PF_WORK, .svm_notifier_size = XE_DEFAULT_SVM_NOTIFIER_SIZE, /* the rest are 0 by default */ }; @@ -81,6 +82,9 @@ MODULE_PARM_DESC(wedged_mode, "Module's default policy for the wedged mode (0=never, 1=upon-critical-error, 2=upon-any-hang-no-reset " "[default=" XE_DEFAULT_WEDGED_MODE_STR "])"); +module_param_named(num_pf_work, xe_modparam.num_pf_work, uint, 0600); +MODULE_PARM_DESC(num_pf_work, "Number of page fault work threads, default=2, min=1, max=8"); + static int xe_check_nomodeset(void) { if (drm_firmware_drivers_only()) diff --git a/drivers/gpu/drm/xe/xe_module.h b/drivers/gpu/drm/xe/xe_module.h index a0eb7db07770..6272d9e41207 100644 --- a/drivers/gpu/drm/xe/xe_module.h +++ b/drivers/gpu/drm/xe/xe_module.h @@ -23,6 +23,7 @@ struct xe_modparam { unsigned int max_vfs; #endif unsigned int wedged_mode; + unsigned int num_pf_work; u32 svm_notifier_size; }; diff --git a/drivers/gpu/drm/xe/xe_oa.c b/drivers/gpu/drm/xe/xe_oa.c index 9c5384b95c63..b460fcdfca15 100644 --- a/drivers/gpu/drm/xe/xe_oa.c +++ b/drivers/gpu/drm/xe/xe_oa.c @@ -2581,10 +2581,11 @@ static u32 __hwe_oam_unit(struct xe_hw_engine *hwe) return XE_OA_UNIT_INVALID; else if (!IS_DGFX(gt_to_xe(hwe->gt))) return XE_OAM_UNIT_SCMI_0; - else if (hwe->class == XE_ENGINE_CLASS_VIDEO_DECODE) - return (hwe->instance / 2 & 0x1) + 1; - else if (hwe->class == XE_ENGINE_CLASS_VIDEO_ENHANCE) + else if (hwe->class == XE_ENGINE_CLASS_VIDEO_ENHANCE && + MEDIA_VERx100(gt_to_xe(hwe->gt)) < 3500) return (hwe->instance & 0x1) + 1; + else + return (hwe->instance / 2 & 0x1) + 1; return XE_OA_UNIT_INVALID; } diff --git a/drivers/gpu/drm/xe/xe_pagefault.c b/drivers/gpu/drm/xe/xe_pagefault.c index dd3c068e1a39..9e89d7b37395 100644 --- a/drivers/gpu/drm/xe/xe_pagefault.c +++ b/drivers/gpu/drm/xe/xe_pagefault.c @@ -14,6 +14,7 @@ #include "xe_gt_types.h" #include "xe_gt_stats.h" #include "xe_hw_engine.h" +#include "xe_log.h" #include "xe_pagefault.h" #include "xe_pagefault_types.h" #include "xe_svm.h" @@ -35,6 +36,73 @@ * xe_pagefault.c implements the consumer layer. */ +/** + * DOC: Xe page fault cache + * + * Some Xe hardware can trigger “fault storms,” which are many page faults to + * the same address within a short period of time. An example is many EU threads + * faulting on the same page simultaneously. With the current page fault locking + * structure, only one page fault for a given address range can be processed at + * a time. This causes head-of-queue blocking across workers, killing + * parallelism. If the page fault handler must repeatedly look up resources + * (VMAs, ranges) to determine that the pages are valid for each fault in the + * storm, the time complexity grows rapidly. + * + * To address this, each page fault worker maintains a cache of the active fault + * being processed. Subsequent faults that hit in the cache are chained to the + * pending fault, and all chained faults are acknowledged once the initial fault + * completes. This alleviates head-of-queue blocking and quickly chains faults + * in the upper layers, avoiding expensive lookups in the main fault-handling + * path. + * + * Faults are buffered in the page fault queue in a way that provides stable + * storage for outstanding faults. In particular, faults may be chained directly + * while still resident in the queue storage (i.e., outside the worker’s current + * head/tail dequeue position). This allows the IRQ handler to match newly + * arrived faults against the per-worker cache and immediately chain cache hits + * onto the active fault under the queue lock, without allocating memory or + * waiting for the worker to pop the fault first. + * + * A per-fault state field is used to assert correctness of these invariants. + * The state tracks whether an entry is free, queued, chained, or currently + * active. Transitions are performed under the page fault queue lock, and the + * worker acknowledges faults by walking the chain and returning entries to the + * free state once they are complete. + */ + +/** + * enum xe_pagefault_alloc_state - lifetime state for a page fault queue entry + * @XE_PAGEFAULT_ALLOC_STATE_FREE: + * Entry is unused and may be overwritten by the producer, consumer retry + * or requeue.. + * @XE_PAGEFAULT_ALLOC_STATE_QUEUED: + * Entry has been enqueued and may be dequeued by a worker. + * @XE_PAGEFAULT_ALLOC_STATE_ACTIVE: + * Entry has been dequeued and is the worker's currently serviced fault. + * The worker may attach additional faults to it via consumer.next. + * @XE_PAGEFAULT_ALLOC_STATE_CHAINED: + * Entry is not independently serviced; it has been chained onto an + * ACTIVE entry via consumer.next and will be acknowledged when the + * leading fault completes. + * @XE_PAGEFAULT_ALLOC_STATE_COUNT: + * Count of allocation states. + * + * The page fault queue provides stable storage for outstanding faults so the + * IRQ handler can chain new cache hits directly onto a worker's active fault. + * Because entries may remain referenced outside the consumer dequeue window, + * the producer must only write into entries in the FREE state. + * + * State transitions are protected by the page fault queue lock. Workers return + * entries to FREE after acknowledging the fault (either as ACTIVE or CHAINED). + */ +enum xe_pagefault_alloc_state { + XE_PAGEFAULT_ALLOC_STATE_FREE = 0, + XE_PAGEFAULT_ALLOC_STATE_QUEUED = 1, + XE_PAGEFAULT_ALLOC_STATE_CHAINED = 2, + XE_PAGEFAULT_ALLOC_STATE_ACTIVE = 3, + XE_PAGEFAULT_ALLOC_STATE_COUNT = 4, +}; + static int xe_pagefault_entry_size(void) { /* @@ -77,16 +145,16 @@ static int xe_pagefault_begin(struct drm_exec *exec, struct xe_vma *vma, } static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma, - bool atomic) + struct xe_pagefault *pf, bool atomic) { struct xe_vm *vm = xe_vma_vm(vma); struct xe_tile *tile = gt_to_tile(gt); struct xe_validation_ctx ctx; struct drm_exec exec; struct dma_fence *fence; - int err, needs_vram; + int err = 0, needs_vram; - lockdep_assert_held_write(&vm->lock); + lockdep_assert_held(&vm->lock); needs_vram = xe_vma_need_vram_for_atomic(vm->xe, vma, atomic); if (needs_vram < 0 || (needs_vram && xe_vma_is_userptr(vma))) @@ -98,50 +166,59 @@ static int xe_pagefault_handle_vma(struct xe_gt *gt, struct xe_vma *vma, trace_xe_vma_pagefault(vma); + guard(mutex)(&vma->fault_lock); + /* Check if VMA is valid, opportunistic check only */ if (xe_vm_has_valid_gpu_mapping(tile, vma->tile_present, - vma->tile_invalidated) && !atomic) + vma->tile_invalidated) && !atomic) { + xe_pagefault_set_start_addr(pf, xe_vma_start(vma)); + xe_pagefault_set_end_addr(pf, xe_vma_end(vma)); return 0; + } -retry_userptr: - if (xe_vma_is_userptr(vma) && - xe_vma_userptr_check_repin(to_userptr_vma(vma))) { - struct xe_userptr_vma *uvma = to_userptr_vma(vma); + do { + if (xe_vma_is_userptr(vma) && + xe_vma_userptr_check_repin(to_userptr_vma(vma))) { + struct xe_userptr_vma *uvma = to_userptr_vma(vma); - err = xe_vma_userptr_pin_pages(uvma); - if (err) - return err; - } + err = xe_vma_userptr_pin_pages(uvma); + if (err) + return err; + } - /* Lock VM and BOs dma-resv */ - xe_validation_ctx_init(&ctx, &vm->xe->val, &exec, (struct xe_val_flags) {}); - drm_exec_until_all_locked(&exec) { - err = xe_pagefault_begin(&exec, vma, tile->mem.vram, - needs_vram == 1); - drm_exec_retry_on_contention(&exec); - xe_validation_retry_on_oom(&ctx, &err); - if (err) - goto unlock_dma_resv; - - /* Bind VMA only to the GT that has faulted */ - trace_xe_vma_pf_bind(vma); - xe_vm_set_validation_exec(vm, &exec); - fence = xe_vma_rebind(vm, vma, BIT(tile->id)); - xe_vm_set_validation_exec(vm, NULL); - if (IS_ERR(fence)) { - err = PTR_ERR(fence); + /* Lock VM and BOs dma-resv */ + xe_validation_ctx_init(&ctx, &vm->xe->val, &exec, + (struct xe_val_flags) {}); + drm_exec_until_all_locked(&exec) { + err = xe_pagefault_begin(&exec, vma, tile->mem.vram, + needs_vram == 1); + drm_exec_retry_on_contention(&exec); xe_validation_retry_on_oom(&ctx, &err); - goto unlock_dma_resv; + if (err) + break; + + /* Bind VMA only to the GT that has faulted */ + trace_xe_vma_pf_bind(vma); + xe_vm_set_validation_exec(vm, &exec); + fence = xe_vma_rebind(vm, vma, BIT(tile->id)); + xe_vm_set_validation_exec(vm, NULL); + if (IS_ERR(fence)) { + err = PTR_ERR(fence); + xe_validation_retry_on_oom(&ctx, &err); + break; + } } - } + xe_validation_ctx_fini(&ctx); + } while (err == -EAGAIN); - dma_fence_wait(fence, false); - dma_fence_put(fence); + if (!err) { + /* Give hint to immediately ack faults */ + xe_pagefault_set_start_addr(pf, xe_vma_start(vma)); + xe_pagefault_set_end_addr(pf, xe_vma_end(vma)); -unlock_dma_resv: - xe_validation_ctx_fini(&ctx); - if (err == -EAGAIN) - goto retry_userptr; + dma_fence_wait(fence, false); + dma_fence_put(fence); + } return err; } @@ -184,10 +261,7 @@ static int xe_pagefault_service(struct xe_pagefault *pf) if (IS_ERR(vm)) return PTR_ERR(vm); - /* - * TODO: Change to read lock? Using write lock for simplicity. - */ - down_write(&vm->lock); + down_read(&vm->lock); if (xe_vm_is_closed(vm)) { err = -ENOENT; @@ -209,46 +283,275 @@ static int xe_pagefault_service(struct xe_pagefault *pf) atomic = xe_pagefault_access_is_atomic(pf->consumer.access_type); if (xe_vma_is_cpu_addr_mirror(vma)) - err = xe_svm_handle_pagefault(vm, vma, gt, + err = xe_svm_handle_pagefault(vm, vma, pf, gt, pf->consumer.page_addr, atomic); else - err = xe_pagefault_handle_vma(gt, vma, atomic); + err = xe_pagefault_handle_vma(gt, vma, pf, atomic); unlock_vm: - if (!err) - vm->usm.last_fault_vma = vma; - up_write(&vm->lock); + up_read(&vm->lock); xe_vm_put(vm); return err; } -static bool xe_pagefault_queue_pop(struct xe_pagefault_queue *pf_queue, - struct xe_pagefault *pf) +#define XE_PAGEFAULT_CACHE_START_INVALID U64_MAX +#define xe_pagefault_cache_start_invalidate(val) \ + (val = XE_PAGEFAULT_CACHE_START_INVALID) + +static void +xe_pagefault_cache_invalidate(struct xe_pagefault_queue *pf_queue, + struct xe_pagefault_work *pf_work) +{ + lockdep_assert_held(&pf_queue->lock); + + xe_pagefault_cache_start_invalidate(pf_work->cache.start); +} + +static bool xe_pagefault_queue_full(struct xe_pagefault_queue *pf_queue) +{ + lockdep_assert_held(&pf_queue->lock); + + return CIRC_SPACE(pf_queue->head, pf_queue->tail, + pf_queue->size) <= xe_pagefault_entry_size(); +} + +static struct xe_pagefault * +xe_pagefault_queue_add(struct xe_pagefault_queue *pf_queue, + struct xe_pagefault *pf) { - bool found_fault = false; + struct xe_device *xe = container_of(pf_queue, typeof(*xe), + usm.pf_queue); + struct xe_pagefault *lpf; + + lockdep_assert_held(&pf_queue->lock); - spin_lock_irq(&pf_queue->lock); - if (pf_queue->tail != pf_queue->head) { - memcpy(pf, pf_queue->data + pf_queue->tail, sizeof(*pf)); - pf_queue->tail = (pf_queue->tail + xe_pagefault_entry_size()) % + do { + /* Not possible, warn on and drop page fault */ + if (WARN_ON_ONCE(xe_pagefault_queue_full(pf_queue))) { + xe_log_err(xe, PAGEFAULT, -ENOSPC, "Queue full!\n"); + return NULL; + } + + lpf = (pf_queue->data + pf_queue->head); + pf_queue->head = (pf_queue->head + xe_pagefault_entry_size()) % pf_queue->size; - found_fault = true; + } while (lpf->consumer.alloc_state != XE_PAGEFAULT_ALLOC_STATE_FREE); + + xe_assert(xe, lpf != pf); + memcpy(lpf, pf, sizeof(*pf)); + lpf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_QUEUED; + + return lpf; +} + +static struct xe_pagefault * +xe_pagefault_queue_unchain_requeue(struct xe_pagefault_queue *pf_queue, + struct xe_pagefault *pf, struct xe_gt *gt) +{ + struct xe_device *xe = container_of(pf_queue, typeof(*xe), + usm.pf_queue); + struct xe_pagefault *next = pf->consumer.next, *lpf; + + lockdep_assert_held(&pf_queue->lock); + xe_assert(xe, pf->consumer.alloc_state == + XE_PAGEFAULT_ALLOC_STATE_CHAINED); + + xe_gt_stats_incr(gt, XE_GT_STATS_ID_CHAIN_MISMATCH_PAGEFAULT_COUNT, 1); + + pf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_FREE; + lpf = xe_pagefault_queue_add(pf_queue, pf); + if (lpf) { + lpf->consumer.next = NULL; + lpf->consumer.fault_type_level |= XE_PAGEFAULT_REQUEUE_MASK; + } + + return next; +} + +static bool xe_pagefault_match(struct xe_pagefault *pf, u64 start, + u64 end, u64 cache_asid) +{ + struct xe_device *xe = gt_to_xe(pf->gt); + u64 page_addr = pf->consumer.page_addr; + u32 pf_asid = pf->consumer.asid; + + xe_assert(xe, pf->consumer.alloc_state != + XE_PAGEFAULT_ALLOC_STATE_FREE); + + return page_addr >= start && page_addr < end && + pf_asid == cache_asid; +} + +static bool xe_pagefault_try_chain(struct xe_pagefault_queue *pf_queue, + struct xe_pagefault *pf) +{ + struct xe_device *xe = container_of(pf_queue, typeof(*xe), + usm.pf_queue); + struct xe_pagefault_work *pf_work; + bool requeue = FIELD_GET(XE_PAGEFAULT_REQUEUE_MASK, + pf->consumer.fault_type_level); + int i; + + lockdep_assert_held(&pf_queue->lock); + xe_assert(xe, pf->consumer.alloc_state == + XE_PAGEFAULT_ALLOC_STATE_QUEUED); + + /* + * If this is a retry, we may already have a chain attached. In that + * case, we cannot hit in the cache because chains cannot easily be + * combined. + */ + if (pf->consumer.next) + return false; + + for (i = 0, pf_work = xe->usm.pf_workers; + i < xe->info.num_pf_work; ++i, ++pf_work) { + u64 start = pf_work->cache.start; + u64 end = requeue ? start + SZ_4K : pf_work->cache.end; + u32 asid = pf_work->cache.asid; + + if (xe_pagefault_match(pf, start, end, asid)) { + xe_assert(xe, pf_work->cache.pf->consumer.alloc_state == + XE_PAGEFAULT_ALLOC_STATE_ACTIVE); + + if (pf->producer.private != + pf_work->cache.pf->producer.private) + continue; + + xe_gt_stats_incr(pf->gt, + XE_GT_STATS_ID_CHAIN_PAGEFAULT_COUNT, + 1); + + pf->consumer.alloc_state = + XE_PAGEFAULT_ALLOC_STATE_CHAINED; + pf->consumer.next = pf_work->cache.pf->consumer.next; + pf_work->cache.pf->consumer.next = pf; + + return true; + } + } + + return false; +} + +static void xe_pagefault_queue_advance(struct xe_pagefault_queue *pf_queue) +{ + lockdep_assert_held(&pf_queue->lock); + + pf_queue->tail = (pf_queue->tail + xe_pagefault_entry_size()) % + pf_queue->size; +} + +static struct xe_pagefault * +xe_pagefault_queue_tail_fault(struct xe_pagefault_queue *pf_queue) +{ + lockdep_assert_held(&pf_queue->lock); + + return pf_queue->data + pf_queue->tail; +} + +static bool xe_pagefault_queue_empty(struct xe_pagefault_queue *pf_queue) +{ + lockdep_assert_held(&pf_queue->lock); + + return pf_queue->head == pf_queue->tail; +} + +static bool xe_pagefault_queue_pop(struct xe_pagefault_queue *pf_queue, + struct xe_pagefault **pf, int id) +{ + struct xe_device *xe = container_of(pf_queue, typeof(*xe), + usm.pf_queue); + struct xe_pagefault_work *pf_work, *__pf_work; + struct xe_pagefault *lpf; + size_t align = SZ_2M; + int i; + + guard(spinlock_irq)(&pf_queue->lock); + + for (*pf = NULL; !*pf;) { + if (xe_pagefault_queue_empty(pf_queue)) + return false; + + lpf = xe_pagefault_queue_tail_fault(pf_queue); + xe_pagefault_queue_advance(pf_queue); + + if (lpf->consumer.alloc_state != + XE_PAGEFAULT_ALLOC_STATE_QUEUED) + continue; + + if (xe_pagefault_try_chain(pf_queue, lpf)) + continue; + + *pf = lpf; /* Hand back page fault for processing */ } - spin_unlock_irq(&pf_queue->lock); - return found_fault; + /* + * No cache hit; allocate a new cache entry. We assume most faults + * within a 2M range will hit the same pages. If this assumption proves + * false, the mismatched fault is requeued after the initial fault is + * acknowledged. + */ + pf_work = xe->usm.pf_workers + id; + if (FIELD_GET(XE_PAGEFAULT_REQUEUE_MASK, + lpf->consumer.fault_type_level)) + align = SZ_4K; + pf_work->cache.start = ALIGN_DOWN(lpf->consumer.page_addr, align); + pf_work->cache.end = pf_work->cache.start + align; + pf_work->cache.asid = lpf->consumer.asid; + pf_work->cache.pf = lpf; + lpf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_ACTIVE; + + for (i = 0, __pf_work = xe->usm.pf_workers; + i < xe->info.num_pf_work; ++i, ++__pf_work) { + u64 cache_start = __pf_work->cache.start; + + if (__pf_work == pf_work) + continue; + + if (cache_start != XE_PAGEFAULT_CACHE_START_INVALID) { + xe_gt_stats_incr(xe_root_mmio_gt(xe), + XE_GT_STATS_ID_PARALLEL_PAGEFAULT_COUNT, + 1); + break; + } + } + + /* Drain queue until empty or new fault found */ + while (1) { + if (xe_pagefault_queue_empty(pf_queue)) + break; + + lpf = xe_pagefault_queue_tail_fault(pf_queue); + + if (lpf->consumer.alloc_state != + XE_PAGEFAULT_ALLOC_STATE_QUEUED) { + xe_pagefault_queue_advance(pf_queue); + continue; + } + + if (!xe_pagefault_try_chain(pf_queue, lpf)) + break; + + xe_pagefault_queue_advance(pf_queue); + } + + return true; } static void xe_pagefault_print(struct xe_pagefault *pf) { + u8 engine_class = FIELD_GET(XE_PAGEFAULT_ENGINE_CLASS_MASK, + pf->consumer.engine_class_instance); + xe_gt_info(pf->gt, "\n\tASID: %d\n" "\tFaulted Address: 0x%08x%08x\n" "\tFaultType: %lu\n" "\tAccessType: %lu\n" "\tFaultLevel: %lu\n" "\tEngineClass: %d %s\n" - "\tEngineInstance: %d\n", + "\tEngineInstance: %lu\n", pf->consumer.asid, upper_32_bits(pf->consumer.page_addr), lower_32_bits(pf->consumer.page_addr), @@ -258,9 +561,10 @@ static void xe_pagefault_print(struct xe_pagefault *pf) pf->consumer.access_type), FIELD_GET(XE_PAGEFAULT_LEVEL_MASK, pf->consumer.fault_type_level), - pf->consumer.engine_class, - xe_hw_engine_class_to_str(pf->consumer.engine_class), - pf->consumer.engine_instance); + engine_class, + xe_hw_engine_class_to_str(engine_class), + FIELD_GET(XE_PAGEFAULT_ENGINE_INSTANCE_MASK, + pf->consumer.engine_class_instance)); } static void xe_pagefault_save_to_vm(struct xe_device *xe, struct xe_pagefault *pf) @@ -290,42 +594,104 @@ static void xe_pagefault_save_to_vm(struct xe_device *xe, struct xe_pagefault *p static void xe_pagefault_queue_work(struct work_struct *w) { - struct xe_pagefault_queue *pf_queue = - container_of(w, typeof(*pf_queue), worker); - struct xe_pagefault pf; + struct xe_pagefault_work *pf_work = + container_of(w, typeof(*pf_work), work); + struct xe_device *xe = pf_work->xe; + struct xe_pagefault_queue *pf_queue = &xe->usm.pf_queue; + struct xe_pagefault *pf; + ktime_t start = xe_gt_stats_ktime_get(); unsigned long threshold; + u64 cache_start = XE_PAGEFAULT_CACHE_START_INVALID, cache_end = 0; + u32 cache_asid = 0; #define USM_QUEUE_MAX_RUNTIME_MS 20 threshold = jiffies + msecs_to_jiffies(USM_QUEUE_MAX_RUNTIME_MS); - while (xe_pagefault_queue_pop(pf_queue, &pf)) { - int err; + while (xe_pagefault_queue_pop(pf_queue, &pf, pf_work->id)) { + const struct xe_pagefault_ops *ops = pf->producer.ops; + void *private = pf->producer.private; + struct xe_gt *gt = pf->gt; + u32 asid = pf->consumer.asid; + int err = 0; + bool invalidated = false; + + /* Last fault same address, ack immediately */ + if (xe_pagefault_match(pf, cache_start, cache_end, cache_asid)) { + xe_gt_stats_incr(gt, XE_GT_STATS_ID_LAST_PAGEFAULT_COUNT, 1); + goto ack_fault; + } - if (!pf.gt) /* Fault squashed during reset */ - continue; + err = xe_pagefault_service(pf); - err = xe_pagefault_service(&pf); if (err) { - xe_pagefault_save_to_vm(gt_to_xe(pf.gt), &pf); - if (!(pf.consumer.access_type & XE_PAGEFAULT_ACCESS_PREFETCH)) { - xe_pagefault_print(&pf); - xe_gt_info(pf.gt, "Fault response: Unsuccessful %pe\n", - ERR_PTR(err)); + if (!(pf->consumer.access_type & XE_PAGEFAULT_ACCESS_PREFETCH)) { + xe_pagefault_save_to_vm(gt_to_xe(gt), pf); + xe_pagefault_cache_start_invalidate(cache_start); + xe_pagefault_print(pf); + xe_log_err_info(pf->gt, PAGEFAULT, err, "Unsuccessful response\n"); } else { - xe_gt_stats_incr(pf.gt, XE_GT_STATS_ID_INVALID_PREFETCH_PAGEFAULT_COUNT, 1); - xe_gt_dbg(pf.gt, "Prefetch Fault response: Unsuccessful %pe\n", + xe_gt_stats_incr(pf->gt, XE_GT_STATS_ID_INVALID_PREFETCH_PAGEFAULT_COUNT, 1); + xe_gt_dbg(pf->gt, "Prefetch Fault response: Unsuccessful %pe\n", ERR_PTR(err)); } + } else { + /* Cache valid fault locally */ + cache_start = xe_pagefault_start_addr(pf); + cache_end = xe_pagefault_end_addr(pf); + cache_asid = asid; } - pf.producer.ops->ack_fault(&pf, err); +ack_fault: + xe_assert(xe, pf->consumer.alloc_state == + XE_PAGEFAULT_ALLOC_STATE_ACTIVE); + xe_assert(xe, pf == pf_work->cache.pf); + + ops->ack_fault_begin(private); + while (pf) { + xe_assert(xe, pf->consumer.alloc_state == + XE_PAGEFAULT_ALLOC_STATE_ACTIVE); + xe_assert(xe, ops == pf->producer.ops); + xe_assert(xe, gt == pf->gt); + + ops->ack_fault(pf, err); + + spin_lock_irq(&pf_queue->lock); + + if (!invalidated) { + invalidated = true; + xe_pagefault_cache_invalidate(pf_queue, + pf_work); + } + + pf->consumer.alloc_state = XE_PAGEFAULT_ALLOC_STATE_FREE; + pf = pf->consumer.next; + + /* + * Requeue chained faults which do not match the last + * fault processed + */ + while (pf && !xe_pagefault_match(pf, cache_start, + cache_end, cache_asid)) + pf = xe_pagefault_queue_unchain_requeue(pf_queue, pf, gt); + + /* Ensure resets are safe */ + if (pf) + pf->consumer.alloc_state = + XE_PAGEFAULT_ALLOC_STATE_ACTIVE; + + spin_unlock_irq(&pf_queue->lock); + } + ops->ack_fault_end(private); if (time_after(jiffies, threshold)) { - queue_work(gt_to_xe(pf.gt)->usm.pf_wq, w); + queue_work(xe->usm.pagefault_wq, w); break; } } #undef USM_QUEUE_MAX_RUNTIME_MS + + xe_gt_stats_incr(xe_root_mmio_gt(xe), XE_GT_STATS_ID_PAGEFAULT_US, + xe_gt_stats_ktime_us_delta(start)); } static int xe_pagefault_queue_init(struct xe_device *xe, @@ -366,7 +732,6 @@ static int xe_pagefault_queue_init(struct xe_device *xe, xe_pagefault_entry_size(), total_num_eus, pf_queue->size); spin_lock_init(&pf_queue->lock); - INIT_WORK(&pf_queue->worker, xe_pagefault_queue_work); pf_queue->data = drmm_kzalloc(&xe->drm, pf_queue->size, GFP_KERNEL); if (!pf_queue->data) @@ -379,7 +744,8 @@ static void xe_pagefault_fini(void *arg) { struct xe_device *xe = arg; - destroy_workqueue(xe->usm.pf_wq); + destroy_workqueue(xe->usm.prefetch_wq); + destroy_workqueue(xe->usm.pagefault_wq); } /** @@ -397,22 +763,39 @@ int xe_pagefault_init(struct xe_device *xe) if (!xe->info.has_usm) return 0; - xe->usm.pf_wq = alloc_workqueue("xe_page_fault_work_queue", - WQ_UNBOUND | WQ_HIGHPRI, - XE_PAGEFAULT_QUEUE_COUNT); - if (!xe->usm.pf_wq) + xe->usm.pagefault_wq = alloc_workqueue("xe_page_fault_work_queue", + WQ_UNBOUND | WQ_HIGHPRI, + xe->info.num_pf_work); + if (!xe->usm.pagefault_wq) return -ENOMEM; - for (i = 0; i < XE_PAGEFAULT_QUEUE_COUNT; ++i) { - err = xe_pagefault_queue_init(xe, xe->usm.pf_queue + i); - if (err) - goto err_out; + xe->usm.prefetch_wq = alloc_workqueue("xe_prefetch_work_queue", + WQ_UNBOUND, + xe->info.num_pf_work); + if (!xe->usm.prefetch_wq) { + err = -ENOMEM; + goto err_pagefault_wq; + } + + err = xe_pagefault_queue_init(xe, &xe->usm.pf_queue); + if (err) + goto err_out; + + for (i = 0; i < xe->info.num_pf_work; ++i) { + struct xe_pagefault_work *pf_work = xe->usm.pf_workers + i; + + pf_work->xe = xe; + pf_work->id = i; + xe_pagefault_cache_start_invalidate(pf_work->cache.start); + INIT_WORK(&pf_work->work, xe_pagefault_queue_work); } return devm_add_action_or_reset(xe->drm.dev, xe_pagefault_fini, xe); err_out: - destroy_workqueue(xe->usm.pf_wq); + destroy_workqueue(xe->usm.prefetch_wq); +err_pagefault_wq: + destroy_workqueue(xe->usm.pagefault_wq); return err; } @@ -427,15 +810,23 @@ static void xe_pagefault_queue_reset(struct xe_device *xe, struct xe_gt *gt, /* Squash all pending faults on the GT */ - spin_lock_irq(&pf_queue->lock); - for (i = pf_queue->tail; i != pf_queue->head; - i = (i + xe_pagefault_entry_size()) % pf_queue->size) { + guard(spinlock_irq)(&pf_queue->lock); + + for (i = 0; i < pf_queue->size; i += xe_pagefault_entry_size()) { struct xe_pagefault *pf = pf_queue->data + i; + bool active = pf->consumer.alloc_state == + XE_PAGEFAULT_ALLOC_STATE_ACTIVE; - if (pf->gt == gt) - pf->gt = NULL; + if (pf->gt != gt || active) { + if (active) + pf->consumer.next = NULL; + continue; + } + + pf->consumer.alloc_state = + XE_PAGEFAULT_ALLOC_STATE_FREE; + pf->consumer.next = NULL; } - spin_unlock_irq(&pf_queue->lock); } /** @@ -448,18 +839,18 @@ static void xe_pagefault_queue_reset(struct xe_device *xe, struct xe_gt *gt, */ void xe_pagefault_reset(struct xe_device *xe, struct xe_gt *gt) { - int i; - - for (i = 0; i < XE_PAGEFAULT_QUEUE_COUNT; ++i) - xe_pagefault_queue_reset(xe, gt, xe->usm.pf_queue + i); + xe_pagefault_queue_reset(xe, gt, &xe->usm.pf_queue); } -static bool xe_pagefault_queue_full(struct xe_pagefault_queue *pf_queue) +/* + * This function can race with multiple page fault producers, but worst case we + * stick a page fault on the same queue for consumption. + */ +static int xe_pagefault_work_index(struct xe_device *xe) { - lockdep_assert_held(&pf_queue->lock); + lockdep_assert_held(&xe->usm.pf_queue.lock); - return CIRC_SPACE(pf_queue->head, pf_queue->tail, pf_queue->size) <= - xe_pagefault_entry_size(); + return xe->usm.current_pf_work++ % xe->info.num_pf_work; } /** @@ -474,24 +865,95 @@ static bool xe_pagefault_queue_full(struct xe_pagefault_queue *pf_queue) */ int xe_pagefault_handler(struct xe_device *xe, struct xe_pagefault *pf) { - struct xe_pagefault_queue *pf_queue = xe->usm.pf_queue + - (pf->consumer.asid % XE_PAGEFAULT_QUEUE_COUNT); - unsigned long flags; - bool full; - - spin_lock_irqsave(&pf_queue->lock, flags); - full = xe_pagefault_queue_full(pf_queue); - if (!full) { - memcpy(pf_queue->data + pf_queue->head, pf, sizeof(*pf)); - pf_queue->head = (pf_queue->head + xe_pagefault_entry_size()) % - pf_queue->size; - queue_work(xe->usm.pf_wq, &pf_queue->worker); + struct xe_pagefault_queue *pf_queue = &xe->usm.pf_queue; + struct xe_pagefault *lpf; + bool empty; + + guard(spinlock_irqsave)(&pf_queue->lock); + + empty = xe_pagefault_queue_empty(pf_queue); + lpf = xe_pagefault_queue_add(pf_queue, pf); + if (!lpf) + return -ENOSPC; + + lpf->consumer.next = NULL; + if (xe_pagefault_try_chain(pf_queue, lpf)) { + xe_gt_stats_incr(pf->gt, XE_GT_STATS_ID_CHAIN_IRQ_PAGEFAULT_COUNT, 1); + if (empty) { + xe_gt_stats_incr(pf->gt, + XE_GT_STATS_ID_CHAIN_DRAIN_IRQ_PAGEFAULT_COUNT, 1); + xe_pagefault_queue_advance(pf_queue); + } } else { - drm_warn(&xe->drm, - "PageFault Queue (%d) full, shouldn't be possible\n", - pf->consumer.asid % XE_PAGEFAULT_QUEUE_COUNT); + int work_index = xe_pagefault_work_index(xe); + + queue_work(xe->usm.pagefault_wq, + &xe->usm.pf_workers[work_index].work); + } + + return 0; +} + +/** + * xe_pagefault_print_info() - dump page fault queue/cache debug information + * @xe: Xe device + * @p: DRM printer to emit output to + * + * Print a snapshot of the page fault queue state for debugging. The output + * includes queue parameters (entry size, total size, head/tail), a histogram + * of per-entry allocation state values, and the validity of each per-worker + * page fault cache. + * + * This function is intended for debugfs and similar diagnostics. It acquires + * the page fault queue spinlock internally to serialize against IRQ-side + * producers and the worker consumer path, so callers must not hold the queue + * lock. + */ +void xe_pagefault_print_info(struct xe_device *xe, struct drm_printer *p) +{ + struct xe_pagefault_queue *pf_queue = &xe->usm.pf_queue; + struct xe_pagefault_work *pf_work; + static const char * const alloc_state_names[] = { + [XE_PAGEFAULT_ALLOC_STATE_FREE] = "free", + [XE_PAGEFAULT_ALLOC_STATE_QUEUED] = "queued", + [XE_PAGEFAULT_ALLOC_STATE_CHAINED] = "chained", + [XE_PAGEFAULT_ALLOC_STATE_ACTIVE] = "active", + }; + u32 i, counts[XE_PAGEFAULT_ALLOC_STATE_COUNT] = {}; + + /* Driver load failure guard / USM not enabled guard */ + if (!pf_queue->data) + return; + + guard(spinlock_irq)(&pf_queue->lock); + + drm_printf(p, "pagefault size: %u\n", xe_pagefault_entry_size()); + drm_printf(p, "pagefault queue size: %u\n", pf_queue->size); + drm_printf(p, "pagefault queue head: %u\n", pf_queue->head); + drm_printf(p, "pagefault queue tail: %u\n", pf_queue->tail); + + for (i = 0; i < pf_queue->size; i += xe_pagefault_entry_size()) { + struct xe_pagefault *pf = pf_queue->data + i; + + if (pf->consumer.alloc_state >= + XE_PAGEFAULT_ALLOC_STATE_COUNT) { + drm_printf(p, "pagefault[%u] corrupted alloc_state=%u\n", + i, pf->consumer.alloc_state); + continue; + } + + counts[pf->consumer.alloc_state]++; } - spin_unlock_irqrestore(&pf_queue->lock, flags); - return full ? -ENOSPC : 0; + for (i = 0; i < XE_PAGEFAULT_ALLOC_STATE_COUNT; ++i) + drm_printf(p, "pagefault queue %s count: %u\n", + alloc_state_names[i], counts[i]); + + for (i = 0, pf_work = xe->usm.pf_workers; + i < xe->info.num_pf_work; ++i, ++pf_work) { + if (pf_work->cache.start == XE_PAGEFAULT_CACHE_START_INVALID) + drm_printf(p, "pagefault work[%u] cache invalid\n", i); + else + drm_printf(p, "pagefault work[%u] cache valid\n", i); + } } diff --git a/drivers/gpu/drm/xe/xe_pagefault.h b/drivers/gpu/drm/xe/xe_pagefault.h index bd0cdf9ed37f..e9c5d1f03760 100644 --- a/drivers/gpu/drm/xe/xe_pagefault.h +++ b/drivers/gpu/drm/xe/xe_pagefault.h @@ -6,6 +6,9 @@ #ifndef _XE_PAGEFAULT_H_ #define _XE_PAGEFAULT_H_ +#include "xe_pagefault_types.h" + +struct drm_printer; struct xe_device; struct xe_gt; struct xe_pagefault; @@ -16,4 +19,75 @@ void xe_pagefault_reset(struct xe_device *xe, struct xe_gt *gt); int xe_pagefault_handler(struct xe_device *xe, struct xe_pagefault *pf); +void xe_pagefault_print_info(struct xe_device *xe, struct drm_printer *p); + +#define XE_PAGEFAULT_END_ADDR_MASK (~0xfffull) + +/** + * xe_pagefault_set_end_addr() - store serviced range end for a pagefault + * @pf: Pagefault entry + * @end_addr: Inclusive end address of the serviced fault range + * + * The pagefault consumer stores the resolved fault range so subsequent faults + * hitting the same range can be immediately acknowledged without re-running + * the full fault handling path. + * + * The end address shares storage with other consumer metadata and therefore + * must be masked with %XE_PAGEFAULT_END_ADDR_MASK before storing. Bits outside + * the mask are reserved for internal state tracking and must be preserved. + */ +static inline void +xe_pagefault_set_end_addr(struct xe_pagefault *pf, u64 end_addr) +{ + pf->consumer.end_addr &= ~XE_PAGEFAULT_END_ADDR_MASK; + pf->consumer.end_addr |= end_addr; +} + +/** + * xe_pagefault_end_addr() - read serviced range end for a pagefault + * @pf: Pagefault entry + * + * Returns the inclusive end address of the range previously recorded by + * xe_pagefault_set_end_addr(). Only the bits covered by + * %XE_PAGEFAULT_END_ADDR_MASK are returned; other bits in the storage are + * reserved for internal state. + * + * Return: End address of the serviced fault range. + */ +static inline u64 xe_pagefault_end_addr(struct xe_pagefault *pf) +{ + return pf->consumer.end_addr & XE_PAGEFAULT_END_ADDR_MASK; +} + +#undef XE_PAGEFAULT_END_ADDR_MASK + +/** + * xe_pagefault_set_start_addr() - store serviced range start for a pagefault + * @pf: Pagefault entry + * @start_addr: Start address of the serviced fault range + * + * The pagefault consumer stores the resolved fault range so subsequent faults + * hitting the same range can be immediately acknowledged without re-running + * the full fault handling path. + */ +static inline void +xe_pagefault_set_start_addr(struct xe_pagefault *pf, u64 start_addr) +{ + pf->consumer.page_addr = start_addr; +} + +/** + * xe_pagefault_start_addr() - read serviced range start for a pagefault + * @pf: Pagefault entry + * + * Returns the inclusive start address of the range previously recorded by + * xe_pagefault_set_start_addr(). + * + * Return: Start address of the serviced fault range. + */ +static inline u64 xe_pagefault_start_addr(struct xe_pagefault *pf) +{ + return pf->consumer.page_addr; +} + #endif diff --git a/drivers/gpu/drm/xe/xe_pagefault_types.h b/drivers/gpu/drm/xe/xe_pagefault_types.h index c4ee625b93dd..efeba5c3a58b 100644 --- a/drivers/gpu/drm/xe/xe_pagefault_types.h +++ b/drivers/gpu/drm/xe/xe_pagefault_types.h @@ -34,6 +34,13 @@ enum xe_pagefault_type { /** struct xe_pagefault_ops - Xe pagefault ops (producer) */ struct xe_pagefault_ops { /** + * @ack_fault_begin: Ack fault begin + * @private: producer private data + * + * Page fault producer begins acknowledgment from the consumer. + */ + void (*ack_fault_begin)(void *private); + /** * @ack_fault: Ack fault * @pf: Page fault * @err: Error state of fault @@ -42,6 +49,13 @@ struct xe_pagefault_ops { * sends the result to the HW/FW interface. */ void (*ack_fault)(struct xe_pagefault *pf, int err); + /** + * @ack_fault_end: Ack fault end + * @private: producer private data + * + * Page fault producer ends acknowledgment from the consumer. + */ + void (*ack_fault_end)(void *private); }; /** @@ -60,34 +74,58 @@ struct xe_pagefault { /** * @consumer: State for the software handling the fault. Populated by * the producer and may be modified by the consumer to communicate - * information back to the producer upon fault acknowledgment. + * information back to the producer upon fault acknowledgment. After + * fault acknowledgment, the producer should only access consumer fields + * via well defined helpers. */ struct { - /** @consumer.page_addr: address of page fault */ - u64 page_addr; - /** @consumer.asid: address space ID */ - u32 asid; /** - * @consumer.access_type: access type and prefetch flag packed - * into a u8. + * @consumer.page_addr: address of page fault, populated by + * consumer after fault completion */ - u8 access_type; + u64 page_addr; + union { + struct { + /** + * @consumer.alloc_state: page fault allocation + * state + */ + u8 alloc_state; + /** + * @consumer.access_type: access type, u8 rather + * than enum to keep size compact + */ + u8 access_type; #define XE_PAGEFAULT_ACCESS_TYPE_MASK GENMASK(1, 0) #define XE_PAGEFAULT_ACCESS_PREFETCH BIT(7) + /** + * @consumer.fault_type_level: fault type and + * level, u8 rather than enum to keep size + * compact + */ + u8 fault_type_level; +#define XE_PAGEFAULT_TYPE_LEVEL_NACK 0xff /* Producer indicates nack fault */ +#define XE_PAGEFAULT_LEVEL_MASK GENMASK(2, 0) +#define XE_PAGEFAULT_TYPE_MASK GENMASK(6, 3) +#define XE_PAGEFAULT_REQUEUE_MASK BIT(7) + /** @consumer.engine_class_instance: engine class and instance */ + u8 engine_class_instance; +#define XE_PAGEFAULT_ENGINE_CLASS_MASK GENMASK(3, 0) +#define XE_PAGEFAULT_ENGINE_INSTANCE_MASK GENMASK(7, 4) + /** @consumer.asid: address space ID */ + u32 asid; + }; + /** + * @consumer.end_addr: end address of page fault, + * populated by consumer after fault completion + */ + u64 end_addr; + }; /** - * @consumer.fault_type_level: fault type and level, u8 rather - * than enum to keep size compact + * @consumer.next: next pagefault chained to this fault, + * protected by pf_queue lock */ - u8 fault_type_level; -#define XE_PAGEFAULT_TYPE_LEVEL_NACK 0xff /* Producer indicates nack fault */ -#define XE_PAGEFAULT_LEVEL_MASK GENMASK(3, 0) -#define XE_PAGEFAULT_TYPE_MASK GENMASK(7, 4) - /** @consumer.engine_class: engine class */ - u8 engine_class; - /** @consumer.engine_instance: engine instance */ - u8 engine_instance; - /** @consumer.reserved: reserved bits for future expansion */ - u64 reserved; + struct xe_pagefault *next; } consumer; /** * @producer: State for the producer (i.e., HW/FW interface). Populated @@ -129,10 +167,38 @@ struct xe_pagefault_queue { u32 head; /** @tail: Tail pointer in bytes, moved by consumer, protected by @lock */ u32 tail; - /** @lock: protects page fault queue */ + /** @lock: protects page fault queue, workers caches */ spinlock_t lock; - /** @worker: to process page faults */ - struct work_struct worker; +}; + +/** + * struct xe_pagefault_work - Xe page fault work item (consumer) + * + * Represents a worker that pops a &struct xe_pagefault from the page fault + * queue and processes it. + */ +struct xe_pagefault_work { + /** @xe: Back-pointer to the Xe device */ + struct xe_device *xe; + /** @id: Identifier for this work item */ + int id; + /** + * @cache: Page fault cache for the currently processed fault + * + * Protected by the page fault queue lock. + */ + struct { + /** @cache.start: Start address of the current page fault */ + u64 start; + /** @cache.end: End address of the current page fault */ + u64 end; + /** @cache.asid: Address space ID of the current page fault */ + u32 asid; + /** @cache.pf: Pointer to the current page fault */ + struct xe_pagefault *pf; + } cache; + /** @work: Work item used to process the page fault */ + struct work_struct work; }; #endif diff --git a/drivers/gpu/drm/xe/xe_pci.c b/drivers/gpu/drm/xe/xe_pci.c index 5f2a0b19839d..1e04e8ef2611 100644 --- a/drivers/gpu/drm/xe/xe_pci.c +++ b/drivers/gpu/drm/xe/xe_pci.c @@ -25,6 +25,7 @@ #include "xe_gt_printk.h" #include "xe_gt_sriov_vf.h" #include "xe_guc.h" +#include "xe_log.h" #include "xe_mmio.h" #include "xe_module.h" #include "xe_pci_error.h" @@ -1145,17 +1146,12 @@ static void xe_pci_remove(struct pci_dev *pdev) * caller. Therefore there is no consequence on those specific callers when * function error injection skips the whole function. */ +static int __xe_pci_probe(struct pci_dev *pdev, const struct xe_device_desc *desc); static int xe_pci_probe(struct pci_dev *pdev, const struct pci_device_id *ent) { - struct xe_probed_info probed_info = {}; const struct xe_device_desc *desc = (const void *)ent->driver_data; - const struct xe_subplatform_desc *subplatform_desc; - struct xe_device *xe; - void *group; int err; - subplatform_desc = find_subplatform(desc, pdev->device); - xe_configfs_check_device(pdev); if (desc->require_force_probe && !id_forced(pdev->device)) { @@ -1171,14 +1167,34 @@ static int xe_pci_probe(struct pci_dev *pdev, const struct pci_device_id *ent) } if (id_blocked(pdev->device)) { - dev_info(&pdev->dev, "Probe blocked for device [%04x:%04x].\n", - pdev->vendor, pdev->device); + xe_log_info(pdev, PROBE, "driver loading blocked for device '%04x'\n", + pdev->device); return -ENODEV; } if (xe_display_driver_probe_defer(pdev)) return -EPROBE_DEFER; + err = __xe_pci_probe(pdev, desc); + if (err) { + xe_log_err_fatal(pdev, PROBE, err, "driver loading failed for device '%04x'\n", + pdev->device); + return err; + } + + return 0; +} + +static int __xe_pci_probe(struct pci_dev *pdev, const struct xe_device_desc *desc) +{ + const struct xe_subplatform_desc *subplatform_desc; + struct xe_probed_info probed_info = {}; + struct xe_device *xe; + void *group; + int err; + + subplatform_desc = find_subplatform(desc, pdev->device); + /* Group all devres so xe_pci_error_slot_reset() can release them as a unit. */ group = devres_open_group(&pdev->dev, NULL, GFP_KERNEL); if (!group) diff --git a/drivers/gpu/drm/xe/xe_pci_error.c b/drivers/gpu/drm/xe/xe_pci_error.c index e41af2ac7f23..79ce0c671549 100644 --- a/drivers/gpu/drm/xe/xe_pci_error.c +++ b/drivers/gpu/drm/xe/xe_pci_error.c @@ -7,6 +7,7 @@ #include "xe_device.h" #include "xe_gt.h" +#include "xe_log.h" #include "xe_pci.h" #include "xe_pm.h" #include "xe_printk.h" @@ -83,6 +84,12 @@ static pci_ers_result_t xe_pci_error_mmio_enabled(struct pci_dev *pdev) xe_info(xe, "PCI error: MMIO enabled\n"); action = xe_ras_process_errors(xe); + /* User wants to debug the error, prevent reset */ + if (xe->wedged.mode == XE_WEDGED_MODE_UPON_ANY_HANG_NO_RESET) { + xe_device_declare_wedged(xe); + return PCI_ERS_RESULT_DISCONNECT; + } + return ras_action_to_pci_result(pdev, action); } @@ -90,13 +97,15 @@ static pci_ers_result_t xe_pci_error_slot_reset(struct pci_dev *pdev) { const struct pci_device_id *ent = pci_match_id(pdev->driver->id_table, pdev); struct xe_device *xe = pdev_to_xe_device(pdev); + int err; xe_info(xe, "PCI error: slot reset\n"); pci_restore_state(pdev); - if (pci_enable_device(pdev)) { - xe_err(xe, "Cannot re-enable PCI device after reset\n"); + err = pci_enable_device(pdev); + if (err) { + xe_log_err_fatal(xe, PCI, err, "Cannot re-enable PCI device after reset\n"); return PCI_ERS_RESULT_DISCONNECT; } diff --git a/drivers/gpu/drm/xe/xe_pcode.c b/drivers/gpu/drm/xe/xe_pcode.c index ccc3bdeed6bb..e1b8062541a9 100644 --- a/drivers/gpu/drm/xe/xe_pcode.c +++ b/drivers/gpu/drm/xe/xe_pcode.c @@ -14,6 +14,7 @@ #include "regs/xe_pmt.h" #include "xe_assert.h" #include "xe_device.h" +#include "xe_log.h" #include "xe_mmio.h" #include "xe_pcode_api.h" #include "xe_pm.h" @@ -61,9 +62,7 @@ static int pcode_mailbox_status(struct xe_tile *tile) } if (err) { - drm_err(&tile_to_xe(tile)->drm, "PCODE Mailbox failed: %d %s", - err_decode, err_str); - + xe_log_err(tile, PCODE, err_decode, "Mailbox failed: %s\n", err_str); return err_decode; } @@ -219,8 +218,7 @@ int xe_pcode_request(struct xe_tile *tile, u32 mbox, u32 request, * requests, and for any quirks of the PCODE firmware that delays * the request completion. */ - drm_err(&tile_to_xe(tile)->drm, - "PCODE timeout, retrying with preemption disabled\n"); + xe_log_err(tile, PCODE, ret, "timeout, retrying with preemption disabled\n"); preempt_disable(); ret = pcode_try_request(tile, mbox, request, reply_mask, reply, &status, true, 50 * 1000, true); @@ -299,7 +297,7 @@ int xe_pcode_ready(struct xe_device *xe, bool locked) { u32 status, request = DGFX_GET_INIT_STATUS; struct xe_tile *tile = xe_device_get_root_tile(xe); - int timeout_us = 180000000; /* 3 min */ + long timeout_us = 3 * 60 * USEC_PER_SEC; /* 3 min */ int ret; if (xe->info.skip_pcode) @@ -320,8 +318,8 @@ int xe_pcode_ready(struct xe_device *xe, bool locked) mutex_unlock(&tile->pcode.lock); if (ret) - drm_err(&xe->drm, - "PCODE initialization timedout after: 3 min\n"); + xe_log_err(tile, PCODE, ret, "initialization timedout after %ld seconds\n", + timeout_us / USEC_PER_SEC); return ret; } diff --git a/drivers/gpu/drm/xe/xe_pm.c b/drivers/gpu/drm/xe/xe_pm.c index a5289a9df8d2..f517bf453b54 100644 --- a/drivers/gpu/drm/xe/xe_pm.c +++ b/drivers/gpu/drm/xe/xe_pm.c @@ -905,6 +905,11 @@ static bool xe_pm_suspending_or_resuming(struct xe_device *xe) * break scope-based handling, or when the lifetime of the runtime PM reference * does not match a specific scope (e.g., runtime PM obtained in one function * and released in a different one). + * + * This helper assumes the caller already holds a runtime PM reference and + * only warns when it cannot see one. After hot-unplug runtime PM is disabled + * and the check fails even when a reference is held, so callers that may run + * after unplug must guard it with drm_dev_enter()/drm_dev_exit() instead. */ void xe_pm_runtime_get_noresume(struct xe_device *xe) { diff --git a/drivers/gpu/drm/xe/xe_printk.h b/drivers/gpu/drm/xe/xe_printk.h index c5be2385aa95..87f51d7edfa8 100644 --- a/drivers/gpu/drm/xe/xe_printk.h +++ b/drivers/gpu/drm/xe/xe_printk.h @@ -36,6 +36,9 @@ #define xe_dbg(_xe, _fmt, ...) \ xe_printk((_xe), dbg, _fmt, ##__VA_ARGS__) +#define xe_dbg_ratelimited(_xe, _fmt, ...) \ + xe_printk((_xe), dbg_ratelimited, _fmt, ##__VA_ARGS__) + #define xe_WARN_type(_xe, _type, _condition, _fmt, ...) \ drm_WARN##_type(&(_xe)->drm, _condition, _fmt, ## __VA_ARGS__) diff --git a/drivers/gpu/drm/xe/xe_pt.c b/drivers/gpu/drm/xe/xe_pt.c index a07316a45d79..5d990c1c3740 100644 --- a/drivers/gpu/drm/xe/xe_pt.c +++ b/drivers/gpu/drm/xe/xe_pt.c @@ -2414,6 +2414,12 @@ static int op_prepare(struct xe_vm *vm, xa_for_each(&op->prefetch_range.range, i, range) { err = bind_range_prepare(vm, tile, pt_update_ops, vma, range); + /* + * Don't tell user space to retry, rather let + * page faults fixup the pages. + */ + if (err == -EAGAIN) + err = -ENODATA; if (err) return err; } diff --git a/drivers/gpu/drm/xe/xe_pxp_submit.c b/drivers/gpu/drm/xe/xe_pxp_submit.c index e60526e30030..5de86a8cf27d 100644 --- a/drivers/gpu/drm/xe/xe_pxp_submit.c +++ b/drivers/gpu/drm/xe/xe_pxp_submit.c @@ -46,7 +46,7 @@ static int allocate_vcs_execution_resources(struct xe_pxp *pxp) return -ENODEV; q = xe_exec_queue_create(xe, NULL, BIT(hwe->logical_instance), 1, hwe, - EXEC_QUEUE_FLAG_KERNEL | EXEC_QUEUE_FLAG_PERMANENT, 0); + EXEC_QUEUE_FLAG_KERNEL, 0); if (IS_ERR(q)) return PTR_ERR(q); @@ -144,8 +144,7 @@ static int allocate_gsc_client_resources(struct xe_gt *gt, } q = xe_exec_queue_create(xe, vm, BIT(hwe->logical_instance), 1, hwe, - EXEC_QUEUE_FLAG_KERNEL | - EXEC_QUEUE_FLAG_PERMANENT, 0); + EXEC_QUEUE_FLAG_KERNEL, 0); if (IS_ERR(q)) { err = PTR_ERR(q); goto bo_out; diff --git a/drivers/gpu/drm/xe/xe_ras.c b/drivers/gpu/drm/xe/xe_ras.c index d98ff9453f60..d25d25f77531 100644 --- a/drivers/gpu/drm/xe/xe_ras.c +++ b/drivers/gpu/drm/xe/xe_ras.c @@ -3,8 +3,10 @@ * Copyright © 2026 Intel Corporation */ +#include "xe_debugfs.h" #include "xe_device.h" #include "xe_drm_ras.h" +#include "xe_log.h" #include "xe_pm.h" #include "xe_printk.h" #include "xe_ras.h" @@ -45,6 +47,16 @@ enum xe_ras_component { XE_RAS_COMP_MAX }; +#define CHECK_COMPONENT(RAS_COMP, LOG_COMP) \ + static_assert(MAKE_XE_LOG_COMPONENT(HARDWARE, (RAS_COMP)) == (LOG_COMP)) + /* make sure components definitions maintain stable relation */ + CHECK_COMPONENT(XE_RAS_COMP_DEVICE_MEMORY, XE_LOG_COMPONENT_DEVICE_MEMORY); + CHECK_COMPONENT(XE_RAS_COMP_CORE_COMPUTE, XE_LOG_COMPONENT_CORE_COMPUTE); + CHECK_COMPONENT(XE_RAS_COMP_PCIE, XE_LOG_COMPONENT_PCIE); + CHECK_COMPONENT(XE_RAS_COMP_FABRIC, XE_LOG_COMPONENT_FABRIC); + CHECK_COMPONENT(XE_RAS_COMP_SOC_INTERNAL, XE_LOG_COMPONENT_SOC_INTERNAL); +#undef CHECK_COMPONENT + /* RAS response status codes */ enum xe_ras_response_status { XE_RAS_STATUS_SUCCESS = 0, @@ -90,6 +102,8 @@ static const char * const gpu_health_states[] = { }; static_assert(ARRAY_SIZE(gpu_health_states) == XE_RAS_HEALTH_MAX); +static int get_counter(struct xe_device *xe, struct xe_ras_error_class *counter, u32 *value); + static u8 drm_to_xe_ras_severity(u8 severity) { switch (severity) { @@ -102,6 +116,18 @@ static u8 drm_to_xe_ras_severity(u8 severity) } } +static u8 xe_to_drm_ras_severity(u8 severity) +{ + switch (severity) { + case XE_RAS_SEV_CORRECTABLE: + return DRM_XE_RAS_ERR_SEV_CORRECTABLE; + case XE_RAS_SEV_UNCORRECTABLE: + return DRM_XE_RAS_ERR_SEV_UNCORRECTABLE; + default: + return DRM_XE_RAS_ERR_SEV_MAX; + } +} + static u8 drm_to_xe_ras_component(u8 component) { switch (component) { @@ -120,6 +146,24 @@ static u8 drm_to_xe_ras_component(u8 component) } } +static u8 xe_to_drm_ras_component(u8 component) +{ + switch (component) { + case XE_RAS_COMP_DEVICE_MEMORY: + return DRM_XE_RAS_ERR_COMP_DEVICE_MEMORY; + case XE_RAS_COMP_CORE_COMPUTE: + return DRM_XE_RAS_ERR_COMP_CORE_COMPUTE; + case XE_RAS_COMP_PCIE: + return DRM_XE_RAS_ERR_COMP_PCIE; + case XE_RAS_COMP_FABRIC: + return DRM_XE_RAS_ERR_COMP_FABRIC; + case XE_RAS_COMP_SOC_INTERNAL: + return DRM_XE_RAS_ERR_COMP_SOC_INTERNAL; + default: + return DRM_XE_RAS_ERR_COMP_MAX; + } +} + static int ras_status_to_errno(u32 status) { switch (status) { @@ -156,6 +200,24 @@ static inline const char *comp_to_str(u8 component) return xe_ras_components[component]; } +static bool ras_counter_is_valid(struct xe_device *xe, struct xe_ras_error_class *counter) +{ + u8 severity = counter->common.severity; + u8 component = counter->common.component; + + if (!in_range(severity, XE_RAS_SEV_NOT_SUPPORTED + 1, XE_RAS_SEV_MAX - 1)) { + xe_err(xe, "sysctrl: unexpected severity %u\n", severity); + return false; + } + + if (!in_range(component, XE_RAS_COMP_NOT_SUPPORTED + 1, XE_RAS_COMP_MAX - 1)) { + xe_err(xe, "sysctrl: unexpected component %u\n", component); + return false; + } + + return true; +} + static struct pci_dev *find_usp_dev(struct pci_dev *pdev) { struct pci_dev *vsp; @@ -218,6 +280,26 @@ static void ras_usp_aer_init(struct xe_device *xe) dev_dbg(&usp->dev, "Uncorrectable Internal Errors downgraded and unmasked\n"); } +static void ras_send_error_event(struct xe_device *xe, u8 severity, u8 component) +{ + struct xe_ras_error_class counter = {0}; + u8 drm_severity, drm_component; + u32 value; + int ret; + + counter.common.severity = severity; + counter.common.component = component; + + ret = get_counter(xe, &counter, &value); + if (ret) + return; + + drm_severity = xe_to_drm_ras_severity(severity); + drm_component = xe_to_drm_ras_component(component); + + xe_drm_ras_event(xe, drm_component, drm_severity, value); +} + static u8 handle_core_compute_errors(struct xe_ras_error_array *arr) { struct xe_ras_compute_error *error_info = (void *)arr->details; @@ -236,6 +318,12 @@ static u8 handle_core_compute_errors(struct xe_ras_error_array *arr) return XE_RAS_RECOVERY_ACTION_RECOVERED; } +static void punit_error_handler(struct xe_device *xe) +{ + xe_device_set_wedged_method(xe, DRM_WEDGE_RECOVERY_COLD_RESET); + xe_device_declare_wedged(xe); +} + static u8 handle_soc_internal_errors(struct xe_device *xe, struct xe_ras_error_array *arr) { struct xe_ras_soc_error *info = (void *)arr->details; @@ -267,7 +355,7 @@ static u8 handle_soc_internal_errors(struct xe_device *xe, struct xe_ras_error_a xe_err(xe, "[RAS]: PUNIT %s detected: 0x%x\n", sev_to_str(counter->common.severity), ieh_error->global_error_status); - /* TODO: Add PUNIT error handling */ + punit_error_handler(xe); return XE_RAS_RECOVERY_ACTION_DISCONNECT; } } @@ -312,8 +400,10 @@ void xe_ras_counter_threshold_crossed(struct xe_device *xe, struct xe_ras_threshold_crossed *pending = (void *)&response->data; struct xe_ras_error_class *errors = pending->counters; u32 id, ncounters = pending->ncounters; + u8 sent = 0; BUILD_BUG_ON(sizeof(response->data) < sizeof(*pending)); + BUILD_BUG_ON(BITS_PER_TYPE(sent) < XE_RAS_COMP_MAX); xe_device_assert_mem_access(xe); if (!ncounters || ncounters > XE_RAS_NUM_COUNTERS) @@ -327,8 +417,18 @@ void xe_ras_counter_threshold_crossed(struct xe_device *xe, severity = errors[id].common.severity; component = errors[id].common.component; + if (!ras_counter_is_valid(xe, &errors[id])) + continue; + xe_warn(xe, "[RAS]: %s %s detected\n", comp_to_str(component), sev_to_str(severity)); + + /* Send event once per component */ + if (sent & BIT(component)) + continue; + sent |= BIT(component); + + ras_send_error_event(xe, severity, component); } } @@ -358,6 +458,9 @@ static int get_counter(struct xe_device *xe, struct xe_ras_error_class *counter, return -EIO; } + if (!ras_counter_is_valid(xe, &response.counter)) + return -EBADMSG; + common = &response.counter.common; *value = response.value; @@ -382,12 +485,20 @@ enum xe_ras_recovery_action xe_ras_process_errors(struct xe_device *xe) enum xe_ras_recovery_action final_action; u32 remaining = XE_SYSCTRL_FLOOD_LIMIT; struct xe_ras_get_soc_error response; + u8 sent = 0; size_t rlen; int ret; + if (xe_fault_wedge_cold_reset()) { + xe_err(xe, "[RAS]: cold-reset wedge injected\n"); + punit_error_handler(xe); + return XE_RAS_RECOVERY_ACTION_DISCONNECT; + } + if (!xe->info.has_sysctrl) return XE_RAS_RECOVERY_ACTION_RESET; + BUILD_BUG_ON(BITS_PER_TYPE(sent) < XE_RAS_COMP_MAX); /* Default action */ final_action = XE_RAS_RECOVERY_ACTION_RECOVERED; @@ -422,9 +533,18 @@ enum xe_ras_recovery_action xe_ras_process_errors(struct xe_device *xe) component = arr->counter.common.component; severity = arr->counter.common.severity; + if (!ras_counter_is_valid(xe, &arr->counter)) + continue; + xe_info(xe, "[RAS]: %s %s detected\n", comp_to_str(component), sev_to_str(severity)); + /* Send event once per component */ + if (!(sent & BIT(component))) { + sent |= BIT(component); + ras_send_error_event(xe, severity, component); + } + switch (component) { case XE_RAS_COMP_CORE_COMPUTE: action = handle_core_compute_errors(arr); @@ -532,6 +652,9 @@ int xe_ras_clear_counter(struct xe_device *xe, u8 severity, u8 component) counter = &response.counter; + if (!ras_counter_is_valid(xe, counter)) + return -EBADMSG; + xe_dbg(xe, "[RAS]: clear counter for %s %s\n", comp_to_str(counter->common.component), sev_to_str(counter->common.severity)); diff --git a/drivers/gpu/drm/xe/xe_sriov_printk.h b/drivers/gpu/drm/xe/xe_sriov_printk.h index 4c6b5c3d2190..5931c918655f 100644 --- a/drivers/gpu/drm/xe/xe_sriov_printk.h +++ b/drivers/gpu/drm/xe/xe_sriov_printk.h @@ -36,6 +36,9 @@ #define xe_sriov_dbg(xe, fmt, ...) \ xe_sriov_printk((xe), dbg, fmt, ##__VA_ARGS__) +#define xe_sriov_dbg_ratelimited(xe, fmt, ...) \ + xe_sriov_printk((xe), dbg_ratelimited, fmt, ##__VA_ARGS__) + /* for low level noisy debug messages */ #ifdef CONFIG_DRM_XE_DEBUG_SRIOV #define xe_sriov_dbg_verbose(xe, fmt, ...) xe_sriov_dbg(xe, fmt, ##__VA_ARGS__) diff --git a/drivers/gpu/drm/xe/xe_sriov_vf_ccs.c b/drivers/gpu/drm/xe/xe_sriov_vf_ccs.c index a8c831fbee3b..a54138461f44 100644 --- a/drivers/gpu/drm/xe/xe_sriov_vf_ccs.c +++ b/drivers/gpu/drm/xe/xe_sriov_vf_ccs.c @@ -354,7 +354,6 @@ int xe_sriov_vf_ccs_init(struct xe_device *xe) ctx->ctx_id = ctx_id; flags = EXEC_QUEUE_FLAG_KERNEL | - EXEC_QUEUE_FLAG_PERMANENT | EXEC_QUEUE_FLAG_MIGRATE; q = xe_exec_queue_create_bind(xe, tile, NULL, flags, 0); if (IS_ERR(q)) { diff --git a/drivers/gpu/drm/xe/xe_survivability_mode.c b/drivers/gpu/drm/xe/xe_survivability_mode.c index 4c506027fa94..45d44ebe288b 100644 --- a/drivers/gpu/drm/xe/xe_survivability_mode.c +++ b/drivers/gpu/drm/xe/xe_survivability_mode.c @@ -14,9 +14,11 @@ #include "xe_device.h" #include "xe_heci_gsc.h" #include "xe_i2c.h" +#include "xe_log.h" #include "xe_mmio.h" #include "xe_nvm.h" #include "xe_pcode_api.h" +#include "xe_printk.h" #include "xe_vsec.h" /** @@ -172,18 +174,32 @@ static void populate_survivability_info(struct xe_device *xe) } } -static void log_survivability_info(struct pci_dev *pdev) +static const char *boot_status_str(u8 boot_status) +{ + switch (boot_status) { + case CRITICAL_FAILURE: + return "Critical Failure"; + case NON_CRITICAL_FAILURE: + return "Non Critical Failure"; + default: + return "Other"; + } +} + +static void log_survivability_info(struct xe_device *xe) { - struct xe_device *xe = pdev_to_xe_device(pdev); struct xe_survivability *survivability = &xe->survivability; u32 *info = survivability->info; int id; - dev_info(&pdev->dev, "Survivability Boot Status : Critical Failure (%d)\n", - survivability->boot_status); + xe_log_info(xe, SURVIVABILITY, "Boot Status: %#x (%s)\n", + survivability->boot_status, + boot_status_str(survivability->boot_status)); + for (id = 0; id < MAX_SCRATCH_REG; id++) { - if (info[id]) - dev_info(&pdev->dev, "%s: 0x%x\n", reg_map[id], info[id]); + if (!info[id]) + continue; + xe_log_info(xe, SURVIVABILITY, "%s: %#x\n", reg_map[id], info[id]); } } @@ -289,41 +305,46 @@ static const struct attribute_group survivability_info_group = { static int create_survivability_sysfs(struct pci_dev *pdev) { - struct device *dev = &pdev->dev; + /* Survivability info is required if not enabled via configfs */ + bool needs_info = !xe_configfs_get_survivability_mode(pdev); struct xe_device *xe = pdev_to_xe_device(pdev); + struct device *dev = &pdev->dev; int ret; ret = device_create_file(dev, &dev_attr_survivability_mode); - if (ret) { - dev_warn(dev, "Failed to create survivability sysfs files\n"); - return ret; - } + if (ret) + goto failed; ret = devm_add_action_or_reset(xe->drm.dev, xe_survivability_mode_fini, xe); if (ret) - return ret; + goto failed; - /* Survivability info is not required if enabled via configfs */ - if (!xe_configfs_get_survivability_mode(pdev)) { + if (needs_info) { ret = devm_device_add_group(dev, &survivability_info_group); if (ret) - return ret; + goto failed; } return 0; + +failed: + xe_err(xe, "Failed to create survivability sysfs files: %pe\n", ERR_PTR(ret)); + /* no sysfs, dump Survivability info to dmesg instead */ + if (needs_info) + log_survivability_info(xe); + return ret; } static int enable_boot_survivability_mode(struct pci_dev *pdev) { - struct device *dev = &pdev->dev; struct xe_device *xe = pdev_to_xe_device(pdev); struct xe_survivability *survivability = &xe->survivability; - int ret = 0; + int ret; ret = create_survivability_sysfs(pdev); if (ret) - return ret; + goto failed; /* Make sure xe_heci_gsc_init() and xe_i2c_probe() are aware of survivability */ survivability->mode = true; @@ -335,19 +356,22 @@ static int enable_boot_survivability_mode(struct pci_dev *pdev) if (survivability->fdo_mode) { ret = xe_nvm_init(xe); if (ret) - goto err; + goto failed; } ret = xe_i2c_probe(xe); if (ret) - goto err; + goto failed; - dev_err(dev, "In Survivability Mode\n"); + if (check_boot_failure(xe)) + xe_log_err_fatal(pdev, SURVIVABILITY, 0, "Boot Mode enabled!\n"); + else + xe_log_info(pdev, SURVIVABILITY, "Boot Mode enabled!\n"); return 0; -err: - dev_err(dev, "Failed to enable Survivability Mode\n"); +failed: + xe_log_err_fatal(pdev, SURVIVABILITY, ret, "Failed to enable Boot Mode!\n"); survivability->mode = false; return ret; } @@ -412,21 +436,22 @@ void xe_survivability_mode_runtime_enable(struct xe_device *xe) struct pci_dev *pdev = to_pci_dev(xe->drm.dev); if (!IS_DGFX(xe) || IS_SRIOV_VF(xe) || xe->info.platform < XE_BATTLEMAGE) { - dev_err(&pdev->dev, "Runtime Survivability Mode not supported\n"); + xe_log_err(xe, SURVIVABILITY, -EOPNOTSUPP, "Runtime Mode not supported!\n"); return; } populate_survivability_info(xe); - - if (create_survivability_sysfs(pdev)) - dev_err(&pdev->dev, "Failed to create survivability sysfs\n"); + create_survivability_sysfs(pdev); survivability->type = XE_SURVIVABILITY_TYPE_RUNTIME; - dev_err(&pdev->dev, "Runtime Survivability mode enabled\n"); + xe_log_err(xe, SURVIVABILITY, 0, "Runtime Mode enabled!\n"); xe_device_set_wedged_method(xe, DRM_WEDGE_RECOVERY_VENDOR); xe_device_declare_wedged(xe); - dev_err(&pdev->dev, "Firmware flash required, Please refer to the userspace documentation for more details!\n"); + + xe_log_err(xe, SURVIVABILITY, 0, "Firmware flash required!\n"); + xe_info(xe, "Please refer to the userspace documentation for more details how to flash the firmware on %s!\n", + xe->info.platform_name); } /** @@ -452,7 +477,7 @@ int xe_survivability_mode_boot_enable(struct xe_device *xe) * v2 supports survivability mode for critical errors */ if (survivability->version < 2 && survivability->boot_status == CRITICAL_FAILURE) { - log_survivability_info(pdev); + log_survivability_info(xe); return -ENXIO; } diff --git a/drivers/gpu/drm/xe/xe_svm.c b/drivers/gpu/drm/xe/xe_svm.c index b228a737cfd6..627a741293d5 100644 --- a/drivers/gpu/drm/xe/xe_svm.c +++ b/drivers/gpu/drm/xe/xe_svm.c @@ -15,6 +15,7 @@ #include "xe_gt_stats.h" #include "xe_migrate.h" #include "xe_module.h" +#include "xe_pagefault.h" #include "xe_pm.h" #include "xe_pt.h" #include "xe_svm.h" @@ -115,6 +116,7 @@ xe_svm_range_alloc(struct drm_gpusvm *gpusvm) return NULL; INIT_LIST_HEAD(&range->garbage_collector_link); + mutex_init(&range->lock); drm_gpusvm_init_pages(&range->pages, &gpusvm_to_vm(gpusvm)->xe->drm); xe_vm_get(gpusvm_to_vm(gpusvm)); @@ -125,6 +127,7 @@ static void xe_svm_range_free(struct drm_gpusvm_range *range) { drm_gpusvm_free_pages(range->gpusvm, &(to_xe_range(range)->pages), drm_gpusvm_range_size(range) >> PAGE_SHIFT); + mutex_destroy(&to_xe_range(range)->lock); xe_vm_put(range_to_vm(range)); kfree(to_xe_range(range)); } @@ -140,13 +143,13 @@ xe_svm_garbage_collector_add_range(struct xe_vm *vm, struct xe_svm_range *range, drm_gpusvm_range_set_unmapped(&range->base, &range->pages, 1, mmu_range); - spin_lock(&vm->svm.garbage_collector.lock); + spin_lock(&vm->svm.garbage_collector.list_lock); if (list_empty(&range->garbage_collector_link)) list_add_tail(&range->garbage_collector_link, &vm->svm.garbage_collector.range_list); - spin_unlock(&vm->svm.garbage_collector.lock); + spin_unlock(&vm->svm.garbage_collector.list_lock); - queue_work(xe->usm.pf_wq, &vm->svm.garbage_collector.work); + queue_work(xe->usm.pagefault_wq, &vm->svm.garbage_collector.work); } static void xe_svm_tlb_inval_count_stats_incr(struct xe_gt *gt) @@ -309,18 +312,30 @@ static int __xe_svm_garbage_collector(struct xe_vm *vm, range_debug(range, "GARBAGE COLLECTOR"); - xe_vm_lock(vm, false); - fence = xe_vm_range_unbind(vm, range); - xe_vm_unlock(vm); - if (IS_ERR(fence)) - return PTR_ERR(fence); - dma_fence_put(fence); + scoped_guard(mutex, &range->lock) { + drm_gpusvm_range_get(&range->base); + range->removed = true; + + range_debug(range, "GARBAGE COLLECTOR"); + + xe_vm_lock(vm, false); + fence = xe_vm_range_unbind(vm, range); + xe_vm_unlock(vm); + if (IS_ERR(fence)) { + drm_gpusvm_range_put(&range->base); + return PTR_ERR(fence); + } + dma_fence_put(fence); + + drm_gpusvm_unmap_pages(&vm->svm.gpusvm, &range->pages, + drm_gpusvm_range_size(&range->base) >> PAGE_SHIFT, + &ctx); - drm_gpusvm_unmap_pages(&vm->svm.gpusvm, &range->pages, - drm_gpusvm_range_size(&range->base) >> PAGE_SHIFT, - &ctx); + scoped_guard(mutex, &vm->svm.range_lock) + drm_gpusvm_range_remove(&vm->svm.gpusvm, &range->base); + } - drm_gpusvm_range_remove(&vm->svm.gpusvm, &range->base); + drm_gpusvm_range_put(&range->base); return 0; } @@ -393,13 +408,15 @@ static int xe_svm_garbage_collector(struct xe_vm *vm) u64 range_end; int err, ret = 0; - lockdep_assert_held_write(&vm->lock); + lockdep_assert_held(&vm->lock); if (xe_vm_is_closed_or_banned(vm)) return -ENOENT; + guard(mutex)(&vm->svm.garbage_collector.lock); + for (;;) { - spin_lock(&vm->svm.garbage_collector.lock); + spin_lock(&vm->svm.garbage_collector.list_lock); range = list_first_entry_or_null(&vm->svm.garbage_collector.range_list, typeof(*range), garbage_collector_link); @@ -410,7 +427,7 @@ static int xe_svm_garbage_collector(struct xe_vm *vm) range_end = xe_svm_range_end(range); list_del(&range->garbage_collector_link); - spin_unlock(&vm->svm.garbage_collector.lock); + spin_unlock(&vm->svm.garbage_collector.list_lock); err = __xe_svm_garbage_collector(vm, range); if (err) { @@ -429,7 +446,7 @@ static int xe_svm_garbage_collector(struct xe_vm *vm) return err; } } - spin_unlock(&vm->svm.garbage_collector.lock); + spin_unlock(&vm->svm.garbage_collector.list_lock); return ret; } @@ -439,9 +456,8 @@ static void xe_svm_garbage_collector_work_func(struct work_struct *w) struct xe_vm *vm = container_of(w, struct xe_vm, svm.garbage_collector.work); - down_write(&vm->lock); + guard(rwsem_read)(&vm->lock); xe_svm_garbage_collector(vm); - up_write(&vm->lock); } #if IS_ENABLED(CONFIG_DRM_XE_PAGEMAP) @@ -893,8 +909,11 @@ int xe_svm_init(struct xe_vm *vm) { int err; + mutex_init(&vm->svm.range_lock); + mutex_init(&vm->svm.garbage_collector.lock); + if (vm->flags & XE_VM_FLAG_FAULT_MODE) { - spin_lock_init(&vm->svm.garbage_collector.lock); + spin_lock_init(&vm->svm.garbage_collector.list_lock); INIT_LIST_HEAD(&vm->svm.garbage_collector.range_list); INIT_WORK(&vm->svm.garbage_collector.work, xe_svm_garbage_collector_work_func); @@ -903,12 +922,12 @@ int xe_svm_init(struct xe_vm *vm) err = drm_pagemap_acquire_owner(&vm->svm.peer, &xe_owner_list, xe_has_interconnect); if (err) - return err; + goto out_err; err = xe_svm_get_pagemaps(vm); if (err) { drm_pagemap_release_owner(&vm->svm.peer); - return err; + goto out_err; } err = drm_gpusvm_init(&vm->svm.gpusvm, "Xe SVM", @@ -916,19 +935,27 @@ int xe_svm_init(struct xe_vm *vm) xe_modparam.svm_notifier_size * SZ_1M, &gpusvm_ops, fault_chunk_sizes, ARRAY_SIZE(fault_chunk_sizes)); - drm_gpusvm_driver_set_lock(&vm->svm.gpusvm, &vm->lock); + drm_gpusvm_driver_set_lock(&vm->svm.gpusvm, &vm->svm.range_lock); if (err) { xe_svm_put_pagemaps(vm); drm_pagemap_release_owner(&vm->svm.peer); - return err; + goto out_err; } } else { err = drm_gpusvm_init(&vm->svm.gpusvm, "Xe SVM (simple)", NULL, 0, 0, 0, NULL, NULL, 0); + if (err) + goto out_err; } + return 0; + +out_err: + mutex_destroy(&vm->svm.range_lock); + mutex_destroy(&vm->svm.garbage_collector.lock); + return err; } @@ -969,7 +996,10 @@ void xe_svm_fini(struct xe_vm *vm) &ctx); } - drm_gpusvm_fini(&vm->svm.gpusvm); + scoped_guard(mutex, &vm->svm.range_lock) + drm_gpusvm_fini(&vm->svm.gpusvm); + mutex_destroy(&vm->svm.range_lock); + mutex_destroy(&vm->svm.garbage_collector.lock); } static bool xe_svm_range_has_pagemap_locked(const struct xe_svm_range *range, @@ -1022,6 +1052,7 @@ void xe_svm_range_migrate_to_smem(struct xe_vm *vm, struct xe_svm_range *range) * @tile_mask: Mask representing the tiles to be checked * @dpagemap: if !%NULL, the range is expected to be present * in device memory identified by this parameter. + * @valid_pages: Pages are valid, result written back to caller * * The xe_svm_range_validate() function checks if a range is * valid and located in the desired memory region. @@ -1030,7 +1061,8 @@ void xe_svm_range_migrate_to_smem(struct xe_vm *vm, struct xe_svm_range *range) */ bool xe_svm_range_validate(struct xe_vm *vm, struct xe_svm_range *range, - u8 tile_mask, const struct drm_pagemap *dpagemap) + u8 tile_mask, const struct drm_pagemap *dpagemap, + bool *valid_pages) { bool ret; @@ -1042,6 +1074,8 @@ bool xe_svm_range_validate(struct xe_vm *vm, else ret = ret && !range->pages.dpagemap; + *valid_pages = xe_svm_range_pages_valid(range); + xe_svm_notifier_unlock(vm); return ret; @@ -1231,8 +1265,8 @@ DECL_SVM_RANGE_US_STATS(bind, BIND) DECL_SVM_RANGE_US_STATS(fault, PAGEFAULT) static int __xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma, - struct xe_gt *gt, u64 fault_addr, - bool need_vram) + struct xe_pagefault *pf, struct xe_gt *gt, + u64 fault_addr, bool need_vram) { int devmem_possible = IS_DGFX(vm->xe) && IS_ENABLED(CONFIG_DRM_XE_PAGEMAP); @@ -1246,21 +1280,27 @@ static int __xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma, }; struct xe_validation_ctx vctx; struct drm_exec exec; - struct xe_svm_range *range; + struct xe_svm_range *range = NULL; struct drm_gpusvm_range_flags range_flags; struct dma_fence *fence; struct drm_pagemap *dpagemap; struct xe_tile *tile = gt_to_tile(gt); int migrate_try_count = ctx.devmem_only ? 3 : 1; ktime_t start = xe_gt_stats_ktime_get(), bind_start, get_pages_start; - int err; + int err = 0; - lockdep_assert_held_write(&vm->lock); + lockdep_assert_held(&vm->lock); xe_assert(vm->xe, xe_vma_is_cpu_addr_mirror(vma)); xe_gt_stats_incr(gt, XE_GT_STATS_ID_SVM_PAGEFAULT_COUNT, 1); retry: + /* Release old range */ + if (range) { + mutex_unlock(&range->lock); + drm_gpusvm_range_put(&range->base); + } + /* Always process UNMAPs first so view SVM ranges is current */ err = xe_svm_garbage_collector(vm); if (err) @@ -1276,10 +1316,17 @@ retry: xe_svm_range_fault_count_stats_incr(gt, range); + mutex_lock(&range->lock); + + if (xe_svm_range_is_removed(range)) + goto retry; + /* READ_ONCE pairs with WRITE_ONCE in drm_gpusvm_range_set_unmapped() */ range_flags.__flags = READ_ONCE(range->base.flags.__flags); - if (ctx.devmem_only && !range_flags.migrate_devmem) - return -EACCES; + if (ctx.devmem_only && !range_flags.migrate_devmem) { + err = -EACCES; + goto err_out; + } if (xe_svm_range_is_valid(range, tile, ctx.devmem_only, dpagemap)) { xe_svm_range_valid_fault_count_stats_incr(gt, range); @@ -1317,7 +1364,7 @@ retry: drm_err(&vm->xe->drm, "VRAM allocation failed, retry count exceeded, asid=%u, errno=%pe\n", vm->usm.asid, ERR_PTR(err)); - return err; + goto err_out; } } } @@ -1344,7 +1391,7 @@ get_pages: } if (err) { range_debug(range, "PAGE FAULT - FAIL PAGE COLLECT"); - goto out; + goto err_out; } else if (IS_ENABLED(CONFIG_DRM_XE_DEBUG_VM)) { drm_dbg(&vm->xe->drm, "After page collect data location is %sin \"%s\".\n", xe_svm_range_has_pagemap(range, dpagemap) ? "" : "NOT ", @@ -1378,7 +1425,13 @@ get_pages: xe_svm_range_bind_us_stats_incr(gt, range, bind_start); out: + /* Give hint to immediately ack faults */ + xe_pagefault_set_start_addr(pf, xe_svm_range_start(range)); + xe_pagefault_set_end_addr(pf, xe_svm_range_end(range)); + xe_svm_range_fault_us_stats_incr(gt, range, start); + mutex_unlock(&range->lock); + drm_gpusvm_range_put(&range->base); return 0; err_out: @@ -1388,6 +1441,9 @@ err_out: goto retry; } + mutex_unlock(&range->lock); + drm_gpusvm_range_put(&range->base); + return err; } @@ -1395,6 +1451,7 @@ err_out: * xe_svm_handle_pagefault() - SVM handle page fault * @vm: The VM. * @vma: The CPU address mirror VMA. + * @pf: Pagefault structure * @gt: The gt upon the fault occurred. * @fault_addr: The GPU fault address. * @atomic: The fault atomic access bit. @@ -1405,8 +1462,8 @@ err_out: * Return: 0 on success, negative error code on error. */ int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma, - struct xe_gt *gt, u64 fault_addr, - bool atomic) + struct xe_pagefault *pf, struct xe_gt *gt, + u64 fault_addr, bool atomic) { int need_vram, ret; retry: @@ -1414,7 +1471,7 @@ retry: if (need_vram < 0) return need_vram; - ret = __xe_svm_handle_pagefault(vm, vma, gt, fault_addr, + ret = __xe_svm_handle_pagefault(vm, vma, pf, gt, fault_addr, need_vram ? true : false); if (ret == -EAGAIN) { /* @@ -1470,9 +1527,9 @@ void xe_svm_unmap_address_range(struct xe_vm *vm, u64 start, u64 end) drm_gpusvm_range_get(range); __xe_svm_garbage_collector(vm, to_xe_range(range)); if (!list_empty(&to_xe_range(range)->garbage_collector_link)) { - spin_lock(&vm->svm.garbage_collector.lock); + spin_lock(&vm->svm.garbage_collector.list_lock); list_del(&to_xe_range(range)->garbage_collector_link); - spin_unlock(&vm->svm.garbage_collector.lock); + spin_unlock(&vm->svm.garbage_collector.list_lock); } drm_gpusvm_range_put(range); } @@ -1502,7 +1559,7 @@ int xe_svm_bo_evict(struct xe_bo *bo) * @ctx: GPU SVM context * * This function finds or inserts a newly allocated a SVM range based on the - * address. + * address. Take a reference to SVM range on success. * * Return: Pointer to the SVM range on success, ERR_PTR() on failure. */ @@ -1511,11 +1568,15 @@ struct xe_svm_range *xe_svm_range_find_or_insert(struct xe_vm *vm, u64 addr, { struct drm_gpusvm_range *r; + guard(mutex)(&vm->svm.range_lock); + r = drm_gpusvm_range_find_or_insert(&vm->svm.gpusvm, max(addr, xe_vma_start(vma)), xe_vma_start(vma), xe_vma_end(vma), ctx); if (IS_ERR(r)) return ERR_CAST(r); + drm_gpusvm_range_get(r); + return to_xe_range(r); } @@ -1535,6 +1596,8 @@ int xe_svm_range_get_pages(struct xe_vm *vm, struct xe_svm_range *range, { int err = 0; + lockdep_assert_held(&range->lock); + err = drm_gpusvm_get_pages(&vm->svm.gpusvm, &range->pages, vm->svm.gpusvm.mm, &range->base.notifier->notifier, @@ -1659,6 +1722,7 @@ int xe_svm_alloc_vram(struct xe_svm_range *range, const struct drm_gpusvm_ctx *c .__flags = READ_ONCE(range->base.flags.__flags), }; + lockdep_assert_held(&range->lock); xe_assert(range_to_vm(&range->base)->xe, flags.migrate_devmem); range_debug(range, "ALLOCATE VRAM"); diff --git a/drivers/gpu/drm/xe/xe_svm.h b/drivers/gpu/drm/xe/xe_svm.h index a921556d3466..2a0dc0d125c9 100644 --- a/drivers/gpu/drm/xe/xe_svm.h +++ b/drivers/gpu/drm/xe/xe_svm.h @@ -21,6 +21,7 @@ struct drm_file; struct xe_bo; struct xe_gt; struct xe_device; +struct xe_pagefault; struct xe_vram_region; struct xe_tile; struct xe_vm; @@ -39,6 +40,13 @@ struct xe_svm_range { */ struct list_head garbage_collector_link; /** + * @lock: Protects fault handler, garbage collector, and prefetch + * critical sections, ensuring only one thread operates on a range at a + * time. Locking order: inside vm->lock and garbage collector, outside + * dma-resv locks, vm->svm.range_lock. + */ + struct mutex lock; + /** * @tile_present: Tile mask of binding is present for this range. * Protected by GPU SVM notifier lock. */ @@ -48,9 +56,23 @@ struct xe_svm_range { * range. Protected by GPU SVM notifier lock. */ u8 tile_invalidated; + /** + * @removed: Range has been removed from GPU SVM tree, protected by + * @lock. + */ + bool removed; }; /** + * xe_svm_range_put() - SVM range put + * @range: SVM range + */ +static inline void xe_svm_range_put(struct xe_svm_range *range) +{ + drm_gpusvm_range_put(&range->base); +} + +/** * struct xe_pagemap - Manages xe device_private memory for SVM. * @pagemap: The struct dev_pagemap providing the struct pages. * @dpagemap: The drm_pagemap managing allocation and migration. @@ -88,8 +110,8 @@ void xe_svm_fini(struct xe_vm *vm); void xe_svm_close(struct xe_vm *vm); int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma, - struct xe_gt *gt, u64 fault_addr, - bool atomic); + struct xe_pagefault *pf, struct xe_gt *gt, + u64 fault_addr, bool atomic); bool xe_svm_has_mapping(struct xe_vm *vm, u64 start, u64 end); @@ -113,7 +135,8 @@ void xe_svm_range_migrate_to_smem(struct xe_vm *vm, struct xe_svm_range *range); bool xe_svm_range_validate(struct xe_vm *vm, struct xe_svm_range *range, - u8 tile_mask, const struct drm_pagemap *dpagemap); + u8 tile_mask, const struct drm_pagemap *dpagemap, + bool *valid_pages); u64 xe_svm_find_vma_start(struct xe_vm *vm, u64 addr, u64 end, struct xe_vma *vma); @@ -138,6 +161,19 @@ static inline bool xe_svm_range_has_dma_mapping(struct xe_svm_range *range) } /** + * xe_svm_range_is_removed() - SVM range is removed from GPU SVM tree + * @range: SVM range + * + * Return: True if SVM range is removed from GPU SVM tree, False otherwise + */ +static inline bool xe_svm_range_is_removed(struct xe_svm_range *range) +{ + lockdep_assert_held(&range->lock); + + return range->removed; +} + +/** * to_xe_range - Convert a drm_gpusvm_range pointer to a xe_svm_range * @r: Pointer to the drm_gpusvm_range structure * @@ -216,10 +252,15 @@ struct xe_svm_range { struct { const struct drm_pagemap_addr *dma_addr; } pages; + struct mutex lock; u32 tile_present; u32 tile_invalidated; }; +static inline void xe_svm_range_put(struct xe_svm_range *range) +{ +} + static inline bool xe_svm_range_pages_valid(struct xe_svm_range *range) { return false; @@ -258,8 +299,8 @@ void xe_svm_close(struct xe_vm *vm) static inline int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma, - struct xe_gt *gt, u64 fault_addr, - bool atomic) + struct xe_pagefault *pf, struct xe_gt *gt, + u64 fault_addr, bool atomic) { return 0; } @@ -337,7 +378,8 @@ void xe_svm_range_migrate_to_smem(struct xe_vm *vm, struct xe_svm_range *range) static inline bool xe_svm_range_validate(struct xe_vm *vm, struct xe_svm_range *range, - u8 tile_mask, bool devmem_preferred) + u8 tile_mask, const struct drm_pagemap *dpagemap, + bool *valid_pages) { return false; } @@ -389,6 +431,11 @@ static inline struct drm_pagemap *xe_drm_pagemap_from_fd(int fd, u32 region_inst return ERR_PTR(-ENOENT); } +static inline bool xe_svm_range_is_removed(struct xe_svm_range *range) +{ + return false; +} + #define xe_svm_range_has_dma_mapping(...) false #endif /* CONFIG_DRM_XE_GPUSVM */ diff --git a/drivers/gpu/drm/xe/xe_tile_printk.h b/drivers/gpu/drm/xe/xe_tile_printk.h index 738433a764bd..c87c07e14fd1 100644 --- a/drivers/gpu/drm/xe/xe_tile_printk.h +++ b/drivers/gpu/drm/xe/xe_tile_printk.h @@ -34,6 +34,9 @@ #define xe_tile_dbg(_tile, _fmt, ...) \ xe_tile_printk((_tile), dbg, _fmt, ##__VA_ARGS__) +#define xe_tile_dbg_ratelimited(_tile, _fmt, ...) \ + xe_tile_printk((_tile), dbg_ratelimited, _fmt, ##__VA_ARGS__) + #define xe_tile_WARN_type(_tile, _type, _condition, _fmt, ...) \ xe_WARN##_type((_tile)->xe, _condition, _fmt, ## __VA_ARGS__) diff --git a/drivers/gpu/drm/xe/xe_tile_sriov_printk.h b/drivers/gpu/drm/xe/xe_tile_sriov_printk.h index 68323512872c..c23e698bc8db 100644 --- a/drivers/gpu/drm/xe/xe_tile_sriov_printk.h +++ b/drivers/gpu/drm/xe/xe_tile_sriov_printk.h @@ -27,6 +27,9 @@ #define xe_tile_sriov_dbg(_tile, _fmt, ...) \ xe_tile_sriov_printk(_tile, dbg, _fmt, ##__VA_ARGS__) +#define xe_tile_sriov_dbg_ratelimited(_tile, _fmt, ...) \ + xe_tile_sriov_printk(_tile, dbg_ratelimited, _fmt, ##__VA_ARGS__) + #define xe_tile_sriov_dbg_verbose(_tile, _fmt, ...) \ xe_tile_sriov_printk(_tile, dbg_verbose, _fmt, ##__VA_ARGS__) diff --git a/drivers/gpu/drm/xe/xe_userptr.c b/drivers/gpu/drm/xe/xe_userptr.c index 8b2d461ea0b2..90ac141fc12d 100644 --- a/drivers/gpu/drm/xe/xe_userptr.c +++ b/drivers/gpu/drm/xe/xe_userptr.c @@ -57,6 +57,23 @@ int __xe_vm_userptr_needs_repin(struct xe_vm *vm) list_empty(&vm->userptr.invalidated)) ? 0 : -EAGAIN; } +#if IS_ENABLED(CONFIG_PROVE_LOCKING) +static bool __xe_vma_userptr_lockdep(struct xe_userptr_vma *uvma) +{ + struct xe_vma *vma = &uvma->vma; + struct xe_vm *vm = xe_vma_vm(vma); + + return lockdep_is_held_type(&vm->lock, 0) || + (lockdep_is_held_type(&vm->lock, 1) && + lockdep_is_held_type(&vma->fault_lock, 0)); +} + +#define xe_vma_userptr_lockdep(uvma) \ + lockdep_assert(__xe_vma_userptr_lockdep(uvma)) +#else +#define xe_vma_userptr_lockdep(uvma) +#endif + int xe_vma_userptr_pin_pages(struct xe_userptr_vma *uvma) { struct xe_vma *vma = &uvma->vma; @@ -68,7 +85,7 @@ int xe_vma_userptr_pin_pages(struct xe_userptr_vma *uvma) .allow_mixed = true, }; - lockdep_assert_held(&vm->lock); + xe_vma_userptr_lockdep(uvma); xe_assert(xe, xe_vma_is_userptr(vma)); if (vma->gpuva.flags & XE_VMA_DESTROYED) diff --git a/drivers/gpu/drm/xe/xe_vm.c b/drivers/gpu/drm/xe/xe_vm.c index 9e0176861cb6..19b3d0be7928 100644 --- a/drivers/gpu/drm/xe/xe_vm.c +++ b/drivers/gpu/drm/xe/xe_vm.c @@ -615,10 +615,13 @@ void xe_vm_add_fault_entry_pf(struct xe_vm *vm, struct xe_pagefault *pf) { struct xe_vm_fault_entry *e; struct xe_hw_engine *hwe; + u8 engine_class = FIELD_GET(XE_PAGEFAULT_ENGINE_CLASS_MASK, + pf->consumer.engine_class_instance); + u8 engine_instance = FIELD_GET(XE_PAGEFAULT_ENGINE_INSTANCE_MASK, + pf->consumer.engine_class_instance); /* Do not report faults on reserved engines */ - hwe = xe_gt_hw_engine(pf->gt, pf->consumer.engine_class, - pf->consumer.engine_instance, false); + hwe = xe_gt_hw_engine(pf->gt, engine_class, engine_instance, false); if (!hwe || xe_hw_engine_is_reserved(hwe)) return; @@ -689,6 +692,17 @@ static int xe_vma_ops_alloc(struct xe_vma_ops *vops, bool array_of_binds) } ALLOW_ERROR_INJECTION(xe_vma_ops_alloc, ERRNO); +static void xe_vma_svm_prefetch_ranges_fini(struct xe_vma_op *op) +{ + struct xe_svm_range *svm_range; + unsigned long i; + + xa_for_each(&op->prefetch_range.range, i, svm_range) + xe_svm_range_put(svm_range); + + xa_destroy(&op->prefetch_range.range); +} + static void xe_vma_svm_prefetch_op_fini(struct xe_vma_op *op) { struct xe_vma *vma; @@ -696,7 +710,7 @@ static void xe_vma_svm_prefetch_op_fini(struct xe_vma_op *op) vma = gpuva_to_vma(op->base.prefetch.va); if (op->base.op == DRM_GPUVA_OP_PREFETCH && xe_vma_is_cpu_addr_mirror(vma)) - xa_destroy(&op->prefetch_range.range); + xe_vma_svm_prefetch_ranges_fini(op); } static void xe_vma_svm_prefetch_ops_fini(struct xe_vma_ops *vops) @@ -930,6 +944,7 @@ struct dma_fence *xe_vm_range_rebind(struct xe_vm *vm, u8 id; int err; + lockdep_assert_held(&range->lock); lockdep_assert_held(&vm->lock); xe_vm_assert_held(vm); xe_assert(vm->xe, xe_vm_in_fault_mode(vm)); @@ -1012,6 +1027,7 @@ struct dma_fence *xe_vm_range_unbind(struct xe_vm *vm, u8 id; int err; + lockdep_assert_held(&range->lock); lockdep_assert_held(&vm->lock); xe_vm_assert_held(vm); xe_assert(vm->xe, xe_vm_in_fault_mode(vm)); @@ -1187,6 +1203,8 @@ static struct xe_vma *xe_vma_create(struct xe_vm *vm, xe_vm_get(vm); } + mutex_init(&vma->fault_lock); + return vma; } @@ -1211,6 +1229,7 @@ static void xe_vma_destroy_late(struct xe_vma *vma) xe_bo_put(bo); } + mutex_destroy(&vma->fault_lock); xe_vma_free(vma); } @@ -1231,12 +1250,19 @@ static void vma_destroy_cb(struct dma_fence *fence, queue_work(system_dfl_wq, &vma->destroy_work); } +static void xe_vm_assert_write_mode_or_garbage_collector(struct xe_vm *vm) +{ + lockdep_assert(lockdep_is_held_type(&vm->lock, 0) || + (lockdep_is_held_type(&vm->lock, 1) && + lockdep_is_held_type(&vm->svm.garbage_collector.lock, 0))); +} + static void xe_vma_destroy(struct xe_vma *vma, struct dma_fence *fence) { struct xe_vm *vm = xe_vma_vm(vma); struct xe_bo *bo = xe_vma_bo(vma); - lockdep_assert_held_write(&vm->lock); + xe_vm_assert_write_mode_or_garbage_collector(vm); xe_assert(vm->xe, list_empty(&vma->combined_links.destroy)); if (xe_vma_is_userptr(vma)) { @@ -1320,7 +1346,9 @@ xe_vm_find_overlapping_vma(struct xe_vm *vm, u64 start, u64 range) xe_assert(vm->xe, start + range <= vm->size); + mutex_lock(&vm->snap_mutex); gpuva = drm_gpuva_find_first(&vm->gpuvm, start, range); + mutex_unlock(&vm->snap_mutex); return gpuva ? gpuva_to_vma(gpuva) : NULL; } @@ -1330,7 +1358,7 @@ static int xe_vm_insert_vma(struct xe_vm *vm, struct xe_vma *vma) int err; xe_assert(vm->xe, xe_vma_vm(vma) == vm); - lockdep_assert_held(&vm->lock); + xe_vm_assert_write_mode_or_garbage_collector(vm); mutex_lock(&vm->snap_mutex); err = drm_gpuva_insert(&vm->gpuvm, &vma->gpuva); @@ -1343,13 +1371,11 @@ static int xe_vm_insert_vma(struct xe_vm *vm, struct xe_vma *vma) static void xe_vm_remove_vma(struct xe_vm *vm, struct xe_vma *vma) { xe_assert(vm->xe, xe_vma_vm(vma) == vm); - lockdep_assert_held(&vm->lock); + xe_vm_assert_write_mode_or_garbage_collector(vm); mutex_lock(&vm->snap_mutex); drm_gpuva_remove(&vma->gpuva); mutex_unlock(&vm->snap_mutex); - if (vm->usm.last_fault_vma == vma) - vm->usm.last_fault_vma = NULL; } static struct drm_gpuva_op *xe_vm_op_alloc(void) @@ -2185,7 +2211,7 @@ static int xe_vm_query_vmas(struct xe_vm *vm, u64 start, u64 end) struct drm_gpuva *gpuva; u32 num_vmas = 0; - lockdep_assert_held(&vm->lock); + xe_vm_assert_write_mode_or_garbage_collector(vm); drm_gpuvm_for_each_va_range(gpuva, &vm->gpuvm, start, end) num_vmas++; @@ -2198,7 +2224,7 @@ static int get_mem_attrs(struct xe_vm *vm, u32 *num_vmas, u64 start, struct drm_gpuva *gpuva; int i = 0; - lockdep_assert_held(&vm->lock); + xe_vm_assert_write_mode_or_garbage_collector(vm); drm_gpuvm_for_each_va_range(gpuva, &vm->gpuvm, start, end) { struct xe_vma *vma = gpuva_to_vma(gpuva); @@ -2244,7 +2270,7 @@ int xe_vm_query_vmas_attrs_ioctl(struct drm_device *dev, void *data, struct drm_ if (XE_IOCTL_DBG(xe, !vm)) return -EINVAL; - err = down_read_interruptible(&vm->lock); + err = down_write_killable(&vm->lock); if (err) goto put_vm; @@ -2278,21 +2304,12 @@ int xe_vm_query_vmas_attrs_ioctl(struct drm_device *dev, void *data, struct drm_ free_mem_attrs: kvfree(mem_attrs); unlock_vm: - up_read(&vm->lock); + up_write(&vm->lock); put_vm: xe_vm_put(vm); return err; } -static bool vma_matches(struct xe_vma *vma, u64 page_addr) -{ - if (page_addr > xe_vma_end(vma) - 1 || - page_addr + SZ_4K - 1 < xe_vma_start(vma)) - return false; - - return true; -} - /** * xe_vm_find_vma_by_addr() - Find a VMA by its address * @@ -2301,16 +2318,7 @@ static bool vma_matches(struct xe_vma *vma, u64 page_addr) */ struct xe_vma *xe_vm_find_vma_by_addr(struct xe_vm *vm, u64 page_addr) { - struct xe_vma *vma = NULL; - - if (vm->usm.last_fault_vma) { /* Fast lookup */ - if (vma_matches(vm->usm.last_fault_vma, page_addr)) - vma = vm->usm.last_fault_vma; - } - if (!vma) - vma = xe_vm_find_overlapping_vma(vm, page_addr, SZ_4K); - - return vma; + return xe_vm_find_overlapping_vma(vm, page_addr, SZ_4K); } static const u32 region_to_mem_type[] = { @@ -2423,7 +2431,7 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops, u64 range_end = addr + range; int err; - lockdep_assert_held_write(&vm->lock); + xe_vm_assert_write_mode_or_garbage_collector(vm); vm_dbg(&vm->xe->drm, "op=%d, addr=0x%016llx, range=0x%016llx, bo_offset_or_userptr=0x%016llx", @@ -2446,10 +2454,12 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops, .map.gem.offset = bo_offset_or_userptr, }; + vops->flags |= XE_VMA_OPS_FLAG_MODIFIES_GPUVA; ops = drm_gpuvm_sm_map_ops_create(&vm->gpuvm, &map_req); break; } case DRM_XE_VM_BIND_OP_UNMAP: + vops->flags |= XE_VMA_OPS_FLAG_MODIFIES_GPUVA; ops = drm_gpuvm_sm_unmap_ops_create(&vm->gpuvm, addr, range); break; case DRM_XE_VM_BIND_OP_PREFETCH: @@ -2458,6 +2468,7 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops, case DRM_XE_VM_BIND_OP_UNMAP_ALL: xe_assert(vm->xe, bo); + vops->flags |= XE_VMA_OPS_FLAG_MODIFIES_GPUVA; err = xe_bo_lock(bo, true); if (err) return ERR_PTR(err); @@ -2479,6 +2490,16 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops, if (IS_ERR(ops)) return ops; + /* Setup safe unwind */ + drm_gpuva_for_each_op(__op, ops) { + struct xe_vma_op *op = gpuva_op_to_vma_op(__op); + + if (__op->op == DRM_GPUVA_OP_PREFETCH) { + xa_init_flags(&op->prefetch_range.range, XA_FLAGS_ALLOC); + op->prefetch_range.ranges_count = 0; + } + } + drm_gpuva_for_each_op(__op, ops) { struct xe_vma_op *op = gpuva_op_to_vma_op(__op); @@ -2507,10 +2528,14 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops, struct drm_pagemap *dpagemap = NULL; u8 id, tile_mask = 0; u32 i; + bool need_put, valid_pages; + + if (xe_vma_is_userptr(vma)) + vops->flags |= XE_VMA_OPS_FLAG_MODIFIES_GPUVA; if (!xe_vma_is_cpu_addr_mirror(vma)) { op->prefetch.region = prefetch_region; - break; + continue; } ctx.read_only = xe_vma_read_only(vma); @@ -2520,9 +2545,6 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops, for_each_tile(tile, vm->xe, id) tile_mask |= 0x1 << id; - xa_init_flags(&op->prefetch_range.range, XA_FLAGS_ALLOC); - op->prefetch_range.ranges_count = 0; - if (prefetch_region == DRM_XE_CONSULT_MEM_ADVISE_PREF_LOC) { dpagemap = xe_vma_resolve_pagemap(vma, xe_device_get_root_tile(vm->xe)); @@ -2534,6 +2556,7 @@ vm_bind_ioctl_ops_create(struct xe_vm *vm, struct xe_vma_ops *vops, op->prefetch_range.dpagemap = dpagemap; alloc_next_range: + need_put = false; svm_range = xe_svm_range_find_or_insert(vm, addr, vma, &ctx); if (PTR_ERR(svm_range) == -ENOENT) { @@ -2551,8 +2574,11 @@ alloc_next_range: goto unwind_prefetch_ops; } - if (xe_svm_range_validate(vm, svm_range, tile_mask, dpagemap)) { + if (xe_svm_range_validate(vm, svm_range, tile_mask, + dpagemap, &valid_pages)) { xe_svm_range_debug(svm_range, "PREFETCH - RANGE IS VALID"); + xe_assert(vm->xe, valid_pages); + need_put = true; goto check_next_range; } @@ -2560,18 +2586,27 @@ alloc_next_range: &i, svm_range, xa_limit_32b, GFP_KERNEL); - if (err) + if (err) { + xe_svm_range_put(svm_range); goto unwind_prefetch_ops; + } op->prefetch_range.ranges_count++; vops->flags |= XE_VMA_OPS_FLAG_HAS_SVM_PREFETCH; + if (valid_pages) + vops->flags |= XE_VMA_OPS_FLAG_HAS_SVM_VALID_RANGE; xe_svm_range_debug(svm_range, "PREFETCH - RANGE CREATED"); check_next_range: if (range_end > xe_svm_range_end(svm_range) && xe_svm_range_end(svm_range) < xe_vma_end(vma)) { addr = xe_svm_range_end(svm_range); + if (need_put) + xe_svm_range_put(svm_range); goto alloc_next_range; } + if (need_put) + xe_svm_range_put(svm_range); + } print_op_label: print_op(vm->xe, __op); @@ -2596,7 +2631,7 @@ static struct xe_vma *new_vma(struct xe_vm *vm, struct drm_gpuva_op_map *op, struct xe_vma *vma; int err = 0; - lockdep_assert_held_write(&vm->lock); + xe_vm_assert_write_mode_or_garbage_collector(vm); if (bo) { err = 0; @@ -2693,10 +2728,12 @@ static int xe_vma_op_commit(struct xe_vm *vm, struct xe_vma_op *op) { int err = 0; - lockdep_assert_held_write(&vm->lock); + lockdep_assert_held(&vm->lock); switch (op->base.op) { case DRM_GPUVA_OP_MAP: + xe_vm_assert_write_mode_or_garbage_collector(vm); + err |= xe_vm_insert_vma(vm, op->map.vma); if (!err) op->flags |= XE_VMA_OP_COMMITTED; @@ -2706,6 +2743,8 @@ static int xe_vma_op_commit(struct xe_vm *vm, struct xe_vma_op *op) u8 tile_present = gpuva_to_vma(op->base.remap.unmap->va)->tile_present; + xe_vm_assert_write_mode_or_garbage_collector(vm); + prep_vma_destroy(vm, gpuva_to_vma(op->base.remap.unmap->va), true); op->flags |= XE_VMA_OP_COMMITTED; @@ -2740,6 +2779,8 @@ static int xe_vma_op_commit(struct xe_vm *vm, struct xe_vma_op *op) break; } case DRM_GPUVA_OP_UNMAP: + xe_vm_assert_write_mode_or_garbage_collector(vm); + prep_vma_destroy(vm, gpuva_to_vma(op->base.unmap.va), true); op->flags |= XE_VMA_OP_COMMITTED; break; @@ -2785,7 +2826,7 @@ static int vm_bind_ioctl_ops_parse(struct xe_vm *vm, struct drm_gpuva_ops *ops, u8 id, tile_mask = 0; int err = 0; - lockdep_assert_held_write(&vm->lock); + xe_vm_assert_write_mode_or_garbage_collector(vm); for_each_tile(tile, vm->xe, id) tile_mask |= 0x1 << id; @@ -2964,10 +3005,12 @@ static void xe_vma_op_unwind(struct xe_vm *vm, struct xe_vma_op *op, bool post_commit, bool prev_post_commit, bool next_post_commit) { - lockdep_assert_held_write(&vm->lock); + lockdep_assert_held(&vm->lock); switch (op->base.op) { case DRM_GPUVA_OP_MAP: + xe_vm_assert_write_mode_or_garbage_collector(vm); + if (op->map.vma) { prep_vma_destroy(vm, op->map.vma, post_commit); xe_vma_destroy_unlocked(op->map.vma); @@ -2977,6 +3020,8 @@ static void xe_vma_op_unwind(struct xe_vm *vm, struct xe_vma_op *op, { struct xe_vma *vma = gpuva_to_vma(op->base.unmap.va); + xe_vm_assert_write_mode_or_garbage_collector(vm); + if (vma) { xe_svm_notifier_lock(vm); vma->gpuva.flags &= ~XE_VMA_DESTROYED; @@ -2990,6 +3035,8 @@ static void xe_vma_op_unwind(struct xe_vm *vm, struct xe_vma_op *op, { struct xe_vma *vma = gpuva_to_vma(op->base.remap.unmap->va); + xe_vm_assert_write_mode_or_garbage_collector(vm); + if (op->remap.prev) { prep_vma_destroy(vm, op->remap.prev, prev_post_commit); xe_vma_destroy_unlocked(op->remap.prev); @@ -3118,16 +3165,87 @@ static int check_ufence(struct xe_vma *vma) return 0; } -static int prefetch_ranges(struct xe_vm *vm, struct xe_vma_op *op) +struct prefetch_thread { + struct work_struct work; + struct drm_gpusvm_ctx *ctx; + struct xe_vma *vma; + struct xe_svm_range *svm_range; + struct drm_pagemap *dpagemap; + int err; +}; + +static void prefetch_thread_func(struct prefetch_thread *thread) +{ + struct xe_vma *vma = thread->vma; + struct xe_vm *vm = xe_vma_vm(vma); + struct xe_svm_range *svm_range = thread->svm_range; + struct drm_pagemap *dpagemap = thread->dpagemap; + int err = 0; + + guard(mutex)(&svm_range->lock); + + if (xe_svm_range_is_removed(svm_range)) + return; + + if (!dpagemap) + xe_svm_range_migrate_to_smem(vm, svm_range); + + if (IS_ENABLED(CONFIG_DRM_XE_DEBUG_VM)) { + drm_dbg(&vm->xe->drm, + "Prefetch pagemap is %s start 0x%016lx end 0x%016lx\n", + dpagemap ? dpagemap->drm->unique : "system", + xe_svm_range_start(svm_range), xe_svm_range_end(svm_range)); + } + + if (xe_svm_range_needs_migrate_to_vram(svm_range, vma, dpagemap)) { + err = xe_svm_alloc_vram(svm_range, thread->ctx, dpagemap); + if (err) { + drm_dbg(&vm->xe->drm, "VRAM allocation failed, retry from userspace, asid=%u, gpusvm=%p, errno=%pe\n", + vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err)); + /* + * We intentionally return -ENODATA on any races to + * commit any VMA updates from other ops without + * updating any page tables deferring to page faults to + * page updates skipped in the IOCTL. + */ + thread->err = -ENODATA; + return; + } + xe_svm_range_debug(svm_range, "PREFETCH - RANGE MIGRATED TO VRAM"); + } + + err = xe_svm_range_get_pages(vm, svm_range, thread->ctx); + if (err) { + drm_dbg(&vm->xe->drm, "Get pages failed, asid=%u, gpusvm=%p, errno=%pe\n", + vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err)); + if (err == -EOPNOTSUPP || err == -EFAULT || err == -EPERM) + err = -ENODATA; + thread->err = err; + return; + } + xe_svm_range_debug(svm_range, "PREFETCH - RANGE GET PAGES DONE"); +} + +static void prefetch_work_func(struct work_struct *w) +{ + struct prefetch_thread *thread = + container_of(w, struct prefetch_thread, work); + + prefetch_thread_func(thread); +} + +static int prefetch_ranges(struct xe_vm *vm, struct xe_vma_ops *vops, + struct xe_vma_op *op) { bool devmem_possible = IS_DGFX(vm->xe) && IS_ENABLED(CONFIG_DRM_XE_PAGEMAP); struct xe_vma *vma = gpuva_to_vma(op->base.prefetch.va); struct drm_pagemap *dpagemap = op->prefetch_range.dpagemap; - int err = 0; - struct xe_svm_range *svm_range; struct drm_gpusvm_ctx ctx = {}; + struct prefetch_thread stack_thread, *thread, *prefetches; unsigned long i; + int err = 0, idx = 0; + bool skip_threads; if (!xe_vma_is_cpu_addr_mirror(vma)) return 0; @@ -3137,37 +3255,50 @@ static int prefetch_ranges(struct xe_vm *vm, struct xe_vma_op *op) ctx.check_pages_threshold = devmem_possible ? SZ_64K : 0; ctx.device_private_page_owner = xe_svm_private_page_owner(vm, !dpagemap); - /* TODO: Threading the migration */ + skip_threads = op->prefetch_range.ranges_count == 1 || + (!dpagemap && !(vops->flags & + XE_VMA_OPS_FLAG_HAS_SVM_VALID_RANGE)) || + !(vops->flags & XE_VMA_OPS_FLAG_DOWNGRADE_LOCK) || + vm->xe->info.num_pf_work == 1; + thread = skip_threads ? &stack_thread : NULL; + + if (!skip_threads) { + prefetches = kvmalloc_array(op->prefetch_range.ranges_count, + sizeof(*prefetches), GFP_KERNEL); + if (!prefetches) + return -ENOMEM; + } + xa_for_each(&op->prefetch_range.range, i, svm_range) { - if (!dpagemap) - xe_svm_range_migrate_to_smem(vm, svm_range); - - if (IS_ENABLED(CONFIG_DRM_XE_DEBUG_VM)) { - drm_dbg(&vm->xe->drm, - "Prefetch pagemap is %s start 0x%016lx end 0x%016lx\n", - dpagemap ? dpagemap->drm->unique : "system", - xe_svm_range_start(svm_range), xe_svm_range_end(svm_range)); + if (!skip_threads) { + thread = prefetches + idx++; + INIT_WORK(&thread->work, prefetch_work_func); } - if (xe_svm_range_needs_migrate_to_vram(svm_range, vma, dpagemap)) { - err = xe_svm_alloc_vram(svm_range, &ctx, dpagemap); - if (err) { - drm_dbg(&vm->xe->drm, "VRAM allocation failed, retry from userspace, asid=%u, gpusvm=%p, errno=%pe\n", - vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err)); - return -ENODATA; - } - xe_svm_range_debug(svm_range, "PREFETCH - RANGE MIGRATED TO VRAM"); + thread->ctx = &ctx; + thread->vma = vma; + thread->svm_range = svm_range; + thread->dpagemap = dpagemap; + thread->err = 0; + + if (skip_threads) { + prefetch_thread_func(thread); + if (thread->err) + return thread->err; + } else { + queue_work(vm->xe->usm.prefetch_wq, &thread->work); } + } - err = xe_svm_range_get_pages(vm, svm_range, &ctx); - if (err) { - drm_dbg(&vm->xe->drm, "Get pages failed, asid=%u, gpusvm=%p, errno=%pe\n", - vm->usm.asid, &vm->svm.gpusvm, ERR_PTR(err)); - if (err == -EOPNOTSUPP || err == -EFAULT || err == -EPERM) - err = -ENODATA; - return err; + if (!skip_threads) { + for (i = 0; i < idx; ++i) { + thread = prefetches + i; + + flush_work(&thread->work); + if (thread->err && !err) + err = thread->err; } - xe_svm_range_debug(svm_range, "PREFETCH - RANGE GET PAGES DONE"); + kvfree(prefetches); } return err; @@ -3298,7 +3429,8 @@ static int op_lock_and_prep(struct drm_exec *exec, struct xe_vm *vm, return err; } -static int vm_bind_ioctl_ops_prefetch_ranges(struct xe_vm *vm, struct xe_vma_ops *vops) +static int vm_bind_ioctl_ops_prefetch_ranges(struct xe_vm *vm, + struct xe_vma_ops *vops) { struct xe_vma_op *op; int err; @@ -3308,7 +3440,7 @@ static int vm_bind_ioctl_ops_prefetch_ranges(struct xe_vm *vm, struct xe_vma_ops list_for_each_entry(op, &vops->list, link) { if (op->base.op == DRM_GPUVA_OP_PREFETCH) { - err = prefetch_ranges(vm, op); + err = prefetch_ranges(vm, vops, op); if (err) return err; } @@ -3569,7 +3701,7 @@ static struct dma_fence *vm_bind_ioctl_ops_execute(struct xe_vm *vm, struct dma_fence *fence; int err = 0; - lockdep_assert_held_write(&vm->lock); + lockdep_assert_held(&vm->lock); xe_validation_guard(&ctx, &vm->xe->val, &exec, ((struct xe_val_flags) { @@ -3892,7 +4024,7 @@ int xe_vm_bind_ioctl(struct drm_device *dev, void *data, struct drm_file *file) u32 num_syncs, num_ufence = 0; struct xe_sync_entry *syncs = NULL; struct drm_xe_vm_bind_op *bind_ops = NULL; - struct xe_vma_ops vops; + struct xe_vma_ops vops = { .flags = 0, }; struct dma_fence *fence; int err; int i; @@ -4067,6 +4199,11 @@ int xe_vm_bind_ioctl(struct drm_device *dev, void *data, struct drm_file *file) goto unwind_ops; } + if (!(vops.flags & XE_VMA_OPS_FLAG_MODIFIES_GPUVA)) { + vops.flags |= XE_VMA_OPS_FLAG_DOWNGRADE_LOCK; + downgrade_write(&vm->lock); + } + err = xe_vma_ops_alloc(&vops, args->num_binds > 1); if (err) goto unwind_ops; @@ -4103,7 +4240,10 @@ put_obj: free_bos: kvfree(bos); release_vm_lock: - up_write(&vm->lock); + if (vops.flags & XE_VMA_OPS_FLAG_DOWNGRADE_LOCK) + up_read(&vm->lock); + else + up_write(&vm->lock); put_exec_queue: if (q) xe_exec_queue_put(q); @@ -4735,7 +4875,7 @@ static int xe_vm_alloc_vma(struct xe_vm *vm, u16 default_pat; int err; - lockdep_assert_held_write(&vm->lock); + xe_vm_assert_write_mode_or_garbage_collector(vm); if (is_madvise) ops = drm_gpuvm_madvise_ops_create(&vm->gpuvm, map_req); @@ -4869,7 +5009,7 @@ int xe_vm_alloc_madvise_vma(struct xe_vm *vm, uint64_t start, uint64_t range) .map.va.range = range, }; - lockdep_assert_held_write(&vm->lock); + xe_vm_assert_write_mode_or_garbage_collector(vm); vm_dbg(&vm->xe->drm, "MADVISE_OPS_CREATE: addr=0x%016llx, size=0x%016llx", start, range); @@ -4901,7 +5041,7 @@ void xe_vm_find_cpu_addr_mirror_vma_range(struct xe_vm *vm, u64 *start, u64 *end { struct xe_vma *prev, *next; - lockdep_assert_held(&vm->lock); + xe_vm_assert_write_mode_or_garbage_collector(vm); if (*start >= SZ_4K) { prev = xe_vm_find_vma_by_addr(vm, *start - SZ_4K); @@ -4933,7 +5073,7 @@ int xe_vm_alloc_cpu_addr_mirror_vma(struct xe_vm *vm, uint64_t start, uint64_t r .map.va.range = range, }; - lockdep_assert_held_write(&vm->lock); + xe_vm_assert_write_mode_or_garbage_collector(vm); vm_dbg(&vm->xe->drm, "CPU_ADDR_MIRROR_VMA_OPS_CREATE: addr=0x%016llx, size=0x%016llx", start, range); @@ -4955,7 +5095,6 @@ void xe_vm_add_exec_queue(struct xe_vm *vm, struct xe_exec_queue *q) /* User VMs and queues only */ xe_assert(xe, !(q->flags & EXEC_QUEUE_FLAG_KERNEL)); - xe_assert(xe, !(q->flags & EXEC_QUEUE_FLAG_PERMANENT)); xe_assert(xe, !(q->flags & EXEC_QUEUE_FLAG_VM)); xe_assert(xe, !(q->flags & EXEC_QUEUE_FLAG_MIGRATE)); xe_assert(xe, vm->xef); diff --git a/drivers/gpu/drm/xe/xe_vm_types.h b/drivers/gpu/drm/xe/xe_vm_types.h index 635ed29b9a69..68588b624212 100644 --- a/drivers/gpu/drm/xe/xe_vm_types.h +++ b/drivers/gpu/drm/xe/xe_vm_types.h @@ -133,6 +133,12 @@ struct xe_vma { }; /** + * @fault_lock: Synchronizes fault processing. Locking order: inside + * vm->lock, outside dma-resv. + */ + struct mutex fault_lock; + + /** * @tile_invalidated: Tile mask of binding are invalidated for this VMA. * protected by BO's resv and for userptrs, vm->svm.gpusvm.notifier_lock in * write mode for writing or vm->svm.gpusvm.notifier_lock in read mode and @@ -215,12 +221,26 @@ struct xe_vm { /** @svm.gpusvm: base GPUSVM used to track fault allocations */ struct drm_gpusvm gpusvm; /** + * @svm.range_lock: Protects insertion and removal of ranges + * from GPU SVM tree. + */ + struct mutex range_lock; + /** * @svm.garbage_collector: Garbage collector which is used unmap * SVM range's GPU bindings and destroy the ranges. */ struct { - /** @svm.garbage_collector.lock: Protect's range list */ - spinlock_t lock; + /** + * @svm.garbage_collector.lock: Ensures only one thread + * runs the garbage collector at a time. Locking order: + * inside vm->lock, outside range->lock and dma-resv. + */ + struct mutex lock; + /** + * @svm.garbage_collector.list_lock: Protect's range + * list + */ + spinlock_t list_lock; /** * @svm.garbage_collector.range_list: List of SVM ranges * in the garbage collector. @@ -350,11 +370,6 @@ struct xe_vm { struct { /** @asid: address space ID, unique to each VM */ u32 asid; - /** - * @last_fault_vma: Last fault VMA, used for fast lookup when we - * get a flood of faults to the same VMA - */ - struct xe_vma *last_fault_vma; } usm; /** @error_capture: allow to track errors */ @@ -541,11 +556,14 @@ struct xe_vma_ops { /** @pt_update_ops: page table update operations */ struct xe_vm_pgtable_update_ops pt_update_ops[XE_MAX_TILES_PER_DEVICE]; /** @flag: signify the properties within xe_vma_ops*/ -#define XE_VMA_OPS_FLAG_HAS_SVM_PREFETCH BIT(0) -#define XE_VMA_OPS_FLAG_MADVISE BIT(1) -#define XE_VMA_OPS_ARRAY_OF_BINDS BIT(2) -#define XE_VMA_OPS_FLAG_SKIP_TLB_WAIT BIT(3) -#define XE_VMA_OPS_FLAG_ALLOW_SVM_UNMAP BIT(4) +#define XE_VMA_OPS_FLAG_HAS_SVM_PREFETCH BIT(0) +#define XE_VMA_OPS_FLAG_MADVISE BIT(1) +#define XE_VMA_OPS_ARRAY_OF_BINDS BIT(2) +#define XE_VMA_OPS_FLAG_SKIP_TLB_WAIT BIT(3) +#define XE_VMA_OPS_FLAG_ALLOW_SVM_UNMAP BIT(4) +#define XE_VMA_OPS_FLAG_MODIFIES_GPUVA BIT(5) +#define XE_VMA_OPS_FLAG_DOWNGRADE_LOCK BIT(6) +#define XE_VMA_OPS_FLAG_HAS_SVM_VALID_RANGE BIT(7) u32 flags; #ifdef TEST_VM_OPS_ERROR /** @inject_error: inject error to test error handling */ diff --git a/drivers/gpu/drm/xe/xe_wa_oob.rules b/drivers/gpu/drm/xe/xe_wa_oob.rules index f02ac9bf7424..dd69ad07f7a9 100644 --- a/drivers/gpu/drm/xe/xe_wa_oob.rules +++ b/drivers/gpu/drm/xe/xe_wa_oob.rules @@ -63,7 +63,7 @@ 16026007364 MEDIA_VERSION(3000) 14020316580 MEDIA_VERSION(1301) -14025883347 MEDIA_VERSION_RANGE(1301, 3503) +14025883347 MEDIA_VERSION_RANGE(1301, 3500) GRAPHICS_VERSION_RANGE(2004, 3005) 16029380221 MEDIA_VERSION(3500) 22022079272 MEDIA_VERSION(3503) |
