diff options
Diffstat (limited to 'drivers/gpu/drm/xe/xe_guc_submit.c')
| -rw-r--r-- | drivers/gpu/drm/xe/xe_guc_submit.c | 434 |
1 files changed, 376 insertions, 58 deletions
diff --git a/drivers/gpu/drm/xe/xe_guc_submit.c b/drivers/gpu/drm/xe/xe_guc_submit.c index 12416bfa3255..8aaed4fd13ea 100644 --- a/drivers/gpu/drm/xe/xe_guc_submit.c +++ b/drivers/gpu/drm/xe/xe_guc_submit.c @@ -10,6 +10,7 @@ #include <linux/circ_buf.h> #include <linux/dma-fence-array.h> +#include <drm/drm_drv.h> #include <drm/drm_managed.h> #include "abi/guc_actions_abi.h" @@ -37,6 +38,7 @@ #include "xe_macros.h" #include "xe_map.h" #include "xe_mocs.h" +#include "xe_module.h" #include "xe_pm.h" #include "xe_ring_ops_types.h" #include "xe_sched_job.h" @@ -232,17 +234,9 @@ static bool exec_queue_killed_or_banned_or_wedged(struct xe_exec_queue *q) static void guc_submit_sw_fini(struct drm_device *drm, void *arg) { struct xe_guc *guc = arg; - struct xe_device *xe = guc_to_xe(guc); struct xe_gt *gt = guc_to_gt(guc); - int ret; - - ret = wait_event_timeout(guc->submission_state.fini_wq, - xa_empty(&guc->submission_state.exec_queue_lookup), - HZ * 5); - - drain_workqueue(xe->destroy_wq); - xe_gt_assert(gt, ret); + xe_gt_assert(gt, xa_empty(&guc->submission_state.exec_queue_lookup)); xa_destroy(&guc->submission_state.exec_queue_lookup); } @@ -319,8 +313,6 @@ int xe_guc_submit_init(struct xe_guc *guc, unsigned int num_ids) xa_init(&guc->submission_state.exec_queue_lookup); - init_waitqueue_head(&guc->submission_state.fini_wq); - primelockdep(guc); guc->submission_state.initialized = true; @@ -411,9 +403,6 @@ static void __release_guc_id(struct xe_guc *guc, struct xe_exec_queue *q, xe_guc_id_mgr_release_locked(&guc->submission_state.idm, q->guc->id, q->width); - if (xa_empty(&guc->submission_state.exec_queue_lookup)) - wake_up(&guc->submission_state.fini_wq); - mutex_unlock(&guc->submission_state.lock); } @@ -972,6 +961,27 @@ static void __register_exec_queue(struct xe_guc *guc, xe_guc_ct_send(&guc->ct, action, ARRAY_SIZE(action), 0, 0); } +static u32 xe_hwe_guc_logical_to_submit_mask(struct xe_hw_engine *hwe, u32 logical_mask) +{ + struct xe_gt *gt = hwe->gt; + + if (xe_gt_is_usm_hwe(gt, hwe)) { + int shift = gt->usm.paging_hwe0->logical_instance; + u32 paging_logical_mask = gt->usm.paging_logical_mask; + + xe_gt_assert(gt, (logical_mask & paging_logical_mask) == logical_mask); + + /* + * Remap to GUC_PAGING_CLASS logical instance mask, if + * applicable. + */ + if (xe_guc_has_paging_engine(&hwe->gt->uc.guc)) + return logical_mask >> shift; + } + + return logical_mask; +} + static void register_exec_queue(struct xe_exec_queue *q, int ctx_type) { struct xe_guc *guc = exec_queue_to_guc(q); @@ -984,8 +994,9 @@ static void register_exec_queue(struct xe_exec_queue *q, int ctx_type) memset(&info, 0, sizeof(info)); info.context_idx = q->guc->id; - info.engine_class = xe_engine_class_to_guc_class(q->class); - info.engine_submit_mask = q->logical_mask; + info.engine_class = xe_hwe_to_guc_class(q->hwe); + info.engine_submit_mask = + xe_hwe_guc_logical_to_submit_mask(q->hwe, q->logical_mask); info.hwlrca_lo = lower_32_bits(xe_lrc_descriptor(lrc)); info.hwlrca_hi = upper_32_bits(xe_lrc_descriptor(lrc)); info.flags = CONTEXT_REGISTRATION_FLAG_KMD | @@ -1682,10 +1693,36 @@ handle_vf_resume: return DRM_GPU_SCHED_STAT_NO_HANG; } +static void guc_exec_queue_multi_queue_drop_suspend(struct xe_exec_queue *q); +static int guc_exec_queue_suspend_wait_blocking(struct xe_exec_queue *q); + static void guc_exec_queue_fini(struct xe_exec_queue *q) { struct xe_guc_exec_queue *ge = q->guc; struct xe_guc *guc = exec_queue_to_guc(q); + struct drm_device *drm = &guc_to_xe(guc)->drm; + + /* + * A secondary can leave the group while still preempt suspended (e.g. + * xe_vm_remove_compute_exec_queue() forces its preempt fence to signal, + * which suspends it). It holds one forwarded suspend reference on the + * primary, so drop it and resume the primary if it was the last member + * that had it suspended. Primaries forward to nobody, so they don't need + * this. + * + * First make sure the primary's forwarded suspend has completed. If the + * secondary was killed/reset before its preempt fence worker ran, that + * worker skips suspend_wait() (see preempt_fence_work_func()), leaving + * the primary's suspend possibly in flight. drop_suspend() runs under a + * spinlock and cannot wait, so drain it here with the uninterruptible + * blocking wait; otherwise resuming the primary in drop_suspend() could + * trip the !suspend_pending assert. + */ + if (xe_exec_queue_is_multi_queue_secondary(q)) { + if (READ_ONCE(q->guc->suspend_count)) + guc_exec_queue_suspend_wait_blocking(q); + guc_exec_queue_multi_queue_drop_suspend(q); + } if (xe_exec_queue_is_multi_queue_secondary(q)) { struct xe_exec_queue_group *group = q->multi_queue.group; @@ -1704,36 +1741,52 @@ static void guc_exec_queue_fini(struct xe_exec_queue *q) * (timeline name). */ kfree_rcu(ge, rcu); + + drm_dev_put(drm); } -static void __guc_exec_queue_destroy_async(struct work_struct *w) +static void guc_exec_queue_do_destroy(struct xe_exec_queue *q) { - struct xe_guc_exec_queue *ge = - container_of(w, struct xe_guc_exec_queue, destroy_async); - struct xe_exec_queue *q = ge->q; + struct xe_guc_exec_queue *ge = q->guc; struct xe_guc *guc = exec_queue_to_guc(q); + struct xe_device *xe = guc_to_xe(guc); + struct drm_device *drm = &xe->drm; - guard(xe_pm_runtime)(guc_to_xe(guc)); - trace_xe_exec_queue_destroy(q); + /* + * guc_exec_queue_fini() drops the queue's drm_device ref. + * Keep the device alive until the PM-runtime guard unwinds. + */ + drm_dev_get(drm); + + scoped_guard(xe_pm_runtime, xe) { + trace_xe_exec_queue_destroy(q); - /* Confirm no work left behind accessing device structures */ - cancel_delayed_work_sync(&ge->sched.base.work_tdr); + /* Confirm no work left behind accessing device structures */ + cancel_delayed_work_sync(&ge->sched.base.work_tdr); - xe_exec_queue_fini(q); + xe_exec_queue_fini(q); + } + + drm_dev_put(drm); } -static void guc_exec_queue_destroy_async(struct xe_exec_queue *q) +static void __guc_exec_queue_destroy_async(struct work_struct *w) { - struct xe_guc *guc = exec_queue_to_guc(q); - struct xe_device *xe = guc_to_xe(guc); + struct xe_guc_exec_queue *ge = + container_of(w, struct xe_guc_exec_queue, destroy_async); + + guc_exec_queue_do_destroy(ge->q); +} +static void guc_exec_queue_destroy_async(struct xe_exec_queue *q) +{ INIT_WORK(&q->guc->destroy_async, __guc_exec_queue_destroy_async); /* We must block on kernel engines so slabs are empty on driver unload */ if (q->flags & EXEC_QUEUE_FLAG_PERMANENT || exec_queue_wedged(q)) - __guc_exec_queue_destroy_async(&q->guc->destroy_async); + guc_exec_queue_do_destroy(q); else - queue_work(xe->destroy_wq, &q->guc->destroy_async); + xe_destroy_wq_queue(&q->guc->destroy_async); } static void __guc_exec_queue_destroy(struct xe_guc *guc, struct xe_exec_queue *q) @@ -1928,6 +1981,7 @@ static int guc_exec_queue_init(struct xe_exec_queue *q) { struct xe_gpu_scheduler *sched; struct xe_guc *guc = exec_queue_to_guc(q); + struct drm_device *drm = &guc_to_xe(guc)->drm; struct workqueue_struct *submit_wq = NULL; struct xe_guc_exec_queue *ge; long timeout; @@ -1939,6 +1993,8 @@ static int guc_exec_queue_init(struct xe_exec_queue *q) if (!ge) return -ENOMEM; + drm_dev_get(drm); + q->guc = ge; ge->q = q; init_rcu_head(&ge->rcu); @@ -1956,6 +2012,8 @@ static int guc_exec_queue_init(struct xe_exec_queue *q) xe_exec_queue_assign_name(q, q->guc->id); + strscpy(ge->name, q->name, sizeof(ge->name)); + /* * Use primary queue's submit_wq for all secondary queues of a * multi queue group. This serialization avoids any locking around @@ -1970,7 +2028,7 @@ static int guc_exec_queue_init(struct xe_exec_queue *q) err = xe_sched_init(&ge->sched, &drm_sched_ops, &xe_sched_ops, submit_wq, xe_lrc_ring_size() / MAX_JOB_SIZE_BYTES, 64, timeout, guc_to_gt(guc)->ordered_wq, NULL, - q->name, gt_to_xe(q->gt)->drm.dev); + ge->name, gt_to_xe(q->gt)->drm.dev); if (err) goto err_release_id; @@ -2015,6 +2073,7 @@ err_release_id: release_guc_id(guc, q); err_free: kfree(ge); + drm_dev_put(drm); return err; } @@ -2164,23 +2223,147 @@ static int guc_exec_queue_set_multi_queue_priority(struct xe_exec_queue *q, return 0; } -static int guc_exec_queue_suspend(struct xe_exec_queue *q) +/* + * Core suspend: take a suspend reference on @q and, on the first reference, + * disable its GuC context so the GPU is actually preempted. Caller must have + * ensured @q is not killed/banned/wedged. Returns true if this was the first + * suspend reference (the 0->1 transition). + */ +static bool __guc_exec_queue_suspend(struct xe_exec_queue *q) { - struct xe_gpu_scheduler *sched = &q->guc->sched; - struct xe_sched_msg *msg = q->guc->static_msgs + STATIC_MSG_SUSPEND; + struct xe_guc_exec_queue *ge = q->guc; + struct xe_gpu_scheduler *sched = &ge->sched; + struct xe_sched_msg *msg = ge->static_msgs + STATIC_MSG_SUSPEND; + bool first; - if (exec_queue_killed_or_banned_or_wedged(q)) - return -EINVAL; + xe_sched_msg_lock(sched); + first = (++ge->suspend_count == 1); + if (first) { + bool added = guc_exec_queue_try_add_msg(q, msg, SUSPEND); + + /* slot must be free at 0->1 */ + xe_gt_assert(guc_to_gt(exec_queue_to_guc(q)), added); + ge->suspend_pending = true; + } + xe_sched_msg_unlock(sched); + + return first; +} + +/* + * Core resume: drop a suspend reference on @q and, on the last reference, + * re-enable its GuC context. Returns true if this dropped the last suspend + * reference (the 1->0 transition). + */ +static bool __guc_exec_queue_resume(struct xe_exec_queue *q) +{ + struct xe_guc_exec_queue *ge = q->guc; + struct xe_gpu_scheduler *sched = &ge->sched; + struct xe_sched_msg *msg = ge->static_msgs + STATIC_MSG_RESUME; + struct xe_guc *guc = exec_queue_to_guc(q); + bool last; xe_sched_msg_lock(sched); - if (guc_exec_queue_try_add_msg(q, msg, SUSPEND)) - q->guc->suspend_pending = true; + xe_gt_assert(guc_to_gt(guc), !ge->suspend_pending); + xe_gt_assert(guc_to_gt(guc), ge->suspend_count > 0); + last = (--ge->suspend_count == 0); + if (last) { + bool added = guc_exec_queue_try_add_msg(q, msg, RESUME); + + /* slot must be free at 1->0 */ + xe_gt_assert(guc_to_gt(guc), added); + } xe_sched_msg_unlock(sched); + return last; +} + +static int guc_exec_queue_suspend(struct xe_exec_queue *q) +{ + if (exec_queue_killed_or_banned_or_wedged(q)) + return -EINVAL; + + /* + * Non-multi-queue queues and multi-queue primaries suspend themselves + * directly: their own msg_lock makes the suspend_count 0->1 transition + * and the suspend_pending update atomic, so no group level serialization + * is needed. + */ + if (!xe_exec_queue_is_multi_queue_secondary(q)) { + __guc_exec_queue_suspend(q); + return 0; + } + + /* + * A secondary's suspend is meaningless once the primary - which owns the + * group's GuC context - is gone, so fail it too. This keeps the + * secondary's effective state consistent with guc_exec_queue_reset_status(), + * which already reports the primary's killed/banned/wedged state for + * secondaries. A primary killed *after* this check is still handled at + * message-processing time, where the SUSPEND is a no-op for a killed + * context; this only covers an already-dead primary. + */ + if (exec_queue_killed_or_banned_or_wedged(xe_exec_queue_multi_queue_primary(q))) + return -EINVAL; + + /* + * A secondary doesn't interface with GuC: suspend it like any other + * queue (its own suspend_count drives its internally handled scheduler + * state) and, only on its own 0->1 transition, forward the suspend to the + * primary so the GPU is actually preempted. Hold @suspend_lock so that + * observing the secondary's transition and forwarding it to the primary + * happen atomically; this keeps the primary's refcount paired with member + * transitions even if the same secondary is suspended and resumed + * concurrently across rebind cycles. + */ + scoped_guard(spinlock, &q->multi_queue.group->suspend_lock) { + if (__guc_exec_queue_suspend(q)) + __guc_exec_queue_suspend(xe_exec_queue_multi_queue_primary(q)); + } + return 0; } -static int guc_exec_queue_suspend_wait(struct xe_exec_queue *q) +static void guc_exec_queue_suspend_timeout_ban(struct xe_exec_queue *q) +{ + struct xe_guc *guc = exec_queue_to_guc(q); + + xe_gt_warn(guc_to_gt(guc), + "Suspend fence, guc_id=%d, failed to respond, banning queue", + q->guc->id); + /* + * The GuC failed to respond to the suspend within the timeout. This is + * not recoverable for this context, so ban it and tear it down via + * cleanup rather than leave it suspended forever. __suspend_fence_signal + * clears suspend_pending and wakes any waiter. + * + * @q is the primary here; it owns the group's GuC context, so a failure + * to suspend it wedges the whole group. Ban and tear down the entire + * group in the multi-queue case. + */ + if (xe_exec_queue_is_multi_queue(q)) { + set_exec_queue_group_banned(q); + __suspend_fence_signal(q); + xe_guc_exec_queue_group_trigger_cleanup(q); + } else { + set_exec_queue_banned(q); + __suspend_fence_signal(q); + xe_guc_exec_queue_trigger_cleanup(q); + } +} + +/* + * Wait for @q's own suspend to complete: suspend_pending cleared, or the queue + * killed / GuC stopped. With @blocking, wait uninterruptibly and do not handle + * VF recovery (for callers that must complete on behalf of a possibly + * cross-process queue); otherwise wait interruptibly. + * + * Returns 0 on completion or -ETIME on timeout. Interruptible waits may also + * return -EAGAIN (VF recovery in progress, retry) or -ERESTARTSYS (aborted by a + * signal; suspend_pending may still be set, so callers must not resume() + * without re-confirming the suspend). + */ +static int guc_exec_queue_wait_suspend_done(struct xe_exec_queue *q, bool blocking) { struct xe_guc *guc = exec_queue_to_guc(q); struct xe_device *xe = guc_to_xe(guc); @@ -2196,44 +2379,146 @@ static int guc_exec_queue_suspend_wait(struct xe_exec_queue *q) xe_guc_read_stopped(guc)) retry: - if (IS_SRIOV_VF(xe)) + if (blocking) { + if (IS_SRIOV_VF(xe)) + ret = wait_event_timeout(guc->ct.wq, WAIT_COND, HZ * 5); + else + ret = wait_event_timeout(q->guc->suspend_wait, WAIT_COND, + HZ * 5); + } else if (IS_SRIOV_VF(xe)) { ret = wait_event_interruptible_timeout(guc->ct.wq, WAIT_COND || - vf_recovery(guc), - HZ * 5); - else + vf_recovery(guc), HZ * 5); + } else { ret = wait_event_interruptible_timeout(q->guc->suspend_wait, WAIT_COND, HZ * 5); + } - if (vf_recovery(guc) && !xe_device_wedged((guc_to_xe(guc)))) + if (!blocking && vf_recovery(guc) && !xe_device_wedged(xe)) return -EAGAIN; - if (!ret) { - xe_gt_warn(guc_to_gt(guc), - "Suspend fence, guc_id=%d, failed to respond", - q->guc->id); - /* XXX: Trigger GT reset? */ + if (!ret) return -ETIME; - } else if (IS_SRIOV_VF(xe) && !WAIT_COND) { + else if (!blocking && IS_SRIOV_VF(xe) && !WAIT_COND) /* Corner case on RESFIX DONE where vf_recovery() changes */ goto retry; - } #undef WAIT_COND return ret < 0 ? ret : 0; } +static int guc_exec_queue_suspend_wait_common(struct xe_exec_queue *q, bool blocking) +{ + int ret; + + /* + * A secondary's suspend rides the sched-message worker (short-circuited, + * no GuC round-trip) and so is not synchronous with + * guc_exec_queue_suspend(): its own suspend_pending may still be set + * here. Waiting on the primary alone is not sufficient - if the primary + * was already suspended, the forward is a refcount-only transition that + * queues no new primary SUSPEND and leaves the primary's suspend_pending + * clear, so the primary wait would return immediately while the + * secondary's suspend is still in flight, and a later resume() would trip + * the secondary's !suspend_pending assert. So first wait for the + * secondary's own suspend to complete, then wait on the primary. + * + * A timeout on either bans the queue (being multi-queue, that tears down + * the whole group). A secondary suspend has no real GuC round-trip, so + * its timeout is a software scheduler stall rather than a GuC fault, but + * banning is still the safe recovery: otherwise the queue is left with + * suspend_pending set and a subsequent resume() trips the !suspend_pending + * assert. + */ + if (xe_exec_queue_is_multi_queue_secondary(q)) { + ret = guc_exec_queue_wait_suspend_done(q, blocking); + if (ret == -ETIME) + guc_exec_queue_suspend_timeout_ban(q); + if (ret) + return ret; + } + + q = xe_exec_queue_multi_queue_primary(q); + ret = guc_exec_queue_wait_suspend_done(q, blocking); + if (ret == -ETIME) + guc_exec_queue_suspend_timeout_ban(q); + + return ret; +} + +static int guc_exec_queue_suspend_wait(struct xe_exec_queue *q) +{ + return guc_exec_queue_suspend_wait_common(q, false); +} + +/* + * Uninterruptible variant of guc_exec_queue_suspend_wait() for callers that + * must complete the wait on behalf of a queue possibly owned by a different + * process (e.g. cleanup/undo paths). An interruptible wait could return + * -ERESTARTSYS if the calling task is signalled, leaving that queue suspended + * forever (cross-process DoS). VF recovery is deliberately not handled (no + * -EAGAIN) since a blocking caller cannot retry. + */ +static int guc_exec_queue_suspend_wait_blocking(struct xe_exec_queue *q) +{ + return guc_exec_queue_suspend_wait_common(q, true); +} + static void guc_exec_queue_resume(struct xe_exec_queue *q) { - struct xe_gpu_scheduler *sched = &q->guc->sched; - struct xe_sched_msg *msg = q->guc->static_msgs + STATIC_MSG_RESUME; - struct xe_guc *guc = exec_queue_to_guc(q); + /* + * Non-multi-queue queues and multi-queue primaries resume themselves + * directly; their own msg_lock is sufficient. + */ + if (!xe_exec_queue_is_multi_queue_secondary(q)) { + __guc_exec_queue_resume(q); + return; + } - xe_gt_assert(guc_to_gt(guc), !q->guc->suspend_pending); + /* + * Mirror of guc_exec_queue_suspend(): resume the secondary like any + * other queue and, only on its own 1->0 transition, forward the resume + * to the primary so the primary's GuC context is re-enabled once the + * last member that suspended it resumes. @suspend_lock keeps the + * secondary transition and the primary forward atomic. + */ + scoped_guard(spinlock, &q->multi_queue.group->suspend_lock) { + if (__guc_exec_queue_resume(q)) + __guc_exec_queue_resume(xe_exec_queue_multi_queue_primary(q)); + } +} - xe_sched_msg_lock(sched); - guc_exec_queue_try_add_msg(q, msg, RESUME); - xe_sched_msg_unlock(sched); +/* + * Drop a leaving secondary's forwarded suspend reference on the primary and + * resume the primary if this was the last member that had it suspended. + * See guc_exec_queue_fini(). + */ +static void guc_exec_queue_multi_queue_drop_suspend(struct xe_exec_queue *q) +{ + scoped_guard(spinlock, &q->multi_queue.group->suspend_lock) { + struct xe_exec_queue *primary = xe_exec_queue_multi_queue_primary(q); + + /* + * A suspended secondary holds exactly one suspend reference on the + * primary (forwarded on its 0->1 transition). If it leaves while + * still suspended, release that reference so the primary is not + * kept disabled forever. + */ + if (!READ_ONCE(q->guc->suspend_count)) + break; + + if (exec_queue_killed_or_banned_or_wedged(primary)) + break; + + /* + * No suspend_wait() here (and we can't - suspend_lock is a + * spinlock). guc_exec_queue_fini() has already drained the + * primary's forwarded suspend with the blocking wait, so its + * suspend has completed (suspend_pending cleared) by the time we + * resume it here. __guc_exec_queue_resume() asserts this. + */ + __guc_exec_queue_resume(primary); + } } static bool guc_exec_queue_reset_status(struct xe_exec_queue *q) @@ -2262,6 +2547,7 @@ static const struct xe_exec_queue_ops guc_exec_queue_ops = { .set_multi_queue_priority = guc_exec_queue_set_multi_queue_priority, .suspend = guc_exec_queue_suspend, .suspend_wait = guc_exec_queue_suspend_wait, + .suspend_wait_blocking = guc_exec_queue_suspend_wait_blocking, .resume = guc_exec_queue_resume, .reset_status = guc_exec_queue_reset_status, }; @@ -3022,6 +3308,38 @@ int xe_guc_exec_queue_memory_cat_error_handler(struct xe_guc *guc, u32 *msg, return 0; } +int xe_guc_uncorrectable_error_handler(struct xe_guc *guc, u32 *msg, u32 len) +{ + struct xe_gt *gt = guc_to_gt(guc); + struct xe_exec_queue *q; + u32 guc_id; + + if (unlikely(!len || len > 1)) + return -EPROTO; + + guc_id = msg[0]; + + if (guc_id == GUC_ID_UNKNOWN) { + xe_gt_err(gt, "GuC: Uncorrectable local error with unknown GuC id\n"); + return 0; + } + + q = g2h_exec_queue_lookup(guc, guc_id); + if (unlikely(!q)) + return -EPROTO; + + xe_gt_err(gt, + "GuC: Uncorrectable local error! guc_id=%d class=%s, logical_mask=0x%x", + guc_id, xe_hw_engine_class_to_str(q->class), q->logical_mask); + + trace_xe_guc_uncorrectable_error(q); + + /* Treat the same as engine reset */ + xe_guc_exec_queue_reset_trigger_cleanup(q); + + return 0; +} + int xe_guc_exec_queue_reset_failure_handler(struct xe_guc *guc, u32 *msg, u32 len) { struct xe_gt *gt = guc_to_gt(guc); |
