summaryrefslogtreecommitdiff
path: root/drivers/gpu
diff options
context:
space:
mode:
authorNiranjana Vishwanathapura <niranjana.vishwanathapura@intel.com>2026-07-13 13:23:17 -0700
committerNiranjana Vishwanathapura <niranjana.vishwanathapura@intel.com>2026-07-13 13:29:21 -0700
commitdad2af2da9ead8a390153c0f042feacd5056888b (patch)
tree1be9331ec6b5b1c1ba4f63ec0e05263751dc84ea /drivers/gpu
parent65f40fa5022bcdb90d3a39e4948adf2aa16732a3 (diff)
drm/xe/guc: ban exec queue on suspend timeout
Harden guc_exec_queue_suspend_wait(): - In multi-queue mode the primary owns the group's GuC scheduling context, so wait on the primary's suspend to complete. - On timeout, ban the queue and trigger cleanup rather than leaving it suspended forever. Clearing suspend_pending via __suspend_fence_signal() lets a subsequent resume() proceed without tripping the !suspend_pending assert. A timeout on the primary wedges the whole group, so ban and tear down the entire group in the multi-queue case. The ban/cleanup is factored into guc_exec_queue_suspend_timeout_ban(). Add a note that on a signal (-ERESTARTSYS) the queue is not banned and the suspend is not confirmed complete, so callers must not resume() without re-confirming. Assisted-by: Github-Copilot:Claude-opus-4.8 Signed-off-by: Niranjana Vishwanathapura <niranjana.vishwanathapura@intel.com> Reviewed-by: Matthew Brost <matthew.brost@intel.com> Link: https://patch.msgid.link/20260713202317.2187787-10-niranjana.vishwanathapura@intel.com
Diffstat (limited to 'drivers/gpu')
-rw-r--r--drivers/gpu/drm/xe/xe_guc_submit.c53
1 files changed, 49 insertions, 4 deletions
diff --git a/drivers/gpu/drm/xe/xe_guc_submit.c b/drivers/gpu/drm/xe/xe_guc_submit.c
index cec3bbf3a10e..3ece51451f86 100644
--- a/drivers/gpu/drm/xe/xe_guc_submit.c
+++ b/drivers/gpu/drm/xe/xe_guc_submit.c
@@ -2202,6 +2202,34 @@ static int guc_exec_queue_suspend(struct xe_exec_queue *q)
return 0;
}
+static void guc_exec_queue_suspend_timeout_ban(struct xe_exec_queue *q)
+{
+ struct xe_guc *guc = exec_queue_to_guc(q);
+
+ xe_gt_warn(guc_to_gt(guc),
+ "Suspend fence, guc_id=%d, failed to respond, banning queue",
+ q->guc->id);
+ /*
+ * The GuC failed to respond to the suspend within the timeout. This is
+ * not recoverable for this context, so ban it and tear it down via
+ * cleanup rather than leave it suspended forever. __suspend_fence_signal
+ * clears suspend_pending and wakes any waiter.
+ *
+ * @q is the primary here; it owns the group's GuC context, so a failure
+ * to suspend it wedges the whole group. Ban and tear down the entire
+ * group in the multi-queue case.
+ */
+ if (xe_exec_queue_is_multi_queue(q)) {
+ set_exec_queue_group_banned(q);
+ __suspend_fence_signal(q);
+ xe_guc_exec_queue_group_trigger_cleanup(q);
+ } else {
+ set_exec_queue_banned(q);
+ __suspend_fence_signal(q);
+ xe_guc_exec_queue_trigger_cleanup(q);
+ }
+}
+
static int guc_exec_queue_suspend_wait(struct xe_exec_queue *q)
{
struct xe_guc *guc = exec_queue_to_guc(q);
@@ -2209,6 +2237,19 @@ static int guc_exec_queue_suspend_wait(struct xe_exec_queue *q)
int ret;
/*
+ * In multi-queue mode the primary owns the GuC scheduling context for
+ * the whole group, so wait on the primary's suspend to complete. All
+ * group members share the same GuC/device, so guc, xe and timeout above
+ * are computed from @q directly.
+ *
+ * A secondary's suspend is short-circuited (no GuC round-trip) and, as
+ * its SUSPEND message precedes the primary's on the shared FIFO
+ * submit_wq, completes before the primary's. So waiting on the primary
+ * is sufficient.
+ */
+ q = xe_exec_queue_multi_queue_primary(q);
+
+ /*
* Likely don't need to check exec_queue_killed() as we clear
* suspend_pending upon kill but to be paranoid but races in which
* suspend_pending is set after kill also check kill here.
@@ -2230,10 +2271,7 @@ retry:
return -EAGAIN;
if (!ret) {
- xe_gt_warn(guc_to_gt(guc),
- "Suspend fence, guc_id=%d, failed to respond",
- q->guc->id);
- /* XXX: Trigger GT reset? */
+ guc_exec_queue_suspend_timeout_ban(q);
return -ETIME;
} else if (IS_SRIOV_VF(xe) && !WAIT_COND) {
/* Corner case on RESFIX DONE where vf_recovery() changes */
@@ -2242,6 +2280,13 @@ retry:
#undef WAIT_COND
+ /*
+ * ret < 0 (-ERESTARTSYS): the interruptible wait was aborted by a
+ * signal. The queue is not banned - the failure is in the waiter, not
+ * the queue. The suspend is not confirmed complete, so suspend_pending
+ * may still be set; callers must not resume() on this error without
+ * re-confirming the suspend.
+ */
return ret < 0 ? ret : 0;
}