* [PATCH v4 1/7] drm/xe/guc: Use different error codes for GuC load errors
2026-09-17 23:59 [PATCH v4 0/7] Use SIG_ID logs for GuC component Umesh Nerlige Ramappa
@ 2026-09-17 23:59 ` Umesh Nerlige Ramappa
2026-09-17 23:59 ` [PATCH v4 2/7] drm/xe/guc: Handle CRASH and EXCEPTION G2H with separate helpers Umesh Nerlige Ramappa
` (6 subsequent siblings)
7 siblings, 0 replies; 14+ messages in thread
From: Umesh Nerlige Ramappa @ 2026-09-17 23:59 UTC (permalink / raw)
To: intel-xe
Cc: daniele.ceraolospurio, michal.wajdeczko, aravind.iddamsetty,
mallesh.koujalagi, alan.previn.teres.alexis, julia.filipchuk
From: Daniele Ceraolo Spurio <daniele.ceraolospurio@intel.com>
Instead of always returning -EPROTO no matter what goes wrong, use
different error codes based on the error type.
Signed-off-by: Daniele Ceraolo Spurio <daniele.ceraolospurio@intel.com>
Reviewed-by: Michal Wajdeczko <michal.wajdeczko@intel.com>
---
v2: split to its own patch
---
drivers/gpu/drm/xe/xe_guc.c | 22 ++++++++++++----------
1 file changed, 12 insertions(+), 10 deletions(-)
diff --git a/drivers/gpu/drm/xe/xe_guc.c b/drivers/gpu/drm/xe/xe_guc.c
index c7f8bbd4cb92..5285d4cecbc8 100644
--- a/drivers/gpu/drm/xe/xe_guc.c
+++ b/drivers/gpu/drm/xe/xe_guc.c
@@ -1144,8 +1144,8 @@ static void print_load_status_err(struct xe_gt *gt, u32 status)
* Check GUC_STATUS looking for known terminal states (either completion or
* failure) of either the microkernel status field or the boot ROM status field.
*
- * Returns 1 for successful completion, -1 for failure and 0 for any
- * intermediate state.
+ * Returns 0 for successful completion, -EBUSY for any intermediate state and
+ * other errno values for failure cases.
*/
static int guc_load_done(struct xe_gt *gt, u32 *status, u32 *tries)
{
@@ -1157,20 +1157,22 @@ static int guc_load_done(struct xe_gt *gt, u32 *status, u32 *tries)
switch (ukernel) {
case XE_GUC_LOAD_STATUS_READY:
- return 1;
+ return 0;
case XE_GUC_LOAD_STATUS_ERROR_DEVID_BUILD_MISMATCH:
case XE_GUC_LOAD_STATUS_GUC_PREPROD_BUILD_MISMATCH:
case XE_GUC_LOAD_STATUS_ERROR_DEVID_INVALID_GUCTYPE:
case XE_GUC_LOAD_STATUS_HWCONFIG_ERROR:
case XE_GUC_LOAD_STATUS_BOOTROM_VERSION_MISMATCH:
+ return -ENOEXEC;
case XE_GUC_LOAD_STATUS_DPC_ERROR:
case XE_GUC_LOAD_STATUS_EXCEPTION:
+ return -EIO;
case XE_GUC_LOAD_STATUS_INIT_DATA_INVALID:
case XE_GUC_LOAD_STATUS_MPU_DATA_INVALID:
case XE_GUC_LOAD_STATUS_INIT_MMIO_SAVE_RESTORE_INVALID:
case XE_GUC_LOAD_STATUS_KLV_WORKAROUND_INIT_ERROR:
case XE_GUC_LOAD_STATUS_INVALID_FTR_FLAG:
- return -1;
+ return -EINVAL;
}
switch (bootrom) {
@@ -1184,7 +1186,7 @@ static int guc_load_done(struct xe_gt *gt, u32 *status, u32 *tries)
case XE_BOOTROM_STATUS_MPUMAP_INCORRECT:
case XE_BOOTROM_STATUS_EXCEPTION:
case XE_BOOTROM_STATUS_PROD_KEY_CHECK_FAILURE:
- return -1;
+ return -ENOEXEC;
}
if (++*tries >= 100) {
@@ -1197,7 +1199,7 @@ static int guc_load_done(struct xe_gt *gt, u32 *status, u32 *tries)
*status, ukernel, bootrom);
}
- return 0;
+ return -EBUSY;
}
static int guc_wait_ucode(struct xe_guc *guc)
@@ -1213,21 +1215,21 @@ static int guc_wait_ucode(struct xe_guc *guc)
before_freq = xe_guc_pc_get_act_freq(guc_pc);
before = ktime_get();
- ret = poll_timeout_us(load_result = guc_load_done(gt, &status, &tries), load_result,
- 10 * USEC_PER_MSEC,
+ ret = poll_timeout_us(load_result = guc_load_done(gt, &status, &tries),
+ load_result != -EBUSY, 10 * USEC_PER_MSEC,
GUC_LOAD_TIMEOUT_SEC * USEC_PER_SEC, false);
delta_ms = ktime_to_ms(ktime_sub(ktime_get(), before));
act_freq = xe_guc_pc_get_act_freq(guc_pc);
cur_freq = xe_guc_pc_get_cur_freq_fw(guc_pc);
- if (ret || load_result <= 0) {
+ if (ret || load_result) {
xe_gt_err(gt, "load failed: status = 0x%08X, time = %lldms, freq = %dMHz (req %dMHz)\n",
status, delta_ms, xe_guc_pc_get_act_freq(guc_pc),
xe_guc_pc_get_cur_freq_fw(guc_pc));
print_load_status_err(gt, status);
- return -EPROTO;
+ return ret ?: load_result;
}
if (delta_ms > GUC_LOAD_TIME_WARN_MSEC) {
--
2.55.0
^ permalink raw reply related [flat|nested] 14+ messages in thread* [PATCH v4 2/7] drm/xe/guc: Handle CRASH and EXCEPTION G2H with separate helpers
2026-09-17 23:59 [PATCH v4 0/7] Use SIG_ID logs for GuC component Umesh Nerlige Ramappa
2026-09-17 23:59 ` [PATCH v4 1/7] drm/xe/guc: Use different error codes for GuC load errors Umesh Nerlige Ramappa
@ 2026-09-17 23:59 ` Umesh Nerlige Ramappa
2026-09-18 0:08 ` sashiko-bot
2026-09-17 23:59 ` [PATCH v4 3/7] drm/xe/guc: Make ct_dead_capture available on non-debug config Umesh Nerlige Ramappa
` (5 subsequent siblings)
7 siblings, 1 reply; 14+ messages in thread
From: Umesh Nerlige Ramappa @ 2026-09-17 23:59 UTC (permalink / raw)
To: intel-xe
Cc: daniele.ceraolospurio, michal.wajdeczko, aravind.iddamsetty,
mallesh.koujalagi, alan.previn.teres.alexis, julia.filipchuk
Refactor to handle the CRASH and EXCEPTION G2H separately. In the
process drop the else case that existed earlier but was unreachable
code.
Suggested-by: Michal Wajdeczko <michal.wajdeczko@intel.com>
Signed-off-by: Umesh Nerlige Ramappa <umesh.nerlige.ramappa@intel.com>
---
drivers/gpu/drm/xe/xe_guc_ct.c | 24 +++++++++++++-----------
1 file changed, 13 insertions(+), 11 deletions(-)
diff --git a/drivers/gpu/drm/xe/xe_guc_ct.c b/drivers/gpu/drm/xe/xe_guc_ct.c
index 5c4733da385c..b3a6aa37808b 100644
--- a/drivers/gpu/drm/xe/xe_guc_ct.c
+++ b/drivers/gpu/drm/xe/xe_guc_ct.c
@@ -1553,22 +1553,22 @@ static int parse_g2h_event(struct xe_guc_ct *ct, u32 *msg, u32 len)
return 0;
}
-static int guc_crash_process_msg(struct xe_guc_ct *ct, u32 action)
+static int guc_handle_crash_dump_msg(struct xe_guc_ct *ct, u32 action)
{
- struct xe_gt *gt = ct_to_gt(ct);
+ xe_gt_err(ct_to_gt(ct), "GuC Crash dump notification\n");
+ CT_DEAD(ct, NULL, CRASH);
+ kick_reset(ct);
- if (action == XE_GUC_ACTION_NOTIFY_CRASH_DUMP_POSTED)
- xe_gt_err(gt, "GuC Crash dump notification\n");
- else if (action == XE_GUC_ACTION_NOTIFY_EXCEPTION)
- xe_gt_err(gt, "GuC Exception notification\n");
- else
- xe_gt_err(gt, "Unknown GuC crash notification: 0x%04X\n", action);
+ return -EHOSTDOWN;
+}
+static int guc_handle_exception_msg(struct xe_guc_ct *ct, u32 action)
+{
+ xe_gt_err(ct_to_gt(ct), "GuC Exception notification\n");
CT_DEAD(ct, NULL, CRASH);
-
kick_reset(ct);
- return 0;
+ return -EHOSTDOWN;
}
static int parse_g2h_response(struct xe_guc_ct *ct, u32 *msg, u32 len)
@@ -1772,8 +1772,10 @@ static int process_g2h_msg(struct xe_guc_ct *ct, u32 *msg, u32 len)
ret = xe_gt_sriov_pf_monitor_process_guc2pf(gt, hxg, hxg_len);
break;
case XE_GUC_ACTION_NOTIFY_CRASH_DUMP_POSTED:
+ ret = guc_handle_crash_dump_msg(ct, action);
+ break;
case XE_GUC_ACTION_NOTIFY_EXCEPTION:
- ret = guc_crash_process_msg(ct, action);
+ ret = guc_handle_exception_msg(ct, action);
break;
#if IS_ENABLED(CONFIG_DRM_XE_KUNIT_TEST)
case XE_GUC_ACTION_TEST_G2G_RECV:
--
2.55.0
^ permalink raw reply related [flat|nested] 14+ messages in thread* Re: [PATCH v4 2/7] drm/xe/guc: Handle CRASH and EXCEPTION G2H with separate helpers
2026-09-17 23:59 ` [PATCH v4 2/7] drm/xe/guc: Handle CRASH and EXCEPTION G2H with separate helpers Umesh Nerlige Ramappa
@ 2026-09-18 0:08 ` sashiko-bot
0 siblings, 0 replies; 14+ messages in thread
From: sashiko-bot @ 2026-09-18 0:08 UTC (permalink / raw)
To: Umesh Nerlige Ramappa; +Cc: intel-xe
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [Medium] xe: spurious PROCESS_FAILED on GuC crash due to -EHOSTDOWN return
--
commit 0030c089dd9bc1455460d3cc58697be48bcade56
Author: Umesh Nerlige Ramappa <umesh.nerlige.ramappa@intel.com>
drm/xe/guc: Handle CRASH and EXCEPTION G2H with separate helpers
This commit refactors the handling of CRASH and EXCEPTION G2H messages
into separate helper functions. It also drops an unreachable code path
that existed in the earlier combined function.
> diff --git a/drivers/gpu/drm/xe/xe_guc_ct.c b/drivers/gpu/drm/xe/xe_guc_ct.c
> index 5c4733da385c5..b3a6aa37808b4 100644
> --- a/drivers/gpu/drm/xe/xe_guc_ct.c
> +++ b/drivers/gpu/drm/xe/xe_guc_ct.c
> @@ -1553,22 +1553,22 @@ static int parse_g2h_event(struct xe_guc_ct *ct, u32 *msg, u32 len)
> return 0;
> }
>
> -static int guc_crash_process_msg(struct xe_guc_ct *ct, u32 action)
> +static int guc_handle_crash_dump_msg(struct xe_guc_ct *ct, u32 action)
> {
> - struct xe_gt *gt = ct_to_gt(ct);
> + xe_gt_err(ct_to_gt(ct), "GuC Crash dump notification\n");
> + CT_DEAD(ct, NULL, CRASH);
> + kick_reset(ct);
>
> - if (action == XE_GUC_ACTION_NOTIFY_CRASH_DUMP_POSTED)
> - xe_gt_err(gt, "GuC Crash dump notification\n");
[ ... ]
> + return -EHOSTDOWN;
> +}
[Severity: Medium]
Since this function now returns -EHOSTDOWN instead of 0, will this trigger the
failure path in process_g2h_msg()?
When process_g2h_msg() receives a non-zero return value, it logs a spurious
"G2H action failed" error and redundantly flags the CT as dead:
if (ret) {
xe_gt_err(gt, "G2H action %#04x failed (%pe) len %u msg %*ph\n", ...);
CT_DEAD(ct, NULL, PROCESS_FAILED);
}
Does this misrepresent a successfully processed firmware notification as a
host-side parsing failure?
>
> +static int guc_handle_exception_msg(struct xe_guc_ct *ct, u32 action)
> +{
> + xe_gt_err(ct_to_gt(ct), "GuC Exception notification\n");
> CT_DEAD(ct, NULL, CRASH);
> -
> kick_reset(ct);
>
> - return 0;
> + return -EHOSTDOWN;
> }
[Severity: Medium]
Similarly here, returning -EHOSTDOWN instead of 0 will cause process_g2h_msg()
to execute its error handling path. Should these helpers continue to return 0
to indicate the notification was successfully processed by the host?
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260917235923.1521112-9-umesh.nerlige.ramappa@intel.com?part=2
^ permalink raw reply [flat|nested] 14+ messages in thread
* [PATCH v4 3/7] drm/xe/guc: Make ct_dead_capture available on non-debug config
2026-09-17 23:59 [PATCH v4 0/7] Use SIG_ID logs for GuC component Umesh Nerlige Ramappa
2026-09-17 23:59 ` [PATCH v4 1/7] drm/xe/guc: Use different error codes for GuC load errors Umesh Nerlige Ramappa
2026-09-17 23:59 ` [PATCH v4 2/7] drm/xe/guc: Handle CRASH and EXCEPTION G2H with separate helpers Umesh Nerlige Ramappa
@ 2026-09-17 23:59 ` Umesh Nerlige Ramappa
2026-09-18 0:06 ` sashiko-bot
2026-09-17 23:59 ` [PATCH v4 4/7] drm/xe/guc: Cleanup error codes and handling for CT errors Umesh Nerlige Ramappa
` (4 subsequent siblings)
7 siblings, 1 reply; 14+ messages in thread
From: Umesh Nerlige Ramappa @ 2026-09-17 23:59 UTC (permalink / raw)
To: intel-xe
Cc: daniele.ceraolospurio, michal.wajdeczko, aravind.iddamsetty,
mallesh.koujalagi, alan.previn.teres.alexis, julia.filipchuk
Define ct_dead_capture for non-debug as well to simplify future
refactor.
Signed-off-by: Umesh Nerlige Ramappa <umesh.nerlige.ramappa@intel.com>
---
drivers/gpu/drm/xe/xe_guc_ct.c | 19 +++++++++----------
1 file changed, 9 insertions(+), 10 deletions(-)
diff --git a/drivers/gpu/drm/xe/xe_guc_ct.c b/drivers/gpu/drm/xe/xe_guc_ct.c
index b3a6aa37808b..3353515aa874 100644
--- a/drivers/gpu/drm/xe/xe_guc_ct.c
+++ b/drivers/gpu/drm/xe/xe_guc_ct.c
@@ -59,6 +59,9 @@ static struct xe_device *ct_to_xe(struct xe_guc_ct *ct)
return gt_to_xe(ct_to_gt(ct));
}
+#define ct_id(reason_code) CT_DEAD_##reason_code
+#define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
+
#if IS_ENABLED(CONFIG_DRM_XE_DEBUG)
enum {
/* Internal states, not error conditions */
@@ -162,9 +165,6 @@ static void fast_req_track(struct xe_guc_ct *ct, u16 fence, u16 action)
ct->fast_req[slot].fence = fence;
ct->fast_req[slot].action = action;
}
-
-#define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), CT_DEAD_##reason_code)
-
#else
static void ct_dead_fini(struct xe_guc_ct *ct) { }
@@ -173,13 +173,6 @@ static void ct_dead_init(struct xe_guc_ct *ct) { }
static void fast_req_report(struct xe_guc_ct *ct, u16 fence) { }
static void fast_req_track(struct xe_guc_ct *ct, u16 fence, u16 action) { }
-#define CT_DEAD(ct, ctb, reason) \
- do { \
- struct guc_ctb *_ctb = (ctb); \
- if (_ctb) \
- _ctb->info.broken = true; \
- } while (0)
-
#endif
/* Used when a CT send wants to block and / or receive data */
@@ -2376,4 +2369,10 @@ static void ct_dead_worker_func(struct work_struct *w)
spin_unlock_irq(&ct->dead.lock);
}
+#else
+static void ct_dead_capture(struct xe_guc_ct *ct, struct guc_ctb *ctb, u32 reason_code)
+{
+ if (ctb)
+ ctb->info.broken = true;
+}
#endif
--
2.55.0
^ permalink raw reply related [flat|nested] 14+ messages in thread* Re: [PATCH v4 3/7] drm/xe/guc: Make ct_dead_capture available on non-debug config
2026-09-17 23:59 ` [PATCH v4 3/7] drm/xe/guc: Make ct_dead_capture available on non-debug config Umesh Nerlige Ramappa
@ 2026-09-18 0:06 ` sashiko-bot
0 siblings, 0 replies; 14+ messages in thread
From: sashiko-bot @ 2026-09-18 0:06 UTC (permalink / raw)
To: Umesh Nerlige Ramappa; +Cc: intel-xe
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [Low] build: xe_guc_ct compile error on !CONFIG_DRM_XE_DEBUG
--
commit 1f4499efdc2f257f0698c2212dc79db666b74590
Author: Umesh Nerlige Ramappa <umesh.nerlige.ramappa@intel.com>
drm/xe/guc: Make ct_dead_capture available on non-debug config
Define ct_dead_capture for non-debug as well to simplify future
refactor.
> diff --git a/drivers/gpu/drm/xe/xe_guc_ct.c b/drivers/gpu/drm/xe/xe_guc_ct.c
> index b3a6aa37808b4..3353515aa8747 100644
> --- a/drivers/gpu/drm/xe/xe_guc_ct.c
> +++ b/drivers/gpu/drm/xe/xe_guc_ct.c
> @@ -59,6 +59,9 @@ static struct xe_device *ct_to_xe(struct xe_guc_ct *ct)
> return gt_to_xe(ct_to_gt(ct));
> }
>
> +#define ct_id(reason_code) CT_DEAD_##reason_code
> +#define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
> +
> #if IS_ENABLED(CONFIG_DRM_XE_DEBUG)
> enum {
> /* Internal states, not error conditions */
[Severity: Low]
Will this break compilation for non-debug builds?
The macro CT_DEAD now unconditionally references CT_DEAD_##reason_code, but
the enumeration defining these reason codes remains strictly guarded by
#if IS_ENABLED(CONFIG_DRM_XE_DEBUG) just below.
Any invocation of CT_DEAD in common code could cause an undeclared identifier
error when CONFIG_DRM_XE_DEBUG is disabled.
[ ... ]
> @@ -2376,4 +2369,10 @@ static void ct_dead_worker_func(struct work_struct *w)
>
> spin_unlock_irq(&ct->dead.lock);
> }
> +#else
> +static void ct_dead_capture(struct xe_guc_ct *ct, struct guc_ctb *ctb, u32 reason_code)
> +{
> + if (ctb)
> + ctb->info.broken = true;
> +}
> #endif
[Severity: Low]
Does ct_dead_capture() also need a forward declaration for the non-debug
path?
Since this non-debug implementation is defined at the end of the file with
no preceding forward declaration, calling CT_DEAD() earlier in the file may
result in an implicit function declaration error.
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260917235923.1521112-9-umesh.nerlige.ramappa@intel.com?part=3
^ permalink raw reply [flat|nested] 14+ messages in thread
* [PATCH v4 4/7] drm/xe/guc: Cleanup error codes and handling for CT errors
2026-09-17 23:59 [PATCH v4 0/7] Use SIG_ID logs for GuC component Umesh Nerlige Ramappa
` (2 preceding siblings ...)
2026-09-17 23:59 ` [PATCH v4 3/7] drm/xe/guc: Make ct_dead_capture available on non-debug config Umesh Nerlige Ramappa
@ 2026-09-17 23:59 ` Umesh Nerlige Ramappa
2026-09-18 0:17 ` sashiko-bot
2026-09-17 23:59 ` [PATCH v4 5/7] drm/xe/uc: Report DMA failure using SIGID Umesh Nerlige Ramappa
` (3 subsequent siblings)
7 siblings, 1 reply; 14+ messages in thread
From: Umesh Nerlige Ramappa @ 2026-09-17 23:59 UTC (permalink / raw)
To: intel-xe
Cc: daniele.ceraolospurio, michal.wajdeczko, aravind.iddamsetty,
mallesh.koujalagi, alan.previn.teres.alexis, julia.filipchuk
Instead of always returning -EPROTO for most errors, use different error
codes based on the error type. -EPROTO is retained for the cases that
are genuine protocol violations by the GuC. The remaining cases now
report what actually went wrong.
receive_g2h() used to escalate to CT_DEAD + kick_reset() by matching the
two error codes that dequeue_one_g2h() could return on a fatal error.
Invert the check to simplify reset handling.
Signed-off-by: Umesh Nerlige Ramappa <umesh.nerlige.ramappa@intel.com>
Assisted-by: Claude:claude-opus-5
---
v2: (Michal)
- Sync order of errors in g2h_read and __guc_ct_send_locked
- For CT errors use EPIPE and for HXG errors use EPROTO
- Convert the non-fatal EPIPE to EPERM in the helper
v3: (Michal)
- s/EPIPE/EPERM/ in __guc_ct_send_locked
- Add ct_corrupted helper
- Add error documentation
- Remove "reset required" string from logs
---
drivers/gpu/drm/xe/xe_guc_ct.c | 190 ++++++++++++++++++++++++---------
1 file changed, 141 insertions(+), 49 deletions(-)
diff --git a/drivers/gpu/drm/xe/xe_guc_ct.c b/drivers/gpu/drm/xe/xe_guc_ct.c
index 3353515aa874..31ecddab3057 100644
--- a/drivers/gpu/drm/xe/xe_guc_ct.c
+++ b/drivers/gpu/drm/xe/xe_guc_ct.c
@@ -37,6 +37,71 @@
#include "xe_sriov_vf.h"
#include "xe_trace_guc.h"
+/**
+ * DOC: GuC CT Error Codes
+ *
+ * Send path - xe_guc_ct_send(), xe_guc_ct_send_locked(), h2g_write()
+ * ------------------------------------------------------------------
+ *
+ * Channel-state rejections - the message was never queued, nothing is corrupt:
+ *
+ * ``-ENOTRECOVERABLE``: the device is wedged. No CT traffic can make progress
+ * until the device is recovered.
+ * ``-ENODEV``: the CT channel is disabled. Don't retry until it is enabled.
+ * ``-ECANCELED``: the CT channel is stopped or a GT recovery is pending, so the
+ * message was dropped. Often benign; retry after the reset/recovery completes.
+ * ``-EPERM``: the H2G CTB is flagged broken and stays unusable until the CT is
+ * restarted, which clears the flag.
+ *
+ * Internal flow control - never seen by callers, handled by the send helpers:
+ *
+ * ``-EBUSY``: not enough H2G room or G2H credits. Triggers a flush, a wait and
+ * a retry.
+ * ``-EAGAIN``: the message would wrap the H2G ring, so the tail was NOP-filled
+ * and reset to 0. The write is reissued immediately.
+ *
+ * Hard failures:
+ *
+ * ``-EPIPE``: H2G descriptor corruption found before writing - non-zero status,
+ * or head/tail out of range. Marks the CTB broken.
+ * ``-EDEADLK``: H2G credits never recovered after waiting ~1s. An async GT reset
+ * is requested before this is returned.
+ * ``-EPROTO``: the MMIO CT enable/disable handshake got an unexpected reply.
+ * ``-ENOMEM``: internal allocation failure (fence lookup under GFP_ATOMIC, or the
+ * G2H workqueue at init). The blocking path retries the allocation with
+ * GFP_KERNEL.
+ *
+ * Blocking send only (xe_guc_ct_send_recv()) - the H2G went out, the reply is the
+ * problem:
+ *
+ * ``-ETIME``: no G2H response within 1s, even after flushing the G2H worker.
+ * ``-EIO``: the GuC replied RESPONSE_FAILURE; error and hint codes are logged.
+ * ``-ELOOP``: the GuC kept replying NO_RESPONSE_RETRY until the request hit
+ * GUC_SEND_RETRY_LIMIT. (NO_RESPONSE_BUSY is not an error - the waiter re-arms.)
+ *
+ * Receive path - g2h_read(), parse_g2h_msg(), dequeue_one_g2h()
+ * -------------------------------------------------------------
+ *
+ * Channel-state rejections - the same four codes as the send path, checked against
+ * the G2H CTB. g2h_err_is_fatal() treats exactly these as benign; anything else
+ * declares the CT dead and kicks a reset:
+ *
+ * ``-ENOTRECOVERABLE``: device wedged.
+ * ``-ENODEV``: CT disabled.
+ * ``-ECANCELED``: CT stopped.
+ * ``-EPERM``: G2H CTB already flagged broken.
+ *
+ * Hard failures:
+ *
+ * ``-EPIPE``: G2H descriptor corruption - non-zero status, head/tail out of
+ * range, or a declared message length larger than the data actually available.
+ * ``-EPROTO``: HXG protocol violation. Either the G2H origin field is not the
+ * GuC, or a FAST_REQUEST H2G came back rejected. FAST_REQUEST fences aren't
+ * tracked, so there is no caller to report to and the failure is escalated to a
+ * reset instead.
+ * ``-EOPNOTSUPP``: G2H carried an HXG type this driver does not handle.
+ */
+
static void receive_g2h(struct xe_guc_ct *ct);
static void g2h_worker_func(struct work_struct *w);
static void safe_mode_worker_func(struct work_struct *w);
@@ -929,6 +994,23 @@ static bool vf_action_can_safely_fail(struct xe_device *xe, u32 action)
action == XE_GUC_ACTION_REGISTER_CONTEXT);
}
+static int ct_corrupted(struct xe_guc_ct *ct, struct guc_ctb *ctb,
+ u32 reason_code, const char *msg, ...)
+{
+ struct va_format vaf;
+ va_list va_args;
+ int err = -EPIPE;
+
+ va_start(va_args, msg);
+ vaf.fmt = msg;
+ vaf.va = &va_args;
+ xe_gt_err(ct_to_gt(ct), "GUC: CT: %pV", &vaf);
+ va_end(va_args);
+
+ ct_dead_capture(ct, ctb, reason_code);
+ return err;
+}
+
#define H2G_CT_HEADERS (GUC_CTB_HDR_LEN + 1) /* one DW CTB header and one DW HxG header */
static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
@@ -953,23 +1035,22 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
u32 desc_status;
desc_status = desc_read(xe, h2g, status);
- if (desc_status) {
- xe_gt_err(gt, "CT write: non-zero status: %u\n", desc_status);
- goto corrupted;
- }
+ if (desc_status)
+ return ct_corrupted(ct, &ct->ctbs.h2g, ct_id(H2G_WRITE),
+ "write: non-zero status: %u\n", desc_status);
if (tail > h2g->info.size) {
desc_write(xe, h2g, status, desc_status | GUC_CTB_STATUS_OVERFLOW);
- xe_gt_err(gt, "CT write: tail out of range: %u vs %u\n",
- tail, h2g->info.size);
- goto corrupted;
+ return ct_corrupted(ct, &ct->ctbs.h2g, ct_id(H2G_WRITE),
+ "write: tail out of range: %u vs %u\n",
+ tail, h2g->info.size);
}
if (desc_head >= h2g->info.size) {
desc_write(xe, h2g, status, desc_status | GUC_CTB_STATUS_OVERFLOW);
- xe_gt_err(gt, "CT write: invalid head offset %u >= %u)\n",
- desc_head, h2g->info.size);
- goto corrupted;
+ return ct_corrupted(ct, &ct->ctbs.h2g, ct_id(H2G_WRITE),
+ "write: invalid head offset %u >= %u)\n",
+ desc_head, h2g->info.size);
}
}
@@ -1034,10 +1115,6 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
desc_read(xe, h2g, head), h2g->info.tail);
return 0;
-
-corrupted:
- CT_DEAD(ct, &ct->ctbs.h2g, H2G_WRITE);
- return -EPIPE;
}
static int __guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action,
@@ -1060,11 +1137,6 @@ static int __guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action,
goto out;
}
- if (unlikely(ct->ctbs.h2g.info.broken)) {
- ret = -EPIPE;
- goto out;
- }
-
if (ct->state == XE_GUC_CT_STATE_DISABLED) {
ret = -ENODEV;
goto out;
@@ -1075,6 +1147,11 @@ static int __guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action,
goto out;
}
+ if (unlikely(ct->ctbs.h2g.info.broken)) {
+ ret = -EPERM;
+ goto out;
+ }
+
xe_gt_assert(gt, xe_guc_ct_enabled(ct));
if (g2h_fence) {
@@ -1213,7 +1290,7 @@ static int guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len,
return ret;
broken:
- xe_gt_err(gt, "No forward process on H2G, reset required\n");
+ xe_gt_err(gt, "No forward progress on H2G\n");
CT_DEAD(ct, &ct->ctbs.h2g, DEADLOCK);
return -EDEADLK;
@@ -1667,8 +1744,7 @@ static int parse_g2h_msg(struct xe_guc_ct *ct, u32 *msg, u32 len)
origin = FIELD_GET(GUC_HXG_MSG_0_ORIGIN, hxg[0]);
if (unlikely(origin != GUC_HXG_ORIGIN_GUC)) {
- xe_gt_err(gt, "G2H channel broken on read, origin=%u, reset required\n",
- origin);
+ xe_gt_err(gt, "Invalid G2H origin=%u\n", origin);
CT_DEAD(ct, &ct->ctbs.g2h, PARSE_G2H_ORIGIN);
return -EPROTO;
@@ -1686,8 +1762,7 @@ static int parse_g2h_msg(struct xe_guc_ct *ct, u32 *msg, u32 len)
ret = parse_g2h_response(ct, msg, len);
break;
default:
- xe_gt_err(gt, "G2H channel broken on read, type=%u, reset required\n",
- type);
+ xe_gt_err(gt, "Unexpected G2H message type=%u\n", type);
CT_DEAD(ct, &ct->ctbs.g2h, PARSE_G2H_TYPE);
ret = -EOPNOTSUPP;
@@ -1808,6 +1883,7 @@ static int g2h_read(struct xe_guc_ct *ct, u32 *msg, bool fast_path)
xe_gt_assert(gt, xe_guc_ct_initialized(ct));
lockdep_assert_held(&ct->fast_lock);
+ /* Keep in sync with g2h_err_is_fatal() */
if (xe_device_wedged(xe))
return -ENOTRECOVERABLE;
@@ -1818,7 +1894,7 @@ static int g2h_read(struct xe_guc_ct *ct, u32 *msg, bool fast_path)
return -ECANCELED;
if (g2h->info.broken)
- return -EPIPE;
+ return -EPERM;
xe_gt_assert(gt, xe_guc_ct_enabled(ct));
@@ -1834,10 +1910,9 @@ static int g2h_read(struct xe_guc_ct *ct, u32 *msg, bool fast_path)
desc_status &= ~GUC_CTB_STATUS_DISABLED;
}
- if (desc_status) {
- xe_gt_err(gt, "CT read: non-zero status: %u\n", desc_status);
- goto corrupted;
- }
+ if (desc_status)
+ return ct_corrupted(ct, &ct->ctbs.g2h, ct_id(G2H_READ),
+ "read: non-zero status: %u\n", desc_status);
}
if (IS_ENABLED(CONFIG_DRM_XE_DEBUG)) {
@@ -1858,24 +1933,24 @@ static int g2h_read(struct xe_guc_ct *ct, u32 *msg, bool fast_path)
*
if (g2h->info.head != desc_head) {
desc_write(xe, g2h, status, desc_status | GUC_CTB_STATUS_MISMATCH);
- xe_gt_err(gt, "CT read: head was modified %u != %u\n",
- desc_head, g2h->info.head);
- goto corrupted;
+ return ct_corrupted(ct, &ct->ctbs.g2h, ct_id(G2H_READ),
+ "read: head was modified %u != %u\n",
+ desc_head, g2h->info.head);
}
*/
if (g2h->info.head > g2h->info.size) {
desc_write(xe, g2h, status, desc_status | GUC_CTB_STATUS_OVERFLOW);
- xe_gt_err(gt, "CT read: head out of range: %u vs %u\n",
- g2h->info.head, g2h->info.size);
- goto corrupted;
+ return ct_corrupted(ct, &ct->ctbs.g2h, ct_id(G2H_READ),
+ "read: head out of range: %u vs %u\n",
+ g2h->info.head, g2h->info.size);
}
if (desc_tail >= g2h->info.size) {
desc_write(xe, g2h, status, desc_status | GUC_CTB_STATUS_OVERFLOW);
- xe_gt_err(gt, "CT read: invalid tail offset %u >= %u)\n",
- desc_tail, g2h->info.size);
- goto corrupted;
+ return ct_corrupted(ct, &ct->ctbs.g2h, ct_id(G2H_READ),
+ "read: invalid tail offset %u >= %u)\n",
+ desc_tail, g2h->info.size);
}
}
@@ -1892,11 +1967,10 @@ static int g2h_read(struct xe_guc_ct *ct, u32 *msg, bool fast_path)
xe_map_memcpy_from(xe, msg, &g2h->cmds, sizeof(u32) * g2h->info.head,
sizeof(u32));
len = FIELD_GET(GUC_CTB_MSG_0_NUM_DWORDS, msg[0]) + GUC_CTB_MSG_MIN_LEN;
- if (len > avail) {
- xe_gt_err(gt, "G2H channel broken on read, avail=%d, len=%d, reset required\n",
- avail, len);
- goto corrupted;
- }
+ if (len > avail)
+ return ct_corrupted(ct, &ct->ctbs.g2h, ct_id(G2H_READ),
+ "G2H channel broken on read, avail=%d, len=%d\n",
+ avail, len);
head = (g2h->info.head + 1) % g2h->info.size;
avail = len - 1;
@@ -1942,10 +2016,6 @@ static int g2h_read(struct xe_guc_ct *ct, u32 *msg, bool fast_path)
action, len, g2h->info.head, tail);
return len;
-
-corrupted:
- CT_DEAD(ct, &ct->ctbs.g2h, G2H_READ);
- return -EPROTO;
}
static void g2h_fast_path(struct xe_guc_ct *ct, u32 *msg, u32 len)
@@ -2037,6 +2107,27 @@ static int dequeue_one_g2h(struct xe_guc_ct *ct)
return 1;
}
+/*
+ * Errors reported by dequeue_one_g2h() come in two flavours: either the channel
+ * is simply not available right now, which is expected and handled gracefully,
+ * or the channel state or the message itself is inconsistent, in which case the
+ * only way forward is to declare the CT dead and reset the GuC. Since the
+ * former is a short and well known list, check against that and treat anything
+ * else as fatal, so that new error codes don't silently escape the escalation.
+ */
+static bool g2h_err_is_fatal(int err)
+{
+ switch (err) {
+ case -ENOTRECOVERABLE: /* device wedged */
+ case -ENODEV: /* CT disabled */
+ case -ECANCELED: /* CT stopped */
+ case -EPERM: /* CT already declared broken */
+ return false;
+ default:
+ return true;
+ }
+}
+
static void receive_g2h(struct xe_guc_ct *ct)
{
bool ongoing;
@@ -2074,8 +2165,9 @@ static void receive_g2h(struct xe_guc_ct *ct)
ret = dequeue_one_g2h(ct);
mutex_unlock(&ct->lock);
- if (unlikely(ret == -EPROTO || ret == -EOPNOTSUPP)) {
- xe_gt_err(ct_to_gt(ct), "CT dequeue failed: %d\n", ret);
+ if (unlikely(ret < 0 && g2h_err_is_fatal(ret))) {
+ xe_gt_err(ct_to_gt(ct), "CT dequeue failed, forcing GT reset(%pe)\n",
+ ERR_PTR(ret));
CT_DEAD(ct, NULL, G2H_RECV);
kick_reset(ct);
}
--
2.55.0
^ permalink raw reply related [flat|nested] 14+ messages in thread* Re: [PATCH v4 4/7] drm/xe/guc: Cleanup error codes and handling for CT errors
2026-09-17 23:59 ` [PATCH v4 4/7] drm/xe/guc: Cleanup error codes and handling for CT errors Umesh Nerlige Ramappa
@ 2026-09-18 0:17 ` sashiko-bot
0 siblings, 0 replies; 14+ messages in thread
From: sashiko-bot @ 2026-09-18 0:17 UTC (permalink / raw)
To: Umesh Nerlige Ramappa; +Cc: intel-xe
Thank you for your contribution! Sashiko AI review found 3 potential issue(s) to consider:
- [High] drm/xe/guc: retry_failure() not updated for -EPERM breaks no_fail requests
- [Low] drm/xe/guc: build failure due to undeclared CT_DEAD_ reason codes
- [Low] drm/xe/guc: kernel-doc mismatch for xe_guc_ct_send return value
--
commit fcb132fd1e8cabdb5fd371a7a87d1bb78381dd54
Author: Umesh Nerlige Ramappa <umesh.nerlige.ramappa@intel.com>
Subject: drm/xe/guc: Cleanup error codes and handling for CT errors
This commit updates GuC CT error handling to return more specific error
codes instead of uniformly returning -EPROTO. It retains -EPROTO for
genuine protocol violations while adjusting others to reflect the actual
failure, and simplifies the reset handling logic.
> diff --git a/drivers/gpu/drm/xe/xe_guc_ct.c b/drivers/gpu/drm/xe/xe_guc_ct.c
> index 3353515aa8747..31ecddab30578 100644
> --- a/drivers/gpu/drm/xe/xe_guc_ct.c
> +++ b/drivers/gpu/drm/xe/xe_guc_ct.c
[ ... ]
> @@ -953,23 +1035,22 @@ static int h2g_write(struct xe_guc_ct *ct, const u32 *action, u32 len,
> u32 desc_status;
>
> desc_status = desc_read(xe, h2g, status);
> - if (desc_status) {
> - xe_gt_err(gt, "CT write: non-zero status: %u\n", desc_status);
> - goto corrupted;
> - }
> + if (desc_status)
> + return ct_corrupted(ct, &ct->ctbs.h2g, ct_id(H2G_WRITE),
> + "write: non-zero status: %u\n", desc_status);
[Severity: Low]
Does this code break the build when CONFIG_DRM_XE_DEBUG is disabled?
The macro ct_id() evaluates to CT_DEAD_##reason_code. If those enum
values are conditionally defined globally (such as inside an
#if IS_ENABLED(CONFIG_DRM_XE_DEBUG) block), using them here in h2g_write()
outside of any #if guards will cause undeclared identifier compilation
errors on non-debug builds.
[ ... ]
> @@ -1060,11 +1137,6 @@ static int __guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action,
> goto out;
> }
>
> - if (unlikely(ct->ctbs.h2g.info.broken)) {
> - ret = -EPIPE;
> - goto out;
> - }
> -
> if (ct->state == XE_GUC_CT_STATE_DISABLED) {
> ret = -ENODEV;
> goto out;
> @@ -1075,6 +1147,11 @@ static int __guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action,
> goto out;
> }
>
> + if (unlikely(ct->ctbs.h2g.info.broken)) {
> + ret = -EPERM;
> + goto out;
> + }
> +
[Severity: High]
Does this return code change break the retry logic for no-fail requests?
When the H2G CTB is broken, __guc_ct_send_locked() now returns -EPERM
instead of the legacy -EPIPE. However, retry_failure() still only checks
for -EPIPE:
if (!(ret == -EDEADLK || ret == -EPIPE || ret == -ENODEV))
return false;
Without checking for -EPERM, retry_failure() will return false, causing
no-fail messages (such as critical TLB invalidations) to fail immediately
instead of waiting for a GT reset and retrying.
[Severity: Low]
Also, does the kernel-doc for xe_guc_ct_send() need an update to reflect
this change?
The documentation for xe_guc_ct_send() still lists -EPIPE for a broken
CTB:
* -EPIPE: the H2G CTB is marked broken. The channel stays unusable until the
* CT is restarted, which clears the broken flag.
Since this patch changes the return value for a broken CTB to -EPERM,
this documentation might need to be updated to match.
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260917235923.1521112-9-umesh.nerlige.ramappa@intel.com?part=4
^ permalink raw reply [flat|nested] 14+ messages in thread
* [PATCH v4 5/7] drm/xe/uc: Report DMA failure using SIGID
2026-09-17 23:59 [PATCH v4 0/7] Use SIG_ID logs for GuC component Umesh Nerlige Ramappa
` (3 preceding siblings ...)
2026-09-17 23:59 ` [PATCH v4 4/7] drm/xe/guc: Cleanup error codes and handling for CT errors Umesh Nerlige Ramappa
@ 2026-09-17 23:59 ` Umesh Nerlige Ramappa
2026-09-18 0:07 ` sashiko-bot
2026-09-17 23:59 ` [PATCH v4 6/7] drm/xe/guc: Report major GuC failures " Umesh Nerlige Ramappa
` (2 subsequent siblings)
7 siblings, 1 reply; 14+ messages in thread
From: Umesh Nerlige Ramappa @ 2026-09-17 23:59 UTC (permalink / raw)
To: intel-xe
Cc: daniele.ceraolospurio, michal.wajdeczko, aravind.iddamsetty,
mallesh.koujalagi, alan.previn.teres.alexis, julia.filipchuk
From: Daniele Ceraolo Spurio <daniele.ceraolospurio@intel.com>
Convert the uC DMA failure error to use the xe_log_err() helper.
Signed-off-by: Daniele Ceraolo Spurio <daniele.ceraolospurio@intel.com>
Reviewed-by: Michal Wajdeczko <michal.wajdeczko@intel.com>
---
v2: add assert and simplify if/else
---
drivers/gpu/drm/xe/xe_uc_fw.c | 13 ++++++++++---
1 file changed, 10 insertions(+), 3 deletions(-)
diff --git a/drivers/gpu/drm/xe/xe_uc_fw.c b/drivers/gpu/drm/xe/xe_uc_fw.c
index e29878b255ba..ada39c1266a8 100644
--- a/drivers/gpu/drm/xe/xe_uc_fw.c
+++ b/drivers/gpu/drm/xe/xe_uc_fw.c
@@ -18,6 +18,7 @@
#include "xe_gt_sriov_vf.h"
#include "xe_gt_types.h"
#include "xe_guc.h"
+#include "xe_log.h"
#include "xe_map.h"
#include "xe_mmio.h"
#include "xe_module.h"
@@ -914,8 +915,12 @@ static int uc_fw_xfer(struct xe_uc_fw *uc_fw, u32 offset, u32 dma_flags)
int xe_uc_fw_upload(struct xe_uc_fw *uc_fw, u32 offset, u32 dma_flags)
{
struct xe_device *xe = uc_fw_to_xe(uc_fw);
+ struct xe_gt *gt = uc_fw_to_gt(uc_fw);
int err;
+ xe_gt_assert(gt, uc_fw->type == XE_UC_FW_TYPE_GUC ||
+ uc_fw->type == XE_UC_FW_TYPE_HUC);
+
/* make sure the status was cleared the last time we reset the uc */
xe_assert(xe, !xe_uc_fw_is_loaded(uc_fw));
@@ -931,9 +936,11 @@ int xe_uc_fw_upload(struct xe_uc_fw *uc_fw, u32 offset, u32 dma_flags)
return 0;
fail:
- drm_err(&xe->drm, "Failed to load %s firmware %s (%d)\n",
- xe_uc_fw_type_repr(uc_fw->type), uc_fw->path,
- err);
+ if (uc_fw->type == XE_UC_FW_TYPE_GUC)
+ xe_log_err(gt, GUC, err, "Failed to load firmware %s\n", uc_fw->path);
+ else
+ xe_log_err(gt, HUC, err, "Failed to load firmware %s\n", uc_fw->path);
+
xe_uc_fw_change_status(uc_fw, XE_UC_FIRMWARE_LOAD_FAIL);
return err;
}
--
2.55.0
^ permalink raw reply related [flat|nested] 14+ messages in thread* Re: [PATCH v4 5/7] drm/xe/uc: Report DMA failure using SIGID
2026-09-17 23:59 ` [PATCH v4 5/7] drm/xe/uc: Report DMA failure using SIGID Umesh Nerlige Ramappa
@ 2026-09-18 0:07 ` sashiko-bot
0 siblings, 0 replies; 14+ messages in thread
From: sashiko-bot @ 2026-09-18 0:07 UTC (permalink / raw)
To: Umesh Nerlige Ramappa; +Cc: intel-xe
Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [Medium] xe: missing conversion of DMA failure message to xe_log_err() in uc_fw_xfer()
--
commit 519565d5579d72f99be56f54099a18eadbbceb69
Author: Daniele Ceraolo Spurio <daniele.ceraolospurio@intel.com>
drm/xe/uc: Report DMA failure using SIGID
Convert the uC DMA failure error to use the xe_log_err() helper.
> diff --git a/drivers/gpu/drm/xe/xe_uc_fw.c b/drivers/gpu/drm/xe/xe_uc_fw.c
> index a8e6f18cc9b4a..21e2c1be08842 100644
> --- a/drivers/gpu/drm/xe/xe_uc_fw.c
> +++ b/drivers/gpu/drm/xe/xe_uc_fw.c
[ ... ]
> @@ -927,9 +932,11 @@ int xe_uc_fw_upload(struct xe_uc_fw *uc_fw, u32 offset, u32 dma_flags)
> return 0;
>
> fail:
> - drm_err(&xe->drm, "Failed to load %s firmware %s (%d)\n",
> - xe_uc_fw_type_repr(uc_fw->type), uc_fw->path,
> - err);
> + if (uc_fw->type == XE_UC_FW_TYPE_GUC)
> + xe_log_err(gt, GUC, err, "Failed to load firmware %s\n", uc_fw->path);
> + else
> + xe_log_err(gt, HUC, err, "Failed to load firmware %s\n", uc_fw->path);
> +
[Severity: Medium]
Does this commit match its description? The commit message explicitly claims
to "Convert the uC DMA failure error to use the xe_log_err() helper."
While it did convert the generic firmware load failure message here in
xe_uc_fw_upload(), it entirely missed the actual DMA-specific error message
in uc_fw_xfer():
if (ret)
drm_err(&xe->drm, "DMA for %s fw failed, DMA_CTRL=%u\n",
xe_uc_fw_type_repr(uc_fw->type), dma_ctrl);
If the uC firmware DMA transfer fails (e.g., timeout in xe_mmio_wait32), the
error is still logged using the old drm_err() format without the intended
SIGID logging.
> xe_uc_fw_change_status(uc_fw, XE_UC_FIRMWARE_LOAD_FAIL);
> return err;
> }
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260917235923.1521112-9-umesh.nerlige.ramappa@intel.com?part=5
^ permalink raw reply [flat|nested] 14+ messages in thread
* [PATCH v4 6/7] drm/xe/guc: Report major GuC failures using SIGID
2026-09-17 23:59 [PATCH v4 0/7] Use SIG_ID logs for GuC component Umesh Nerlige Ramappa
` (4 preceding siblings ...)
2026-09-17 23:59 ` [PATCH v4 5/7] drm/xe/uc: Report DMA failure using SIGID Umesh Nerlige Ramappa
@ 2026-09-17 23:59 ` Umesh Nerlige Ramappa
2026-09-17 23:59 ` [PATCH v4 7/7] drm/xe/guc: Report errors that cause a CT shutdown " Umesh Nerlige Ramappa
2026-09-18 0:58 ` ✗ CI.KUnit: failure for Use SIG_ID logs for GuC component (rev3) Patchwork
7 siblings, 0 replies; 14+ messages in thread
From: Umesh Nerlige Ramappa @ 2026-09-17 23:59 UTC (permalink / raw)
To: intel-xe
Cc: daniele.ceraolospurio, michal.wajdeczko, aravind.iddamsetty,
mallesh.koujalagi, alan.previn.teres.alexis, julia.filipchuk
From: Daniele Ceraolo Spurio <daniele.ceraolospurio@intel.com>
Convert errors during critical GuC flows (init, load, reset, suspend)
to use the xe_log_err() helper.
While at it, simplify the error log for the GuC load failure to use the
cached frequency values instead of fetching them again.
Signed-off-by: Daniele Ceraolo Spurio <daniele.ceraolospurio@intel.com>
Signed-off-by: Umesh Nerlige Ramappa <umesh.nerlige.ramappa@intel.com>
Reviewed-by: Michal Wajdeczko <michal.wajdeczko@intel.com>
---
v2: split guc_load_done changes to their own patch, use available ret
value for reset errors (Michal)
v3: Remove 'GuC' string from logs and cleanup logs (Michal)
v4: (Michal)
- Add s-o-b
- Differentiate Early vs later initialization messages
- Use %u for frequencies
- Remove unnecessary newline
---
drivers/gpu/drm/xe/xe_guc.c | 18 +++++++++---------
1 file changed, 9 insertions(+), 9 deletions(-)
diff --git a/drivers/gpu/drm/xe/xe_guc.c b/drivers/gpu/drm/xe/xe_guc.c
index 5285d4cecbc8..3bb4aea30a5c 100644
--- a/drivers/gpu/drm/xe/xe_guc.c
+++ b/drivers/gpu/drm/xe/xe_guc.c
@@ -777,7 +777,7 @@ int xe_guc_init_noalloc(struct xe_guc *guc)
return 0;
out:
- xe_gt_err(gt, "GuC init failed with %pe\n", ERR_PTR(ret));
+ xe_log_err(gt, GUC, ret, "Early initialization error\n");
return ret;
}
@@ -845,7 +845,7 @@ int xe_guc_init(struct xe_guc *guc)
return 0;
out:
- xe_gt_err(gt, "GuC init failed with %pe\n", ERR_PTR(ret));
+ xe_log_err(gt, GUC, ret, "Initialization error\n");
return ret;
}
@@ -998,15 +998,15 @@ int xe_guc_reset(struct xe_guc *guc)
ret = xe_mmio_wait32(mmio, GDRST, GRDOM_GUC, 0, 5000, &gdrst, false);
if (ret) {
- xe_gt_err(gt, "GuC reset timed out, GDRST=%#x\n", gdrst);
+ xe_log_err(gt, GUC, ret, "Reset timeout, GDRST=%#x\n", gdrst);
goto err_out;
}
guc_status = xe_mmio_read32(mmio, GUC_STATUS);
if (!(guc_status & GS_MIA_IN_RESET)) {
- xe_gt_err(gt, "GuC status: %#x, MIA core expected to be in reset\n",
- guc_status);
ret = -EIO;
+ xe_log_err(gt, GUC, ret, "MIA core not in reset, GUC_STATUS=%#x\n",
+ guc_status);
goto err_out;
}
@@ -1224,9 +1224,9 @@ static int guc_wait_ucode(struct xe_guc *guc)
cur_freq = xe_guc_pc_get_cur_freq_fw(guc_pc);
if (ret || load_result) {
- xe_gt_err(gt, "load failed: status = 0x%08X, time = %lldms, freq = %dMHz (req %dMHz)\n",
- status, delta_ms, xe_guc_pc_get_act_freq(guc_pc),
- xe_guc_pc_get_cur_freq_fw(guc_pc));
+ xe_log_err(gt, GUC, ret ?: load_result,
+ "Load failed, GUC_STATUS=%#x, time = %lldms, freq = %uMHz (req %uMHz)\n",
+ status, delta_ms, act_freq, cur_freq);
print_load_status_err(gt, status);
return ret ?: load_result;
@@ -1463,7 +1463,7 @@ int xe_guc_suspend(struct xe_guc *guc)
ret = xe_guc_softreset(guc);
if (ret) {
- xe_gt_err(gt, "GuC suspend failed: %pe\n", ERR_PTR(ret));
+ xe_log_err(gt, GUC, ret, "Suspend aborted\n");
return ret;
}
--
2.55.0
^ permalink raw reply related [flat|nested] 14+ messages in thread* [PATCH v4 7/7] drm/xe/guc: Report errors that cause a CT shutdown using SIGID
2026-09-17 23:59 [PATCH v4 0/7] Use SIG_ID logs for GuC component Umesh Nerlige Ramappa
` (5 preceding siblings ...)
2026-09-17 23:59 ` [PATCH v4 6/7] drm/xe/guc: Report major GuC failures " Umesh Nerlige Ramappa
@ 2026-09-17 23:59 ` Umesh Nerlige Ramappa
2026-09-18 0:08 ` sashiko-bot
2026-09-18 0:58 ` ✗ CI.KUnit: failure for Use SIG_ID logs for GuC component (rev3) Patchwork
7 siblings, 1 reply; 14+ messages in thread
From: Umesh Nerlige Ramappa @ 2026-09-17 23:59 UTC (permalink / raw)
To: intel-xe
Cc: daniele.ceraolospurio, michal.wajdeczko, aravind.iddamsetty,
mallesh.koujalagi, alan.previn.teres.alexis, julia.filipchuk
From: Daniele Ceraolo Spurio <daniele.ceraolospurio@intel.com>
Convert any errors that can cause the CT to be declared as dead to
use the xe_log_err() helper. Errors that are escalated to the callers
are left for the caller to report with SIGID if needed.
While at it, update some of the error messages to make what went wrong
clearer.
Signed-off-by: Daniele Ceraolo Spurio <daniele.ceraolospurio@intel.com>
Signed-off-by: Umesh Nerlige Ramappa <umesh.nerlige.ramappa@intel.com>
Assisted-by: Claude:claude-opus-5
---
v2:
- use different error codes and better messages (Michal)
v3:
- Drop GuC from log messages (Michal)
- Clean up log messages
v4: (Michal)
- Add s-o-b and move revision history to end
- s/EPROTO/EINVAL/ or FAST_REQ H2G fence failure
---
drivers/gpu/drm/xe/xe_guc_ct.c | 53 ++++++++++++++++++----------------
1 file changed, 28 insertions(+), 25 deletions(-)
diff --git a/drivers/gpu/drm/xe/xe_guc_ct.c b/drivers/gpu/drm/xe/xe_guc_ct.c
index 31ecddab3057..63d79987c5d2 100644
--- a/drivers/gpu/drm/xe/xe_guc_ct.c
+++ b/drivers/gpu/drm/xe/xe_guc_ct.c
@@ -30,6 +30,7 @@
#include "xe_guc_relay.h"
#include "xe_guc_submit.h"
#include "xe_guc_tlb_inval.h"
+#include "xe_log.h"
#include "xe_map.h"
#include "xe_page_reclaim.h"
#include "xe_pm.h"
@@ -737,7 +738,7 @@ static int __xe_guc_ct_start(struct xe_guc_ct *ct, bool needs_register)
return 0;
err_out:
- xe_gt_err(gt, "Failed to enable GuC CT (%pe)\n", ERR_PTR(err));
+ xe_log_err(gt, GUC, err, "CT: Failed to enable\n");
CT_DEAD(ct, NULL, SETUP);
return err;
@@ -861,8 +862,9 @@ static bool h2g_has_room(struct xe_guc_ct *ct, u32 cmd_len)
desc_write(xe, h2g, status, desc_status | GUC_CTB_STATUS_OVERFLOW);
- xe_gt_err(ct_to_gt(ct), "CT: invalid head offset %u >= %u)\n",
- h2g->info.head, h2g->info.size);
+ xe_log_err(ct_to_gt(ct), GUC, -EPIPE,
+ "CT: invalid head offset %u >= %u)\n",
+ h2g->info.head, h2g->info.size);
CT_DEAD(ct, h2g, H2G_HAS_ROOM);
return false;
}
@@ -931,12 +933,13 @@ static void __g2h_release_space(struct xe_guc_ct *ct, u32 g2h_len)
bad |= !ct->g2h_outstanding;
if (bad) {
- xe_gt_err(ct_to_gt(ct), "Invalid G2H release: %d + %d vs %d - %d -> %d vs %d, outstanding = %d!\n",
- ct->ctbs.g2h.info.space, g2h_len,
- ct->ctbs.g2h.info.size, ct->ctbs.g2h.info.resv_space,
- ct->ctbs.g2h.info.space + g2h_len,
- ct->ctbs.g2h.info.size - ct->ctbs.g2h.info.resv_space,
- ct->g2h_outstanding);
+ xe_log_err(ct_to_gt(ct), GUC, -ETOOMANYREFS,
+ "CT: Invalid G2H release: %d + %d vs %d - %d -> %d vs %d, outstanding = %d!\n",
+ ct->ctbs.g2h.info.space, g2h_len,
+ ct->ctbs.g2h.info.size, ct->ctbs.g2h.info.resv_space,
+ ct->ctbs.g2h.info.space + g2h_len,
+ ct->ctbs.g2h.info.size - ct->ctbs.g2h.info.resv_space,
+ ct->g2h_outstanding);
CT_DEAD(ct, &ct->ctbs.g2h, G2H_RELEASE);
return;
}
@@ -1004,7 +1007,7 @@ static int ct_corrupted(struct xe_guc_ct *ct, struct guc_ctb *ctb,
va_start(va_args, msg);
vaf.fmt = msg;
vaf.va = &va_args;
- xe_gt_err(ct_to_gt(ct), "GUC: CT: %pV", &vaf);
+ xe_log_err(ct_to_gt(ct), GUC, err, "CT: %pV", &vaf);
va_end(va_args);
ct_dead_capture(ct, ctb, reason_code);
@@ -1290,7 +1293,7 @@ static int guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len,
return ret;
broken:
- xe_gt_err(gt, "No forward progress on H2G\n");
+ xe_log_err(gt, GUC, -EDEADLK, "CT: No forward progress on H2G\n");
CT_DEAD(ct, &ct->ctbs.h2g, DEADLOCK);
return -EDEADLK;
@@ -1662,13 +1665,15 @@ static int parse_g2h_response(struct xe_guc_ct *ct, u32 *msg, u32 len)
*/
if (fence & CT_SEQNO_UNTRACKED) {
if (type == GUC_HXG_TYPE_RESPONSE_FAILURE)
- xe_gt_err(gt, "FAST_REQ H2G fence 0x%x failed! e=0x%x, h=%u\n",
- fence,
- FIELD_GET(GUC_HXG_FAILURE_MSG_0_ERROR, hxg[0]),
- FIELD_GET(GUC_HXG_FAILURE_MSG_0_HINT, hxg[0]));
+ xe_log_err(gt, GUC, -EINVAL,
+ "CT: FAST_REQ H2G fence 0x%x failed! e=0x%x, h=%u\n",
+ fence,
+ FIELD_GET(GUC_HXG_FAILURE_MSG_0_ERROR, hxg[0]),
+ FIELD_GET(GUC_HXG_FAILURE_MSG_0_HINT, hxg[0]));
else
- xe_gt_err(gt, "unexpected response %u for FAST_REQ H2G fence 0x%x!\n",
- type, fence);
+ xe_log_err(gt, GUC, -EPROTO,
+ "CT: unexpected response %u for FAST_REQ H2G fence 0x%x!\n",
+ type, fence);
fast_req_report(ct, fence);
@@ -1744,7 +1749,7 @@ static int parse_g2h_msg(struct xe_guc_ct *ct, u32 *msg, u32 len)
origin = FIELD_GET(GUC_HXG_MSG_0_ORIGIN, hxg[0]);
if (unlikely(origin != GUC_HXG_ORIGIN_GUC)) {
- xe_gt_err(gt, "Invalid G2H origin=%u\n", origin);
+ xe_log_err(gt, GUC, -EBADMSG, "CT: Invalid G2H origin=%u\n", origin);
CT_DEAD(ct, &ct->ctbs.g2h, PARSE_G2H_ORIGIN);
return -EPROTO;
@@ -1762,7 +1767,7 @@ static int parse_g2h_msg(struct xe_guc_ct *ct, u32 *msg, u32 len)
ret = parse_g2h_response(ct, msg, len);
break;
default:
- xe_gt_err(gt, "Unexpected G2H message type=%u\n", type);
+ xe_log_err(gt, GUC, -EOPNOTSUPP, "CT: Unexpected G2H message type %u\n", type);
CT_DEAD(ct, &ct->ctbs.g2h, PARSE_G2H_TYPE);
ret = -EOPNOTSUPP;
@@ -1862,8 +1867,8 @@ static int process_g2h_msg(struct xe_guc_ct *ct, u32 *msg, u32 len)
}
if (ret) {
- xe_gt_err(gt, "G2H action %#04x failed (%pe) len %u msg %*ph\n",
- action, ERR_PTR(ret), hxg_len, (int)sizeof(u32) * hxg_len, hxg);
+ xe_log_err(gt, GUC, ret, "CT: G2H action %#04x failed len %u msg %*ph\n",
+ action, hxg_len, (int)sizeof(u32) * hxg_len, hxg);
CT_DEAD(ct, NULL, PROCESS_FAILED);
}
@@ -2046,8 +2051,7 @@ static void g2h_fast_path(struct xe_guc_ct *ct, u32 *msg, u32 len)
}
if (ret) {
- xe_gt_err(gt, "G2H action 0x%04x failed (%pe)\n",
- action, ERR_PTR(ret));
+ xe_log_err(gt, GUC, ret, "CT: G2H action 0x%04x failed\n", action);
CT_DEAD(ct, NULL, FAST_G2H);
}
}
@@ -2166,8 +2170,7 @@ static void receive_g2h(struct xe_guc_ct *ct)
mutex_unlock(&ct->lock);
if (unlikely(ret < 0 && g2h_err_is_fatal(ret))) {
- xe_gt_err(ct_to_gt(ct), "CT dequeue failed, forcing GT reset(%pe)\n",
- ERR_PTR(ret));
+ xe_log_err(ct_to_gt(ct), GUC, ret, "CT: dequeue failed, forcing GT reset\n");
CT_DEAD(ct, NULL, G2H_RECV);
kick_reset(ct);
}
--
2.55.0
^ permalink raw reply related [flat|nested] 14+ messages in thread* Re: [PATCH v4 7/7] drm/xe/guc: Report errors that cause a CT shutdown using SIGID
2026-09-17 23:59 ` [PATCH v4 7/7] drm/xe/guc: Report errors that cause a CT shutdown " Umesh Nerlige Ramappa
@ 2026-09-18 0:08 ` sashiko-bot
0 siblings, 0 replies; 14+ messages in thread
From: sashiko-bot @ 2026-09-18 0:08 UTC (permalink / raw)
To: Umesh Nerlige Ramappa; +Cc: intel-xe
Thank you for your contribution! Sashiko AI review found 2 potential issue(s) to consider:
- [Medium] xe: missed xe_log_err conversion in guc crash handlers
- [Medium] xe: mismatched error codes returned in parse_g2h_msg and parse_g2h_response
--
commit 3134241c4666d211fb1d7e98e3ce95c50000c2e9
Author: Daniele Ceraolo Spurio <daniele.ceraolospurio@intel.com>
Subject: drm/xe/guc: Report errors that cause a CT shutdown using SIGID
This patch converts errors that result in a CT being declared dead to use
the xe_log_err helper, which supports SIGID reporting. Escalate-to-caller
errors remain unchanged so callers can apply SIGID if needed. Error
messages are also updated to be clearer about the failure.
> diff --git a/drivers/gpu/drm/xe/xe_guc_ct.c b/drivers/gpu/drm/xe/xe_guc_ct.c
> index 31ecddab30578..63d79987c5d28 100644
> --- a/drivers/gpu/drm/xe/xe_guc_ct.c
> +++ b/drivers/gpu/drm/xe/xe_guc_ct.c
[ ... ]
> @@ -1662,13 +1665,15 @@ static int parse_g2h_response(struct xe_guc_ct *ct, u32 *msg, u32 len)
> */
> if (fence & CT_SEQNO_UNTRACKED) {
> if (type == GUC_HXG_TYPE_RESPONSE_FAILURE)
> - xe_gt_err(gt, "FAST_REQ H2G fence 0x%x failed! e=0x%x, h=%u\n",
> - fence,
> - FIELD_GET(GUC_HXG_FAILURE_MSG_0_ERROR, hxg[0]),
> - FIELD_GET(GUC_HXG_FAILURE_MSG_0_HINT, hxg[0]));
> + xe_log_err(gt, GUC, -EINVAL,
> + "CT: FAST_REQ H2G fence 0x%x failed! e=0x%x, h=%u\n",
> + fence,
> + FIELD_GET(GUC_HXG_FAILURE_MSG_0_ERROR, hxg[0]),
> + FIELD_GET(GUC_HXG_FAILURE_MSG_0_HINT, hxg[0]));
> else
> - xe_gt_err(gt, "unexpected response %u for FAST_REQ H2G fence 0x%x!\n",
> - type, fence);
> + xe_log_err(gt, GUC, -EPROTO,
> + "CT: unexpected response %u for FAST_REQ H2G fence 0x%x!\n",
> + type, fence);
>
> fast_req_report(ct, fence);
>
> /* FIXME: W/A race in the GuC, will get in firmware soon */
> if (xe_gt_recovery_pending(gt))
> return 0;
>
> CT_DEAD(ct, NULL, PARSE_G2H_RESPONSE);
>
> return -EPROTO;
[Severity: Medium]
Does this cause misleading double-logging when receive_g2h() processes the
return value?
Here in parse_g2h_response(), we log the specific -EINVAL error code, but
return -EPROTO to the caller. When the caller receive_g2h() receives this,
it will log the failure a second time as a generic -EPROTO error since the
error codes don't match.
> }
>
[ ... ]
> @@ -1744,7 +1749,7 @@ static int parse_g2h_msg(struct xe_guc_ct *ct, u32 *msg, u32 len)
>
> origin = FIELD_GET(GUC_HXG_MSG_0_ORIGIN, hxg[0]);
> if (unlikely(origin != GUC_HXG_ORIGIN_GUC)) {
> - xe_gt_err(gt, "Invalid G2H origin=%u\n", origin);
> + xe_log_err(gt, GUC, -EBADMSG, "CT: Invalid G2H origin=%u\n", origin);
> CT_DEAD(ct, &ct->ctbs.g2h, PARSE_G2H_ORIGIN);
>
> return -EPROTO;
[Severity: Medium]
Similar to the check in parse_g2h_response(), does logging -EBADMSG here
but returning -EPROTO result in contradictory double-logging in the caller?
> }
>
[ ... ]
> @@ -1862,8 +1867,8 @@ static int process_g2h_msg(struct xe_guc_ct *ct, u32 *msg, u32 len)
> }
>
> if (ret) {
> - xe_gt_err(gt, "G2H action %#04x failed (%pe) len %u msg %*ph\n",
> - action, ERR_PTR(ret), hxg_len, (int)sizeof(u32) * hxg_len, hxg);
> + xe_log_err(gt, GUC, ret, "CT: G2H action %#04x failed len %u msg %*ph\n",
> + action, hxg_len, (int)sizeof(u32) * hxg_len, hxg);
> CT_DEAD(ct, NULL, PROCESS_FAILED);
> }
[Severity: Medium]
The commit message notes this converts any errors causing a CT shutdown to use
xe_log_err, but did it miss the GuC crash and exception handlers?
When a GuC crash or exception occurs, process_g2h_msg() calls
guc_handle_crash_dump_msg() and guc_handle_exception_msg(). Both of these
handlers call CT_DEAD directly and use the older xe_gt_err for logging:
guc_handle_crash_dump_msg()
xe_gt_err(ct_to_gt(ct), "GuC Crash dump notification\n");
CT_DEAD(ct, NULL, CRASH);
Should these be converted to xe_log_err as well to ensure these GuC crashes
receive a SIGID? As it stands, they are untagged, while the fallback generic
PROCESS_FAILED error above does get a SIGID.
--
Sashiko AI review · https://sashiko.dev/#/patchset/20260917235923.1521112-9-umesh.nerlige.ramappa@intel.com?part=7
^ permalink raw reply [flat|nested] 14+ messages in thread
* ✗ CI.KUnit: failure for Use SIG_ID logs for GuC component (rev3)
2026-09-17 23:59 [PATCH v4 0/7] Use SIG_ID logs for GuC component Umesh Nerlige Ramappa
` (6 preceding siblings ...)
2026-09-17 23:59 ` [PATCH v4 7/7] drm/xe/guc: Report errors that cause a CT shutdown " Umesh Nerlige Ramappa
@ 2026-09-18 0:58 ` Patchwork
7 siblings, 0 replies; 14+ messages in thread
From: Patchwork @ 2026-09-18 0:58 UTC (permalink / raw)
To: Umesh Nerlige Ramappa; +Cc: intel-xe
== Series Details ==
Series: Use SIG_ID logs for GuC component (rev3)
URL : https://patchwork.freedesktop.org/series/173177/
State : failure
== Summary ==
+ trap cleanup EXIT
+ kunitconfigs=('/kernel/drivers/gpu/tests/.kunitconfig' '/kernel/drivers/gpu/drm/xe/.kunitconfig' '/kernel/drivers/gpu/drm/tests/.kunitconfig' '/kernel/drivers/gpu/drm/ttm/tests/.kunitconfig' '/kernel/drivers/dma-buf/.kunitconfig')
+ for kcfg in "${kunitconfigs[@]}"
+ [[ ! -f /kernel/drivers/gpu/tests/.kunitconfig ]]
+ /kernel/tools/testing/kunit/kunit.py run --kunitconfig /kernel/drivers/gpu/tests/.kunitconfig
[00:57:41] Configuring KUnit Kernel ...
Generating .config ...
Populating config with:
$ make ARCH=um O=.kunit olddefconfig
[00:57:46] Building KUnit Kernel ...
Populating config with:
$ make ARCH=um O=.kunit olddefconfig
Building with:
$ make all compile_commands.json scripts_gdb ARCH=um O=.kunit --jobs=48
[00:58:06] Starting KUnit Kernel (1/1)...
[00:58:06] ============================================================
Running tests with:
$ .kunit/linux kunit.enable=1 mem=1G console=tty kunit_shutdown=halt
[00:58:06] ============= refcount_interrupt (4 subtests) ==============
[00:58:06] [PASSED] test_single_irq_change
[00:58:06] [PASSED] test_nested_irq_change
[00:58:06] [PASSED] test_multiple_irq_change
[00:58:06] [PASSED] test_irq_save
[00:58:06] =============== [PASSED] refcount_interrupt ================
[00:58:06] ================= gpu_buddy (14 subtests) ==================
[00:58:06] [PASSED] gpu_test_buddy_alloc_limit
[00:58:06] [PASSED] gpu_test_buddy_alloc_optimistic
[00:58:06] [PASSED] gpu_test_buddy_alloc_pessimistic
[00:58:06] [PASSED] gpu_test_buddy_alloc_pathological
[00:58:06] [PASSED] gpu_test_buddy_alloc_contiguous
[00:58:06] [PASSED] gpu_test_buddy_alloc_clear
[00:58:06] [PASSED] gpu_test_buddy_alloc_range
[00:58:07] [PASSED] gpu_test_buddy_alloc_range_bias
[00:58:08] [PASSED] gpu_test_buddy_fragmentation_performance
[00:58:08] [PASSED] gpu_test_buddy_dirty_tracker_performance
[00:58:08] [PASSED] gpu_test_buddy_alloc_exceeds_max_order
[00:58:08] [PASSED] gpu_test_buddy_offset_aligned_allocation
[00:58:08] [PASSED] gpu_test_buddy_subtree_offset_alignment_stress
[00:58:08] [PASSED] gpu_test_buddy_addr_to_block
[00:58:08] ==================== [PASSED] gpu_buddy ====================
[00:58:08] ============================================================
[00:58:08] Testing complete. Ran 18 tests: passed: 18
[00:58:08] Elapsed time: 26.713s total, 4.453s configuring, 20.342s building, 1.868s running
+ for kcfg in "${kunitconfigs[@]}"
+ [[ ! -f /kernel/drivers/gpu/drm/xe/.kunitconfig ]]
+ /kernel/tools/testing/kunit/kunit.py run --kunitconfig /kernel/drivers/gpu/drm/xe/.kunitconfig
[00:58:08] Configuring KUnit Kernel ...
Regenerating .config ...
Populating config with:
$ make ARCH=um O=.kunit olddefconfig
[00:58:10] Building KUnit Kernel ...
Populating config with:
$ make ARCH=um O=.kunit olddefconfig
Building with:
$ make all compile_commands.json scripts_gdb ARCH=um O=.kunit --jobs=48
ERROR:root:../drivers/gpu/drm/xe/xe_guc_ct.c: In function ‘__xe_guc_ct_start’:
../drivers/gpu/drm/xe/xe_guc_ct.c:129:41: error: implicit declaration of function ‘ct_dead_capture’ [-Werror=implicit-function-declaration]
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~~~~~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:742:9: note: in expansion of macro ‘CT_DEAD’
742 | CT_DEAD(ct, NULL, SETUP);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_SETUP’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:129:70: note: in expansion of macro ‘ct_id’
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:742:9: note: in expansion of macro ‘CT_DEAD’
742 | CT_DEAD(ct, NULL, SETUP);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: note: each undeclared identifier is reported only once for each function it appears in
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:129:70: note: in expansion of macro ‘ct_id’
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:742:9: note: in expansion of macro ‘CT_DEAD’
742 | CT_DEAD(ct, NULL, SETUP);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c: In function ‘h2g_has_room’:
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_H2G_HAS_ROOM’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:129:70: note: in expansion of macro ‘ct_id’
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:868:25: note: in expansion of macro ‘CT_DEAD’
868 | CT_DEAD(ct, h2g, H2G_HAS_ROOM);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c: In function ‘__g2h_release_space’:
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_G2H_RELEASE’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:129:70: note: in expansion of macro ‘ct_id’
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:943:17: note: in expansion of macro ‘CT_DEAD’
943 | CT_DEAD(ct, &ct->ctbs.g2h, G2H_RELEASE);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c: In function ‘h2g_write’:
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_H2G_WRITE’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:1042:65: note: in expansion of macro ‘ct_id’
1042 | return ct_corrupted(ct, &ct->ctbs.h2g, ct_id(H2G_WRITE),
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c: In function ‘guc_ct_send_locked’:
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_DEADLOCK’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:129:70: note: in expansion of macro ‘ct_id’
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:1297:9: note: in expansion of macro ‘CT_DEAD’
1297 | CT_DEAD(ct, &ct->ctbs.h2g, DEADLOCK);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c: In function ‘guc_handle_crash_dump_msg’:
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_CRASH’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:129:70: note: in expansion of macro ‘ct_id’
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:1632:9: note: in expansion of macro ‘CT_DEAD’
1632 | CT_DEAD(ct, NULL, CRASH);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c: In function ‘guc_handle_exception_msg’:
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_CRASH’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:129:70: note: in expansion of macro ‘ct_id’
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:1641:9: note: in expansion of macro ‘CT_DEAD’
1641 | CT_DEAD(ct, NULL, CRASH);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c: In function ‘parse_g2h_response’:
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_PARSE_G2H_RESPONSE’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:129:70: note: in expansion of macro ‘ct_id’
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:1684:17: note: in expansion of macro ‘CT_DEAD’
1684 | CT_DEAD(ct, NULL, PARSE_G2H_RESPONSE);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c: In function ‘parse_g2h_msg’:
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_PARSE_G2H_ORIGIN’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:129:70: note: in expansion of macro ‘ct_id’
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:1753:17: note: in expansion of macro ‘CT_DEAD’
1753 | CT_DEAD(ct, &ct->ctbs.g2h, PARSE_G2H_ORIGIN);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_PARSE_G2H_TYPE’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:129:70: note: in expansion of macro ‘ct_id’
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:1771:17: note: in expansion of macro ‘CT_DEAD’
1771 | CT_DEAD(ct, &ct->ctbs.g2h, PARSE_G2H_TYPE);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c: In function ‘process_g2h_msg’:
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_PROCESS_FAILED’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:129:70: note: in expansion of macro ‘ct_id’
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:1872:17: note: in expansion of macro ‘CT_DEAD’
1872 | CT_DEAD(ct, NULL, PROCESS_FAILED);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c: In function ‘g2h_read’:
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_G2H_READ’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:1919:65: note: in expansion of macro ‘ct_id’
1919 | return ct_corrupted(ct, &ct->ctbs.g2h, ct_id(G2H_READ),
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c: In function ‘g2h_fast_path’:
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_FAST_G2H’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:129:70: note: in expansion of macro ‘ct_id’
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:2055:17: note: in expansion of macro ‘CT_DEAD’
2055 | CT_DEAD(ct, NULL, FAST_G2H);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c: In function ‘receive_g2h’:
../drivers/gpu/drm/xe/xe_guc_ct.c:128:29: error: ‘CT_DEAD_G2H_RECV’ undeclared (first use in this function)
128 | #define ct_id(reason_code) CT_DEAD_##reason_code
| ^~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:129:70: note: in expansion of macro ‘ct_id’
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:2174:25: note: in expansion of macro ‘CT_DEAD’
2174 | CT_DEAD(ct, NULL, G2H_RECV);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c: At top level:
../drivers/gpu/drm/xe/xe_guc_ct.c:2468:13: warning: conflicting types for ‘ct_dead_capture’; have ‘void(struct xe_guc_ct *, struct guc_ctb *, u32)’ {aka ‘void(struct xe_guc_ct *, struct guc_ctb *, unsigned int)’}
2468 | static void ct_dead_capture(struct xe_guc_ct *ct, struct guc_ctb *ctb, u32 reason_code)
| ^~~~~~~~~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:2468:13: error: static declaration of ‘ct_dead_capture’ follows non-static declaration
../drivers/gpu/drm/xe/xe_guc_ct.c:129:41: note: previous implicit declaration of ‘ct_dead_capture’ with type ‘void(struct xe_guc_ct *, struct guc_ctb *, u32)’ {aka ‘void(struct xe_guc_ct *, struct guc_ctb *, unsigned int)’}
129 | #define CT_DEAD(ct, ctb, reason_code) ct_dead_capture((ct), (ctb), ct_id(reason_code))
| ^~~~~~~~~~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:742:9: note: in expansion of macro ‘CT_DEAD’
742 | CT_DEAD(ct, NULL, SETUP);
| ^~~~~~~
../drivers/gpu/drm/xe/xe_guc_ct.c:2468:13: warning: ‘ct_dead_capture’ defined but not used [-Wunused-function]
2468 | static void ct_dead_capture(struct xe_guc_ct *ct, struct guc_ctb *ctb, u32 reason_code)
| ^~~~~~~~~~~~~~~
cc1: some warnings being treated as errors
make[7]: *** [../scripts/Makefile.build:290: drivers/gpu/drm/xe/xe_guc_ct.o] Error 1
make[7]: *** Waiting for unfinished jobs....
make[6]: *** [../scripts/Makefile.build:551: drivers/gpu/drm/xe] Error 2
make[6]: *** Waiting for unfinished jobs....
make[5]: *** [../scripts/Makefile.build:551: drivers/gpu/drm] Error 2
make[4]: *** [../scripts/Makefile.build:551: drivers/gpu] Error 2
make[3]: *** [../scripts/Makefile.build:551: drivers] Error 2
make[3]: *** Waiting for unfinished jobs....
make[2]: *** [/kernel/Makefile:2229: .] Error 2
make[1]: *** [/kernel/Makefile:248: __sub-make] Error 2
make: *** [Makefile:248: __sub-make] Error 2
+ cleanup
++ stat -c %u:%g /kernel
+ chown -R 1003:1003 /kernel
^ permalink raw reply [flat|nested] 14+ messages in thread