Intel-XE Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: Umesh Nerlige Ramappa <umesh.nerlige.ramappa@intel.com>
To: intel-xe@lists.freedesktop.org
Cc: daniele.ceraolospurio@intel.com, michal.wajdeczko@intel.com,
	aravind.iddamsetty@intel.com, mallesh.koujalagi@intel.com,
	alan.previn.teres.alexis@intel.com, julia.filipchuk@intel.com
Subject: [PATCH v4 7/7] drm/xe/guc: Report errors that cause a CT shutdown using SIGID
Date: Thu, 17 Sep 2026 16:59:31 -0700	[thread overview]
Message-ID: <20260917235923.1521112-16-umesh.nerlige.ramappa@intel.com> (raw)
In-Reply-To: <20260917235923.1521112-9-umesh.nerlige.ramappa@intel.com>

From: Daniele Ceraolo Spurio <daniele.ceraolospurio@intel.com>

Convert any errors that can cause the CT to be declared as dead to
use the xe_log_err() helper. Errors that are escalated to the callers
are left for the caller to report with SIGID if needed.
While at it, update some of the error messages to make what went wrong
clearer.

Signed-off-by: Daniele Ceraolo Spurio <daniele.ceraolospurio@intel.com>
Signed-off-by: Umesh Nerlige Ramappa <umesh.nerlige.ramappa@intel.com>
Assisted-by: Claude:claude-opus-5
---
v2:
- use different error codes and better messages (Michal)
v3:
- Drop GuC from log messages (Michal)
- Clean up log messages
v4: (Michal)
- Add s-o-b and move revision history to end
- s/EPROTO/EINVAL/ or FAST_REQ H2G fence failure
---
 drivers/gpu/drm/xe/xe_guc_ct.c | 53 ++++++++++++++++++----------------
 1 file changed, 28 insertions(+), 25 deletions(-)

diff --git a/drivers/gpu/drm/xe/xe_guc_ct.c b/drivers/gpu/drm/xe/xe_guc_ct.c
index 31ecddab3057..63d79987c5d2 100644
--- a/drivers/gpu/drm/xe/xe_guc_ct.c
+++ b/drivers/gpu/drm/xe/xe_guc_ct.c
@@ -30,6 +30,7 @@
 #include "xe_guc_relay.h"
 #include "xe_guc_submit.h"
 #include "xe_guc_tlb_inval.h"
+#include "xe_log.h"
 #include "xe_map.h"
 #include "xe_page_reclaim.h"
 #include "xe_pm.h"
@@ -737,7 +738,7 @@ static int __xe_guc_ct_start(struct xe_guc_ct *ct, bool needs_register)
 	return 0;
 
 err_out:
-	xe_gt_err(gt, "Failed to enable GuC CT (%pe)\n", ERR_PTR(err));
+	xe_log_err(gt, GUC, err, "CT: Failed to enable\n");
 	CT_DEAD(ct, NULL, SETUP);
 
 	return err;
@@ -861,8 +862,9 @@ static bool h2g_has_room(struct xe_guc_ct *ct, u32 cmd_len)
 
 			desc_write(xe, h2g, status, desc_status | GUC_CTB_STATUS_OVERFLOW);
 
-			xe_gt_err(ct_to_gt(ct), "CT: invalid head offset %u >= %u)\n",
-				  h2g->info.head, h2g->info.size);
+			xe_log_err(ct_to_gt(ct), GUC, -EPIPE,
+				   "CT: invalid head offset %u >= %u)\n",
+				   h2g->info.head, h2g->info.size);
 			CT_DEAD(ct, h2g, H2G_HAS_ROOM);
 			return false;
 		}
@@ -931,12 +933,13 @@ static void __g2h_release_space(struct xe_guc_ct *ct, u32 g2h_len)
 	bad |= !ct->g2h_outstanding;
 
 	if (bad) {
-		xe_gt_err(ct_to_gt(ct), "Invalid G2H release: %d + %d vs %d - %d -> %d vs %d, outstanding = %d!\n",
-			  ct->ctbs.g2h.info.space, g2h_len,
-			  ct->ctbs.g2h.info.size, ct->ctbs.g2h.info.resv_space,
-			  ct->ctbs.g2h.info.space + g2h_len,
-			  ct->ctbs.g2h.info.size - ct->ctbs.g2h.info.resv_space,
-			  ct->g2h_outstanding);
+		xe_log_err(ct_to_gt(ct), GUC, -ETOOMANYREFS,
+			   "CT: Invalid G2H release: %d + %d vs %d - %d -> %d vs %d, outstanding = %d!\n",
+			   ct->ctbs.g2h.info.space, g2h_len,
+			   ct->ctbs.g2h.info.size, ct->ctbs.g2h.info.resv_space,
+			   ct->ctbs.g2h.info.space + g2h_len,
+			   ct->ctbs.g2h.info.size - ct->ctbs.g2h.info.resv_space,
+			   ct->g2h_outstanding);
 		CT_DEAD(ct, &ct->ctbs.g2h, G2H_RELEASE);
 		return;
 	}
@@ -1004,7 +1007,7 @@ static int ct_corrupted(struct xe_guc_ct *ct, struct guc_ctb *ctb,
 	va_start(va_args, msg);
 	vaf.fmt = msg;
 	vaf.va  = &va_args;
-	xe_gt_err(ct_to_gt(ct), "GUC: CT: %pV", &vaf);
+	xe_log_err(ct_to_gt(ct), GUC, err, "CT: %pV", &vaf);
 	va_end(va_args);
 
 	ct_dead_capture(ct, ctb, reason_code);
@@ -1290,7 +1293,7 @@ static int guc_ct_send_locked(struct xe_guc_ct *ct, const u32 *action, u32 len,
 	return ret;
 
 broken:
-	xe_gt_err(gt, "No forward progress on H2G\n");
+	xe_log_err(gt, GUC, -EDEADLK, "CT: No forward progress on H2G\n");
 	CT_DEAD(ct, &ct->ctbs.h2g, DEADLOCK);
 
 	return -EDEADLK;
@@ -1662,13 +1665,15 @@ static int parse_g2h_response(struct xe_guc_ct *ct, u32 *msg, u32 len)
 	 */
 	if (fence & CT_SEQNO_UNTRACKED) {
 		if (type == GUC_HXG_TYPE_RESPONSE_FAILURE)
-			xe_gt_err(gt, "FAST_REQ H2G fence 0x%x failed! e=0x%x, h=%u\n",
-				  fence,
-				  FIELD_GET(GUC_HXG_FAILURE_MSG_0_ERROR, hxg[0]),
-				  FIELD_GET(GUC_HXG_FAILURE_MSG_0_HINT, hxg[0]));
+			xe_log_err(gt, GUC, -EINVAL,
+				   "CT: FAST_REQ H2G fence 0x%x failed! e=0x%x, h=%u\n",
+				   fence,
+				   FIELD_GET(GUC_HXG_FAILURE_MSG_0_ERROR, hxg[0]),
+				   FIELD_GET(GUC_HXG_FAILURE_MSG_0_HINT, hxg[0]));
 		else
-			xe_gt_err(gt, "unexpected response %u for FAST_REQ H2G fence 0x%x!\n",
-				  type, fence);
+			xe_log_err(gt, GUC, -EPROTO,
+				   "CT: unexpected response %u for FAST_REQ H2G fence 0x%x!\n",
+				    type, fence);
 
 		fast_req_report(ct, fence);
 
@@ -1744,7 +1749,7 @@ static int parse_g2h_msg(struct xe_guc_ct *ct, u32 *msg, u32 len)
 
 	origin = FIELD_GET(GUC_HXG_MSG_0_ORIGIN, hxg[0]);
 	if (unlikely(origin != GUC_HXG_ORIGIN_GUC)) {
-		xe_gt_err(gt, "Invalid G2H origin=%u\n", origin);
+		xe_log_err(gt, GUC, -EBADMSG, "CT: Invalid G2H origin=%u\n", origin);
 		CT_DEAD(ct, &ct->ctbs.g2h, PARSE_G2H_ORIGIN);
 
 		return -EPROTO;
@@ -1762,7 +1767,7 @@ static int parse_g2h_msg(struct xe_guc_ct *ct, u32 *msg, u32 len)
 		ret = parse_g2h_response(ct, msg, len);
 		break;
 	default:
-		xe_gt_err(gt, "Unexpected G2H message type=%u\n", type);
+		xe_log_err(gt, GUC, -EOPNOTSUPP, "CT: Unexpected G2H message type %u\n", type);
 		CT_DEAD(ct, &ct->ctbs.g2h, PARSE_G2H_TYPE);
 
 		ret = -EOPNOTSUPP;
@@ -1862,8 +1867,8 @@ static int process_g2h_msg(struct xe_guc_ct *ct, u32 *msg, u32 len)
 	}
 
 	if (ret) {
-		xe_gt_err(gt, "G2H action %#04x failed (%pe) len %u msg %*ph\n",
-			  action, ERR_PTR(ret), hxg_len, (int)sizeof(u32) * hxg_len, hxg);
+		xe_log_err(gt, GUC, ret, "CT: G2H action %#04x failed len %u msg %*ph\n",
+			   action, hxg_len, (int)sizeof(u32) * hxg_len, hxg);
 		CT_DEAD(ct, NULL, PROCESS_FAILED);
 	}
 
@@ -2046,8 +2051,7 @@ static void g2h_fast_path(struct xe_guc_ct *ct, u32 *msg, u32 len)
 	}
 
 	if (ret) {
-		xe_gt_err(gt, "G2H action 0x%04x failed (%pe)\n",
-			  action, ERR_PTR(ret));
+		xe_log_err(gt, GUC, ret, "CT: G2H action 0x%04x failed\n", action);
 		CT_DEAD(ct, NULL, FAST_G2H);
 	}
 }
@@ -2166,8 +2170,7 @@ static void receive_g2h(struct xe_guc_ct *ct)
 		mutex_unlock(&ct->lock);
 
 		if (unlikely(ret < 0 && g2h_err_is_fatal(ret))) {
-			xe_gt_err(ct_to_gt(ct), "CT dequeue failed, forcing GT reset(%pe)\n",
-				  ERR_PTR(ret));
+			xe_log_err(ct_to_gt(ct), GUC, ret, "CT: dequeue failed, forcing GT reset\n");
 			CT_DEAD(ct, NULL, G2H_RECV);
 			kick_reset(ct);
 		}
-- 
2.55.0


  parent reply	other threads:[~2026-09-17 23:59 UTC|newest]

Thread overview: 14+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-17 23:59 [PATCH v4 0/7] Use SIG_ID logs for GuC component Umesh Nerlige Ramappa
2026-09-17 23:59 ` [PATCH v4 1/7] drm/xe/guc: Use different error codes for GuC load errors Umesh Nerlige Ramappa
2026-09-17 23:59 ` [PATCH v4 2/7] drm/xe/guc: Handle CRASH and EXCEPTION G2H with separate helpers Umesh Nerlige Ramappa
2026-09-18  0:08   ` sashiko-bot
2026-09-17 23:59 ` [PATCH v4 3/7] drm/xe/guc: Make ct_dead_capture available on non-debug config Umesh Nerlige Ramappa
2026-09-18  0:06   ` sashiko-bot
2026-09-17 23:59 ` [PATCH v4 4/7] drm/xe/guc: Cleanup error codes and handling for CT errors Umesh Nerlige Ramappa
2026-09-18  0:17   ` sashiko-bot
2026-09-17 23:59 ` [PATCH v4 5/7] drm/xe/uc: Report DMA failure using SIGID Umesh Nerlige Ramappa
2026-09-18  0:07   ` sashiko-bot
2026-09-17 23:59 ` [PATCH v4 6/7] drm/xe/guc: Report major GuC failures " Umesh Nerlige Ramappa
2026-09-17 23:59 ` Umesh Nerlige Ramappa [this message]
2026-09-18  0:08   ` [PATCH v4 7/7] drm/xe/guc: Report errors that cause a CT shutdown " sashiko-bot
2026-09-18  0:58 ` ✗ CI.KUnit: failure for Use SIG_ID logs for GuC component (rev3) Patchwork

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260917235923.1521112-16-umesh.nerlige.ramappa@intel.com \
    --to=umesh.nerlige.ramappa@intel.com \
    --cc=alan.previn.teres.alexis@intel.com \
    --cc=aravind.iddamsetty@intel.com \
    --cc=daniele.ceraolospurio@intel.com \
    --cc=intel-xe@lists.freedesktop.org \
    --cc=julia.filipchuk@intel.com \
    --cc=mallesh.koujalagi@intel.com \
    --cc=michal.wajdeczko@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox