Intel-XE Archive on lore.kernel.org
 help / color / mirror / Atom feed
* [CI ONLY] drm/xe: Convert job preparation in 2 stages
@ 2026-09-25 17:42 Maarten Lankhorst
  2026-09-25 19:39 ` ✗ CI.checkpatch: warning for " Patchwork
                   ` (3 more replies)
  0 siblings, 4 replies; 5+ messages in thread
From: Maarten Lankhorst @ 2026-09-25 17:42 UTC (permalink / raw)
  To: intel-xe; +Cc: Maarten Lankhorst

Instead of writing the entire job at the time of emission, prepare it in
advance.

This allows us to calculate the size before we submit, in preparation of
allowing more jobs to run simultaneously.

Signed-off-by: Maarten Lankhorst <dev@lankhorst.se>
---
DO NOT REVIEW, this is sent as a patch in a 2-part series, but I'm testing
in advance if test failures still occur, whether they originate from this
patch the second patch.

drivers/gpu/drm/xe/xe_ring_ops.c        | 150 ++++++++++++++----------
 drivers/gpu/drm/xe/xe_ring_ops_types.h  |   3 +
 drivers/gpu/drm/xe/xe_sched_job.c       |   4 +
 drivers/gpu/drm/xe/xe_sched_job_types.h |   5 +
 4 files changed, 97 insertions(+), 65 deletions(-)

diff --git a/drivers/gpu/drm/xe/xe_ring_ops.c b/drivers/gpu/drm/xe/xe_ring_ops.c
index 3dd8cf4e21316..0b9f7018fabd2 100644
--- a/drivers/gpu/drm/xe/xe_ring_ops.c
+++ b/drivers/gpu/drm/xe/xe_ring_ops.c
@@ -330,15 +330,12 @@ static int emit_fake_watchdog(struct xe_lrc *lrc, u32 *dw, int i)
 }
 
 /* for engines that don't require any special HW handling (no EUs, no aux inval, etc) */
-static void __emit_job_gen12_simple(struct xe_sched_job *job, struct xe_lrc *lrc,
-				    u64 batch_addr, u32 *head, u32 seqno)
+static u32 __prepare_job_gen12_simple(struct xe_sched_job *job, struct xe_lrc *lrc,
+				      u32 *dw, u64 batch_addr, u32 seqno)
 {
-	u32 dw[MAX_JOB_SIZE_DW], i = 0;
-	u32 ppgtt_flag = get_ppgtt_flag(job);
+	u32 i = 0, ppgtt_flag = get_ppgtt_flag(job);
 	struct xe_gt *gt = job->q->gt;
 
-	*head = lrc->ring.tail;
-
 	if (job->ring_ops_force_reset)
 		i = emit_fake_watchdog(lrc, dw, i);
 
@@ -372,7 +369,7 @@ static void __emit_job_gen12_simple(struct xe_sched_job *job, struct xe_lrc *lrc
 
 	xe_gt_assert(gt, i <= MAX_JOB_SIZE_DW);
 
-	xe_lrc_write_ring(lrc, dw, i * sizeof(*dw));
+	return i * sizeof(*dw);
 }
 
 static bool has_aux_ccs(struct xe_device *xe)
@@ -389,16 +386,13 @@ static bool has_aux_ccs(struct xe_device *xe)
 	return !xe->info.has_flat_ccs;
 }
 
-static void __emit_job_gen12_video(struct xe_sched_job *job, struct xe_lrc *lrc,
-				   u64 batch_addr, u32 *head, u32 seqno)
+static u32 __prepare_job_gen12_video(struct xe_sched_job *job, struct xe_lrc *lrc,
+				   u32 *dw, u64 batch_addr, u32 seqno)
 {
-	u32 dw[MAX_JOB_SIZE_DW], i = 0;
-	u32 ppgtt_flag = get_ppgtt_flag(job);
+	u32 i = 0, ppgtt_flag = get_ppgtt_flag(job);
 	struct xe_gt *gt = job->q->gt;
 	struct xe_device *xe = gt_to_xe(gt);
 
-	*head = lrc->ring.tail;
-
 	if (job->ring_ops_force_reset)
 		i = emit_fake_watchdog(lrc, dw, i);
 
@@ -437,23 +431,20 @@ static void __emit_job_gen12_video(struct xe_sched_job *job, struct xe_lrc *lrc,
 
 	xe_gt_assert(gt, i <= MAX_JOB_SIZE_DW);
 
-	xe_lrc_write_ring(lrc, dw, i * sizeof(*dw));
+	return i * sizeof(*dw);
 }
 
-static void __emit_job_gen12_render_compute(struct xe_sched_job *job,
-					    struct xe_lrc *lrc,
-					    u64 batch_addr, u32 *head,
-					    u32 seqno)
+static u32 __prepare_job_gen12_render_compute(struct xe_sched_job *job,
+					      struct xe_lrc *lrc,
+					      u32 *dw, u64 batch_addr,
+					      u32 seqno)
 {
-	u32 dw[MAX_JOB_SIZE_DW], i = 0;
-	u32 ppgtt_flag = get_ppgtt_flag(job);
+	u32 i = 0, ppgtt_flag = get_ppgtt_flag(job);
 	struct xe_gt *gt = job->q->gt;
 	struct xe_device *xe = gt_to_xe(gt);
 	bool lacks_render = !(gt->info.engine_mask & XE_HW_ENGINE_RCS_MASK);
 	u32 mask_flags = 0;
 
-	*head = lrc->ring.tail;
-
 	if (job->ring_ops_force_reset)
 		i = emit_fake_watchdog(lrc, dw, i);
 
@@ -501,19 +492,17 @@ static void __emit_job_gen12_render_compute(struct xe_sched_job *job,
 
 	xe_gt_assert(gt, i <= MAX_JOB_SIZE_DW);
 
-	xe_lrc_write_ring(lrc, dw, i * sizeof(*dw));
+	return i * sizeof(*dw);
 }
 
-static void emit_migration_job_gen12(struct xe_sched_job *job,
-				     struct xe_lrc *lrc, u32 *head,
-				     u32 seqno)
+static u32 prepare_migration_job_gen12(struct xe_sched_job *job,
+				       struct xe_lrc *lrc, u32 *dw,
+				       u32 seqno)
 {
 	struct xe_gt *gt = job->q->gt;
 	struct xe_device *xe = gt_to_xe(gt);
 	u32 saddr = xe_lrc_start_seqno_ggtt_addr(lrc);
-	u32 dw[MAX_JOB_SIZE_DW], i = 0;
-
-	*head = lrc->ring.tail;
+	u32 i = 0;
 
 	xe_gt_assert(gt, !job->ring_ops_force_reset);
 
@@ -539,94 +528,125 @@ static void emit_migration_job_gen12(struct xe_sched_job *job,
 
 	xe_gt_assert(job->q->gt, i <= MAX_JOB_SIZE_DW);
 
-	xe_lrc_write_ring(lrc, dw, i * sizeof(*dw));
+	return i * sizeof(*dw);
 }
 
-static void emit_job_gen12_gsc(struct xe_sched_job *job)
+static void prepare_job_gen12_gsc(struct xe_sched_job *job)
 {
 	struct xe_gt *gt = job->q->gt;
 
-	xe_gt_assert(gt, job->q->width <= 1); /* no parallel submission for GSCCS */
+	xe_gt_assert(gt, job->q->width == 1); /* no parallel submission for GSCCS */
 
-	__emit_job_gen12_simple(job, job->q->lrc[0],
-				job->ptrs[0].batch_addr,
-				&job->ptrs[0].head,
-				xe_sched_job_lrc_seqno(job));
+	job->ptrs[0].job_size =
+		__prepare_job_gen12_simple(job, job->q->lrc[0],
+					   job->ptrs[0].dw,
+					   job->ptrs[0].batch_addr,
+					   xe_sched_job_lrc_seqno(job));
 }
 
-static void emit_job_gen12_copy(struct xe_sched_job *job)
+static void prepare_job_gen12_copy(struct xe_sched_job *job)
 {
 	int i;
 
 	if (xe_sched_job_is_migration(job->q)) {
-		emit_migration_job_gen12(job, job->q->lrc[0],
-					 &job->ptrs[0].head,
-					 xe_sched_job_lrc_seqno(job));
+		xe_gt_assert(job->q->gt, job->q->width == 1);
+
+		job->ptrs[0].job_size =
+			prepare_migration_job_gen12(job, job->q->lrc[0],
+						    job->ptrs[0].dw,
+						    xe_sched_job_lrc_seqno(job));
 		return;
 	}
 
-	for (i = 0; i < job->q->width; ++i)
-		__emit_job_gen12_simple(job, job->q->lrc[i],
-					job->ptrs[i].batch_addr,
-					&job->ptrs[i].head,
-					xe_sched_job_lrc_seqno(job));
+	for (i = 0; i < job->q->width; ++i) {
+		job->ptrs[i].job_size =
+			__prepare_job_gen12_simple(job, job->q->lrc[i],
+						   job->ptrs[i].dw,
+						   job->ptrs[i].batch_addr,
+						   xe_sched_job_lrc_seqno(job));
+		xe_gt_assert(job->q->gt, job->ptrs[i].job_size == job->ptrs[0].job_size);
+	}
 }
 
-static void emit_job_gen12_video(struct xe_sched_job *job)
+static void prepare_job_gen12_video(struct xe_sched_job *job)
 {
 	int i;
 
 	/* FIXME: Not doing parallel handshake for now */
-	for (i = 0; i < job->q->width; ++i)
-		__emit_job_gen12_video(job, job->q->lrc[i],
-				       job->ptrs[i].batch_addr,
-				       &job->ptrs[i].head,
-				       xe_sched_job_lrc_seqno(job));
+	for (i = 0; i < job->q->width; ++i) {
+		job->ptrs[i].job_size =
+			__prepare_job_gen12_video(job, job->q->lrc[i],
+						  job->ptrs[i].dw,
+						  job->ptrs[i].batch_addr,
+						  xe_sched_job_lrc_seqno(job));
+		xe_gt_assert(job->q->gt, job->ptrs[i].job_size == job->ptrs[0].job_size);
+	}
 }
 
-static void emit_job_gen12_render_compute(struct xe_sched_job *job)
+static void prepare_job_gen12_render_compute(struct xe_sched_job *job)
 {
 	int i;
 
-	for (i = 0; i < job->q->width; ++i)
-		__emit_job_gen12_render_compute(job, job->q->lrc[i],
-						job->ptrs[i].batch_addr,
-						&job->ptrs[i].head,
-						xe_sched_job_lrc_seqno(job));
+	for (i = 0; i < job->q->width; ++i) {
+		job->ptrs[i].job_size =
+			__prepare_job_gen12_render_compute(job, job->q->lrc[i],
+							   job->ptrs[i].dw,
+							   job->ptrs[i].batch_addr,
+							   xe_sched_job_lrc_seqno(job));
+		xe_gt_assert(job->q->gt, job->ptrs[i].job_size == job->ptrs[0].job_size);
+	}
+}
+
+static void emit_prepared_job(struct xe_sched_job *job)
+{
+	for (u32 i = 0; i < job->q->width; ++i) {
+		struct xe_lrc *lrc = job->q->lrc[i];
+
+		job->ptrs[i].head = lrc->ring.tail;
+		xe_lrc_write_ring(lrc, job->ptrs[i].dw, job->ptrs[i].job_size);
+	}
 }
 
 static const struct xe_ring_ops ring_ops_gen12_gsc = {
-	.emit_job = emit_job_gen12_gsc,
+	.prepare_job = prepare_job_gen12_gsc,
+	.emit_job = emit_prepared_job,
 };
 
 static const struct xe_ring_ops ring_ops_gen12_copy = {
-	.emit_job = emit_job_gen12_copy,
+	.prepare_job = prepare_job_gen12_copy,
+	.emit_job = emit_prepared_job,
 };
 
 static const struct xe_ring_ops ring_ops_gen12_video_decode = {
-	.emit_job = emit_job_gen12_video,
+	.prepare_job = prepare_job_gen12_video,
+	.emit_job = emit_prepared_job,
 };
 
 static const struct xe_ring_ops ring_ops_gen12_video_enhance = {
-	.emit_job = emit_job_gen12_video,
+	.prepare_job = prepare_job_gen12_video,
+	.emit_job = emit_prepared_job,
 };
 
 static const struct xe_ring_ops ring_ops_gen12_render_compute = {
-	.emit_job = emit_job_gen12_render_compute,
+	.prepare_job = prepare_job_gen12_render_compute,
+	.emit_job = emit_prepared_job,
 };
 
 static const struct xe_ring_ops auxccs_ring_ops_gen12_video_decode = {
-	.emit_job = emit_job_gen12_video,
+	.prepare_job = prepare_job_gen12_video,
+	.emit_job = emit_prepared_job,
 	.emit_aux_table_inv = emit_aux_table_inv_video_decode,
 };
 
 static const struct xe_ring_ops auxccs_ring_ops_gen12_video_enhance = {
-	.emit_job = emit_job_gen12_video,
+	.prepare_job = prepare_job_gen12_video,
+	.emit_job = emit_prepared_job,
 	.emit_aux_table_inv = emit_aux_table_inv_video_enhance,
 };
 
 static const struct xe_ring_ops auxccs_ring_ops_gen12_render_compute = {
-	.emit_job = emit_job_gen12_render_compute,
+	.prepare_job = prepare_job_gen12_render_compute,
+	.emit_job = emit_prepared_job,
 	.emit_aux_table_inv = emit_aux_table_inv_render_compute,
 };
 
diff --git a/drivers/gpu/drm/xe/xe_ring_ops_types.h b/drivers/gpu/drm/xe/xe_ring_ops_types.h
index 52ff96bc41004..5774ef6f3a265 100644
--- a/drivers/gpu/drm/xe/xe_ring_ops_types.h
+++ b/drivers/gpu/drm/xe/xe_ring_ops_types.h
@@ -18,6 +18,9 @@ struct xe_sched_job;
  * struct xe_ring_ops - Ring operations
  */
 struct xe_ring_ops {
+	/** @prepare_job: Write job to @job struct */
+	void (*prepare_job)(struct xe_sched_job *job);
+
 	/** @emit_job: Write job to ring */
 	void (*emit_job)(struct xe_sched_job *job);
 
diff --git a/drivers/gpu/drm/xe/xe_sched_job.c b/drivers/gpu/drm/xe/xe_sched_job.c
index a4fa00632a303..57ed6734d31d5 100644
--- a/drivers/gpu/drm/xe/xe_sched_job.c
+++ b/drivers/gpu/drm/xe/xe_sched_job.c
@@ -294,6 +294,10 @@ void xe_sched_job_push(struct xe_sched_job *job)
 {
 	xe_sched_job_get(job);
 	trace_xe_sched_job_exec(job);
+
+	if (!job->restore_replay)
+		job->q->ring_ops->prepare_job(job);
+
 	drm_sched_entity_push_job(&job->drm);
 	xe_sched_job_put(job);
 }
diff --git a/drivers/gpu/drm/xe/xe_sched_job_types.h b/drivers/gpu/drm/xe/xe_sched_job_types.h
index 0490b1247a6e9..5d4ea9dd09912 100644
--- a/drivers/gpu/drm/xe/xe_sched_job_types.h
+++ b/drivers/gpu/drm/xe/xe_sched_job_types.h
@@ -9,6 +9,7 @@
 #include <linux/kref.h>
 
 #include <drm/gpu_scheduler.h>
+#include "xe_ring_ops_types.h"
 
 struct xe_exec_queue;
 struct dma_fence;
@@ -29,6 +30,10 @@ struct xe_job_ptrs {
 	 * job was submitted
 	 */
 	u32 head;
+	/** @job_size: size of job in bytes */
+	u32 job_size;
+	/** @dw: the raw bytes to be added to lrc */
+	u32 dw[MAX_JOB_SIZE_DW];
 };
 
 /**
-- 
2.55.0


^ permalink raw reply related	[flat|nested] 5+ messages in thread

end of thread, other threads:[~2026-09-26  5:35 UTC | newest]

Thread overview: 5+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-25 17:42 [CI ONLY] drm/xe: Convert job preparation in 2 stages Maarten Lankhorst
2026-09-25 19:39 ` ✗ CI.checkpatch: warning for " Patchwork
2026-09-25 19:41 ` ✓ CI.KUnit: success " Patchwork
2026-09-25 20:28 ` ✓ Xe.CI.BAT: " Patchwork
2026-09-26  5:35 ` ✓ Xe.CI.FULL: " Patchwork

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox