Igt-dev Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: Junhua Shen <Junhua.Shen@amd.com>
To: <igt-dev@lists.freedesktop.org>
Cc: Vitaly Prosyak <vitaly.prosyak@amd.com>,
	Jesse Zhang <Jesse.Zhang@amd.com>,
	Sunil Khatri <sunil.khatri@amd.com>,
	Honglei Huang <honglei1.huang@amd.com>,
	Huang Rui <ray.huang@amd.com>, Yiru Ma <yiru.ma@amd.com>,
	Junhua Shen <Junhua.Shen@amd.com>
Subject: [PATCH i-g-t v2 4/6] lib/amdgpu: add CONST_FILL packet type and pass caller fill value through
Date: Wed, 9 Sep 2026 18:46:00 +0800	[thread overview]
Message-ID: <20260909104602.13807-5-Junhua.Shen@amd.com> (raw)
In-Reply-To: <20260909104602.13807-1-Junhua.Shen@amd.com>

Add a CMD_PACKET_CONST_FILL packet type with a per-packet size guard, and
pass the caller-supplied fill value (write_data/write_data_valid) through
write_linear and const_fill in the IP block builders.

Signed-off-by: Junhua Shen <Junhua.Shen@amd.com>
---
 lib/amdgpu/amd_command_submission.c | 74 +++++++++++++++++++++++++++++
 lib/amdgpu/amd_command_submission.h |  3 ++
 lib/amdgpu/amd_ip_blocks.c          | 20 +++++---
 lib/amdgpu/amd_ip_blocks.h          |  2 +
 4 files changed, 93 insertions(+), 6 deletions(-)

diff --git a/lib/amdgpu/amd_command_submission.c b/lib/amdgpu/amd_command_submission.c
index 6d55bd107..fd1915221 100644
--- a/lib/amdgpu/amd_command_submission.c
+++ b/lib/amdgpu/amd_command_submission.c
@@ -1087,6 +1087,8 @@ static void cmd_clear_packet_inputs(cmd_context_t *ctx)
 
 	rc->bo_mc = 0;
 	rc->bo_mc2 = 0;
+	rc->write_data = 0;
+	rc->write_data_valid = false;
 	rc->write_length = 0;
 
 	/* Drop the per-submit residency set. */
@@ -1211,6 +1213,38 @@ int cmd_submit_packet(cmd_context_t *ctx)
 	return 0;
 }
 
+static void cmd_assert_size_within_packet_limit(cmd_context_t *ctx, uint32_t size,
+						const char *op)
+{
+	struct amdgpu_dma_limits limits;
+	uint64_t max_bytes;
+
+	amdgpu_dma_limits_query(ctx->device, &limits);
+	max_bytes = amdgpu_dma_max_bytes(&limits, (unsigned int)ctx->ip_type);
+
+	igt_assert_f(size <= max_bytes,
+		     "%s size %u exceeds single-packet max %llu bytes for this IP; split the transfer\n",
+		     op, size, (unsigned long long)max_bytes);
+}
+
+/*
+ * WRITE_LINEAR inlines the payload directly into the PM4 stream as
+ * write_length/4 dwords (plus a small packet header), so its real limit is the
+ * PM4 buffer capacity, not the DMA engine's per-packet transfer maximum. Reject
+ * transfers whose inlined form would overrun pm4[] before the builder writes it.
+ */
+static void cmd_assert_inline_payload_fits(cmd_context_t *ctx, uint32_t size,
+					   const char *op)
+{
+	/* Upper bound on the header across the SDMA/GFX write_linear encodings. */
+	const uint32_t header_dw = 8;
+	uint64_t need_dw = (uint64_t)header_dw + (size / 4);
+
+	igt_assert_f(need_dw <= ctx->ring_ctx->pm4_size,
+		     "%s size %u needs %llu dw, exceeds PM4 buffer (%u dw); split the transfer\n",
+		     op, size, (unsigned long long)need_dw, ctx->ring_ctx->pm4_size);
+}
+
 int cmd_place_packet(cmd_context_t *ctx, const cmd_packet_params_t *params)
 {
 	int result = -EINVAL;
@@ -1232,7 +1266,10 @@ int cmd_place_packet(cmd_context_t *ctx, const cmd_packet_params_t *params)
 			ctx->ring_ctx->bo_mc = params->dst_addr;
 		igt_assert_f(ctx->ring_ctx->bo_mc,
 			     "WRITE_LINEAR requires a destination address\n");
+		cmd_assert_inline_payload_fits(ctx, params->size, "WRITE_LINEAR");
 		ctx->ring_ctx->write_length = params->size;
+		ctx->ring_ctx->write_data = params->data;
+		ctx->ring_ctx->write_data_valid = true;
 
 		/* Build PM4 packet */
 		result = ctx->ip_block->funcs->write_linear(ctx->ip_block->funcs,
@@ -1269,6 +1306,7 @@ int cmd_place_packet(cmd_context_t *ctx, const cmd_packet_params_t *params)
 			ctx->ring_ctx->bo_mc2 = params->dst_addr;
 		igt_assert_f(ctx->ring_ctx->bo_mc && ctx->ring_ctx->bo_mc2,
 			     "COPY_LINEAR requires source and destination addresses\n");
+		cmd_assert_size_within_packet_limit(ctx, params->size, "COPY_LINEAR");
 		ctx->ring_ctx->write_length = params->size;
 
 		result = ctx->ip_block->funcs->copy_linear(ctx->ip_block->funcs,
@@ -1276,6 +1314,27 @@ int cmd_place_packet(cmd_context_t *ctx, const cmd_packet_params_t *params)
 						   &ctx->ring_ctx->pm4_dw);
 		return result;
 
+	case CMD_PACKET_CONST_FILL:
+		if (!ctx->ip_block->funcs->const_fill)
+			return -ENOTSUP;
+
+		/* Destination operand comes from the caller; params->data is the
+		 * fill pattern written to every DWORD of the range.
+		 */
+		if (params->dst_addr)
+			ctx->ring_ctx->bo_mc = params->dst_addr;
+		igt_assert_f(ctx->ring_ctx->bo_mc,
+			     "CONST_FILL requires a destination address\n");
+		cmd_assert_size_within_packet_limit(ctx, params->size, "CONST_FILL");
+		ctx->ring_ctx->write_length = params->size;
+		ctx->ring_ctx->write_data = params->data;
+		ctx->ring_ctx->write_data_valid = true;
+
+		result = ctx->ip_block->funcs->const_fill(ctx->ip_block->funcs,
+							  ctx->ring_ctx,
+							  &ctx->ring_ctx->pm4_dw);
+		return result;
+
 	case CMD_PACKET_COPY_ATOMIC:
 		/* TODO: Implement atomic copy if supported */
 		igt_info("cmd_place_packet: ATOMIC_COPY not implemented\n");
@@ -1341,6 +1400,21 @@ int cmd_submit_copy_linear(cmd_context_t *ctx, uint64_t src_addr, uint64_t dst_a
 	return cmd_place_and_submit_packet(ctx, &params);
 }
 
+/**
+ * Convenience function for constant fill
+ */
+int cmd_submit_const_fill(cmd_context_t *ctx, uint64_t dst_addr, uint32_t size, uint32_t data)
+{
+	cmd_packet_params_t params = {
+		.type = CMD_PACKET_CONST_FILL,
+		.dst_addr = dst_addr,
+		.size = size,
+		.data = data,
+	};
+
+	return cmd_place_and_submit_packet(ctx, &params);
+}
+
 /**
  * Convenience function for atomic operation
  */
diff --git a/lib/amdgpu/amd_command_submission.h b/lib/amdgpu/amd_command_submission.h
index 8644f25ed..cf20ff457 100644
--- a/lib/amdgpu/amd_command_submission.h
+++ b/lib/amdgpu/amd_command_submission.h
@@ -36,6 +36,7 @@ typedef enum {
     CMD_PACKET_WRITE_ATOMIC,
     CMD_PACKET_COPY_LINEAR,
     CMD_PACKET_COPY_ATOMIC,
+    CMD_PACKET_CONST_FILL,
     CMD_PACKET_FENCE,
     CMD_PACKET_TIMESTAMP,
 } cmd_packet_type_t;
@@ -196,6 +197,8 @@ int cmd_submit_write_linear(cmd_context_t *ctx, uint64_t dst_addr, uint32_t size
 
 int cmd_submit_copy_linear(cmd_context_t *ctx, uint64_t src_addr, uint64_t dst_addr, uint32_t size);
 
+int cmd_submit_const_fill(cmd_context_t *ctx, uint64_t dst_addr, uint32_t size, uint32_t data);
+
 int cmd_submit_atomic(cmd_context_t *ctx, uint64_t dst_addr, uint32_t data);
 
 bool cmd_ring_available(amdgpu_device_handle device,
diff --git a/lib/amdgpu/amd_ip_blocks.c b/lib/amdgpu/amd_ip_blocks.c
index 7d31a582a..c6a189fdf 100644
--- a/lib/amdgpu/amd_ip_blocks.c
+++ b/lib/amdgpu/amd_ip_blocks.c
@@ -32,6 +32,8 @@ sdma_ring_write_linear(const struct amdgpu_ip_funcs *func,
 		       uint32_t *pm4_dw)
 {
 	uint32_t i, j;
+	uint32_t fill = ring_context->write_data_valid ?
+			ring_context->write_data : func->deadbeaf;
 
 	i = 0;
 	j = 0;
@@ -56,7 +58,7 @@ sdma_ring_write_linear(const struct amdgpu_ip_funcs *func,
 		ring_context->pm4[i++] = ring_context->write_length / 4;
 
 	while (j++ < ring_context->write_length / 4)
-		ring_context->pm4[i++] = func->deadbeaf;
+		ring_context->pm4[i++] = fill;
 
 	*pm4_dw = i;
 
@@ -156,20 +158,22 @@ sdma_ring_const_fill(const struct amdgpu_ip_funcs *func,
 		     uint32_t *pm4_dw)
 {
 	uint32_t i;
+	uint32_t fill = context->write_data_valid ?
+			context->write_data : func->deadbeaf;
 
 	i = 0;
 	if (func->family_id == AMDGPU_FAMILY_SI) {
 		context->pm4[i++] = SDMA_PACKET_SI(SDMA_OPCODE_CONSTANT_FILL_SI,
 						   0, 0, 0, context->write_length / 4);
 		context->pm4[i++] = lower_32_bits(context->bo_mc);
-		context->pm4[i++] = 0xdeadbeaf;
+		context->pm4[i++] = fill;
 		context->pm4[i++] = upper_32_bits(context->bo_mc) >> 16;
 	} else {
 		context->pm4[i++] = SDMA_PACKET(SDMA_OPCODE_CONSTANT_FILL, 0,
 						SDMA_CONSTANT_FILL_EXTRA_SIZE(2));
 		context->pm4[i++] = lower_32_bits(context->bo_mc);
 		context->pm4[i++] = upper_32_bits(context->bo_mc);
-		context->pm4[i++] = func->deadbeaf;
+		context->pm4[i++] = fill;
 
 		if (func->family_id >= AMDGPU_FAMILY_AI)
 			context->pm4[i++] = context->write_length - 1;
@@ -239,6 +243,8 @@ gfx_ring_write_linear(const struct amdgpu_ip_funcs *func,
 		      uint32_t *pm4_dw)
 {
 	uint32_t i, j;
+	uint32_t fill = ring_context->write_data_valid ?
+			ring_context->write_data : func->deadbeaf;
 
 	i = 0;
 	j = 0;
@@ -258,7 +264,7 @@ gfx_ring_write_linear(const struct amdgpu_ip_funcs *func,
 	ring_context->pm4[i++] = lower_32_bits(ring_context->bo_mc);
 	ring_context->pm4[i++] = upper_32_bits(ring_context->bo_mc);
 	while (j++ < ring_context->write_length / 4)
-		ring_context->pm4[i++] = func->deadbeaf;
+		ring_context->pm4[i++] = fill;
 
 	*pm4_dw = i;
 	if (ring_context->pm4_size > 0)
@@ -357,11 +363,13 @@ gfx_ring_const_fill(const struct amdgpu_ip_funcs *func,
 		     uint32_t *pm4_dw)
 {
 	uint32_t i;
+	uint32_t fill = ring_context->write_data_valid ?
+			ring_context->write_data : func->deadbeaf;
 
 	i = 0;
 	if (func->family_id == AMDGPU_FAMILY_SI) {
 		ring_context->pm4[i++] = PACKET3(PACKET3_DMA_DATA_SI, 4);
-		ring_context->pm4[i++] = func->deadbeaf;
+		ring_context->pm4[i++] = fill;
 		ring_context->pm4[i++] = PACKET3_DMA_DATA_SI_ENGINE(0) |
 					 PACKET3_DMA_DATA_SI_DST_SEL(0) |
 					 PACKET3_DMA_DATA_SI_SRC_SEL(2) |
@@ -375,7 +383,7 @@ gfx_ring_const_fill(const struct amdgpu_ip_funcs *func,
 					 PACKET3_DMA_DATA_DST_SEL(0) |
 					 PACKET3_DMA_DATA_SRC_SEL(2) |
 					 PACKET3_DMA_DATA_CP_SYNC;
-		ring_context->pm4[i++] = func->deadbeaf;
+		ring_context->pm4[i++] = fill;
 		ring_context->pm4[i++] = 0;
 		ring_context->pm4[i++] = lower_32_bits(ring_context->bo_mc);
 		ring_context->pm4[i++] = upper_32_bits(ring_context->bo_mc);
diff --git a/lib/amdgpu/amd_ip_blocks.h b/lib/amdgpu/amd_ip_blocks.h
index 37f38ed44..19d698e91 100644
--- a/lib/amdgpu/amd_ip_blocks.h
+++ b/lib/amdgpu/amd_ip_blocks.h
@@ -307,6 +307,8 @@ struct amdgpu_ring_context {
 	 */
 	uint64_t write_length;
 	uint64_t write_length2; /* IN: transfer size in bytes, second packet */
+	uint32_t write_data;    /* IN: payload for write_linear/const_fill */
+	bool write_data_valid;  /* IN: use write_data; else IP deadbeaf pattern */
 	uint32_t *pm4;		/* INTERNAL: packet buffer (submitter-owned, builder fills) */
 	uint32_t pm4_size;	/* IN: capacity of pm4[] in dwords (required, non-zero) */
 	bool secure;		/* IN: secure or not */
-- 
2.34.1


  parent reply	other threads:[~2026-09-09 10:51 UTC|newest]

Thread overview: 12+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-09 10:45 [PATCH i-g-t v2 0/6] lib/amdgpu: refactor cmd_context ownership and packet operands Junhua Shen
2026-09-09 10:45 ` [PATCH i-g-t v2 1/6] lib/amdgpu: document amdgpu_ring_context field contract Junhua Shen
2026-09-09 10:45 ` [PATCH i-g-t v2 2/6] lib/amdgpu: drop external BO from cmd_context Junhua Shen
2026-09-14  2:18   ` Zhang, Jesse(Jie)
2026-09-09 10:45 ` [PATCH i-g-t v2 3/6] lib/amdgpu: source packet operands from params and size the IB from pm4_size Junhua Shen
2026-09-09 10:46 ` Junhua Shen [this message]
2026-09-09 10:46 ` [PATCH i-g-t v2 5/6] tests/amdgpu: derive compute dispatch version from hw_ip_version_major Junhua Shen
2026-09-09 10:46 ` [PATCH i-g-t v2 6/6] tests/amdgpu: query KFD aperture node count before filling array Junhua Shen
2026-09-09 18:19 ` ✓ i915.CI.BAT: success for lib/amdgpu: refactor cmd_context ownership and packet operands (rev2) Patchwork
2026-09-09 18:20 ` ✓ Xe.CI.BAT: " Patchwork
2026-09-10  3:38 ` ✓ Xe.CI.FULL: " Patchwork
2026-09-10 11:49 ` ✗ i915.CI.Full: failure " Patchwork

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260909104602.13807-5-Junhua.Shen@amd.com \
    --to=junhua.shen@amd.com \
    --cc=Jesse.Zhang@amd.com \
    --cc=honglei1.huang@amd.com \
    --cc=igt-dev@lists.freedesktop.org \
    --cc=ray.huang@amd.com \
    --cc=sunil.khatri@amd.com \
    --cc=vitaly.prosyak@amd.com \
    --cc=yiru.ma@amd.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox