* [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage
@ 2026-09-01 14:01 nishit.sharma
2026-09-01 18:13 ` Kamil Konieczny
2026-09-02 0:04 ` ✗ i915.CI.BAT: failure for tests/intel/xe_exec_reset: add GT reset fault injection stress coverage (rev4) Patchwork
0 siblings, 2 replies; 10+ messages in thread
From: nishit.sharma @ 2026-09-01 14:01 UTC (permalink / raw)
To: igt-dev, kamil.konieczny
From: Nishit Sharma <nishit.sharma@intel.com>
Inject a GT reset failure via the KMD fail_gt_reset debugfs hook, verify
the device wedges as expected, tolerate the resulting -ECANCELED errors in
the submitting threads, then recover by rebind and confirm the driver is
usable again.
Signed-off-by: Nishit Sharma <nishit.sharma@intel.com>
---
tests/intel/xe_exec_reset.c | 136 ++++++++++++++++++++++++++++++++++--
1 file changed, 131 insertions(+), 5 deletions(-)
diff --git a/tests/intel/xe_exec_reset.c b/tests/intel/xe_exec_reset.c
index 6eda71c32..5c5857b25 100644
--- a/tests/intel/xe_exec_reset.c
+++ b/tests/intel/xe_exec_reset.c
@@ -15,6 +15,8 @@
#include <fcntl.h>
#include "igt.h"
+#include "igt_device.h"
+#include "igt_kmod.h"
#include "igt_sysfs.h"
#include "lib/igt_syncobj.h"
#include "lib/intel_reg.h"
@@ -140,6 +142,7 @@ static void test_spin(int fd, struct drm_xe_engine_class_instance *eci,
#define DESTROY_VM_CTX_STRESS (0x1 << 20)
#define MIXED_ENGINE_STRESS (0x1 << 21)
#define PM_TRANSITION_STRESS (0x1 << 22)
+#define FAULT_INJECT_STRESS (0x1 << 23)
/**
* SUBTEST: %s-cat-error
@@ -760,7 +763,7 @@ static void submit_jobs(struct gt_thread_data *t)
uint32_t pressure_bos[PRESSURE_COUNT];
uint32_t *data;
int pressure_count;
- int i = 0;
+ int i = 0, exec_ret;
bo = xe_bo_create(fd, vm, bo_size, vram_if_possible(fd, t->gt),
DRM_XE_GEM_CREATE_FLAG_NEEDS_VISIBLE_VRAM);
@@ -794,8 +797,25 @@ static void submit_jobs(struct gt_thread_data *t)
continue;
}
- xe_exec(fd, &exec);
- xe_exec_queue_destroy(fd, exec.exec_queue_id);
+ /*
+ * Once an injected GT reset failure wedges the device, exec and
+ * queue teardown return -ECANCELED. That is the expected outcome
+ * for the fault-injection stress
+ */
+ if (t->flags & FAULT_INJECT_STRESS) {
+ struct drm_xe_exec_queue_destroy destroy = {
+ .exec_queue_id = exec.exec_queue_id,
+ };
+
+ exec_ret = __xe_exec(fd, &exec);
+ igt_assert_f(exec_ret == 0 || exec_ret == -ECANCELED,
+ "exec returned unexpected error %d (expected 0 or -ECANCELED)\n",
+ exec_ret);
+ igt_ioctl(fd, DRM_IOCTL_XE_EXEC_QUEUE_DESTROY, &destroy);
+ } else {
+ xe_exec(fd, &exec);
+ xe_exec_queue_destroy(fd, exec.exec_queue_id);
+ }
(*t->num_submit)++;
if ((t->flags & MEM_PRESSURE_STRESS) && !(i % 128)) {
@@ -829,8 +849,20 @@ static void submit_jobs(struct gt_thread_data *t)
pressure_bo_destroy(fd, pressure_bos, PRESSURE_COUNT);
munmap(data, bo_size);
- gem_close(fd, bo);
- xe_vm_destroy(fd, vm);
+ /*
+ * On a device wedged by injected GT reset failures, BO close and VM
+ * destroy also return -ECANCELED.
+ */
+ if (t->flags & FAULT_INJECT_STRESS) {
+ struct drm_gem_close close_bo = { .handle = bo };
+ struct drm_xe_vm_destroy vm_destroy = { .vm_id = vm };
+
+ igt_ioctl(fd, DRM_IOCTL_GEM_CLOSE, &close_bo);
+ igt_ioctl(fd, DRM_IOCTL_XE_VM_DESTROY, &vm_destroy);
+ } else {
+ gem_close(fd, bo);
+ xe_vm_destroy(fd, vm);
+ }
}
static void *gt_reset_thread(void *data)
@@ -850,6 +882,60 @@ static void *gt_reset_thread(void *data)
return NULL;
}
+static int fault_inject_fd = -1;
+
+static void gt_reset_disable_fault_injection(int sig)
+{
+ if (fault_inject_fd < 0)
+ return;
+
+ igt_debugfs_write(fault_inject_fd, "fail_gt_reset/probability", "0");
+ igt_debugfs_write(fault_inject_fd, "fail_gt_reset/times", "1");
+ fault_inject_fd = -1;
+}
+
+static void gt_reset_enable_fault_injection(int fd)
+{
+ static bool exit_handler_installed;
+
+ fault_inject_fd = fd;
+
+ if (!exit_handler_installed) {
+ igt_install_exit_handler(gt_reset_disable_fault_injection);
+ exit_handler_installed = true;
+ }
+
+ igt_debugfs_write(fd, "fail_gt_reset/probability", "100");
+ igt_debugfs_write(fd, "fail_gt_reset/times", "2");
+}
+
+static void gt_reset_fault_injection_forget_fd(int fd)
+{
+ if (fault_inject_fd == fd)
+ fault_inject_fd = -1;
+}
+
+static int try_vm_create(int fd)
+{
+ struct drm_xe_vm_create create = { 0 };
+ int err = 0;
+
+ if (igt_ioctl(fd, DRM_IOCTL_XE_VM_CREATE, &create))
+ err = -errno;
+ else
+ xe_vm_destroy(fd, create.vm_id);
+
+ return err;
+}
+
+static void ignore_gt_reset_fault_dmesg(void)
+{
+ igt_emit_ignore_dmesg_regex("reset failed \\(-ECANCELED\\)"
+ "|declared device .* as wedged"
+ "|GPU HANG"
+ "|Failed to reset");
+}
+
/**
* SUBTEST: gt-reset-stress
* Description: Stress GT reset
@@ -891,6 +977,9 @@ static void *gt_reset_thread(void *data)
* Description: Test GT reset while long spinner workload is active
* Test category: stress test
*
+ * SUBTEST: gt-reset-fault-injection
+ * Description: Stress concurrent GT resets and job submissions with GT reset failures injected via debugfs
+ * Test category: fault injection
*/
static void
gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
@@ -907,6 +996,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
pthread_mutex_init(&mutex, 0);
pthread_cond_init(&cond, 0);
+ if (flags & FAULT_INJECT_STRESS)
+ gt_reset_enable_fault_injection(fd);
+
for (i = 0; i < n_threads; ++i) {
threads[i].mutex = &mutex;
threads[i].cond = &cond;
@@ -940,6 +1032,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
igt_info("number of resets %d, submissions %d, submit fails %d vm_recreate %d\n",
num_reset, num_submit, num_submit_fail, num_vm_recreate);
+ if (flags & FAULT_INJECT_STRESS)
+ gt_reset_disable_fault_injection(0);
+
igt_assert_neq(num_reset, 0);
igt_assert_neq(num_submit, 0);
free(threads);
@@ -1122,6 +1217,7 @@ int igt_main()
int gt;
int class;
int fd;
+ char pci_slot[NAME_MAX];
igt_fixture()
fd = drm_open_driver(DRIVER_XE);
@@ -1326,6 +1422,36 @@ int igt_main()
break;
}
+ igt_subtest("gt-reset-fault-injection") {
+ igt_require_f(igt_debugfs_exists(fd, "fail_gt_reset/probability",
+ O_RDWR),
+ "GT reset fault injection not available; "
+ "CONFIG_DRM_XE_KUNIT_TEST/fault-injection must be "
+ "enabled in the KMD\n");
+
+ igt_device_get_pci_slot_name(fd, pci_slot);
+ ignore_gt_reset_fault_dmesg();
+
+ gt_reset(fd, 0, 8, 2, FAULT_INJECT_STRESS);
+
+ igt_assert_f(try_vm_create(fd) != 0,
+ "Device did not wedge after injected GT reset failure\n");
+
+ /*
+ * The device is about to be rebind, which recreates the debugfs
+ * fault-injection state from scratch - there is nothing to restore
+ * across the rebind. Dropping now-stale fd so the exit handler
+ * won't touch a closed descriptor.
+ */
+ gt_reset_fault_injection_forget_fd(fd);
+ drm_close_driver(fd);
+ igt_kmod_rebind("xe", pci_slot);
+ fd = drm_open_driver(DRIVER_XE);
+
+ igt_assert_f(try_vm_create(fd) == 0,
+ "Device not functional after rebind recovery\n");
+ }
+
igt_subtest("gt-mocs-reset")
xe_for_each_gt(fd, gt)
gt_mocs_reset(fd, gt);
--
2.43.0
^ permalink raw reply related [flat|nested] 10+ messages in thread* Re: [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage
2026-09-01 14:01 [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage nishit.sharma
@ 2026-09-01 18:13 ` Kamil Konieczny
2026-09-02 0:04 ` ✗ i915.CI.BAT: failure for tests/intel/xe_exec_reset: add GT reset fault injection stress coverage (rev4) Patchwork
1 sibling, 0 replies; 10+ messages in thread
From: Kamil Konieczny @ 2026-09-01 18:13 UTC (permalink / raw)
To: nishit.sharma; +Cc: igt-dev, kamil.konieczny
Hi nishit.sharma,
On 2026-09-01 at 14:01:31 +0000, nishit.sharma@intel.com wrote:
> From: Nishit Sharma <nishit.sharma@intel.com>
>
> Inject a GT reset failure via the KMD fail_gt_reset debugfs hook, verify
> the device wedges as expected, tolerate the resulting -ECANCELED errors in
> the submitting threads, then recover by rebind and confirm the driver is
> usable again.
>
> Signed-off-by: Nishit Sharma <nishit.sharma@intel.com>
> ---
> tests/intel/xe_exec_reset.c | 136 ++++++++++++++++++++++++++++++++++--
> 1 file changed, 131 insertions(+), 5 deletions(-)
>
> diff --git a/tests/intel/xe_exec_reset.c b/tests/intel/xe_exec_reset.c
> index 6eda71c32..5c5857b25 100644
> --- a/tests/intel/xe_exec_reset.c
> +++ b/tests/intel/xe_exec_reset.c
> @@ -15,6 +15,8 @@
> #include <fcntl.h>
>
> #include "igt.h"
> +#include "igt_device.h"
> +#include "igt_kmod.h"
> #include "igt_sysfs.h"
> #include "lib/igt_syncobj.h"
> #include "lib/intel_reg.h"
> @@ -140,6 +142,7 @@ static void test_spin(int fd, struct drm_xe_engine_class_instance *eci,
> #define DESTROY_VM_CTX_STRESS (0x1 << 20)
> #define MIXED_ENGINE_STRESS (0x1 << 21)
> #define PM_TRANSITION_STRESS (0x1 << 22)
> +#define FAULT_INJECT_STRESS (0x1 << 23)
>
> /**
> * SUBTEST: %s-cat-error
> @@ -760,7 +763,7 @@ static void submit_jobs(struct gt_thread_data *t)
> uint32_t pressure_bos[PRESSURE_COUNT];
> uint32_t *data;
> int pressure_count;
> - int i = 0;
> + int i = 0, exec_ret;
>
> bo = xe_bo_create(fd, vm, bo_size, vram_if_possible(fd, t->gt),
> DRM_XE_GEM_CREATE_FLAG_NEEDS_VISIBLE_VRAM);
> @@ -794,8 +797,25 @@ static void submit_jobs(struct gt_thread_data *t)
> continue;
> }
>
> - xe_exec(fd, &exec);
> - xe_exec_queue_destroy(fd, exec.exec_queue_id);
> + /*
> + * Once an injected GT reset failure wedges the device, exec and
> + * queue teardown return -ECANCELED. That is the expected outcome
> + * for the fault-injection stress
> + */
> + if (t->flags & FAULT_INJECT_STRESS) {
> + struct drm_xe_exec_queue_destroy destroy = {
> + .exec_queue_id = exec.exec_queue_id,
> + };
> +
> + exec_ret = __xe_exec(fd, &exec);
> + igt_assert_f(exec_ret == 0 || exec_ret == -ECANCELED,
> + "exec returned unexpected error %d (expected 0 or -ECANCELED)\n",
> + exec_ret);
> + igt_ioctl(fd, DRM_IOCTL_XE_EXEC_QUEUE_DESTROY, &destroy);
> + } else {
> + xe_exec(fd, &exec);
> + xe_exec_queue_destroy(fd, exec.exec_queue_id);
> + }
> (*t->num_submit)++;
>
> if ((t->flags & MEM_PRESSURE_STRESS) && !(i % 128)) {
> @@ -829,8 +849,20 @@ static void submit_jobs(struct gt_thread_data *t)
> pressure_bo_destroy(fd, pressure_bos, PRESSURE_COUNT);
>
> munmap(data, bo_size);
> - gem_close(fd, bo);
> - xe_vm_destroy(fd, vm);
> + /*
> + * On a device wedged by injected GT reset failures, BO close and VM
> + * destroy also return -ECANCELED.
> + */
> + if (t->flags & FAULT_INJECT_STRESS) {
> + struct drm_gem_close close_bo = { .handle = bo };
> + struct drm_xe_vm_destroy vm_destroy = { .vm_id = vm };
> +
> + igt_ioctl(fd, DRM_IOCTL_GEM_CLOSE, &close_bo);
> + igt_ioctl(fd, DRM_IOCTL_XE_VM_DESTROY, &vm_destroy);
> + } else {
> + gem_close(fd, bo);
> + xe_vm_destroy(fd, vm);
> + }
> }
>
> static void *gt_reset_thread(void *data)
> @@ -850,6 +882,60 @@ static void *gt_reset_thread(void *data)
> return NULL;
> }
>
> +static int fault_inject_fd = -1;
> +
> +static void gt_reset_disable_fault_injection(int sig)
> +{
> + if (fault_inject_fd < 0)
> + return;
> +
> + igt_debugfs_write(fault_inject_fd, "fail_gt_reset/probability", "0");
> + igt_debugfs_write(fault_inject_fd, "fail_gt_reset/times", "1");
> + fault_inject_fd = -1;
> +}
> +
> +static void gt_reset_enable_fault_injection(int fd)
> +{
> + static bool exit_handler_installed;
> +
> + fault_inject_fd = fd;
> +
> + if (!exit_handler_installed) {
> + igt_install_exit_handler(gt_reset_disable_fault_injection);
> + exit_handler_installed = true;
> + }
> +
> + igt_debugfs_write(fd, "fail_gt_reset/probability", "100");
> + igt_debugfs_write(fd, "fail_gt_reset/times", "2");
> +}
> +
> +static void gt_reset_fault_injection_forget_fd(int fd)
> +{
> + if (fault_inject_fd == fd)
> + fault_inject_fd = -1;
> +}
Why new function? imho call disable just before closing fd,
see below.
> +
> +static int try_vm_create(int fd)
> +{
> + struct drm_xe_vm_create create = { 0 };
> + int err = 0;
> +
> + if (igt_ioctl(fd, DRM_IOCTL_XE_VM_CREATE, &create))
> + err = -errno;
> + else
> + xe_vm_destroy(fd, create.vm_id);
> +
> + return err;
> +}
> +
> +static void ignore_gt_reset_fault_dmesg(void)
> +{
> + igt_emit_ignore_dmesg_regex("reset failed \\(-ECANCELED\\)"
> + "|declared device .* as wedged"
> + "|GPU HANG"
> + "|Failed to reset");
> +}
> +
> /**
> * SUBTEST: gt-reset-stress
> * Description: Stress GT reset
> @@ -891,6 +977,9 @@ static void *gt_reset_thread(void *data)
> * Description: Test GT reset while long spinner workload is active
> * Test category: stress test
> *
> + * SUBTEST: gt-reset-fault-injection
> + * Description: Stress concurrent GT resets and job submissions with GT reset failures injected via debugfs
> + * Test category: fault injection
> */
> static void
> gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
> @@ -907,6 +996,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
> pthread_mutex_init(&mutex, 0);
> pthread_cond_init(&cond, 0);
>
> + if (flags & FAULT_INJECT_STRESS)
> + gt_reset_enable_fault_injection(fd);
> +
> for (i = 0; i < n_threads; ++i) {
> threads[i].mutex = &mutex;
> threads[i].cond = &cond;
> @@ -940,6 +1032,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
> igt_info("number of resets %d, submissions %d, submit fails %d vm_recreate %d\n",
> num_reset, num_submit, num_submit_fail, num_vm_recreate);
>
> + if (flags & FAULT_INJECT_STRESS)
> + gt_reset_disable_fault_injection(0);
> +
> igt_assert_neq(num_reset, 0);
> igt_assert_neq(num_submit, 0);
> free(threads);
> @@ -1122,6 +1217,7 @@ int igt_main()
> int gt;
> int class;
> int fd;
> + char pci_slot[NAME_MAX];
>
> igt_fixture()
> fd = drm_open_driver(DRIVER_XE);
> @@ -1326,6 +1422,36 @@ int igt_main()
> break;
> }
>
> + igt_subtest("gt-reset-fault-injection") {
> + igt_require_f(igt_debugfs_exists(fd, "fail_gt_reset/probability",
> + O_RDWR),
> + "GT reset fault injection not available; "
> + "CONFIG_DRM_XE_KUNIT_TEST/fault-injection must be "
> + "enabled in the KMD\n");
> +
> + igt_device_get_pci_slot_name(fd, pci_slot);
> + ignore_gt_reset_fault_dmesg();
> +
> + gt_reset(fd, 0, 8, 2, FAULT_INJECT_STRESS);
> +
> + igt_assert_f(try_vm_create(fd) != 0,
> + "Device did not wedge after injected GT reset failure\n");
> +
> + /*
> + * The device is about to be rebind, which recreates the debugfs
> + * fault-injection state from scratch - there is nothing to restore
> + * across the rebind. Dropping now-stale fd so the exit handler
> + * won't touch a closed descriptor.
You could write that disabling is done here, in case if
any unexpected signal happen.
> + */
> + gt_reset_fault_injection_forget_fd(fd);
imho just call
gt_reset_disable_fault_injection(0);
Rest looks good.
Regards,
Kamil
> + drm_close_driver(fd);
> + igt_kmod_rebind("xe", pci_slot);
> + fd = drm_open_driver(DRIVER_XE);
> +
> + igt_assert_f(try_vm_create(fd) == 0,
> + "Device not functional after rebind recovery\n");
> + }
> +
> igt_subtest("gt-mocs-reset")
> xe_for_each_gt(fd, gt)
> gt_mocs_reset(fd, gt);
> --
> 2.43.0
>
^ permalink raw reply [flat|nested] 10+ messages in thread* ✗ i915.CI.BAT: failure for tests/intel/xe_exec_reset: add GT reset fault injection stress coverage (rev4)
2026-09-01 14:01 [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage nishit.sharma
2026-09-01 18:13 ` Kamil Konieczny
@ 2026-09-02 0:04 ` Patchwork
1 sibling, 0 replies; 10+ messages in thread
From: Patchwork @ 2026-09-02 0:04 UTC (permalink / raw)
To: nishit.sharma; +Cc: igt-dev
[-- Attachment #1: Type: text/plain, Size: 2479 bytes --]
== Series Details ==
Series: tests/intel/xe_exec_reset: add GT reset fault injection stress coverage (rev4)
URL : https://patchwork.freedesktop.org/series/172441/
State : failure
== Summary ==
CI Bug Log - changes from IGT_9079 -> IGTPW_15767
====================================================
Summary
-------
**FAILURE**
Serious unknown changes coming with IGTPW_15767 absolutely need to be
verified manually.
If you think the reported changes have nothing to do with the changes
introduced in IGTPW_15767, please notify your bug team (I915-ci-infra@lists.freedesktop.org) to allow them
to document this new failure mode, which will reduce false positives in CI.
External URL: https://intel-gfx-ci.01.org/tree/drm-tip/IGTPW_15767/index.html
Participating hosts (37 -> 36)
------------------------------
Missing (1): bat-dg2-13
Possible new issues
-------------------
Here are the unknown changes that may have been introduced in IGTPW_15767:
### IGT changes ###
#### Possible regressions ####
* igt@i915_selftest@live:
- bat-arlh-2: [PASS][1] -> [ABORT][2] +1 other test abort
[1]: https://intel-gfx-ci.01.org/tree/drm-tip/IGT_9079/bat-arlh-2/igt@i915_selftest@live.html
[2]: https://intel-gfx-ci.01.org/tree/drm-tip/IGTPW_15767/bat-arlh-2/igt@i915_selftest@live.html
Known issues
------------
Here are the changes found in IGTPW_15767 that come from known issues:
### IGT changes ###
#### Possible fixes ####
* igt@core_hotunplug@unbind-rebind:
- bat-rpls-4: [DMESG-WARN][3] ([i915#13400]) -> [PASS][4]
[3]: https://intel-gfx-ci.01.org/tree/drm-tip/IGT_9079/bat-rpls-4/igt@core_hotunplug@unbind-rebind.html
[4]: https://intel-gfx-ci.01.org/tree/drm-tip/IGTPW_15767/bat-rpls-4/igt@core_hotunplug@unbind-rebind.html
[i915#13400]: https://gitlab.freedesktop.org/drm/i915/kernel/-/issues/13400
Build changes
-------------
* CI: CI-20190529 -> None
* IGT: IGT_9079 -> IGTPW_15767
* Linux: CI_DRM_19066 -> CI_DRM_19069
CI-20190529: 20190529
CI_DRM_19066: b8f17b16c0f8638b9df4f090511e3ae97210eb77 @ git://anongit.freedesktop.org/gfx-ci/linux
CI_DRM_19069: c53467440a5196138d2d6610e269a43b6b47bf6b @ git://anongit.freedesktop.org/gfx-ci/linux
IGTPW_15767: cb4a33192808eed13ff7df657d50457b44fe9489 @ https://gitlab.freedesktop.org/drm/igt-gpu-tools.git
IGT_9079: 9079
== Logs ==
For more details see: https://intel-gfx-ci.01.org/tree/drm-tip/IGTPW_15767/index.html
[-- Attachment #2: Type: text/html, Size: 3120 bytes --]
^ permalink raw reply [flat|nested] 10+ messages in thread
* [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage
@ 2026-09-02 3:33 nishit.sharma
2026-09-02 10:12 ` Kamil Konieczny
0 siblings, 1 reply; 10+ messages in thread
From: nishit.sharma @ 2026-09-02 3:33 UTC (permalink / raw)
To: igt-dev, kamil.konieczny
From: Nishit Sharma <nishit.sharma@intel.com>
Inject a GT reset failure via the KMD fail_gt_reset debugfs hook, verify
the device wedges as expected, tolerate the resulting -ECANCELED errors in
the submitting threads, then recover by rebind and confirm the driver is
usable again.
Signed-off-by: Nishit Sharma <nishit.sharma@intel.com>
---
tests/intel/xe_exec_reset.c | 124 ++++++++++++++++++++++++++++++++++--
1 file changed, 119 insertions(+), 5 deletions(-)
diff --git a/tests/intel/xe_exec_reset.c b/tests/intel/xe_exec_reset.c
index 6eda71c32..f0071cc67 100644
--- a/tests/intel/xe_exec_reset.c
+++ b/tests/intel/xe_exec_reset.c
@@ -15,6 +15,8 @@
#include <fcntl.h>
#include "igt.h"
+#include "igt_device.h"
+#include "igt_kmod.h"
#include "igt_sysfs.h"
#include "lib/igt_syncobj.h"
#include "lib/intel_reg.h"
@@ -140,6 +142,7 @@ static void test_spin(int fd, struct drm_xe_engine_class_instance *eci,
#define DESTROY_VM_CTX_STRESS (0x1 << 20)
#define MIXED_ENGINE_STRESS (0x1 << 21)
#define PM_TRANSITION_STRESS (0x1 << 22)
+#define FAULT_INJECT_STRESS (0x1 << 23)
/**
* SUBTEST: %s-cat-error
@@ -760,7 +763,7 @@ static void submit_jobs(struct gt_thread_data *t)
uint32_t pressure_bos[PRESSURE_COUNT];
uint32_t *data;
int pressure_count;
- int i = 0;
+ int i = 0, exec_ret;
bo = xe_bo_create(fd, vm, bo_size, vram_if_possible(fd, t->gt),
DRM_XE_GEM_CREATE_FLAG_NEEDS_VISIBLE_VRAM);
@@ -794,8 +797,25 @@ static void submit_jobs(struct gt_thread_data *t)
continue;
}
- xe_exec(fd, &exec);
- xe_exec_queue_destroy(fd, exec.exec_queue_id);
+ /*
+ * Once an injected GT reset failure wedges the device, exec and
+ * queue teardown return -ECANCELED. That is the expected outcome
+ * for the fault-injection stress
+ */
+ if (t->flags & FAULT_INJECT_STRESS) {
+ struct drm_xe_exec_queue_destroy destroy = {
+ .exec_queue_id = exec.exec_queue_id,
+ };
+
+ exec_ret = __xe_exec(fd, &exec);
+ igt_assert_f(exec_ret == 0 || exec_ret == -ECANCELED,
+ "exec returned unexpected error %d (expected 0 or -ECANCELED)\n",
+ exec_ret);
+ igt_ioctl(fd, DRM_IOCTL_XE_EXEC_QUEUE_DESTROY, &destroy);
+ } else {
+ xe_exec(fd, &exec);
+ xe_exec_queue_destroy(fd, exec.exec_queue_id);
+ }
(*t->num_submit)++;
if ((t->flags & MEM_PRESSURE_STRESS) && !(i % 128)) {
@@ -829,8 +849,20 @@ static void submit_jobs(struct gt_thread_data *t)
pressure_bo_destroy(fd, pressure_bos, PRESSURE_COUNT);
munmap(data, bo_size);
- gem_close(fd, bo);
- xe_vm_destroy(fd, vm);
+ /*
+ * On a device wedged by injected GT reset failures, BO close and VM
+ * destroy also return -ECANCELED.
+ */
+ if (t->flags & FAULT_INJECT_STRESS) {
+ struct drm_gem_close close_bo = { .handle = bo };
+ struct drm_xe_vm_destroy vm_destroy = { .vm_id = vm };
+
+ igt_ioctl(fd, DRM_IOCTL_GEM_CLOSE, &close_bo);
+ igt_ioctl(fd, DRM_IOCTL_XE_VM_DESTROY, &vm_destroy);
+ } else {
+ gem_close(fd, bo);
+ xe_vm_destroy(fd, vm);
+ }
}
static void *gt_reset_thread(void *data)
@@ -850,6 +882,54 @@ static void *gt_reset_thread(void *data)
return NULL;
}
+static int fault_inject_fd = -1;
+
+static void gt_reset_disable_fault_injection(int sig)
+{
+ if (fault_inject_fd < 0)
+ return;
+
+ igt_debugfs_write(fault_inject_fd, "fail_gt_reset/probability", "0");
+ igt_debugfs_write(fault_inject_fd, "fail_gt_reset/times", "1");
+ fault_inject_fd = -1;
+}
+
+static void gt_reset_enable_fault_injection(int fd)
+{
+ static bool exit_handler_installed;
+
+ fault_inject_fd = fd;
+
+ if (!exit_handler_installed) {
+ igt_install_exit_handler(gt_reset_disable_fault_injection);
+ exit_handler_installed = true;
+ }
+
+ igt_debugfs_write(fd, "fail_gt_reset/probability", "100");
+ igt_debugfs_write(fd, "fail_gt_reset/times", "2");
+}
+
+static int try_vm_create(int fd)
+{
+ struct drm_xe_vm_create create = { 0 };
+ int err = 0;
+
+ if (igt_ioctl(fd, DRM_IOCTL_XE_VM_CREATE, &create))
+ err = -errno;
+ else
+ xe_vm_destroy(fd, create.vm_id);
+
+ return err;
+}
+
+static void ignore_gt_reset_fault_dmesg(void)
+{
+ igt_emit_ignore_dmesg_regex("reset failed \\(-ECANCELED\\)"
+ "|declared device .* as wedged"
+ "|GPU HANG"
+ "|Failed to reset");
+}
+
/**
* SUBTEST: gt-reset-stress
* Description: Stress GT reset
@@ -891,6 +971,9 @@ static void *gt_reset_thread(void *data)
* Description: Test GT reset while long spinner workload is active
* Test category: stress test
*
+ * SUBTEST: gt-reset-fault-injection
+ * Description: Stress concurrent GT resets and job submissions with GT reset failures injected via debugfs
+ * Test category: fault injection
*/
static void
gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
@@ -907,6 +990,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
pthread_mutex_init(&mutex, 0);
pthread_cond_init(&cond, 0);
+ if (flags & FAULT_INJECT_STRESS)
+ gt_reset_enable_fault_injection(fd);
+
for (i = 0; i < n_threads; ++i) {
threads[i].mutex = &mutex;
threads[i].cond = &cond;
@@ -940,6 +1026,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
igt_info("number of resets %d, submissions %d, submit fails %d vm_recreate %d\n",
num_reset, num_submit, num_submit_fail, num_vm_recreate);
+ if (flags & FAULT_INJECT_STRESS)
+ gt_reset_disable_fault_injection(0);
+
igt_assert_neq(num_reset, 0);
igt_assert_neq(num_submit, 0);
free(threads);
@@ -1122,6 +1211,7 @@ int igt_main()
int gt;
int class;
int fd;
+ char pci_slot[NAME_MAX];
igt_fixture()
fd = drm_open_driver(DRIVER_XE);
@@ -1326,6 +1416,30 @@ int igt_main()
break;
}
+ igt_subtest("gt-reset-fault-injection") {
+ igt_require_f(igt_debugfs_exists(fd, "fail_gt_reset/probability",
+ O_RDWR),
+ "GT reset fault injection not available; "
+ "CONFIG_DRM_XE_KUNIT_TEST/fault-injection must be "
+ "enabled in the KMD\n");
+
+ igt_device_get_pci_slot_name(fd, pci_slot);
+ ignore_gt_reset_fault_dmesg();
+
+ gt_reset(fd, 0, 8, 2, FAULT_INJECT_STRESS);
+
+ igt_assert_f(try_vm_create(fd) != 0,
+ "Device did not wedge after injected GT reset failure\n");
+
+ gt_reset_disable_fault_injection(0);
+ drm_close_driver(fd);
+ igt_kmod_rebind("xe", pci_slot);
+ fd = drm_open_driver(DRIVER_XE);
+
+ igt_assert_f(try_vm_create(fd) == 0,
+ "Device not functional after rebind recovery\n");
+ }
+
igt_subtest("gt-mocs-reset")
xe_for_each_gt(fd, gt)
gt_mocs_reset(fd, gt);
--
2.43.0
^ permalink raw reply related [flat|nested] 10+ messages in thread* Re: [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage
2026-09-02 3:33 [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage nishit.sharma
@ 2026-09-02 10:12 ` Kamil Konieczny
0 siblings, 0 replies; 10+ messages in thread
From: Kamil Konieczny @ 2026-09-02 10:12 UTC (permalink / raw)
To: nishit.sharma; +Cc: igt-dev, kamil.konieczny
Hi nishit.sharma,
On 2026-09-02 at 03:33:15 +0000, nishit.sharma@intel.com wrote:
> From: Nishit Sharma <nishit.sharma@intel.com>
>
> Inject a GT reset failure via the KMD fail_gt_reset debugfs hook, verify
> the device wedges as expected, tolerate the resulting -ECANCELED errors in
> the submitting threads, then recover by rebind and confirm the driver is
> usable again.
>
> Signed-off-by: Nishit Sharma <nishit.sharma@intel.com>
> ---
> tests/intel/xe_exec_reset.c | 124 ++++++++++++++++++++++++++++++++++--
> 1 file changed, 119 insertions(+), 5 deletions(-)
>
> diff --git a/tests/intel/xe_exec_reset.c b/tests/intel/xe_exec_reset.c
> index 6eda71c32..f0071cc67 100644
> --- a/tests/intel/xe_exec_reset.c
> +++ b/tests/intel/xe_exec_reset.c
> @@ -15,6 +15,8 @@
> #include <fcntl.h>
>
> #include "igt.h"
> +#include "igt_device.h"
> +#include "igt_kmod.h"
> #include "igt_sysfs.h"
> #include "lib/igt_syncobj.h"
> #include "lib/intel_reg.h"
> @@ -140,6 +142,7 @@ static void test_spin(int fd, struct drm_xe_engine_class_instance *eci,
> #define DESTROY_VM_CTX_STRESS (0x1 << 20)
> #define MIXED_ENGINE_STRESS (0x1 << 21)
> #define PM_TRANSITION_STRESS (0x1 << 22)
> +#define FAULT_INJECT_STRESS (0x1 << 23)
>
> /**
> * SUBTEST: %s-cat-error
> @@ -760,7 +763,7 @@ static void submit_jobs(struct gt_thread_data *t)
> uint32_t pressure_bos[PRESSURE_COUNT];
> uint32_t *data;
> int pressure_count;
> - int i = 0;
> + int i = 0, exec_ret;
>
> bo = xe_bo_create(fd, vm, bo_size, vram_if_possible(fd, t->gt),
> DRM_XE_GEM_CREATE_FLAG_NEEDS_VISIBLE_VRAM);
> @@ -794,8 +797,25 @@ static void submit_jobs(struct gt_thread_data *t)
> continue;
> }
>
> - xe_exec(fd, &exec);
> - xe_exec_queue_destroy(fd, exec.exec_queue_id);
> + /*
> + * Once an injected GT reset failure wedges the device, exec and
> + * queue teardown return -ECANCELED. That is the expected outcome
> + * for the fault-injection stress
> + */
> + if (t->flags & FAULT_INJECT_STRESS) {
> + struct drm_xe_exec_queue_destroy destroy = {
> + .exec_queue_id = exec.exec_queue_id,
> + };
This seems a break in formatting, interesting it wasn't catched
by checkpatch.pl
> +
> + exec_ret = __xe_exec(fd, &exec);
> + igt_assert_f(exec_ret == 0 || exec_ret == -ECANCELED,
> + "exec returned unexpected error %d (expected 0 or -ECANCELED)\n",
> + exec_ret);
> + igt_ioctl(fd, DRM_IOCTL_XE_EXEC_QUEUE_DESTROY, &destroy);
Above should be shifted to right by one tab, to align
with below 'else'. It could be corrected at merge time so
Reviewed-by: Kamil Konieczny <kamil.konieczny@linux.intel.com>
Regards,
Kamil
> + } else {
> + xe_exec(fd, &exec);
> + xe_exec_queue_destroy(fd, exec.exec_queue_id);
> + }
> (*t->num_submit)++;
>
> if ((t->flags & MEM_PRESSURE_STRESS) && !(i % 128)) {
> @@ -829,8 +849,20 @@ static void submit_jobs(struct gt_thread_data *t)
> pressure_bo_destroy(fd, pressure_bos, PRESSURE_COUNT);
>
> munmap(data, bo_size);
> - gem_close(fd, bo);
> - xe_vm_destroy(fd, vm);
> + /*
> + * On a device wedged by injected GT reset failures, BO close and VM
> + * destroy also return -ECANCELED.
> + */
> + if (t->flags & FAULT_INJECT_STRESS) {
> + struct drm_gem_close close_bo = { .handle = bo };
> + struct drm_xe_vm_destroy vm_destroy = { .vm_id = vm };
> +
> + igt_ioctl(fd, DRM_IOCTL_GEM_CLOSE, &close_bo);
> + igt_ioctl(fd, DRM_IOCTL_XE_VM_DESTROY, &vm_destroy);
> + } else {
> + gem_close(fd, bo);
> + xe_vm_destroy(fd, vm);
> + }
> }
>
> static void *gt_reset_thread(void *data)
> @@ -850,6 +882,54 @@ static void *gt_reset_thread(void *data)
> return NULL;
> }
>
> +static int fault_inject_fd = -1;
> +
> +static void gt_reset_disable_fault_injection(int sig)
> +{
> + if (fault_inject_fd < 0)
> + return;
> +
> + igt_debugfs_write(fault_inject_fd, "fail_gt_reset/probability", "0");
> + igt_debugfs_write(fault_inject_fd, "fail_gt_reset/times", "1");
> + fault_inject_fd = -1;
> +}
> +
> +static void gt_reset_enable_fault_injection(int fd)
> +{
> + static bool exit_handler_installed;
> +
> + fault_inject_fd = fd;
> +
> + if (!exit_handler_installed) {
> + igt_install_exit_handler(gt_reset_disable_fault_injection);
> + exit_handler_installed = true;
> + }
> +
> + igt_debugfs_write(fd, "fail_gt_reset/probability", "100");
> + igt_debugfs_write(fd, "fail_gt_reset/times", "2");
> +}
> +
> +static int try_vm_create(int fd)
> +{
> + struct drm_xe_vm_create create = { 0 };
> + int err = 0;
> +
> + if (igt_ioctl(fd, DRM_IOCTL_XE_VM_CREATE, &create))
> + err = -errno;
> + else
> + xe_vm_destroy(fd, create.vm_id);
> +
> + return err;
> +}
> +
> +static void ignore_gt_reset_fault_dmesg(void)
> +{
> + igt_emit_ignore_dmesg_regex("reset failed \\(-ECANCELED\\)"
> + "|declared device .* as wedged"
> + "|GPU HANG"
> + "|Failed to reset");
> +}
> +
> /**
> * SUBTEST: gt-reset-stress
> * Description: Stress GT reset
> @@ -891,6 +971,9 @@ static void *gt_reset_thread(void *data)
> * Description: Test GT reset while long spinner workload is active
> * Test category: stress test
> *
> + * SUBTEST: gt-reset-fault-injection
> + * Description: Stress concurrent GT resets and job submissions with GT reset failures injected via debugfs
> + * Test category: fault injection
> */
> static void
> gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
> @@ -907,6 +990,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
> pthread_mutex_init(&mutex, 0);
> pthread_cond_init(&cond, 0);
>
> + if (flags & FAULT_INJECT_STRESS)
> + gt_reset_enable_fault_injection(fd);
> +
> for (i = 0; i < n_threads; ++i) {
> threads[i].mutex = &mutex;
> threads[i].cond = &cond;
> @@ -940,6 +1026,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
> igt_info("number of resets %d, submissions %d, submit fails %d vm_recreate %d\n",
> num_reset, num_submit, num_submit_fail, num_vm_recreate);
>
> + if (flags & FAULT_INJECT_STRESS)
> + gt_reset_disable_fault_injection(0);
> +
> igt_assert_neq(num_reset, 0);
> igt_assert_neq(num_submit, 0);
> free(threads);
> @@ -1122,6 +1211,7 @@ int igt_main()
> int gt;
> int class;
> int fd;
> + char pci_slot[NAME_MAX];
>
> igt_fixture()
> fd = drm_open_driver(DRIVER_XE);
> @@ -1326,6 +1416,30 @@ int igt_main()
> break;
> }
>
> + igt_subtest("gt-reset-fault-injection") {
> + igt_require_f(igt_debugfs_exists(fd, "fail_gt_reset/probability",
> + O_RDWR),
> + "GT reset fault injection not available; "
> + "CONFIG_DRM_XE_KUNIT_TEST/fault-injection must be "
> + "enabled in the KMD\n");
> +
> + igt_device_get_pci_slot_name(fd, pci_slot);
> + ignore_gt_reset_fault_dmesg();
> +
> + gt_reset(fd, 0, 8, 2, FAULT_INJECT_STRESS);
> +
> + igt_assert_f(try_vm_create(fd) != 0,
> + "Device did not wedge after injected GT reset failure\n");
> +
> + gt_reset_disable_fault_injection(0);
> + drm_close_driver(fd);
> + igt_kmod_rebind("xe", pci_slot);
> + fd = drm_open_driver(DRIVER_XE);
> +
> + igt_assert_f(try_vm_create(fd) == 0,
> + "Device not functional after rebind recovery\n");
> + }
> +
> igt_subtest("gt-mocs-reset")
> xe_for_each_gt(fd, gt)
> gt_mocs_reset(fd, gt);
> --
> 2.43.0
>
^ permalink raw reply [flat|nested] 10+ messages in thread
* [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage
@ 2026-09-01 3:44 nishit.sharma
2026-09-01 12:48 ` Kamil Konieczny
0 siblings, 1 reply; 10+ messages in thread
From: nishit.sharma @ 2026-09-01 3:44 UTC (permalink / raw)
To: igt-dev, kamil.konieczny
From: Nishit Sharma <nishit.sharma@intel.com>
Inject a GT reset failure via the KMD fail_gt_reset debugfs hook, verify
the device wedges as expected, tolerate the resulting -ECANCELED errors in
the submitting threads, then recover by rebind and confirm the driver is
usable again.
Signed-off-by: Nishit Sharma <nishit.sharma@intel.com>
---
tests/intel/xe_exec_reset.c | 123 ++++++++++++++++++++++++++++++++++--
1 file changed, 118 insertions(+), 5 deletions(-)
diff --git a/tests/intel/xe_exec_reset.c b/tests/intel/xe_exec_reset.c
index 6eda71c32..f3226579f 100644
--- a/tests/intel/xe_exec_reset.c
+++ b/tests/intel/xe_exec_reset.c
@@ -16,6 +16,8 @@
#include "igt.h"
#include "igt_sysfs.h"
+#include "igt_device.h"
+#include "igt_kmod.h"
#include "lib/igt_syncobj.h"
#include "lib/intel_reg.h"
#include "xe_drm.h"
@@ -140,6 +142,7 @@ static void test_spin(int fd, struct drm_xe_engine_class_instance *eci,
#define DESTROY_VM_CTX_STRESS (0x1 << 20)
#define MIXED_ENGINE_STRESS (0x1 << 21)
#define PM_TRANSITION_STRESS (0x1 << 22)
+#define FAULT_INJECT_STRESS (0x1 << 23)
/**
* SUBTEST: %s-cat-error
@@ -760,7 +763,7 @@ static void submit_jobs(struct gt_thread_data *t)
uint32_t pressure_bos[PRESSURE_COUNT];
uint32_t *data;
int pressure_count;
- int i = 0;
+ int i = 0, exec_ret;
bo = xe_bo_create(fd, vm, bo_size, vram_if_possible(fd, t->gt),
DRM_XE_GEM_CREATE_FLAG_NEEDS_VISIBLE_VRAM);
@@ -794,8 +797,25 @@ static void submit_jobs(struct gt_thread_data *t)
continue;
}
- xe_exec(fd, &exec);
- xe_exec_queue_destroy(fd, exec.exec_queue_id);
+ /*
+ * Once an injected GT reset failure wedges the device, exec and
+ * queue teardown return -ECANCELED. That is the expected outcome
+ * for the fault-injection stress
+ */
+ if (t->flags & FAULT_INJECT_STRESS) {
+ struct drm_xe_exec_queue_destroy destroy = {
+ .exec_queue_id = exec.exec_queue_id,
+ };
+
+ exec_ret = __xe_exec(fd, &exec);
+ igt_assert_f(exec_ret == 0 || exec_ret == -ECANCELED,
+ "exec returned unexpected error %d (expected 0 or -ECANCELED)\n",
+ exec_ret);
+ igt_ioctl(fd, DRM_IOCTL_XE_EXEC_QUEUE_DESTROY, &destroy);
+ } else {
+ xe_exec(fd, &exec);
+ xe_exec_queue_destroy(fd, exec.exec_queue_id);
+ }
(*t->num_submit)++;
if ((t->flags & MEM_PRESSURE_STRESS) && !(i % 128)) {
@@ -829,8 +849,20 @@ static void submit_jobs(struct gt_thread_data *t)
pressure_bo_destroy(fd, pressure_bos, PRESSURE_COUNT);
munmap(data, bo_size);
- gem_close(fd, bo);
- xe_vm_destroy(fd, vm);
+ /*
+ * On a device wedged by injected GT reset failures, BO close and VM
+ * destroy also return -ECANCELED.
+ */
+ if (t->flags & FAULT_INJECT_STRESS) {
+ struct drm_gem_close close_bo = { .handle = bo };
+ struct drm_xe_vm_destroy vm_destroy = { .vm_id = vm };
+
+ igt_ioctl(fd, DRM_IOCTL_GEM_CLOSE, &close_bo);
+ igt_ioctl(fd, DRM_IOCTL_XE_VM_DESTROY, &vm_destroy);
+ } else {
+ gem_close(fd, bo);
+ xe_vm_destroy(fd, vm);
+ }
}
static void *gt_reset_thread(void *data)
@@ -850,6 +882,54 @@ static void *gt_reset_thread(void *data)
return NULL;
}
+static int fault_inject_fd = -1;
+
+static void gt_reset_disable_fault_injection(int sig)
+{
+ if (fault_inject_fd < 0)
+ return;
+
+ igt_debugfs_write(fault_inject_fd, "fail_gt_reset/probability", "0");
+ igt_debugfs_write(fault_inject_fd, "fail_gt_reset/times", "1");
+ fault_inject_fd = -1;
+}
+
+static void gt_reset_enable_fault_injection(int fd)
+{
+ static bool exit_handler_installed;
+
+ fault_inject_fd = fd;
+
+ if (!exit_handler_installed) {
+ igt_install_exit_handler(gt_reset_disable_fault_injection);
+ exit_handler_installed = true;
+ }
+
+ igt_debugfs_write(fd, "fail_gt_reset/probability", "100");
+ igt_debugfs_write(fd, "fail_gt_reset/times", "2");
+}
+
+static int try_vm_create(int fd)
+{
+ struct drm_xe_vm_create create = { 0 };
+ int err = 0;
+
+ if (igt_ioctl(fd, DRM_IOCTL_XE_VM_CREATE, &create))
+ err = -errno;
+ else
+ xe_vm_destroy(fd, create.vm_id);
+
+ return err;
+}
+
+static void ignore_gt_reset_fault_dmesg(void)
+{
+ igt_emit_ignore_dmesg_regex("reset failed \\(-ECANCELED\\)"
+ "|declared device .* as wedged"
+ "|GPU HANG"
+ "|Failed to reset");
+}
+
/**
* SUBTEST: gt-reset-stress
* Description: Stress GT reset
@@ -891,6 +971,9 @@ static void *gt_reset_thread(void *data)
* Description: Test GT reset while long spinner workload is active
* Test category: stress test
*
+ * SUBTEST: gt-reset-fault-injection
+ * Description: Stress concurrent GT resets and job submissions with GT reset failures injected via debugfs
+ * Test category: fault injection
*/
static void
gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
@@ -907,6 +990,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
pthread_mutex_init(&mutex, 0);
pthread_cond_init(&cond, 0);
+ if (flags & FAULT_INJECT_STRESS)
+ gt_reset_enable_fault_injection(fd);
+
for (i = 0; i < n_threads; ++i) {
threads[i].mutex = &mutex;
threads[i].cond = &cond;
@@ -940,6 +1026,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
igt_info("number of resets %d, submissions %d, submit fails %d vm_recreate %d\n",
num_reset, num_submit, num_submit_fail, num_vm_recreate);
+ if (flags & FAULT_INJECT_STRESS)
+ gt_reset_disable_fault_injection(0);
+
igt_assert_neq(num_reset, 0);
igt_assert_neq(num_submit, 0);
free(threads);
@@ -1122,6 +1211,7 @@ int igt_main()
int gt;
int class;
int fd;
+ char pci_slot[NAME_MAX];
igt_fixture()
fd = drm_open_driver(DRIVER_XE);
@@ -1326,6 +1416,29 @@ int igt_main()
break;
}
+ igt_subtest("gt-reset-fault-injection") {
+ igt_require_f(igt_debugfs_exists(fd, "fail_gt_reset/probability",
+ O_RDWR),
+ "GT reset fault injection not available; "
+ "CONFIG_DRM_XE_KUNIT_TEST/fault-injection must be "
+ "enabled in the KMD\n");
+
+ igt_device_get_pci_slot_name(fd, pci_slot);
+ ignore_gt_reset_fault_dmesg();
+
+ gt_reset(fd, 0, 8, 2, FAULT_INJECT_STRESS);
+
+ igt_assert_f(try_vm_create(fd) != 0,
+ "Device did not wedge after injected GT reset failure\n");
+
+ drm_close_driver(fd);
+ igt_kmod_rebind("xe", pci_slot);
+ fd = drm_open_driver(DRIVER_XE);
+
+ igt_assert_f(try_vm_create(fd) == 0,
+ "Device not functional after rebind recovery\n");
+ }
+
igt_subtest("gt-mocs-reset")
xe_for_each_gt(fd, gt)
gt_mocs_reset(fd, gt);
--
2.43.0
^ permalink raw reply related [flat|nested] 10+ messages in thread* Re: [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage
2026-09-01 3:44 nishit.sharma
@ 2026-09-01 12:48 ` Kamil Konieczny
0 siblings, 0 replies; 10+ messages in thread
From: Kamil Konieczny @ 2026-09-01 12:48 UTC (permalink / raw)
To: nishit.sharma; +Cc: igt-dev, kamil.konieczny
Hi Nishit,
On 2026-09-01 at 03:44:03 +0000, nishit.sharma@intel.com wrote:
> From: Nishit Sharma <nishit.sharma@intel.com>
>
> Inject a GT reset failure via the KMD fail_gt_reset debugfs hook, verify
> the device wedges as expected, tolerate the resulting -ECANCELED errors in
> the submitting threads, then recover by rebind and confirm the driver is
> usable again.
>
> Signed-off-by: Nishit Sharma <nishit.sharma@intel.com>
> ---
> tests/intel/xe_exec_reset.c | 123 ++++++++++++++++++++++++++++++++++--
> 1 file changed, 118 insertions(+), 5 deletions(-)
>
> diff --git a/tests/intel/xe_exec_reset.c b/tests/intel/xe_exec_reset.c
> index 6eda71c32..f3226579f 100644
> --- a/tests/intel/xe_exec_reset.c
> +++ b/tests/intel/xe_exec_reset.c
> @@ -16,6 +16,8 @@
>
> #include "igt.h"
> #include "igt_sysfs.h"
> +#include "igt_device.h"
> +#include "igt_kmod.h"
Move these up before igt_sysfs.h, keep it sorted.
> #include "lib/igt_syncobj.h"
> #include "lib/intel_reg.h"
> #include "xe_drm.h"
> @@ -140,6 +142,7 @@ static void test_spin(int fd, struct drm_xe_engine_class_instance *eci,
> #define DESTROY_VM_CTX_STRESS (0x1 << 20)
> #define MIXED_ENGINE_STRESS (0x1 << 21)
> #define PM_TRANSITION_STRESS (0x1 << 22)
> +#define FAULT_INJECT_STRESS (0x1 << 23)
>
> /**
> * SUBTEST: %s-cat-error
> @@ -760,7 +763,7 @@ static void submit_jobs(struct gt_thread_data *t)
> uint32_t pressure_bos[PRESSURE_COUNT];
> uint32_t *data;
> int pressure_count;
> - int i = 0;
> + int i = 0, exec_ret;
>
> bo = xe_bo_create(fd, vm, bo_size, vram_if_possible(fd, t->gt),
> DRM_XE_GEM_CREATE_FLAG_NEEDS_VISIBLE_VRAM);
> @@ -794,8 +797,25 @@ static void submit_jobs(struct gt_thread_data *t)
> continue;
> }
>
> - xe_exec(fd, &exec);
> - xe_exec_queue_destroy(fd, exec.exec_queue_id);
> + /*
> + * Once an injected GT reset failure wedges the device, exec and
> + * queue teardown return -ECANCELED. That is the expected outcome
> + * for the fault-injection stress
> + */
> + if (t->flags & FAULT_INJECT_STRESS) {
> + struct drm_xe_exec_queue_destroy destroy = {
> + .exec_queue_id = exec.exec_queue_id,
> + };
> +
> + exec_ret = __xe_exec(fd, &exec);
> + igt_assert_f(exec_ret == 0 || exec_ret == -ECANCELED,
> + "exec returned unexpected error %d (expected 0 or -ECANCELED)\n",
> + exec_ret);
> + igt_ioctl(fd, DRM_IOCTL_XE_EXEC_QUEUE_DESTROY, &destroy);
> + } else {
> + xe_exec(fd, &exec);
> + xe_exec_queue_destroy(fd, exec.exec_queue_id);
> + }
> (*t->num_submit)++;
>
> if ((t->flags & MEM_PRESSURE_STRESS) && !(i % 128)) {
> @@ -829,8 +849,20 @@ static void submit_jobs(struct gt_thread_data *t)
> pressure_bo_destroy(fd, pressure_bos, PRESSURE_COUNT);
>
> munmap(data, bo_size);
> - gem_close(fd, bo);
> - xe_vm_destroy(fd, vm);
> + /*
> + * On a device wedged by injected GT reset failures, BO close and VM
> + * destroy also return -ECANCELED.
> + */
> + if (t->flags & FAULT_INJECT_STRESS) {
> + struct drm_gem_close close_bo = { .handle = bo };
> + struct drm_xe_vm_destroy vm_destroy = { .vm_id = vm };
> +
> + igt_ioctl(fd, DRM_IOCTL_GEM_CLOSE, &close_bo);
> + igt_ioctl(fd, DRM_IOCTL_XE_VM_DESTROY, &vm_destroy);
> + } else {
> + gem_close(fd, bo);
> + xe_vm_destroy(fd, vm);
> + }
> }
>
> static void *gt_reset_thread(void *data)
> @@ -850,6 +882,54 @@ static void *gt_reset_thread(void *data)
> return NULL;
> }
>
> +static int fault_inject_fd = -1;
> +
> +static void gt_reset_disable_fault_injection(int sig)
> +{
> + if (fault_inject_fd < 0)
> + return;
> +
> + igt_debugfs_write(fault_inject_fd, "fail_gt_reset/probability", "0");
> + igt_debugfs_write(fault_inject_fd, "fail_gt_reset/times", "1");
> + fault_inject_fd = -1;
> +}
Now this looks ok but see a note at a test.
> +
> +static void gt_reset_enable_fault_injection(int fd)
> +{
> + static bool exit_handler_installed;
> +
> + fault_inject_fd = fd;
> +
> + if (!exit_handler_installed) {
> + igt_install_exit_handler(gt_reset_disable_fault_injection);
> + exit_handler_installed = true;
> + }
> +
> + igt_debugfs_write(fd, "fail_gt_reset/probability", "100");
> + igt_debugfs_write(fd, "fail_gt_reset/times", "2");
> +}
> +
> +static int try_vm_create(int fd)
> +{
> + struct drm_xe_vm_create create = { 0 };
> + int err = 0;
> +
> + if (igt_ioctl(fd, DRM_IOCTL_XE_VM_CREATE, &create))
> + err = -errno;
> + else
> + xe_vm_destroy(fd, create.vm_id);
> +
> + return err;
> +}
> +
> +static void ignore_gt_reset_fault_dmesg(void)
> +{
> + igt_emit_ignore_dmesg_regex("reset failed \\(-ECANCELED\\)"
> + "|declared device .* as wedged"
> + "|GPU HANG"
> + "|Failed to reset");
> +}
> +
> /**
> * SUBTEST: gt-reset-stress
> * Description: Stress GT reset
> @@ -891,6 +971,9 @@ static void *gt_reset_thread(void *data)
> * Description: Test GT reset while long spinner workload is active
> * Test category: stress test
> *
> + * SUBTEST: gt-reset-fault-injection
> + * Description: Stress concurrent GT resets and job submissions with GT reset failures injected via debugfs
> + * Test category: fault injection
> */
> static void
> gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
> @@ -907,6 +990,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
> pthread_mutex_init(&mutex, 0);
> pthread_cond_init(&cond, 0);
>
> + if (flags & FAULT_INJECT_STRESS)
> + gt_reset_enable_fault_injection(fd);
> +
> for (i = 0; i < n_threads; ++i) {
> threads[i].mutex = &mutex;
> threads[i].cond = &cond;
> @@ -940,6 +1026,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
> igt_info("number of resets %d, submissions %d, submit fails %d vm_recreate %d\n",
> num_reset, num_submit, num_submit_fail, num_vm_recreate);
>
> + if (flags & FAULT_INJECT_STRESS)
> + gt_reset_disable_fault_injection(0);
> +
> igt_assert_neq(num_reset, 0);
> igt_assert_neq(num_submit, 0);
> free(threads);
> @@ -1122,6 +1211,7 @@ int igt_main()
> int gt;
> int class;
> int fd;
> + char pci_slot[NAME_MAX];
>
> igt_fixture()
> fd = drm_open_driver(DRIVER_XE);
> @@ -1326,6 +1416,29 @@ int igt_main()
> break;
> }
>
> + igt_subtest("gt-reset-fault-injection") {
> + igt_require_f(igt_debugfs_exists(fd, "fail_gt_reset/probability",
> + O_RDWR),
> + "GT reset fault injection not available; "
> + "CONFIG_DRM_XE_KUNIT_TEST/fault-injection must be "
> + "enabled in the KMD\n");
> +
> + igt_device_get_pci_slot_name(fd, pci_slot);
> + ignore_gt_reset_fault_dmesg();
> +
> + gt_reset(fd, 0, 8, 2, FAULT_INJECT_STRESS);
> +
> + igt_assert_f(try_vm_create(fd) != 0,
> + "Device did not wedge after injected GT reset failure\n");
> +
> + drm_close_driver(fd);
Now fault_inject_fd becomes stale.
> + igt_kmod_rebind("xe", pci_slot);
After rebind there is no point in restoring params, or do I miss
something?
Regards,
Kamil
> + fd = drm_open_driver(DRIVER_XE);
> +
> + igt_assert_f(try_vm_create(fd) == 0,
> + "Device not functional after rebind recovery\n");
> + }
> +
> igt_subtest("gt-mocs-reset")
> xe_for_each_gt(fd, gt)
> gt_mocs_reset(fd, gt);
> --
> 2.43.0
>
^ permalink raw reply [flat|nested] 10+ messages in thread
* [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage
@ 2026-08-25 3:25 nishit.sharma
2026-08-31 17:20 ` Kamil Konieczny
0 siblings, 1 reply; 10+ messages in thread
From: nishit.sharma @ 2026-08-25 3:25 UTC (permalink / raw)
To: igt-dev, kamil.konieczny
From: Nishit Sharma <nishit.sharma@intel.com>
Inject a GT reset failure via the KMD fail_gt_reset debugfs hook, verify
the device wedges as expected, tolerate the resulting -ECANCELED errors in
the submitting threads, then recover by rebind and confirm the driver is
usable again.
Signed-off-by: Nishit Sharma <nishit.sharma@intel.com>
---
tests/intel/xe_exec_reset.c | 107 ++++++++++++++++++++++++++++++++++--
1 file changed, 101 insertions(+), 6 deletions(-)
diff --git a/tests/intel/xe_exec_reset.c b/tests/intel/xe_exec_reset.c
index 6eda71c32..1796e4ec1 100644
--- a/tests/intel/xe_exec_reset.c
+++ b/tests/intel/xe_exec_reset.c
@@ -16,6 +16,8 @@
#include "igt.h"
#include "igt_sysfs.h"
+#include "igt_device.h"
+#include "igt_kmod.h"
#include "lib/igt_syncobj.h"
#include "lib/intel_reg.h"
#include "xe_drm.h"
@@ -140,6 +142,7 @@ static void test_spin(int fd, struct drm_xe_engine_class_instance *eci,
#define DESTROY_VM_CTX_STRESS (0x1 << 20)
#define MIXED_ENGINE_STRESS (0x1 << 21)
#define PM_TRANSITION_STRESS (0x1 << 22)
+#define FAULT_INJECT_STRESS (0x1 << 23)
/**
* SUBTEST: %s-cat-error
@@ -760,7 +763,7 @@ static void submit_jobs(struct gt_thread_data *t)
uint32_t pressure_bos[PRESSURE_COUNT];
uint32_t *data;
int pressure_count;
- int i = 0;
+ int i = 0, exec_ret;
bo = xe_bo_create(fd, vm, bo_size, vram_if_possible(fd, t->gt),
DRM_XE_GEM_CREATE_FLAG_NEEDS_VISIBLE_VRAM);
@@ -794,8 +797,25 @@ static void submit_jobs(struct gt_thread_data *t)
continue;
}
- xe_exec(fd, &exec);
- xe_exec_queue_destroy(fd, exec.exec_queue_id);
+ /*
+ * Once an injected GT reset failure wedges the device, exec and
+ * queue teardown return -ECANCELED. That is the expected outcome
+ * for the fault-injection stress
+ */
+ if (t->flags & FAULT_INJECT_STRESS) {
+ struct drm_xe_exec_queue_destroy destroy = {
+ .exec_queue_id = exec.exec_queue_id,
+ };
+
+ exec_ret = __xe_exec(fd, &exec);
+ igt_assert_f(exec_ret == 0 || exec_ret == -ECANCELED,
+ "exec returned unexpected error %d (expected 0 or -ECANCELED)\n",
+ exec_ret);
+ igt_ioctl(fd, DRM_IOCTL_XE_EXEC_QUEUE_DESTROY, &destroy);
+ } else {
+ xe_exec(fd, &exec);
+ xe_exec_queue_destroy(fd, exec.exec_queue_id);
+ }
(*t->num_submit)++;
if ((t->flags & MEM_PRESSURE_STRESS) && !(i % 128)) {
@@ -829,8 +849,20 @@ static void submit_jobs(struct gt_thread_data *t)
pressure_bo_destroy(fd, pressure_bos, PRESSURE_COUNT);
munmap(data, bo_size);
- gem_close(fd, bo);
- xe_vm_destroy(fd, vm);
+ /*
+ * On a device wedged by injected GT reset failures, BO close and VM
+ * destroy also return -ECANCELED.
+ */
+ if (t->flags & FAULT_INJECT_STRESS) {
+ struct drm_gem_close close_bo = { .handle = bo };
+ struct drm_xe_vm_destroy vm_destroy = { .vm_id = vm };
+
+ igt_ioctl(fd, DRM_IOCTL_GEM_CLOSE, &close_bo);
+ igt_ioctl(fd, DRM_IOCTL_XE_VM_DESTROY, &vm_destroy);
+ } else {
+ gem_close(fd, bo);
+ xe_vm_destroy(fd, vm);
+ }
}
static void *gt_reset_thread(void *data)
@@ -850,6 +882,33 @@ static void *gt_reset_thread(void *data)
return NULL;
}
+static void gt_reset_fault_injection(int fd, bool enable)
+{
+ igt_debugfs_write(fd, "fail_gt_reset/probability", enable ? "100" : "0");
+ igt_debugfs_write(fd, "fail_gt_reset/times", enable ? "2" : "1");
+}
+
+static int try_vm_create(int fd)
+{
+ struct drm_xe_vm_create create = { 0 };
+ int err = 0;
+
+ if (igt_ioctl(fd, DRM_IOCTL_XE_VM_CREATE, &create))
+ err = -errno;
+ else
+ xe_vm_destroy(fd, create.vm_id);
+
+ return err;
+}
+
+static void ignore_gt_reset_fault_dmesg(void)
+{
+ igt_emit_ignore_dmesg_regex("reset failed \\(-ECANCELED\\)"
+ "|declared device .* as wedged"
+ "|GPU HANG"
+ "|Failed to reset");
+}
+
/**
* SUBTEST: gt-reset-stress
* Description: Stress GT reset
@@ -891,6 +950,9 @@ static void *gt_reset_thread(void *data)
* Description: Test GT reset while long spinner workload is active
* Test category: stress test
*
+ * SUBTEST: gt-reset-fault-injection
+ * Description: Stress concurrent GT resets and job submissions with GT reset failures injected via debugfs
+ * Test category: fault injection
*/
static void
gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
@@ -907,6 +969,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
pthread_mutex_init(&mutex, 0);
pthread_cond_init(&cond, 0);
+ if (flags & FAULT_INJECT_STRESS)
+ gt_reset_fault_injection(fd, true);
+
for (i = 0; i < n_threads; ++i) {
threads[i].mutex = &mutex;
threads[i].cond = &cond;
@@ -940,6 +1005,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
igt_info("number of resets %d, submissions %d, submit fails %d vm_recreate %d\n",
num_reset, num_submit, num_submit_fail, num_vm_recreate);
+ if (flags & FAULT_INJECT_STRESS)
+ gt_reset_fault_injection(fd, true);
+
igt_assert_neq(num_reset, 0);
igt_assert_neq(num_submit, 0);
free(threads);
@@ -1122,6 +1190,7 @@ int igt_main()
int gt;
int class;
int fd;
+ char pci_slot[NAME_MAX];
igt_fixture()
fd = drm_open_driver(DRIVER_XE);
@@ -1326,6 +1395,29 @@ int igt_main()
break;
}
+ igt_subtest("gt-reset-fault-injection") {
+ igt_require_f(igt_debugfs_exists(fd, "fail_gt_reset/probability",
+ O_RDWR),
+ "GT reset fault injection not available; "
+ "CONFIG_DRM_XE_KUNIT_TEST/fault-injection must be "
+ "enabled in the KMD\n");
+
+ igt_device_get_pci_slot_name(fd, pci_slot);
+ ignore_gt_reset_fault_dmesg();
+
+ gt_reset(fd, 0, 8, 2, FAULT_INJECT_STRESS);
+
+ igt_assert_f(try_vm_create(fd) != 0,
+ "Device did not wedge after injected GT reset failure\n");
+
+ drm_close_driver(fd);
+ igt_kmod_rebind("xe", pci_slot);
+ fd = drm_open_driver(DRIVER_XE);
+
+ igt_assert_f(try_vm_create(fd) == 0,
+ "Device not functional after rebind recovery\n");
+ }
+
igt_subtest("gt-mocs-reset")
xe_for_each_gt(fd, gt)
gt_mocs_reset(fd, gt);
@@ -1455,6 +1547,9 @@ int igt_main()
}
}
- igt_fixture()
+ igt_fixture() {
+ if (igt_debugfs_exists(fd, "fail_gt_reset/probability", O_RDWR))
+ gt_reset_fault_injection(fd, false);
drm_close_driver(fd);
+ }
}
--
2.34.1
^ permalink raw reply related [flat|nested] 10+ messages in thread* Re: [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage
2026-08-25 3:25 nishit.sharma
@ 2026-08-31 17:20 ` Kamil Konieczny
0 siblings, 0 replies; 10+ messages in thread
From: Kamil Konieczny @ 2026-08-31 17:20 UTC (permalink / raw)
To: nishit.sharma; +Cc: igt-dev, kamil.konieczny
Hi Nishit,
On 2026-08-25 at 03:25:20 +0000, nishit.sharma@intel.com wrote:
> From: Nishit Sharma <nishit.sharma@intel.com>
>
> Inject a GT reset failure via the KMD fail_gt_reset debugfs hook, verify
> the device wedges as expected, tolerate the resulting -ECANCELED errors in
> the submitting threads, then recover by rebind and confirm the driver is
> usable again.
>
> Signed-off-by: Nishit Sharma <nishit.sharma@intel.com>
> ---
> tests/intel/xe_exec_reset.c | 107 ++++++++++++++++++++++++++++++++++--
> 1 file changed, 101 insertions(+), 6 deletions(-)
>
> diff --git a/tests/intel/xe_exec_reset.c b/tests/intel/xe_exec_reset.c
> index 6eda71c32..1796e4ec1 100644
> --- a/tests/intel/xe_exec_reset.c
> +++ b/tests/intel/xe_exec_reset.c
> @@ -16,6 +16,8 @@
>
> #include "igt.h"
> #include "igt_sysfs.h"
> +#include "igt_device.h"
> +#include "igt_kmod.h"
> #include "lib/igt_syncobj.h"
> #include "lib/intel_reg.h"
> #include "xe_drm.h"
> @@ -140,6 +142,7 @@ static void test_spin(int fd, struct drm_xe_engine_class_instance *eci,
> #define DESTROY_VM_CTX_STRESS (0x1 << 20)
> #define MIXED_ENGINE_STRESS (0x1 << 21)
> #define PM_TRANSITION_STRESS (0x1 << 22)
> +#define FAULT_INJECT_STRESS (0x1 << 23)
>
> /**
> * SUBTEST: %s-cat-error
> @@ -760,7 +763,7 @@ static void submit_jobs(struct gt_thread_data *t)
> uint32_t pressure_bos[PRESSURE_COUNT];
> uint32_t *data;
> int pressure_count;
> - int i = 0;
> + int i = 0, exec_ret;
>
> bo = xe_bo_create(fd, vm, bo_size, vram_if_possible(fd, t->gt),
> DRM_XE_GEM_CREATE_FLAG_NEEDS_VISIBLE_VRAM);
> @@ -794,8 +797,25 @@ static void submit_jobs(struct gt_thread_data *t)
> continue;
> }
>
> - xe_exec(fd, &exec);
> - xe_exec_queue_destroy(fd, exec.exec_queue_id);
> + /*
> + * Once an injected GT reset failure wedges the device, exec and
> + * queue teardown return -ECANCELED. That is the expected outcome
> + * for the fault-injection stress
> + */
> + if (t->flags & FAULT_INJECT_STRESS) {
> + struct drm_xe_exec_queue_destroy destroy = {
> + .exec_queue_id = exec.exec_queue_id,
> + };
> +
> + exec_ret = __xe_exec(fd, &exec);
> + igt_assert_f(exec_ret == 0 || exec_ret == -ECANCELED,
> + "exec returned unexpected error %d (expected 0 or -ECANCELED)\n",
> + exec_ret);
> + igt_ioctl(fd, DRM_IOCTL_XE_EXEC_QUEUE_DESTROY, &destroy);
> + } else {
> + xe_exec(fd, &exec);
> + xe_exec_queue_destroy(fd, exec.exec_queue_id);
> + }
> (*t->num_submit)++;
>
> if ((t->flags & MEM_PRESSURE_STRESS) && !(i % 128)) {
> @@ -829,8 +849,20 @@ static void submit_jobs(struct gt_thread_data *t)
> pressure_bo_destroy(fd, pressure_bos, PRESSURE_COUNT);
>
> munmap(data, bo_size);
> - gem_close(fd, bo);
> - xe_vm_destroy(fd, vm);
> + /*
> + * On a device wedged by injected GT reset failures, BO close and VM
> + * destroy also return -ECANCELED.
> + */
> + if (t->flags & FAULT_INJECT_STRESS) {
> + struct drm_gem_close close_bo = { .handle = bo };
> + struct drm_xe_vm_destroy vm_destroy = { .vm_id = vm };
> +
> + igt_ioctl(fd, DRM_IOCTL_GEM_CLOSE, &close_bo);
> + igt_ioctl(fd, DRM_IOCTL_XE_VM_DESTROY, &vm_destroy);
> + } else {
> + gem_close(fd, bo);
> + xe_vm_destroy(fd, vm);
> + }
> }
>
> static void *gt_reset_thread(void *data)
> @@ -850,6 +882,33 @@ static void *gt_reset_thread(void *data)
> return NULL;
> }
>
> +static void gt_reset_fault_injection(int fd, bool enable)
> +{
> + igt_debugfs_write(fd, "fail_gt_reset/probability", enable ? "100" : "0");
> + igt_debugfs_write(fd, "fail_gt_reset/times", enable ? "2" : "1");
> +}
imho this function should be two functions, one for enabling
and second for disabling. Second one should be non-fd and
registered as at_exit() function, in case any of signals
happens (not catched by final igt_fixture).
So here:
static void gt_reset_enable_fault_injection(int fd)
static void gt_reset_disable_fault_injection(void)
> +
> +static int try_vm_create(int fd)
> +{
> + struct drm_xe_vm_create create = { 0 };
> + int err = 0;
> +
> + if (igt_ioctl(fd, DRM_IOCTL_XE_VM_CREATE, &create))
> + err = -errno;
> + else
> + xe_vm_destroy(fd, create.vm_id);
> +
> + return err;
> +}
> +
> +static void ignore_gt_reset_fault_dmesg(void)
> +{
> + igt_emit_ignore_dmesg_regex("reset failed \\(-ECANCELED\\)"
> + "|declared device .* as wedged"
> + "|GPU HANG"
> + "|Failed to reset");
> +}
> +
> /**
> * SUBTEST: gt-reset-stress
> * Description: Stress GT reset
> @@ -891,6 +950,9 @@ static void *gt_reset_thread(void *data)
> * Description: Test GT reset while long spinner workload is active
> * Test category: stress test
> *
> + * SUBTEST: gt-reset-fault-injection
> + * Description: Stress concurrent GT resets and job submissions with GT reset failures injected via debugfs
> + * Test category: fault injection
> */
> static void
> gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
> @@ -907,6 +969,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
> pthread_mutex_init(&mutex, 0);
> pthread_cond_init(&cond, 0);
>
> + if (flags & FAULT_INJECT_STRESS)
> + gt_reset_fault_injection(fd, true);
> +
> for (i = 0; i < n_threads; ++i) {
> threads[i].mutex = &mutex;
> threads[i].cond = &cond;
> @@ -940,6 +1005,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
> igt_info("number of resets %d, submissions %d, submit fails %d vm_recreate %d\n",
> num_reset, num_submit, num_submit_fail, num_vm_recreate);
>
> + if (flags & FAULT_INJECT_STRESS)
> + gt_reset_fault_injection(fd, true);
> +
> igt_assert_neq(num_reset, 0);
> igt_assert_neq(num_submit, 0);
> free(threads);
> @@ -1122,6 +1190,7 @@ int igt_main()
> int gt;
> int class;
> int fd;
> + char pci_slot[NAME_MAX];
>
> igt_fixture()
> fd = drm_open_driver(DRIVER_XE);
> @@ -1326,6 +1395,29 @@ int igt_main()
> break;
> }
>
> + igt_subtest("gt-reset-fault-injection") {
> + igt_require_f(igt_debugfs_exists(fd, "fail_gt_reset/probability",
> + O_RDWR),
> + "GT reset fault injection not available; "
> + "CONFIG_DRM_XE_KUNIT_TEST/fault-injection must be "
> + "enabled in the KMD\n");
> +
> + igt_device_get_pci_slot_name(fd, pci_slot);
> + ignore_gt_reset_fault_dmesg();
> +
> + gt_reset(fd, 0, 8, 2, FAULT_INJECT_STRESS);
> +
> + igt_assert_f(try_vm_create(fd) != 0,
> + "Device did not wedge after injected GT reset failure\n");
> +
> + drm_close_driver(fd);
> + igt_kmod_rebind("xe", pci_slot);
> + fd = drm_open_driver(DRIVER_XE);
> +
> + igt_assert_f(try_vm_create(fd) == 0,
> + "Device not functional after rebind recovery\n");
> + }
> +
> igt_subtest("gt-mocs-reset")
> xe_for_each_gt(fd, gt)
> gt_mocs_reset(fd, gt);
> @@ -1455,6 +1547,9 @@ int igt_main()
> }
> }
>
> - igt_fixture()
> + igt_fixture() {
> + if (igt_debugfs_exists(fd, "fail_gt_reset/probability", O_RDWR))
> + gt_reset_fault_injection(fd, false);
See note above, some errors like SIGSEGV would make it
unreachable.
Regards,
Kamil
> drm_close_driver(fd);
> + }
> }
> --
> 2.34.1
>
^ permalink raw reply [flat|nested] 10+ messages in thread
* [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage
@ 2026-08-19 7:36 nishit.sharma
0 siblings, 0 replies; 10+ messages in thread
From: nishit.sharma @ 2026-08-19 7:36 UTC (permalink / raw)
To: igt-dev, kamil.konieczny
From: Nishit Sharma <nishit.sharma@intel.com>
Inject a GT reset failure via the KMD fail_gt_reset debugfs hook, verify
the device wedges as expected, tolerate the resulting -ECANCELED errors in
the submitting threads, then recover by rebind and confirm the driver is
usable again.
Signed-off-by: Nishit Sharma <nishit.sharma@intel.com>
---
tests/intel/xe_exec_reset.c | 107 ++++++++++++++++++++++++++++++++++--
1 file changed, 101 insertions(+), 6 deletions(-)
diff --git a/tests/intel/xe_exec_reset.c b/tests/intel/xe_exec_reset.c
index 6eda71c32..1796e4ec1 100644
--- a/tests/intel/xe_exec_reset.c
+++ b/tests/intel/xe_exec_reset.c
@@ -16,6 +16,8 @@
#include "igt.h"
#include "igt_sysfs.h"
+#include "igt_device.h"
+#include "igt_kmod.h"
#include "lib/igt_syncobj.h"
#include "lib/intel_reg.h"
#include "xe_drm.h"
@@ -140,6 +142,7 @@ static void test_spin(int fd, struct drm_xe_engine_class_instance *eci,
#define DESTROY_VM_CTX_STRESS (0x1 << 20)
#define MIXED_ENGINE_STRESS (0x1 << 21)
#define PM_TRANSITION_STRESS (0x1 << 22)
+#define FAULT_INJECT_STRESS (0x1 << 23)
/**
* SUBTEST: %s-cat-error
@@ -760,7 +763,7 @@ static void submit_jobs(struct gt_thread_data *t)
uint32_t pressure_bos[PRESSURE_COUNT];
uint32_t *data;
int pressure_count;
- int i = 0;
+ int i = 0, exec_ret;
bo = xe_bo_create(fd, vm, bo_size, vram_if_possible(fd, t->gt),
DRM_XE_GEM_CREATE_FLAG_NEEDS_VISIBLE_VRAM);
@@ -794,8 +797,25 @@ static void submit_jobs(struct gt_thread_data *t)
continue;
}
- xe_exec(fd, &exec);
- xe_exec_queue_destroy(fd, exec.exec_queue_id);
+ /*
+ * Once an injected GT reset failure wedges the device, exec and
+ * queue teardown return -ECANCELED. That is the expected outcome
+ * for the fault-injection stress
+ */
+ if (t->flags & FAULT_INJECT_STRESS) {
+ struct drm_xe_exec_queue_destroy destroy = {
+ .exec_queue_id = exec.exec_queue_id,
+ };
+
+ exec_ret = __xe_exec(fd, &exec);
+ igt_assert_f(exec_ret == 0 || exec_ret == -ECANCELED,
+ "exec returned unexpected error %d (expected 0 or -ECANCELED)\n",
+ exec_ret);
+ igt_ioctl(fd, DRM_IOCTL_XE_EXEC_QUEUE_DESTROY, &destroy);
+ } else {
+ xe_exec(fd, &exec);
+ xe_exec_queue_destroy(fd, exec.exec_queue_id);
+ }
(*t->num_submit)++;
if ((t->flags & MEM_PRESSURE_STRESS) && !(i % 128)) {
@@ -829,8 +849,20 @@ static void submit_jobs(struct gt_thread_data *t)
pressure_bo_destroy(fd, pressure_bos, PRESSURE_COUNT);
munmap(data, bo_size);
- gem_close(fd, bo);
- xe_vm_destroy(fd, vm);
+ /*
+ * On a device wedged by injected GT reset failures, BO close and VM
+ * destroy also return -ECANCELED.
+ */
+ if (t->flags & FAULT_INJECT_STRESS) {
+ struct drm_gem_close close_bo = { .handle = bo };
+ struct drm_xe_vm_destroy vm_destroy = { .vm_id = vm };
+
+ igt_ioctl(fd, DRM_IOCTL_GEM_CLOSE, &close_bo);
+ igt_ioctl(fd, DRM_IOCTL_XE_VM_DESTROY, &vm_destroy);
+ } else {
+ gem_close(fd, bo);
+ xe_vm_destroy(fd, vm);
+ }
}
static void *gt_reset_thread(void *data)
@@ -850,6 +882,33 @@ static void *gt_reset_thread(void *data)
return NULL;
}
+static void gt_reset_fault_injection(int fd, bool enable)
+{
+ igt_debugfs_write(fd, "fail_gt_reset/probability", enable ? "100" : "0");
+ igt_debugfs_write(fd, "fail_gt_reset/times", enable ? "2" : "1");
+}
+
+static int try_vm_create(int fd)
+{
+ struct drm_xe_vm_create create = { 0 };
+ int err = 0;
+
+ if (igt_ioctl(fd, DRM_IOCTL_XE_VM_CREATE, &create))
+ err = -errno;
+ else
+ xe_vm_destroy(fd, create.vm_id);
+
+ return err;
+}
+
+static void ignore_gt_reset_fault_dmesg(void)
+{
+ igt_emit_ignore_dmesg_regex("reset failed \\(-ECANCELED\\)"
+ "|declared device .* as wedged"
+ "|GPU HANG"
+ "|Failed to reset");
+}
+
/**
* SUBTEST: gt-reset-stress
* Description: Stress GT reset
@@ -891,6 +950,9 @@ static void *gt_reset_thread(void *data)
* Description: Test GT reset while long spinner workload is active
* Test category: stress test
*
+ * SUBTEST: gt-reset-fault-injection
+ * Description: Stress concurrent GT resets and job submissions with GT reset failures injected via debugfs
+ * Test category: fault injection
*/
static void
gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
@@ -907,6 +969,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
pthread_mutex_init(&mutex, 0);
pthread_cond_init(&cond, 0);
+ if (flags & FAULT_INJECT_STRESS)
+ gt_reset_fault_injection(fd, true);
+
for (i = 0; i < n_threads; ++i) {
threads[i].mutex = &mutex;
threads[i].cond = &cond;
@@ -940,6 +1005,9 @@ gt_reset(int fd, int gt, int n_threads, int n_sec, unsigned int flags)
igt_info("number of resets %d, submissions %d, submit fails %d vm_recreate %d\n",
num_reset, num_submit, num_submit_fail, num_vm_recreate);
+ if (flags & FAULT_INJECT_STRESS)
+ gt_reset_fault_injection(fd, true);
+
igt_assert_neq(num_reset, 0);
igt_assert_neq(num_submit, 0);
free(threads);
@@ -1122,6 +1190,7 @@ int igt_main()
int gt;
int class;
int fd;
+ char pci_slot[NAME_MAX];
igt_fixture()
fd = drm_open_driver(DRIVER_XE);
@@ -1326,6 +1395,29 @@ int igt_main()
break;
}
+ igt_subtest("gt-reset-fault-injection") {
+ igt_require_f(igt_debugfs_exists(fd, "fail_gt_reset/probability",
+ O_RDWR),
+ "GT reset fault injection not available; "
+ "CONFIG_DRM_XE_KUNIT_TEST/fault-injection must be "
+ "enabled in the KMD\n");
+
+ igt_device_get_pci_slot_name(fd, pci_slot);
+ ignore_gt_reset_fault_dmesg();
+
+ gt_reset(fd, 0, 8, 2, FAULT_INJECT_STRESS);
+
+ igt_assert_f(try_vm_create(fd) != 0,
+ "Device did not wedge after injected GT reset failure\n");
+
+ drm_close_driver(fd);
+ igt_kmod_rebind("xe", pci_slot);
+ fd = drm_open_driver(DRIVER_XE);
+
+ igt_assert_f(try_vm_create(fd) == 0,
+ "Device not functional after rebind recovery\n");
+ }
+
igt_subtest("gt-mocs-reset")
xe_for_each_gt(fd, gt)
gt_mocs_reset(fd, gt);
@@ -1455,6 +1547,9 @@ int igt_main()
}
}
- igt_fixture()
+ igt_fixture() {
+ if (igt_debugfs_exists(fd, "fail_gt_reset/probability", O_RDWR))
+ gt_reset_fault_injection(fd, false);
drm_close_driver(fd);
+ }
}
--
2.43.0
^ permalink raw reply related [flat|nested] 10+ messages in thread
end of thread, other threads:[~2026-09-02 10:13 UTC | newest]
Thread overview: 10+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-01 14:01 [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage nishit.sharma
2026-09-01 18:13 ` Kamil Konieczny
2026-09-02 0:04 ` ✗ i915.CI.BAT: failure for tests/intel/xe_exec_reset: add GT reset fault injection stress coverage (rev4) Patchwork
-- strict thread matches above, loose matches on Subject: below --
2026-09-02 3:33 [PATCH] tests/intel/xe_exec_reset: add GT reset fault injection stress coverage nishit.sharma
2026-09-02 10:12 ` Kamil Konieczny
2026-09-01 3:44 nishit.sharma
2026-09-01 12:48 ` Kamil Konieczny
2026-08-25 3:25 nishit.sharma
2026-08-31 17:20 ` Kamil Konieczny
2026-08-19 7:36 nishit.sharma
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.