AMD-GFX Archive on lore.kernel.org
 help / color / mirror / Atom feed
* [PATCH v2 1/3] drm/amdgpu: gfx_v11_0_cp_ecc_error_irq_funcs is not needed any more
@ 2023-04-25  6:38 Horatio Zhang
  2023-04-25  6:38 ` [PATCH v2 2/3] drm/amdgpu: fix amdgpu_irq_put call trace in gmc_v11_0_hw_fini Horatio Zhang
                   ` (3 more replies)
  0 siblings, 4 replies; 6+ messages in thread
From: Horatio Zhang @ 2023-04-25  6:38 UTC (permalink / raw)
  To: hawking.zhang, christian.koenig, amd-gfx
  Cc: longlong.yao, feifei.xu, Horatio Zhang, Guchun.Chen

The gfx.cp_ecc_error_irq is retired in gfx11. In gfx_v11_0_hw_fini still
use amdgpu_irq_put to disable this interrupt, which caused the call trace
in this function.

[  102.873958] Call Trace:
[  102.873959]  <TASK>
[  102.873961]  gfx_v11_0_hw_fini+0x23/0x1e0 [amdgpu]
[  102.874019]  gfx_v11_0_suspend+0xe/0x20 [amdgpu]
[  102.874072]  amdgpu_device_ip_suspend_phase2+0x240/0x460 [amdgpu]
[  102.874122]  amdgpu_device_ip_suspend+0x3d/0x80 [amdgpu]
[  102.874172]  amdgpu_device_pre_asic_reset+0xd9/0x490 [amdgpu]
[  102.874223]  amdgpu_device_gpu_recover.cold+0x548/0xce6 [amdgpu]
[  102.874321]  amdgpu_debugfs_reset_work+0x4c/0x70 [amdgpu]
[  102.874375]  process_one_work+0x21f/0x3f0
[  102.874377]  worker_thread+0x200/0x3e0
[  102.874378]  ? process_one_work+0x3f0/0x3f0
[  102.874379]  kthread+0xfd/0x130
[  102.874380]  ? kthread_complete_and_exit+0x20/0x20
[  102.874381]  ret_from_fork+0x22/0x30

Signed-off-by: Horatio Zhang <Hongkun.Zhang@amd.com>
---
 drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c | 38 --------------------------
 1 file changed, 38 deletions(-)

diff --git a/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c b/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
index 8a4c4769e607..e9491aec3cae 100644
--- a/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
@@ -1355,13 +1355,6 @@ static int gfx_v11_0_sw_init(void *handle)
 	if (r)
 		return r;
 
-	/* ECC error */
-	r = amdgpu_irq_add_id(adev, SOC21_IH_CLIENTID_GRBM_CP,
-				  GFX_11_0_0__SRCID__CP_ECC_ERROR,
-				  &adev->gfx.cp_ecc_error_irq);
-	if (r)
-		return r;
-
 	/* FED error */
 	r = amdgpu_irq_add_id(adev, SOC21_IH_CLIENTID_GFX,
 				  GFX_11_0_0__SRCID__RLC_GC_FED_INTERRUPT,
@@ -4483,7 +4476,6 @@ static int gfx_v11_0_hw_fini(void *handle)
 	struct amdgpu_device *adev = (struct amdgpu_device *)handle;
 	int r;
 
-	amdgpu_irq_put(adev, &adev->gfx.cp_ecc_error_irq, 0);
 	amdgpu_irq_put(adev, &adev->gfx.priv_reg_irq, 0);
 	amdgpu_irq_put(adev, &adev->gfx.priv_inst_irq, 0);
 
@@ -5970,28 +5962,6 @@ static void gfx_v11_0_set_compute_eop_interrupt_state(struct amdgpu_device *adev
 		WREG32_SOC15_IP(GC, reg_addr, tmp); \
 	} while (0)
 
-static int gfx_v11_0_set_cp_ecc_error_state(struct amdgpu_device *adev,
-							struct amdgpu_irq_src *source,
-							unsigned type,
-							enum amdgpu_interrupt_state state)
-{
-	uint32_t ecc_irq_state = 0;
-	uint32_t pipe0_int_cntl_addr = 0;
-	int i = 0;
-
-	ecc_irq_state = (state == AMDGPU_IRQ_STATE_ENABLE) ? 1 : 0;
-
-	pipe0_int_cntl_addr = SOC15_REG_OFFSET(GC, 0, regCP_ME1_PIPE0_INT_CNTL);
-
-	WREG32_FIELD15_PREREG(GC, 0, CP_INT_CNTL_RING0, CP_ECC_ERROR_INT_ENABLE, ecc_irq_state);
-
-	for (i = 0; i < adev->gfx.mec.num_pipe_per_mec; i++)
-		SET_ECC_ME_PIPE_STATE(pipe0_int_cntl_addr + i * CP_ME1_PIPE_INST_ADDR_INTERVAL,
-					ecc_irq_state);
-
-	return 0;
-}
-
 static int gfx_v11_0_set_eop_interrupt_state(struct amdgpu_device *adev,
 					    struct amdgpu_irq_src *src,
 					    unsigned type,
@@ -6408,11 +6378,6 @@ static const struct amdgpu_irq_src_funcs gfx_v11_0_priv_inst_irq_funcs = {
 	.process = gfx_v11_0_priv_inst_irq,
 };
 
-static const struct amdgpu_irq_src_funcs gfx_v11_0_cp_ecc_error_irq_funcs = {
-	.set = gfx_v11_0_set_cp_ecc_error_state,
-	.process = amdgpu_gfx_cp_ecc_error_irq,
-};
-
 static const struct amdgpu_irq_src_funcs gfx_v11_0_rlc_gc_fed_irq_funcs = {
 	.process = gfx_v11_0_rlc_gc_fed_irq,
 };
@@ -6428,9 +6393,6 @@ static void gfx_v11_0_set_irq_funcs(struct amdgpu_device *adev)
 	adev->gfx.priv_inst_irq.num_types = 1;
 	adev->gfx.priv_inst_irq.funcs = &gfx_v11_0_priv_inst_irq_funcs;
 
-	adev->gfx.cp_ecc_error_irq.num_types = 1; /* CP ECC error */
-	adev->gfx.cp_ecc_error_irq.funcs = &gfx_v11_0_cp_ecc_error_irq_funcs;
-
 	adev->gfx.rlc_gc_fed_irq.num_types = 1; /* 0x80 FED error */
 	adev->gfx.rlc_gc_fed_irq.funcs = &gfx_v11_0_rlc_gc_fed_irq_funcs;
 
-- 
2.34.1


^ permalink raw reply related	[flat|nested] 6+ messages in thread

* [PATCH v2 2/3] drm/amdgpu: fix amdgpu_irq_put call trace in gmc_v11_0_hw_fini
  2023-04-25  6:38 [PATCH v2 1/3] drm/amdgpu: gfx_v11_0_cp_ecc_error_irq_funcs is not needed any more Horatio Zhang
@ 2023-04-25  6:38 ` Horatio Zhang
  2023-04-25  6:38 ` [PATCH v2 3/3] drm/amdgpu: fix amdgpu_irq_put call trace in gmc_v10_0_hw_fini Horatio Zhang
                   ` (2 subsequent siblings)
  3 siblings, 0 replies; 6+ messages in thread
From: Horatio Zhang @ 2023-04-25  6:38 UTC (permalink / raw)
  To: hawking.zhang, christian.koenig, amd-gfx
  Cc: longlong.yao, feifei.xu, Horatio Zhang, Guchun.Chen

The gmc.ecc_irq is enabled by firmware per IFWI setting,
and the host driver is not privileged to enable/disable
the interrupt. So, it is meaningless to use the amdgpu_irq_put
function in gmc_v11_0_hw_fini, which also leads to the call
trace.

[  102.980303] Call Trace:
[  102.980303]  <TASK>
[  102.980304]  gmc_v11_0_hw_fini+0x54/0x90 [amdgpu]
[  102.980357]  gmc_v11_0_suspend+0xe/0x20 [amdgpu]
[  102.980409]  amdgpu_device_ip_suspend_phase2+0x240/0x460 [amdgpu]
[  102.980459]  amdgpu_device_ip_suspend+0x3d/0x80 [amdgpu]
[  102.980520]  amdgpu_device_pre_asic_reset+0xd9/0x490 [amdgpu]
[  102.980573]  amdgpu_device_gpu_recover.cold+0x548/0xce6 [amdgpu]
[  102.980687]  amdgpu_debugfs_reset_work+0x4c/0x70 [amdgpu]
[  102.980740]  process_one_work+0x21f/0x3f0
[  102.980741]  worker_thread+0x200/0x3e0
[  102.980742]  ? process_one_work+0x3f0/0x3f0
[  102.980743]  kthread+0xfd/0x130
[  102.980743]  ? kthread_complete_and_exit+0x20/0x20
[  102.980744]  ret_from_fork+0x22/0x30

Signed-off-by: Horatio Zhang <Hongkun.Zhang@amd.com>
---
 drivers/gpu/drm/amd/amdgpu/gmc_v11_0.c | 1 -
 1 file changed, 1 deletion(-)

diff --git a/drivers/gpu/drm/amd/amdgpu/gmc_v11_0.c b/drivers/gpu/drm/amd/amdgpu/gmc_v11_0.c
index 3828ca95899f..f73c238f3145 100644
--- a/drivers/gpu/drm/amd/amdgpu/gmc_v11_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/gmc_v11_0.c
@@ -951,7 +951,6 @@ static int gmc_v11_0_hw_fini(void *handle)
 		return 0;
 	}
 
-	amdgpu_irq_put(adev, &adev->gmc.ecc_irq, 0);
 	amdgpu_irq_put(adev, &adev->gmc.vm_fault, 0);
 	gmc_v11_0_gart_disable(adev);
 
-- 
2.34.1


^ permalink raw reply related	[flat|nested] 6+ messages in thread

* [PATCH v2 3/3] drm/amdgpu: fix amdgpu_irq_put call trace in gmc_v10_0_hw_fini
  2023-04-25  6:38 [PATCH v2 1/3] drm/amdgpu: gfx_v11_0_cp_ecc_error_irq_funcs is not needed any more Horatio Zhang
  2023-04-25  6:38 ` [PATCH v2 2/3] drm/amdgpu: fix amdgpu_irq_put call trace in gmc_v11_0_hw_fini Horatio Zhang
@ 2023-04-25  6:38 ` Horatio Zhang
  2023-04-25  8:35   ` Zhang, Hawking
  2023-04-25  8:56 ` [PATCH v2 1/3] drm/amdgpu: gfx_v11_0_cp_ecc_error_irq_funcs is not needed any more Christian König
  2023-04-25  9:04 ` Chen, Guchun
  3 siblings, 1 reply; 6+ messages in thread
From: Horatio Zhang @ 2023-04-25  6:38 UTC (permalink / raw)
  To: hawking.zhang, christian.koenig, amd-gfx
  Cc: longlong.yao, feifei.xu, Horatio Zhang, Guchun.Chen

The gmc.ecc_irq is enabled by firmware per IFWI setting,
and the host driver is not privileged to enable/disable
the interrupt. So, it is meaningless to use the amdgpu_irq_put
function in gmc_v10_0_hw_fini, which also leads to the call
trace.

[   82.340264] Call Trace:
[   82.340265]  <TASK>
[   82.340269]  gmc_v10_0_hw_fini+0x83/0xa0 [amdgpu]
[   82.340447]  gmc_v10_0_suspend+0xe/0x20 [amdgpu]
[   82.340623]  amdgpu_device_ip_suspend_phase2+0x127/0x1c0 [amdgpu]
[   82.340789]  amdgpu_device_ip_suspend+0x3d/0x80 [amdgpu]
[   82.340955]  amdgpu_device_pre_asic_reset+0xdd/0x2b0 [amdgpu]
[   82.341122]  amdgpu_device_gpu_recover.cold+0x4dd/0xbb2 [amdgpu]
[   82.341359]  amdgpu_debugfs_reset_work+0x4c/0x70 [amdgpu]
[   82.341529]  process_one_work+0x21d/0x3f0
[   82.341535]  worker_thread+0x1fa/0x3c0
[   82.341538]  ? process_one_work+0x3f0/0x3f0
[   82.341540]  kthread+0xff/0x130
[   82.341544]  ? kthread_complete_and_exit+0x20/0x20
[   82.341547]  ret_from_fork+0x22/0x30

Signed-off-by: Horatio Zhang <Hongkun.Zhang@amd.com>
---
 drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c | 1 -
 1 file changed, 1 deletion(-)

diff --git a/drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c b/drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c
index 23d4081eca00..5697b66bf0de 100644
--- a/drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c
@@ -1143,7 +1143,6 @@ static int gmc_v10_0_hw_fini(void *handle)
 		return 0;
 	}
 
-	amdgpu_irq_put(adev, &adev->gmc.ecc_irq, 0);
 	amdgpu_irq_put(adev, &adev->gmc.vm_fault, 0);
 
 	return 0;
-- 
2.34.1


^ permalink raw reply related	[flat|nested] 6+ messages in thread

* Re: [PATCH v2 3/3] drm/amdgpu: fix amdgpu_irq_put call trace in gmc_v10_0_hw_fini
  2023-04-25  6:38 ` [PATCH v2 3/3] drm/amdgpu: fix amdgpu_irq_put call trace in gmc_v10_0_hw_fini Horatio Zhang
@ 2023-04-25  8:35   ` Zhang, Hawking
  0 siblings, 0 replies; 6+ messages in thread
From: Zhang, Hawking @ 2023-04-25  8:35 UTC (permalink / raw)
  To: Zhang, Horatio, Koenig, Christian, amd-gfx@lists.freedesktop.org
  Cc: Yao, Longlong, Xu, Feifei, Zhang, Horatio, Chen, Guchun

[-- Attachment #1: Type: text/plain, Size: 2226 bytes --]

[AMD Official Use Only - General]

Series is

Reviewed-by: Hawking Zhang <Hawking.Zhang@amd.com>

Regards,
Hawking
From: Horatio Zhang <Hongkun.Zhang@amd.com>
Date: Tuesday, April 25, 2023 at 14:38
To: Zhang, Hawking <Hawking.Zhang@amd.com>, Koenig, Christian <Christian.Koenig@amd.com>, amd-gfx@lists.freedesktop.org <amd-gfx@lists.freedesktop.org>
Cc: Chen, Guchun <Guchun.Chen@amd.com>, Xu, Feifei <Feifei.Xu@amd.com>, Yao, Longlong <Longlong.Yao@amd.com>, Zhang, Horatio <Hongkun.Zhang@amd.com>
Subject: [PATCH v2 3/3] drm/amdgpu: fix amdgpu_irq_put call trace in gmc_v10_0_hw_fini
The gmc.ecc_irq is enabled by firmware per IFWI setting,
and the host driver is not privileged to enable/disable
the interrupt. So, it is meaningless to use the amdgpu_irq_put
function in gmc_v10_0_hw_fini, which also leads to the call
trace.

[   82.340264] Call Trace:
[   82.340265]  <TASK>
[   82.340269]  gmc_v10_0_hw_fini+0x83/0xa0 [amdgpu]
[   82.340447]  gmc_v10_0_suspend+0xe/0x20 [amdgpu]
[   82.340623]  amdgpu_device_ip_suspend_phase2+0x127/0x1c0 [amdgpu]
[   82.340789]  amdgpu_device_ip_suspend+0x3d/0x80 [amdgpu]
[   82.340955]  amdgpu_device_pre_asic_reset+0xdd/0x2b0 [amdgpu]
[   82.341122]  amdgpu_device_gpu_recover.cold+0x4dd/0xbb2 [amdgpu]
[   82.341359]  amdgpu_debugfs_reset_work+0x4c/0x70 [amdgpu]
[   82.341529]  process_one_work+0x21d/0x3f0
[   82.341535]  worker_thread+0x1fa/0x3c0
[   82.341538]  ? process_one_work+0x3f0/0x3f0
[   82.341540]  kthread+0xff/0x130
[   82.341544]  ? kthread_complete_and_exit+0x20/0x20
[   82.341547]  ret_from_fork+0x22/0x30

Signed-off-by: Horatio Zhang <Hongkun.Zhang@amd.com>
---
 drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c | 1 -
 1 file changed, 1 deletion(-)

diff --git a/drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c b/drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c
index 23d4081eca00..5697b66bf0de 100644
--- a/drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c
@@ -1143,7 +1143,6 @@ static int gmc_v10_0_hw_fini(void *handle)
                 return 0;
         }

-       amdgpu_irq_put(adev, &adev->gmc.ecc_irq, 0);
         amdgpu_irq_put(adev, &adev->gmc.vm_fault, 0);

         return 0;
--
2.34.1

[-- Attachment #2: Type: text/html, Size: 5362 bytes --]

^ permalink raw reply related	[flat|nested] 6+ messages in thread

* Re: [PATCH v2 1/3] drm/amdgpu: gfx_v11_0_cp_ecc_error_irq_funcs is not needed any more
  2023-04-25  6:38 [PATCH v2 1/3] drm/amdgpu: gfx_v11_0_cp_ecc_error_irq_funcs is not needed any more Horatio Zhang
  2023-04-25  6:38 ` [PATCH v2 2/3] drm/amdgpu: fix amdgpu_irq_put call trace in gmc_v11_0_hw_fini Horatio Zhang
  2023-04-25  6:38 ` [PATCH v2 3/3] drm/amdgpu: fix amdgpu_irq_put call trace in gmc_v10_0_hw_fini Horatio Zhang
@ 2023-04-25  8:56 ` Christian König
  2023-04-25  9:04 ` Chen, Guchun
  3 siblings, 0 replies; 6+ messages in thread
From: Christian König @ 2023-04-25  8:56 UTC (permalink / raw)
  To: Horatio Zhang, hawking.zhang, amd-gfx
  Cc: longlong.yao, feifei.xu, Guchun.Chen

Am 25.04.23 um 08:38 schrieb Horatio Zhang:
> The gfx.cp_ecc_error_irq is retired in gfx11. In gfx_v11_0_hw_fini still
> use amdgpu_irq_put to disable this interrupt, which caused the call trace
> in this function.
>
> [  102.873958] Call Trace:
> [  102.873959]  <TASK>
> [  102.873961]  gfx_v11_0_hw_fini+0x23/0x1e0 [amdgpu]
> [  102.874019]  gfx_v11_0_suspend+0xe/0x20 [amdgpu]
> [  102.874072]  amdgpu_device_ip_suspend_phase2+0x240/0x460 [amdgpu]
> [  102.874122]  amdgpu_device_ip_suspend+0x3d/0x80 [amdgpu]
> [  102.874172]  amdgpu_device_pre_asic_reset+0xd9/0x490 [amdgpu]
> [  102.874223]  amdgpu_device_gpu_recover.cold+0x548/0xce6 [amdgpu]
> [  102.874321]  amdgpu_debugfs_reset_work+0x4c/0x70 [amdgpu]
> [  102.874375]  process_one_work+0x21f/0x3f0
> [  102.874377]  worker_thread+0x200/0x3e0
> [  102.874378]  ? process_one_work+0x3f0/0x3f0
> [  102.874379]  kthread+0xfd/0x130
> [  102.874380]  ? kthread_complete_and_exit+0x20/0x20
> [  102.874381]  ret_from_fork+0x22/0x30
>
> Signed-off-by: Horatio Zhang <Hongkun.Zhang@amd.com>

This goes deeper than my understanding of the hw so I can't fully judge 
if this is correct or not.

But from the general idea looks good to me, so feel free to add my Acked-by.

Regards,
Christian.

> ---
>   drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c | 38 --------------------------
>   1 file changed, 38 deletions(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c b/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
> index 8a4c4769e607..e9491aec3cae 100644
> --- a/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
> +++ b/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
> @@ -1355,13 +1355,6 @@ static int gfx_v11_0_sw_init(void *handle)
>   	if (r)
>   		return r;
>   
> -	/* ECC error */
> -	r = amdgpu_irq_add_id(adev, SOC21_IH_CLIENTID_GRBM_CP,
> -				  GFX_11_0_0__SRCID__CP_ECC_ERROR,
> -				  &adev->gfx.cp_ecc_error_irq);
> -	if (r)
> -		return r;
> -
>   	/* FED error */
>   	r = amdgpu_irq_add_id(adev, SOC21_IH_CLIENTID_GFX,
>   				  GFX_11_0_0__SRCID__RLC_GC_FED_INTERRUPT,
> @@ -4483,7 +4476,6 @@ static int gfx_v11_0_hw_fini(void *handle)
>   	struct amdgpu_device *adev = (struct amdgpu_device *)handle;
>   	int r;
>   
> -	amdgpu_irq_put(adev, &adev->gfx.cp_ecc_error_irq, 0);
>   	amdgpu_irq_put(adev, &adev->gfx.priv_reg_irq, 0);
>   	amdgpu_irq_put(adev, &adev->gfx.priv_inst_irq, 0);
>   
> @@ -5970,28 +5962,6 @@ static void gfx_v11_0_set_compute_eop_interrupt_state(struct amdgpu_device *adev
>   		WREG32_SOC15_IP(GC, reg_addr, tmp); \
>   	} while (0)
>   
> -static int gfx_v11_0_set_cp_ecc_error_state(struct amdgpu_device *adev,
> -							struct amdgpu_irq_src *source,
> -							unsigned type,
> -							enum amdgpu_interrupt_state state)
> -{
> -	uint32_t ecc_irq_state = 0;
> -	uint32_t pipe0_int_cntl_addr = 0;
> -	int i = 0;
> -
> -	ecc_irq_state = (state == AMDGPU_IRQ_STATE_ENABLE) ? 1 : 0;
> -
> -	pipe0_int_cntl_addr = SOC15_REG_OFFSET(GC, 0, regCP_ME1_PIPE0_INT_CNTL);
> -
> -	WREG32_FIELD15_PREREG(GC, 0, CP_INT_CNTL_RING0, CP_ECC_ERROR_INT_ENABLE, ecc_irq_state);
> -
> -	for (i = 0; i < adev->gfx.mec.num_pipe_per_mec; i++)
> -		SET_ECC_ME_PIPE_STATE(pipe0_int_cntl_addr + i * CP_ME1_PIPE_INST_ADDR_INTERVAL,
> -					ecc_irq_state);
> -
> -	return 0;
> -}
> -
>   static int gfx_v11_0_set_eop_interrupt_state(struct amdgpu_device *adev,
>   					    struct amdgpu_irq_src *src,
>   					    unsigned type,
> @@ -6408,11 +6378,6 @@ static const struct amdgpu_irq_src_funcs gfx_v11_0_priv_inst_irq_funcs = {
>   	.process = gfx_v11_0_priv_inst_irq,
>   };
>   
> -static const struct amdgpu_irq_src_funcs gfx_v11_0_cp_ecc_error_irq_funcs = {
> -	.set = gfx_v11_0_set_cp_ecc_error_state,
> -	.process = amdgpu_gfx_cp_ecc_error_irq,
> -};
> -
>   static const struct amdgpu_irq_src_funcs gfx_v11_0_rlc_gc_fed_irq_funcs = {
>   	.process = gfx_v11_0_rlc_gc_fed_irq,
>   };
> @@ -6428,9 +6393,6 @@ static void gfx_v11_0_set_irq_funcs(struct amdgpu_device *adev)
>   	adev->gfx.priv_inst_irq.num_types = 1;
>   	adev->gfx.priv_inst_irq.funcs = &gfx_v11_0_priv_inst_irq_funcs;
>   
> -	adev->gfx.cp_ecc_error_irq.num_types = 1; /* CP ECC error */
> -	adev->gfx.cp_ecc_error_irq.funcs = &gfx_v11_0_cp_ecc_error_irq_funcs;
> -
>   	adev->gfx.rlc_gc_fed_irq.num_types = 1; /* 0x80 FED error */
>   	adev->gfx.rlc_gc_fed_irq.funcs = &gfx_v11_0_rlc_gc_fed_irq_funcs;
>   


^ permalink raw reply	[flat|nested] 6+ messages in thread

* RE: [PATCH v2 1/3] drm/amdgpu: gfx_v11_0_cp_ecc_error_irq_funcs is not needed any more
  2023-04-25  6:38 [PATCH v2 1/3] drm/amdgpu: gfx_v11_0_cp_ecc_error_irq_funcs is not needed any more Horatio Zhang
                   ` (2 preceding siblings ...)
  2023-04-25  8:56 ` [PATCH v2 1/3] drm/amdgpu: gfx_v11_0_cp_ecc_error_irq_funcs is not needed any more Christian König
@ 2023-04-25  9:04 ` Chen, Guchun
  3 siblings, 0 replies; 6+ messages in thread
From: Chen, Guchun @ 2023-04-25  9:04 UTC (permalink / raw)
  To: Zhang, Horatio, Zhang, Hawking, Koenig, Christian,
	amd-gfx@lists.freedesktop.org
  Cc: Yao, Longlong, Xu, Feifei, Zhang, Horatio

I guess it's more simple if updating the subject to "drm/amdgpu: drop gfx_v11_0_cp_ecc_error_irq_funcs"

With this improved, the series are:
Reviewed-by: Guchun Chen <guchun.chen@amd.com>

Regards,
Guchun

> -----Original Message-----
> From: Horatio Zhang <Hongkun.Zhang@amd.com>
> Sent: Tuesday, April 25, 2023 2:39 PM
> To: Zhang, Hawking <Hawking.Zhang@amd.com>; Koenig, Christian
> <Christian.Koenig@amd.com>; amd-gfx@lists.freedesktop.org
> Cc: Chen, Guchun <Guchun.Chen@amd.com>; Xu, Feifei
> <Feifei.Xu@amd.com>; Yao, Longlong <Longlong.Yao@amd.com>; Zhang,
> Horatio <Hongkun.Zhang@amd.com>
> Subject: [PATCH v2 1/3] drm/amdgpu: gfx_v11_0_cp_ecc_error_irq_funcs is
> not needed any more
> 
> The gfx.cp_ecc_error_irq is retired in gfx11. In gfx_v11_0_hw_fini still use
> amdgpu_irq_put to disable this interrupt, which caused the call trace in this
> function.
> 
> [  102.873958] Call Trace:
> [  102.873959]  <TASK>
> [  102.873961]  gfx_v11_0_hw_fini+0x23/0x1e0 [amdgpu] [  102.874019]
> gfx_v11_0_suspend+0xe/0x20 [amdgpu] [  102.874072]
> amdgpu_device_ip_suspend_phase2+0x240/0x460 [amdgpu] [  102.874122]
> amdgpu_device_ip_suspend+0x3d/0x80 [amdgpu] [  102.874172]
> amdgpu_device_pre_asic_reset+0xd9/0x490 [amdgpu] [  102.874223]
> amdgpu_device_gpu_recover.cold+0x548/0xce6 [amdgpu] [  102.874321]
> amdgpu_debugfs_reset_work+0x4c/0x70 [amdgpu] [  102.874375]
> process_one_work+0x21f/0x3f0 [  102.874377]  worker_thread+0x200/0x3e0
> [  102.874378]  ? process_one_work+0x3f0/0x3f0 [  102.874379]
> kthread+0xfd/0x130 [  102.874380]  ?
> kthread_complete_and_exit+0x20/0x20
> [  102.874381]  ret_from_fork+0x22/0x30
> 
> Signed-off-by: Horatio Zhang <Hongkun.Zhang@amd.com>
> ---
>  drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c | 38 --------------------------
>  1 file changed, 38 deletions(-)
> 
> diff --git a/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
> b/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
> index 8a4c4769e607..e9491aec3cae 100644
> --- a/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
> +++ b/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
> @@ -1355,13 +1355,6 @@ static int gfx_v11_0_sw_init(void *handle)
>  	if (r)
>  		return r;
> 
> -	/* ECC error */
> -	r = amdgpu_irq_add_id(adev, SOC21_IH_CLIENTID_GRBM_CP,
> -				  GFX_11_0_0__SRCID__CP_ECC_ERROR,
> -				  &adev->gfx.cp_ecc_error_irq);
> -	if (r)
> -		return r;
> -
>  	/* FED error */
>  	r = amdgpu_irq_add_id(adev, SOC21_IH_CLIENTID_GFX,
> 
> GFX_11_0_0__SRCID__RLC_GC_FED_INTERRUPT,
> @@ -4483,7 +4476,6 @@ static int gfx_v11_0_hw_fini(void *handle)
>  	struct amdgpu_device *adev = (struct amdgpu_device *)handle;
>  	int r;
> 
> -	amdgpu_irq_put(adev, &adev->gfx.cp_ecc_error_irq, 0);
>  	amdgpu_irq_put(adev, &adev->gfx.priv_reg_irq, 0);
>  	amdgpu_irq_put(adev, &adev->gfx.priv_inst_irq, 0);
> 
> @@ -5970,28 +5962,6 @@ static void
> gfx_v11_0_set_compute_eop_interrupt_state(struct amdgpu_device *adev
>  		WREG32_SOC15_IP(GC, reg_addr, tmp); \
>  	} while (0)
> 
> -static int gfx_v11_0_set_cp_ecc_error_state(struct amdgpu_device *adev,
> -							struct
> amdgpu_irq_src *source,
> -							unsigned type,
> -							enum
> amdgpu_interrupt_state state)
> -{
> -	uint32_t ecc_irq_state = 0;
> -	uint32_t pipe0_int_cntl_addr = 0;
> -	int i = 0;
> -
> -	ecc_irq_state = (state == AMDGPU_IRQ_STATE_ENABLE) ? 1 : 0;
> -
> -	pipe0_int_cntl_addr = SOC15_REG_OFFSET(GC, 0,
> regCP_ME1_PIPE0_INT_CNTL);
> -
> -	WREG32_FIELD15_PREREG(GC, 0, CP_INT_CNTL_RING0,
> CP_ECC_ERROR_INT_ENABLE, ecc_irq_state);
> -
> -	for (i = 0; i < adev->gfx.mec.num_pipe_per_mec; i++)
> -		SET_ECC_ME_PIPE_STATE(pipe0_int_cntl_addr + i *
> CP_ME1_PIPE_INST_ADDR_INTERVAL,
> -					ecc_irq_state);
> -
> -	return 0;
> -}
> -
>  static int gfx_v11_0_set_eop_interrupt_state(struct amdgpu_device *adev,
>  					    struct amdgpu_irq_src *src,
>  					    unsigned type,
> @@ -6408,11 +6378,6 @@ static const struct amdgpu_irq_src_funcs
> gfx_v11_0_priv_inst_irq_funcs = {
>  	.process = gfx_v11_0_priv_inst_irq,
>  };
> 
> -static const struct amdgpu_irq_src_funcs gfx_v11_0_cp_ecc_error_irq_funcs
> = {
> -	.set = gfx_v11_0_set_cp_ecc_error_state,
> -	.process = amdgpu_gfx_cp_ecc_error_irq,
> -};
> -
>  static const struct amdgpu_irq_src_funcs gfx_v11_0_rlc_gc_fed_irq_funcs = {
>  	.process = gfx_v11_0_rlc_gc_fed_irq,
>  };
> @@ -6428,9 +6393,6 @@ static void gfx_v11_0_set_irq_funcs(struct
> amdgpu_device *adev)
>  	adev->gfx.priv_inst_irq.num_types = 1;
>  	adev->gfx.priv_inst_irq.funcs = &gfx_v11_0_priv_inst_irq_funcs;
> 
> -	adev->gfx.cp_ecc_error_irq.num_types = 1; /* CP ECC error */
> -	adev->gfx.cp_ecc_error_irq.funcs =
> &gfx_v11_0_cp_ecc_error_irq_funcs;
> -
>  	adev->gfx.rlc_gc_fed_irq.num_types = 1; /* 0x80 FED error */
>  	adev->gfx.rlc_gc_fed_irq.funcs = &gfx_v11_0_rlc_gc_fed_irq_funcs;
> 
> --
> 2.34.1


^ permalink raw reply	[flat|nested] 6+ messages in thread

end of thread, other threads:[~2023-04-25  9:04 UTC | newest]

Thread overview: 6+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2023-04-25  6:38 [PATCH v2 1/3] drm/amdgpu: gfx_v11_0_cp_ecc_error_irq_funcs is not needed any more Horatio Zhang
2023-04-25  6:38 ` [PATCH v2 2/3] drm/amdgpu: fix amdgpu_irq_put call trace in gmc_v11_0_hw_fini Horatio Zhang
2023-04-25  6:38 ` [PATCH v2 3/3] drm/amdgpu: fix amdgpu_irq_put call trace in gmc_v10_0_hw_fini Horatio Zhang
2023-04-25  8:35   ` Zhang, Hawking
2023-04-25  8:56 ` [PATCH v2 1/3] drm/amdgpu: gfx_v11_0_cp_ecc_error_irq_funcs is not needed any more Christian König
2023-04-25  9:04 ` Chen, Guchun

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox