From: "Christian König" <ckoenig.leichtzumerken-Re5JQEeQqe8AvxtiuMwx3w@public.gmane.org>
To: Emily Deng <Emily.Deng-5C7GfCeVMHo@public.gmane.org>,
amd-gfx-PD4FTy7X32lNgt0PjOBp9y5qC8QIuHrW@public.gmane.org
Cc: Monk Liu <Monk.Liu-5C7GfCeVMHo@public.gmane.org>
Subject: Re: [PATCH] drm/amdgpu: fix a kcq hang issue for SRIOV
Date: Tue, 27 Mar 2018 09:48:07 +0200 [thread overview]
Message-ID: <04acc6d7-680b-861d-2a3e-e4206b72345c@gmail.com> (raw)
In-Reply-To: <1522130286-25401-1-git-send-email-Emily.Deng-5C7GfCeVMHo@public.gmane.org>
Am 27.03.2018 um 07:58 schrieb Emily Deng:
> issue:
> the vmflush in KCQ could be preempted (not like GFX ring
> which doesn't allow preemption in ring buffer) and this lead
> to vm flush fail when there is a world switch during
> the vm flush procedure (between write invalidate request
> and query invalidate ack)
>
> fix:
> separate vm flush for gfx and compute ring, and use
> the new format command in compute's vm flush which
> use only one package so no preemption could allowed
NAK, as already discussed multiple times now that only circumvents the
problem, but not really fixes it.
Just executing the "amdgpu_ring_emit_wreg(ring, hub->vm_inv_eng0_req +
eng, req);" multiple times has the same effect and we need to figure out
why.
Regards,
Christian.
>
> Signed-off-by: Monk Liu <Monk.Liu@amd.com>
> Signed-off-by: Emily Deng <Emily.Deng@amd.com>
> ---
> drivers/gpu/drm/amd/amdgpu/amdgpu.h | 1 +
> drivers/gpu/drm/amd/amdgpu/amdgpu_ring.h | 2 ++
> drivers/gpu/drm/amd/amdgpu/gfx_v9_0.c | 10 +++++++++-
> drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c | 18 +++++++++++++-----
> 4 files changed, 25 insertions(+), 6 deletions(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu.h b/drivers/gpu/drm/amd/amdgpu/amdgpu.h
> index a7e2229..986659f 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu.h
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu.h
> @@ -1790,6 +1790,7 @@ amdgpu_get_sdma_instance(struct amdgpu_ring *ring)
> #define amdgpu_ring_emit_rreg(r, d) (r)->funcs->emit_rreg((r), (d))
> #define amdgpu_ring_emit_wreg(r, d, v) (r)->funcs->emit_wreg((r), (d), (v))
> #define amdgpu_ring_emit_reg_wait(r, d, v, m) (r)->funcs->emit_reg_wait((r), (d), (v), (m))
> +#define amdgpu_ring_emit_reg_wait1(r, d0, d1, v, m) (r)->funcs->emit_reg_wait1((r), (d0), (d1), (v), (m))
> #define amdgpu_ring_emit_tmz(r, b) (r)->funcs->emit_tmz((r), (b))
> #define amdgpu_ring_pad_ib(r, ib) ((r)->funcs->pad_ib((r), (ib)))
> #define amdgpu_ring_init_cond_exec(r) (r)->funcs->init_cond_exec((r))
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ring.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_ring.h
> index 1d0d250..d85df5d 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ring.h
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ring.h
> @@ -152,6 +152,8 @@ struct amdgpu_ring_funcs {
> void (*emit_wreg)(struct amdgpu_ring *ring, uint32_t reg, uint32_t val);
> void (*emit_reg_wait)(struct amdgpu_ring *ring, uint32_t reg,
> uint32_t val, uint32_t mask);
> + void (*emit_reg_wait1)(struct amdgpu_ring *ring, uint32_t reg0,
> + uint32_t reg1, uint32_t val, uint32_t mask);
> void (*emit_tmz)(struct amdgpu_ring *ring, bool start);
> /* priority functions */
> void (*set_priority) (struct amdgpu_ring *ring,
> diff --git a/drivers/gpu/drm/amd/amdgpu/gfx_v9_0.c b/drivers/gpu/drm/amd/amdgpu/gfx_v9_0.c
> index 1ae3de1..509c9d2 100644
> --- a/drivers/gpu/drm/amd/amdgpu/gfx_v9_0.c
> +++ b/drivers/gpu/drm/amd/amdgpu/gfx_v9_0.c
> @@ -4078,6 +4078,13 @@ static void gfx_v9_0_ring_emit_reg_wait(struct amdgpu_ring *ring, uint32_t reg,
> gfx_v9_0_wait_reg_mem(ring, 0, 0, 0, reg, 0, val, mask, 0x20);
> }
>
> +static void gfx_v9_0_ring_emit_reg_wait_compute(struct amdgpu_ring *ring,
> + uint32_t reg0, uint32_t reg1,
> + uint32_t val, uint32_t mask)
> +{
> + gfx_v9_0_wait_reg_mem(ring, 0, 0, 1, reg0, reg1, val, mask, 0x20);
> +}
> +
> static void gfx_v9_0_set_gfx_eop_interrupt_state(struct amdgpu_device *adev,
> enum amdgpu_interrupt_state state)
> {
> @@ -4415,7 +4422,7 @@ static const struct amdgpu_ring_funcs gfx_v9_0_ring_funcs_compute = {
> 7 + /* gfx_v9_0_ring_emit_hdp_flush */
> 5 + /* hdp invalidate */
> 7 + /* gfx_v9_0_ring_emit_pipeline_sync */
> - SOC15_FLUSH_GPU_TLB_NUM_WREG * 5 +
> + (SOC15_FLUSH_GPU_TLB_NUM_WREG - 1) * 5 +
> SOC15_FLUSH_GPU_TLB_NUM_REG_WAIT * 7 +
> 2 + /* gfx_v9_0_ring_emit_vm_flush */
> 8 + 8 + 8, /* gfx_v9_0_ring_emit_fence x3 for user fence, vm fence */
> @@ -4433,6 +4440,7 @@ static const struct amdgpu_ring_funcs gfx_v9_0_ring_funcs_compute = {
> .set_priority = gfx_v9_0_ring_set_priority_compute,
> .emit_wreg = gfx_v9_0_ring_emit_wreg,
> .emit_reg_wait = gfx_v9_0_ring_emit_reg_wait,
> + .emit_reg_wait1 = gfx_v9_0_ring_emit_reg_wait_compute,
> };
>
> static const struct amdgpu_ring_funcs gfx_v9_0_ring_funcs_kiq = {
> diff --git a/drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c b/drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c
> index e687363..968447d 100644
> --- a/drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c
> +++ b/drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c
> @@ -385,11 +385,19 @@ static uint64_t gmc_v9_0_emit_flush_gpu_tlb(struct amdgpu_ring *ring,
> amdgpu_ring_emit_wreg(ring, hub->ctx0_ptb_addr_hi32 + (2 * vmid),
> upper_32_bits(pd_addr));
>
> - amdgpu_ring_emit_wreg(ring, hub->vm_inv_eng0_req + eng, req);
> -
> - /* wait for the invalidate to complete */
> - amdgpu_ring_emit_reg_wait(ring, hub->vm_inv_eng0_ack + eng,
> - 1 << vmid, 1 << vmid);
> + /* The world switch cannot be allowed to occur while
> + some invalidation controller code is waiting for an ack.
> + To workaround the hardware restriction, replace the original
> + two command with one command for compute ring */
> + if (ring->funcs->type == AMDGPU_RING_TYPE_COMPUTE && amdgpu_sriov_vf(adev)) {
> + amdgpu_ring_emit_reg_wait1(ring, hub->vm_inv_eng0_req + eng,
> + hub->vm_inv_eng0_ack + eng, req, 1 << vmid);
> + } else {
> + amdgpu_ring_emit_wreg(ring, hub->vm_inv_eng0_req + eng, req);
> + /* wait for the invalidate to complete */
> + amdgpu_ring_emit_reg_wait(ring, hub->vm_inv_eng0_ack + eng,
> + 1 << vmid, 1 << vmid);
> + }
>
> return pd_addr;
> }
_______________________________________________
amd-gfx mailing list
amd-gfx@lists.freedesktop.org
https://lists.freedesktop.org/mailman/listinfo/amd-gfx
next prev parent reply other threads:[~2018-03-27 7:48 UTC|newest]
Thread overview: 17+ messages / expand[flat|nested] mbox.gz Atom feed top
2018-03-27 5:58 [PATCH] drm/amdgpu: fix a kcq hang issue for SRIOV Emily Deng
[not found] ` <1522130286-25401-1-git-send-email-Emily.Deng-5C7GfCeVMHo@public.gmane.org>
2018-03-27 7:48 ` Christian König [this message]
[not found] ` <04acc6d7-680b-861d-2a3e-e4206b72345c-Re5JQEeQqe8AvxtiuMwx3w@public.gmane.org>
2018-03-27 8:23 ` Deng, Emily
[not found] ` <CY4PR12MB1125AE6F5482FD386603A1188FAC0-rpdhrqHFk07v2MZdTKcfDgdYzm3356FpvxpqHgZTriW3zl9H0oFU5g@public.gmane.org>
2018-03-27 8:26 ` Christian König
[not found] ` <221c0352-0e9d-5f11-24a5-49842f1a3fa4-Re5JQEeQqe8AvxtiuMwx3w@public.gmane.org>
2018-03-27 8:31 ` Deng, Emily
2018-03-27 15:15 ` Alex Deucher
2018-03-27 15:26 ` Alex Deucher
2018-03-27 15:37 ` Alex Deucher
[not found] ` <CADnq5_MaXurcEuMw2d_jYnb2_4iL9xLy7Z4VYgW6pr9Kcm7nKg-JsoAwUIsXosN+BqQ9rBEUg@public.gmane.org>
2018-03-27 15:43 ` Christian König
[not found] ` <183c370b-bcf5-fd80-429e-8418f2b1a105-Re5JQEeQqe8AvxtiuMwx3w@public.gmane.org>
2018-03-27 15:52 ` Alex Deucher
[not found] ` <CADnq5_MPtDiJ3vx-yqWfa65xb3etO-e1xRuiM9mJR1bON=5Ecw-JsoAwUIsXosN+BqQ9rBEUg@public.gmane.org>
2018-03-27 16:30 ` Christian König
[not found] ` <97d802df-4563-7a6a-eae4-d14313762015-5C7GfCeVMHo@public.gmane.org>
2018-03-27 16:56 ` Alex Deucher
[not found] ` <CADnq5_OXS2tA4VEJ3Jo-2py-2gOCFphODDUeUt_wR=uZ51_XYA-JsoAwUIsXosN+BqQ9rBEUg@public.gmane.org>
2018-03-27 17:18 ` Christian König
[not found] ` <5c2b0241-4ddd-522e-3d94-9a29ba04ab2e-Re5JQEeQqe8AvxtiuMwx3w@public.gmane.org>
2018-03-28 4:22 ` Liu, Monk
2018-03-28 4:36 ` Liu, Monk
[not found] ` <BLUPR12MB04491F1B2A1BBAFCA738900E84A30-7LeqcoF/hwpTIQvHjXdJlwdYzm3356FpvxpqHgZTriW3zl9H0oFU5g@public.gmane.org>
2018-03-28 7:36 ` Christian König
-- strict thread matches above, loose matches on Subject: below --
2018-03-20 6:29 Emily Deng
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=04acc6d7-680b-861d-2a3e-e4206b72345c@gmail.com \
--to=ckoenig.leichtzumerken-re5jqeeqqe8avxtiumwx3w@public.gmane.org \
--cc=Emily.Deng-5C7GfCeVMHo@public.gmane.org \
--cc=Monk.Liu-5C7GfCeVMHo@public.gmane.org \
--cc=amd-gfx-PD4FTy7X32lNgt0PjOBp9y5qC8QIuHrW@public.gmane.org \
--cc=christian.koenig-5C7GfCeVMHo@public.gmane.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox