* [PATCH v3 1/4] drm/amdgpu: add engine_retains_context to amdgpu_ring_funcs
@ 2025-11-06 9:39 Pierre-Eric Pelloux-Prayer
2025-11-06 9:39 ` [PATCH v3 2/4] drm/amdgpu: jump to the correct label on failure Pierre-Eric Pelloux-Prayer
` (2 more replies)
0 siblings, 3 replies; 11+ messages in thread
From: Pierre-Eric Pelloux-Prayer @ 2025-11-06 9:39 UTC (permalink / raw)
To: Alex Deucher, Christian König, David Airlie, Simona Vetter
Cc: Pierre-Eric Pelloux-Prayer, amd-gfx, dri-devel, linux-kernel
If true, the hw engine retains context among dependent jobs, which means
load balancing between schedulers cannot be used at the job level.
amdgpu_ctx_init_entity uses this information to disable load balancing,
but it's best to store it as a property rather than deduce it based on
hw_ip.
Signed-off-by: Pierre-Eric Pelloux-Prayer <pierre-eric.pelloux-prayer@amd.com>
---
drivers/gpu/drm/amd/amdgpu/amdgpu_ring.h | 1 +
drivers/gpu/drm/amd/amdgpu/uvd_v3_1.c | 1 +
drivers/gpu/drm/amd/amdgpu/uvd_v4_2.c | 1 +
drivers/gpu/drm/amd/amdgpu/uvd_v5_0.c | 1 +
drivers/gpu/drm/amd/amdgpu/uvd_v6_0.c | 3 +++
drivers/gpu/drm/amd/amdgpu/uvd_v7_0.c | 2 ++
drivers/gpu/drm/amd/amdgpu/vcn_v1_0.c | 2 ++
drivers/gpu/drm/amd/amdgpu/vcn_v2_0.c | 2 ++
drivers/gpu/drm/amd/amdgpu/vcn_v2_5.c | 2 ++
drivers/gpu/drm/amd/amdgpu/vcn_v3_0.c | 3 +++
drivers/gpu/drm/amd/amdgpu/vcn_v4_0.c | 1 +
drivers/gpu/drm/amd/amdgpu/vcn_v4_0_3.c | 1 +
drivers/gpu/drm/amd/amdgpu/vcn_v4_0_5.c | 1 +
drivers/gpu/drm/amd/amdgpu/vcn_v5_0_0.c | 1 +
drivers/gpu/drm/amd/amdgpu/vcn_v5_0_1.c | 1 +
15 files changed, 23 insertions(+)
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ring.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_ring.h
index 4b46e3c26ff3..a10efac2fc54 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ring.h
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ring.h
@@ -211,6 +211,7 @@ struct amdgpu_ring_funcs {
bool support_64bit_ptrs;
bool no_user_fence;
bool secure_submission_supported;
+ bool engine_retains_context;
/**
* @extra_bytes:
diff --git a/drivers/gpu/drm/amd/amdgpu/uvd_v3_1.c b/drivers/gpu/drm/amd/amdgpu/uvd_v3_1.c
index 2e79a3afc774..4a85b5465bb2 100644
--- a/drivers/gpu/drm/amd/amdgpu/uvd_v3_1.c
+++ b/drivers/gpu/drm/amd/amdgpu/uvd_v3_1.c
@@ -181,6 +181,7 @@ static const struct amdgpu_ring_funcs uvd_v3_1_ring_funcs = {
.align_mask = 0xf,
.support_64bit_ptrs = false,
.no_user_fence = true,
+ .engine_retains_context = true,
.get_rptr = uvd_v3_1_ring_get_rptr,
.get_wptr = uvd_v3_1_ring_get_wptr,
.set_wptr = uvd_v3_1_ring_set_wptr,
diff --git a/drivers/gpu/drm/amd/amdgpu/uvd_v4_2.c b/drivers/gpu/drm/amd/amdgpu/uvd_v4_2.c
index 4b96fd583772..e7c1d12f0596 100644
--- a/drivers/gpu/drm/amd/amdgpu/uvd_v4_2.c
+++ b/drivers/gpu/drm/amd/amdgpu/uvd_v4_2.c
@@ -775,6 +775,7 @@ static const struct amdgpu_ring_funcs uvd_v4_2_ring_funcs = {
.align_mask = 0xf,
.support_64bit_ptrs = false,
.no_user_fence = true,
+ .engine_retains_context = true,
.get_rptr = uvd_v4_2_ring_get_rptr,
.get_wptr = uvd_v4_2_ring_get_wptr,
.set_wptr = uvd_v4_2_ring_set_wptr,
diff --git a/drivers/gpu/drm/amd/amdgpu/uvd_v5_0.c b/drivers/gpu/drm/amd/amdgpu/uvd_v5_0.c
index 71409ad8b7ed..a62788e4af96 100644
--- a/drivers/gpu/drm/amd/amdgpu/uvd_v5_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/uvd_v5_0.c
@@ -882,6 +882,7 @@ static const struct amdgpu_ring_funcs uvd_v5_0_ring_funcs = {
.align_mask = 0xf,
.support_64bit_ptrs = false,
.no_user_fence = true,
+ .engine_retains_context = true,
.get_rptr = uvd_v5_0_ring_get_rptr,
.get_wptr = uvd_v5_0_ring_get_wptr,
.set_wptr = uvd_v5_0_ring_set_wptr,
diff --git a/drivers/gpu/drm/amd/amdgpu/uvd_v6_0.c b/drivers/gpu/drm/amd/amdgpu/uvd_v6_0.c
index ceb94bbb03a4..0435577b9b3b 100644
--- a/drivers/gpu/drm/amd/amdgpu/uvd_v6_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/uvd_v6_0.c
@@ -1552,6 +1552,7 @@ static const struct amdgpu_ring_funcs uvd_v6_0_ring_phys_funcs = {
.align_mask = 0xf,
.support_64bit_ptrs = false,
.no_user_fence = true,
+ .engine_retains_context = true,
.get_rptr = uvd_v6_0_ring_get_rptr,
.get_wptr = uvd_v6_0_ring_get_wptr,
.set_wptr = uvd_v6_0_ring_set_wptr,
@@ -1578,6 +1579,7 @@ static const struct amdgpu_ring_funcs uvd_v6_0_ring_vm_funcs = {
.align_mask = 0xf,
.support_64bit_ptrs = false,
.no_user_fence = true,
+ .engine_retains_context = true,
.get_rptr = uvd_v6_0_ring_get_rptr,
.get_wptr = uvd_v6_0_ring_get_wptr,
.set_wptr = uvd_v6_0_ring_set_wptr,
@@ -1607,6 +1609,7 @@ static const struct amdgpu_ring_funcs uvd_v6_0_enc_ring_vm_funcs = {
.nop = HEVC_ENC_CMD_NO_OP,
.support_64bit_ptrs = false,
.no_user_fence = true,
+ .engine_retains_context = true,
.get_rptr = uvd_v6_0_enc_ring_get_rptr,
.get_wptr = uvd_v6_0_enc_ring_get_wptr,
.set_wptr = uvd_v6_0_enc_ring_set_wptr,
diff --git a/drivers/gpu/drm/amd/amdgpu/uvd_v7_0.c b/drivers/gpu/drm/amd/amdgpu/uvd_v7_0.c
index 1f8866f3f63c..3720d72f2c3e 100644
--- a/drivers/gpu/drm/amd/amdgpu/uvd_v7_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/uvd_v7_0.c
@@ -1539,6 +1539,7 @@ static const struct amdgpu_ring_funcs uvd_v7_0_ring_vm_funcs = {
.align_mask = 0xf,
.support_64bit_ptrs = false,
.no_user_fence = true,
+ .engine_retains_context = true,
.get_rptr = uvd_v7_0_ring_get_rptr,
.get_wptr = uvd_v7_0_ring_get_wptr,
.set_wptr = uvd_v7_0_ring_set_wptr,
@@ -1571,6 +1572,7 @@ static const struct amdgpu_ring_funcs uvd_v7_0_enc_ring_vm_funcs = {
.nop = HEVC_ENC_CMD_NO_OP,
.support_64bit_ptrs = false,
.no_user_fence = true,
+ .engine_retains_context = true,
.get_rptr = uvd_v7_0_enc_ring_get_rptr,
.get_wptr = uvd_v7_0_enc_ring_get_wptr,
.set_wptr = uvd_v7_0_enc_ring_set_wptr,
diff --git a/drivers/gpu/drm/amd/amdgpu/vcn_v1_0.c b/drivers/gpu/drm/amd/amdgpu/vcn_v1_0.c
index a316797875a8..1691d0f955a9 100644
--- a/drivers/gpu/drm/amd/amdgpu/vcn_v1_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/vcn_v1_0.c
@@ -2117,6 +2117,7 @@ static const struct amdgpu_ring_funcs vcn_v1_0_dec_ring_vm_funcs = {
.support_64bit_ptrs = false,
.no_user_fence = true,
.secure_submission_supported = true,
+ .engine_retains_context = true,
.get_rptr = vcn_v1_0_dec_ring_get_rptr,
.get_wptr = vcn_v1_0_dec_ring_get_wptr,
.set_wptr = vcn_v1_0_dec_ring_set_wptr,
@@ -2150,6 +2151,7 @@ static const struct amdgpu_ring_funcs vcn_v1_0_enc_ring_vm_funcs = {
.align_mask = 0x3f,
.nop = VCN_ENC_CMD_NO_OP,
.support_64bit_ptrs = false,
+ .engine_retains_context = true,
.no_user_fence = true,
.get_rptr = vcn_v1_0_enc_ring_get_rptr,
.get_wptr = vcn_v1_0_enc_ring_get_wptr,
diff --git a/drivers/gpu/drm/amd/amdgpu/vcn_v2_0.c b/drivers/gpu/drm/amd/amdgpu/vcn_v2_0.c
index 8897dcc9c1a0..046dd6b216e9 100644
--- a/drivers/gpu/drm/amd/amdgpu/vcn_v2_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/vcn_v2_0.c
@@ -2113,6 +2113,7 @@ static const struct amdgpu_ring_funcs vcn_v2_0_dec_ring_vm_funcs = {
.type = AMDGPU_RING_TYPE_VCN_DEC,
.align_mask = 0xf,
.secure_submission_supported = true,
+ .engine_retains_context = true,
.get_rptr = vcn_v2_0_dec_ring_get_rptr,
.get_wptr = vcn_v2_0_dec_ring_get_wptr,
.set_wptr = vcn_v2_0_dec_ring_set_wptr,
@@ -2144,6 +2145,7 @@ static const struct amdgpu_ring_funcs vcn_v2_0_enc_ring_vm_funcs = {
.type = AMDGPU_RING_TYPE_VCN_ENC,
.align_mask = 0x3f,
.nop = VCN_ENC_CMD_NO_OP,
+ .engine_retains_context = true,
.get_rptr = vcn_v2_0_enc_ring_get_rptr,
.get_wptr = vcn_v2_0_enc_ring_get_wptr,
.set_wptr = vcn_v2_0_enc_ring_set_wptr,
diff --git a/drivers/gpu/drm/amd/amdgpu/vcn_v2_5.c b/drivers/gpu/drm/amd/amdgpu/vcn_v2_5.c
index cebee453871c..063f88da120b 100644
--- a/drivers/gpu/drm/amd/amdgpu/vcn_v2_5.c
+++ b/drivers/gpu/drm/amd/amdgpu/vcn_v2_5.c
@@ -1777,6 +1777,7 @@ static const struct amdgpu_ring_funcs vcn_v2_5_dec_ring_vm_funcs = {
.type = AMDGPU_RING_TYPE_VCN_DEC,
.align_mask = 0xf,
.secure_submission_supported = true,
+ .engine_retains_context = true,
.get_rptr = vcn_v2_5_dec_ring_get_rptr,
.get_wptr = vcn_v2_5_dec_ring_get_wptr,
.set_wptr = vcn_v2_5_dec_ring_set_wptr,
@@ -1877,6 +1878,7 @@ static const struct amdgpu_ring_funcs vcn_v2_5_enc_ring_vm_funcs = {
.type = AMDGPU_RING_TYPE_VCN_ENC,
.align_mask = 0x3f,
.nop = VCN_ENC_CMD_NO_OP,
+ .engine_retains_context = true,
.get_rptr = vcn_v2_5_enc_ring_get_rptr,
.get_wptr = vcn_v2_5_enc_ring_get_wptr,
.set_wptr = vcn_v2_5_enc_ring_set_wptr,
diff --git a/drivers/gpu/drm/amd/amdgpu/vcn_v3_0.c b/drivers/gpu/drm/amd/amdgpu/vcn_v3_0.c
index d9cf8f0feeb3..8dcc07b3f631 100644
--- a/drivers/gpu/drm/amd/amdgpu/vcn_v3_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/vcn_v3_0.c
@@ -1857,6 +1857,7 @@ static const struct amdgpu_ring_funcs vcn_v3_0_dec_sw_ring_vm_funcs = {
.align_mask = 0x3f,
.nop = VCN_DEC_SW_CMD_NO_OP,
.secure_submission_supported = true,
+ .engine_retains_context = true,
.get_rptr = vcn_v3_0_dec_ring_get_rptr,
.get_wptr = vcn_v3_0_dec_ring_get_wptr,
.set_wptr = vcn_v3_0_dec_ring_set_wptr,
@@ -2021,6 +2022,7 @@ static const struct amdgpu_ring_funcs vcn_v3_0_dec_ring_vm_funcs = {
.type = AMDGPU_RING_TYPE_VCN_DEC,
.align_mask = 0xf,
.secure_submission_supported = true,
+ .engine_retains_context = true,
.get_rptr = vcn_v3_0_dec_ring_get_rptr,
.get_wptr = vcn_v3_0_dec_ring_get_wptr,
.set_wptr = vcn_v3_0_dec_ring_set_wptr,
@@ -2122,6 +2124,7 @@ static const struct amdgpu_ring_funcs vcn_v3_0_enc_ring_vm_funcs = {
.type = AMDGPU_RING_TYPE_VCN_ENC,
.align_mask = 0x3f,
.nop = VCN_ENC_CMD_NO_OP,
+ .engine_retains_context = true,
.get_rptr = vcn_v3_0_enc_ring_get_rptr,
.get_wptr = vcn_v3_0_enc_ring_get_wptr,
.set_wptr = vcn_v3_0_enc_ring_set_wptr,
diff --git a/drivers/gpu/drm/amd/amdgpu/vcn_v4_0.c b/drivers/gpu/drm/amd/amdgpu/vcn_v4_0.c
index 3ae666522d57..f1306316dc3c 100644
--- a/drivers/gpu/drm/amd/amdgpu/vcn_v4_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/vcn_v4_0.c
@@ -1977,6 +1977,7 @@ static struct amdgpu_ring_funcs vcn_v4_0_unified_ring_vm_funcs = {
.type = AMDGPU_RING_TYPE_VCN_ENC,
.align_mask = 0x3f,
.nop = VCN_ENC_CMD_NO_OP,
+ .engine_retains_context = true,
.extra_bytes = sizeof(struct amdgpu_vcn_rb_metadata),
.get_rptr = vcn_v4_0_unified_ring_get_rptr,
.get_wptr = vcn_v4_0_unified_ring_get_wptr,
diff --git a/drivers/gpu/drm/amd/amdgpu/vcn_v4_0_3.c b/drivers/gpu/drm/amd/amdgpu/vcn_v4_0_3.c
index eacf4e93ba2f..5a935c07352a 100644
--- a/drivers/gpu/drm/amd/amdgpu/vcn_v4_0_3.c
+++ b/drivers/gpu/drm/amd/amdgpu/vcn_v4_0_3.c
@@ -1628,6 +1628,7 @@ static const struct amdgpu_ring_funcs vcn_v4_0_3_unified_ring_vm_funcs = {
.type = AMDGPU_RING_TYPE_VCN_ENC,
.align_mask = 0x3f,
.nop = VCN_ENC_CMD_NO_OP,
+ .engine_retains_context = true,
.get_rptr = vcn_v4_0_3_unified_ring_get_rptr,
.get_wptr = vcn_v4_0_3_unified_ring_get_wptr,
.set_wptr = vcn_v4_0_3_unified_ring_set_wptr,
diff --git a/drivers/gpu/drm/amd/amdgpu/vcn_v4_0_5.c b/drivers/gpu/drm/amd/amdgpu/vcn_v4_0_5.c
index b107ee80e472..1a485f5825dd 100644
--- a/drivers/gpu/drm/amd/amdgpu/vcn_v4_0_5.c
+++ b/drivers/gpu/drm/amd/amdgpu/vcn_v4_0_5.c
@@ -1481,6 +1481,7 @@ static struct amdgpu_ring_funcs vcn_v4_0_5_unified_ring_vm_funcs = {
.type = AMDGPU_RING_TYPE_VCN_ENC,
.align_mask = 0x3f,
.nop = VCN_ENC_CMD_NO_OP,
+ .engine_retains_context = true,
.get_rptr = vcn_v4_0_5_unified_ring_get_rptr,
.get_wptr = vcn_v4_0_5_unified_ring_get_wptr,
.set_wptr = vcn_v4_0_5_unified_ring_set_wptr,
diff --git a/drivers/gpu/drm/amd/amdgpu/vcn_v5_0_0.c b/drivers/gpu/drm/amd/amdgpu/vcn_v5_0_0.c
index 0202df5db1e1..2d8214f591f1 100644
--- a/drivers/gpu/drm/amd/amdgpu/vcn_v5_0_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/vcn_v5_0_0.c
@@ -1203,6 +1203,7 @@ static const struct amdgpu_ring_funcs vcn_v5_0_0_unified_ring_vm_funcs = {
.type = AMDGPU_RING_TYPE_VCN_ENC,
.align_mask = 0x3f,
.nop = VCN_ENC_CMD_NO_OP,
+ .engine_retains_context = true,
.get_rptr = vcn_v5_0_0_unified_ring_get_rptr,
.get_wptr = vcn_v5_0_0_unified_ring_get_wptr,
.set_wptr = vcn_v5_0_0_unified_ring_set_wptr,
diff --git a/drivers/gpu/drm/amd/amdgpu/vcn_v5_0_1.c b/drivers/gpu/drm/amd/amdgpu/vcn_v5_0_1.c
index 714350cabf2f..bd3a04f1414d 100644
--- a/drivers/gpu/drm/amd/amdgpu/vcn_v5_0_1.c
+++ b/drivers/gpu/drm/amd/amdgpu/vcn_v5_0_1.c
@@ -1328,6 +1328,7 @@ static const struct amdgpu_ring_funcs vcn_v5_0_1_unified_ring_vm_funcs = {
.type = AMDGPU_RING_TYPE_VCN_ENC,
.align_mask = 0x3f,
.nop = VCN_ENC_CMD_NO_OP,
+ .engine_retains_context = true,
.get_rptr = vcn_v5_0_1_unified_ring_get_rptr,
.get_wptr = vcn_v5_0_1_unified_ring_get_wptr,
.set_wptr = vcn_v5_0_1_unified_ring_set_wptr,
--
2.43.0
^ permalink raw reply [flat|nested] 11+ messages in thread* [PATCH v3 2/4] drm/amdgpu: jump to the correct label on failure
2025-11-06 9:39 [PATCH v3 1/4] drm/amdgpu: add engine_retains_context to amdgpu_ring_funcs Pierre-Eric Pelloux-Prayer
@ 2025-11-06 9:39 ` Pierre-Eric Pelloux-Prayer
2025-11-06 9:56 ` Tvrtko Ursulin
2025-11-06 9:39 ` [PATCH v3 3/4] drm/amdgpu: increment sched score on entity selection Pierre-Eric Pelloux-Prayer
2025-11-06 9:39 ` [PATCH v3 4/4] drm/sched: limit sched score update to jobs change Pierre-Eric Pelloux-Prayer
2 siblings, 1 reply; 11+ messages in thread
From: Pierre-Eric Pelloux-Prayer @ 2025-11-06 9:39 UTC (permalink / raw)
To: Alex Deucher, Christian König, David Airlie, Simona Vetter
Cc: Pierre-Eric Pelloux-Prayer, amd-gfx, dri-devel, linux-kernel
drm_sched_entity_init wasn't called yet, so the only thing to
do is to release allocated memory.
This doesn't fix any bug since entity is zero allocated and
drm_sched_entity_fini does nothing in this case.
Signed-off-by: Pierre-Eric Pelloux-Prayer <pierre-eric.pelloux-prayer@amd.com>
---
drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
index f5d5c45ddc0d..afedea02188d 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
@@ -236,7 +236,7 @@ static int amdgpu_ctx_init_entity(struct amdgpu_ctx *ctx, u32 hw_ip,
r = amdgpu_xcp_select_scheds(adev, hw_ip, hw_prio, fpriv,
&num_scheds, &scheds);
if (r)
- goto cleanup_entity;
+ goto error_free_entity;
}
/* disable load balance if the hw engine retains context among dependent jobs */
--
2.43.0
^ permalink raw reply [flat|nested] 11+ messages in thread
* Re: [PATCH v3 2/4] drm/amdgpu: jump to the correct label on failure
2025-11-06 9:39 ` [PATCH v3 2/4] drm/amdgpu: jump to the correct label on failure Pierre-Eric Pelloux-Prayer
@ 2025-11-06 9:56 ` Tvrtko Ursulin
2025-11-06 10:21 ` Christian König
0 siblings, 1 reply; 11+ messages in thread
From: Tvrtko Ursulin @ 2025-11-06 9:56 UTC (permalink / raw)
To: Pierre-Eric Pelloux-Prayer, Alex Deucher, Christian König,
David Airlie, Simona Vetter
Cc: amd-gfx, dri-devel, linux-kernel
On 06/11/2025 09:39, Pierre-Eric Pelloux-Prayer wrote:
> drm_sched_entity_init wasn't called yet, so the only thing to
> do is to release allocated memory.
> This doesn't fix any bug since entity is zero allocated and
> drm_sched_entity_fini does nothing in this case.
>
> Signed-off-by: Pierre-Eric Pelloux-Prayer <pierre-eric.pelloux-prayer@amd.com>
> ---
> drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c | 2 +-
> 1 file changed, 1 insertion(+), 1 deletion(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
> index f5d5c45ddc0d..afedea02188d 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
> @@ -236,7 +236,7 @@ static int amdgpu_ctx_init_entity(struct amdgpu_ctx *ctx, u32 hw_ip,
> r = amdgpu_xcp_select_scheds(adev, hw_ip, hw_prio, fpriv,
> &num_scheds, &scheds);
> if (r)
> - goto cleanup_entity;
> + goto error_free_entity;
> }
>
> /* disable load balance if the hw engine retains context among dependent jobs */
Reviewed-by: Tvrtko Ursulin <tvrtko.ursulin@igalia.com>
Regards,
Tvrtko
^ permalink raw reply [flat|nested] 11+ messages in thread
* Re: [PATCH v3 2/4] drm/amdgpu: jump to the correct label on failure
2025-11-06 9:56 ` Tvrtko Ursulin
@ 2025-11-06 10:21 ` Christian König
2025-11-07 9:05 ` Pierre-Eric Pelloux-Prayer
0 siblings, 1 reply; 11+ messages in thread
From: Christian König @ 2025-11-06 10:21 UTC (permalink / raw)
To: Tvrtko Ursulin, Pierre-Eric Pelloux-Prayer, Alex Deucher,
David Airlie, Simona Vetter
Cc: amd-gfx, dri-devel, linux-kernel
On 11/6/25 10:56, Tvrtko Ursulin wrote:
>
> On 06/11/2025 09:39, Pierre-Eric Pelloux-Prayer wrote:
>> drm_sched_entity_init wasn't called yet, so the only thing to
>> do is to release allocated memory.
>> This doesn't fix any bug since entity is zero allocated and
>> drm_sched_entity_fini does nothing in this case.
>>
>> Signed-off-by: Pierre-Eric Pelloux-Prayer <pierre-eric.pelloux-prayer@amd.com>
>> ---
>> drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c | 2 +-
>> 1 file changed, 1 insertion(+), 1 deletion(-)
>>
>> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
>> index f5d5c45ddc0d..afedea02188d 100644
>> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
>> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
>> @@ -236,7 +236,7 @@ static int amdgpu_ctx_init_entity(struct amdgpu_ctx *ctx, u32 hw_ip,
>> r = amdgpu_xcp_select_scheds(adev, hw_ip, hw_prio, fpriv,
>> &num_scheds, &scheds);
>> if (r)
>> - goto cleanup_entity;
>> + goto error_free_entity;
>> }
>> /* disable load balance if the hw engine retains context among dependent jobs */
>
> Reviewed-by: Tvrtko Ursulin <tvrtko.ursulin@igalia.com>
Acked-by: Christian König <christian.koenig@amd.com>
Since this is still a fix please push it to amd-staging-drm-next independent of the remaining patch set.
Thanks,
Christian.
>
> Regards,
>
> Tvrtko
>
^ permalink raw reply [flat|nested] 11+ messages in thread
* Re: [PATCH v3 2/4] drm/amdgpu: jump to the correct label on failure
2025-11-06 10:21 ` Christian König
@ 2025-11-07 9:05 ` Pierre-Eric Pelloux-Prayer
0 siblings, 0 replies; 11+ messages in thread
From: Pierre-Eric Pelloux-Prayer @ 2025-11-07 9:05 UTC (permalink / raw)
To: Christian König, Tvrtko Ursulin, Pierre-Eric Pelloux-Prayer,
Alex Deucher, David Airlie, Simona Vetter
Cc: amd-gfx, dri-devel, linux-kernel
Le 06/11/2025 à 11:21, Christian König a écrit :
> On 11/6/25 10:56, Tvrtko Ursulin wrote:
>>
>> On 06/11/2025 09:39, Pierre-Eric Pelloux-Prayer wrote:
>>> drm_sched_entity_init wasn't called yet, so the only thing to
>>> do is to release allocated memory.
>>> This doesn't fix any bug since entity is zero allocated and
>>> drm_sched_entity_fini does nothing in this case.
>>>
>>> Signed-off-by: Pierre-Eric Pelloux-Prayer <pierre-eric.pelloux-prayer@amd.com>
>>> ---
>>> drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c | 2 +-
>>> 1 file changed, 1 insertion(+), 1 deletion(-)
>>>
>>> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
>>> index f5d5c45ddc0d..afedea02188d 100644
>>> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
>>> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
>>> @@ -236,7 +236,7 @@ static int amdgpu_ctx_init_entity(struct amdgpu_ctx *ctx, u32 hw_ip,
>>> r = amdgpu_xcp_select_scheds(adev, hw_ip, hw_prio, fpriv,
>>> &num_scheds, &scheds);
>>> if (r)
>>> - goto cleanup_entity;
>>> + goto error_free_entity;
>>> }
>>> /* disable load balance if the hw engine retains context among dependent jobs */
>>
>> Reviewed-by: Tvrtko Ursulin <tvrtko.ursulin@igalia.com>
>
> Acked-by: Christian König <christian.koenig@amd.com>
>
> Since this is still a fix please push it to amd-staging-drm-next independent of the remaining patch set.
Ok, I removed this patch from v4 and will push it to amd-staging-drm-next.
Pierre-Eric
^ permalink raw reply [flat|nested] 11+ messages in thread
* [PATCH v3 3/4] drm/amdgpu: increment sched score on entity selection
2025-11-06 9:39 [PATCH v3 1/4] drm/amdgpu: add engine_retains_context to amdgpu_ring_funcs Pierre-Eric Pelloux-Prayer
2025-11-06 9:39 ` [PATCH v3 2/4] drm/amdgpu: jump to the correct label on failure Pierre-Eric Pelloux-Prayer
@ 2025-11-06 9:39 ` Pierre-Eric Pelloux-Prayer
2025-11-06 10:00 ` Tvrtko Ursulin
2025-11-06 9:39 ` [PATCH v3 4/4] drm/sched: limit sched score update to jobs change Pierre-Eric Pelloux-Prayer
2 siblings, 1 reply; 11+ messages in thread
From: Pierre-Eric Pelloux-Prayer @ 2025-11-06 9:39 UTC (permalink / raw)
To: Alex Deucher, Christian König, David Airlie, Simona Vetter
Cc: Pierre-Eric Pelloux-Prayer, amd-gfx, dri-devel, linux-kernel
For hw engines that can't load balance jobs, entities are
"statically" load balanced: on their first submit, they select
the best scheduler based on its score.
The score is made up of 2 parts:
* the job queue depth (how much jobs are executing/waiting)
* the number of entities assigned
The second part is only relevant for the static load balance:
it's a way to consider how many entities are attached to this
scheduler, knowing that if they ever submit jobs they will go
to this one.
For rings that can load balance jobs freely, idle entities
aren't a concern and shouldn't impact the scheduler's decisions.
Signed-off-by: Pierre-Eric Pelloux-Prayer <pierre-eric.pelloux-prayer@amd.com>
---
drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c | 21 ++++++++++++++++-----
drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h | 1 +
2 files changed, 17 insertions(+), 5 deletions(-)
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
index afedea02188d..953c81c928c1 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
@@ -209,6 +209,7 @@ static int amdgpu_ctx_init_entity(struct amdgpu_ctx *ctx, u32 hw_ip,
struct amdgpu_ctx_entity *entity;
enum drm_sched_priority drm_prio;
unsigned int hw_prio, num_scheds;
+ struct amdgpu_ring *aring;
int32_t ctx_prio;
int r;
@@ -239,11 +240,13 @@ static int amdgpu_ctx_init_entity(struct amdgpu_ctx *ctx, u32 hw_ip,
goto error_free_entity;
}
- /* disable load balance if the hw engine retains context among dependent jobs */
- if (hw_ip == AMDGPU_HW_IP_VCN_ENC ||
- hw_ip == AMDGPU_HW_IP_VCN_DEC ||
- hw_ip == AMDGPU_HW_IP_UVD_ENC ||
- hw_ip == AMDGPU_HW_IP_UVD) {
+ sched = scheds[0];
+ aring = container_of(sched, struct amdgpu_ring, sched);
+
+ if (aring->funcs->engine_retains_context) {
+ /* Disable load balancing between multiple schedulers if the hw
+ * engine retains context among dependent jobs.
+ */
sched = drm_sched_pick_best(scheds, num_scheds);
scheds = &sched;
num_scheds = 1;
@@ -258,6 +261,11 @@ static int amdgpu_ctx_init_entity(struct amdgpu_ctx *ctx, u32 hw_ip,
if (cmpxchg(&ctx->entities[hw_ip][ring], NULL, entity))
goto cleanup_entity;
+ if (aring->funcs->engine_retains_context) {
+ entity->sched_score = sched->score;
+ atomic_inc(entity->sched_score);
+ }
+
return 0;
cleanup_entity:
@@ -514,6 +522,9 @@ static void amdgpu_ctx_do_release(struct kref *ref)
if (!ctx->entities[i][j])
continue;
+ if (ctx->entities[i][j]->sched_score)
+ atomic_dec(ctx->entities[i][j]->sched_score);
+
drm_sched_entity_destroy(&ctx->entities[i][j]->entity);
}
}
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h
index 090dfe86f75b..f7b44f96f374 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h
@@ -39,6 +39,7 @@ struct amdgpu_ctx_entity {
uint32_t hw_ip;
uint64_t sequence;
struct drm_sched_entity entity;
+ atomic_t *sched_score;
struct dma_fence *fences[];
};
--
2.43.0
^ permalink raw reply [flat|nested] 11+ messages in thread* Re: [PATCH v3 3/4] drm/amdgpu: increment sched score on entity selection
2025-11-06 9:39 ` [PATCH v3 3/4] drm/amdgpu: increment sched score on entity selection Pierre-Eric Pelloux-Prayer
@ 2025-11-06 10:00 ` Tvrtko Ursulin
2025-11-06 10:10 ` Pierre-Eric Pelloux-Prayer
0 siblings, 1 reply; 11+ messages in thread
From: Tvrtko Ursulin @ 2025-11-06 10:00 UTC (permalink / raw)
To: Pierre-Eric Pelloux-Prayer, Alex Deucher, Christian König,
David Airlie, Simona Vetter
Cc: amd-gfx, dri-devel, linux-kernel
On 06/11/2025 09:39, Pierre-Eric Pelloux-Prayer wrote:
> For hw engines that can't load balance jobs, entities are
> "statically" load balanced: on their first submit, they select
> the best scheduler based on its score.
> The score is made up of 2 parts:
> * the job queue depth (how much jobs are executing/waiting)
> * the number of entities assigned
>
> The second part is only relevant for the static load balance:
> it's a way to consider how many entities are attached to this
> scheduler, knowing that if they ever submit jobs they will go
> to this one.
>
> For rings that can load balance jobs freely, idle entities
> aren't a concern and shouldn't impact the scheduler's decisions.
>
> Signed-off-by: Pierre-Eric Pelloux-Prayer <pierre-eric.pelloux-prayer@amd.com>
> ---
> drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c | 21 ++++++++++++++++-----
> drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h | 1 +
> 2 files changed, 17 insertions(+), 5 deletions(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
> index afedea02188d..953c81c928c1 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
> @@ -209,6 +209,7 @@ static int amdgpu_ctx_init_entity(struct amdgpu_ctx *ctx, u32 hw_ip,
> struct amdgpu_ctx_entity *entity;
> enum drm_sched_priority drm_prio;
> unsigned int hw_prio, num_scheds;
> + struct amdgpu_ring *aring;
> int32_t ctx_prio;
> int r;
>
> @@ -239,11 +240,13 @@ static int amdgpu_ctx_init_entity(struct amdgpu_ctx *ctx, u32 hw_ip,
> goto error_free_entity;
> }
>
> - /* disable load balance if the hw engine retains context among dependent jobs */
> - if (hw_ip == AMDGPU_HW_IP_VCN_ENC ||
> - hw_ip == AMDGPU_HW_IP_VCN_DEC ||
> - hw_ip == AMDGPU_HW_IP_UVD_ENC ||
> - hw_ip == AMDGPU_HW_IP_UVD) {
> + sched = scheds[0];
> + aring = container_of(sched, struct amdgpu_ring, sched);
> +
> + if (aring->funcs->engine_retains_context) {
> + /* Disable load balancing between multiple schedulers if the hw
> + * engine retains context among dependent jobs.
> + */
> sched = drm_sched_pick_best(scheds, num_scheds);
> scheds = &sched;
> num_scheds = 1;
> @@ -258,6 +261,11 @@ static int amdgpu_ctx_init_entity(struct amdgpu_ctx *ctx, u32 hw_ip,
> if (cmpxchg(&ctx->entities[hw_ip][ring], NULL, entity))
> goto cleanup_entity;
>
> + if (aring->funcs->engine_retains_context) {
> + entity->sched_score = sched->score;
> + atomic_inc(entity->sched_score);
Maybe you missed it, in the last round I asked this:
"""
Here is would always be sched->score == aring->sched_score, right?
If so it would probably be good to either add that assert, or even to
just fetch it from there. Otherwise it can look potentially concerning
to be fishing out the pointer from scheduler internals.
The rest looks good to me.
"""
Because grabbing a pointer from drm_sched->score and storing it in AMD
entity can look scary, since sched->score can be scheduler owned.
Hence I was suggesting to either fish it out from aring->sched_score. If
it is true that they are always the same atomic_t at this point.
Regards,
Tvrtko
> + }
> +
> return 0;
>
> cleanup_entity:
> @@ -514,6 +522,9 @@ static void amdgpu_ctx_do_release(struct kref *ref)
> if (!ctx->entities[i][j])
> continue;
>
> + if (ctx->entities[i][j]->sched_score)
> + atomic_dec(ctx->entities[i][j]->sched_score);
> +
> drm_sched_entity_destroy(&ctx->entities[i][j]->entity);
> }
> }
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h
> index 090dfe86f75b..f7b44f96f374 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h
> @@ -39,6 +39,7 @@ struct amdgpu_ctx_entity {
> uint32_t hw_ip;
> uint64_t sequence;
> struct drm_sched_entity entity;
> + atomic_t *sched_score;
> struct dma_fence *fences[];
> };
>
^ permalink raw reply [flat|nested] 11+ messages in thread* Re: [PATCH v3 3/4] drm/amdgpu: increment sched score on entity selection
2025-11-06 10:00 ` Tvrtko Ursulin
@ 2025-11-06 10:10 ` Pierre-Eric Pelloux-Prayer
2025-11-06 10:23 ` Tvrtko Ursulin
0 siblings, 1 reply; 11+ messages in thread
From: Pierre-Eric Pelloux-Prayer @ 2025-11-06 10:10 UTC (permalink / raw)
To: Tvrtko Ursulin, Pierre-Eric Pelloux-Prayer, Alex Deucher,
Christian König, David Airlie, Simona Vetter
Cc: amd-gfx, dri-devel, linux-kernel
Le 06/11/2025 à 11:00, Tvrtko Ursulin a écrit :
>
> On 06/11/2025 09:39, Pierre-Eric Pelloux-Prayer wrote:
>> For hw engines that can't load balance jobs, entities are
>> "statically" load balanced: on their first submit, they select
>> the best scheduler based on its score.
>> The score is made up of 2 parts:
>> * the job queue depth (how much jobs are executing/waiting)
>> * the number of entities assigned
>>
>> The second part is only relevant for the static load balance:
>> it's a way to consider how many entities are attached to this
>> scheduler, knowing that if they ever submit jobs they will go
>> to this one.
>>
>> For rings that can load balance jobs freely, idle entities
>> aren't a concern and shouldn't impact the scheduler's decisions.
>>
>> Signed-off-by: Pierre-Eric Pelloux-Prayer <pierre-eric.pelloux-prayer@amd.com>
>> ---
>> drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c | 21 ++++++++++++++++-----
>> drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h | 1 +
>> 2 files changed, 17 insertions(+), 5 deletions(-)
>>
>> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c b/drivers/gpu/drm/amd/
>> amdgpu/amdgpu_ctx.c
>> index afedea02188d..953c81c928c1 100644
>> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
>> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
>> @@ -209,6 +209,7 @@ static int amdgpu_ctx_init_entity(struct amdgpu_ctx *ctx,
>> u32 hw_ip,
>> struct amdgpu_ctx_entity *entity;
>> enum drm_sched_priority drm_prio;
>> unsigned int hw_prio, num_scheds;
>> + struct amdgpu_ring *aring;
>> int32_t ctx_prio;
>> int r;
>> @@ -239,11 +240,13 @@ static int amdgpu_ctx_init_entity(struct amdgpu_ctx
>> *ctx, u32 hw_ip,
>> goto error_free_entity;
>> }
>> - /* disable load balance if the hw engine retains context among dependent
>> jobs */
>> - if (hw_ip == AMDGPU_HW_IP_VCN_ENC ||
>> - hw_ip == AMDGPU_HW_IP_VCN_DEC ||
>> - hw_ip == AMDGPU_HW_IP_UVD_ENC ||
>> - hw_ip == AMDGPU_HW_IP_UVD) {
>> + sched = scheds[0];
>> + aring = container_of(sched, struct amdgpu_ring, sched);
>> +
>> + if (aring->funcs->engine_retains_context) {
>> + /* Disable load balancing between multiple schedulers if the hw
>> + * engine retains context among dependent jobs.
>> + */
>> sched = drm_sched_pick_best(scheds, num_scheds);
>> scheds = &sched;
>> num_scheds = 1;
>> @@ -258,6 +261,11 @@ static int amdgpu_ctx_init_entity(struct amdgpu_ctx *ctx,
>> u32 hw_ip,
>> if (cmpxchg(&ctx->entities[hw_ip][ring], NULL, entity))
>> goto cleanup_entity;
>> + if (aring->funcs->engine_retains_context) {
>> + entity->sched_score = sched->score;
>> + atomic_inc(entity->sched_score);
>
> Maybe you missed it, in the last round I asked this:
I missed it, sorry.
>
> """
> Here is would always be sched->score == aring->sched_score, right?
Yes because drm_sched_init is called with args.score = ring->sched_score
>
> If so it would probably be good to either add that assert, or even to just fetch
> it from there. Otherwise it can look potentially concerning to be fishing out
> the pointer from scheduler internals.
>
> The rest looks good to me.
> """
>
> Because grabbing a pointer from drm_sched->score and storing it in AMD entity
> can look scary, since sched->score can be scheduler owned.
>
> Hence I was suggesting to either fish it out from aring->sched_score. If it is
> true that they are always the same atomic_t at this point.
I used sched->score, because aring->sched_score is not the one we want (the
existing aring points to scheds[0], not the selected sched). But I can change
the code to:
if (aring->funcs->engine_retains_context) {
aring = container_of(sched, struct amdgpu_ring, sched)
entity->sched_score = aring->sched_score;
atomic_inc(entity->sched_score);
}
If it's preferred.
Pierre-Eric
>
> Regards,
>
> Tvrtko
>
>> + }
>> +
>> return 0;
>> cleanup_entity:
>> @@ -514,6 +522,9 @@ static void amdgpu_ctx_do_release(struct kref *ref)
>> if (!ctx->entities[i][j])
>> continue;
>> + if (ctx->entities[i][j]->sched_score)
>> + atomic_dec(ctx->entities[i][j]->sched_score);
>> +
>> drm_sched_entity_destroy(&ctx->entities[i][j]->entity);
>> }
>> }
>> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h b/drivers/gpu/drm/amd/
>> amdgpu/amdgpu_ctx.h
>> index 090dfe86f75b..f7b44f96f374 100644
>> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h
>> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h
>> @@ -39,6 +39,7 @@ struct amdgpu_ctx_entity {
>> uint32_t hw_ip;
>> uint64_t sequence;
>> struct drm_sched_entity entity;
>> + atomic_t *sched_score;
>> struct dma_fence *fences[];
>> };
^ permalink raw reply [flat|nested] 11+ messages in thread* Re: [PATCH v3 3/4] drm/amdgpu: increment sched score on entity selection
2025-11-06 10:10 ` Pierre-Eric Pelloux-Prayer
@ 2025-11-06 10:23 ` Tvrtko Ursulin
0 siblings, 0 replies; 11+ messages in thread
From: Tvrtko Ursulin @ 2025-11-06 10:23 UTC (permalink / raw)
To: Pierre-Eric Pelloux-Prayer, Pierre-Eric Pelloux-Prayer,
Alex Deucher, Christian König, David Airlie, Simona Vetter
Cc: amd-gfx, dri-devel, linux-kernel
On 06/11/2025 10:10, Pierre-Eric Pelloux-Prayer wrote:
>
>
> Le 06/11/2025 à 11:00, Tvrtko Ursulin a écrit :
>>
>> On 06/11/2025 09:39, Pierre-Eric Pelloux-Prayer wrote:
>>> For hw engines that can't load balance jobs, entities are
>>> "statically" load balanced: on their first submit, they select
>>> the best scheduler based on its score.
>>> The score is made up of 2 parts:
>>> * the job queue depth (how much jobs are executing/waiting)
>>> * the number of entities assigned
>>>
>>> The second part is only relevant for the static load balance:
>>> it's a way to consider how many entities are attached to this
>>> scheduler, knowing that if they ever submit jobs they will go
>>> to this one.
>>>
>>> For rings that can load balance jobs freely, idle entities
>>> aren't a concern and shouldn't impact the scheduler's decisions.
>>>
>>> Signed-off-by: Pierre-Eric Pelloux-Prayer <pierre-eric.pelloux-
>>> prayer@amd.com>
>>> ---
>>> drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c | 21 ++++++++++++++++-----
>>> drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h | 1 +
>>> 2 files changed, 17 insertions(+), 5 deletions(-)
>>>
>>> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c b/drivers/gpu/
>>> drm/amd/ amdgpu/amdgpu_ctx.c
>>> index afedea02188d..953c81c928c1 100644
>>> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
>>> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c
>>> @@ -209,6 +209,7 @@ static int amdgpu_ctx_init_entity(struct
>>> amdgpu_ctx *ctx, u32 hw_ip,
>>> struct amdgpu_ctx_entity *entity;
>>> enum drm_sched_priority drm_prio;
>>> unsigned int hw_prio, num_scheds;
>>> + struct amdgpu_ring *aring;
>>> int32_t ctx_prio;
>>> int r;
>>> @@ -239,11 +240,13 @@ static int amdgpu_ctx_init_entity(struct
>>> amdgpu_ctx *ctx, u32 hw_ip,
>>> goto error_free_entity;
>>> }
>>> - /* disable load balance if the hw engine retains context among
>>> dependent jobs */
>>> - if (hw_ip == AMDGPU_HW_IP_VCN_ENC ||
>>> - hw_ip == AMDGPU_HW_IP_VCN_DEC ||
>>> - hw_ip == AMDGPU_HW_IP_UVD_ENC ||
>>> - hw_ip == AMDGPU_HW_IP_UVD) {
>>> + sched = scheds[0];
>>> + aring = container_of(sched, struct amdgpu_ring, sched);
>>> +
>>> + if (aring->funcs->engine_retains_context) {
>>> + /* Disable load balancing between multiple schedulers if the hw
>>> + * engine retains context among dependent jobs.
>>> + */
>>> sched = drm_sched_pick_best(scheds, num_scheds);
>>> scheds = &sched;
>>> num_scheds = 1;
>>> @@ -258,6 +261,11 @@ static int amdgpu_ctx_init_entity(struct
>>> amdgpu_ctx *ctx, u32 hw_ip,
>>> if (cmpxchg(&ctx->entities[hw_ip][ring], NULL, entity))
>>> goto cleanup_entity;
>>> + if (aring->funcs->engine_retains_context) {
>>> + entity->sched_score = sched->score;
>>> + atomic_inc(entity->sched_score);
>>
>> Maybe you missed it, in the last round I asked this:
>
> I missed it, sorry.
>
>>
>> """
>> Here is would always be sched->score == aring->sched_score, right?
>
> Yes because drm_sched_init is called with args.score = ring->sched_score
>
>>
>> If so it would probably be good to either add that assert, or even to
>> just fetch it from there. Otherwise it can look potentially concerning
>> to be fishing out the pointer from scheduler internals.
>>
>> The rest looks good to me.
>> """
>>
>> Because grabbing a pointer from drm_sched->score and storing it in AMD
>> entity can look scary, since sched->score can be scheduler owned.
>>
>> Hence I was suggesting to either fish it out from aring->sched_score.
>> If it is true that they are always the same atomic_t at this point.
>
> I used sched->score, because aring->sched_score is not the one we want
> (the existing aring points to scheds[0], not the selected sched). But I
> can change the code to:
>
> if (aring->funcs->engine_retains_context) {
> aring = container_of(sched, struct amdgpu_ring, sched)
> entity->sched_score = aring->sched_score;
> atomic_inc(entity->sched_score);
> }
>
> If it's preferred.
For me it is, yes. Because it removes any doubt on who owns the atomic_t
pointed to. And it also isolates the driver from any changes in DRM
scheduler structures.
With that:
Reviewed-by: Tvrtko Ursulin <tvrtko.ursulin@igalia.com>
Regards,
Tvrtko
>>> + }
>>> +
>>> return 0;
>>> cleanup_entity:
>>> @@ -514,6 +522,9 @@ static void amdgpu_ctx_do_release(struct kref *ref)
>>> if (!ctx->entities[i][j])
>>> continue;
>>> + if (ctx->entities[i][j]->sched_score)
>>> + atomic_dec(ctx->entities[i][j]->sched_score);
>>> +
>>> drm_sched_entity_destroy(&ctx->entities[i][j]->entity);
>>> }
>>> }
>>> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h b/drivers/gpu/
>>> drm/amd/ amdgpu/amdgpu_ctx.h
>>> index 090dfe86f75b..f7b44f96f374 100644
>>> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h
>>> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.h
>>> @@ -39,6 +39,7 @@ struct amdgpu_ctx_entity {
>>> uint32_t hw_ip;
>>> uint64_t sequence;
>>> struct drm_sched_entity entity;
>>> + atomic_t *sched_score;
>>> struct dma_fence *fences[];
>>> };
^ permalink raw reply [flat|nested] 11+ messages in thread
* [PATCH v3 4/4] drm/sched: limit sched score update to jobs change
2025-11-06 9:39 [PATCH v3 1/4] drm/amdgpu: add engine_retains_context to amdgpu_ring_funcs Pierre-Eric Pelloux-Prayer
2025-11-06 9:39 ` [PATCH v3 2/4] drm/amdgpu: jump to the correct label on failure Pierre-Eric Pelloux-Prayer
2025-11-06 9:39 ` [PATCH v3 3/4] drm/amdgpu: increment sched score on entity selection Pierre-Eric Pelloux-Prayer
@ 2025-11-06 9:39 ` Pierre-Eric Pelloux-Prayer
2025-11-06 9:50 ` Philipp Stanner
2 siblings, 1 reply; 11+ messages in thread
From: Pierre-Eric Pelloux-Prayer @ 2025-11-06 9:39 UTC (permalink / raw)
To: Matthew Brost, Danilo Krummrich, Philipp Stanner,
Christian König, Maarten Lankhorst, Maxime Ripard,
Thomas Zimmermann, David Airlie, Simona Vetter
Cc: Pierre-Eric Pelloux-Prayer, Tvrtko Ursulin, Tomeu Vizoso,
dri-devel, linux-kernel
Currently, the scheduler score is incremented when a job is pushed to an
entity and when an entity is attached to the scheduler.
This leads to some bad scheduling decision where the score value is
largely made of idle entities.
For instance, a scenario with 2 schedulers and where 10 entities submit
a single job, then do nothing, each scheduler will probably end up with
a score of 5.
Now, 5 userspace apps exit, so their entities will be dropped. In
the worst case, these apps' entities where all attached to the same
scheduler and we end up with score=5 (the 5 remaining entities) and
score=0, despite the 2 schedulers being idle.
When new entities show up, they will all select the second scheduler
based on its low score value, instead of alternating between the 2.
Some amdgpu rings depended on this feature, but the previous commit
implemented the same thing in amdgpu directly so it can be safely
removed from drm/sched.
Signed-off-by: Pierre-Eric Pelloux-Prayer <pierre-eric.pelloux-prayer@amd.com>
Reviewed-by: Tvrtko Ursulin <tvrtko.ursulin@igalia.com>
Acked-by: Tomeu Vizoso <tomeu@tomeuvizoso.net>
---
drivers/gpu/drm/scheduler/sched_main.c | 2 --
1 file changed, 2 deletions(-)
diff --git a/drivers/gpu/drm/scheduler/sched_main.c b/drivers/gpu/drm/scheduler/sched_main.c
index c39f0245e3a9..8a3d99a86090 100644
--- a/drivers/gpu/drm/scheduler/sched_main.c
+++ b/drivers/gpu/drm/scheduler/sched_main.c
@@ -206,7 +206,6 @@ void drm_sched_rq_add_entity(struct drm_sched_rq *rq,
if (!list_empty(&entity->list))
return;
- atomic_inc(rq->sched->score);
list_add_tail(&entity->list, &rq->entities);
}
@@ -228,7 +227,6 @@ void drm_sched_rq_remove_entity(struct drm_sched_rq *rq,
spin_lock(&rq->lock);
- atomic_dec(rq->sched->score);
list_del_init(&entity->list);
if (rq->current_entity == entity)
--
2.43.0
^ permalink raw reply [flat|nested] 11+ messages in thread
* Re: [PATCH v3 4/4] drm/sched: limit sched score update to jobs change
2025-11-06 9:39 ` [PATCH v3 4/4] drm/sched: limit sched score update to jobs change Pierre-Eric Pelloux-Prayer
@ 2025-11-06 9:50 ` Philipp Stanner
0 siblings, 0 replies; 11+ messages in thread
From: Philipp Stanner @ 2025-11-06 9:50 UTC (permalink / raw)
To: Pierre-Eric Pelloux-Prayer, Matthew Brost, Danilo Krummrich,
Philipp Stanner, Christian König, Maarten Lankhorst,
Maxime Ripard, Thomas Zimmermann, David Airlie, Simona Vetter
Cc: Tvrtko Ursulin, Tomeu Vizoso, dri-devel, linux-kernel
nit: s/limit/Limit
On Thu, 2025-11-06 at 10:39 +0100, Pierre-Eric Pelloux-Prayer wrote:
> Currently, the scheduler score is incremented when a job is pushed to an
> entity and when an entity is attached to the scheduler.
>
> This leads to some bad scheduling decision where the score value is
> largely made of idle entities.
>
> For instance, a scenario with 2 schedulers and where 10 entities submit
> a single job, then do nothing, each scheduler will probably end up with
> a score of 5.
> Now, 5 userspace apps exit, so their entities will be dropped. In
s/Now,/Now, let's imagine
> the worst case, these apps' entities where all attached to the same
s/where/were
> scheduler and we end up with score=5 (the 5 remaining entities) and
> score=0, despite the 2 schedulers being idle.
> When new entities show up, they will all select the second scheduler
> based on its low score value, instead of alternating between the 2.
>
> Some amdgpu rings depended on this feature, but the previous commit
> implemented the same thing in amdgpu directly so it can be safely
> removed from drm/sched.
>
> Signed-off-by: Pierre-Eric Pelloux-Prayer <pierre-eric.pelloux-prayer@amd.com>
> Reviewed-by: Tvrtko Ursulin <tvrtko.ursulin@igalia.com>
> Acked-by: Tomeu Vizoso <tomeu@tomeuvizoso.net>
With the commit message fixed up a little bit:
Acked-by: Philipp Stanner <phasta@kernel.org
Apply how you want :)
P.
> ---
> drivers/gpu/drm/scheduler/sched_main.c | 2 --
> 1 file changed, 2 deletions(-)
>
> diff --git a/drivers/gpu/drm/scheduler/sched_main.c b/drivers/gpu/drm/scheduler/sched_main.c
> index c39f0245e3a9..8a3d99a86090 100644
> --- a/drivers/gpu/drm/scheduler/sched_main.c
> +++ b/drivers/gpu/drm/scheduler/sched_main.c
> @@ -206,7 +206,6 @@ void drm_sched_rq_add_entity(struct drm_sched_rq *rq,
> if (!list_empty(&entity->list))
> return;
>
> - atomic_inc(rq->sched->score);
> list_add_tail(&entity->list, &rq->entities);
> }
>
> @@ -228,7 +227,6 @@ void drm_sched_rq_remove_entity(struct drm_sched_rq *rq,
>
> spin_lock(&rq->lock);
>
> - atomic_dec(rq->sched->score);
> list_del_init(&entity->list);
>
> if (rq->current_entity == entity)
^ permalink raw reply [flat|nested] 11+ messages in thread
end of thread, other threads:[~2025-11-07 9:05 UTC | newest]
Thread overview: 11+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2025-11-06 9:39 [PATCH v3 1/4] drm/amdgpu: add engine_retains_context to amdgpu_ring_funcs Pierre-Eric Pelloux-Prayer
2025-11-06 9:39 ` [PATCH v3 2/4] drm/amdgpu: jump to the correct label on failure Pierre-Eric Pelloux-Prayer
2025-11-06 9:56 ` Tvrtko Ursulin
2025-11-06 10:21 ` Christian König
2025-11-07 9:05 ` Pierre-Eric Pelloux-Prayer
2025-11-06 9:39 ` [PATCH v3 3/4] drm/amdgpu: increment sched score on entity selection Pierre-Eric Pelloux-Prayer
2025-11-06 10:00 ` Tvrtko Ursulin
2025-11-06 10:10 ` Pierre-Eric Pelloux-Prayer
2025-11-06 10:23 ` Tvrtko Ursulin
2025-11-06 9:39 ` [PATCH v3 4/4] drm/sched: limit sched score update to jobs change Pierre-Eric Pelloux-Prayer
2025-11-06 9:50 ` Philipp Stanner
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®