mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Yuri Andriaccio <yurand2000@gmail.com>
To: Ingo Molnar <mingo@redhat.com>,
	Peter Zijlstra <peterz@infradead.org>,
	Juri Lelli <juri.lelli@redhat.com>,
	Vincent Guittot <vincent.guittot@linaro.org>,
	Dietmar Eggemann <dietmar.eggemann@arm.com>,
	Steven Rostedt <rostedt@goodmis.org>,
	Ben Segall <bsegall@google.com>, Mel Gorman <mgorman@suse.de>,
	Valentin Schneider <vschneid@redhat.com>
Cc: linux-kernel@vger.kernel.org,
	Luca Abeni <luca.abeni@santannapisa.it>,
	Yuri Andriaccio <yuri.andriaccio@santannapisa.it>
Subject: [RFC PATCH v4 12/28] sched/rt: Implement dl-server operations for rt-cgroups.
Date: Mon,  1 Dec 2025 13:41:45 +0100	[thread overview]
Message-ID: <20251201124205.11169-13-yurand2000@gmail.com> (raw)
In-Reply-To: <20251201124205.11169-1-yurand2000@gmail.com>

- Implement rt_server_pick, the callback that deadline servers use to
  pick a task to schedule.
  - rt_server_pick(): pick the next runnable rt task and tell the
    scheduler that it is going to be scheduled next.

- Let enqueue/dequeue_task_rt function start/stop the attached deadline
  server when the first/last task is enqueued/dequeued on a specific
  rq/server.

- Change update_curr_rt to perform a deadline server update if the
  updated task is served by non-root group.

- Update inc/dec_dl_tasks to account the number of active tasks in the
  local runqueue for rt-cgroups servers, as their local runqueue is
  different from the global runqueue, and thus when a rt-group server is
  activated/deactivated, the number of served tasks' must be
  added/removed. This uses nr_running to be compatible with future
  dl-server interfaces.

- Update inc/dec_rt_prio_smp to change a rq's cpupri only if the rt_rq
  is the global runqueue, since cgroups are scheduled via their
  dl-server priority.

- Update inc/dec_rt_tasks to account for waking/sleeping tasks on the
  global runqueue, when the task runs on the root cgroup, or its local
  dl server is active. The accounting is not done when servers are
  throttled, as they will add/sub the number of tasks running when they
  get enqueued/dequeued. For rt cgroups, account for the number of active
  tasks in the nr_running field of the local runqueue
  (add/sub_nr_running), as this number is used when a dl server is
  enqueued/dequeued.

- Update set_task_rq to record the dl_rq, tracking which deadline
  server manages a task.

- Update set_task_rq to not use the parent field anymore, as it is
  unused by this patchset's code. Remove the unused parent field from
  sched_rt_entity.

Co-developed-by: Alessio Balsini <a.balsini@sssup.it>
Signed-off-by: Alessio Balsini <a.balsini@sssup.it>
Co-developed-by: Andrea Parri <parri.andrea@gmail.com>
Signed-off-by: Andrea Parri <parri.andrea@gmail.com>
Co-developed-by: luca abeni <luca.abeni@santannapisa.it>
Signed-off-by: luca abeni <luca.abeni@santannapisa.it>
Signed-off-by: Yuri Andriaccio <yurand2000@gmail.com>
---
 include/linux/sched.h   |  1 -
 kernel/sched/deadline.c |  8 +++++
 kernel/sched/rt.c       | 68 ++++++++++++++++++++++++++++++++++++++---
 kernel/sched/sched.h    |  8 ++++-
 4 files changed, 79 insertions(+), 6 deletions(-)

diff --git a/include/linux/sched.h b/include/linux/sched.h
index 000aa3b2b1..3f1f15b6d2 100644
--- a/include/linux/sched.h
+++ b/include/linux/sched.h
@@ -629,7 +629,6 @@ struct sched_rt_entity {

 	struct sched_rt_entity		*back;
 #ifdef CONFIG_RT_GROUP_SCHED
-	struct sched_rt_entity		*parent;
 	/* rq on which this entity is (to be) queued: */
 	struct rt_rq			*rt_rq;
 	/* rq "owned" by this entity/group: */
diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c
index 089fd2c9b7..b890fdd4b2 100644
--- a/kernel/sched/deadline.c
+++ b/kernel/sched/deadline.c
@@ -1847,6 +1847,10 @@ void inc_dl_tasks(struct sched_dl_entity *dl_se, struct dl_rq *dl_rq)

 	if (!dl_server(dl_se))
 		add_nr_running(rq_of_dl_rq(dl_rq), 1);
+	else if (rq_of_dl_se(dl_se) != dl_se->my_q) {
+		WARN_ON(dl_se->my_q->rt.rt_nr_running != dl_se->my_q->nr_running);
+		add_nr_running(rq_of_dl_rq(dl_rq), dl_se->my_q->nr_running);
+	}

 	inc_dl_deadline(dl_rq, deadline);
 }
@@ -1859,6 +1863,10 @@ void dec_dl_tasks(struct sched_dl_entity *dl_se, struct dl_rq *dl_rq)

 	if (!dl_server(dl_se))
 		sub_nr_running(rq_of_dl_rq(dl_rq), 1);
+	else if (rq_of_dl_se(dl_se) != dl_se->my_q) {
+		WARN_ON(dl_se->my_q->rt.rt_nr_running != dl_se->my_q->nr_running);
+		sub_nr_running(rq_of_dl_rq(dl_rq), dl_se->my_q->nr_running);
+	}

 	dec_dl_deadline(dl_rq, dl_se->deadline);
 }
diff --git a/kernel/sched/rt.c b/kernel/sched/rt.c
index 2301efc03f..7ec117a18d 100644
--- a/kernel/sched/rt.c
+++ b/kernel/sched/rt.c
@@ -128,9 +128,22 @@ void free_rt_sched_group(struct task_group *tg)
 	kfree(tg->dl_se);
 }

+static struct sched_rt_entity *pick_next_rt_entity(struct rt_rq *rt_rq);
+static inline void set_next_task_rt(struct rq *rq, struct task_struct *p, bool first);
+
 static struct task_struct *rt_server_pick(struct sched_dl_entity *dl_se)
 {
-	return NULL;
+	struct rt_rq *rt_rq = &dl_se->my_q->rt;
+	struct rq *rq = rq_of_rt_rq(rt_rq);
+	struct task_struct *p;
+
+	if (dl_se->my_q->rt.rt_nr_running == 0)
+		return NULL;
+
+	p = rt_task_of(pick_next_rt_entity(rt_rq));
+	set_next_task_rt(rq, p, true);
+
+	return p;
 }

 static inline void __rt_rq_free(struct rt_rq **rt_rq)
@@ -435,6 +448,7 @@ static inline int rt_se_prio(struct sched_rt_entity *rt_se)
 static void update_curr_rt(struct rq *rq)
 {
 	struct task_struct *donor = rq->donor;
+	struct rt_rq *rt_rq;
 	s64 delta_exec;

 	if (donor->sched_class != &rt_sched_class)
@@ -444,8 +458,18 @@ static void update_curr_rt(struct rq *rq)
 	if (unlikely(delta_exec <= 0))
 		return;

-	if (!rt_bandwidth_enabled())
+	if (!rt_group_sched_enabled())
 		return;
+
+	if (!dl_bandwidth_enabled())
+		return;
+
+	rt_rq = rt_rq_of_se(&donor->rt);
+	if (is_dl_group(rt_rq)) {
+		struct sched_dl_entity *dl_se = dl_group_of(rt_rq);
+
+		dl_server_update(dl_se, delta_exec);
+	}
 }

 static void
@@ -456,7 +480,7 @@ inc_rt_prio_smp(struct rt_rq *rt_rq, int prio, int prev_prio)
 	/*
 	 * Change rq's cpupri only if rt_rq is the top queue.
 	 */
-	if (IS_ENABLED(CONFIG_RT_GROUP_SCHED) && &rq->rt != rt_rq)
+	if (IS_ENABLED(CONFIG_RT_GROUP_SCHED) && is_dl_group(rt_rq))
 		return;

 	if (rq->online && prio < prev_prio)
@@ -471,7 +495,7 @@ dec_rt_prio_smp(struct rt_rq *rt_rq, int prio, int prev_prio)
 	/*
 	 * Change rq's cpupri only if rt_rq is the top queue.
 	 */
-	if (IS_ENABLED(CONFIG_RT_GROUP_SCHED) && &rq->rt != rt_rq)
+	if (IS_ENABLED(CONFIG_RT_GROUP_SCHED) && is_dl_group(rt_rq))
 		return;

 	if (rq->online && rt_rq->highest_prio.curr != prev_prio)
@@ -534,6 +558,16 @@ void inc_rt_tasks(struct sched_rt_entity *rt_se, struct rt_rq *rt_rq)
 	rt_rq->rr_nr_running += is_rr_task(rt_se);

 	inc_rt_prio(rt_rq, rt_se_prio(rt_se));
+
+	if (IS_ENABLED(CONFIG_RT_GROUP_SCHED) && is_dl_group(rt_rq)) {
+		struct sched_dl_entity *dl_se = dl_group_of(rt_rq);
+
+		if (!dl_se->dl_throttled)
+			add_nr_running(rq_of_rt_rq(rt_rq), 1);
+		add_nr_running(served_rq_of_rt_rq(rt_rq), 1);
+	} else {
+		add_nr_running(rq_of_rt_rq(rt_rq), 1);
+	}
 }

 static inline
@@ -544,6 +578,16 @@ void dec_rt_tasks(struct sched_rt_entity *rt_se, struct rt_rq *rt_rq)
 	rt_rq->rr_nr_running -= is_rr_task(rt_se);

 	dec_rt_prio(rt_rq, rt_se_prio(rt_se));
+
+	if (IS_ENABLED(CONFIG_RT_GROUP_SCHED) && is_dl_group(rt_rq)) {
+		struct sched_dl_entity *dl_se = dl_group_of(rt_rq);
+
+		if (!dl_se->dl_throttled)
+			sub_nr_running(rq_of_rt_rq(rt_rq), 1);
+		sub_nr_running(served_rq_of_rt_rq(rt_rq), 1);
+	} else {
+		sub_nr_running(rq_of_rt_rq(rt_rq), 1);
+	}
 }

 /*
@@ -725,6 +769,14 @@ enqueue_task_rt(struct rq *rq, struct task_struct *p, int flags)
 	check_schedstat_required();
 	update_stats_wait_start_rt(rt_rq_of_se(rt_se), rt_se);

+	/* Task arriving in an idle group of tasks. */
+	if (IS_ENABLED(CONFIG_RT_GROUP_SCHED) &&
+	    is_dl_group(rt_rq) && rt_rq->rt_nr_running == 0) {
+		struct sched_dl_entity *dl_se = dl_group_of(rt_rq);
+
+		dl_server_start(dl_se);
+	}
+
 	enqueue_rt_entity(rt_se, flags);

 	if (task_is_blocked(p))
@@ -744,6 +796,14 @@ static bool dequeue_task_rt(struct rq *rq, struct task_struct *p, int flags)

 	dequeue_pushable_task(rt_rq, p);

+	/* Last task of the task group. */
+	if (IS_ENABLED(CONFIG_RT_GROUP_SCHED) &&
+	    is_dl_group(rt_rq) && rt_rq->rt_nr_running == 0) {
+		struct sched_dl_entity *dl_se = dl_group_of(rt_rq);
+
+		dl_server_stop(dl_se);
+	}
+
 	return true;
 }

diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h
index f42bef06a9..fb4dcb4551 100644
--- a/kernel/sched/sched.h
+++ b/kernel/sched/sched.h
@@ -2203,7 +2203,7 @@ static inline void set_task_rq(struct task_struct *p, unsigned int cpu)
 	if (!rt_group_sched_enabled())
 		tg = &root_task_group;
 	p->rt.rt_rq  = tg->rt_rq[cpu];
-	p->rt.parent = tg->rt_se[cpu];
+	p->dl.dl_rq  = &cpu_rq(cpu)->dl;
 #endif /* CONFIG_RT_GROUP_SCHED */
 }

@@ -2750,6 +2750,9 @@ static inline void add_nr_running(struct rq *rq, unsigned count)
 	unsigned prev_nr = rq->nr_running;

 	rq->nr_running = prev_nr + count;
+	if (rq != cpu_rq(rq->cpu))
+		return;
+
 	if (trace_sched_update_nr_running_tp_enabled()) {
 		call_trace_sched_update_nr_running(rq, count);
 	}
@@ -2763,6 +2766,9 @@ static inline void add_nr_running(struct rq *rq, unsigned count)
 static inline void sub_nr_running(struct rq *rq, unsigned count)
 {
 	rq->nr_running -= count;
+	if (rq != cpu_rq(rq->cpu))
+		return;
+
 	if (trace_sched_update_nr_running_tp_enabled()) {
 		call_trace_sched_update_nr_running(rq, -count);
 	}
--
2.51.0


  parent reply	other threads:[~2025-12-01 12:42 UTC|newest]

Thread overview: 47+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2025-12-01 12:41 [RFC PATCH v4 00/28] Hierarchical Constant Bandwidth Server Yuri Andriaccio
2025-12-01 12:41 ` [RFC PATCH v4 01/28] sched/deadline: Do not access dl_se->rq directly Yuri Andriaccio
2026-01-14 11:03   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 02/28] sched/deadline: Distinct between dl_rq and my_q Yuri Andriaccio
2026-01-14 16:10   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 03/28] sched/rt: Pass an rt_rq instead of an rq where needed Yuri Andriaccio
2026-01-15 14:33   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 04/28] sched/rt: Move some functions from rt.c to sched.h Yuri Andriaccio
2026-01-15 14:40   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 05/28] sched/rt: Disable RT_GROUP_SCHED Yuri Andriaccio
2026-01-15 15:56   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 06/28] sched/rt: Remove rq field in struct rt_rq Yuri Andriaccio
2026-01-15 16:12   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 07/28] sched/rt: Introduce HCBS specific structs in task_group Yuri Andriaccio
2026-01-16 13:50   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 08/28] sched/core: Initialize HCBS specific structures Yuri Andriaccio
2026-01-16 14:09   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 09/28] sched/deadline: Add dl_init_tg Yuri Andriaccio
2026-01-16 15:02   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 10/28] sched/rt: Add {alloc/free}_rt_sched_group Yuri Andriaccio
2026-01-19 13:49   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 11/28] sched/deadline: Account rt-cgroups bandwidth in deadline tasks schedulability tests Yuri Andriaccio
2026-01-19 14:00   ` Juri Lelli
2025-12-01 12:41 ` Yuri Andriaccio [this message]
2026-01-19 15:01   ` [RFC PATCH v4 12/28] sched/rt: Implement dl-server operations for rt-cgroups Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 13/28] sched/rt: Update task event callbacks for HCBS scheduling Yuri Andriaccio
2026-01-21  9:46   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 14/28] sched/rt: Update rt-cgroup schedulability checks Yuri Andriaccio
2026-01-21 10:21   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 15/28] sched/rt: Allow zeroing the runtime of the root control group Yuri Andriaccio
2026-01-21 10:32   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 16/28] sched/rt: Remove old RT_GROUP_SCHED data structures Yuri Andriaccio
2026-01-21 10:41   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 17/28] sched/core: Cgroup v2 support Yuri Andriaccio
2026-01-21 10:51   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 18/28] sched/rt: Remove support for cgroups-v1 Yuri Andriaccio
2026-01-21 11:08   ` Juri Lelli
2025-12-01 12:41 ` [RFC PATCH v4 19/28] sched/deadline: Allow deeper hierarchies of RT cgroups Yuri Andriaccio
2025-12-01 12:41 ` [RFC PATCH v4 20/28] sched/rt: Add rt-cgroup migration Yuri Andriaccio
2025-12-01 12:41 ` [RFC PATCH v4 21/28] sched/rt: Add HCBS migration related checks and function calls Yuri Andriaccio
2025-12-01 12:41 ` [RFC PATCH v4 22/28] sched/deadline: Introduce dl_server_try_pull_f Yuri Andriaccio
2025-12-01 12:41 ` [RFC PATCH v4 23/28] sched/deadline: Fix HCBS migrations on server stop Yuri Andriaccio
2025-12-01 12:41 ` [RFC PATCH v4 24/28] sched/core: Execute enqueued balance callbacks when changing allowed CPUs Yuri Andriaccio
2025-12-01 12:41 ` [RFC PATCH v4 25/28] sched/core: Execute enqueued balance callbacks when migrating task betweeen cgroups Yuri Andriaccio
2025-12-01 12:41 ` [RFC PATCH v4 26/28] Documentation: Update documentation for real-time cgroups Yuri Andriaccio
2025-12-01 12:42 ` [RFC PATCH v4 27/28] [DEBUG] sched/rt: Add debug BUG_ONs for pre-migration code Yuri Andriaccio
2025-12-01 12:42 ` [RFC PATCH v4 28/28] [DEBUG] sched/rt: Add debug BUG_ONs in migration code Yuri Andriaccio

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20251201124205.11169-13-yurand2000@gmail.com \
    --to=yurand2000@gmail.com \
    --cc=bsegall@google.com \
    --cc=dietmar.eggemann@arm.com \
    --cc=juri.lelli@redhat.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=luca.abeni@santannapisa.it \
    --cc=mgorman@suse.de \
    --cc=mingo@redhat.com \
    --cc=peterz@infradead.org \
    --cc=rostedt@goodmis.org \
    --cc=vincent.guittot@linaro.org \
    --cc=vschneid@redhat.com \
    --cc=yuri.andriaccio@santannapisa.it \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

Powered by JetHome