mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Tao Cui <cui.tao@linux.dev>
To: tj@kernel.org, josef@toxicopanda.com, axboe@kernel.dk
Cc: cgroups@vger.kernel.org, linux-block@vger.kernel.org,
	linux-kernel@vger.kernel.org, bpf@vger.kernel.org,
	andrii@kernel.org, ast@kernel.org, daniel@iogearbox.net,
	linux-kselftest@vger.kernel.org, cui.tao@linux.dev,
	cuitao@kylinos.cn
Subject: [RFC PATCH v3 3/5] blk-iocost: add iocost_ioc_tick tracepoint for per-period device summary
Date: Mon, 14 Sep 2026 15:33:54 +0800	[thread overview]
Message-ID: <20260914073356.791518-4-cui.tao@linux.dev> (raw)
In-Reply-To: <20260914073356.791518-1-cui.tao@linux.dev>

From: Tao Cui <cuitao@kylinos.cn>

The existing iocost tracepoints are state-change driven: vrate_adj
fires only when the adjustment logic runs, inuse_* only on surplus
state transitions, activate/idle only on cgroup state changes.  In a
steady state none of them fire.  The only other way to observe the
controller (period length, vrate, busy level, active cgroup count,
device utilization) is iocost_monitor.py, which reads kernel memory
through drgn and is not usable in most production environments.

Add iocost_ioc_tick, emitted once per period from the tail of
ioc_timer_fn() with the overall controller state: the period number,
period_us, vrate, busy_level, active iocg count, usage percentage
and running state.  It fires every period the controller runs, including steady states,
plus one final tick before the controller goes idle, which makes
dormancy (e.g. a device saturated entirely by uncharged IO) directly
visible.

Depending on the autop profile this is 2-100 events per second per
device; the added cost outside the tracepoint static key is one
increment per active cgroup per period.

Signed-off-by: Tao Cui <cuitao@kylinos.cn>
---
 block/blk-iocost.c            | 51 ++++++++++++++++++++++++++++-------
 include/trace/events/iocost.h | 46 +++++++++++++++++++++++++++++++
 2 files changed, 88 insertions(+), 9 deletions(-)

diff --git a/block/blk-iocost.c b/block/blk-iocost.c
index 3cc21092cf47f..96126a185f9a6 100644
--- a/block/blk-iocost.c
+++ b/block/blk-iocost.c
@@ -2244,6 +2244,7 @@ static void ioc_timer_fn(struct timer_list *timer)
 	struct ioc_now now;
 	LIST_HEAD(surpluses);
 	int nr_debtors, nr_shortages = 0, nr_lagging = 0;
+	int nr_active = 0;
 	u64 usage_us_sum = 0;
 	u32 ppm_rthr;
 	u32 ppm_wthr;
@@ -2280,6 +2281,8 @@ static void ioc_timer_fn(struct timer_list *timer)
 		u64 vdone, vtime, usage_us;
 		u32 hw_active, hw_inuse;
 
+		nr_active++;
+
 		/*
 		 * Collect unused and wind vtime closer to vnow to prevent
 		 * iocgs from accumulating a large amount of budget.
@@ -2441,6 +2444,14 @@ static void ioc_timer_fn(struct timer_list *timer)
 
 	ioc->busy_level = clamp(ioc->busy_level, -1000, 1000);
 
+	/*
+	 * vrate and period_us change right below; snapshot the values
+	 * this period ran in so the tick below reports the period's own
+	 * parameters instead of the next period's.
+	 */
+	u32 tick_period_us = ioc->period_us;
+	u64 tick_vrate = ioc->vtime_base_rate;
+
 	ioc_adjust_base_vrate(ioc, rq_wait_pct, nr_lagging, nr_shortages,
 			      prev_busy_level, missed_ppm);
 
@@ -2454,16 +2465,38 @@ static void ioc_timer_fn(struct timer_list *timer)
 	 */
 	atomic64_inc(&ioc->cur_period);
 
-	if (ioc->running != IOC_STOP) {
-		if (!list_empty(&ioc->active_iocgs)) {
-			ioc_start_period(ioc, &now);
-		} else {
-			ioc->busy_level = 0;
-			ioc->vtime_err = 0;
-			ioc->running = IOC_IDLE;
-		}
+	/*
+	 * Snapshot the state this period ran in before the idle
+	 * transition wipes it, so the final tick reports the period's
+	 * own busy level (e.g. the saturation that drove the controller
+	 * idle) instead of the cleared one.  usage is normalized by the
+	 * measured period length, captured before ioc_start_period()
+	 * overwrites period_at, the same way the donation loop does.
+	 */
+	{
+		int tick_busy = ioc->busy_level;
+		int tick_running = ioc->running;
+		u64 tick_dur = now.now - ioc->period_at;
+		/* cur_period was already advanced to N+1 above; report
+		 * the period that just ended, like the other fields */
+		u64 tick_period = atomic64_read(&ioc->cur_period) - 1;
+
+		trace_iocost_ioc_tick(ioc, nr_active, usage_us_sum,
+				      tick_period, tick_period_us,
+				      tick_vrate, tick_busy, tick_running,
+				      tick_dur);
+
+		if (ioc->running != IOC_STOP) {
+			if (!list_empty(&ioc->active_iocgs)) {
+				ioc_start_period(ioc, &now);
+			} else {
+				ioc->busy_level = 0;
+				ioc->vtime_err = 0;
+				ioc->running = IOC_IDLE;
+			}
 
-		ioc_refresh_vrate(ioc, &now);
+			ioc_refresh_vrate(ioc, &now);
+		}
 	}
 
 	spin_unlock_irq(&ioc->lock);
diff --git a/include/trace/events/iocost.h b/include/trace/events/iocost.h
index e772b1bc60d60..ec5d9c453d55a 100644
--- a/include/trace/events/iocost.h
+++ b/include/trace/events/iocost.h
@@ -178,6 +178,52 @@ TRACE_EVENT(iocost_ioc_vrate_adj,
 	)
 );
 
+/*
+ * Periodic per-device summary, emitted once per period from the tail of
+ * ioc_timer_fn().  Unlike the state-change events above, this fires every
+ * period the controller is running, including steady states, and carries
+ * the overall controller state so basic monitoring doesn't require drgn.
+ */
+TRACE_EVENT(iocost_ioc_tick,
+
+	TP_PROTO(struct ioc *ioc, int nr_active, u64 usage_us_sum,
+		 u64 cur_period, u32 tick_period_us, u64 tick_vrate,
+		 int tick_busy, int tick_running, u64 tick_dur),
+
+	TP_ARGS(ioc, nr_active, usage_us_sum, cur_period,
+		tick_period_us, tick_vrate,
+		tick_busy, tick_running, tick_dur),
+
+	TP_STRUCT__entry (
+		__string(devname, ioc_name(ioc))
+		__field(u64, cur_period)
+		__field(u32, period_us)
+		__field(u64, vrate)
+		__field(int, busy_level)
+		__field(int, nr_active)
+		__field(u32, usage_pct)
+		__field(int, running)
+	),
+
+	TP_fast_assign(
+		__assign_str(devname);
+		__entry->cur_period = cur_period;
+		__entry->period_us = tick_period_us;
+		__entry->vrate = tick_vrate;
+		__entry->busy_level = tick_busy;
+		__entry->nr_active = nr_active;
+		__entry->usage_pct = tick_dur ?
+			div_u64(usage_us_sum * 100, tick_dur) : 0;
+		__entry->running = tick_running;
+	),
+
+	TP_printk("[%s] period=%llu:%uus vrate=%llu busy=%d active=%d usage=%u%% running=%d",
+		__get_str(devname), __entry->cur_period, __entry->period_us,
+		__entry->vrate, __entry->busy_level, __entry->nr_active,
+		__entry->usage_pct, __entry->running
+	)
+);
+
 TRACE_EVENT(iocost_iocg_forgive_debt,
 
 	TP_PROTO(struct ioc_gq *iocg, const char *path, struct ioc_now *now,
-- 
2.43.0


  parent reply	other threads:[~2026-09-14  7:34 UTC|newest]

Thread overview: 6+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-14  7:33 [RFC PATCH v3 0/5] blk-iocost: BPF struct_ops cost model Tao Cui
2026-09-14  7:33 ` [RFC PATCH v3 1/5] blk-iocost: add BPF struct_ops cost model support Tao Cui
2026-09-14  7:33 ` [RFC PATCH v3 2/5] selftests/bpf: add iocost cost model test Tao Cui
2026-09-14  7:33 ` Tao Cui [this message]
2026-09-14  7:33 ` [RFC PATCH v3 4/5] selftests/bpf: add multi-stream sequentiality example model Tao Cui
2026-09-14  7:33 ` [RFC PATCH v3 5/5] docs: cgroup-v2: document io.cost model=<name> binding Tao Cui

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260914073356.791518-4-cui.tao@linux.dev \
    --to=cui.tao@linux.dev \
    --cc=andrii@kernel.org \
    --cc=ast@kernel.org \
    --cc=axboe@kernel.dk \
    --cc=bpf@vger.kernel.org \
    --cc=cgroups@vger.kernel.org \
    --cc=cuitao@kylinos.cn \
    --cc=daniel@iogearbox.net \
    --cc=josef@toxicopanda.com \
    --cc=linux-block@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-kselftest@vger.kernel.org \
    --cc=tj@kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®