sched: Introduce task block time in schedstats
authorYafang Shao <laoar.shao@gmail.com>
Sun, 5 Sep 2021 14:35:43 +0000 (14:35 +0000)
committerPeter Zijlstra <peterz@infradead.org>
Tue, 5 Oct 2021 13:51:48 +0000 (15:51 +0200)
Currently in schedstats we have sum_sleep_runtime and iowait_sum, but
there's no metric to show how long the task is in D state.  Once a task in
D state, it means the task is blocked in the kernel, for example the
task may be waiting for a mutex. The D state is more frequent than
iowait, and it is more critital than S state. So it is worth to add a
metric to measure it.

Signed-off-by: Yafang Shao <laoar.shao@gmail.com>
Signed-off-by: Peter Zijlstra (Intel) <peterz@infradead.org>
Link: https://lore.kernel.org/r/20210905143547.4668-5-laoar.shao@gmail.com
include/linux/sched.h
kernel/sched/debug.c
kernel/sched/stats.c

index 2bc4c72..193e16e 100644 (file)
@@ -503,6 +503,8 @@ struct sched_statistics {
 
        u64                             block_start;
        u64                             block_max;
+       s64                             sum_block_runtime;
+
        u64                             exec_max;
        u64                             slice_max;
 
index 76fd38e..26fac5e 100644 (file)
@@ -540,10 +540,11 @@ print_task(struct seq_file *m, struct rq *rq, struct task_struct *p)
                (long long)(p->nvcsw + p->nivcsw),
                p->prio);
 
-       SEQ_printf(m, "%9Ld.%06ld %9Ld.%06ld %9Ld.%06ld",
+       SEQ_printf(m, "%9lld.%06ld %9lld.%06ld %9lld.%06ld %9lld.%06ld",
                SPLIT_NS(schedstat_val_or_zero(p->stats.wait_sum)),
                SPLIT_NS(p->se.sum_exec_runtime),
-               SPLIT_NS(schedstat_val_or_zero(p->stats.sum_sleep_runtime)));
+               SPLIT_NS(schedstat_val_or_zero(p->stats.sum_sleep_runtime)),
+               SPLIT_NS(schedstat_val_or_zero(p->stats.sum_block_runtime)));
 
 #ifdef CONFIG_NUMA_BALANCING
        SEQ_printf(m, " %d %d", task_node(p), task_numa_group_id(p));
@@ -977,6 +978,7 @@ void proc_sched_show_task(struct task_struct *p, struct pid_namespace *ns,
                u64 avg_atom, avg_per_cpu;
 
                PN_SCHEDSTAT(sum_sleep_runtime);
+               PN_SCHEDSTAT(sum_block_runtime);
                PN_SCHEDSTAT(wait_start);
                PN_SCHEDSTAT(sleep_start);
                PN_SCHEDSTAT(block_start);
index fad781c..07dde29 100644 (file)
@@ -82,6 +82,7 @@ void __update_stats_enqueue_sleeper(struct rq *rq, struct task_struct *p,
 
                __schedstat_set(stats->block_start, 0);
                __schedstat_add(stats->sum_sleep_runtime, delta);
+               __schedstat_add(stats->sum_block_runtime, delta);
 
                if (p) {
                        if (p->in_iowait) {