[PATCH v3] sched_ext: Add tracepoint for scheduler exit

Pat Somaru <[email protected]>
Newsgroups dev.linux.lists.sched-ext,org.kernel.vger.linux-kernel
Message-ID <[email protected]>
sched_ext schedulers have state in BPF programs and kernel. scx_dump
provides kernel state and BPF program state on error, but this is static
in what it can provide.

Add a sched_ext_exit tracepoint in scx_claim_exit() so that BPF programs
can dynamically inspect scheduler specific state at the moment of exit.
Pass the exiting scx_sched so attached programs can read its state, and,
since exits propagate through a hierarchy of sub-schedulers, identify
which scheduler each event belongs to.

Signed-off-by: Pat Somaru <[email protected]>
---
Hi Tejun,

> Thanks, but name and level don't pin down a scheduler in a hierarchy -
> same-program instances under sibling cgroups share both. Recording
> sub_cgroup_id and cgrp_path too would make the events self-identifying. Both
> are already on sch. Care to add them in a v3?

Good idea, I added this, thanks!

v3: also record sub_cgroup_id and cgrp_path so events are self-identifying
    in a scheduler hierarchy - name and level don't pin down an instance
    when the same program runs under sibling cgroups (Tejun)
v2: https://lore.kernel.org/all/[email protected]/
v1: https://lore.kernel.org/all/[email protected]/

Tested by running scx_mitosis, attaching to the raw tracepoint and
unregistering the scheduler:

  # bpftrace -e 'rawtracepoint:sched_ext_exit {
        $sch = (struct scx_sched *)arg0;
        printf("sch=%p name=%s level=%d sub_cgroup_id=%llu cgrp_path=%s kind=%ld\n",
               $sch, $sch->ops.name, $sch->level, $sch->ops.sub_cgroup_id,
               str($sch->cgrp_path), (int64)arg1); }'

  Attached 1 probe
  sch=0xff3bd35691464000 name=mitosis_1.1.0_x86_64_unknown_linux_gnu level=0 sub_cgroup_id=0 cgrp_path=/ kind=64

and via the tracepoint's formatted event:

  # grep 'sched_ext_exit:' /sys/kernel/tracing/trace
  scx_mitosis-51 [000] ...1. 30.552740: sched_ext_exit: sched mitosis_1.1.0_x86_64_unknown_linux_gnu level 0 sub_cgroup_id 0 cgrp_path / kind 64

 include/trace/events/sched_ext.h | 28 ++++++++++++++++++++++++++++
 kernel/sched/ext/ext.c           |  2 ++
 kernel/sched/ext/sub.h           |  6 ++++++
 3 files changed, 36 insertions(+)

diff --git a/include/trace/events/sched_ext.h b/include/trace/events/sched_ext.h
index d1bf5acd59c5..1e54f9564ef3 100644
--- a/include/trace/events/sched_ext.h
+++ b/include/trace/events/sched_ext.h
@@ -84,6 +84,34 @@ TRACE_EVENT(sched_ext_bypass_lb,
 	)
 );
 
+TRACE_EVENT(sched_ext_exit,
+
+	TP_PROTO(struct scx_sched *sch, __u32 kind),
+
+	TP_ARGS(sch, kind),
+
+	TP_STRUCT__entry(
+		__string(	name,		sch->ops.name		)
+		__field(	__s32,		level			)
+		__field(	__u64,		sub_cgroup_id		)
+		__string(	cgrp_path,	sch_cgrp_path(sch)	)
+		__field(	__u32,		kind			)
+	),
+
+	TP_fast_assign(
+		__assign_str(name);
+		__entry->level		= sch->level;
+		__entry->sub_cgroup_id	= sch->ops.sub_cgroup_id;
+		__assign_str(cgrp_path);
+		__entry->kind		= kind;
+	),
+
+	TP_printk("sched %s level %d sub_cgroup_id %llu cgrp_path %s kind %u",
+		  __get_str(name), __entry->level, __entry->sub_cgroup_id,
+		  __get_str(cgrp_path), __entry->kind
+	)
+);
+
 #endif /* _TRACE_SCHED_EXT_H */
 
 /* This part must be outside protection */
diff --git a/kernel/sched/ext/ext.c b/kernel/sched/ext/ext.c
index 1a0ec985da77..6ec774435cd1 100644
--- a/kernel/sched/ext/ext.c
+++ b/kernel/sched/ext/ext.c
@@ -5730,6 +5730,8 @@ static bool scx_claim_exit(struct scx_sched *sch, enum scx_exit_kind kind)
 	 */
 	WRITE_ONCE(sch->aborting, true);
 
+	trace_sched_ext_exit(sch, kind);
+
 	/*
 	 * Propagate exits to descendants immediately. Each has a dedicated
 	 * helper kthread and can run in parallel. While most of disabling is
diff --git a/kernel/sched/ext/sub.h b/kernel/sched/ext/sub.h
index 460a9fd196dc..9b5ac07e5e76 100644
--- a/kernel/sched/ext/sub.h
+++ b/kernel/sched/ext/sub.h
@@ -25,11 +25,17 @@ void scx_sub_disable(struct scx_sched *sch);
 void scx_sub_enable_workfn(struct kthread_work *work);
 bool scx_bpf_sub_dispatch(u64 cgroup_id, const struct bpf_prog_aux *aux);
 
+static inline const char *sch_cgrp_path(struct scx_sched *sch)
+{
+	return sch->cgrp_path;
+}
+
 #else	/* CONFIG_EXT_SUB_SCHED */
 
 static inline struct scx_sched *scx_next_descendant_pre(struct scx_sched *pos, struct scx_sched *root) { return pos ? NULL : root; }
 static inline void scx_set_task_sched(struct task_struct *p, struct scx_sched *sch) {}
 static inline struct cgroup *sch_cgroup(struct scx_sched *sch) { return NULL; }
+static inline const char *sch_cgrp_path(struct scx_sched *sch) { return "/"; }
 static inline void set_cgroup_sched(struct cgroup *cgrp, struct scx_sched *sch) {}
 static inline void drain_descendants(struct scx_sched *sch) { }
 static inline void scx_sub_disable(struct scx_sched *sch) { }

base-commit: 57194a3172ba0123e8f37c4574a8e2863ab67622
-- 
2.55.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.