[PATCH v4] perf stat: Include PMU name and split uncore events per PMU in metric-only JSON output
Chun-Tse Shao <[email protected]> Tue, 4 Aug 2026 14:06:59 -0700
| Newsgroups | org.kernel.vger.linux-perf-users,org.kernel.vger.linux-kernel |
|---|---|
| Message-ID | <[email protected]> |
When running `perf stat` with `-A` (--no-aggr) and `--metric-only` in
JSON output mode (`-j`), `perf stat` evaluates metric expressions
across all matching PMUs (including uncore PMUs like `uncore_iio_0`,
`uncore_iio_1`, etc.).
However, `perf stat` previously formatted JSON output by printing only
"cpu" : "<id>" and grouping all metric values on a single line per CPU
without identifying which PMU instance evaluated each metric. As a
result, when an uncore event spans multiple PMU boxes, `perf stat`
printed repeated, ambiguous metric keys without PMU names.
Fix this by:
1. Including "pmu" : "<pmu_name>" in print_aggr_id_json when evsel->pmu
is a non-core or hybrid PMU in AGGR_NONE mode (-A).
2. Starting a new JSON metric line in AGGR_NONE mode (-A) whenever the
underlying hardware PMU instance changes across PMU events.
3. Bypassing line initialization for tool events (e.g. -e user_time) in
metric-only mode so that JSON lines are opened only by hardware PMUs,
preventing tool events from corrupting PMU name labels or creating
spurious empty {"pmu": "tool"} JSON objects.
4. Updating perf_json_output_lint.py to recognize the new "pmu" key in
the JSON test suite.
Before the fix:
$ perf stat -M iio_bandwidth_read -a -A --metric-only -j -I 1000
{"interval" : 1.001017947, "cpu" : "0", "MB/s iio_bandwidth_read" : "0.0=
", "MB/s iio_bandwidth_read" : "0.0", "MB/s iio_bandwidth_read" : "22.5",=
"MB/s iio_bandwidth_read" : "0.0", "MB/s iio_bandwidth_read" : "0.2", "M=
B/s iio_bandwidth_read" : "0.0", "MB/s iio_bandwidth_read" : "0.0", "MB/s=
iio_bandwidth_read" : "0.0", "MB/s iio_bandwidth_read" : "0.0", "MB/s i=
io_bandwidth_read" : "0.0"}
There is no way to determine which uncore device generated the metrics.
After:
$ perf stat -M iio_bandwidth_read -a -A --metric-only -j -I 1000
{"interval" : 1.000314908, "cpu" : "0", "pmu" : "uncore_iio_0", "MB/s ii=
o_bandwidth_read" : "0.0"}
{"interval" : 1.000314908, "cpu" : "0", "pmu" : "uncore_iio_1", "MB/s ii=
o_bandwidth_read" : "0.1"}
{"interval" : 1.000314908, "cpu" : "0", "pmu" : "uncore_iio_11", "MB/s i=
io_bandwidth_read" : "0.0"}
...
{"interval" : 1.000314908, "cpu" : "56", "pmu" : "uncore_iio_0", "MB/s i=
io_bandwidth_read" : "0.0"}
{"interval" : 1.000314908, "cpu" : "56", "pmu" : "uncore_iio_1", "MB/s i=
io_bandwidth_read" : "0.0"}
{"interval" : 1.000314908, "cpu" : "56", "pmu" : "uncore_iio_11", "MB/s =
iio_bandwidth_read" : "0.0"}
Signed-off-by: Chun-Tse Shao <[email protected]>
Assisted-by: Gemini:gemini-3.1-pro-preview
Reviewed-by: Ian Rogers <[email protected]>
---
v4:
- Fix the line splitting issue, causing spurious empty {"pmu": "tool"}
objects and order-dependent PMU mislabeling.
v3: lore.kernel.org/[email protected]
- Add before the fix output.
v2: lore.kernel.org/[email protected]
- Fix print_metric_begin() initialization in print_no_aggr_metric()
to ensure tool events always open lines cleanly.
- Simplify and fix PMU line-split transition tracking when switching
between different PMU instances or transitioning from tool events.
- Add bounds check on aggr_map index in print_no_aggr_metric() to
prevent out-of-bounds array reads if sysfs CPU topology is incomplete.
v1: lore.kernel.org/[email protected]
.../tests/shell/lib/perf_json_output_lint.py | 1 +
tools/perf/util/stat-display.c | 62 +++++++++++++++----
2 files changed, 51 insertions(+), 12 deletions(-)
diff --git a/tools/perf/tests/shell/lib/perf_json_output_lint.py b/tools/pe=
rf/tests/shell/lib/perf_json_output_lint.py
index dafbde56cc76..36767905fcdb 100644
--- a/tools/perf/tests/shell/lib/perf_json_output_lint.py
+++ b/tools/perf/tests/shell/lib/perf_json_output_lint.py
@@ -64,6 +64,7 @@ def check_json_output(expected_items):
'metric-threshold': lambda x: x in ['unknown', 'good', 'less good', =
'nearly bad', 'bad'],
'metricgroup': lambda x: True,
'node': lambda x: True,
+ 'pmu': lambda x: True,
'pcnt-running': lambda x: isfloat(x),
'socket': lambda x: True,
'thread': lambda x: True,
diff --git a/tools/perf/util/stat-display.c b/tools/perf/util/stat-display.=
c
index b337cc23f413..681f83c7295d 100644
--- a/tools/perf/util/stat-display.c
+++ b/tools/perf/util/stat-display.c
@@ -397,6 +397,9 @@ static void print_aggr_id_json(struct perf_stat_config =
*config, struct outstate
json_out(os, "\"cpu\" : \"%d\"",
id.cpu.cpu);
}
+ if (evsel && !evsel__is_tool(evsel) && evsel->pmu &&
+ (!evsel->pmu->is_core || evsel__is_hybrid(evsel)))
+ json_out(os, "\"pmu\" : \"%s\"", evsel->pmu->name);
break;
case AGGR_THREAD:
json_out(os, "\"thread\" : \"%s-%d\"",
@@ -1012,11 +1015,12 @@ static void print_counter_aggrdata(struct perf_stat=
_config *config,
static void print_metric_begin(struct perf_stat_config *config,
struct evlist *evlist,
- struct outstate *os, int aggr_idx)
+ struct outstate *os, int aggr_idx,
+ struct evsel *counter)
{
struct perf_stat_aggr *aggr;
struct aggr_cpu_id id;
- struct evsel *evsel;
+ struct evsel *evsel =3D counter ?: evlist__first(evlist);
os->first =3D true;
if (!config->metric_only)
@@ -1031,7 +1035,6 @@ static void print_metric_begin(struct perf_stat_confi=
g *config,
else
fprintf(config->output, "%s", os->timestamp);
}
- evsel =3D evlist__first(evlist);
id =3D config->aggr_map->map[aggr_idx];
aggr =3D &evsel->stats->aggr[aggr_idx];
aggr_printout(config, os, evsel, id, aggr->nr);
@@ -1069,7 +1072,7 @@ static void print_aggr(struct perf_stat_config *confi=
g,
* Without each counter has its own line.
*/
cpu_aggr_map__for_each_idx(aggr_idx, config->aggr_map) {
- print_metric_begin(config, evlist, os, aggr_idx);
+ print_metric_begin(config, evlist, os, aggr_idx, NULL);
evlist__for_each_entry(evlist, counter) {
print_counter_aggrdata(config, counter, aggr_idx, os);
@@ -1095,7 +1098,7 @@ static void print_aggr_cgroup(struct perf_stat_config=
*config,
os->cgrp =3D evsel->cgrp;
cpu_aggr_map__for_each_idx(aggr_idx, config->aggr_map) {
- print_metric_begin(config, evlist, os, aggr_idx);
+ print_metric_begin(config, evlist, os, aggr_idx, NULL);
evlist__for_each_entry(evlist, counter) {
if (counter->cgrp !=3D os->cgrp)
@@ -1131,7 +1134,9 @@ static void print_no_aggr_metric(struct perf_stat_con=
fig *config,
perf_cpu_map__for_each_cpu(cpu, all_idx, evlist__core(evlist)->user_reque=
sted_cpus) {
struct evsel *counter;
- bool first =3D true;
+ struct evsel *last_evsel =3D NULL;
+ struct perf_pmu *last_pmu =3D NULL;
+ bool line_open =3D false;
evlist__for_each_entry(evlist, counter) {
u64 ena, run, val;
@@ -1146,13 +1151,46 @@ static void print_no_aggr_metric(struct perf_stat_c=
onfig *config,
if (config->aggr_map->map[aggr_idx].cpu.cpu =3D=3D cpu.cpu)
break;
}
+ if (aggr_idx >=3D config->aggr_map->nr)
+ continue;
os->evsel =3D counter;
os->id =3D aggr_cpu_id__cpu(cpu, /*data=3D*/NULL);
- if (first) {
- print_metric_begin(config, evlist, os, aggr_idx);
- first =3D false;
+
+ if (config->metric_only) {
+ struct perf_pmu *pmu =3D counter->pmu;
+ bool is_tool =3D evsel__is_tool(counter);
+
+ if (config->json_output && line_open) {
+ bool pmu_changed;
+
+ if (is_tool) {
+ pmu_changed =3D last_pmu &&
+ (!last_pmu->is_core ||
+ (last_evsel &&
+ evsel__is_hybrid(last_evsel)));
+ } else if (last_pmu) {
+ pmu_changed =3D (pmu !=3D last_pmu);
+ } else {
+ pmu_changed =3D pmu &&
+ (!pmu->is_core ||
+ evsel__is_hybrid(counter));
+ }
+
+ if (pmu_changed) {
+ print_metric_end(config, os);
+ line_open =3D false;
+ }
+ }
+ if (!line_open) {
+ print_metric_begin(config, evlist, os,
+ aggr_idx, counter);
+ line_open =3D true;
+ }
+ last_pmu =3D is_tool ? NULL : pmu;
+ last_evsel =3D counter;
}
+
val =3D ps->aggr[aggr_idx].counts.val;
ena =3D ps->aggr[aggr_idx].counts.ena;
run =3D ps->aggr[aggr_idx].counts.run;
@@ -1160,7 +1198,7 @@ static void print_no_aggr_metric(struct perf_stat_con=
fig *config,
uval =3D val * counter->scale;
printout(config, os, uval, run, ena, 1.0, aggr_idx);
}
- if (!first)
+ if (line_open)
print_metric_end(config, os);
}
}
@@ -1523,7 +1561,7 @@ static void print_cgroup_counter(struct perf_stat_con=
fig *config, struct evlist
print_metric_end(config, os);
os->cgrp =3D counter->cgrp;
- print_metric_begin(config, evlist, os, /*aggr_idx=3D*/0);
+ print_metric_begin(config, evlist, os, /*aggr_idx=3D*/0, NULL);
}
print_counter(config, counter, os);
@@ -1573,7 +1611,7 @@ void evlist__print_counters(struct evlist *evlist, st=
ruct perf_stat_config *conf
} else if (config->cgroup_list) {
print_cgroup_counter(config, evlist, &os);
} else {
- print_metric_begin(config, evlist, &os, /*aggr_idx=3D*/0);
+ print_metric_begin(config, evlist, &os, /*aggr_idx=3D*/0, NULL);
evlist__for_each_entry(evlist, counter) {
print_counter(config, counter, &os);
}
--
2.55.0.571.g244d577d93-goog