* [RFC net-next 1/6] selftests: net: py: add timestamped metric output
2026-09-16 19:04 [RFC net-next 0/6] selftests: net: add performance metric reporting Stanislav Fomichev
@ 2026-09-16 19:04 ` Stanislav Fomichev
2026-09-22 16:10 ` Paolo Abeni
2026-09-16 19:04 ` [RFC net-next 2/6] selftests: net: py: add metric policy output Stanislav Fomichev
` (5 subsequent siblings)
6 siblings, 1 reply; 10+ messages in thread
From: Stanislav Fomichev @ 2026-09-16 19:04 UTC (permalink / raw)
To: netdev; +Cc: davem, edumazet, kuba, pabeni
Add ksft_metric() to record timestamped JSON observations for the current
test case.
Signed-off-by: Stanislav Fomichev <sdf@fomichev.me>
---
.../testing/selftests/drivers/net/README.rst | 27 ++++++++++++
.../drivers/net/hw/lib/py/__init__.py | 8 ++--
.../selftests/drivers/net/lib/py/__init__.py | 8 ++--
.../testing/selftests/net/lib/py/__init__.py | 9 ++--
tools/testing/selftests/net/lib/py/ksft.py | 44 ++++++++++++++++++-
5 files changed, 83 insertions(+), 13 deletions(-)
diff --git a/tools/testing/selftests/drivers/net/README.rst b/tools/testing/selftests/drivers/net/README.rst
index 3fe49bce4f3a..a6a8605844eb 100644
--- a/tools/testing/selftests/drivers/net/README.rst
+++ b/tools/testing/selftests/drivers/net/README.rst
@@ -259,6 +259,33 @@ ksft_pr()
Use ``ksft_pr()`` instead of ``print()`` to avoid breaking TAP format.
+ksft_metric()
+~~~~~~~~~~~~~
+
+Use ``ksft_metric()`` to report a numeric measurement for the current test
+case. The helper records the time relative to the start of the case and emits
+each observation immediately before the test result as compact JSON. For
+example::
+
+ ksft_metric("throughput", 100.5, shape="scalar", direction="rx")
+
+emits output similar to::
+
+ # ktap-metric-json: {"direction":"rx","name":"throughput","shape":"scalar","time":1.234,"value":100.5}
+
+When ``run_kselftest.sh`` nests the test output, its diagnostic prefix makes
+this ``# # ktap-metric-json: <json>``. A background sampler can call
+``ksft_metric()`` repeatedly. Pass ``shape`` explicitly to describe the
+layout of each value. For per-CPU data, the array index is the CPU number and
+``None`` represents a missing or offline CPU::
+
+ ksft_metric("cpu.utilization", [75.0, 22.0, None, 100.0],
+ shape="per-cpu", host="local")
+
+The helper does not infer the shape from the Python value, since an array may
+represent something other than per-CPU data. Metrics are measurements only and
+do not affect the pass or fail result.
+
ksft_disruptive
~~~~~~~~~~~~~~~
diff --git a/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py b/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py
index 81e1d1865cd5..17e3daae10e2 100644
--- a/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py
+++ b/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py
@@ -27,8 +27,8 @@ KSFT_DIR = (Path(__file__).parent / "../../../../..").resolve()
wait_file, ctl_file_write, tool
from net.lib.py import bpf_map_set, bpf_map_dump, bpf_prog_map_ids
from net.lib.py import KsftSkipEx, KsftFailEx, KsftXfailEx
- from net.lib.py import ksft_disruptive, ksft_exit, ksft_pr, ksft_run, \
- ksft_setup, ksft_variants, KsftNamedVariant
+ from net.lib.py import ksft_disruptive, ksft_exit, ksft_metric, ksft_pr, \
+ ksft_run, ksft_setup, ksft_variants, KsftNamedVariant
from net.lib.py import ksft_eq, ksft_ge, ksft_in, ksft_is, ksft_lt, \
ksft_ne, ksft_not_in, ksft_raises, ksft_true, ksft_gt, ksft_not_none
from drivers.net.lib.py import GenerateTraffic, Remote, Iperf3Runner
@@ -43,8 +43,8 @@ KSFT_DIR = (Path(__file__).parent / "../../../../..").resolve()
"wait_port_listen", "wait_file", "ctl_file_write", "tool",
"bpf_map_set", "bpf_map_dump", "bpf_prog_map_ids",
"KsftSkipEx", "KsftFailEx", "KsftXfailEx",
- "ksft_disruptive", "ksft_exit", "ksft_pr", "ksft_run",
- "ksft_setup", "ksft_variants", "KsftNamedVariant",
+ "ksft_disruptive", "ksft_exit", "ksft_metric", "ksft_pr",
+ "ksft_run", "ksft_setup", "ksft_variants", "KsftNamedVariant",
"ksft_eq", "ksft_ge", "ksft_in", "ksft_is", "ksft_lt",
"ksft_ne", "ksft_not_in", "ksft_raises", "ksft_true", "ksft_gt",
"ksft_not_none", "ksft_not_none",
diff --git a/tools/testing/selftests/drivers/net/lib/py/__init__.py b/tools/testing/selftests/drivers/net/lib/py/__init__.py
index 591b1e6c7eea..d22f36b189e4 100644
--- a/tools/testing/selftests/drivers/net/lib/py/__init__.py
+++ b/tools/testing/selftests/drivers/net/lib/py/__init__.py
@@ -27,8 +27,8 @@ KSFT_DIR = (Path(__file__).parent / "../../../..").resolve()
wait_file, ctl_file_write
from net.lib.py import bpf_map_set, bpf_map_dump, bpf_prog_map_ids
from net.lib.py import KsftSkipEx, KsftFailEx, KsftXfailEx
- from net.lib.py import ksft_disruptive, ksft_exit, ksft_pr, ksft_run, \
- ksft_setup, ksft_variants, KsftNamedVariant
+ from net.lib.py import ksft_disruptive, ksft_exit, ksft_metric, ksft_pr, \
+ ksft_run, ksft_setup, ksft_variants, KsftNamedVariant
from net.lib.py import ksft_eq, ksft_ge, ksft_in, ksft_is, ksft_lt, \
ksft_ne, ksft_not_in, ksft_raises, ksft_true, ksft_gt, ksft_not_none
@@ -41,8 +41,8 @@ KSFT_DIR = (Path(__file__).parent / "../../../..").resolve()
"wait_port_listen", "wait_file", "ctl_file_write",
"bpf_map_set", "bpf_map_dump", "bpf_prog_map_ids",
"KsftSkipEx", "KsftFailEx", "KsftXfailEx",
- "ksft_disruptive", "ksft_exit", "ksft_pr", "ksft_run",
- "ksft_setup", "ksft_variants", "KsftNamedVariant",
+ "ksft_disruptive", "ksft_exit", "ksft_metric", "ksft_pr",
+ "ksft_run", "ksft_setup", "ksft_variants", "KsftNamedVariant",
"ksft_eq", "ksft_ge", "ksft_in", "ksft_is", "ksft_lt",
"ksft_ne", "ksft_not_in", "ksft_raises", "ksft_true", "ksft_gt",
"ksft_not_none", "ksft_not_none"]
diff --git a/tools/testing/selftests/net/lib/py/__init__.py b/tools/testing/selftests/net/lib/py/__init__.py
index 71df5880b356..3d9e6f6b402c 100644
--- a/tools/testing/selftests/net/lib/py/__init__.py
+++ b/tools/testing/selftests/net/lib/py/__init__.py
@@ -8,8 +8,8 @@ from .consts import KSRC
from .ksft import KsftFailEx, KsftSkipEx, KsftXfailEx, ksft_pr, ksft_eq, \
ksft_ne, ksft_true, ksft_not_none, ksft_in, ksft_not_in, ksft_is, \
ksft_ge, ksft_gt, ksft_lt, ksft_raises, ksft_busy_wait, \
- ktap_result, ksft_disruptive, ksft_setup, ksft_run, ksft_exit, \
- ksft_variants, KsftNamedVariant
+ ktap_result, ksft_disruptive, ksft_metric, ksft_setup, ksft_run, \
+ ksft_exit, ksft_variants, KsftNamedVariant
from .netns import NetNS, NetNSEnter, UserNetNS
from .nsim import NetdevSim, NetdevSimDev
from .utils import CmdExitFailure, fd_read_timeout, cmd, bkg, defer, \
@@ -24,8 +24,9 @@ __all__ = ["KSRC",
"KsftFailEx", "KsftSkipEx", "KsftXfailEx", "ksft_pr", "ksft_eq",
"ksft_ne", "ksft_true", "ksft_not_none", "ksft_in", "ksft_not_in",
"ksft_is", "ksft_ge", "ksft_gt", "ksft_lt", "ksft_raises",
- "ksft_busy_wait", "ktap_result", "ksft_disruptive", "ksft_setup",
- "ksft_run", "ksft_exit", "ksft_variants", "KsftNamedVariant",
+ "ksft_busy_wait", "ktap_result", "ksft_disruptive", "ksft_metric",
+ "ksft_setup", "ksft_run", "ksft_exit", "ksft_variants",
+ "KsftNamedVariant",
"NetNS", "NetNSEnter", "UserNetNS",
"CmdExitFailure", "fd_read_timeout", "cmd", "bkg", "defer",
"bpftool", "ip", "ethtool", "bpftrace", "rand_port", "rand_ports",
diff --git a/tools/testing/selftests/net/lib/py/ksft.py b/tools/testing/selftests/net/lib/py/ksft.py
index 81287c2daff0..fb0865df86c9 100644
--- a/tools/testing/selftests/net/lib/py/ksft.py
+++ b/tools/testing/selftests/net/lib/py/ksft.py
@@ -4,9 +4,11 @@ import fnmatch
import functools
import getopt
import inspect
+import json
import os
import signal
import sys
+import threading
import time
import traceback
from collections import namedtuple
@@ -16,6 +18,9 @@ from . import utils
KSFT_RESULT = None
KSFT_RESULT_ALL = True
KSFT_DISRUPTIVE = True
+KSFT_METRICS = None
+KSFT_METRICS_START = None
+KSFT_METRICS_LOCK = threading.Lock()
class KsftFailEx(Exception):
@@ -93,6 +98,39 @@ KSFT_DISRUPTIVE = True
print(pfx, prefixed, **kwargs)
+def ksft_metric(name, value, *, shape, **labels):
+ """Record a timestamped metric with an explicitly described shape."""
+ with KSFT_METRICS_LOCK:
+ if KSFT_METRICS is None or KSFT_METRICS_START is None:
+ raise RuntimeError("ksft_metric() called outside of a test case")
+
+ metric = {
+ "name": name,
+ "shape": shape,
+ "value": value,
+ "time": round(time.monotonic() - KSFT_METRICS_START, 6),
+ }
+ for key, label_value in sorted(labels.items()):
+ if key in metric:
+ raise ValueError(f"Metric label uses reserved name: {key}")
+ metric[key] = label_value
+ KSFT_METRICS.append(metric)
+
+
+def _ksft_flush_metrics():
+ global KSFT_METRICS, KSFT_METRICS_START
+
+ with KSFT_METRICS_LOCK:
+ metrics = KSFT_METRICS
+ KSFT_METRICS = None
+ KSFT_METRICS_START = None
+
+ for metric in metrics or []:
+ payload = json.dumps(metric, allow_nan=False, separators=(",", ":"),
+ sort_keys=True)
+ ksft_pr(f"ktap-metric-json: {payload}")
+
+
def _fail(*args):
global KSFT_RESULT
KSFT_RESULT = False
@@ -400,7 +438,7 @@ KsftCaseFunction = namedtuple("KsftCaseFunction",
totals = {"pass": 0, "fail": 0, "skip": 0, "xfail": 0}
- global KSFT_RESULT
+ global KSFT_RESULT, KSFT_METRICS, KSFT_METRICS_START
if KSFT_RESULT is not None:
raise RuntimeError("ksft_run() can't be called multiple times.")
@@ -411,6 +449,9 @@ KsftCaseFunction = namedtuple("KsftCaseFunction",
stop = False
for func, args, name in test_cases:
KSFT_RESULT = True
+ with KSFT_METRICS_LOCK:
+ KSFT_METRICS = []
+ KSFT_METRICS_START = time.monotonic()
cnt += 1
comment = ""
cnt_key = ""
@@ -450,6 +491,7 @@ KsftCaseFunction = namedtuple("KsftCaseFunction",
if not cnt_key:
cnt_key = 'pass' if KSFT_RESULT else 'fail'
+ _ksft_flush_metrics()
ktap_result(KSFT_RESULT, cnt, name, comment=comment)
totals[cnt_key] += 1
--
2.53.0-Meta
^ permalink raw reply related [flat|nested] 10+ messages in thread* [RFC net-next 2/6] selftests: net: py: add metric policy output
2026-09-16 19:04 [RFC net-next 0/6] selftests: net: add performance metric reporting Stanislav Fomichev
2026-09-16 19:04 ` [RFC net-next 1/6] selftests: net: py: add timestamped metric output Stanislav Fomichev
@ 2026-09-16 19:04 ` Stanislav Fomichev
2026-09-22 16:16 ` Paolo Abeni
2026-09-16 19:04 ` [RFC net-next 3/6] selftests: drv-net: add a system performance monitor Stanislav Fomichev
` (4 subsequent siblings)
6 siblings, 1 reply; 10+ messages in thread
From: Stanislav Fomichev @ 2026-09-16 19:04 UTC (permalink / raw)
To: netdev; +Cc: davem, edumazet, kuba, pabeni
Add ksft_metric_aggregate() to publish processing and display policy
without tying consumers to metric names. Emit each policy once and reject
conflicting or late registration.
Signed-off-by: Stanislav Fomichev <sdf@fomichev.me>
---
.../testing/selftests/drivers/net/README.rst | 21 ++++
.../drivers/net/hw/lib/py/__init__.py | 11 +-
.../selftests/drivers/net/lib/py/__init__.py | 11 +-
.../testing/selftests/net/lib/py/__init__.py | 6 +-
tools/testing/selftests/net/lib/py/ksft.py | 114 +++++++++++++++++-
5 files changed, 152 insertions(+), 11 deletions(-)
diff --git a/tools/testing/selftests/drivers/net/README.rst b/tools/testing/selftests/drivers/net/README.rst
index a6a8605844eb..03373d6cecb0 100644
--- a/tools/testing/selftests/drivers/net/README.rst
+++ b/tools/testing/selftests/drivers/net/README.rst
@@ -286,6 +286,27 @@ The helper does not infer the shape from the Python value, since an array may
represent something other than per-CPU data. Metrics are measurements only and
do not affect the pass or fail result.
+ksft_metric_aggregate()
+~~~~~~~~~~~~~~~~~~~~~~~
+
+Use ``ksft_metric_aggregate()`` before the first matching observation to
+register processing, display, and regression policy for a metric. The policy
+is emitted once per test case rather than repeated in every observation. For
+example::
+
+ ksft_metric_aggregate("nic.rx.dropped", summarize="distribution",
+ transform="rate", display_range={"min": 0})
+ ksft_metric("nic.rx.dropped", 0, shape="scalar", host="local")
+
+emits separate policy and observation records::
+
+ # ktap-metric-policy-json: {"display_range":{"min":0},"name":"nic.rx.dropped","summarize":"distribution","transform":"rate"}
+ # ktap-metric-json: {"host":"local","name":"nic.rx.dropped","shape":"scalar","time":1.234,"value":0}
+
+Consumers apply a policy to observations with the same metric name. The
+function docstring documents the supported summary, transform, aggregation,
+regression, and display options.
+
ksft_disruptive
~~~~~~~~~~~~~~~
diff --git a/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py b/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py
index 17e3daae10e2..349647c2307d 100644
--- a/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py
+++ b/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py
@@ -27,8 +27,9 @@ KSFT_DIR = (Path(__file__).parent / "../../../../..").resolve()
wait_file, ctl_file_write, tool
from net.lib.py import bpf_map_set, bpf_map_dump, bpf_prog_map_ids
from net.lib.py import KsftSkipEx, KsftFailEx, KsftXfailEx
- from net.lib.py import ksft_disruptive, ksft_exit, ksft_metric, ksft_pr, \
- ksft_run, ksft_setup, ksft_variants, KsftNamedVariant
+ from net.lib.py import ksft_disruptive, ksft_exit, ksft_metric, \
+ ksft_metric_aggregate, ksft_pr, ksft_run, ksft_setup, \
+ ksft_variants, KsftNamedVariant
from net.lib.py import ksft_eq, ksft_ge, ksft_in, ksft_is, ksft_lt, \
ksft_ne, ksft_not_in, ksft_raises, ksft_true, ksft_gt, ksft_not_none
from drivers.net.lib.py import GenerateTraffic, Remote, Iperf3Runner
@@ -43,8 +44,10 @@ KSFT_DIR = (Path(__file__).parent / "../../../../..").resolve()
"wait_port_listen", "wait_file", "ctl_file_write", "tool",
"bpf_map_set", "bpf_map_dump", "bpf_prog_map_ids",
"KsftSkipEx", "KsftFailEx", "KsftXfailEx",
- "ksft_disruptive", "ksft_exit", "ksft_metric", "ksft_pr",
- "ksft_run", "ksft_setup", "ksft_variants", "KsftNamedVariant",
+ "ksft_disruptive", "ksft_exit", "ksft_metric",
+ "ksft_metric_aggregate", "ksft_pr", "ksft_run",
+ "ksft_setup", "ksft_variants",
+ "KsftNamedVariant",
"ksft_eq", "ksft_ge", "ksft_in", "ksft_is", "ksft_lt",
"ksft_ne", "ksft_not_in", "ksft_raises", "ksft_true", "ksft_gt",
"ksft_not_none", "ksft_not_none",
diff --git a/tools/testing/selftests/drivers/net/lib/py/__init__.py b/tools/testing/selftests/drivers/net/lib/py/__init__.py
index d22f36b189e4..afad9d5392ca 100644
--- a/tools/testing/selftests/drivers/net/lib/py/__init__.py
+++ b/tools/testing/selftests/drivers/net/lib/py/__init__.py
@@ -27,8 +27,9 @@ KSFT_DIR = (Path(__file__).parent / "../../../..").resolve()
wait_file, ctl_file_write
from net.lib.py import bpf_map_set, bpf_map_dump, bpf_prog_map_ids
from net.lib.py import KsftSkipEx, KsftFailEx, KsftXfailEx
- from net.lib.py import ksft_disruptive, ksft_exit, ksft_metric, ksft_pr, \
- ksft_run, ksft_setup, ksft_variants, KsftNamedVariant
+ from net.lib.py import ksft_disruptive, ksft_exit, ksft_metric, \
+ ksft_metric_aggregate, ksft_pr, ksft_run, ksft_setup, \
+ ksft_variants, KsftNamedVariant
from net.lib.py import ksft_eq, ksft_ge, ksft_in, ksft_is, ksft_lt, \
ksft_ne, ksft_not_in, ksft_raises, ksft_true, ksft_gt, ksft_not_none
@@ -41,8 +42,10 @@ KSFT_DIR = (Path(__file__).parent / "../../../..").resolve()
"wait_port_listen", "wait_file", "ctl_file_write",
"bpf_map_set", "bpf_map_dump", "bpf_prog_map_ids",
"KsftSkipEx", "KsftFailEx", "KsftXfailEx",
- "ksft_disruptive", "ksft_exit", "ksft_metric", "ksft_pr",
- "ksft_run", "ksft_setup", "ksft_variants", "KsftNamedVariant",
+ "ksft_disruptive", "ksft_exit", "ksft_metric",
+ "ksft_metric_aggregate", "ksft_pr", "ksft_run",
+ "ksft_setup", "ksft_variants",
+ "KsftNamedVariant",
"ksft_eq", "ksft_ge", "ksft_in", "ksft_is", "ksft_lt",
"ksft_ne", "ksft_not_in", "ksft_raises", "ksft_true", "ksft_gt",
"ksft_not_none", "ksft_not_none"]
diff --git a/tools/testing/selftests/net/lib/py/__init__.py b/tools/testing/selftests/net/lib/py/__init__.py
index 3d9e6f6b402c..21d89ae76b49 100644
--- a/tools/testing/selftests/net/lib/py/__init__.py
+++ b/tools/testing/selftests/net/lib/py/__init__.py
@@ -8,7 +8,8 @@ from .consts import KSRC
from .ksft import KsftFailEx, KsftSkipEx, KsftXfailEx, ksft_pr, ksft_eq, \
ksft_ne, ksft_true, ksft_not_none, ksft_in, ksft_not_in, ksft_is, \
ksft_ge, ksft_gt, ksft_lt, ksft_raises, ksft_busy_wait, \
- ktap_result, ksft_disruptive, ksft_metric, ksft_setup, ksft_run, \
+ ktap_result, ksft_disruptive, ksft_metric, ksft_metric_aggregate, \
+ ksft_setup, ksft_run, \
ksft_exit, ksft_variants, KsftNamedVariant
from .netns import NetNS, NetNSEnter, UserNetNS
from .nsim import NetdevSim, NetdevSimDev
@@ -25,7 +26,8 @@ __all__ = ["KSRC",
"ksft_ne", "ksft_true", "ksft_not_none", "ksft_in", "ksft_not_in",
"ksft_is", "ksft_ge", "ksft_gt", "ksft_lt", "ksft_raises",
"ksft_busy_wait", "ktap_result", "ksft_disruptive", "ksft_metric",
- "ksft_setup", "ksft_run", "ksft_exit", "ksft_variants",
+ "ksft_metric_aggregate", "ksft_setup", "ksft_run",
+ "ksft_exit", "ksft_variants",
"KsftNamedVariant",
"NetNS", "NetNSEnter", "UserNetNS",
"CmdExitFailure", "fd_read_timeout", "cmd", "bkg", "defer",
diff --git a/tools/testing/selftests/net/lib/py/ksft.py b/tools/testing/selftests/net/lib/py/ksft.py
index fb0865df86c9..f0cbc7307117 100644
--- a/tools/testing/selftests/net/lib/py/ksft.py
+++ b/tools/testing/selftests/net/lib/py/ksft.py
@@ -20,6 +20,7 @@ KSFT_RESULT_ALL = True
KSFT_DISRUPTIVE = True
KSFT_METRICS = None
KSFT_METRICS_START = None
+KSFT_METRIC_AGGREGATES = None
KSFT_METRICS_LOCK = threading.Lock()
@@ -98,6 +99,105 @@ KSFT_METRICS_LOCK = threading.Lock()
print(pfx, prefixed, **kwargs)
+def ksft_metric_aggregate(name, summarize="total", *, aggregation=None,
+ transform=None, regression=None,
+ display_range=None, display_scale=None,
+ display_label=None):
+ """Register processing and presentation metadata for a metric.
+
+ The policy applies to every observation with ``name`` in the current test
+ case. It is emitted once as a ``ktap-metric-policy-json`` record before
+ the matching ``ktap-metric-json`` observations, so a consumer can process
+ the metric without recognizing its name. Register a policy before the
+ first matching :func:`ksft_metric` call. Registering the same policy more
+ than once is allowed; changing it or registering it after an observation
+ is rejected.
+
+ Args:
+ name: Metric name passed to :func:`ksft_metric`.
+ summarize: Operation used to reduce the processed observations.
+ Supported values are ``last``, ``total``, ``p50``, ``p90``,
+ ``p99``, ``max``, and ``distribution``. ``distribution`` keeps
+ ``last`` for one scalar observation and produces p50, p90, p99,
+ and max for sampled observations. The default is ``total``.
+ aggregation: Optional operation applied to each transformed
+ observation before summarization. An aggregation may reduce a
+ structured value, such as an array, to a scalar.
+ ``busy-core-equivalents`` expects a CPU-indexed array and sums
+ transformed per-CPU rates. Without a transform, it treats values
+ as percentages and divides their sum by 100. The resulting series
+ measures concurrently busy cores.
+ transform: Optional operation applied before aggregation and summary.
+ ``rate`` calculates ``(current - previous) / elapsed_time`` from
+ timestamped cumulative observations. Counter decreases and
+ intervals with nonpositive elapsed time do not produce a rate.
+ regression: Optional regression-tracking policy dictionary:
+
+ ``compare``
+ Summary operation to compare: ``last``, ``total``, ``p50``,
+ ``p90``, ``p99``, or ``max``. This is independent of the
+ summaries selected for display.
+ ``better``
+ ``higher`` or ``lower``, indicating which direction is an
+ improvement.
+ ``relative_tolerance``
+ Optional nonnegative fractional deterioration, for example
+ ``0.05`` for five percent.
+ ``absolute_tolerance``
+ Optional nonnegative deterioration in the processed metric's
+ native units. At least one tolerance is required; consumers
+ use the larger allowance when both are present.
+
+ The producer describes comparison semantics, while the consumer
+ selects historical baselines and determines regression status.
+ display_range: Optional graph range dictionary containing ``min``,
+ ``max``, or both. It describes displayed individual series. An
+ upper bound does not cap a post-aggregation total such as multiple
+ busy-core equivalents.
+ display_scale: Optional positive multiplier applied only when values
+ are displayed, after transformation. It does not alter stored
+ observations, aggregation, summaries, or regression comparison.
+ display_label: Optional nonempty axis label for displayed values.
+
+ Example::
+
+ ksft_metric_aggregate(
+ "cpu.time.usr", summarize="p90", transform="rate",
+ aggregation="busy-core-equivalents",
+ display_range={"min": 0, "max": 100},
+ display_scale=100,
+ display_label="Percent of one CPU")
+
+ Raises:
+ RuntimeError: If called outside a test case or after a matching metric.
+ ValueError: If the policy conflicts with an earlier registration.
+ """
+ metadata = {"summarize": summarize}
+ optional = {
+ "aggregation": aggregation,
+ "transform": transform,
+ "regression": regression,
+ "display_range": display_range,
+ "display_scale": display_scale,
+ "display_label": display_label,
+ }
+ metadata.update({key: value for key, value in optional.items()
+ if value is not None})
+
+ with KSFT_METRICS_LOCK:
+ if KSFT_METRIC_AGGREGATES is None:
+ raise RuntimeError(
+ "ksft_metric_aggregate() called outside of a test case")
+ previous = KSFT_METRIC_AGGREGATES.get(name)
+ if previous is not None and previous != metadata:
+ raise ValueError(f"Conflicting aggregation for metric {name}")
+ if previous is None and any(metric["name"] == name
+ for metric in KSFT_METRICS):
+ raise RuntimeError(
+ f"Aggregation registered after metric {name} was recorded")
+ KSFT_METRIC_AGGREGATES[name] = metadata
+
+
def ksft_metric(name, value, *, shape, **labels):
"""Record a timestamped metric with an explicitly described shape."""
with KSFT_METRICS_LOCK:
@@ -118,13 +218,23 @@ KSFT_METRICS_LOCK = threading.Lock()
def _ksft_flush_metrics():
- global KSFT_METRICS, KSFT_METRICS_START
+ global KSFT_METRICS, KSFT_METRICS_START, KSFT_METRIC_AGGREGATES
with KSFT_METRICS_LOCK:
metrics = KSFT_METRICS
+ aggregates = KSFT_METRIC_AGGREGATES
KSFT_METRICS = None
KSFT_METRICS_START = None
+ KSFT_METRIC_AGGREGATES = None
+ metric_names = {metric["name"] for metric in metrics or []}
+ for name, metadata in (aggregates or {}).items():
+ if name not in metric_names:
+ continue
+ policy = {"name": name, **metadata}
+ payload = json.dumps(policy, allow_nan=False, separators=(",", ":"),
+ sort_keys=True)
+ ksft_pr(f"ktap-metric-policy-json: {payload}")
for metric in metrics or []:
payload = json.dumps(metric, allow_nan=False, separators=(",", ":"),
sort_keys=True)
@@ -439,6 +549,7 @@ KsftCaseFunction = namedtuple("KsftCaseFunction",
totals = {"pass": 0, "fail": 0, "skip": 0, "xfail": 0}
global KSFT_RESULT, KSFT_METRICS, KSFT_METRICS_START
+ global KSFT_METRIC_AGGREGATES
if KSFT_RESULT is not None:
raise RuntimeError("ksft_run() can't be called multiple times.")
@@ -452,6 +563,7 @@ KsftCaseFunction = namedtuple("KsftCaseFunction",
with KSFT_METRICS_LOCK:
KSFT_METRICS = []
KSFT_METRICS_START = time.monotonic()
+ KSFT_METRIC_AGGREGATES = {}
cnt += 1
comment = ""
cnt_key = ""
--
2.53.0-Meta
^ permalink raw reply related [flat|nested] 10+ messages in thread* [RFC net-next 3/6] selftests: drv-net: add a system performance monitor
2026-09-16 19:04 [RFC net-next 0/6] selftests: net: add performance metric reporting Stanislav Fomichev
2026-09-16 19:04 ` [RFC net-next 1/6] selftests: net: py: add timestamped metric output Stanislav Fomichev
2026-09-16 19:04 ` [RFC net-next 2/6] selftests: net: py: add metric policy output Stanislav Fomichev
@ 2026-09-16 19:04 ` Stanislav Fomichev
2026-09-16 19:04 ` [RFC net-next 4/6] selftests: drv-net: add an iperf performance test Stanislav Fomichev
` (3 subsequent siblings)
6 siblings, 0 replies; 10+ messages in thread
From: Stanislav Fomichev @ 2026-09-16 19:04 UTC (permalink / raw)
To: netdev; +Cc: davem, edumazet, kuba, pabeni
Add a context manager which emits per-CPU time, utilization, and NIC drop
counters for both endpoints. Use one persistent remote process for
one-second sampling and enable benchmark metrics while it is active.
Tested on two mlx5 hosts. First local CPU sample:
# # ktap-metric-policy-json: {"aggregation":"busy-core-equivalents","display_label":"Percent of one CPU","display_range":{"max":100.0,"min":0.0},"display_scale":100.0,"name":"cpu.time.usr","summarize":"p90","transform":"rate"}
# # ktap-metric-policy-json: {"aggregation":"busy-core-equivalents","display_label":"Percent of one CPU","display_range":{"max":100.0,"min":0.0},"display_scale":100.0,"name":"cpu.time.sys","summarize":"p90","transform":"rate"}
# # ktap-metric-policy-json: {"aggregation":"busy-core-equivalents","display_label":"Percent of one CPU","display_range":{"max":100.0,"min":0.0},"display_scale":100.0,"name":"cpu.time.sirq","summarize":"p90","transform":"rate"}
# # ktap-metric-policy-json: {"aggregation":"busy-core-equivalents","display_label":"Percent of one CPU","display_range":{"max":100.0,"min":0.0},"name":"cpu.utilization","summarize":"p90"}
# # ktap-metric-json: {"host":"local","name":"cpu.time.usr","shape":"per-cpu","time":1.615669,"value":[0.0,0.0,0.02,0.0,0.0,0.01,0.0,0.0,0.01,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.06,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.02,0.0,0.16,0.0,0.01,0.01,0.0,0.0,0.0,0.01,0.01,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.01,0.0,0.0,0.02,0.01,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.01,0.0,0.0,0.01,0.0,0.03,0.0,0.0,0.0,0.0,0.01,0.02,0.06,0.01,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.02,0.01,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.02,0.0,0.0,0.0,0.0,0.02,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.04,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.02,0.0,0.0,0.0,0.0,0.01]}
# # ktap-metric-json: {"host":"local","name":"cpu.time.sys","shape":"per-cpu","time":1.615747,"value":[0.0,0.0,0.0,0.0,0.0,0.02,0.0,0.0,0.0,0.01,0.01,0.0,0.0,0.01,0.0,0.01,0.01,0.0,0.0,0.04,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.01,0.02,0.0,0.0,0.0,0.01,0.13,0.0,0.02,0.01,0.0,0.02,0.01,0.01,0.02,0.0,0.01,0.02,0.0,0.04,0.0,0.0,0.01,0.01,0.01,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.01,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.01,0.06,0.02,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.01,0.01,0.0,0.0,0.0,0.0,0.0,0.01,0.02,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.01,0.01,0.02,0.0,0.0,0.0,0.02,0.0,0.0,0.0,0.0,0.0,0.02,0.01,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.15,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.01]}
# # ktap-metric-json: {"host":"local","name":"cpu.time.sirq","shape":"per-cpu","time":1.615814,"value":[0.09,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.08,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.01,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0]}
# # ktap-metric-json: {"host":"local","name":"cpu.utilization","shape":"per-cpu","time":1.615881,"value":[9.0909,0.0,2.0202,0.0,0.0,3.0,0.0,0.9901,0.9901,1.0,2.9412,0.0,0.0,0.9901,0.0,1.0,0.9901,0.0,0.0,3.9604,0.0,0.0,0.0,0.0,0.0,0.0,0.9901,0.0,0.0,0.0,0.0,1.0,0.0,6.0,0.0,8.9109,0.0,1.0,2.0,0.9901,0.0,2.0,2.9412,30.0,0.0,3.9604,2.9412,0.0,2.0202,1.0,2.0,2.9703,0.0,1.0101,2.0,0.0,5.0,0.0,0.0,1.0,2.9126,0.9901,0.9901,2.0,0.9901,0.0,0.0,0.9901,0.0,0.0,0.0,0.0,0.0,0.0,0.9901,2.0,0.9901,0.0,0.0,0.9901,0.0,3.9604,0.0,0.0,0.0,0.0,1.0,3.0,12.0,3.0,0.0,0.0,1.0,0.0,0.0,1.9802,0.0,0.0,2.0,0.0,0.9901,0.0,0.9901,0.0,0.0,0.9901,0.0,1.9802,0.9901,0.0,0.0,0.9901,1.0,0.9901,0.9901,0.0,0.0,0.0,0.0,0.9901,0.9901,3.8835,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.9901,0.9901,0.0,0.9901,0.0,0.0,0.0,0.0,0.0,0.0,0.0,1.9608,0.0,0.0,0.0,1.9802,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,1.9802,0.0,0.0,1.0,1.0,2.9412,2.0,0.0,1.0,0.0,2.0202,0.0,0.0,0.0,0.0,0.0,2.9703,0.9901,0.0,1.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.9901,0.0,0.0,0.0,0.0,1.9802,0.0,0.0,0.0,0.9901,0.0,0.0,19.1919,0.0,0.0,0.0,0.0,1.0,0.9901,0.0,0.0,0.0,0.0,0.0,1.0,0.0,0.0,0.0,0.0,0.0,0.0,0.9901,0.9901,1.9802,0.0,0.0,0.0,0.0,1.9802]}
# ok 1 iperf.test_iperf
# # Totals: pass:1 fail:0 xfail:0 xpass:0 skip:0 error:0
Signed-off-by: Stanislav Fomichev <sdf@fomichev.me>
---
.../drivers/net/hw/lib/py/__init__.py | 11 +-
.../selftests/drivers/net/lib/py/__init__.py | 9 +-
.../drivers/net/lib/py/system_monitor.py | 240 ++++++++++++++++++
.../testing/selftests/net/lib/py/__init__.py | 5 +-
tools/testing/selftests/net/lib/py/ksft.py | 14 +-
5 files changed, 269 insertions(+), 10 deletions(-)
create mode 100644 tools/testing/selftests/drivers/net/lib/py/system_monitor.py
diff --git a/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py b/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py
index 349647c2307d..1077930d1677 100644
--- a/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py
+++ b/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py
@@ -28,11 +28,13 @@ KSFT_DIR = (Path(__file__).parent / "../../../../..").resolve()
from net.lib.py import bpf_map_set, bpf_map_dump, bpf_prog_map_ids
from net.lib.py import KsftSkipEx, KsftFailEx, KsftXfailEx
from net.lib.py import ksft_disruptive, ksft_exit, ksft_metric, \
- ksft_metric_aggregate, ksft_pr, ksft_run, ksft_setup, \
+ ksft_metric_aggregate, ksft_metric_reporting, ksft_pr, ksft_run, \
+ ksft_setup, \
ksft_variants, KsftNamedVariant
from net.lib.py import ksft_eq, ksft_ge, ksft_in, ksft_is, ksft_lt, \
ksft_ne, ksft_not_in, ksft_raises, ksft_true, ksft_gt, ksft_not_none
- from drivers.net.lib.py import GenerateTraffic, Remote, Iperf3Runner
+ from drivers.net.lib.py import GenerateTraffic, Remote, SystemMonitor, \
+ Iperf3Runner
from drivers.net.lib.py import NetDrvEnv, NetDrvEpEnv, NetDrvContEnv
__all__ = ["NetNS", "NetNSEnter", "NetdevSimDev", "UserNetNS",
@@ -45,14 +47,15 @@ KSFT_DIR = (Path(__file__).parent / "../../../../..").resolve()
"bpf_map_set", "bpf_map_dump", "bpf_prog_map_ids",
"KsftSkipEx", "KsftFailEx", "KsftXfailEx",
"ksft_disruptive", "ksft_exit", "ksft_metric",
- "ksft_metric_aggregate", "ksft_pr", "ksft_run",
+ "ksft_metric_aggregate", "ksft_metric_reporting", "ksft_pr",
+ "ksft_run",
"ksft_setup", "ksft_variants",
"KsftNamedVariant",
"ksft_eq", "ksft_ge", "ksft_in", "ksft_is", "ksft_lt",
"ksft_ne", "ksft_not_in", "ksft_raises", "ksft_true", "ksft_gt",
"ksft_not_none", "ksft_not_none",
"NetDrvEnv", "NetDrvEpEnv", "NetDrvContEnv", "GenerateTraffic",
- "Remote", "Iperf3Runner"]
+ "Remote", "Iperf3Runner", "SystemMonitor"]
except ModuleNotFoundError as e:
print("Failed importing `net` library from kernel sources")
print(str(e))
diff --git a/tools/testing/selftests/drivers/net/lib/py/__init__.py b/tools/testing/selftests/drivers/net/lib/py/__init__.py
index afad9d5392ca..d779f48ea6c5 100644
--- a/tools/testing/selftests/drivers/net/lib/py/__init__.py
+++ b/tools/testing/selftests/drivers/net/lib/py/__init__.py
@@ -28,7 +28,8 @@ KSFT_DIR = (Path(__file__).parent / "../../../..").resolve()
from net.lib.py import bpf_map_set, bpf_map_dump, bpf_prog_map_ids
from net.lib.py import KsftSkipEx, KsftFailEx, KsftXfailEx
from net.lib.py import ksft_disruptive, ksft_exit, ksft_metric, \
- ksft_metric_aggregate, ksft_pr, ksft_run, ksft_setup, \
+ ksft_metric_aggregate, ksft_metric_reporting, ksft_pr, ksft_run, \
+ ksft_setup, \
ksft_variants, KsftNamedVariant
from net.lib.py import ksft_eq, ksft_ge, ksft_in, ksft_is, ksft_lt, \
ksft_ne, ksft_not_in, ksft_raises, ksft_true, ksft_gt, ksft_not_none
@@ -43,7 +44,8 @@ KSFT_DIR = (Path(__file__).parent / "../../../..").resolve()
"bpf_map_set", "bpf_map_dump", "bpf_prog_map_ids",
"KsftSkipEx", "KsftFailEx", "KsftXfailEx",
"ksft_disruptive", "ksft_exit", "ksft_metric",
- "ksft_metric_aggregate", "ksft_pr", "ksft_run",
+ "ksft_metric_aggregate", "ksft_metric_reporting", "ksft_pr",
+ "ksft_run",
"ksft_setup", "ksft_variants",
"KsftNamedVariant",
"ksft_eq", "ksft_ge", "ksft_in", "ksft_is", "ksft_lt",
@@ -52,10 +54,11 @@ KSFT_DIR = (Path(__file__).parent / "../../../..").resolve()
from .env import NetDrvEnv, NetDrvEpEnv, NetDrvContEnv
from .load import GenerateTraffic, Iperf3Runner
+ from .system_monitor import SystemMonitor
from .remote import Remote
__all__ += ["NetDrvEnv", "NetDrvEpEnv", "NetDrvContEnv", "GenerateTraffic",
- "Remote", "Iperf3Runner"]
+ "Remote", "Iperf3Runner", "SystemMonitor"]
except ModuleNotFoundError as e:
print("Failed importing `net` library from kernel sources")
print(str(e))
diff --git a/tools/testing/selftests/drivers/net/lib/py/system_monitor.py b/tools/testing/selftests/drivers/net/lib/py/system_monitor.py
new file mode 100644
index 000000000000..81ded4bea1b9
--- /dev/null
+++ b/tools/testing/selftests/drivers/net/lib/py/system_monitor.py
@@ -0,0 +1,240 @@
+# SPDX-License-Identifier: GPL-2.0
+
+"""Performance measurement helpers for network driver selftests."""
+
+import shlex
+import subprocess
+import threading
+import time
+
+from lib.py import cmd, ksft_metric, ksft_metric_aggregate, \
+ ksft_metric_reporting, ksft_pr
+
+
+class SystemMonitor:
+ """Emit CPU and NIC metrics for the local and remote test endpoints."""
+
+ _CPU_FIELDS = ("user", "nice", "sys", "idle", "iowait", "irq", "sirq",
+ "steal")
+ _CPU_TIME_METRICS = {
+ "cpu.time.usr": ("user", "nice"),
+ "cpu.time.sys": ("sys",),
+ "cpu.time.sirq": ("sirq",),
+ }
+ _NIC_METRICS = {
+ "rx_dropped": "nic.rx.dropped",
+ "rx_missed_errors": "nic.rx.missed_errors",
+ "tx_dropped": "nic.tx.dropped",
+ }
+ _SNAPSHOT_END = "__KSFT_SYSTEM_STAT_END__"
+ _NIC_STATS = "__KSFT_NIC_STATS__"
+ _CLOCK_TICKS = 100
+ _SAMPLE_INTERVAL = 1
+
+ def __init__(self, env):
+ self.env = env
+ self._remote_sampler = None
+
+ @classmethod
+ def _parse_cpu(cls, text):
+ stats = {}
+ for line in text.splitlines():
+ fields = line.split()
+ if not fields or not fields[0].startswith("cpu") or \
+ not fields[0][3:].isdigit():
+ continue
+ stats[int(fields[0][3:])] = {
+ name: int(value)
+ for name, value in zip(cls._CPU_FIELDS, fields[1:])
+ }
+ return stats
+
+ @classmethod
+ def _parse_nic(cls, text):
+ fields = next(line.split() for line in text.splitlines()
+ if line.startswith(cls._NIC_STATS))
+ return {name: int(value)
+ for name, value in zip(cls._NIC_METRICS, fields[1:])}
+
+ @classmethod
+ def _read_nic(cls, ifname):
+ stats = {}
+ for name in cls._NIC_METRICS:
+ path = f"/sys/class/net/{ifname}/statistics/{name}"
+ with open(path, encoding="utf-8") as stat_file:
+ stats[name] = int(stat_file.read())
+ return stats
+
+ @classmethod
+ def _cpu_deltas(cls, before, after):
+ return {
+ cpu: {
+ field: after[cpu][field] - before[cpu][field]
+ for field in cls._CPU_FIELDS
+ }
+ for cpu in before.keys() & after.keys()
+ }
+
+ @staticmethod
+ def _per_cpu_array(values):
+ if not values:
+ return None
+ result = [None] * (max(values) + 1)
+ for cpu, value in values.items():
+ result[cpu] = value
+ return result
+
+ def _start_remote_sampler(self):
+ nic_reads = ""
+ for name in self._NIC_METRICS:
+ path = shlex.quote(
+ f"/sys/class/net/{self.env.remote_ifname}/statistics/{name}")
+ nic_reads += (f"read value < {path} || exit; "
+ f"printf ' %s' \"$value\"; ")
+ command = (
+ "while :; do cat /proc/stat || exit; "
+ f"printf '{self._NIC_STATS}'; {nic_reads}"
+ f"printf '\\n{self._SNAPSHOT_END}\\n'; sleep 1 || exit; done"
+ )
+ self._remote_sampler = cmd(command, host=self.env.remote,
+ background=True)
+
+ def _read_remote(self):
+ lines = []
+ while True:
+ line = self._remote_sampler.proc.stdout.readline()
+ if not line:
+ status = self._remote_sampler.proc.poll()
+ raise RuntimeError(
+ f"Remote system sampler stopped unexpectedly ({status})")
+ line = line.decode("utf-8", "replace").rstrip("\n")
+ if line == self._SNAPSHOT_END:
+ return "\n".join(lines)
+ lines.append(line)
+
+ def _stop_remote_sampler(self):
+ sampler = self._remote_sampler
+ self._remote_sampler = None
+ if sampler is None:
+ return
+ try:
+ sampler.process(terminate=True, fail=False, timeout=5)
+ except subprocess.TimeoutExpired:
+ sampler.proc.kill()
+ sampler.process(terminate=False, fail=False, timeout=5)
+
+ def _snapshot(self):
+ remote = self._read_remote()
+ with open("/proc/stat", encoding="utf-8") as stat_file:
+ local_cpu = self._parse_cpu(stat_file.read())
+ return {
+ "local": {
+ "cpu": local_cpu,
+ "nic": self._read_nic(self.env.ifname),
+ },
+ "remote": {
+ "cpu": self._parse_cpu(remote),
+ "nic": self._parse_nic(remote),
+ },
+ }
+
+ def _record_sample(self, snapshots):
+ for host, current in snapshots.items():
+ baseline = self._baseline[host]
+ previous = self._previous[host]
+ elapsed = self._cpu_deltas(baseline["cpu"], current["cpu"])
+
+ for metric, fields in self._CPU_TIME_METRICS.items():
+ values = self._per_cpu_array({
+ cpu: sum(delta[field] for field in fields) /
+ self._CLOCK_TICKS
+ for cpu, delta in elapsed.items()
+ })
+ if values is not None:
+ ksft_metric(metric, values, shape="per-cpu", host=host)
+
+ interval = self._cpu_deltas(previous["cpu"], current["cpu"])
+ utilization = {}
+ for cpu, delta in interval.items():
+ total = sum(delta.values())
+ if total > 0:
+ busy = total - delta["idle"] - delta["iowait"]
+ utilization[cpu] = round(100 * busy / total, 4)
+ values = self._per_cpu_array(utilization)
+ if values is not None:
+ ksft_metric("cpu.utilization", values, shape="per-cpu",
+ host=host)
+
+ for counter, metric in self._NIC_METRICS.items():
+ value = current["nic"][counter] - baseline["nic"][counter]
+ ksft_metric(metric, value, shape="scalar", host=host)
+
+ self._previous = snapshots
+
+ def _register_aggregations(self):
+ for metric in self._CPU_TIME_METRICS:
+ ksft_metric_aggregate(
+ metric, summarize="p90", transform="rate",
+ aggregation="busy-core-equivalents",
+ display_range={"min": 0, "max": 100},
+ display_scale=100,
+ display_label="Percent of one CPU")
+ ksft_metric_aggregate(
+ "cpu.utilization", summarize="p90",
+ aggregation="busy-core-equivalents",
+ display_range={"min": 0, "max": 100},
+ display_label="Percent of one CPU")
+ for metric in self._NIC_METRICS.values():
+ ksft_metric_aggregate(
+ metric, summarize="distribution", transform="rate",
+ display_range={"min": 0})
+
+ def _sample_loop(self):
+ next_sample = time.monotonic() + self._SAMPLE_INTERVAL
+ try:
+ while not self._stop_event.wait(
+ max(0, next_sample - time.monotonic())):
+ self._record_sample(self._snapshot())
+ next_sample += self._SAMPLE_INTERVAL
+ while next_sample <= time.monotonic():
+ next_sample += self._SAMPLE_INTERVAL
+ except Exception as error:
+ self._sample_error = error
+
+ def __enter__(self):
+ ksft_metric_reporting(True)
+ try:
+ self._register_aggregations()
+ self._start_remote_sampler()
+ self._baseline = self._snapshot()
+ self._previous = self._baseline
+ for host in self._baseline:
+ for metric in self._NIC_METRICS.values():
+ ksft_metric(metric, 0, shape="scalar", host=host)
+
+ self._stop_event = threading.Event()
+ self._sample_error = None
+ self._thread = threading.Thread(target=self._sample_loop,
+ daemon=True)
+ self._thread.start()
+ return self
+ except Exception:
+ ksft_metric_reporting(False)
+ self._stop_remote_sampler()
+ raise
+
+ def __exit__(self, exc_type, _exc_value, _exc_tb):
+ ksft_metric_reporting(False)
+ try:
+ self._stop_event.set()
+ self._thread.join()
+ if self._sample_error is not None:
+ raise self._sample_error
+ self._record_sample(self._snapshot())
+ except Exception as error:
+ if exc_type is None:
+ raise
+ ksft_pr(f"System metric collection failed: {error}")
+ finally:
+ self._stop_remote_sampler()
+ return False
diff --git a/tools/testing/selftests/net/lib/py/__init__.py b/tools/testing/selftests/net/lib/py/__init__.py
index 21d89ae76b49..9795a83d662a 100644
--- a/tools/testing/selftests/net/lib/py/__init__.py
+++ b/tools/testing/selftests/net/lib/py/__init__.py
@@ -9,7 +9,7 @@ from .ksft import KsftFailEx, KsftSkipEx, KsftXfailEx, ksft_pr, ksft_eq, \
ksft_ne, ksft_true, ksft_not_none, ksft_in, ksft_not_in, ksft_is, \
ksft_ge, ksft_gt, ksft_lt, ksft_raises, ksft_busy_wait, \
ktap_result, ksft_disruptive, ksft_metric, ksft_metric_aggregate, \
- ksft_setup, ksft_run, \
+ ksft_metric_reporting, ksft_setup, ksft_run, \
ksft_exit, ksft_variants, KsftNamedVariant
from .netns import NetNS, NetNSEnter, UserNetNS
from .nsim import NetdevSim, NetdevSimDev
@@ -26,7 +26,8 @@ __all__ = ["KSRC",
"ksft_ne", "ksft_true", "ksft_not_none", "ksft_in", "ksft_not_in",
"ksft_is", "ksft_ge", "ksft_gt", "ksft_lt", "ksft_raises",
"ksft_busy_wait", "ktap_result", "ksft_disruptive", "ksft_metric",
- "ksft_metric_aggregate", "ksft_setup", "ksft_run",
+ "ksft_metric_aggregate", "ksft_metric_reporting", "ksft_setup",
+ "ksft_run",
"ksft_exit", "ksft_variants",
"KsftNamedVariant",
"NetNS", "NetNSEnter", "UserNetNS",
diff --git a/tools/testing/selftests/net/lib/py/ksft.py b/tools/testing/selftests/net/lib/py/ksft.py
index f0cbc7307117..e44029213e4f 100644
--- a/tools/testing/selftests/net/lib/py/ksft.py
+++ b/tools/testing/selftests/net/lib/py/ksft.py
@@ -21,6 +21,7 @@ KSFT_DISRUPTIVE = True
KSFT_METRICS = None
KSFT_METRICS_START = None
KSFT_METRIC_AGGREGATES = None
+KSFT_METRIC_REPORTING = 0
KSFT_METRICS_LOCK = threading.Lock()
@@ -99,6 +100,14 @@ KSFT_METRICS_LOCK = threading.Lock()
print(pfx, prefixed, **kwargs)
+def ksft_metric_reporting(enable):
+ """Enter or leave a scope which enables benchmark metric producers."""
+ global KSFT_METRIC_REPORTING
+
+ with KSFT_METRICS_LOCK:
+ KSFT_METRIC_REPORTING += 1 if enable else -1
+
+
def ksft_metric_aggregate(name, summarize="total", *, aggregation=None,
transform=None, regression=None,
display_range=None, display_scale=None,
@@ -219,6 +228,7 @@ def ksft_metric_aggregate(name, summarize="total", *, aggregation=None,
def _ksft_flush_metrics():
global KSFT_METRICS, KSFT_METRICS_START, KSFT_METRIC_AGGREGATES
+ global KSFT_METRIC_REPORTING
with KSFT_METRICS_LOCK:
metrics = KSFT_METRICS
@@ -226,6 +236,7 @@ def ksft_metric_aggregate(name, summarize="total", *, aggregation=None,
KSFT_METRICS = None
KSFT_METRICS_START = None
KSFT_METRIC_AGGREGATES = None
+ KSFT_METRIC_REPORTING = 0
metric_names = {metric["name"] for metric in metrics or []}
for name, metadata in (aggregates or {}).items():
@@ -549,7 +560,7 @@ KsftCaseFunction = namedtuple("KsftCaseFunction",
totals = {"pass": 0, "fail": 0, "skip": 0, "xfail": 0}
global KSFT_RESULT, KSFT_METRICS, KSFT_METRICS_START
- global KSFT_METRIC_AGGREGATES
+ global KSFT_METRIC_AGGREGATES, KSFT_METRIC_REPORTING
if KSFT_RESULT is not None:
raise RuntimeError("ksft_run() can't be called multiple times.")
@@ -564,6 +575,7 @@ KsftCaseFunction = namedtuple("KsftCaseFunction",
KSFT_METRICS = []
KSFT_METRICS_START = time.monotonic()
KSFT_METRIC_AGGREGATES = {}
+ KSFT_METRIC_REPORTING = 0
cnt += 1
comment = ""
cnt_key = ""
--
2.53.0-Meta
^ permalink raw reply related [flat|nested] 10+ messages in thread* [RFC net-next 4/6] selftests: drv-net: add an iperf performance test
2026-09-16 19:04 [RFC net-next 0/6] selftests: net: add performance metric reporting Stanislav Fomichev
` (2 preceding siblings ...)
2026-09-16 19:04 ` [RFC net-next 3/6] selftests: drv-net: add a system performance monitor Stanislav Fomichev
@ 2026-09-16 19:04 ` Stanislav Fomichev
2026-09-16 19:04 ` [RFC net-next 5/6] selftests: drv-net: add a kperf runner Stanislav Fomichev
` (2 subsequent siblings)
6 siblings, 0 replies; 10+ messages in thread
From: Stanislav Fomichev @ 2026-09-16 19:04 UTC (permalink / raw)
To: netdev; +Cc: davem, edumazet, kuba, pabeni
Report Iperf3Runner throughput and retransmits when metric collection is
active.
Tested on two mlx5 hosts. Benchmark metric output:
# # ktap-metric-policy-json: {"name":"throughput","regression":{"better":"higher","compare":"p50","relative_tolerance":0.05},"summarize":"distribution"}
# # ktap-metric-policy-json: {"name":"tcp.retransmits","summarize":"max"}
# # ktap-metric-json: {"host":"local","name":"throughput","shape":"scalar","time":11.42686,"value":32.62285180182741}
# # ktap-metric-json: {"host":"remote","name":"tcp.retransmits","shape":"scalar","time":11.426883,"value":75895.0}
# ok 1 iperf.test_iperf
# # Totals: pass:1 fail:0 xfail:0 xpass:0 skip:0 error:0
Signed-off-by: Stanislav Fomichev <sdf@fomichev.me>
---
.../testing/selftests/drivers/net/hw/Makefile | 1 +
.../testing/selftests/drivers/net/hw/iperf.py | 21 ++++++++++++++++
.../drivers/net/hw/lib/py/__init__.py | 8 +++---
.../selftests/drivers/net/lib/py/__init__.py | 8 +++---
.../selftests/drivers/net/lib/py/load.py | 25 +++++++++++++++++--
.../testing/selftests/net/lib/py/__init__.py | 6 ++---
tools/testing/selftests/net/lib/py/ksft.py | 6 +++++
7 files changed, 62 insertions(+), 13 deletions(-)
create mode 100755 tools/testing/selftests/drivers/net/hw/iperf.py
diff --git a/tools/testing/selftests/drivers/net/hw/Makefile b/tools/testing/selftests/drivers/net/hw/Makefile
index 8aebdc6feb17..03ecc81e18e0 100644
--- a/tools/testing/selftests/drivers/net/hw/Makefile
+++ b/tools/testing/selftests/drivers/net/hw/Makefile
@@ -32,6 +32,7 @@ TEST_PROGS = \
hw_stats_l3.sh \
hw_stats_l3_gre.sh \
iou-zcrx.py \
+ iperf.py \
ipsec_vxlan.py \
irq.py \
loopback.sh \
diff --git a/tools/testing/selftests/drivers/net/hw/iperf.py b/tools/testing/selftests/drivers/net/hw/iperf.py
new file mode 100755
index 000000000000..77e315c9547e
--- /dev/null
+++ b/tools/testing/selftests/drivers/net/hw/iperf.py
@@ -0,0 +1,21 @@
+#!/usr/bin/env python3
+# SPDX-License-Identifier: GPL-2.0
+
+from lib.py import (Iperf3Runner, NetDrvEpEnv, SystemMonitor, ksft_exit,
+ ksft_run)
+
+
+def test_iperf(cfg):
+ """Measure TCP throughput while sampling system utilization."""
+ with SystemMonitor(cfg):
+ Iperf3Runner(cfg).measure_bandwidth()
+
+
+def main():
+ with NetDrvEpEnv(__file__, nsim_test=False) as cfg:
+ ksft_run([test_iperf], args=(cfg,))
+ ksft_exit()
+
+
+if __name__ == "__main__":
+ main()
diff --git a/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py b/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py
index 1077930d1677..857bd016e5e0 100644
--- a/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py
+++ b/tools/testing/selftests/drivers/net/hw/lib/py/__init__.py
@@ -28,8 +28,8 @@ KSFT_DIR = (Path(__file__).parent / "../../../../..").resolve()
from net.lib.py import bpf_map_set, bpf_map_dump, bpf_prog_map_ids
from net.lib.py import KsftSkipEx, KsftFailEx, KsftXfailEx
from net.lib.py import ksft_disruptive, ksft_exit, ksft_metric, \
- ksft_metric_aggregate, ksft_metric_reporting, ksft_pr, ksft_run, \
- ksft_setup, \
+ ksft_metric_aggregate, ksft_metric_reporting, \
+ ksft_metric_reporting_enabled, ksft_pr, ksft_run, ksft_setup, \
ksft_variants, KsftNamedVariant
from net.lib.py import ksft_eq, ksft_ge, ksft_in, ksft_is, ksft_lt, \
ksft_ne, ksft_not_in, ksft_raises, ksft_true, ksft_gt, ksft_not_none
@@ -47,8 +47,8 @@ KSFT_DIR = (Path(__file__).parent / "../../../../..").resolve()
"bpf_map_set", "bpf_map_dump", "bpf_prog_map_ids",
"KsftSkipEx", "KsftFailEx", "KsftXfailEx",
"ksft_disruptive", "ksft_exit", "ksft_metric",
- "ksft_metric_aggregate", "ksft_metric_reporting", "ksft_pr",
- "ksft_run",
+ "ksft_metric_aggregate", "ksft_metric_reporting",
+ "ksft_metric_reporting_enabled", "ksft_pr", "ksft_run",
"ksft_setup", "ksft_variants",
"KsftNamedVariant",
"ksft_eq", "ksft_ge", "ksft_in", "ksft_is", "ksft_lt",
diff --git a/tools/testing/selftests/drivers/net/lib/py/__init__.py b/tools/testing/selftests/drivers/net/lib/py/__init__.py
index d779f48ea6c5..e418837f0b00 100644
--- a/tools/testing/selftests/drivers/net/lib/py/__init__.py
+++ b/tools/testing/selftests/drivers/net/lib/py/__init__.py
@@ -28,8 +28,8 @@ KSFT_DIR = (Path(__file__).parent / "../../../..").resolve()
from net.lib.py import bpf_map_set, bpf_map_dump, bpf_prog_map_ids
from net.lib.py import KsftSkipEx, KsftFailEx, KsftXfailEx
from net.lib.py import ksft_disruptive, ksft_exit, ksft_metric, \
- ksft_metric_aggregate, ksft_metric_reporting, ksft_pr, ksft_run, \
- ksft_setup, \
+ ksft_metric_aggregate, ksft_metric_reporting, \
+ ksft_metric_reporting_enabled, ksft_pr, ksft_run, ksft_setup, \
ksft_variants, KsftNamedVariant
from net.lib.py import ksft_eq, ksft_ge, ksft_in, ksft_is, ksft_lt, \
ksft_ne, ksft_not_in, ksft_raises, ksft_true, ksft_gt, ksft_not_none
@@ -44,8 +44,8 @@ KSFT_DIR = (Path(__file__).parent / "../../../..").resolve()
"bpf_map_set", "bpf_map_dump", "bpf_prog_map_ids",
"KsftSkipEx", "KsftFailEx", "KsftXfailEx",
"ksft_disruptive", "ksft_exit", "ksft_metric",
- "ksft_metric_aggregate", "ksft_metric_reporting", "ksft_pr",
- "ksft_run",
+ "ksft_metric_aggregate", "ksft_metric_reporting",
+ "ksft_metric_reporting_enabled", "ksft_pr", "ksft_run",
"ksft_setup", "ksft_variants",
"KsftNamedVariant",
"ksft_eq", "ksft_ge", "ksft_in", "ksft_is", "ksft_lt",
diff --git a/tools/testing/selftests/drivers/net/lib/py/load.py b/tools/testing/selftests/drivers/net/lib/py/load.py
index e24660e5c27f..f5f5cf780fbe 100644
--- a/tools/testing/selftests/drivers/net/lib/py/load.py
+++ b/tools/testing/selftests/drivers/net/lib/py/load.py
@@ -4,7 +4,8 @@ import re
import time
import json
-from lib.py import ksft_pr, cmd, ip, rand_port, wait_port_listen
+from lib.py import cmd, ip, ksft_metric, ksft_metric_aggregate, \
+ ksft_metric_reporting_enabled, ksft_pr, rand_port, wait_port_listen
class Iperf3Runner:
@@ -57,8 +58,17 @@ from lib.py import ksft_pr, cmd, ip, rand_port, wait_port_listen
"""
Runs an iperf3 measurement and returns the average bandwidth (Gbps).
Discards the first and last few reporting intervals and uses only the
- middle part of the run where throughput is typically stable.
+ middle part of the run where throughput is typically stable. When a
+ SystemMonitor is active, also emit throughput and retransmit metrics.
"""
+ report_metrics = ksft_metric_reporting_enabled()
+ if report_metrics:
+ ksft_metric_aggregate(
+ "throughput", summarize="distribution",
+ regression={"compare": "p50", "better": "higher",
+ "relative_tolerance": 0.05})
+ ksft_metric_aggregate("tcp.retransmits", summarize="max")
+
self.start_server()
result = self.start_client(duration=10, reverse=reverse)
@@ -77,6 +87,17 @@ from lib.py import ksft_pr, cmd, ip, rand_port, wait_port_listen
stable = samples[3:-3]
avg = sum(stable) / len(stable)
+ retransmits = out.get("end", {}).get("sum_sent", {}).get(
+ "retransmits")
+ if report_metrics:
+ if not isinstance(retransmits, int) or \
+ isinstance(retransmits, bool):
+ raise ValueError("iperf3 did not report TCP retransmits")
+ receiver = "remote" if reverse else "local"
+ sender = "local" if reverse else "remote"
+ ksft_metric("throughput", avg, shape="scalar", host=receiver)
+ ksft_metric("tcp.retransmits", retransmits, shape="scalar",
+ host=sender)
return avg
diff --git a/tools/testing/selftests/net/lib/py/__init__.py b/tools/testing/selftests/net/lib/py/__init__.py
index 9795a83d662a..5b91de1650c5 100644
--- a/tools/testing/selftests/net/lib/py/__init__.py
+++ b/tools/testing/selftests/net/lib/py/__init__.py
@@ -9,7 +9,7 @@ from .ksft import KsftFailEx, KsftSkipEx, KsftXfailEx, ksft_pr, ksft_eq, \
ksft_ne, ksft_true, ksft_not_none, ksft_in, ksft_not_in, ksft_is, \
ksft_ge, ksft_gt, ksft_lt, ksft_raises, ksft_busy_wait, \
ktap_result, ksft_disruptive, ksft_metric, ksft_metric_aggregate, \
- ksft_metric_reporting, ksft_setup, ksft_run, \
+ ksft_metric_reporting, ksft_metric_reporting_enabled, ksft_setup, ksft_run, \
ksft_exit, ksft_variants, KsftNamedVariant
from .netns import NetNS, NetNSEnter, UserNetNS
from .nsim import NetdevSim, NetdevSimDev
@@ -26,8 +26,8 @@ __all__ = ["KSRC",
"ksft_ne", "ksft_true", "ksft_not_none", "ksft_in", "ksft_not_in",
"ksft_is", "ksft_ge", "ksft_gt", "ksft_lt", "ksft_raises",
"ksft_busy_wait", "ktap_result", "ksft_disruptive", "ksft_metric",
- "ksft_metric_aggregate", "ksft_metric_reporting", "ksft_setup",
- "ksft_run",
+ "ksft_metric_aggregate", "ksft_metric_reporting",
+ "ksft_metric_reporting_enabled", "ksft_setup", "ksft_run",
"ksft_exit", "ksft_variants",
"KsftNamedVariant",
"NetNS", "NetNSEnter", "UserNetNS",
diff --git a/tools/testing/selftests/net/lib/py/ksft.py b/tools/testing/selftests/net/lib/py/ksft.py
index e44029213e4f..566ec52cd9c9 100644
--- a/tools/testing/selftests/net/lib/py/ksft.py
+++ b/tools/testing/selftests/net/lib/py/ksft.py
@@ -108,6 +108,12 @@ KSFT_METRICS_LOCK = threading.Lock()
KSFT_METRIC_REPORTING += 1 if enable else -1
+def ksft_metric_reporting_enabled():
+ """Return whether benchmark metric producers should emit metrics."""
+ with KSFT_METRICS_LOCK:
+ return KSFT_METRIC_REPORTING > 0
+
+
def ksft_metric_aggregate(name, summarize="total", *, aggregation=None,
transform=None, regression=None,
display_range=None, display_scale=None,
--
2.53.0-Meta
^ permalink raw reply related [flat|nested] 10+ messages in thread