This is an automated email from the ASF dual-hosted git repository.
github-merge-queue[bot] pushed a commit to branch dev
in repository https://gitbox.apache.org/repos/asf/seatunnel.git
The following commit(s) were added to refs/heads/dev by this push:
new 0301f7cd57 [Improve][Benchmark] Expose JMH CV and error in comparison
report (#12089)
0301f7cd57 is described below
commit 0301f7cd576f3d4918ef857bf2579ca6ae14ddf9
Author: zhiweiniu <[email protected]>
AuthorDate: Fri Sep 4 15:05:47 2026 +0000
[Improve][Benchmark] Expose JMH CV and error in comparison report (#12089)
---
tools/benchmarks/regression_report.py | 38 +++++++++++++++++++++-----
tools/benchmarks/test_regression_report.py | 44 ++++++++++++++++++++++++++++++
2 files changed, 75 insertions(+), 7 deletions(-)
diff --git a/tools/benchmarks/regression_report.py
b/tools/benchmarks/regression_report.py
index b3c3f0a270..79c1e585de 100644
--- a/tools/benchmarks/regression_report.py
+++ b/tools/benchmarks/regression_report.py
@@ -370,10 +370,22 @@ def median_value(metrics, target_unit=None):
return statistics.median(values) if values else None
-def adjusted_change(baseline, candidate, direction):
+def median_statistic(metrics, statistic):
+ values = [statistic(metric) for metric in metrics]
+ values = [value for value in values if value is not None]
+ return statistics.median(values) if values else None
+
+
+def relative_change(baseline, candidate):
if baseline in (None, 0.0) or candidate is None:
return None
- raw = (candidate / baseline - 1.0) * 100.0
+ return (candidate / baseline - 1.0) * 100.0
+
+
+def adjusted_change(baseline, candidate, direction):
+ raw = relative_change(baseline, candidate)
+ if raw is None:
+ return None
return -raw if direction == "lower" else raw
@@ -396,8 +408,11 @@ def jmh_comparison_lines(baselines, candidates):
lines = [
"### JMH comparison",
"",
- "| Benchmark | Parameters | Baseline | Candidate | Change | Unit |",
- "| --- | --- | ---: | ---: | ---: | --- |",
+ "> `B` = Baseline, `C` = Candidate. Score is the median benchmark
result; CV measures variability and Error represents the relative JMH
confidence interval, both aggregated as medians across runs.",
+ "> Score Change is direction-adjusted so positive is favorable; CV and
Error changes are relative changes.",
+ "",
+ "| Benchmark | Parameters | Score B | Score C | Score Change | CV B |
CV C | CV Change | Error B | Error C | Error Change | Unit |",
+ "| --- | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |
---: | --- |",
]
names.sort(
key=lambda name: jmh_sort_key(
@@ -411,13 +426,23 @@ def jmh_comparison_lines(baselines, candidates):
target_unit = (candidate_group or baseline_group)[0]["unit"]
baseline = median_value(baseline_group, target_unit)
candidate = median_value(candidate_group, target_unit)
+ baseline_cv = median_statistic(baseline_group,
coefficient_of_variation)
+ candidate_cv = median_statistic(candidate_group,
coefficient_of_variation)
+ baseline_error = median_statistic(baseline_group, relative_error)
+ candidate_error = median_statistic(candidate_group, relative_error)
lines.append(
- "| `{}` | `{}` | {} | {} | {} | {} |".format(
+ "| `{}` | `{}` | {} | {} | {} | {} | {} | {} | {} | {} | {} | {}
|".format(
short_benchmark_name(metric),
compact_params(metric.get("params", {})),
format_number(baseline),
format_number(candidate),
format_percent(adjusted_change(baseline, candidate,
metric["direction"])),
+ format_percent(baseline_cv, signed=False),
+ format_percent(candidate_cv, signed=False),
+ format_percent(relative_change(baseline_cv, candidate_cv)),
+ format_percent(baseline_error, signed=False),
+ format_percent(candidate_error, signed=False),
+ format_percent(relative_change(baseline_error,
candidate_error)),
target_unit,
)
)
@@ -556,8 +581,7 @@ def comparison_lines(baselines, candidates):
"- Runner image: `{}`".format(environment.get("runner_image",
"unknown")),
"- CPU: `{}`".format(environment.get("cpu_model", "unknown")),
"",
- "> Baseline and candidate ran alternately on the same worker. Positive
adjusted change is "
- "favorable, but this observational report does not enforce a
regression threshold.",
+ "> Baseline and candidate ran alternately on the same worker.",
]
for section in (
jmh_comparison_lines(baselines, candidates),
diff --git a/tools/benchmarks/test_regression_report.py
b/tools/benchmarks/test_regression_report.py
index 3cb3686d4b..b4dd5deef7 100644
--- a/tools/benchmarks/test_regression_report.py
+++ b/tools/benchmarks/test_regression_report.py
@@ -55,6 +55,50 @@ class RegressionReportTest(unittest.TestCase):
self.assertIn("+10.00%", markdown)
self.assertIn("ops/ms", markdown)
+ def test_jmh_comparison_reports_score_cv_error_and_changes(self):
+ baseline_one = self.jmh_metric(100.0, "ops/s")
+ baseline_one.update({"score_error": 10.0, "sample_standard_deviation":
20.0})
+ baseline_two = self.jmh_metric(100.0, "ops/s")
+ baseline_two.update({"score_error": 10.0, "sample_standard_deviation":
10.0})
+ candidate_one = self.jmh_metric(110.0, "ops/s")
+ candidate_one.update({"score_error": 22.0,
"sample_standard_deviation": 11.0})
+ candidate_two = self.jmh_metric(110.0, "ops/s")
+ candidate_two.update({"score_error": 22.0,
"sample_standard_deviation": 11.0})
+ baselines = [
+ self.report("dev", baseline_one),
+ self.report("dev", baseline_two),
+ ]
+ candidates = [
+ self.report("PR #123", candidate_one),
+ self.report("PR #123", candidate_two),
+ ]
+
+ markdown = "\n".join(
+ regression_report.jmh_comparison_lines(baselines, candidates)
+ )
+
+ self.assertIn(
+ "> `B` = Baseline, `C` = Candidate. Score is the median benchmark
result; CV measures "
+ "variability and Error represents the relative JMH confidence
interval, both "
+ "aggregated as medians across runs.",
+ markdown,
+ )
+ self.assertIn(
+ "> Score Change is direction-adjusted so positive is favorable; CV
and Error changes "
+ "are relative changes.",
+ markdown,
+ )
+ self.assertIn(
+ "| Benchmark | Parameters | Score B | Score C | Score Change | CV
B | CV C | "
+ "CV Change | Error B | Error C | Error Change | Unit |",
+ markdown,
+ )
+ self.assertIn(
+ "| `Queue.publish` | `capacity=1024` | 100.000 | 110.000 | +10.00%
| "
+ "15.00% | 10.00% | -33.33% | 10.00% | 20.00% | +100.00% | ops/s |",
+ markdown,
+ )
+
def test_lower_is_better_change_is_reported_as_positive(self):
metric = self.jmh_metric(10.0, "ms/op", direction="lower")
baseline = self.report("dev", metric)