This is an automated email from the ASF dual-hosted git repository.

github-merge-queue[bot] pushed a commit to branch dev
in repository https://gitbox.apache.org/repos/asf/seatunnel.git


The following commit(s) were added to refs/heads/dev by this push:
     new 0301f7cd57 [Improve][Benchmark] Expose JMH CV and error in comparison 
report (#12089)
0301f7cd57 is described below

commit 0301f7cd576f3d4918ef857bf2579ca6ae14ddf9
Author: zhiweiniu <[email protected]>
AuthorDate: Fri Sep 4 15:05:47 2026 +0000

    [Improve][Benchmark] Expose JMH CV and error in comparison report (#12089)
---
 tools/benchmarks/regression_report.py      | 38 +++++++++++++++++++++-----
 tools/benchmarks/test_regression_report.py | 44 ++++++++++++++++++++++++++++++
 2 files changed, 75 insertions(+), 7 deletions(-)

diff --git a/tools/benchmarks/regression_report.py 
b/tools/benchmarks/regression_report.py
index b3c3f0a270..79c1e585de 100644
--- a/tools/benchmarks/regression_report.py
+++ b/tools/benchmarks/regression_report.py
@@ -370,10 +370,22 @@ def median_value(metrics, target_unit=None):
     return statistics.median(values) if values else None
 
 
-def adjusted_change(baseline, candidate, direction):
+def median_statistic(metrics, statistic):
+    values = [statistic(metric) for metric in metrics]
+    values = [value for value in values if value is not None]
+    return statistics.median(values) if values else None
+
+
+def relative_change(baseline, candidate):
     if baseline in (None, 0.0) or candidate is None:
         return None
-    raw = (candidate / baseline - 1.0) * 100.0
+    return (candidate / baseline - 1.0) * 100.0
+
+
+def adjusted_change(baseline, candidate, direction):
+    raw = relative_change(baseline, candidate)
+    if raw is None:
+        return None
     return -raw if direction == "lower" else raw
 
 
@@ -396,8 +408,11 @@ def jmh_comparison_lines(baselines, candidates):
     lines = [
         "### JMH comparison",
         "",
-        "| Benchmark | Parameters | Baseline | Candidate | Change | Unit |",
-        "| --- | --- | ---: | ---: | ---: | --- |",
+        "> `B` = Baseline, `C` = Candidate. Score is the median benchmark 
result; CV measures variability and Error represents the relative JMH 
confidence interval, both aggregated as medians across runs.",
+        "> Score Change is direction-adjusted so positive is favorable; CV and 
Error changes are relative changes.",
+        "",
+        "| Benchmark | Parameters | Score B | Score C | Score Change | CV B | 
CV C | CV Change | Error B | Error C | Error Change | Unit |",
+        "| --- | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | 
---: | --- |",
     ]
     names.sort(
         key=lambda name: jmh_sort_key(
@@ -411,13 +426,23 @@ def jmh_comparison_lines(baselines, candidates):
         target_unit = (candidate_group or baseline_group)[0]["unit"]
         baseline = median_value(baseline_group, target_unit)
         candidate = median_value(candidate_group, target_unit)
+        baseline_cv = median_statistic(baseline_group, 
coefficient_of_variation)
+        candidate_cv = median_statistic(candidate_group, 
coefficient_of_variation)
+        baseline_error = median_statistic(baseline_group, relative_error)
+        candidate_error = median_statistic(candidate_group, relative_error)
         lines.append(
-            "| `{}` | `{}` | {} | {} | {} | {} |".format(
+            "| `{}` | `{}` | {} | {} | {} | {} | {} | {} | {} | {} | {} | {} 
|".format(
                 short_benchmark_name(metric),
                 compact_params(metric.get("params", {})),
                 format_number(baseline),
                 format_number(candidate),
                 format_percent(adjusted_change(baseline, candidate, 
metric["direction"])),
+                format_percent(baseline_cv, signed=False),
+                format_percent(candidate_cv, signed=False),
+                format_percent(relative_change(baseline_cv, candidate_cv)),
+                format_percent(baseline_error, signed=False),
+                format_percent(candidate_error, signed=False),
+                format_percent(relative_change(baseline_error, 
candidate_error)),
                 target_unit,
             )
         )
@@ -556,8 +581,7 @@ def comparison_lines(baselines, candidates):
         "- Runner image: `{}`".format(environment.get("runner_image", 
"unknown")),
         "- CPU: `{}`".format(environment.get("cpu_model", "unknown")),
         "",
-        "> Baseline and candidate ran alternately on the same worker. Positive 
adjusted change is "
-        "favorable, but this observational report does not enforce a 
regression threshold.",
+        "> Baseline and candidate ran alternately on the same worker.",
     ]
     for section in (
         jmh_comparison_lines(baselines, candidates),
diff --git a/tools/benchmarks/test_regression_report.py 
b/tools/benchmarks/test_regression_report.py
index 3cb3686d4b..b4dd5deef7 100644
--- a/tools/benchmarks/test_regression_report.py
+++ b/tools/benchmarks/test_regression_report.py
@@ -55,6 +55,50 @@ class RegressionReportTest(unittest.TestCase):
         self.assertIn("+10.00%", markdown)
         self.assertIn("ops/ms", markdown)
 
+    def test_jmh_comparison_reports_score_cv_error_and_changes(self):
+        baseline_one = self.jmh_metric(100.0, "ops/s")
+        baseline_one.update({"score_error": 10.0, "sample_standard_deviation": 
20.0})
+        baseline_two = self.jmh_metric(100.0, "ops/s")
+        baseline_two.update({"score_error": 10.0, "sample_standard_deviation": 
10.0})
+        candidate_one = self.jmh_metric(110.0, "ops/s")
+        candidate_one.update({"score_error": 22.0, 
"sample_standard_deviation": 11.0})
+        candidate_two = self.jmh_metric(110.0, "ops/s")
+        candidate_two.update({"score_error": 22.0, 
"sample_standard_deviation": 11.0})
+        baselines = [
+            self.report("dev", baseline_one),
+            self.report("dev", baseline_two),
+        ]
+        candidates = [
+            self.report("PR #123", candidate_one),
+            self.report("PR #123", candidate_two),
+        ]
+
+        markdown = "\n".join(
+            regression_report.jmh_comparison_lines(baselines, candidates)
+        )
+
+        self.assertIn(
+            "> `B` = Baseline, `C` = Candidate. Score is the median benchmark 
result; CV measures "
+            "variability and Error represents the relative JMH confidence 
interval, both "
+            "aggregated as medians across runs.",
+            markdown,
+        )
+        self.assertIn(
+            "> Score Change is direction-adjusted so positive is favorable; CV 
and Error changes "
+            "are relative changes.",
+            markdown,
+        )
+        self.assertIn(
+            "| Benchmark | Parameters | Score B | Score C | Score Change | CV 
B | CV C | "
+            "CV Change | Error B | Error C | Error Change | Unit |",
+            markdown,
+        )
+        self.assertIn(
+            "| `Queue.publish` | `capacity=1024` | 100.000 | 110.000 | +10.00% 
| "
+            "15.00% | 10.00% | -33.33% | 10.00% | 20.00% | +100.00% | ops/s |",
+            markdown,
+        )
+
     def test_lower_is_better_change_is_reported_as_positive(self):
         metric = self.jmh_metric(10.0, "ms/op", direction="lower")
         baseline = self.report("dev", metric)

Reply via email to