This is an automated email from the ASF dual-hosted git repository.

davsclaus pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/camel-performance-tests.git

commit 06327ef8c7f5b1d3f14b146d04c425d323b99f59
Author: Claus Ibsen <[email protected]>
AuthorDate: Tue Oct 6 10:17:33 2026 +0200

    A third score: the last version right, without the errors of attempts the 
model fixed
    
    ok counts every error a step logged, also those of an attempt the model 
fixed later, or of test data it wrote itself
    an earlier step; ok_final ignores errors altogether. ok_lastwrite is in 
between: the final state is right and no
    errors were logged after the model's last accepted write. The harness 
snapshots the errors at each accepted write;
    the summary shows the new column, and counts runs from before it as ok.
    
    Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
    Claude-Session: https://claude.ai/code/session_01STT6whBgK1AqsSsUKrnE8m
---
 ai-benchmark/agent_mcp_stepwise.py | 20 ++++++++++++++++++++
 ai-benchmark/summarize_stepwise.py | 16 ++++++++++------
 2 files changed, 30 insertions(+), 6 deletions(-)

diff --git a/ai-benchmark/agent_mcp_stepwise.py 
b/ai-benchmark/agent_mcp_stepwise.py
index c185c0d..a821f48 100755
--- a/ai-benchmark/agent_mcp_stepwise.py
+++ b/ai-benchmark/agent_mcp_stepwise.py
@@ -340,6 +340,21 @@ def score(step, project, cfg, mcp, name, before):
     # ok_final: the end state is right (files and log), whatever happened on 
the way; ok also needs no errors
     result["ok_final"] = result["file_ok"] and result["props_ok"] and 
result["log_ok"]
     result["ok"] = result["ok_final"] and errors_fine
+    # ok_lastwrite: the end state is right and the model's last version logged 
no errors; the errors of attempts it
+    # fixed on the way do not count (they do in ok). Without a write in the 
step it is the same as ok
+    last = before.get("last_write")
+    if last is not None:
+        after = sum(1 for l in lines if isinstance(l, dict) and lvl(l) == 
"ERROR" and error_key(l) not in last[0]
+                    and error_key(l) not in seen)
+        if isinstance(data, dict):
+            after += max(0, len(data.get("errors", []) or []) - max(last[1], 
before.get("error_count", 0)))
+    else:
+        after = result["errors"]
+    result["errors_after_last_write"] = after
+    lastwrite_fine = after == 0 or bool(chk.get("errors_ok"))
+    if chk.get("min_errors") and result["errors"] < chk["min_errors"]:
+        lastwrite_fine = False
+    result["ok_lastwrite"] = result["ok_final"] and lastwrite_fine
     result["log_sample"] = msgs[:5]
     return result, route, props
 
@@ -425,6 +440,9 @@ def main():
             trace = open(os.path.join(OUT, f"step{sid}.trace.jsonl"), "w")
             messages.append({"role": "user", "content": step["request"]})
             calls = 0; tokens = 0; t0 = time.time(); writes = 0; refused = 0; 
answer = ""
+            # the errors as they stood after the model's last accepted write: 
what came after is what its final
+            # version did; what came before were attempts it fixed 
(ok_lastwrite)
+            before["last_write"] = None
             if REFERENCE:
                 # the reference files stand in for the model's edits; the 
checks then score the steps file itself
                 writes += apply_reference(project, cfg, step.get("reference") 
or {}, mcp, name, before)
@@ -466,6 +484,8 @@ def main():
                         if fn["name"] in ("camel_write_file", 
"camel_edit_file") and ('"invalid"' in out
                                 or '"not-found"' in out or '"ambiguous"' in 
out or out.startswith("ERROR")):
                             refused += 1
+                        elif fn["name"] in ("camel_write_file", 
"camel_edit_file"):
+                            before["last_write"] = error_snapshot(mcp, name)
                         out = out[:TOOL_RESULT_CAP]
                         trace.write(json.dumps({"tool": fn["name"], "args": 
{k: (v if k != "content" else v[:1500]) for k, v in args.items()}, "result": 
out[:800]}) + "\n"); trace.flush()
                         messages.append({"role": "tool", "content": out, 
"tool_name": fn["name"]})
diff --git a/ai-benchmark/summarize_stepwise.py 
b/ai-benchmark/summarize_stepwise.py
index 8173c1b..0ba1a72 100644
--- a/ai-benchmark/summarize_stepwise.py
+++ b/ai-benchmark/summarize_stepwise.py
@@ -3,14 +3,15 @@
 
 A step passes (ok) when its files, log and probes are right and it logged no 
new errors; final state right (ok_final)
 leaves the errors out: a model that tests its own app (the HTTP request tool) 
causes errors on the way to a right
-answer, which ok counts against it."""
+answer, which ok counts against it. Last version right (ok_lastwrite) is in 
between: the final state is right and no
+errors were logged after the model's last accepted write; runs before it was 
recorded count it as ok."""
 import json, os, sys
 tag = sys.argv[1]; k = int(sys.argv[2]) if len(sys.argv) > 2 else 1
 tags = [f"{tag}-{i}" for i in range(1, k + 1)] if k > 1 else [tag]
 names = sorted(os.path.splitext(f)[0] for f in os.listdir("steps-ladder") if 
f.endswith(".json"))
-print(f"| example | steps | passes per step ({' '.join(tags)}) | all steps 
passed | final state right |")
-print("|---|---|---|---|---|")
-total = 0; possible = 0; clean = 0; final = 0
+print(f"| example | steps | passes per step ({' '.join(tags)}) | all steps 
passed | last version right | final state right |")
+print("|---|---|---|---|---|---|")
+total = 0; possible = 0; clean = 0; final = 0; lastw = 0
 for n in names:
     per = []; runs = []
     for t in tags:
@@ -26,5 +27,8 @@ for n in names:
         ok = sum(1 for r in runs if s < len(r) and r[s]["ok"]); 
cols.append(f"{ok}/{len(runs)}"); total += ok; possible += len(runs)
     allok = sum(1 for r in runs if r and all(x["ok"] for x in r)); clean += 
allok
     fin = sum(1 for r in runs for x in r if x.get("ok_final", x["ok"])); final 
+= fin
-    print(f"| {n} | {steps} | {' '.join(cols)} | {allok}/{len(runs)} | 
{fin}/{sum(len(r) for r in runs)} |")
-print(f"\nsteps passed: {total} of {possible}; final state right: {final} of 
{possible}; runs with every step passed: {clean}")
+    lw = sum(1 for r in runs for x in r if x.get("ok_lastwrite", x["ok"])); 
lastw += lw
+    n_steps = sum(len(r) for r in runs)
+    print(f"| {n} | {steps} | {' '.join(cols)} | {allok}/{len(runs)} | 
{lw}/{n_steps} | {fin}/{n_steps} |")
+print(f"\nsteps passed: {total} of {possible}; last version right: {lastw} of 
{possible}; final state right: {final} of"
+      f" {possible}; runs with every step passed: {clean}")

Reply via email to