This is an automated email from the ASF dual-hosted git repository. davsclaus pushed a commit to branch main in repository https://gitbox.apache.org/repos/asf/camel-performance-tests.git
commit 06327ef8c7f5b1d3f14b146d04c425d323b99f59 Author: Claus Ibsen <[email protected]> AuthorDate: Tue Oct 6 10:17:33 2026 +0200 A third score: the last version right, without the errors of attempts the model fixed ok counts every error a step logged, also those of an attempt the model fixed later, or of test data it wrote itself an earlier step; ok_final ignores errors altogether. ok_lastwrite is in between: the final state is right and no errors were logged after the model's last accepted write. The harness snapshots the errors at each accepted write; the summary shows the new column, and counts runs from before it as ok. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]> Claude-Session: https://claude.ai/code/session_01STT6whBgK1AqsSsUKrnE8m --- ai-benchmark/agent_mcp_stepwise.py | 20 ++++++++++++++++++++ ai-benchmark/summarize_stepwise.py | 16 ++++++++++------ 2 files changed, 30 insertions(+), 6 deletions(-) diff --git a/ai-benchmark/agent_mcp_stepwise.py b/ai-benchmark/agent_mcp_stepwise.py index c185c0d..a821f48 100755 --- a/ai-benchmark/agent_mcp_stepwise.py +++ b/ai-benchmark/agent_mcp_stepwise.py @@ -340,6 +340,21 @@ def score(step, project, cfg, mcp, name, before): # ok_final: the end state is right (files and log), whatever happened on the way; ok also needs no errors result["ok_final"] = result["file_ok"] and result["props_ok"] and result["log_ok"] result["ok"] = result["ok_final"] and errors_fine + # ok_lastwrite: the end state is right and the model's last version logged no errors; the errors of attempts it + # fixed on the way do not count (they do in ok). Without a write in the step it is the same as ok + last = before.get("last_write") + if last is not None: + after = sum(1 for l in lines if isinstance(l, dict) and lvl(l) == "ERROR" and error_key(l) not in last[0] + and error_key(l) not in seen) + if isinstance(data, dict): + after += max(0, len(data.get("errors", []) or []) - max(last[1], before.get("error_count", 0))) + else: + after = result["errors"] + result["errors_after_last_write"] = after + lastwrite_fine = after == 0 or bool(chk.get("errors_ok")) + if chk.get("min_errors") and result["errors"] < chk["min_errors"]: + lastwrite_fine = False + result["ok_lastwrite"] = result["ok_final"] and lastwrite_fine result["log_sample"] = msgs[:5] return result, route, props @@ -425,6 +440,9 @@ def main(): trace = open(os.path.join(OUT, f"step{sid}.trace.jsonl"), "w") messages.append({"role": "user", "content": step["request"]}) calls = 0; tokens = 0; t0 = time.time(); writes = 0; refused = 0; answer = "" + # the errors as they stood after the model's last accepted write: what came after is what its final + # version did; what came before were attempts it fixed (ok_lastwrite) + before["last_write"] = None if REFERENCE: # the reference files stand in for the model's edits; the checks then score the steps file itself writes += apply_reference(project, cfg, step.get("reference") or {}, mcp, name, before) @@ -466,6 +484,8 @@ def main(): if fn["name"] in ("camel_write_file", "camel_edit_file") and ('"invalid"' in out or '"not-found"' in out or '"ambiguous"' in out or out.startswith("ERROR")): refused += 1 + elif fn["name"] in ("camel_write_file", "camel_edit_file"): + before["last_write"] = error_snapshot(mcp, name) out = out[:TOOL_RESULT_CAP] trace.write(json.dumps({"tool": fn["name"], "args": {k: (v if k != "content" else v[:1500]) for k, v in args.items()}, "result": out[:800]}) + "\n"); trace.flush() messages.append({"role": "tool", "content": out, "tool_name": fn["name"]}) diff --git a/ai-benchmark/summarize_stepwise.py b/ai-benchmark/summarize_stepwise.py index 8173c1b..0ba1a72 100644 --- a/ai-benchmark/summarize_stepwise.py +++ b/ai-benchmark/summarize_stepwise.py @@ -3,14 +3,15 @@ A step passes (ok) when its files, log and probes are right and it logged no new errors; final state right (ok_final) leaves the errors out: a model that tests its own app (the HTTP request tool) causes errors on the way to a right -answer, which ok counts against it.""" +answer, which ok counts against it. Last version right (ok_lastwrite) is in between: the final state is right and no +errors were logged after the model's last accepted write; runs before it was recorded count it as ok.""" import json, os, sys tag = sys.argv[1]; k = int(sys.argv[2]) if len(sys.argv) > 2 else 1 tags = [f"{tag}-{i}" for i in range(1, k + 1)] if k > 1 else [tag] names = sorted(os.path.splitext(f)[0] for f in os.listdir("steps-ladder") if f.endswith(".json")) -print(f"| example | steps | passes per step ({' '.join(tags)}) | all steps passed | final state right |") -print("|---|---|---|---|---|") -total = 0; possible = 0; clean = 0; final = 0 +print(f"| example | steps | passes per step ({' '.join(tags)}) | all steps passed | last version right | final state right |") +print("|---|---|---|---|---|---|") +total = 0; possible = 0; clean = 0; final = 0; lastw = 0 for n in names: per = []; runs = [] for t in tags: @@ -26,5 +27,8 @@ for n in names: ok = sum(1 for r in runs if s < len(r) and r[s]["ok"]); cols.append(f"{ok}/{len(runs)}"); total += ok; possible += len(runs) allok = sum(1 for r in runs if r and all(x["ok"] for x in r)); clean += allok fin = sum(1 for r in runs for x in r if x.get("ok_final", x["ok"])); final += fin - print(f"| {n} | {steps} | {' '.join(cols)} | {allok}/{len(runs)} | {fin}/{sum(len(r) for r in runs)} |") -print(f"\nsteps passed: {total} of {possible}; final state right: {final} of {possible}; runs with every step passed: {clean}") + lw = sum(1 for r in runs for x in r if x.get("ok_lastwrite", x["ok"])); lastw += lw + n_steps = sum(len(r) for r in runs) + print(f"| {n} | {steps} | {' '.join(cols)} | {allok}/{len(runs)} | {lw}/{n_steps} | {fin}/{n_steps} |") +print(f"\nsteps passed: {total} of {possible}; last version right: {lastw} of {possible}; final state right: {final} of" + f" {possible}; runs with every step passed: {clean}")
