This is an automated email from the ASF dual-hosted git repository. davsclaus pushed a commit to branch main in repository https://gitbox.apache.org/repos/asf/camel-performance-tests.git
commit 39312f27c7c5205caec72a20d17340db8f7df6a7 Author: Claus Ibsen <[email protected]> AuthorDate: Sat Sep 26 15:52:12 2026 +0200 The ladder gives the model 20 tool calls per step, not 12 At 12, 115 of 720 step-runs hit the ceiling and only 34% of those passed against 80% overall: the number was partly measuring the budget. Re-running the four examples that hit it most with a ceiling of 24 scored 93 of 130 against 81, so the room was the constraint for three of those steps. 20 rather than 24: of 130 step-runs at a ceiling of 24, 101 used 20 calls or fewer and only 3 finished naturally needing 21 to 24. The 26 runs pinned at 24 are runs that loop -- one asked camel_catalog_doc(sql) six times for a syntax that answer does not contain -- so the last four calls buy looping, not passes. Capping at 20 leaves 29 of 130 pinned instead of 26, and the same 130 step-runs took 168 minutes at 24 against 133 at 12. Co-Authored-By: Claude Opus 5 (1M context) <[email protected]> Claude-Session: https://claude.ai/code/session_01Bp3538HRBPMQkb5ta9xRaj --- ai-benchmark/agent_mcp_stepwise.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/ai-benchmark/agent_mcp_stepwise.py b/ai-benchmark/agent_mcp_stepwise.py index 219229d..e1dd732 100755 --- a/ai-benchmark/agent_mcp_stepwise.py +++ b/ai-benchmark/agent_mcp_stepwise.py @@ -23,7 +23,7 @@ MCP_URL = os.environ.get("MCP_URL", "http://127.0.0.1:9090/mcp") HERE = os.path.dirname(os.path.abspath(__file__)) TAG = os.environ.get("BENCH_TAG", "mcp-" + MODEL.replace(":", "_").replace("/", "_")) OUT = os.path.join(HERE, "stepwise", TAG) -MAX_TOOL_CALLS = int(os.environ.get("BENCH_TOOL_CALLS", "12")) +MAX_TOOL_CALLS = int(os.environ.get("BENCH_TOOL_CALLS", "20")) # round 2: BENCH_REFERENCE=1 skips the model and applies each step's reference files instead, to check the steps file itself REFERENCE = os.environ.get("BENCH_REFERENCE") == "1" TOOL_RESULT_CAP = 6000
