This is an automated email from the ASF dual-hosted git repository.

davsclaus pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/camel-performance-tests.git

commit 1b947aa79b7a237a7ba3837a4fad8ee3124c8552
Author: Claus Ibsen <[email protected]>
AuthorDate: Sat Sep 26 11:23:22 2026 +0200

    Round 2 of the one-shot suite: k runs, pass@k and pass^k, a held-out set 
and services
    
    run-suite.sh <tag> <k> runs the suite k times and ends with passk.py, which
    prints per-example passes out of k plus pass@k (passed at least once) and
    pass^k (passed every time) -- the two consistency measures of the MuleSoft
    integration-skill post, so the series can be compared with it.
    
    Set B (examples-intermediate.json) is six intermediate examples the model 
had
    never been tested on. An entry declares what it needs in the JSON rather 
than
    in the harness: infra for services started with camel infra run and whose
    connection data is appended to the prompt as a developer would read it, seed
    for files copied in before the model's, hint for one extra sentence, and
    pre/post for commands around the example.
    
    Co-Authored-By: Claude Opus 5 (1M context) <[email protected]>
    Claude-Session: https://claude.ai/code/session_01Bp3538HRBPMQkb5ta9xRaj
---
 ai-benchmark/README.md                  |  20 ++
 ai-benchmark/agent_local.py             | 250 ++++++++++++++--
 ai-benchmark/examples-intermediate.json |  61 ++++
 ai-benchmark/examples-ladder-10.json    | 235 +++++++++++++++
 ai-benchmark/examples-ladder-dry.json   | 109 +++++++
 ai-benchmark/examples-ladder.json       | 508 ++++++++++++++++++++++++++++++++
 ai-benchmark/gen_ladder.py              | 132 +++++++++
 ai-benchmark/gen_local.py               |   7 +-
 ai-benchmark/passk.py                   |  38 +++
 ai-benchmark/run-suite.sh               |  40 ++-
 ai-benchmark/run_one.sh                 |   8 +-
 ai-benchmark/steps.json                 |   7 +-
 12 files changed, 1380 insertions(+), 35 deletions(-)

diff --git a/ai-benchmark/README.md b/ai-benchmark/README.md
index bbf4207..b65287c 100644
--- a/ai-benchmark/README.md
+++ b/ai-benchmark/README.md
@@ -107,3 +107,23 @@ were found (about 15 minutes of reading per run).
   those, and `run-suite.sh` runs `caffeinate` on macOS.
 - Keep the examples away from the model: never offer `camel_catalog_examples` 
or `camel_catalog_example_file` in
   `BENCH_TOOL_ALLOW` for a benchmark that uses the examples repository.
+
+## Round 2: k runs, a held-out set, services
+
+Added 2026-09-17 for the second series.
+
+- `run-suite.sh <tag> <k>` runs the suite k times as `<tag>-1 .. <tag>-k` and 
ends with `passk.py`, which prints
+  per-example passes out of k, **pass@k** (passed at least once) and 
**pass^k** (passed every time), the two
+  consistency measures of the MuleSoft integration-skill post so the series 
can be compared with it.
+- `BENCH_EXAMPLES=examples-intermediate.json BENCH_STEPWISE=0 run-suite.sh b 
3` runs set B: six intermediate examples
+  the model has never been tested on (openapi-server, openapi-client, sql, 
artemis, mqtt, route-topology).
+  Each entry may declare, all visible in the JSON rather than hidden in the 
harness:
+  - `infra`: services started with `camel infra run <svc> --background` before 
the example and stopped after it
+    (postgres, artemis, mosquitto, kafka); the connection data from `camel 
infra get <svc> --json` is appended
+    to the prompt, as a developer would read it from the same command. 
Postgres needs about 80 s to come up.
+  - `seed`: files under `seed/<example>/` copied into every attempt folder 
before the model's files (the petstore
+    OpenAPI spec and sample payloads); the prompt lists them and says not to 
rewrite them.
+  - `hint`: one extra sentence in the prompt (the MQTT topic, the petstore 
base path).
+  - `pre` / `post`: shell commands run in this directory around the example 
(openapi-client starts the reference
+    petstore server from `seed/openapi-server-ref/` and stops it after).
+- Docker Desktop must be running for `infra`; `camel infra` pulls the images 
on first use.
diff --git a/ai-benchmark/agent_local.py b/ai-benchmark/agent_local.py
index 3b3a651..9de7e9a 100755
--- a/ai-benchmark/agent_local.py
+++ b/ai-benchmark/agent_local.py
@@ -10,7 +10,8 @@ Per example:
     fails we feed the validator/run errors back and loop.
 Everything (tool calls, timings, tokens) is logged to 
<BENCH_OUT>/<name>/trace.jsonl.
 """
-import json, os, re, subprocess, sys, time, urllib.request
+import glob
+import json, os, re, shutil, subprocess, sys, time, urllib.request
 sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
 from mcp_client import McpClient  # noqa: E402
 from gen_local import write_files, strip_think  # noqa: E402
@@ -19,6 +20,8 @@ MODEL = os.environ.get("BENCH_MODEL", "qwen3.6:35b-a3b")
 HOST = os.environ.get("OLLAMA_HOST", "http://localhost:11434";)
 HERE = os.path.dirname(os.path.abspath(__file__))
 OUT = os.path.join(HERE, os.environ.get("BENCH_OUT", "oneshot"))
+EXAMPLES_FILE = os.environ.get("BENCH_EXAMPLES", "examples.json")   # 
examples-intermediate.json for set B
+INFRA_TIMEOUT = int(os.environ.get("BENCH_INFRA_TIMEOUT", "300"))    # seconds 
to wait for `camel infra` services
 MAX_ROUNDS = int(os.environ.get("BENCH_ROUNDS", "3"))
 MAX_TOOL_CALLS = int(os.environ.get("BENCH_TOOL_CALLS", "10"))
 TOOL_RESULT_CAP = 6000
@@ -48,8 +51,13 @@ Route files use the extension .camel.yaml. Add 
application.properties, Java bean
 
 
 def ollama_chat(messages, tools):
-    body = json.dumps({"model": MODEL, "messages": messages, "tools": tools, 
"stream": False,
-                       "options": {"temperature": 0.2, "num_ctx": 
32768}}).encode()
+    # ladder runs (2026-09-19): thinking off, as the Camel CLI's own Ollama 
client sends ("think": false); the dry run
+    # with thinking on spent 460-500 s and 27-29k tokens per spiral and 
answered nothing three times in four examples.
+    # BENCH_THINK=1 restores the round-1 behaviour. num_predict bounds a 
runaway answer (default 16384).
+    req = {"model": MODEL, "messages": messages, "tools": tools, "stream": 
False,
+           "think": os.environ.get("BENCH_THINK", "0") == "1",
+           "options": {"temperature": 0.2, "num_ctx": 32768, "num_predict": 
int(os.environ.get("BENCH_NUM_PREDICT", "16384"))}}
+    body = json.dumps(req).encode()
     req = urllib.request.Request(HOST + "/api/chat", data=body, 
headers={"Content-Type": "application/json"})
     t0 = time.time()
     with urllib.request.urlopen(req, timeout=1800) as r:
@@ -69,7 +77,14 @@ def to_ollama_tools(mcp_tools):
     return out
 
 
-def run_folder(folder, secs, probe):
+def run_folder(folder, secs, probe, probe_regex=None, checks=None):
+    """checks (round 2, the ladder set): the entry's own checks, all visible 
in the JSON:
+    log_regex      list of regexes the run log must match (the behaviour the 
description promises)
+    log_not_regex  list of regexes the run log must not match (the pending 
order must not reach the warehouse)
+    expected_errors regex: error lines that are the example's own behaviour (a 
retried delivery, a rejected invoice)
+    require_files  globs that must exist in the folder after the run (a Java 
bean, application-prod.properties)
+    output_files   [[glob, min count], ...] files the run must have produced 
(outbox/*.json)"""
+    checks = checks or {}
     subprocess.run([os.path.join(HERE, "run_one.sh"), folder, str(secs), probe 
or ""], check=False)
     v = open(os.path.join(folder, "validate.log")).read()
     r = open(os.path.join(folder, "run.log")).read()
@@ -83,7 +98,15 @@ def run_folder(folder, secs, probe):
     errs = [l for l in r.splitlines()
             if re.search(r"ERROR|Exception|Caused 
by|Unsupported|Unknown|Failed|No bean", l)
             and not (re.search(r"\.(yaml|java):\d+\s", l) and not 
re.search(r"Exception|Caused by|Failed delivery", l))]
-    activity = len(re.findall(r"\.yaml:\d+ |\.java:\d+ ", r)) > 0 or 
bool(p.strip())
+    if checks.get("expected_errors"):
+        errs = [l for l in errs if not re.search(checks["expected_errors"], l)]
+    # round 2: an example may say what the probe must show (probe_regex); a 
404 or an empty body is then not activity
+    probe_ok = bool(p.strip()) and (probe_regex is None or 
re.search(probe_regex, p) is not None)
+    activity = len(re.findall(r"\.yaml:\d+ |\.java:\d+ ", r)) > 0 or probe_ok
+    if probe_regex is not None and not probe_ok:
+        errs_probe = [f"probe did not show the expected result (wanted 
/{probe_regex}/): " + p.strip()[:300]]
+    else:
+        errs_probe = []
     if not activity:
         # run 17 on: a log step with its own logName (priority-logger) logs 
under that name, not the route file;
         # any log line from a logger that is not Camel's own counts as route 
activity
@@ -96,12 +119,132 @@ def run_folder(folder, secs, probe):
     # CAMEL-24701) hides the "Routes startup" line the count is read from
     if nroutes == 0 and activity:
         nroutes = 1
-    ok = (not bad_validate) and nroutes > 0 and not errs and activity
-    return ok, v, "\n".join(errs[:15]), nroutes, activity
+    # the ladder checks: what the description promises, checked on the log and 
the folder; the feedback names the
+    # promised behaviour (the developer's words), never the regex
+    errs_check = []
+    ran = (not bad_validate) and nroutes > 0 and not errs
+    # (only when something went through a route: with no route output at all 
the "no log output" message below,
+    # which names the trigger, is the precise one; dry run 4 read from an 
orders/ directory that did not exist,
+    # three times, and was told the log did not show the behaviour)
+    if ran and activity and checks.get("log_regex"):
+        # a check is [regex, promise] (the promise is the description's own 
words); older sets carry a bare regex
+        missing = [c[1] if isinstance(c, list) else None
+                   for c in checks["log_regex"] if re.search(c[0] if 
isinstance(c, list) else c, r) is None]
+        if missing:
+            if all(missing):
+                errs_check.append("the log does not show: " + "; 
".join(missing))
+            else:
+                errs_check.append("the log does not show the expected 
behaviour: " + checks.get("expect", "see the request"))
+    for x in checks.get("log_not_regex", []):
+        if ran and re.search(x[0] if isinstance(x, list) else x, r):
+            errs_check.append("the log shows something the request rules out: 
" + checks.get("expect", "see the request"))
+            break
+    for g in checks.get("require_files", []):
+        if not glob.glob(os.path.join(folder, g)):
+            errs_check.append(f"the project must contain a file matching {g}")
+    for f, needle in checks.get("require_text", {}).items():
+        path = os.path.join(folder, f)
+        if not (os.path.exists(path) and needle in open(path).read()):
+            errs_check.append(f"{f} must contain {needle}")
+    for g, n in checks.get("output_files", []):
+        found = len(glob.glob(os.path.join(folder, g), recursive=True))
+        if ran and found < n:
+            errs_check.append(f"the run must produce at least {n} file(s) 
matching {g}, found {found}")
+    ok = (not bad_validate) and nroutes > 0 and not errs and activity and not 
errs_probe and not errs_check
+    # CAMEL-24855: the file consumer says when it created the directory it 
reads from; passed on as evidence when
+    # the run failed, since a route reading a directory the project does not 
have is otherwise silent
+    created = re.findall(r"Created starting directory: (\S+) \(it did not 
exist\)", r)
+    if created and not ok:
+        errs_check.append("camel run said it created the starting directory " 
+ ", ".join(created)
+                          + " (it did not exist): the route read an empty 
directory the project does not have;"
+                          + " the given files are in the project folder 
itself")
+    return ok, v, "\n".join((errs + errs_probe + errs_check)[:15]), nroutes, 
activity
+
+
+# --- set B support: services, seed files and hooks (round 2) 
---------------------------------------------
+# An example may declare:
+#   "infra": ["postgres"]        services started with `camel infra run <svc> 
--background` before the example and
+#                                stopped after it; their connection data 
(`camel infra get <svc> --json`) is handed
+#                                to the model in the prompt, as a developer 
would read it from the same command
+#   "seed": true                 files under seed/<example>/ are copied into 
every attempt folder before the model's
+#                                files are written (an OpenAPI spec, sample 
payloads); the prompt lists them
+#   "hint": "..."                one extra sentence in the prompt (a topic 
name, a server URL); kept in the JSON so
+#                                the extra information is visible, not hidden 
in the harness
+#   "pre": "cmd", "post": "cmd"  shell commands run in the harness directory 
before and after the example
+def infra_start(services):
+    for svc in services:
+        subprocess.run(["camel", "infra", "run", svc, "--background"], 
check=False,
+                       stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
+    data, deadline = {}, time.time() + INFRA_TIMEOUT
+    for svc in services:
+        while time.time() < deadline:
+            out = subprocess.run(["camel", "infra", "get", svc, "--json"], 
capture_output=True, text=True).stdout
+            m = re.search(r"\{.*\}", out, re.S)
+            if m:
+                try:
+                    data[svc] = json.loads(m.group(0)); break
+                except json.JSONDecodeError:
+                    pass
+            time.sleep(5)
+        else:
+            data[svc] = {"error": f"{svc} did not come up within 
{INFRA_TIMEOUT}s"}
+    return data
+
+
+def infra_stop(services):
+    for svc in services:
+        subprocess.run(["camel", "infra", "stop", svc], check=False, 
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
+
+
+def infra_prompt(data):
+    lines = []
+    for svc, d in data.items():
+        keep = {k: v for k, v in d.items() if k in ("endpointUri", "jdbcUrl", 
"username", "password", "host", "port",
+                                                    "beanProperties", 
"serviceAddress", "brokerUrl", "remoteURI",
+                                                    "brokers", 
"getBootstrapServers")}
+        lines.append(f"- {svc}: {json.dumps(keep or d)}")
+    return ("\n\nThe following external services are already running on this 
machine; connect to them with these "
+            "details (from `camel infra get`):\n" + "\n".join(lines)) if lines 
else ""
+
+
+def seed_files(name, folder):
+    src = os.path.join(HERE, "seed", name)
+    if not os.path.isdir(src):
+        return []
+    shutil.copytree(src, folder, dirs_exist_ok=True)
+    return sorted(os.path.relpath(os.path.join(r, f), src) for r, _, fs in 
os.walk(src) for f in fs)
+
+
+SEED_SHOW_EXT = (".json", ".csv", ".xml", ".xsl", ".groovy", ".txt", ".yaml", 
".properties")
+SEED_SHOW_MAX = int(os.environ.get("BENCH_SEED_SHOW_MAX", "2500"))   # bytes 
per shown file
+
+
+def seed_contents(name, seeded):
+    """the content of the small text seed files, one per directory, as the 
developer would see them"""
+    src = os.path.join(HERE, "seed", name)
+    shown_dirs, out = set(), []
+    for rel in seeded:
+        d = os.path.dirname(rel)
+        if not rel.endswith(SEED_SHOW_EXT) or d in shown_dirs:
+            continue
+        path = os.path.join(src, rel)
+        if os.path.getsize(path) > SEED_SHOW_MAX:
+            continue
+        shown_dirs.add(d)
+        siblings = [r for r in seeded if os.path.dirname(r) == d and r != rel]
+        note = f" (the other files in {d}/ have the same shape)" if siblings 
and d else ""
+        out.append(f"\n\n{rel}{note}:\n```\n{open(path).read().rstrip()}\n```")
+    return "".join(out)
+
+
+def hook(cmd, log):
+    if cmd:
+        r = subprocess.run(cmd, shell=True, cwd=HERE, capture_output=True, 
text=True)
+        print(f"hook `{cmd}` exit={r.returncode} {r.stdout.strip()[:200]} 
{r.stderr.strip()[:200]}", file=log, flush=True)
 
 
 def main():
-    examples = json.load(open(os.path.join(HERE, "examples.json")))
+    examples = json.load(open(os.path.join(HERE, EXAMPLES_FILE)))
     only = sys.argv[1:]
     mcp = McpClient(); mcp.initialize(); tools = 
to_ollama_tools(mcp.list_tools())
     log = open(os.path.join(HERE, "agent_local.log"), "a")
@@ -115,10 +258,28 @@ def main():
             continue
         os.makedirs(base, exist_ok=True)
         trace = open(os.path.join(base, "trace.jsonl"), "w")
+        hook(ex.get("pre"), log)
+        infra = infra_start(ex.get("infra", [])) if ex.get("infra") else {}
+        if infra:
+            trace.write(json.dumps({"infra": infra}) + "\n"); trace.flush()
+        seeded = seed_files(n, os.path.join(base, "seed-check")) if 
ex.get("seed") else []
+        shutil.rmtree(os.path.join(base, "seed-check"), ignore_errors=True)
+        prompt = f"Create a runnable Camel CLI example: {ex['prompt']}."
+        if ex.get("hint"):
+            prompt += " " + ex["hint"]
+        if seeded:
+            prompt += ("\n\nThe project folder already contains these files; 
use them and do not rewrite them: "
+                       + ", ".join(seeded) + ".")
+            # ladder (09-19): a developer opens the data file before writing 
the route; the model has no file tool,
+            # so the prompt shows the content of the small text seeds (one 
file per directory, the rest have the same
+            # shape). Dry run 3 guessed $.lineItems and $.items for a field 
called lines.
+            prompt += seed_contents(n, seeded)
+        prompt += infra_prompt(infra)
         messages = [{"role": "system", "content": SYSTEM},
-                    {"role": "user", "content": f"Create a runnable Camel CLI 
example: {ex['prompt']}."}]
+                    {"role": "user", "content": prompt}]
         result = {"name": n, "rounds": 0, "tool_calls": 0, "ok": False, 
"seconds": 0, "tokens": 0}
         t_start = time.time()
+        last_call, last_out = None, ""
         for rnd in range(1, MAX_ROUNDS + 1):
             result["rounds"] = rnd
             calls = 0
@@ -134,6 +295,24 @@ def main():
                                         "tool_calls": msg.get("tool_calls"), 
"content": (msg.get("content") or "")[:400]}) + "\n")
                 trace.flush()
                 messages.append({"role": "assistant", "content": 
msg.get("content") or "", "tool_calls": msg.get("tool_calls")})
+                if msg.get("tool_calls") and calls >= MAX_TOOL_CALLS and not 
text:
+                    # round 2 harness fix: the model wants another tool call 
but the round's budget is spent; before,
+                    # the empty content of that message was taken as the 
answer and the round failed on "the file has
+                    # no YAML" (seen on openapi-server). Tell it the budget is 
spent and let it answer without tools.
+                    messages.append({"role": "user", "content":
+                                     f"You have used the {MAX_TOOL_CALLS} tool 
calls allowed in this round. Answer now with the "
+                                     "complete files in the '=== FILE: <name> 
===' format, without further tool calls."})
+                    try:
+                        data, secs = ollama_chat(messages, [])
+                    except Exception as e:
+                        trace.write(json.dumps({"round": rnd, "error": 
str(e)}) + "\n"); break
+                    msg = data["message"]; result["tokens"] += 
data.get("eval_count", 0)
+                    trace.write(json.dumps({"round": rnd, "secs": round(secs, 
1), "eval": data.get("eval_count"),
+                                            "budget_spent": True, "content": 
(msg.get("content") or "")[:400]}) + "\n")
+                    trace.flush()
+                    messages.append({"role": "assistant", "content": 
msg.get("content") or ""})
+                    text = msg.get("content") or ""
+                    break
                 if msg.get("tool_calls") and calls < MAX_TOOL_CALLS:
                     for tc in msg["tool_calls"]:
                         fn = tc["function"]; calls += 1; result["tool_calls"] 
+= 1
@@ -143,10 +322,19 @@ def main():
                                 args = json.loads(args)
                             except Exception:
                                 args = {}
-                        try:
-                            out = mcp.call(fn["name"], args)
-                        except Exception as e:
-                            out = "ERROR: " + str(e)
+                        # ladder (09-19): the same call as the previous one 
(name and arguments) is answered from
+                        # the previous result with a note, so a model that 
repeats a validation of unchanged content
+                        # (error-handling repeated one ten times in the dry 
run) is told instead of charged a round trip
+                        key = (fn["name"], json.dumps(args, sort_keys=True))
+                        if key == last_call:
+                            out = ("This is the same call as your previous 
one, with the same arguments, so the answer is "
+                                   "unchanged. Change the content before 
validating again.\n\n" + last_out)
+                        else:
+                            try:
+                                out = mcp.call(fn["name"], args)
+                            except Exception as e:
+                                out = "ERROR: " + str(e)
+                            last_call, last_out = key, out
                         out = out[:TOOL_RESULT_CAP]
                         trace.write(json.dumps({"round": rnd, "tool": 
fn["name"], "args": args, "result": out[:600]}) + "\n")
                         trace.flush()
@@ -154,29 +342,57 @@ def main():
                     continue
                 text = msg.get("content") or ""
                 break
+            if not text.strip():
+                # an empty answer (a spiral that ran out, a tool budget hit): 
say so, nothing to save or run
+                print(f"{n}: round{rnd} calls={calls} empty answer", file=log, 
flush=True)
+                messages.append({"role": "user", "content": "Your answer was 
empty: it contained no files. Output the complete files in the '=== FILE: 
<name> ===' format."})
+                continue
             folder = os.path.join(base, f"attempt{rnd}")
+            os.makedirs(folder, exist_ok=True)
+            if ex.get("seed"):
+                seed_files(n, folder)
             files = write_files(text, folder)
             with open(os.path.join(folder, "raw.txt"), "w") as f:
                 f.write(text)
-            ok, v, errs, nroutes, activity = run_folder(folder, 
ex["run_seconds"], ex.get("probe"))
+            # after the series (09-19): the given files stay given. The model 
rewrote the seeded stylesheet in all
+            # three xslt attempts of every suite (with a wrong XSL namespace) 
although the prompt says not to; the
+            # seeds are restored after the model's files are written and the 
feedback says which were restored
+            restored = []
+            if ex.get("seed"):
+                src = os.path.join(HERE, "seed", n)
+                for rel in ex.get("seed_files", []):
+                    dst = os.path.join(folder, rel)
+                    if os.path.basename(rel) in files:
+                        shutil.copy2(os.path.join(src, rel), dst)
+                        restored.append(rel)
+                if restored:
+                    print(f"{n}: round{rnd} restored seed files rewritten by 
the model: {restored}", file=log, flush=True)
+            ok, v, errs, nroutes, activity = run_folder(folder, 
ex["run_seconds"], ex.get("probe"), ex.get("probe_regex"), ex)
             print(f"{n}: round{rnd} calls={calls} files={files} ok={ok} 
routes={nroutes} activity={activity}", file=log, flush=True)
             if ok:
                 result["ok"] = True
                 break
             fb = "I saved and ran your files with the Camel CLI.\n\n`camel 
validate yaml` output:\n" + (v.strip() or "(passed)")
+            if restored:
+                fb = ("You rewrote " + ", ".join(restored) + ", which the 
project already provides; the original was kept "
+                      "and your version discarded. Use the given file as it 
is.\n\n" + fb)
             if errs:
                 fb += "\n\nErrors from `camel run`:\n" + errs
             if nroutes == 0 and not errs:
                 fb += "\n\nThe application started but loaded 0 routes."
-            if nroutes > 0 and not errs and not activity:
-                fb += (f"\n\nThe route loaded but produced no log output in 
{ex['run_seconds']} seconds; it must produce "
-                       "output on its own (timer trigger, or create the input 
files it reads).")
+            if nroutes > 0 and not activity and "Failed" not in errs and 
"Exception" not in errs:
+                fb += (f"\n\nThe route loaded but produced no log output in 
{ex['run_seconds']} seconds: no message went "
+                       "through any route. It must produce output on its own 
(a timer trigger, or a file consumer on a "
+                       "directory that contains the input files named in the 
request).")
             fb += "\n\nUse the tools to check the options you are unsure about 
and validate the YAML, then output the complete corrected files again in the 
same '=== FILE: <name> ===' format."
             messages.append({"role": "user", "content": fb})
         result["seconds"] = round(time.time() - t_start, 1)
         json.dump(result, open(os.path.join(base, "result.json"), "w"))
         json.dump(messages, open(os.path.join(base, "messages.json"), "w"))
         trace.close()
+        if ex.get("infra"):
+            infra_stop(ex["infra"])
+        hook(ex.get("post"), log)
         print(f"{n}: DONE ok={result['ok']} rounds={result['rounds']} 
tool_calls={result['tool_calls']} secs={result['seconds']} 
tokens={result['tokens']}", file=log, flush=True)
     log.close()
 
diff --git a/ai-benchmark/examples-intermediate.json 
b/ai-benchmark/examples-intermediate.json
new file mode 100644
index 0000000..49e9912
--- /dev/null
+++ b/ai-benchmark/examples-intermediate.json
@@ -0,0 +1,61 @@
+[
+ {
+  "name": "sql",
+  "prompt": "Use a SQL database with Camel and Postgres",
+  "infra": [
+   "postgres"
+  ],
+  "expect": "a table is created, a row inserted and a periodic select logs the 
rows from the running Postgres",
+  "run_seconds": 15
+ },
+ {
+  "name": "artemis",
+  "prompt": "Setup connection factory to a remote Apache ActiveMQ Artemis 
messaging broker",
+  "infra": [
+   "artemis"
+  ],
+  "expect": "a producer sends messages to a JMS queue on the running Artemis 
broker and a consumer logs them",
+  "run_seconds": 15
+ },
+ {
+  "name": "mqtt",
+  "prompt": "Receive MQTT events from an external MQTT broker",
+  "infra": [
+   "mosquitto"
+  ],
+  "hint": "Events are published to the topic `temperature` as JSON objects 
with a numeric `value` field, for example {\"value\": 25}.",
+  "expect": "a route subscribes to the temperature topic on the running 
broker, reads the value field and logs each event",
+  "run_seconds": 15,
+  "probe": "c=$(docker ps -q --filter ancestor=$(docker ps --format 
'{{.Image}}' | grep -i mosquitto | head -1)); for v in 25 15 30; do docker exec 
$c mosquitto_pub -h localhost -t temperature -m \"{\\\"value\\\": $v}\"; sleep 
1; done; echo published",
+  "probe_regex": "published"
+ },
+ {
+  "name": "route-topology",
+  "prompt": "Demonstrates inter-route topology with triggers, shared routes, 
and external systems",
+  "infra": [
+   "kafka"
+  ],
+  "expect": "several routes connected through direct: and kafka: endpoints; a 
timer generates orders that flow through a shared validation route to Kafka and 
a consumer logs them",
+  "run_seconds": 20
+ },
+ {
+  "name": "ftp",
+  "prompt": "Integrate ActiveMQ messaging with an FTP server",
+  "infra": [
+   "artemis",
+   "ftp"
+  ],
+  "hint": "Messages arrive on the JMS queue `cheese` and each one must be 
uploaded as a file to the FTP server.",
+  "expect": "a route consumes the cheese queue on the running Artemis broker, 
logs each message and writes it as a file to the running FTP server",
+  "run_seconds": 30,
+  "probe": "sleep 14; pid=$(camel ps 2>/dev/null | awk 'NR==2{print $1}'); 
camel cmd send $pid --endpoint=jms:cheese --body='hello ftp' 
--logging-color=false 2>&1 | sed 's/\\x1b\\[[0-9;]*m//g' | grep -i 
'sent\\|error' | head -2",
+  "probe_regex": "Sent \\(success\\)"
+ },
+ {
+  "name": "genai-observability",
+  "prompt": "Observe LangChain4j LLM calls with OpenTelemetry gen_ai spans and 
Micrometer metrics",
+  "hint": "Ollama is running at http://localhost:11434 with the model 
`qwen3.6:35b-a3b`; the Camel CLI needs the dependencies camel-langchain4j-chat, 
camel-ai-observability and langchain4j-ollama, declared with 
camel.jbang.dependencies in application.properties.",
+  "expect": "a timer route sends a prompt to the local Ollama through 
langchain4j-chat and logs the reply and the model name; AI observability is 
enabled",
+  "run_seconds": 45
+ }
+]
\ No newline at end of file
diff --git a/ai-benchmark/examples-ladder-10.json 
b/ai-benchmark/examples-ladder-10.json
new file mode 100644
index 0000000..8f5db21
--- /dev/null
+++ b/ai-benchmark/examples-ladder-10.json
@@ -0,0 +1,235 @@
+[
+  {
+    "name": "run-order-generator",
+    "example": "run/order-generator",
+    "level": "run",
+    "prompt": "A timer creates a shop order every five seconds: a Java bean 
hands out the order number, the body is the order as JSON, and the log shows 
each new order. The order feed every later example starts from",
+    "expect": "A timer creates a shop order every five seconds: a Java bean 
hands out the order number, the body is the order as JSON, and the log shows 
each new order. The order feed every later example starts from.",
+    "run_seconds": 12,
+    "require_files": [
+      "*.java"
+    ],
+    "log_regex": [
+      [
+        "(?s)\\{[^\\n]*\"[^\\n]*\\}.*\\n.*\\{[^\\n]*\"[^\\n]*\\}",
+        "two orders logged as JSON"
+      ]
+    ]
+  },
+  {
+    "name": "run-nightly-report",
+    "example": "run/nightly-report",
+    "level": "run",
+    "prompt": "A cron schedule runs the shop's inventory report, every ten 
seconds in the demo and nightly with a one-line change, and each run logs the 
stock counts with a timestamp",
+    "expect": "A cron schedule runs the shop's inventory report, every ten 
seconds in the demo and nightly with a one-line change, and each run logs the 
stock counts with a timestamp.",
+    "run_seconds": 25,
+    "log_regex": [
+      [
+        
"(?s)\\.yaml:\\d+\\s*:[^\\n]*\\d{4}-\\d{2}-\\d{2}[^\\n]*\\d+.*\\n.*\\.yaml:\\d+\\s*:[^\\n]*\\d{4}-\\d{2}-\\d{2}",
+        "two report runs logged, each with a timestamp and the stock counts"
+      ]
+    ]
+  },
+  {
+    "name": "run-properties-and-profiles",
+    "example": "run/properties-and-profiles",
+    "level": "run",
+    "prompt": "A timer logs a welcome with the shop name and currency from 
application.properties; run with --profile=prod and application-prod.properties 
overrides both, so the same route greets with the production values",
+    "expect": "A timer logs a welcome with the shop name and currency from 
application.properties; run with --profile=prod and application-prod.properties 
overrides both, so the same route greets with the production values.",
+    "run_seconds": 8,
+    "require_files": [
+      "application-prod.properties"
+    ],
+    "log_regex": [
+      [
+        "\\.yaml:\\d+\\s*:[^\\n]*\\b(EUR|USD|DKK|GBP)\\b",
+        "the welcome logged with the currency from application.properties"
+      ]
+    ]
+  },
+  {
+    "name": "transform-json-transform",
+    "example": "transform/json-transform",
+    "level": "transform",
+    "prompt": "The shop's order in order.json is reshaped for the warehouse: 
jsonpath reads the order id and the number of lines into headers, jq builds the 
pick list with only sku and quantity per line, and both the order and the pick 
list are logged",
+    "expect": "The shop's order in order.json is reshaped for the warehouse: 
jsonpath reads the order id and the number of lines into headers, jq builds the 
pick list with only sku and quantity per line, and both the order and the pick 
list are logged.",
+    "seed": true,
+    "seed_files": [
+      "order.json"
+    ],
+    "run_seconds": 10,
+    "log_regex": [
+      [
+        "ORD-1001",
+        "the order ORD-1001 logged"
+      ],
+      [
+        "\"sku\"\\s*:\\s*\"CAMEL-TSHIRT\"",
+        "the pick list logged as JSON with the sku CAMEL-TSHIRT"
+      ]
+    ]
+  },
+  {
+    "name": "transform-xml-to-json",
+    "example": "transform/xml-to-json",
+    "level": "transform",
+    "prompt": "A supplier's XML order dropped in the inbox directory is read 
with the Jackson XML data format and written out as the shop's JSON with the 
Jackson JSON data format, both logged; no mapping code, the XML elements and 
attributes become JSON fields",
+    "expect": "A supplier's XML order dropped in the inbox directory is read 
with the Jackson XML data format and written out as the shop's JSON with the 
Jackson JSON data format, both logged; no mapping code, the XML elements and 
attributes become JSON fields.",
+    "seed": true,
+    "seed_files": [
+      "inbox/supplier-order.xml"
+    ],
+    "run_seconds": 10,
+    "log_regex": [
+      [
+        "<order",
+        "the XML order logged as it was read (the <order element)"
+      ],
+      [
+        "\"CAMEL-TSHIRT\"",
+        "the same order logged as JSON (CAMEL-TSHIRT as a JSON string)"
+      ]
+    ]
+  },
+  {
+    "name": "transform-xslt",
+    "example": "transform/xslt",
+    "level": "transform",
+    "prompt": "A supplier's XML order dropped in the inbox directory is 
transformed by the stylesheet packing-slip.xsl into the packing slip the 
warehouse prints, one item per line and the total pieces to pick, and the slip 
is logged",
+    "expect": "A supplier's XML order dropped in the inbox directory is 
transformed by the stylesheet packing-slip.xsl into the packing slip the 
warehouse prints, one item per line and the total pieces to pick, and the slip 
is logged.",
+    "seed": true,
+    "seed_files": [
+      "inbox/supplier-order.xml",
+      "packing-slip.xsl"
+    ],
+    "run_seconds": 10,
+    "log_regex": [
+      [
+        "<packingSlip",
+        "the packing slip logged (a <packingSlip element)"
+      ],
+      [
+        "<pieces>3</pieces>",
+        "<pieces>3</pieces> in the logged slip"
+      ],
+      [
+        "CAMEL-TSHIRT",
+        "CAMEL-TSHIRT in the logged slip"
+      ]
+    ]
+  },
+  {
+    "name": "route-content-based-router",
+    "example": "route/content-based-router",
+    "level": "route",
+    "prompt": "Three orders from three countries are read from the orders 
directory and a choice routes each by its country: the Danish order to local 
delivery, the German order to EU shipping without customs, the US order to 
export with a customs declaration; each branch logs what it did",
+    "expect": "Three orders from three countries are read from the orders 
directory and a choice routes each by its country: the Danish order to local 
delivery, the German order to EU shipping without customs, the US order to 
export with a customs declaration; each branch logs what it did.",
+    "seed": true,
+    "seed_files": [
+      "orders/order-1001.json",
+      "orders/order-1002.json",
+      "orders/order-1003.json"
+    ],
+    "run_seconds": 10,
+    "log_regex": [
+      [
+        "(?im)^(?=.*ORD-1001)(?=.*(local|copenhagen)).*$",
+        "ORD-1001 logged as local delivery"
+      ],
+      [
+        "(?im)^(?=.*ORD-1002)(?=.*\\bEU\\b).*$",
+        "ORD-1002 logged as EU shipping"
+      ],
+      [
+        "(?im)^(?=.*ORD-1003)(?=.*(export|customs)).*$",
+        "ORD-1003 logged as export with customs"
+      ]
+    ]
+  },
+  {
+    "name": "route-order-lines",
+    "example": "route/order-lines",
+    "level": "route",
+    "prompt": "Each order read from the orders directory is split into one 
message per line, the order id travels along in a header, and the log shows the 
order, one pick line per line, and the parent's confirmation that all lines 
went to picking",
+    "expect": "Each order read from the orders directory is split into one 
message per line, the order id travels along in a header, and the log shows the 
order, one pick line per line, and the parent's confirmation that all lines 
went to picking.",
+    "seed": true,
+    "seed_files": [
+      "orders/order-1001.json",
+      "orders/order-1002.json",
+      "orders/order-1003.json"
+    ],
+    "run_seconds": 10,
+    "log_regex": [
+      [
+        "(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT).*$",
+        "a pick line for CAMEL-TSHIRT of ORD-1001"
+      ],
+      [
+        "(?im)^(?=.*ORD-1001)(?=.*CAMEL-MUG).*$",
+        "a pick line for CAMEL-MUG of ORD-1001"
+      ],
+      [
+        "ORD-1002",
+        "order ORD-1002 logged"
+      ],
+      [
+        "(?im)^(?=.*ORD-1001)(?=.*(all|\\b2\\b))(?=.*(line|pick)).*$",
+        "the confirmation that all lines of ORD-1001 went to picking"
+      ]
+    ]
+  },
+  {
+    "name": "route-filter-and-multicast",
+    "example": "route/filter-and-multicast",
+    "level": "route",
+    "prompt": "Three orders are read from the orders directory; a filter lets 
only the paid ones through and a multicast sends each paid order to both the 
warehouse route and the invoicing route, which log their part; the pending 
order is logged as received and goes no further",
+    "expect": "Three orders are read from the orders directory; a filter lets 
only the paid ones through and a multicast sends each paid order to both the 
warehouse route and the invoicing route, which log their part; the pending 
order is logged as received and goes no further.",
+    "seed": true,
+    "seed_files": [
+      "orders/order-1001.json",
+      "orders/order-1002.json",
+      "orders/order-1003.json"
+    ],
+    "run_seconds": 10,
+    "log_regex": [
+      [
+        "(?im)^(?=.*ORD-1001)(?=.*(warehouse|pick)).*$",
+        "ORD-1001 logged by the warehouse route"
+      ],
+      [
+        "(?im)^(?=.*ORD-1001)(?=.*(invoic|bill)).*$",
+        "ORD-1001 logged by the invoicing route"
+      ],
+      [
+        "ORD-1003",
+        "ORD-1003 logged as received"
+      ]
+    ],
+    "log_not_regex": [
+      "(?im)^(?=.*ORD-1003)(?=.*(warehouse|invoic|pick|bill)).*$"
+    ]
+  },
+  {
+    "name": "fail-well-circuit-breaker",
+    "example": "fail-well/circuit-breaker",
+    "level": "fail-well",
+    "prompt": "A stock check calls the supplier every second; the supplier 
goes down for nine calls, the breaker opens after two failures in its window of 
four, answers from the fallback while open, tries the supplier again after five 
seconds and closes once a call succeeds; each line logs the breaker state",
+    "expect": "A stock check calls the supplier every second; the supplier 
goes down for nine calls, the breaker opens after two failures in its window of 
four, answers from the fallback while open, tries the supplier again after five 
seconds and closes once a call succeeds; each line logs the breaker state.",
+    "run_seconds": 25,
+    "expected_errors": "(?i)supplier|simulat|down",
+    "log_regex": [
+      [
+        "(?i)\\bOPEN\\b",
+        "the breaker logged as OPEN"
+      ],
+      [
+        "(?i)\\bCLOSED\\b",
+        "the breaker logged as CLOSED"
+      ],
+      [
+        "(?i)(fallback|last known|no answer)",
+        "the fallback answer logged while open"
+      ]
+    ]
+  }
+]
\ No newline at end of file
diff --git a/ai-benchmark/examples-ladder-dry.json 
b/ai-benchmark/examples-ladder-dry.json
new file mode 100644
index 0000000..7ddfb67
--- /dev/null
+++ b/ai-benchmark/examples-ladder-dry.json
@@ -0,0 +1,109 @@
+[
+ {
+  "name": "run-order-generator",
+  "example": "run/order-generator",
+  "level": "run",
+  "prompt": "A timer creates a shop order every five seconds: a Java bean 
hands out the order number, the body is the order as JSON, and the log shows 
each new order. The order feed every later example starts from",
+  "expect": "A timer creates a shop order every five seconds: a Java bean 
hands out the order number, the body is the order as JSON, and the log shows 
each new order. The order feed every later example starts from.",
+  "run_seconds": 12,
+  "require_files": [
+   "*.java"
+  ],
+  "log_regex": [
+   "(?s)\\{[^\\n]*\"[^\\n]*\\}.*\\n.*\\{[^\\n]*\"[^\\n]*\\}"
+  ]
+ },
+ {
+  "name": "transform-json-transform",
+  "example": "transform/json-transform",
+  "level": "transform",
+  "prompt": "The shop's order in order.json is reshaped for the warehouse: 
jsonpath reads the order id and the number of lines into headers, jq builds the 
pick list with only sku and quantity per line, and both the order and the pick 
list are logged",
+  "expect": "The shop's order in order.json is reshaped for the warehouse: 
jsonpath reads the order id and the number of lines into headers, jq builds the 
pick list with only sku and quantity per line, and both the order and the pick 
list are logged.",
+  "seed": true,
+  "seed_files": [
+   "order.json"
+  ],
+  "run_seconds": 10,
+  "log_regex": [
+   "ORD-1001",
+   "\"sku\"\\s*:\\s*\"CAMEL-TSHIRT\""
+  ]
+ },
+ {
+  "name": "route-content-based-router",
+  "example": "route/content-based-router",
+  "level": "route",
+  "prompt": "Three orders from three countries are read from the orders 
directory and a choice routes each by its country: the Danish order to local 
delivery, the German order to EU shipping without customs, the US order to 
export with a customs declaration; each branch logs what it did",
+  "expect": "Three orders from three countries are read from the orders 
directory and a choice routes each by its country: the Danish order to local 
delivery, the German order to EU shipping without customs, the US order to 
export with a customs declaration; each branch logs what it did.",
+  "seed": true,
+  "seed_files": [
+   "orders/order-1001.json",
+   "orders/order-1002.json",
+   "orders/order-1003.json"
+  ],
+  "run_seconds": 10,
+  "log_regex": [
+   "(?im)^(?=.*ORD-1001)(?=.*(local|copenhagen)).*$",
+   "(?im)^(?=.*ORD-1002)(?=.*\\bEU\\b).*$",
+   "(?im)^(?=.*ORD-1003)(?=.*(export|customs)).*$"
+  ]
+ },
+ {
+  "name": "fail-well-error-handling",
+  "example": "fail-well/error-handling",
+  "level": "fail-well",
+  "prompt": "The three orders go to a payment provider: the first is charged 
at once, the second gets no answer twice and is charged on the third attempt 
after two retries logged as warnings, and the third is declined, logged as such 
and parked as a file for manual review",
+  "expect": "The three orders go to a payment provider: the first is charged 
at once, the second gets no answer twice and is charged on the third attempt 
after two retries logged as warnings, and the third is declined, logged as such 
and parked as a file for manual review.",
+  "seed": true,
+  "seed_files": [
+   "orders/order-1001.json",
+   "orders/order-1002.json",
+   "orders/order-1003.json"
+  ],
+  "run_seconds": 20,
+  "expected_errors": "(?i)payment|did not 
answer|declined|ConnectException|Failed delivery",
+  "log_regex": [
+   "(?im)^(?=.*ORD-1001)(?=.*charged).*$",
+   "(?im)^(?=.*ORD-1002)(?=.*charged).*$",
+   "(?im)^(?=.*ORD-1003)(?=.*(declin|park|manual|review)).*$",
+   "(?s)(WARN|retr|redeliver|attempt).*(WARN|retr|redeliver|attempt)"
+  ]
+ },
+ {
+  "name": "connect-stock-api",
+  "example": "connect/stock-api",
+  "level": "connect",
+  "prompt": "The shop's stock service on port 8080: GET /stock returns the 
stock file, GET /stock/{sku} returns one SKU as JSON and a 404 with an error 
message for an unknown SKU",
+  "expect": "The shop's stock service on port 8080: GET /stock returns the 
stock file, GET /stock/{sku} returns one SKU as JSON and a 404 with an error 
message for an unknown SKU.",
+  "seed": true,
+  "seed_files": [
+   "stock.json"
+  ],
+  "run_seconds": 15,
+  "probe": "curl -s localhost:8080/stock/CAMEL-MUG; echo; curl -s -o /dev/null 
-w '%{http_code}' localhost:8080/stock/CAMEL-SOCKS; echo; curl -s 
localhost:8080/stock | head -c 400",
+  "probe_regex": "(?s)CAMEL-MUG[^\\n]*42.*\\n404\\n.*CAMEL-TSHIRT"
+ },
+ {
+  "name": "contracts-openapi-client",
+  "example": "contracts/openapi-client",
+  "level": "contracts",
+  "prompt": "The picking desk reserves stock for every order line by calling 
the stock API by contract: rest-openapi turns the operationId reserveStock into 
the HTTP call from stock-api.json; the log shows each reservation and one 409 
for the cap that is out of stock. Needs the openapi-server example running",
+  "expect": "The picking desk reserves stock for every order line by calling 
the stock API by contract: rest-openapi turns the operationId reserveStock into 
the HTTP call from stock-api.json; the log shows each reservation and one 409 
for the cap that is out of stock. Needs the openapi-server example running.",
+  "seed": true,
+  "seed_files": [
+   "orders/order-1001.json",
+   "orders/order-1002.json",
+   "orders/order-1003.json",
+   "stock-api.json"
+  ],
+  "run_seconds": 20,
+  "expected_errors": "409|CAMEL-CAP|HttpOperationFailed",
+  "hint": "The stock API server (the openapi-server example) is already 
running at http://localhost:8080/api and its contract is the file 
stock-api.json.",
+  "pre": "./ref-server.sh start",
+  "post": "./ref-server.sh stop",
+  "log_regex": [
+   "(?im)^(?=.*CAMEL-CAP)(?=.*409).*$",
+   "(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*reserv).*$"
+  ]
+ }
+]
\ No newline at end of file
diff --git a/ai-benchmark/examples-ladder.json 
b/ai-benchmark/examples-ladder.json
new file mode 100644
index 0000000..facbc72
--- /dev/null
+++ b/ai-benchmark/examples-ladder.json
@@ -0,0 +1,508 @@
+[
+ {
+  "name": "run-order-generator",
+  "example": "run/order-generator",
+  "level": "run",
+  "prompt": "A timer creates a shop order every five seconds: a Java bean 
hands out the order number, the body is the order as JSON, and the log shows 
each new order. The order feed every later example starts from",
+  "expect": "A timer creates a shop order every five seconds: a Java bean 
hands out the order number, the body is the order as JSON, and the log shows 
each new order. The order feed every later example starts from.",
+  "run_seconds": 12,
+  "require_files": [
+   "*.java"
+  ],
+  "log_regex": [
+   [
+    "(?s)\\{[^\\n]*\"[^\\n]*\\}.*\\n.*\\{[^\\n]*\"[^\\n]*\\}",
+    "two orders logged as JSON"
+   ]
+  ]
+ },
+ {
+  "name": "run-nightly-report",
+  "example": "run/nightly-report",
+  "level": "run",
+  "prompt": "A cron schedule runs the shop's inventory report, every ten 
seconds in the demo and nightly with a one-line change, and each run logs the 
stock counts with a timestamp",
+  "expect": "A cron schedule runs the shop's inventory report, every ten 
seconds in the demo and nightly with a one-line change, and each run logs the 
stock counts with a timestamp.",
+  "run_seconds": 25,
+  "log_regex": [
+   [
+    
"(?s)\\.yaml:\\d+\\s*:[^\\n]*\\d{4}-\\d{2}-\\d{2}[^\\n]*\\d+.*\\n.*\\.yaml:\\d+\\s*:[^\\n]*\\d{4}-\\d{2}-\\d{2}",
+    "two report runs logged, each with a timestamp and the stock counts"
+   ]
+  ]
+ },
+ {
+  "name": "run-properties-and-profiles",
+  "example": "run/properties-and-profiles",
+  "level": "run",
+  "prompt": "A timer logs a welcome with the shop name and currency from 
application.properties; run with --profile=prod and application-prod.properties 
overrides both, so the same route greets with the production values",
+  "expect": "A timer logs a welcome with the shop name and currency from 
application.properties; run with --profile=prod and application-prod.properties 
overrides both, so the same route greets with the production values.",
+  "run_seconds": 8,
+  "require_files": [
+   "application-prod.properties"
+  ],
+  "log_regex": [
+   [
+    "\\.yaml:\\d+\\s*:[^\\n]*\\b(EUR|USD|DKK|GBP)\\b",
+    "the welcome logged with the currency from application.properties"
+   ]
+  ]
+ },
+ {
+  "name": "transform-json-transform",
+  "example": "transform/json-transform",
+  "level": "transform",
+  "prompt": "The shop's order in order.json is reshaped for the warehouse: 
jsonpath reads the order id and the number of lines into headers, jq builds the 
pick list with only sku and quantity per line, and both the order and the pick 
list are logged",
+  "expect": "The shop's order in order.json is reshaped for the warehouse: 
jsonpath reads the order id and the number of lines into headers, jq builds the 
pick list with only sku and quantity per line, and both the order and the pick 
list are logged.",
+  "seed": true,
+  "seed_files": [
+   "order.json"
+  ],
+  "run_seconds": 10,
+  "log_regex": [
+   [
+    "ORD-1001",
+    "the order ORD-1001 logged"
+   ],
+   [
+    "\"sku\"\\s*:\\s*\"CAMEL-TSHIRT\"",
+    "the pick list logged as JSON with the sku CAMEL-TSHIRT"
+   ]
+  ]
+ },
+ {
+  "name": "transform-xml-to-json",
+  "example": "transform/xml-to-json",
+  "level": "transform",
+  "prompt": "A supplier's XML order dropped in the inbox directory is read 
with the Jackson XML data format and written out as the shop's JSON with the 
Jackson JSON data format, both logged; no mapping code, the XML elements and 
attributes become JSON fields",
+  "expect": "A supplier's XML order dropped in the inbox directory is read 
with the Jackson XML data format and written out as the shop's JSON with the 
Jackson JSON data format, both logged; no mapping code, the XML elements and 
attributes become JSON fields.",
+  "seed": true,
+  "seed_files": [
+   "inbox/supplier-order.xml"
+  ],
+  "run_seconds": 10,
+  "log_regex": [
+   [
+    "<order",
+    "the XML order logged as it was read (the <order element)"
+   ],
+   [
+    "\"CAMEL-TSHIRT\"",
+    "the same order logged as JSON (CAMEL-TSHIRT as a JSON string)"
+   ]
+  ]
+ },
+ {
+  "name": "transform-csv-to-json",
+  "example": "transform/csv-to-json",
+  "level": "transform",
+  "prompt": "A CSV of invoices dropped in the inbox directory is read with the 
CSV data format, its header line naming the fields, split into one message per 
invoice, and each invoice is logged as JSON and written to the outbox directory 
as its own file",
+  "expect": "A CSV of invoices dropped in the inbox directory is read with the 
CSV data format, its header line naming the fields, split into one message per 
invoice, and each invoice is logged as JSON and written to the outbox directory 
as its own file.",
+  "seed": true,
+  "seed_files": [
+   "inbox/invoices.csv"
+  ],
+  "run_seconds": 10,
+  "output_files": [
+   [
+    "outbox/*.json",
+    3
+   ]
+  ],
+  "log_regex": [
+   [
+    "INV-2001",
+    "invoice INV-2001 logged"
+   ],
+   [
+    "INV-2003",
+    "invoice INV-2003 logged"
+   ],
+   [
+    "\"[a-zA-Z]+\"\\s*:\\s*\"INV-200\\d\"",
+    "an invoice logged as JSON (a quoted field with the value INV-200x)"
+   ]
+  ]
+ },
+ {
+  "name": "transform-data-mapping",
+  "example": "transform/data-mapping",
+  "level": "transform",
+  "prompt": "The shop's order in order.json is mapped field by field to the 
courier's shipment format, with renamed fields, a nested recipient, one parcel 
per line, a computed total and a service chosen from the country; the order is 
parsed to a map, the script shipment-mapping.groovy builds the shipment, and it 
is logged as JSON",
+  "expect": "The shop's order in order.json is mapped field by field to the 
courier's shipment format, with renamed fields, a nested recipient, one parcel 
per line, a computed total and a service chosen from the country; the order is 
parsed to a map, the script shipment-mapping.groovy builds the shipment, and it 
is logged as JSON.",
+  "seed": true,
+  "seed_files": [
+   "order.json",
+   "shipment-mapping.groovy"
+  ],
+  "run_seconds": 10,
+  "log_regex": [
+   [
+    "SHIP-1001",
+    "the shipment SHIP-1001 logged"
+   ],
+   [
+    "\"totalPieces\"\\s*:\\s*3",
+    "totalPieces 3 in the logged shipment JSON"
+   ],
+   [
+    "domestic",
+    "the domestic service in the logged shipment"
+   ]
+  ]
+ },
+ {
+  "name": "transform-groovy",
+  "example": "transform/groovy",
+  "level": "transform",
+  "prompt": "Two orders come in, one with a valid customer email and one with 
a bad one; a Groovy expression checks the address with Apache Commons 
Validator, a third-party library declared in application.properties, and the 
log shows one order accepted and one rejected",
+  "expect": "Two orders come in, one with a valid customer email and one with 
a bad one; a Groovy expression checks the address with Apache Commons 
Validator, a third-party library declared in application.properties, and the 
log shows one order accepted and one rejected.",
+  "seed": true,
+  "seed_files": [
+   "order-bad-email.json",
+   "order.json"
+  ],
+  "run_seconds": 10,
+  "require_text": {
+   "application.properties": "commons-validator"
+  },
+  "log_regex": [
+   [
+    "anna@example\\.com",
+    "[email protected] logged as accepted"
+   ],
+   [
+    "not-an-address",
+    "not-an-address logged as rejected"
+   ]
+  ]
+ },
+ {
+  "name": "transform-xslt",
+  "example": "transform/xslt",
+  "level": "transform",
+  "prompt": "A supplier's XML order dropped in the inbox directory is 
transformed by the stylesheet packing-slip.xsl into the packing slip the 
warehouse prints, one item per line and the total pieces to pick, and the slip 
is logged",
+  "expect": "A supplier's XML order dropped in the inbox directory is 
transformed by the stylesheet packing-slip.xsl into the packing slip the 
warehouse prints, one item per line and the total pieces to pick, and the slip 
is logged.",
+  "seed": true,
+  "seed_files": [
+   "inbox/supplier-order.xml",
+   "packing-slip.xsl"
+  ],
+  "run_seconds": 10,
+  "log_regex": [
+   [
+    "<packingSlip",
+    "the packing slip logged (a <packingSlip element)"
+   ],
+   [
+    "<pieces>3</pieces>",
+    "<pieces>3</pieces> in the logged slip"
+   ],
+   [
+    "CAMEL-TSHIRT",
+    "CAMEL-TSHIRT in the logged slip"
+   ]
+  ]
+ },
+ {
+  "name": "route-content-based-router",
+  "example": "route/content-based-router",
+  "level": "route",
+  "prompt": "Three orders from three countries are read from the orders 
directory and a choice routes each by its country: the Danish order to local 
delivery, the German order to EU shipping without customs, the US order to 
export with a customs declaration; each branch logs what it did",
+  "expect": "Three orders from three countries are read from the orders 
directory and a choice routes each by its country: the Danish order to local 
delivery, the German order to EU shipping without customs, the US order to 
export with a customs declaration; each branch logs what it did.",
+  "seed": true,
+  "seed_files": [
+   "orders/order-1001.json",
+   "orders/order-1002.json",
+   "orders/order-1003.json"
+  ],
+  "run_seconds": 10,
+  "log_regex": [
+   [
+    "(?im)^(?=.*ORD-1001)(?=.*(local|copenhagen)).*$",
+    "ORD-1001 logged as local delivery"
+   ],
+   [
+    "(?im)^(?=.*ORD-1002)(?=.*\\bEU\\b).*$",
+    "ORD-1002 logged as EU shipping"
+   ],
+   [
+    "(?im)^(?=.*ORD-1003)(?=.*(export|customs)).*$",
+    "ORD-1003 logged as export with customs"
+   ]
+  ]
+ },
+ {
+  "name": "route-order-lines",
+  "example": "route/order-lines",
+  "level": "route",
+  "prompt": "Each order read from the orders directory is split into one 
message per line, the order id travels along in a header, and the log shows the 
order, one pick line per line, and the parent's confirmation that all lines 
went to picking",
+  "expect": "Each order read from the orders directory is split into one 
message per line, the order id travels along in a header, and the log shows the 
order, one pick line per line, and the parent's confirmation that all lines 
went to picking.",
+  "seed": true,
+  "seed_files": [
+   "orders/order-1001.json",
+   "orders/order-1002.json",
+   "orders/order-1003.json"
+  ],
+  "run_seconds": 10,
+  "log_regex": [
+   [
+    "(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT).*$",
+    "a pick line for CAMEL-TSHIRT of ORD-1001"
+   ],
+   [
+    "(?im)^(?=.*ORD-1001)(?=.*CAMEL-MUG).*$",
+    "a pick line for CAMEL-MUG of ORD-1001"
+   ],
+   [
+    "ORD-1002",
+    "order ORD-1002 logged"
+   ],
+   [
+    "(?im)^(?=.*ORD-1001)(?=.*(all|\\b2\\b))(?=.*(line|pick)).*$",
+    "the confirmation that all lines of ORD-1001 went to picking"
+   ]
+  ]
+ },
+ {
+  "name": "route-aggregator",
+  "example": "route/aggregator",
+  "level": "route",
+  "prompt": "The warehouse reports each picked line on its own and the 
aggregator collects the lines of one order back into a shipment, correlated by 
the order id and complete when as many lines are in as the order had; each 
shipment is logged as JSON",
+  "expect": "The warehouse reports each picked line on its own and the 
aggregator collects the lines of one order back into a shipment, correlated by 
the order id and complete when as many lines are in as the order had; each 
shipment is logged as JSON.",
+  "seed": true,
+  "seed_files": [
+   "orders/order-1001.json",
+   "orders/order-1002.json",
+   "orders/order-1003.json"
+  ],
+  "run_seconds": 15,
+  "log_regex": [
+   [
+    "(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*CAMEL-MUG).*$",
+    "the shipment of ORD-1001 logged with both lines (CAMEL-TSHIRT and 
CAMEL-MUG)"
+   ],
+   [
+    "(?im)^(?=.*ORD-1002)(?=.*CAMEL-MUG)(?=.*\\b3\\b).*$",
+    "the shipment of ORD-1002 logged with 3 x CAMEL-MUG"
+   ]
+  ]
+ },
+ {
+  "name": "route-filter-and-multicast",
+  "example": "route/filter-and-multicast",
+  "level": "route",
+  "prompt": "Three orders are read from the orders directory; a filter lets 
only the paid ones through and a multicast sends each paid order to both the 
warehouse route and the invoicing route, which log their part; the pending 
order is logged as received and goes no further",
+  "expect": "Three orders are read from the orders directory; a filter lets 
only the paid ones through and a multicast sends each paid order to both the 
warehouse route and the invoicing route, which log their part; the pending 
order is logged as received and goes no further.",
+  "seed": true,
+  "seed_files": [
+   "orders/order-1001.json",
+   "orders/order-1002.json",
+   "orders/order-1003.json"
+  ],
+  "run_seconds": 10,
+  "log_regex": [
+   [
+    "(?im)^(?=.*ORD-1001)(?=.*(warehouse|pick)).*$",
+    "ORD-1001 logged by the warehouse route"
+   ],
+   [
+    "(?im)^(?=.*ORD-1001)(?=.*(invoic|bill)).*$",
+    "ORD-1001 logged by the invoicing route"
+   ],
+   [
+    "ORD-1003",
+    "ORD-1003 logged as received"
+   ]
+  ],
+  "log_not_regex": [
+   "(?im)^(?=.*ORD-1003)(?=.*(warehouse|invoic|pick|bill)).*$"
+  ]
+ },
+ {
+  "name": "fail-well-error-handling",
+  "example": "fail-well/error-handling",
+  "level": "fail-well",
+  "prompt": "The three orders go to a payment provider: the first is charged 
at once, the second gets no answer twice and is charged on the third attempt 
after two retries logged as warnings, and the third is declined, logged as such 
and parked as a file for manual review",
+  "expect": "The three orders go to a payment provider: the first is charged 
at once, the second gets no answer twice and is charged on the third attempt 
after two retries logged as warnings, and the third is declined, logged as such 
and parked as a file for manual review.",
+  "seed": true,
+  "seed_files": [
+   "orders/order-1001.json",
+   "orders/order-1002.json",
+   "orders/order-1003.json"
+  ],
+  "run_seconds": 20,
+  "expected_errors": "(?i)payment|did not 
answer|declined|ConnectException|Failed delivery",
+  "log_regex": [
+   [
+    "(?im)^(?=.*ORD-1001)(?=.*charged).*$",
+    "ORD-1001 logged as charged"
+   ],
+   [
+    "(?im)^(?=.*ORD-1002)(?=.*charged).*$",
+    "ORD-1002 logged as charged"
+   ],
+   [
+    "(?im)^(?=.*ORD-1003)(?=.*(declin|park|manual|review)).*$",
+    "ORD-1003 logged as declined and parked"
+   ],
+   [
+    "(?s)(WARN|retr|redeliver|attempt).*(WARN|retr|redeliver|attempt)",
+    "two retries logged (as warnings)"
+   ]
+  ]
+ },
+ {
+  "name": "fail-well-circuit-breaker",
+  "example": "fail-well/circuit-breaker",
+  "level": "fail-well",
+  "prompt": "A stock check calls the supplier every second; the supplier goes 
down for nine calls, the breaker opens after two failures in its window of 
four, answers from the fallback while open, tries the supplier again after five 
seconds and closes once a call succeeds; each line logs the breaker state",
+  "expect": "A stock check calls the supplier every second; the supplier goes 
down for nine calls, the breaker opens after two failures in its window of 
four, answers from the fallback while open, tries the supplier again after five 
seconds and closes once a call succeeds; each line logs the breaker state.",
+  "run_seconds": 25,
+  "expected_errors": "(?i)supplier|simulat|down",
+  "log_regex": [
+   [
+    "(?i)\\bOPEN\\b",
+    "the breaker logged as OPEN"
+   ],
+   [
+    "(?i)\\bCLOSED\\b",
+    "the breaker logged as CLOSED"
+   ],
+   [
+    "(?i)(fallback|last known|no answer)",
+    "the fallback answer logged while open"
+   ]
+  ]
+ },
+ {
+  "name": "connect-file-processing",
+  "example": "connect/file-processing",
+  "level": "connect",
+  "prompt": "A courier route copies five files into an inbox; the invoices are 
checked, archived under a month directory and moved to done, the invoice with a 
negative amount is rejected with a warning and moved to failed, and the 
driver's note is left alone because only .json files are picked up",
+  "expect": "A courier route copies five files into an inbox; the invoices are 
checked, archived under a month directory and moved to done, the invoice with a 
negative amount is rejected with a warning and moved to failed, and the 
driver's note is left alone because only .json files are picked up.",
+  "seed": true,
+  "seed_files": [
+   "samples/invoice-2001.json",
+   "samples/invoice-2002.json",
+   "samples/invoice-2003.json",
+   "samples/invoice-2004.json",
+   "samples/note-2005.txt"
+  ],
+  "run_seconds": 12,
+  "expected_errors": "(?i)Validation|Predicate|Rollback|negative|2003",
+  "require_files": [
+   "inbox/note-2005.txt"
+  ],
+  "output_files": [
+   [
+    "**/done/*.json",
+    3
+   ],
+   [
+    "**/failed/*",
+    1
+   ]
+  ],
+  "log_regex": [
+   [
+    "INV-2001",
+    "INV-2001 logged as archived"
+   ],
+   [
+    "INV-2002",
+    "INV-2002 logged as archived"
+   ],
+   [
+    "INV-2004",
+    "INV-2004 logged as archived"
+   ],
+   [
+    "(?im)^(?=.*(2003|negative))(?=.*(reject|fail|warn|invalid)).*$",
+    "invoice 2003 (the negative amount) logged as rejected"
+   ]
+  ]
+ },
+ {
+  "name": "connect-stock-api",
+  "example": "connect/stock-api",
+  "level": "connect",
+  "prompt": "The shop's stock service on port 8080: GET /stock returns the 
stock file, GET /stock/{sku} returns one SKU as JSON and a 404 with an error 
message for an unknown SKU",
+  "expect": "The shop's stock service on port 8080: GET /stock returns the 
stock file, GET /stock/{sku} returns one SKU as JSON and a 404 with an error 
message for an unknown SKU.",
+  "seed": true,
+  "seed_files": [
+   "stock.json"
+  ],
+  "run_seconds": 15,
+  "probe": "curl -s localhost:8080/stock/CAMEL-MUG; echo; curl -s -o /dev/null 
-w '%{http_code}' localhost:8080/stock/CAMEL-SOCKS; echo; curl -s 
localhost:8080/stock | head -c 400",
+  "probe_regex": "(?s)CAMEL-MUG[^\\n]*42.*\\n404\\n.*CAMEL-TSHIRT"
+ },
+ {
+  "name": "connect-http-client",
+  "example": "connect/http-client",
+  "level": "connect",
+  "prompt": "Every line of the three orders is checked against the stock 
service over HTTP, served by the same example; the log shows each line as ok, 
or back-order when the stock is short, with the stock level from the response",
+  "expect": "Every line of the three orders is checked against the stock 
service over HTTP, served by the same example; the log shows each line as ok, 
or back-order when the stock is short, with the stock level from the response.",
+  "seed": true,
+  "seed_files": [
+   "orders/order-1001.json",
+   "orders/order-1002.json",
+   "orders/order-1003.json",
+   "stock.json"
+  ],
+  "run_seconds": 15,
+  "log_regex": [
+   [
+    "(?im)^(?=.*ORD-1003)(?=.*CAMEL-CAP)(?=.*(back|short|\\b0\\b)).*$",
+    "ORD-1003 CAMEL-CAP logged as back-order with the stock level"
+   ],
+   [
+    "(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*\\b120\\b).*$",
+    "ORD-1001 CAMEL-TSHIRT logged as ok with 120 in stock"
+   ]
+  ]
+ },
+ {
+  "name": "contracts-openapi-server",
+  "example": "contracts/openapi-server",
+  "level": "contracts",
+  "prompt": "The stock API contract first: stock-api.json is the OpenAPI 
contract, the REST DSL serves its three operations on port 8080 with request 
validation, GET /stock/{sku} answers from a file or 404, POST 
/stock/{sku}/reserve answers 200, 409 when the stock is short or 400 for a bad 
reservation, and /openapi serves the contract",
+  "expect": "The stock API contract first: stock-api.json is the OpenAPI 
contract, the REST DSL serves its three operations on port 8080 with request 
validation, GET /stock/{sku} answers from a file or 404, POST 
/stock/{sku}/reserve answers 200, 409 when the stock is short or 400 for a bad 
reservation, and /openapi serves the contract.",
+  "seed": true,
+  "seed_files": [
+   "stock-api.json",
+   "stock.json"
+  ],
+  "run_seconds": 15,
+  "probe": "curl -s localhost:8080/api/stock/CAMEL-MUG; echo; curl -s -o 
/dev/null -w '%{http_code}\\n' -X POST -H 'Content-Type: application/json' -d 
'{\"orderId\": \"ORD-1001\", \"qty\": 2}' 
localhost:8080/api/stock/CAMEL-MUG/reserve; curl -s -o /dev/null -w 
'%{http_code}\\n' -X POST -H 'Content-Type: application/json' -d '{\"orderId\": 
\"ORD-1003\", \"qty\": 1}' localhost:8080/api/stock/CAMEL-CAP/reserve; curl -s 
-o /dev/null -w '%{http_code}\\n' -X POST -H 'Content-Type: application/ [...]
+  "probe_regex": "(?s)CAMEL-MUG[^\\n]*42.*\\n200\\n409\\n400\\n.*openapi"
+ },
+ {
+  "name": "contracts-openapi-client",
+  "example": "contracts/openapi-client",
+  "level": "contracts",
+  "prompt": "The picking desk reserves stock for every order line by calling 
the stock API by contract: rest-openapi turns the operationId reserveStock into 
the HTTP call from stock-api.json; the log shows each reservation and one 409 
for the cap that is out of stock. Needs the openapi-server example running",
+  "expect": "The picking desk reserves stock for every order line by calling 
the stock API by contract: rest-openapi turns the operationId reserveStock into 
the HTTP call from stock-api.json; the log shows each reservation and one 409 
for the cap that is out of stock. Needs the openapi-server example running.",
+  "seed": true,
+  "seed_files": [
+   "orders/order-1001.json",
+   "orders/order-1002.json",
+   "orders/order-1003.json",
+   "stock-api.json"
+  ],
+  "run_seconds": 20,
+  "expected_errors": "409|CAMEL-CAP|HttpOperationFailed",
+  "hint": "The stock API server (the openapi-server example) is already 
running at http://localhost:8080/api and its contract is the file 
stock-api.json.",
+  "pre": "./ref-server.sh start",
+  "post": "./ref-server.sh stop",
+  "log_regex": [
+   [
+    "(?im)^(?=.*CAMEL-CAP)(?=.*409).*$",
+    "the 409 for CAMEL-CAP logged"
+   ],
+   [
+    "(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*reserv).*$",
+    "the reservation of CAMEL-TSHIRT for ORD-1001 logged"
+   ]
+  ]
+ }
+]
\ No newline at end of file
diff --git a/ai-benchmark/gen_ladder.py b/ai-benchmark/gen_ladder.py
new file mode 100755
index 0000000..8101fd7
--- /dev/null
+++ b/ai-benchmark/gen_ladder.py
@@ -0,0 +1,132 @@
+#!/usr/bin/env python3
+"""Builds the round-2 ladder set from the examples catalog: 
examples-ladder.json and seed/<name>/.
+
+    gen_ladder.py [path to camel-jbang-examples]      default 
~/workspace/camel-jbang-examples
+
+The prompt is the catalog description (written as observable behaviour), the 
seeds are the data files the description
+names (orders, inbox, order.json, stock.json, the contract, the stylesheet, 
the Groovy mapping): the model writes the
+routes, the beans and the properties. CHECKS below is the hand-written part: 
what the log and the folder must show for
+the description to count as met; the reference example must pass every check 
(ref_pass.py) before a model sees the set.
+Docker-free rungs only (run, transform, route, fail-well, connect, contracts); 
quick-start was round 1, showcase, ai
+and cloud are out.
+"""
+import json, os, shutil, sys
+
+REPO = os.path.expanduser(sys.argv[1] if len(sys.argv) > 1 else 
"~/workspace/camel-jbang-examples")
+HERE = os.path.dirname(os.path.abspath(__file__))
+LEVELS = ["run", "transform", "route", "fail-well", "connect", "contracts"]
+# files never seeded: the model writes routes, properties and beans; 
README/metadata/tests are not part of the app
+NO_SEED = {"README.md", "metadata.json", "test", "parked"}
+NO_SEED_EXT = (".yaml", ".properties", ".java")
+ORD = r"ORD-\d{4}"
+NL = r"[^\n]*"
+
+
+def L(*terms):
+    """regex for one log line that carries every term, in any order 
(case-insensitive)"""
+    return "(?im)^" + "".join(f"(?=.*{t})" for t in terms) + ".*$"
+
+CHECKS = {
+    "run/order-generator": dict(run_seconds=12, require_files=["*.java"],
+        log_regex=[['(?s)\\{[^\\n]*"[^\\n]*\\}.*\\n.*\\{[^\\n]*"[^\\n]*\\}', 
'two orders logged as JSON']]),   # two JSON orders logged
+    "run/nightly-report": dict(run_seconds=25,
+        
log_regex=[['(?s)\\.yaml:\\d+\\s*:[^\\n]*\\d{4}-\\d{2}-\\d{2}[^\\n]*\\d+.*\\n.*\\.yaml:\\d+\\s*:[^\\n]*\\d{4}-\\d{2}-\\d{2}',
 'two report runs logged, each with a timestamp and the stock counts']]),
+    "run/properties-and-profiles": dict(run_seconds=8, 
require_files=["application-prod.properties"],
+        log_regex=[['\\.yaml:\\d+\\s*:[^\\n]*\\b(EUR|USD|DKK|GBP)\\b', 'the 
welcome logged with the currency from application.properties']]),
+    "transform/json-transform": dict(run_seconds=10,
+        log_regex=[['ORD-1001', 'the order ORD-1001 logged'], 
['"sku"\\s*:\\s*"CAMEL-TSHIRT"', 'the pick list logged as JSON with the sku 
CAMEL-TSHIRT']]),
+    "transform/xml-to-json": dict(run_seconds=10,
+        log_regex=[['<order', 'the XML order logged as it was read (the <order 
element)'], ['"CAMEL-TSHIRT"', 'the same order logged as JSON (CAMEL-TSHIRT as 
a JSON string)']]),
+    "transform/csv-to-json": dict(run_seconds=10, 
output_files=[["outbox/*.json", 3]],
+        log_regex=[['INV-2001', 'invoice INV-2001 logged'], ['INV-2003', 
'invoice INV-2003 logged'], ['"[a-zA-Z]+"\\s*:\\s*"INV-200\\d"', 'an invoice 
logged as JSON (a quoted field with the value INV-200x)']]),
+    "transform/data-mapping": dict(run_seconds=10,
+        log_regex=[['SHIP-1001', 'the shipment SHIP-1001 logged'], 
['"totalPieces"\\s*:\\s*3', 'totalPieces 3 in the logged shipment JSON'], 
['domestic', 'the domestic service in the logged shipment']]),
+    "transform/groovy": dict(run_seconds=10, 
require_text={"application.properties": "commons-validator"},
+        log_regex=[['anna@example\\.com', '[email protected] logged as 
accepted'], ['not-an-address', 'not-an-address logged as rejected']]),
+    "transform/xslt": dict(run_seconds=10,
+        log_regex=[['<packingSlip', 'the packing slip logged (a <packingSlip 
element)'], ['<pieces>3</pieces>', '<pieces>3</pieces> in the logged slip'], 
['CAMEL-TSHIRT', 'CAMEL-TSHIRT in the logged slip']]),
+    "route/content-based-router": dict(run_seconds=10,
+        log_regex=[['(?im)^(?=.*ORD-1001)(?=.*(local|copenhagen)).*$', 
'ORD-1001 logged as local delivery'], ['(?im)^(?=.*ORD-1002)(?=.*\\bEU\\b).*$', 
'ORD-1002 logged as EU shipping'], 
['(?im)^(?=.*ORD-1003)(?=.*(export|customs)).*$', 'ORD-1003 logged as export 
with customs']]),
+    "route/order-lines": dict(run_seconds=10,
+        log_regex=[['(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT).*$', 'a pick line 
for CAMEL-TSHIRT of ORD-1001'], ['(?im)^(?=.*ORD-1001)(?=.*CAMEL-MUG).*$', 'a 
pick line for CAMEL-MUG of ORD-1001'], ['ORD-1002', 'order ORD-1002 logged'], 
['(?im)^(?=.*ORD-1001)(?=.*(all|\\b2\\b))(?=.*(line|pick)).*$', 'the 
confirmation that all lines of ORD-1001 went to picking']]),
+    "route/aggregator": dict(run_seconds=15,
+        
log_regex=[['(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*CAMEL-MUG).*$', 'the 
shipment of ORD-1001 logged with both lines (CAMEL-TSHIRT and CAMEL-MUG)'], 
['(?im)^(?=.*ORD-1002)(?=.*CAMEL-MUG)(?=.*\\b3\\b).*$', 'the shipment of 
ORD-1002 logged with 3 x CAMEL-MUG']]),
+    "route/filter-and-multicast": dict(run_seconds=10,
+        log_regex=[['(?im)^(?=.*ORD-1001)(?=.*(warehouse|pick)).*$', 'ORD-1001 
logged by the warehouse route'], ['(?im)^(?=.*ORD-1001)(?=.*(invoic|bill)).*$', 
'ORD-1001 logged by the invoicing route'], ['ORD-1003', 'ORD-1003 logged as 
received']],
+        log_not_regex=[L("ORD-1003", "(warehouse|invoic|pick|bill)")]),
+    "fail-well/error-handling": dict(run_seconds=20, 
expected_errors=r"(?i)payment|did not answer|declined|ConnectException|Failed 
delivery",
+        log_regex=[['(?im)^(?=.*ORD-1001)(?=.*charged).*$', 'ORD-1001 logged 
as charged'], ['(?im)^(?=.*ORD-1002)(?=.*charged).*$', 'ORD-1002 logged as 
charged'], ['(?im)^(?=.*ORD-1003)(?=.*(declin|park|manual|review)).*$', 
'ORD-1003 logged as declined and parked'], 
['(?s)(WARN|retr|redeliver|attempt).*(WARN|retr|redeliver|attempt)', 'two 
retries logged (as warnings)']]),
+    "fail-well/circuit-breaker": dict(run_seconds=25, 
expected_errors=r"(?i)supplier|simulat|down",
+        log_regex=[['(?i)\\bOPEN\\b', 'the breaker logged as OPEN'], 
['(?i)\\bCLOSED\\b', 'the breaker logged as CLOSED'], ['(?i)(fallback|last 
known|no answer)', 'the fallback answer logged while open']]),
+    "connect/file-processing": dict(run_seconds=12, 
expected_errors=r"(?i)Validation|Predicate|Rollback|negative|2003",
+        require_files=["inbox/note-2005.txt"], 
output_files=[["**/done/*.json", 3], ["**/failed/*", 1]],
+        log_regex=[['INV-2001', 'INV-2001 logged as archived'], ['INV-2002', 
'INV-2002 logged as archived'], ['INV-2004', 'INV-2004 logged as archived'], 
['(?im)^(?=.*(2003|negative))(?=.*(reject|fail|warn|invalid)).*$', 'invoice 
2003 (the negative amount) logged as rejected']]),
+    "connect/stock-api": dict(run_seconds=15,
+        probe="curl -s localhost:8080/stock/CAMEL-MUG; echo; curl -s -o 
/dev/null -w '%{http_code}' localhost:8080/stock/CAMEL-SOCKS; echo; curl -s 
localhost:8080/stock | head -c 400",
+        probe_regex=r"(?s)CAMEL-MUG" + NL + r"42.*\n404\n.*CAMEL-TSHIRT"),
+    "connect/http-client": dict(run_seconds=15,
+        
log_regex=[['(?im)^(?=.*ORD-1003)(?=.*CAMEL-CAP)(?=.*(back|short|\\b0\\b)).*$', 
'ORD-1003 CAMEL-CAP logged as back-order with the stock level'], 
['(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*\\b120\\b).*$', 'ORD-1001 
CAMEL-TSHIRT logged as ok with 120 in stock']]),
+    "contracts/openapi-server": dict(run_seconds=15,
+        probe="curl -s localhost:8080/api/stock/CAMEL-MUG; echo; "
+              "curl -s -o /dev/null -w '%{http_code}\\n' -X POST -H 
'Content-Type: application/json' -d '{\"orderId\": \"ORD-1001\", \"qty\": 2}' 
localhost:8080/api/stock/CAMEL-MUG/reserve; "
+              "curl -s -o /dev/null -w '%{http_code}\\n' -X POST -H 
'Content-Type: application/json' -d '{\"orderId\": \"ORD-1003\", \"qty\": 1}' 
localhost:8080/api/stock/CAMEL-CAP/reserve; "
+              "curl -s -o /dev/null -w '%{http_code}\\n' -X POST -H 
'Content-Type: application/json' localhost:8080/api/stock/CAMEL-MUG/reserve; "
+              "curl -s localhost:8080/openapi localhost:8080/api/openapi | 
head -c 300",
+        probe_regex=r"(?s)CAMEL-MUG" + NL + r"42.*\n200\n409\n400\n.*openapi"),
+    "contracts/openapi-client": dict(run_seconds=20, 
expected_errors=r"409|CAMEL-CAP|HttpOperationFailed",
+        hint="The stock API server (the openapi-server example) is already 
running at http://localhost:8080/api and its contract is the file 
stock-api.json.",
+        pre="./ref-server.sh start", post="./ref-server.sh stop",
+        log_regex=[['(?im)^(?=.*CAMEL-CAP)(?=.*409).*$', 'the 409 for 
CAMEL-CAP logged'], ['(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*reserv).*$', 
'the reservation of CAMEL-TSHIRT for ORD-1001 logged']]),
+}
+
+
+def seed_dir(src, dst):
+    """copies the data files of an example (not routes, properties, beans, 
docs, tests, outputs); returns the list"""
+    out = []
+    for root, dirs, files in os.walk(src):
+        dirs[:] = [d for d in dirs if d not in NO_SEED and not 
d.startswith(".")]
+        for f in files:
+            if f in NO_SEED or f.endswith(NO_SEED_EXT) or f.startswith("."):
+                continue
+            rel = os.path.relpath(os.path.join(root, f), src)
+            os.makedirs(os.path.dirname(os.path.join(dst, rel)), exist_ok=True)
+            shutil.copy2(os.path.join(root, f), os.path.join(dst, rel))
+            out.append(rel)
+    return sorted(out)
+
+
+def main():
+    cat = json.load(open(os.path.join(REPO, 
"camel-jbang-example-catalog.json")))
+    examples = cat["examples"] if isinstance(cat, dict) else cat
+    keep = [e for e in examples if e["level"] in LEVELS and not 
e.get("requiresDocker")]
+    keep.sort(key=lambda e: (LEVELS.index(e["level"]), e.get("order", 99)))
+    missing = [e["name"] for e in keep if e["name"] not in CHECKS]
+    if missing:
+        sys.exit("no checks for: " + ", ".join(missing))
+    seeds = os.path.join(HERE, "seed")
+    out = []
+    for e in keep:
+        name = e["name"].replace("/", "-")
+        c = dict(CHECKS[e["name"]])
+        entry = {"name": name, "example": e["name"], "level": e["level"],
+                 "prompt": e["description"].rstrip(". "),
+                 "expect": e["description"]}
+        shutil.rmtree(os.path.join(seeds, name), ignore_errors=True)
+        seeded = seed_dir(os.path.join(REPO, e["name"]), os.path.join(seeds, 
name))
+        if seeded:
+            entry["seed"] = True
+            entry["seed_files"] = seeded
+        entry.update(c)
+        out.append(entry)
+    # the reference stock API server the openapi-client example calls (started 
by ref-server.sh, port 8080)
+    ref = os.path.join(seeds, "openapi-server-ref")
+    shutil.rmtree(ref, ignore_errors=True)
+    shutil.copytree(os.path.join(REPO, "contracts/openapi-server"), ref, 
ignore=shutil.ignore_patterns("README.md", "metadata.json", "test"))
+    json.dump(out, open(os.path.join(HERE, "examples-ladder.json"), "w"), 
indent=1)
+    print(f"{len(out)} examples -> examples-ladder.json; seeds under seed/")
+    for e in out:
+        print(f"  {e['name']:<34} {e['run_seconds']:>3}s 
seed={len(e.get('seed_files', []))} checks={[k for k in e if k in 
('log_regex','log_not_regex','probe','require_files','require_text','output_files','expected_errors')]}")
+
+
+if __name__ == "__main__":
+    main()
diff --git a/ai-benchmark/gen_local.py b/ai-benchmark/gen_local.py
index 8d6edb5..8c8a282 100755
--- a/ai-benchmark/gen_local.py
+++ b/ai-benchmark/gen_local.py
@@ -63,8 +63,13 @@ def write_files(text, folder):
             f.write(text.strip() + "\n")
         return ["route.camel.yaml (fallback)"]
     for i in range(1, len(parts), 2):
-        name = os.path.basename(parts[i].strip())
+        name = os.path.basename(parts[i].strip().rstrip("/"))
         content = parts[i + 1].strip("\n") + "\n"
+        # ladder (09-19): a "=== FILE: orders/ ===" entry (a directory the 
model lists) has no file name; skip it
+        # instead of opening the attempt folder as a file (the first suite of 
the full run died on it)
+        if not name or os.path.isdir(os.path.join(folder, name)):
+            written.append(name + " (skipped, a directory)")
+            continue
         with open(os.path.join(folder, name), "w") as f:
             f.write(content)
         written.append(name)
diff --git a/ai-benchmark/passk.py b/ai-benchmark/passk.py
new file mode 100755
index 0000000..08e7bfc
--- /dev/null
+++ b/ai-benchmark/passk.py
@@ -0,0 +1,38 @@
+#!/usr/bin/env python3
+"""Consistency over k runs of the same examples: per-example passes out of k, 
pass@k and pass^k.
+
+Usage: passk.py <tag-1> <tag-2> ... (the tags of k runs made with run-suite.sh 
<tag> <k>)
+
+pass@k = share of examples that passed in at least one of the k runs (what a 
user gets with k tries).
+pass^k = share of examples that passed in every run (what a user gets every 
time).
+Both follow the definitions in the MuleSoft integration-skill benchmark post 
(2026-08) so the two can be compared.
+The examples file is BENCH_EXAMPLES (default examples.json), as for the run 
itself.
+"""
+import json, os, sys
+HERE = os.path.dirname(os.path.abspath(__file__))
+names = [e["name"] for e in json.load(open(os.path.join(HERE, 
os.environ.get("BENCH_EXAMPLES", "examples.json"))))]
+tags = sys.argv[1:]
+if len(tags) < 2:
+    print(__doc__); sys.exit(1)
+res = {}
+for t in tags:
+    for n in names:
+        p = os.path.join(HERE, "oneshot-" + t, n, "result.json")
+        res[(t, n)] = json.load(open(p)) if os.path.exists(p) else None
+k = len(tags)
+print("| example | passes of %d | first-round passes | rounds | tokens |" % k)
+print("|---|---|---|---|---|")
+any_pass = all_pass = 0
+for n in names:
+    rs = [res[(t, n)] for t in tags if res[(t, n)]]
+    ok = sum(1 for r in rs if r["ok"])
+    first = sum(1 for r in rs if r["ok"] and r["rounds"] == 1)
+    any_pass += ok > 0
+    all_pass += 1 if rs and ok == len(tags) else 0
+    print(f"| {n} | {ok} | {first} | {', '.join(str(r['rounds']) for r in rs)} 
| {', '.join(f'{r['tokens']:,}' for r in rs)} |")
+n = len(names)
+print()
+print(f"pass@{k}: {any_pass} of {n} ({100 * any_pass / n:.0f}%)")
+print(f"pass^{k}: {all_pass} of {n} ({100 * all_pass / n:.0f}%)")
+per_run = [sum(1 for m in names if res[(t, m)] and res[(t, m)]["ok"]) for t in 
tags]
+print(f"passes per run: {', '.join(map(str, per_run))} (mean {sum(per_run) / 
k:.1f} of {n})")
diff --git a/ai-benchmark/run-suite.sh b/ai-benchmark/run-suite.sh
index 77e2506..e285543 100755
--- a/ai-benchmark/run-suite.sh
+++ b/ai-benchmark/run-suite.sh
@@ -1,24 +1,38 @@
 #!/bin/zsh
-# Runs one full suite: the one-shot examples, then the stepwise edits. Usage: 
run-suite.sh <tag>
+# Runs one full suite: the one-shot examples, then the stepwise edits. Usage: 
run-suite.sh <tag> [k]
+# With k > 1 the suite runs k times as <tag>-1 .. <tag>-k and passk.py reports 
pass@k and pass^k at the end.
 # Results: oneshot-<tag>/, stepwise/<tag>/, <tag>.log (wall clock). Summarise 
with: summarize_runs.py <tag>
+# Environment: BENCH_EXAMPLES=examples-intermediate.json selects set B 
(services started per example with
+# `camel infra`); BENCH_STEPWISE=0 skips the stepwise half (set B has no 
stepwise project).
 set -u
-TAG="${1:?usage: run-suite.sh <tag>}"
+TAG="${1:?usage: run-suite.sh <tag> [k]}"
+K="${2:-1}"
 cd "$(dirname "$0")"
 export MCP_URL="${MCP_URL:-http://127.0.0.1:9090/mcp}";
 export BENCH_VALIDATE_PROPS=1
 export BENCH_VALIDATE_SOURCE=1
+export BENCH_EXAMPLES="${BENCH_EXAMPLES:-examples.json}"
+STEPWISE="${BENCH_STEPWISE:-1}"
 # the one-shot model gets catalog lookups and validation only: no example 
catalog (that would hand it the answer), no runtime tools
 # the shared authoring tools of the catalog and validation kind; after 
CAMEL-24712 these are the only catalog tools
 export 
BENCH_TOOL_ALLOW="${BENCH_TOOL_ALLOW:-^camel_(catalog_(doc|find|sample|docs)|validate_source|component_properties|configuration_validate|error_diagnose|eval_expression)$}"
 command -v caffeinate > /dev/null && caffeinate -i -s -w $$ &   # macOS: keep 
the machine awake for the hour
-echo "[$TAG] one-shot start $(date +%T)" | tee -a "$TAG.log"
-BENCH_OUT="oneshot-$TAG" python3 agent_local.py > "oneshot-$TAG.out" 2>&1
-echo "[$TAG] one-shot done $(date +%T)" | tee -a "$TAG.log"
-# the stepwise project starts from the timer-log example every time
-git checkout -q -- stepwise-project 2>/dev/null || true
-echo "[$TAG] stepwise start $(date +%T)" | tee -a "$TAG.log"
-BENCH_TAG="$TAG" python3 agent_mcp_stepwise.py > "stepwise-$TAG.out" 2>&1
-echo "[$TAG] stepwise done $(date +%T)" | tee -a "$TAG.log"
-git checkout -q -- stepwise-project 2>/dev/null || true
-echo "[$TAG] DONE" | tee -a "$TAG.log"
-python3 summarize_runs.py "$TAG"
+tags=()
+for i in $(seq 1 "$K"); do
+  if (( K > 1 )); then T="$TAG-$i"; else T="$TAG"; fi
+  tags+=("$T")
+  echo "[$T] one-shot start $(date +%T) examples=$BENCH_EXAMPLES" | tee -a 
"$T.log"
+  BENCH_OUT="oneshot-$T" python3 agent_local.py > "oneshot-$T.out" 2>&1
+  echo "[$T] one-shot done $(date +%T)" | tee -a "$T.log"
+  if [[ "$STEPWISE" == "1" ]]; then
+    # the stepwise project starts from the timer-log example every time
+    git checkout -q -- stepwise-project 2>/dev/null || true
+    echo "[$T] stepwise start $(date +%T)" | tee -a "$T.log"
+    BENCH_TAG="$T" python3 agent_mcp_stepwise.py > "stepwise-$T.out" 2>&1
+    echo "[$T] stepwise done $(date +%T)" | tee -a "$T.log"
+    git checkout -q -- stepwise-project 2>/dev/null || true
+  fi
+  echo "[$T] DONE" | tee -a "$T.log"
+done
+python3 summarize_runs.py "${tags[@]}"
+if (( K > 1 )); then python3 passk.py "${tags[@]}"; fi
diff --git a/ai-benchmark/run_one.sh b/ai-benchmark/run_one.sh
index 1eff64c..39f3953 100755
--- a/ai-benchmark/run_one.sh
+++ b/ai-benchmark/run_one.sh
@@ -36,7 +36,13 @@ pid=$!
 ( sleep $((secs + 45)); pkill -TERM -f -- "--max-seconds=$secs 
--logging-color=false" 2>/dev/null; kill -TERM $pid 2>/dev/null ) &
 watchdog=$!
 if [[ -n "$probe" ]]; then
-  sleep 7
+  # ladder (09-19): wait for the routes to be up before probing (jbang 
resolution before the first log line took
+  # longer than the fixed 7 s in the dry run: the stock API answered 000 three 
times while the log showed it started)
+  for i in $(seq 1 $((secs > 4 ? secs - 3 : 1))); do
+    grep -q 'Routes startup\|Started route\|HttpServer started' run.log 
2>/dev/null && break
+    sleep 1
+  done
+  sleep 2
   eval "$probe" > probe.log 2>&1
 fi
 wait $pid
diff --git a/ai-benchmark/steps.json b/ai-benchmark/steps.json
index 3bfa65a..4a8f0ea 100644
--- a/ai-benchmark/steps.json
+++ b/ai-benchmark/steps.json
@@ -23,8 +23,9 @@
    "request": "Set the body to a random number between 0 and 40 instead of the 
greeting message.",
    "check": {
     "file_regex": "random",
-    "log_regex": "^\\d{1,2}$",
-    "min_log": 1
+    "log_regex": "(^|\\D)\\d{1,2}$",
+    "min_log": 1,
+    "note": "round 2: the number may follow a label (run r2-baseline logged 
'Random number: 22'); round 1 required the bare number and counted that as a 
strict failure in runs 4, 6, 20 and r2-baseline"
    },
    "reference": {
     "timer-log.camel.yaml": "- route:\n    id: timer-log\n    from:\n      
uri: timer\n      parameters:\n        timerName: tick\n        period: 
\"{{timer.period}}\"\n      steps:\n        - setBody:\n            
expression:\n              simple:\n                expression: 
\"${random(0,40)}\"\n        - log:\n            message: \"${body}\"\n",
@@ -114,4 +115,4 @@
    "reference": {}
   }
  ]
-}
+}
\ No newline at end of file

Reply via email to