This is an automated email from the ASF dual-hosted git repository. davsclaus pushed a commit to branch main in repository https://gitbox.apache.org/repos/asf/camel-performance-tests.git
commit 1b947aa79b7a237a7ba3837a4fad8ee3124c8552 Author: Claus Ibsen <[email protected]> AuthorDate: Sat Sep 26 11:23:22 2026 +0200 Round 2 of the one-shot suite: k runs, pass@k and pass^k, a held-out set and services run-suite.sh <tag> <k> runs the suite k times and ends with passk.py, which prints per-example passes out of k plus pass@k (passed at least once) and pass^k (passed every time) -- the two consistency measures of the MuleSoft integration-skill post, so the series can be compared with it. Set B (examples-intermediate.json) is six intermediate examples the model had never been tested on. An entry declares what it needs in the JSON rather than in the harness: infra for services started with camel infra run and whose connection data is appended to the prompt as a developer would read it, seed for files copied in before the model's, hint for one extra sentence, and pre/post for commands around the example. Co-Authored-By: Claude Opus 5 (1M context) <[email protected]> Claude-Session: https://claude.ai/code/session_01Bp3538HRBPMQkb5ta9xRaj --- ai-benchmark/README.md | 20 ++ ai-benchmark/agent_local.py | 250 ++++++++++++++-- ai-benchmark/examples-intermediate.json | 61 ++++ ai-benchmark/examples-ladder-10.json | 235 +++++++++++++++ ai-benchmark/examples-ladder-dry.json | 109 +++++++ ai-benchmark/examples-ladder.json | 508 ++++++++++++++++++++++++++++++++ ai-benchmark/gen_ladder.py | 132 +++++++++ ai-benchmark/gen_local.py | 7 +- ai-benchmark/passk.py | 38 +++ ai-benchmark/run-suite.sh | 40 ++- ai-benchmark/run_one.sh | 8 +- ai-benchmark/steps.json | 7 +- 12 files changed, 1380 insertions(+), 35 deletions(-) diff --git a/ai-benchmark/README.md b/ai-benchmark/README.md index bbf4207..b65287c 100644 --- a/ai-benchmark/README.md +++ b/ai-benchmark/README.md @@ -107,3 +107,23 @@ were found (about 15 minutes of reading per run). those, and `run-suite.sh` runs `caffeinate` on macOS. - Keep the examples away from the model: never offer `camel_catalog_examples` or `camel_catalog_example_file` in `BENCH_TOOL_ALLOW` for a benchmark that uses the examples repository. + +## Round 2: k runs, a held-out set, services + +Added 2026-09-17 for the second series. + +- `run-suite.sh <tag> <k>` runs the suite k times as `<tag>-1 .. <tag>-k` and ends with `passk.py`, which prints + per-example passes out of k, **pass@k** (passed at least once) and **pass^k** (passed every time), the two + consistency measures of the MuleSoft integration-skill post so the series can be compared with it. +- `BENCH_EXAMPLES=examples-intermediate.json BENCH_STEPWISE=0 run-suite.sh b 3` runs set B: six intermediate examples + the model has never been tested on (openapi-server, openapi-client, sql, artemis, mqtt, route-topology). + Each entry may declare, all visible in the JSON rather than hidden in the harness: + - `infra`: services started with `camel infra run <svc> --background` before the example and stopped after it + (postgres, artemis, mosquitto, kafka); the connection data from `camel infra get <svc> --json` is appended + to the prompt, as a developer would read it from the same command. Postgres needs about 80 s to come up. + - `seed`: files under `seed/<example>/` copied into every attempt folder before the model's files (the petstore + OpenAPI spec and sample payloads); the prompt lists them and says not to rewrite them. + - `hint`: one extra sentence in the prompt (the MQTT topic, the petstore base path). + - `pre` / `post`: shell commands run in this directory around the example (openapi-client starts the reference + petstore server from `seed/openapi-server-ref/` and stops it after). +- Docker Desktop must be running for `infra`; `camel infra` pulls the images on first use. diff --git a/ai-benchmark/agent_local.py b/ai-benchmark/agent_local.py index 3b3a651..9de7e9a 100755 --- a/ai-benchmark/agent_local.py +++ b/ai-benchmark/agent_local.py @@ -10,7 +10,8 @@ Per example: fails we feed the validator/run errors back and loop. Everything (tool calls, timings, tokens) is logged to <BENCH_OUT>/<name>/trace.jsonl. """ -import json, os, re, subprocess, sys, time, urllib.request +import glob +import json, os, re, shutil, subprocess, sys, time, urllib.request sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) from mcp_client import McpClient # noqa: E402 from gen_local import write_files, strip_think # noqa: E402 @@ -19,6 +20,8 @@ MODEL = os.environ.get("BENCH_MODEL", "qwen3.6:35b-a3b") HOST = os.environ.get("OLLAMA_HOST", "http://localhost:11434") HERE = os.path.dirname(os.path.abspath(__file__)) OUT = os.path.join(HERE, os.environ.get("BENCH_OUT", "oneshot")) +EXAMPLES_FILE = os.environ.get("BENCH_EXAMPLES", "examples.json") # examples-intermediate.json for set B +INFRA_TIMEOUT = int(os.environ.get("BENCH_INFRA_TIMEOUT", "300")) # seconds to wait for `camel infra` services MAX_ROUNDS = int(os.environ.get("BENCH_ROUNDS", "3")) MAX_TOOL_CALLS = int(os.environ.get("BENCH_TOOL_CALLS", "10")) TOOL_RESULT_CAP = 6000 @@ -48,8 +51,13 @@ Route files use the extension .camel.yaml. Add application.properties, Java bean def ollama_chat(messages, tools): - body = json.dumps({"model": MODEL, "messages": messages, "tools": tools, "stream": False, - "options": {"temperature": 0.2, "num_ctx": 32768}}).encode() + # ladder runs (2026-09-19): thinking off, as the Camel CLI's own Ollama client sends ("think": false); the dry run + # with thinking on spent 460-500 s and 27-29k tokens per spiral and answered nothing three times in four examples. + # BENCH_THINK=1 restores the round-1 behaviour. num_predict bounds a runaway answer (default 16384). + req = {"model": MODEL, "messages": messages, "tools": tools, "stream": False, + "think": os.environ.get("BENCH_THINK", "0") == "1", + "options": {"temperature": 0.2, "num_ctx": 32768, "num_predict": int(os.environ.get("BENCH_NUM_PREDICT", "16384"))}} + body = json.dumps(req).encode() req = urllib.request.Request(HOST + "/api/chat", data=body, headers={"Content-Type": "application/json"}) t0 = time.time() with urllib.request.urlopen(req, timeout=1800) as r: @@ -69,7 +77,14 @@ def to_ollama_tools(mcp_tools): return out -def run_folder(folder, secs, probe): +def run_folder(folder, secs, probe, probe_regex=None, checks=None): + """checks (round 2, the ladder set): the entry's own checks, all visible in the JSON: + log_regex list of regexes the run log must match (the behaviour the description promises) + log_not_regex list of regexes the run log must not match (the pending order must not reach the warehouse) + expected_errors regex: error lines that are the example's own behaviour (a retried delivery, a rejected invoice) + require_files globs that must exist in the folder after the run (a Java bean, application-prod.properties) + output_files [[glob, min count], ...] files the run must have produced (outbox/*.json)""" + checks = checks or {} subprocess.run([os.path.join(HERE, "run_one.sh"), folder, str(secs), probe or ""], check=False) v = open(os.path.join(folder, "validate.log")).read() r = open(os.path.join(folder, "run.log")).read() @@ -83,7 +98,15 @@ def run_folder(folder, secs, probe): errs = [l for l in r.splitlines() if re.search(r"ERROR|Exception|Caused by|Unsupported|Unknown|Failed|No bean", l) and not (re.search(r"\.(yaml|java):\d+\s", l) and not re.search(r"Exception|Caused by|Failed delivery", l))] - activity = len(re.findall(r"\.yaml:\d+ |\.java:\d+ ", r)) > 0 or bool(p.strip()) + if checks.get("expected_errors"): + errs = [l for l in errs if not re.search(checks["expected_errors"], l)] + # round 2: an example may say what the probe must show (probe_regex); a 404 or an empty body is then not activity + probe_ok = bool(p.strip()) and (probe_regex is None or re.search(probe_regex, p) is not None) + activity = len(re.findall(r"\.yaml:\d+ |\.java:\d+ ", r)) > 0 or probe_ok + if probe_regex is not None and not probe_ok: + errs_probe = [f"probe did not show the expected result (wanted /{probe_regex}/): " + p.strip()[:300]] + else: + errs_probe = [] if not activity: # run 17 on: a log step with its own logName (priority-logger) logs under that name, not the route file; # any log line from a logger that is not Camel's own counts as route activity @@ -96,12 +119,132 @@ def run_folder(folder, secs, probe): # CAMEL-24701) hides the "Routes startup" line the count is read from if nroutes == 0 and activity: nroutes = 1 - ok = (not bad_validate) and nroutes > 0 and not errs and activity - return ok, v, "\n".join(errs[:15]), nroutes, activity + # the ladder checks: what the description promises, checked on the log and the folder; the feedback names the + # promised behaviour (the developer's words), never the regex + errs_check = [] + ran = (not bad_validate) and nroutes > 0 and not errs + # (only when something went through a route: with no route output at all the "no log output" message below, + # which names the trigger, is the precise one; dry run 4 read from an orders/ directory that did not exist, + # three times, and was told the log did not show the behaviour) + if ran and activity and checks.get("log_regex"): + # a check is [regex, promise] (the promise is the description's own words); older sets carry a bare regex + missing = [c[1] if isinstance(c, list) else None + for c in checks["log_regex"] if re.search(c[0] if isinstance(c, list) else c, r) is None] + if missing: + if all(missing): + errs_check.append("the log does not show: " + "; ".join(missing)) + else: + errs_check.append("the log does not show the expected behaviour: " + checks.get("expect", "see the request")) + for x in checks.get("log_not_regex", []): + if ran and re.search(x[0] if isinstance(x, list) else x, r): + errs_check.append("the log shows something the request rules out: " + checks.get("expect", "see the request")) + break + for g in checks.get("require_files", []): + if not glob.glob(os.path.join(folder, g)): + errs_check.append(f"the project must contain a file matching {g}") + for f, needle in checks.get("require_text", {}).items(): + path = os.path.join(folder, f) + if not (os.path.exists(path) and needle in open(path).read()): + errs_check.append(f"{f} must contain {needle}") + for g, n in checks.get("output_files", []): + found = len(glob.glob(os.path.join(folder, g), recursive=True)) + if ran and found < n: + errs_check.append(f"the run must produce at least {n} file(s) matching {g}, found {found}") + ok = (not bad_validate) and nroutes > 0 and not errs and activity and not errs_probe and not errs_check + # CAMEL-24855: the file consumer says when it created the directory it reads from; passed on as evidence when + # the run failed, since a route reading a directory the project does not have is otherwise silent + created = re.findall(r"Created starting directory: (\S+) \(it did not exist\)", r) + if created and not ok: + errs_check.append("camel run said it created the starting directory " + ", ".join(created) + + " (it did not exist): the route read an empty directory the project does not have;" + + " the given files are in the project folder itself") + return ok, v, "\n".join((errs + errs_probe + errs_check)[:15]), nroutes, activity + + +# --- set B support: services, seed files and hooks (round 2) --------------------------------------------- +# An example may declare: +# "infra": ["postgres"] services started with `camel infra run <svc> --background` before the example and +# stopped after it; their connection data (`camel infra get <svc> --json`) is handed +# to the model in the prompt, as a developer would read it from the same command +# "seed": true files under seed/<example>/ are copied into every attempt folder before the model's +# files are written (an OpenAPI spec, sample payloads); the prompt lists them +# "hint": "..." one extra sentence in the prompt (a topic name, a server URL); kept in the JSON so +# the extra information is visible, not hidden in the harness +# "pre": "cmd", "post": "cmd" shell commands run in the harness directory before and after the example +def infra_start(services): + for svc in services: + subprocess.run(["camel", "infra", "run", svc, "--background"], check=False, + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) + data, deadline = {}, time.time() + INFRA_TIMEOUT + for svc in services: + while time.time() < deadline: + out = subprocess.run(["camel", "infra", "get", svc, "--json"], capture_output=True, text=True).stdout + m = re.search(r"\{.*\}", out, re.S) + if m: + try: + data[svc] = json.loads(m.group(0)); break + except json.JSONDecodeError: + pass + time.sleep(5) + else: + data[svc] = {"error": f"{svc} did not come up within {INFRA_TIMEOUT}s"} + return data + + +def infra_stop(services): + for svc in services: + subprocess.run(["camel", "infra", "stop", svc], check=False, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) + + +def infra_prompt(data): + lines = [] + for svc, d in data.items(): + keep = {k: v for k, v in d.items() if k in ("endpointUri", "jdbcUrl", "username", "password", "host", "port", + "beanProperties", "serviceAddress", "brokerUrl", "remoteURI", + "brokers", "getBootstrapServers")} + lines.append(f"- {svc}: {json.dumps(keep or d)}") + return ("\n\nThe following external services are already running on this machine; connect to them with these " + "details (from `camel infra get`):\n" + "\n".join(lines)) if lines else "" + + +def seed_files(name, folder): + src = os.path.join(HERE, "seed", name) + if not os.path.isdir(src): + return [] + shutil.copytree(src, folder, dirs_exist_ok=True) + return sorted(os.path.relpath(os.path.join(r, f), src) for r, _, fs in os.walk(src) for f in fs) + + +SEED_SHOW_EXT = (".json", ".csv", ".xml", ".xsl", ".groovy", ".txt", ".yaml", ".properties") +SEED_SHOW_MAX = int(os.environ.get("BENCH_SEED_SHOW_MAX", "2500")) # bytes per shown file + + +def seed_contents(name, seeded): + """the content of the small text seed files, one per directory, as the developer would see them""" + src = os.path.join(HERE, "seed", name) + shown_dirs, out = set(), [] + for rel in seeded: + d = os.path.dirname(rel) + if not rel.endswith(SEED_SHOW_EXT) or d in shown_dirs: + continue + path = os.path.join(src, rel) + if os.path.getsize(path) > SEED_SHOW_MAX: + continue + shown_dirs.add(d) + siblings = [r for r in seeded if os.path.dirname(r) == d and r != rel] + note = f" (the other files in {d}/ have the same shape)" if siblings and d else "" + out.append(f"\n\n{rel}{note}:\n```\n{open(path).read().rstrip()}\n```") + return "".join(out) + + +def hook(cmd, log): + if cmd: + r = subprocess.run(cmd, shell=True, cwd=HERE, capture_output=True, text=True) + print(f"hook `{cmd}` exit={r.returncode} {r.stdout.strip()[:200]} {r.stderr.strip()[:200]}", file=log, flush=True) def main(): - examples = json.load(open(os.path.join(HERE, "examples.json"))) + examples = json.load(open(os.path.join(HERE, EXAMPLES_FILE))) only = sys.argv[1:] mcp = McpClient(); mcp.initialize(); tools = to_ollama_tools(mcp.list_tools()) log = open(os.path.join(HERE, "agent_local.log"), "a") @@ -115,10 +258,28 @@ def main(): continue os.makedirs(base, exist_ok=True) trace = open(os.path.join(base, "trace.jsonl"), "w") + hook(ex.get("pre"), log) + infra = infra_start(ex.get("infra", [])) if ex.get("infra") else {} + if infra: + trace.write(json.dumps({"infra": infra}) + "\n"); trace.flush() + seeded = seed_files(n, os.path.join(base, "seed-check")) if ex.get("seed") else [] + shutil.rmtree(os.path.join(base, "seed-check"), ignore_errors=True) + prompt = f"Create a runnable Camel CLI example: {ex['prompt']}." + if ex.get("hint"): + prompt += " " + ex["hint"] + if seeded: + prompt += ("\n\nThe project folder already contains these files; use them and do not rewrite them: " + + ", ".join(seeded) + ".") + # ladder (09-19): a developer opens the data file before writing the route; the model has no file tool, + # so the prompt shows the content of the small text seeds (one file per directory, the rest have the same + # shape). Dry run 3 guessed $.lineItems and $.items for a field called lines. + prompt += seed_contents(n, seeded) + prompt += infra_prompt(infra) messages = [{"role": "system", "content": SYSTEM}, - {"role": "user", "content": f"Create a runnable Camel CLI example: {ex['prompt']}."}] + {"role": "user", "content": prompt}] result = {"name": n, "rounds": 0, "tool_calls": 0, "ok": False, "seconds": 0, "tokens": 0} t_start = time.time() + last_call, last_out = None, "" for rnd in range(1, MAX_ROUNDS + 1): result["rounds"] = rnd calls = 0 @@ -134,6 +295,24 @@ def main(): "tool_calls": msg.get("tool_calls"), "content": (msg.get("content") or "")[:400]}) + "\n") trace.flush() messages.append({"role": "assistant", "content": msg.get("content") or "", "tool_calls": msg.get("tool_calls")}) + if msg.get("tool_calls") and calls >= MAX_TOOL_CALLS and not text: + # round 2 harness fix: the model wants another tool call but the round's budget is spent; before, + # the empty content of that message was taken as the answer and the round failed on "the file has + # no YAML" (seen on openapi-server). Tell it the budget is spent and let it answer without tools. + messages.append({"role": "user", "content": + f"You have used the {MAX_TOOL_CALLS} tool calls allowed in this round. Answer now with the " + "complete files in the '=== FILE: <name> ===' format, without further tool calls."}) + try: + data, secs = ollama_chat(messages, []) + except Exception as e: + trace.write(json.dumps({"round": rnd, "error": str(e)}) + "\n"); break + msg = data["message"]; result["tokens"] += data.get("eval_count", 0) + trace.write(json.dumps({"round": rnd, "secs": round(secs, 1), "eval": data.get("eval_count"), + "budget_spent": True, "content": (msg.get("content") or "")[:400]}) + "\n") + trace.flush() + messages.append({"role": "assistant", "content": msg.get("content") or ""}) + text = msg.get("content") or "" + break if msg.get("tool_calls") and calls < MAX_TOOL_CALLS: for tc in msg["tool_calls"]: fn = tc["function"]; calls += 1; result["tool_calls"] += 1 @@ -143,10 +322,19 @@ def main(): args = json.loads(args) except Exception: args = {} - try: - out = mcp.call(fn["name"], args) - except Exception as e: - out = "ERROR: " + str(e) + # ladder (09-19): the same call as the previous one (name and arguments) is answered from + # the previous result with a note, so a model that repeats a validation of unchanged content + # (error-handling repeated one ten times in the dry run) is told instead of charged a round trip + key = (fn["name"], json.dumps(args, sort_keys=True)) + if key == last_call: + out = ("This is the same call as your previous one, with the same arguments, so the answer is " + "unchanged. Change the content before validating again.\n\n" + last_out) + else: + try: + out = mcp.call(fn["name"], args) + except Exception as e: + out = "ERROR: " + str(e) + last_call, last_out = key, out out = out[:TOOL_RESULT_CAP] trace.write(json.dumps({"round": rnd, "tool": fn["name"], "args": args, "result": out[:600]}) + "\n") trace.flush() @@ -154,29 +342,57 @@ def main(): continue text = msg.get("content") or "" break + if not text.strip(): + # an empty answer (a spiral that ran out, a tool budget hit): say so, nothing to save or run + print(f"{n}: round{rnd} calls={calls} empty answer", file=log, flush=True) + messages.append({"role": "user", "content": "Your answer was empty: it contained no files. Output the complete files in the '=== FILE: <name> ===' format."}) + continue folder = os.path.join(base, f"attempt{rnd}") + os.makedirs(folder, exist_ok=True) + if ex.get("seed"): + seed_files(n, folder) files = write_files(text, folder) with open(os.path.join(folder, "raw.txt"), "w") as f: f.write(text) - ok, v, errs, nroutes, activity = run_folder(folder, ex["run_seconds"], ex.get("probe")) + # after the series (09-19): the given files stay given. The model rewrote the seeded stylesheet in all + # three xslt attempts of every suite (with a wrong XSL namespace) although the prompt says not to; the + # seeds are restored after the model's files are written and the feedback says which were restored + restored = [] + if ex.get("seed"): + src = os.path.join(HERE, "seed", n) + for rel in ex.get("seed_files", []): + dst = os.path.join(folder, rel) + if os.path.basename(rel) in files: + shutil.copy2(os.path.join(src, rel), dst) + restored.append(rel) + if restored: + print(f"{n}: round{rnd} restored seed files rewritten by the model: {restored}", file=log, flush=True) + ok, v, errs, nroutes, activity = run_folder(folder, ex["run_seconds"], ex.get("probe"), ex.get("probe_regex"), ex) print(f"{n}: round{rnd} calls={calls} files={files} ok={ok} routes={nroutes} activity={activity}", file=log, flush=True) if ok: result["ok"] = True break fb = "I saved and ran your files with the Camel CLI.\n\n`camel validate yaml` output:\n" + (v.strip() or "(passed)") + if restored: + fb = ("You rewrote " + ", ".join(restored) + ", which the project already provides; the original was kept " + "and your version discarded. Use the given file as it is.\n\n" + fb) if errs: fb += "\n\nErrors from `camel run`:\n" + errs if nroutes == 0 and not errs: fb += "\n\nThe application started but loaded 0 routes." - if nroutes > 0 and not errs and not activity: - fb += (f"\n\nThe route loaded but produced no log output in {ex['run_seconds']} seconds; it must produce " - "output on its own (timer trigger, or create the input files it reads).") + if nroutes > 0 and not activity and "Failed" not in errs and "Exception" not in errs: + fb += (f"\n\nThe route loaded but produced no log output in {ex['run_seconds']} seconds: no message went " + "through any route. It must produce output on its own (a timer trigger, or a file consumer on a " + "directory that contains the input files named in the request).") fb += "\n\nUse the tools to check the options you are unsure about and validate the YAML, then output the complete corrected files again in the same '=== FILE: <name> ===' format." messages.append({"role": "user", "content": fb}) result["seconds"] = round(time.time() - t_start, 1) json.dump(result, open(os.path.join(base, "result.json"), "w")) json.dump(messages, open(os.path.join(base, "messages.json"), "w")) trace.close() + if ex.get("infra"): + infra_stop(ex["infra"]) + hook(ex.get("post"), log) print(f"{n}: DONE ok={result['ok']} rounds={result['rounds']} tool_calls={result['tool_calls']} secs={result['seconds']} tokens={result['tokens']}", file=log, flush=True) log.close() diff --git a/ai-benchmark/examples-intermediate.json b/ai-benchmark/examples-intermediate.json new file mode 100644 index 0000000..49e9912 --- /dev/null +++ b/ai-benchmark/examples-intermediate.json @@ -0,0 +1,61 @@ +[ + { + "name": "sql", + "prompt": "Use a SQL database with Camel and Postgres", + "infra": [ + "postgres" + ], + "expect": "a table is created, a row inserted and a periodic select logs the rows from the running Postgres", + "run_seconds": 15 + }, + { + "name": "artemis", + "prompt": "Setup connection factory to a remote Apache ActiveMQ Artemis messaging broker", + "infra": [ + "artemis" + ], + "expect": "a producer sends messages to a JMS queue on the running Artemis broker and a consumer logs them", + "run_seconds": 15 + }, + { + "name": "mqtt", + "prompt": "Receive MQTT events from an external MQTT broker", + "infra": [ + "mosquitto" + ], + "hint": "Events are published to the topic `temperature` as JSON objects with a numeric `value` field, for example {\"value\": 25}.", + "expect": "a route subscribes to the temperature topic on the running broker, reads the value field and logs each event", + "run_seconds": 15, + "probe": "c=$(docker ps -q --filter ancestor=$(docker ps --format '{{.Image}}' | grep -i mosquitto | head -1)); for v in 25 15 30; do docker exec $c mosquitto_pub -h localhost -t temperature -m \"{\\\"value\\\": $v}\"; sleep 1; done; echo published", + "probe_regex": "published" + }, + { + "name": "route-topology", + "prompt": "Demonstrates inter-route topology with triggers, shared routes, and external systems", + "infra": [ + "kafka" + ], + "expect": "several routes connected through direct: and kafka: endpoints; a timer generates orders that flow through a shared validation route to Kafka and a consumer logs them", + "run_seconds": 20 + }, + { + "name": "ftp", + "prompt": "Integrate ActiveMQ messaging with an FTP server", + "infra": [ + "artemis", + "ftp" + ], + "hint": "Messages arrive on the JMS queue `cheese` and each one must be uploaded as a file to the FTP server.", + "expect": "a route consumes the cheese queue on the running Artemis broker, logs each message and writes it as a file to the running FTP server", + "run_seconds": 30, + "probe": "sleep 14; pid=$(camel ps 2>/dev/null | awk 'NR==2{print $1}'); camel cmd send $pid --endpoint=jms:cheese --body='hello ftp' --logging-color=false 2>&1 | sed 's/\\x1b\\[[0-9;]*m//g' | grep -i 'sent\\|error' | head -2", + "probe_regex": "Sent \\(success\\)" + }, + { + "name": "genai-observability", + "prompt": "Observe LangChain4j LLM calls with OpenTelemetry gen_ai spans and Micrometer metrics", + "hint": "Ollama is running at http://localhost:11434 with the model `qwen3.6:35b-a3b`; the Camel CLI needs the dependencies camel-langchain4j-chat, camel-ai-observability and langchain4j-ollama, declared with camel.jbang.dependencies in application.properties.", + "expect": "a timer route sends a prompt to the local Ollama through langchain4j-chat and logs the reply and the model name; AI observability is enabled", + "run_seconds": 45 + } +] \ No newline at end of file diff --git a/ai-benchmark/examples-ladder-10.json b/ai-benchmark/examples-ladder-10.json new file mode 100644 index 0000000..8f5db21 --- /dev/null +++ b/ai-benchmark/examples-ladder-10.json @@ -0,0 +1,235 @@ +[ + { + "name": "run-order-generator", + "example": "run/order-generator", + "level": "run", + "prompt": "A timer creates a shop order every five seconds: a Java bean hands out the order number, the body is the order as JSON, and the log shows each new order. The order feed every later example starts from", + "expect": "A timer creates a shop order every five seconds: a Java bean hands out the order number, the body is the order as JSON, and the log shows each new order. The order feed every later example starts from.", + "run_seconds": 12, + "require_files": [ + "*.java" + ], + "log_regex": [ + [ + "(?s)\\{[^\\n]*\"[^\\n]*\\}.*\\n.*\\{[^\\n]*\"[^\\n]*\\}", + "two orders logged as JSON" + ] + ] + }, + { + "name": "run-nightly-report", + "example": "run/nightly-report", + "level": "run", + "prompt": "A cron schedule runs the shop's inventory report, every ten seconds in the demo and nightly with a one-line change, and each run logs the stock counts with a timestamp", + "expect": "A cron schedule runs the shop's inventory report, every ten seconds in the demo and nightly with a one-line change, and each run logs the stock counts with a timestamp.", + "run_seconds": 25, + "log_regex": [ + [ + "(?s)\\.yaml:\\d+\\s*:[^\\n]*\\d{4}-\\d{2}-\\d{2}[^\\n]*\\d+.*\\n.*\\.yaml:\\d+\\s*:[^\\n]*\\d{4}-\\d{2}-\\d{2}", + "two report runs logged, each with a timestamp and the stock counts" + ] + ] + }, + { + "name": "run-properties-and-profiles", + "example": "run/properties-and-profiles", + "level": "run", + "prompt": "A timer logs a welcome with the shop name and currency from application.properties; run with --profile=prod and application-prod.properties overrides both, so the same route greets with the production values", + "expect": "A timer logs a welcome with the shop name and currency from application.properties; run with --profile=prod and application-prod.properties overrides both, so the same route greets with the production values.", + "run_seconds": 8, + "require_files": [ + "application-prod.properties" + ], + "log_regex": [ + [ + "\\.yaml:\\d+\\s*:[^\\n]*\\b(EUR|USD|DKK|GBP)\\b", + "the welcome logged with the currency from application.properties" + ] + ] + }, + { + "name": "transform-json-transform", + "example": "transform/json-transform", + "level": "transform", + "prompt": "The shop's order in order.json is reshaped for the warehouse: jsonpath reads the order id and the number of lines into headers, jq builds the pick list with only sku and quantity per line, and both the order and the pick list are logged", + "expect": "The shop's order in order.json is reshaped for the warehouse: jsonpath reads the order id and the number of lines into headers, jq builds the pick list with only sku and quantity per line, and both the order and the pick list are logged.", + "seed": true, + "seed_files": [ + "order.json" + ], + "run_seconds": 10, + "log_regex": [ + [ + "ORD-1001", + "the order ORD-1001 logged" + ], + [ + "\"sku\"\\s*:\\s*\"CAMEL-TSHIRT\"", + "the pick list logged as JSON with the sku CAMEL-TSHIRT" + ] + ] + }, + { + "name": "transform-xml-to-json", + "example": "transform/xml-to-json", + "level": "transform", + "prompt": "A supplier's XML order dropped in the inbox directory is read with the Jackson XML data format and written out as the shop's JSON with the Jackson JSON data format, both logged; no mapping code, the XML elements and attributes become JSON fields", + "expect": "A supplier's XML order dropped in the inbox directory is read with the Jackson XML data format and written out as the shop's JSON with the Jackson JSON data format, both logged; no mapping code, the XML elements and attributes become JSON fields.", + "seed": true, + "seed_files": [ + "inbox/supplier-order.xml" + ], + "run_seconds": 10, + "log_regex": [ + [ + "<order", + "the XML order logged as it was read (the <order element)" + ], + [ + "\"CAMEL-TSHIRT\"", + "the same order logged as JSON (CAMEL-TSHIRT as a JSON string)" + ] + ] + }, + { + "name": "transform-xslt", + "example": "transform/xslt", + "level": "transform", + "prompt": "A supplier's XML order dropped in the inbox directory is transformed by the stylesheet packing-slip.xsl into the packing slip the warehouse prints, one item per line and the total pieces to pick, and the slip is logged", + "expect": "A supplier's XML order dropped in the inbox directory is transformed by the stylesheet packing-slip.xsl into the packing slip the warehouse prints, one item per line and the total pieces to pick, and the slip is logged.", + "seed": true, + "seed_files": [ + "inbox/supplier-order.xml", + "packing-slip.xsl" + ], + "run_seconds": 10, + "log_regex": [ + [ + "<packingSlip", + "the packing slip logged (a <packingSlip element)" + ], + [ + "<pieces>3</pieces>", + "<pieces>3</pieces> in the logged slip" + ], + [ + "CAMEL-TSHIRT", + "CAMEL-TSHIRT in the logged slip" + ] + ] + }, + { + "name": "route-content-based-router", + "example": "route/content-based-router", + "level": "route", + "prompt": "Three orders from three countries are read from the orders directory and a choice routes each by its country: the Danish order to local delivery, the German order to EU shipping without customs, the US order to export with a customs declaration; each branch logs what it did", + "expect": "Three orders from three countries are read from the orders directory and a choice routes each by its country: the Danish order to local delivery, the German order to EU shipping without customs, the US order to export with a customs declaration; each branch logs what it did.", + "seed": true, + "seed_files": [ + "orders/order-1001.json", + "orders/order-1002.json", + "orders/order-1003.json" + ], + "run_seconds": 10, + "log_regex": [ + [ + "(?im)^(?=.*ORD-1001)(?=.*(local|copenhagen)).*$", + "ORD-1001 logged as local delivery" + ], + [ + "(?im)^(?=.*ORD-1002)(?=.*\\bEU\\b).*$", + "ORD-1002 logged as EU shipping" + ], + [ + "(?im)^(?=.*ORD-1003)(?=.*(export|customs)).*$", + "ORD-1003 logged as export with customs" + ] + ] + }, + { + "name": "route-order-lines", + "example": "route/order-lines", + "level": "route", + "prompt": "Each order read from the orders directory is split into one message per line, the order id travels along in a header, and the log shows the order, one pick line per line, and the parent's confirmation that all lines went to picking", + "expect": "Each order read from the orders directory is split into one message per line, the order id travels along in a header, and the log shows the order, one pick line per line, and the parent's confirmation that all lines went to picking.", + "seed": true, + "seed_files": [ + "orders/order-1001.json", + "orders/order-1002.json", + "orders/order-1003.json" + ], + "run_seconds": 10, + "log_regex": [ + [ + "(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT).*$", + "a pick line for CAMEL-TSHIRT of ORD-1001" + ], + [ + "(?im)^(?=.*ORD-1001)(?=.*CAMEL-MUG).*$", + "a pick line for CAMEL-MUG of ORD-1001" + ], + [ + "ORD-1002", + "order ORD-1002 logged" + ], + [ + "(?im)^(?=.*ORD-1001)(?=.*(all|\\b2\\b))(?=.*(line|pick)).*$", + "the confirmation that all lines of ORD-1001 went to picking" + ] + ] + }, + { + "name": "route-filter-and-multicast", + "example": "route/filter-and-multicast", + "level": "route", + "prompt": "Three orders are read from the orders directory; a filter lets only the paid ones through and a multicast sends each paid order to both the warehouse route and the invoicing route, which log their part; the pending order is logged as received and goes no further", + "expect": "Three orders are read from the orders directory; a filter lets only the paid ones through and a multicast sends each paid order to both the warehouse route and the invoicing route, which log their part; the pending order is logged as received and goes no further.", + "seed": true, + "seed_files": [ + "orders/order-1001.json", + "orders/order-1002.json", + "orders/order-1003.json" + ], + "run_seconds": 10, + "log_regex": [ + [ + "(?im)^(?=.*ORD-1001)(?=.*(warehouse|pick)).*$", + "ORD-1001 logged by the warehouse route" + ], + [ + "(?im)^(?=.*ORD-1001)(?=.*(invoic|bill)).*$", + "ORD-1001 logged by the invoicing route" + ], + [ + "ORD-1003", + "ORD-1003 logged as received" + ] + ], + "log_not_regex": [ + "(?im)^(?=.*ORD-1003)(?=.*(warehouse|invoic|pick|bill)).*$" + ] + }, + { + "name": "fail-well-circuit-breaker", + "example": "fail-well/circuit-breaker", + "level": "fail-well", + "prompt": "A stock check calls the supplier every second; the supplier goes down for nine calls, the breaker opens after two failures in its window of four, answers from the fallback while open, tries the supplier again after five seconds and closes once a call succeeds; each line logs the breaker state", + "expect": "A stock check calls the supplier every second; the supplier goes down for nine calls, the breaker opens after two failures in its window of four, answers from the fallback while open, tries the supplier again after five seconds and closes once a call succeeds; each line logs the breaker state.", + "run_seconds": 25, + "expected_errors": "(?i)supplier|simulat|down", + "log_regex": [ + [ + "(?i)\\bOPEN\\b", + "the breaker logged as OPEN" + ], + [ + "(?i)\\bCLOSED\\b", + "the breaker logged as CLOSED" + ], + [ + "(?i)(fallback|last known|no answer)", + "the fallback answer logged while open" + ] + ] + } +] \ No newline at end of file diff --git a/ai-benchmark/examples-ladder-dry.json b/ai-benchmark/examples-ladder-dry.json new file mode 100644 index 0000000..7ddfb67 --- /dev/null +++ b/ai-benchmark/examples-ladder-dry.json @@ -0,0 +1,109 @@ +[ + { + "name": "run-order-generator", + "example": "run/order-generator", + "level": "run", + "prompt": "A timer creates a shop order every five seconds: a Java bean hands out the order number, the body is the order as JSON, and the log shows each new order. The order feed every later example starts from", + "expect": "A timer creates a shop order every five seconds: a Java bean hands out the order number, the body is the order as JSON, and the log shows each new order. The order feed every later example starts from.", + "run_seconds": 12, + "require_files": [ + "*.java" + ], + "log_regex": [ + "(?s)\\{[^\\n]*\"[^\\n]*\\}.*\\n.*\\{[^\\n]*\"[^\\n]*\\}" + ] + }, + { + "name": "transform-json-transform", + "example": "transform/json-transform", + "level": "transform", + "prompt": "The shop's order in order.json is reshaped for the warehouse: jsonpath reads the order id and the number of lines into headers, jq builds the pick list with only sku and quantity per line, and both the order and the pick list are logged", + "expect": "The shop's order in order.json is reshaped for the warehouse: jsonpath reads the order id and the number of lines into headers, jq builds the pick list with only sku and quantity per line, and both the order and the pick list are logged.", + "seed": true, + "seed_files": [ + "order.json" + ], + "run_seconds": 10, + "log_regex": [ + "ORD-1001", + "\"sku\"\\s*:\\s*\"CAMEL-TSHIRT\"" + ] + }, + { + "name": "route-content-based-router", + "example": "route/content-based-router", + "level": "route", + "prompt": "Three orders from three countries are read from the orders directory and a choice routes each by its country: the Danish order to local delivery, the German order to EU shipping without customs, the US order to export with a customs declaration; each branch logs what it did", + "expect": "Three orders from three countries are read from the orders directory and a choice routes each by its country: the Danish order to local delivery, the German order to EU shipping without customs, the US order to export with a customs declaration; each branch logs what it did.", + "seed": true, + "seed_files": [ + "orders/order-1001.json", + "orders/order-1002.json", + "orders/order-1003.json" + ], + "run_seconds": 10, + "log_regex": [ + "(?im)^(?=.*ORD-1001)(?=.*(local|copenhagen)).*$", + "(?im)^(?=.*ORD-1002)(?=.*\\bEU\\b).*$", + "(?im)^(?=.*ORD-1003)(?=.*(export|customs)).*$" + ] + }, + { + "name": "fail-well-error-handling", + "example": "fail-well/error-handling", + "level": "fail-well", + "prompt": "The three orders go to a payment provider: the first is charged at once, the second gets no answer twice and is charged on the third attempt after two retries logged as warnings, and the third is declined, logged as such and parked as a file for manual review", + "expect": "The three orders go to a payment provider: the first is charged at once, the second gets no answer twice and is charged on the third attempt after two retries logged as warnings, and the third is declined, logged as such and parked as a file for manual review.", + "seed": true, + "seed_files": [ + "orders/order-1001.json", + "orders/order-1002.json", + "orders/order-1003.json" + ], + "run_seconds": 20, + "expected_errors": "(?i)payment|did not answer|declined|ConnectException|Failed delivery", + "log_regex": [ + "(?im)^(?=.*ORD-1001)(?=.*charged).*$", + "(?im)^(?=.*ORD-1002)(?=.*charged).*$", + "(?im)^(?=.*ORD-1003)(?=.*(declin|park|manual|review)).*$", + "(?s)(WARN|retr|redeliver|attempt).*(WARN|retr|redeliver|attempt)" + ] + }, + { + "name": "connect-stock-api", + "example": "connect/stock-api", + "level": "connect", + "prompt": "The shop's stock service on port 8080: GET /stock returns the stock file, GET /stock/{sku} returns one SKU as JSON and a 404 with an error message for an unknown SKU", + "expect": "The shop's stock service on port 8080: GET /stock returns the stock file, GET /stock/{sku} returns one SKU as JSON and a 404 with an error message for an unknown SKU.", + "seed": true, + "seed_files": [ + "stock.json" + ], + "run_seconds": 15, + "probe": "curl -s localhost:8080/stock/CAMEL-MUG; echo; curl -s -o /dev/null -w '%{http_code}' localhost:8080/stock/CAMEL-SOCKS; echo; curl -s localhost:8080/stock | head -c 400", + "probe_regex": "(?s)CAMEL-MUG[^\\n]*42.*\\n404\\n.*CAMEL-TSHIRT" + }, + { + "name": "contracts-openapi-client", + "example": "contracts/openapi-client", + "level": "contracts", + "prompt": "The picking desk reserves stock for every order line by calling the stock API by contract: rest-openapi turns the operationId reserveStock into the HTTP call from stock-api.json; the log shows each reservation and one 409 for the cap that is out of stock. Needs the openapi-server example running", + "expect": "The picking desk reserves stock for every order line by calling the stock API by contract: rest-openapi turns the operationId reserveStock into the HTTP call from stock-api.json; the log shows each reservation and one 409 for the cap that is out of stock. Needs the openapi-server example running.", + "seed": true, + "seed_files": [ + "orders/order-1001.json", + "orders/order-1002.json", + "orders/order-1003.json", + "stock-api.json" + ], + "run_seconds": 20, + "expected_errors": "409|CAMEL-CAP|HttpOperationFailed", + "hint": "The stock API server (the openapi-server example) is already running at http://localhost:8080/api and its contract is the file stock-api.json.", + "pre": "./ref-server.sh start", + "post": "./ref-server.sh stop", + "log_regex": [ + "(?im)^(?=.*CAMEL-CAP)(?=.*409).*$", + "(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*reserv).*$" + ] + } +] \ No newline at end of file diff --git a/ai-benchmark/examples-ladder.json b/ai-benchmark/examples-ladder.json new file mode 100644 index 0000000..facbc72 --- /dev/null +++ b/ai-benchmark/examples-ladder.json @@ -0,0 +1,508 @@ +[ + { + "name": "run-order-generator", + "example": "run/order-generator", + "level": "run", + "prompt": "A timer creates a shop order every five seconds: a Java bean hands out the order number, the body is the order as JSON, and the log shows each new order. The order feed every later example starts from", + "expect": "A timer creates a shop order every five seconds: a Java bean hands out the order number, the body is the order as JSON, and the log shows each new order. The order feed every later example starts from.", + "run_seconds": 12, + "require_files": [ + "*.java" + ], + "log_regex": [ + [ + "(?s)\\{[^\\n]*\"[^\\n]*\\}.*\\n.*\\{[^\\n]*\"[^\\n]*\\}", + "two orders logged as JSON" + ] + ] + }, + { + "name": "run-nightly-report", + "example": "run/nightly-report", + "level": "run", + "prompt": "A cron schedule runs the shop's inventory report, every ten seconds in the demo and nightly with a one-line change, and each run logs the stock counts with a timestamp", + "expect": "A cron schedule runs the shop's inventory report, every ten seconds in the demo and nightly with a one-line change, and each run logs the stock counts with a timestamp.", + "run_seconds": 25, + "log_regex": [ + [ + "(?s)\\.yaml:\\d+\\s*:[^\\n]*\\d{4}-\\d{2}-\\d{2}[^\\n]*\\d+.*\\n.*\\.yaml:\\d+\\s*:[^\\n]*\\d{4}-\\d{2}-\\d{2}", + "two report runs logged, each with a timestamp and the stock counts" + ] + ] + }, + { + "name": "run-properties-and-profiles", + "example": "run/properties-and-profiles", + "level": "run", + "prompt": "A timer logs a welcome with the shop name and currency from application.properties; run with --profile=prod and application-prod.properties overrides both, so the same route greets with the production values", + "expect": "A timer logs a welcome with the shop name and currency from application.properties; run with --profile=prod and application-prod.properties overrides both, so the same route greets with the production values.", + "run_seconds": 8, + "require_files": [ + "application-prod.properties" + ], + "log_regex": [ + [ + "\\.yaml:\\d+\\s*:[^\\n]*\\b(EUR|USD|DKK|GBP)\\b", + "the welcome logged with the currency from application.properties" + ] + ] + }, + { + "name": "transform-json-transform", + "example": "transform/json-transform", + "level": "transform", + "prompt": "The shop's order in order.json is reshaped for the warehouse: jsonpath reads the order id and the number of lines into headers, jq builds the pick list with only sku and quantity per line, and both the order and the pick list are logged", + "expect": "The shop's order in order.json is reshaped for the warehouse: jsonpath reads the order id and the number of lines into headers, jq builds the pick list with only sku and quantity per line, and both the order and the pick list are logged.", + "seed": true, + "seed_files": [ + "order.json" + ], + "run_seconds": 10, + "log_regex": [ + [ + "ORD-1001", + "the order ORD-1001 logged" + ], + [ + "\"sku\"\\s*:\\s*\"CAMEL-TSHIRT\"", + "the pick list logged as JSON with the sku CAMEL-TSHIRT" + ] + ] + }, + { + "name": "transform-xml-to-json", + "example": "transform/xml-to-json", + "level": "transform", + "prompt": "A supplier's XML order dropped in the inbox directory is read with the Jackson XML data format and written out as the shop's JSON with the Jackson JSON data format, both logged; no mapping code, the XML elements and attributes become JSON fields", + "expect": "A supplier's XML order dropped in the inbox directory is read with the Jackson XML data format and written out as the shop's JSON with the Jackson JSON data format, both logged; no mapping code, the XML elements and attributes become JSON fields.", + "seed": true, + "seed_files": [ + "inbox/supplier-order.xml" + ], + "run_seconds": 10, + "log_regex": [ + [ + "<order", + "the XML order logged as it was read (the <order element)" + ], + [ + "\"CAMEL-TSHIRT\"", + "the same order logged as JSON (CAMEL-TSHIRT as a JSON string)" + ] + ] + }, + { + "name": "transform-csv-to-json", + "example": "transform/csv-to-json", + "level": "transform", + "prompt": "A CSV of invoices dropped in the inbox directory is read with the CSV data format, its header line naming the fields, split into one message per invoice, and each invoice is logged as JSON and written to the outbox directory as its own file", + "expect": "A CSV of invoices dropped in the inbox directory is read with the CSV data format, its header line naming the fields, split into one message per invoice, and each invoice is logged as JSON and written to the outbox directory as its own file.", + "seed": true, + "seed_files": [ + "inbox/invoices.csv" + ], + "run_seconds": 10, + "output_files": [ + [ + "outbox/*.json", + 3 + ] + ], + "log_regex": [ + [ + "INV-2001", + "invoice INV-2001 logged" + ], + [ + "INV-2003", + "invoice INV-2003 logged" + ], + [ + "\"[a-zA-Z]+\"\\s*:\\s*\"INV-200\\d\"", + "an invoice logged as JSON (a quoted field with the value INV-200x)" + ] + ] + }, + { + "name": "transform-data-mapping", + "example": "transform/data-mapping", + "level": "transform", + "prompt": "The shop's order in order.json is mapped field by field to the courier's shipment format, with renamed fields, a nested recipient, one parcel per line, a computed total and a service chosen from the country; the order is parsed to a map, the script shipment-mapping.groovy builds the shipment, and it is logged as JSON", + "expect": "The shop's order in order.json is mapped field by field to the courier's shipment format, with renamed fields, a nested recipient, one parcel per line, a computed total and a service chosen from the country; the order is parsed to a map, the script shipment-mapping.groovy builds the shipment, and it is logged as JSON.", + "seed": true, + "seed_files": [ + "order.json", + "shipment-mapping.groovy" + ], + "run_seconds": 10, + "log_regex": [ + [ + "SHIP-1001", + "the shipment SHIP-1001 logged" + ], + [ + "\"totalPieces\"\\s*:\\s*3", + "totalPieces 3 in the logged shipment JSON" + ], + [ + "domestic", + "the domestic service in the logged shipment" + ] + ] + }, + { + "name": "transform-groovy", + "example": "transform/groovy", + "level": "transform", + "prompt": "Two orders come in, one with a valid customer email and one with a bad one; a Groovy expression checks the address with Apache Commons Validator, a third-party library declared in application.properties, and the log shows one order accepted and one rejected", + "expect": "Two orders come in, one with a valid customer email and one with a bad one; a Groovy expression checks the address with Apache Commons Validator, a third-party library declared in application.properties, and the log shows one order accepted and one rejected.", + "seed": true, + "seed_files": [ + "order-bad-email.json", + "order.json" + ], + "run_seconds": 10, + "require_text": { + "application.properties": "commons-validator" + }, + "log_regex": [ + [ + "anna@example\\.com", + "[email protected] logged as accepted" + ], + [ + "not-an-address", + "not-an-address logged as rejected" + ] + ] + }, + { + "name": "transform-xslt", + "example": "transform/xslt", + "level": "transform", + "prompt": "A supplier's XML order dropped in the inbox directory is transformed by the stylesheet packing-slip.xsl into the packing slip the warehouse prints, one item per line and the total pieces to pick, and the slip is logged", + "expect": "A supplier's XML order dropped in the inbox directory is transformed by the stylesheet packing-slip.xsl into the packing slip the warehouse prints, one item per line and the total pieces to pick, and the slip is logged.", + "seed": true, + "seed_files": [ + "inbox/supplier-order.xml", + "packing-slip.xsl" + ], + "run_seconds": 10, + "log_regex": [ + [ + "<packingSlip", + "the packing slip logged (a <packingSlip element)" + ], + [ + "<pieces>3</pieces>", + "<pieces>3</pieces> in the logged slip" + ], + [ + "CAMEL-TSHIRT", + "CAMEL-TSHIRT in the logged slip" + ] + ] + }, + { + "name": "route-content-based-router", + "example": "route/content-based-router", + "level": "route", + "prompt": "Three orders from three countries are read from the orders directory and a choice routes each by its country: the Danish order to local delivery, the German order to EU shipping without customs, the US order to export with a customs declaration; each branch logs what it did", + "expect": "Three orders from three countries are read from the orders directory and a choice routes each by its country: the Danish order to local delivery, the German order to EU shipping without customs, the US order to export with a customs declaration; each branch logs what it did.", + "seed": true, + "seed_files": [ + "orders/order-1001.json", + "orders/order-1002.json", + "orders/order-1003.json" + ], + "run_seconds": 10, + "log_regex": [ + [ + "(?im)^(?=.*ORD-1001)(?=.*(local|copenhagen)).*$", + "ORD-1001 logged as local delivery" + ], + [ + "(?im)^(?=.*ORD-1002)(?=.*\\bEU\\b).*$", + "ORD-1002 logged as EU shipping" + ], + [ + "(?im)^(?=.*ORD-1003)(?=.*(export|customs)).*$", + "ORD-1003 logged as export with customs" + ] + ] + }, + { + "name": "route-order-lines", + "example": "route/order-lines", + "level": "route", + "prompt": "Each order read from the orders directory is split into one message per line, the order id travels along in a header, and the log shows the order, one pick line per line, and the parent's confirmation that all lines went to picking", + "expect": "Each order read from the orders directory is split into one message per line, the order id travels along in a header, and the log shows the order, one pick line per line, and the parent's confirmation that all lines went to picking.", + "seed": true, + "seed_files": [ + "orders/order-1001.json", + "orders/order-1002.json", + "orders/order-1003.json" + ], + "run_seconds": 10, + "log_regex": [ + [ + "(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT).*$", + "a pick line for CAMEL-TSHIRT of ORD-1001" + ], + [ + "(?im)^(?=.*ORD-1001)(?=.*CAMEL-MUG).*$", + "a pick line for CAMEL-MUG of ORD-1001" + ], + [ + "ORD-1002", + "order ORD-1002 logged" + ], + [ + "(?im)^(?=.*ORD-1001)(?=.*(all|\\b2\\b))(?=.*(line|pick)).*$", + "the confirmation that all lines of ORD-1001 went to picking" + ] + ] + }, + { + "name": "route-aggregator", + "example": "route/aggregator", + "level": "route", + "prompt": "The warehouse reports each picked line on its own and the aggregator collects the lines of one order back into a shipment, correlated by the order id and complete when as many lines are in as the order had; each shipment is logged as JSON", + "expect": "The warehouse reports each picked line on its own and the aggregator collects the lines of one order back into a shipment, correlated by the order id and complete when as many lines are in as the order had; each shipment is logged as JSON.", + "seed": true, + "seed_files": [ + "orders/order-1001.json", + "orders/order-1002.json", + "orders/order-1003.json" + ], + "run_seconds": 15, + "log_regex": [ + [ + "(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*CAMEL-MUG).*$", + "the shipment of ORD-1001 logged with both lines (CAMEL-TSHIRT and CAMEL-MUG)" + ], + [ + "(?im)^(?=.*ORD-1002)(?=.*CAMEL-MUG)(?=.*\\b3\\b).*$", + "the shipment of ORD-1002 logged with 3 x CAMEL-MUG" + ] + ] + }, + { + "name": "route-filter-and-multicast", + "example": "route/filter-and-multicast", + "level": "route", + "prompt": "Three orders are read from the orders directory; a filter lets only the paid ones through and a multicast sends each paid order to both the warehouse route and the invoicing route, which log their part; the pending order is logged as received and goes no further", + "expect": "Three orders are read from the orders directory; a filter lets only the paid ones through and a multicast sends each paid order to both the warehouse route and the invoicing route, which log their part; the pending order is logged as received and goes no further.", + "seed": true, + "seed_files": [ + "orders/order-1001.json", + "orders/order-1002.json", + "orders/order-1003.json" + ], + "run_seconds": 10, + "log_regex": [ + [ + "(?im)^(?=.*ORD-1001)(?=.*(warehouse|pick)).*$", + "ORD-1001 logged by the warehouse route" + ], + [ + "(?im)^(?=.*ORD-1001)(?=.*(invoic|bill)).*$", + "ORD-1001 logged by the invoicing route" + ], + [ + "ORD-1003", + "ORD-1003 logged as received" + ] + ], + "log_not_regex": [ + "(?im)^(?=.*ORD-1003)(?=.*(warehouse|invoic|pick|bill)).*$" + ] + }, + { + "name": "fail-well-error-handling", + "example": "fail-well/error-handling", + "level": "fail-well", + "prompt": "The three orders go to a payment provider: the first is charged at once, the second gets no answer twice and is charged on the third attempt after two retries logged as warnings, and the third is declined, logged as such and parked as a file for manual review", + "expect": "The three orders go to a payment provider: the first is charged at once, the second gets no answer twice and is charged on the third attempt after two retries logged as warnings, and the third is declined, logged as such and parked as a file for manual review.", + "seed": true, + "seed_files": [ + "orders/order-1001.json", + "orders/order-1002.json", + "orders/order-1003.json" + ], + "run_seconds": 20, + "expected_errors": "(?i)payment|did not answer|declined|ConnectException|Failed delivery", + "log_regex": [ + [ + "(?im)^(?=.*ORD-1001)(?=.*charged).*$", + "ORD-1001 logged as charged" + ], + [ + "(?im)^(?=.*ORD-1002)(?=.*charged).*$", + "ORD-1002 logged as charged" + ], + [ + "(?im)^(?=.*ORD-1003)(?=.*(declin|park|manual|review)).*$", + "ORD-1003 logged as declined and parked" + ], + [ + "(?s)(WARN|retr|redeliver|attempt).*(WARN|retr|redeliver|attempt)", + "two retries logged (as warnings)" + ] + ] + }, + { + "name": "fail-well-circuit-breaker", + "example": "fail-well/circuit-breaker", + "level": "fail-well", + "prompt": "A stock check calls the supplier every second; the supplier goes down for nine calls, the breaker opens after two failures in its window of four, answers from the fallback while open, tries the supplier again after five seconds and closes once a call succeeds; each line logs the breaker state", + "expect": "A stock check calls the supplier every second; the supplier goes down for nine calls, the breaker opens after two failures in its window of four, answers from the fallback while open, tries the supplier again after five seconds and closes once a call succeeds; each line logs the breaker state.", + "run_seconds": 25, + "expected_errors": "(?i)supplier|simulat|down", + "log_regex": [ + [ + "(?i)\\bOPEN\\b", + "the breaker logged as OPEN" + ], + [ + "(?i)\\bCLOSED\\b", + "the breaker logged as CLOSED" + ], + [ + "(?i)(fallback|last known|no answer)", + "the fallback answer logged while open" + ] + ] + }, + { + "name": "connect-file-processing", + "example": "connect/file-processing", + "level": "connect", + "prompt": "A courier route copies five files into an inbox; the invoices are checked, archived under a month directory and moved to done, the invoice with a negative amount is rejected with a warning and moved to failed, and the driver's note is left alone because only .json files are picked up", + "expect": "A courier route copies five files into an inbox; the invoices are checked, archived under a month directory and moved to done, the invoice with a negative amount is rejected with a warning and moved to failed, and the driver's note is left alone because only .json files are picked up.", + "seed": true, + "seed_files": [ + "samples/invoice-2001.json", + "samples/invoice-2002.json", + "samples/invoice-2003.json", + "samples/invoice-2004.json", + "samples/note-2005.txt" + ], + "run_seconds": 12, + "expected_errors": "(?i)Validation|Predicate|Rollback|negative|2003", + "require_files": [ + "inbox/note-2005.txt" + ], + "output_files": [ + [ + "**/done/*.json", + 3 + ], + [ + "**/failed/*", + 1 + ] + ], + "log_regex": [ + [ + "INV-2001", + "INV-2001 logged as archived" + ], + [ + "INV-2002", + "INV-2002 logged as archived" + ], + [ + "INV-2004", + "INV-2004 logged as archived" + ], + [ + "(?im)^(?=.*(2003|negative))(?=.*(reject|fail|warn|invalid)).*$", + "invoice 2003 (the negative amount) logged as rejected" + ] + ] + }, + { + "name": "connect-stock-api", + "example": "connect/stock-api", + "level": "connect", + "prompt": "The shop's stock service on port 8080: GET /stock returns the stock file, GET /stock/{sku} returns one SKU as JSON and a 404 with an error message for an unknown SKU", + "expect": "The shop's stock service on port 8080: GET /stock returns the stock file, GET /stock/{sku} returns one SKU as JSON and a 404 with an error message for an unknown SKU.", + "seed": true, + "seed_files": [ + "stock.json" + ], + "run_seconds": 15, + "probe": "curl -s localhost:8080/stock/CAMEL-MUG; echo; curl -s -o /dev/null -w '%{http_code}' localhost:8080/stock/CAMEL-SOCKS; echo; curl -s localhost:8080/stock | head -c 400", + "probe_regex": "(?s)CAMEL-MUG[^\\n]*42.*\\n404\\n.*CAMEL-TSHIRT" + }, + { + "name": "connect-http-client", + "example": "connect/http-client", + "level": "connect", + "prompt": "Every line of the three orders is checked against the stock service over HTTP, served by the same example; the log shows each line as ok, or back-order when the stock is short, with the stock level from the response", + "expect": "Every line of the three orders is checked against the stock service over HTTP, served by the same example; the log shows each line as ok, or back-order when the stock is short, with the stock level from the response.", + "seed": true, + "seed_files": [ + "orders/order-1001.json", + "orders/order-1002.json", + "orders/order-1003.json", + "stock.json" + ], + "run_seconds": 15, + "log_regex": [ + [ + "(?im)^(?=.*ORD-1003)(?=.*CAMEL-CAP)(?=.*(back|short|\\b0\\b)).*$", + "ORD-1003 CAMEL-CAP logged as back-order with the stock level" + ], + [ + "(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*\\b120\\b).*$", + "ORD-1001 CAMEL-TSHIRT logged as ok with 120 in stock" + ] + ] + }, + { + "name": "contracts-openapi-server", + "example": "contracts/openapi-server", + "level": "contracts", + "prompt": "The stock API contract first: stock-api.json is the OpenAPI contract, the REST DSL serves its three operations on port 8080 with request validation, GET /stock/{sku} answers from a file or 404, POST /stock/{sku}/reserve answers 200, 409 when the stock is short or 400 for a bad reservation, and /openapi serves the contract", + "expect": "The stock API contract first: stock-api.json is the OpenAPI contract, the REST DSL serves its three operations on port 8080 with request validation, GET /stock/{sku} answers from a file or 404, POST /stock/{sku}/reserve answers 200, 409 when the stock is short or 400 for a bad reservation, and /openapi serves the contract.", + "seed": true, + "seed_files": [ + "stock-api.json", + "stock.json" + ], + "run_seconds": 15, + "probe": "curl -s localhost:8080/api/stock/CAMEL-MUG; echo; curl -s -o /dev/null -w '%{http_code}\\n' -X POST -H 'Content-Type: application/json' -d '{\"orderId\": \"ORD-1001\", \"qty\": 2}' localhost:8080/api/stock/CAMEL-MUG/reserve; curl -s -o /dev/null -w '%{http_code}\\n' -X POST -H 'Content-Type: application/json' -d '{\"orderId\": \"ORD-1003\", \"qty\": 1}' localhost:8080/api/stock/CAMEL-CAP/reserve; curl -s -o /dev/null -w '%{http_code}\\n' -X POST -H 'Content-Type: application/ [...] + "probe_regex": "(?s)CAMEL-MUG[^\\n]*42.*\\n200\\n409\\n400\\n.*openapi" + }, + { + "name": "contracts-openapi-client", + "example": "contracts/openapi-client", + "level": "contracts", + "prompt": "The picking desk reserves stock for every order line by calling the stock API by contract: rest-openapi turns the operationId reserveStock into the HTTP call from stock-api.json; the log shows each reservation and one 409 for the cap that is out of stock. Needs the openapi-server example running", + "expect": "The picking desk reserves stock for every order line by calling the stock API by contract: rest-openapi turns the operationId reserveStock into the HTTP call from stock-api.json; the log shows each reservation and one 409 for the cap that is out of stock. Needs the openapi-server example running.", + "seed": true, + "seed_files": [ + "orders/order-1001.json", + "orders/order-1002.json", + "orders/order-1003.json", + "stock-api.json" + ], + "run_seconds": 20, + "expected_errors": "409|CAMEL-CAP|HttpOperationFailed", + "hint": "The stock API server (the openapi-server example) is already running at http://localhost:8080/api and its contract is the file stock-api.json.", + "pre": "./ref-server.sh start", + "post": "./ref-server.sh stop", + "log_regex": [ + [ + "(?im)^(?=.*CAMEL-CAP)(?=.*409).*$", + "the 409 for CAMEL-CAP logged" + ], + [ + "(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*reserv).*$", + "the reservation of CAMEL-TSHIRT for ORD-1001 logged" + ] + ] + } +] \ No newline at end of file diff --git a/ai-benchmark/gen_ladder.py b/ai-benchmark/gen_ladder.py new file mode 100755 index 0000000..8101fd7 --- /dev/null +++ b/ai-benchmark/gen_ladder.py @@ -0,0 +1,132 @@ +#!/usr/bin/env python3 +"""Builds the round-2 ladder set from the examples catalog: examples-ladder.json and seed/<name>/. + + gen_ladder.py [path to camel-jbang-examples] default ~/workspace/camel-jbang-examples + +The prompt is the catalog description (written as observable behaviour), the seeds are the data files the description +names (orders, inbox, order.json, stock.json, the contract, the stylesheet, the Groovy mapping): the model writes the +routes, the beans and the properties. CHECKS below is the hand-written part: what the log and the folder must show for +the description to count as met; the reference example must pass every check (ref_pass.py) before a model sees the set. +Docker-free rungs only (run, transform, route, fail-well, connect, contracts); quick-start was round 1, showcase, ai +and cloud are out. +""" +import json, os, shutil, sys + +REPO = os.path.expanduser(sys.argv[1] if len(sys.argv) > 1 else "~/workspace/camel-jbang-examples") +HERE = os.path.dirname(os.path.abspath(__file__)) +LEVELS = ["run", "transform", "route", "fail-well", "connect", "contracts"] +# files never seeded: the model writes routes, properties and beans; README/metadata/tests are not part of the app +NO_SEED = {"README.md", "metadata.json", "test", "parked"} +NO_SEED_EXT = (".yaml", ".properties", ".java") +ORD = r"ORD-\d{4}" +NL = r"[^\n]*" + + +def L(*terms): + """regex for one log line that carries every term, in any order (case-insensitive)""" + return "(?im)^" + "".join(f"(?=.*{t})" for t in terms) + ".*$" + +CHECKS = { + "run/order-generator": dict(run_seconds=12, require_files=["*.java"], + log_regex=[['(?s)\\{[^\\n]*"[^\\n]*\\}.*\\n.*\\{[^\\n]*"[^\\n]*\\}', 'two orders logged as JSON']]), # two JSON orders logged + "run/nightly-report": dict(run_seconds=25, + log_regex=[['(?s)\\.yaml:\\d+\\s*:[^\\n]*\\d{4}-\\d{2}-\\d{2}[^\\n]*\\d+.*\\n.*\\.yaml:\\d+\\s*:[^\\n]*\\d{4}-\\d{2}-\\d{2}', 'two report runs logged, each with a timestamp and the stock counts']]), + "run/properties-and-profiles": dict(run_seconds=8, require_files=["application-prod.properties"], + log_regex=[['\\.yaml:\\d+\\s*:[^\\n]*\\b(EUR|USD|DKK|GBP)\\b', 'the welcome logged with the currency from application.properties']]), + "transform/json-transform": dict(run_seconds=10, + log_regex=[['ORD-1001', 'the order ORD-1001 logged'], ['"sku"\\s*:\\s*"CAMEL-TSHIRT"', 'the pick list logged as JSON with the sku CAMEL-TSHIRT']]), + "transform/xml-to-json": dict(run_seconds=10, + log_regex=[['<order', 'the XML order logged as it was read (the <order element)'], ['"CAMEL-TSHIRT"', 'the same order logged as JSON (CAMEL-TSHIRT as a JSON string)']]), + "transform/csv-to-json": dict(run_seconds=10, output_files=[["outbox/*.json", 3]], + log_regex=[['INV-2001', 'invoice INV-2001 logged'], ['INV-2003', 'invoice INV-2003 logged'], ['"[a-zA-Z]+"\\s*:\\s*"INV-200\\d"', 'an invoice logged as JSON (a quoted field with the value INV-200x)']]), + "transform/data-mapping": dict(run_seconds=10, + log_regex=[['SHIP-1001', 'the shipment SHIP-1001 logged'], ['"totalPieces"\\s*:\\s*3', 'totalPieces 3 in the logged shipment JSON'], ['domestic', 'the domestic service in the logged shipment']]), + "transform/groovy": dict(run_seconds=10, require_text={"application.properties": "commons-validator"}, + log_regex=[['anna@example\\.com', '[email protected] logged as accepted'], ['not-an-address', 'not-an-address logged as rejected']]), + "transform/xslt": dict(run_seconds=10, + log_regex=[['<packingSlip', 'the packing slip logged (a <packingSlip element)'], ['<pieces>3</pieces>', '<pieces>3</pieces> in the logged slip'], ['CAMEL-TSHIRT', 'CAMEL-TSHIRT in the logged slip']]), + "route/content-based-router": dict(run_seconds=10, + log_regex=[['(?im)^(?=.*ORD-1001)(?=.*(local|copenhagen)).*$', 'ORD-1001 logged as local delivery'], ['(?im)^(?=.*ORD-1002)(?=.*\\bEU\\b).*$', 'ORD-1002 logged as EU shipping'], ['(?im)^(?=.*ORD-1003)(?=.*(export|customs)).*$', 'ORD-1003 logged as export with customs']]), + "route/order-lines": dict(run_seconds=10, + log_regex=[['(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT).*$', 'a pick line for CAMEL-TSHIRT of ORD-1001'], ['(?im)^(?=.*ORD-1001)(?=.*CAMEL-MUG).*$', 'a pick line for CAMEL-MUG of ORD-1001'], ['ORD-1002', 'order ORD-1002 logged'], ['(?im)^(?=.*ORD-1001)(?=.*(all|\\b2\\b))(?=.*(line|pick)).*$', 'the confirmation that all lines of ORD-1001 went to picking']]), + "route/aggregator": dict(run_seconds=15, + log_regex=[['(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*CAMEL-MUG).*$', 'the shipment of ORD-1001 logged with both lines (CAMEL-TSHIRT and CAMEL-MUG)'], ['(?im)^(?=.*ORD-1002)(?=.*CAMEL-MUG)(?=.*\\b3\\b).*$', 'the shipment of ORD-1002 logged with 3 x CAMEL-MUG']]), + "route/filter-and-multicast": dict(run_seconds=10, + log_regex=[['(?im)^(?=.*ORD-1001)(?=.*(warehouse|pick)).*$', 'ORD-1001 logged by the warehouse route'], ['(?im)^(?=.*ORD-1001)(?=.*(invoic|bill)).*$', 'ORD-1001 logged by the invoicing route'], ['ORD-1003', 'ORD-1003 logged as received']], + log_not_regex=[L("ORD-1003", "(warehouse|invoic|pick|bill)")]), + "fail-well/error-handling": dict(run_seconds=20, expected_errors=r"(?i)payment|did not answer|declined|ConnectException|Failed delivery", + log_regex=[['(?im)^(?=.*ORD-1001)(?=.*charged).*$', 'ORD-1001 logged as charged'], ['(?im)^(?=.*ORD-1002)(?=.*charged).*$', 'ORD-1002 logged as charged'], ['(?im)^(?=.*ORD-1003)(?=.*(declin|park|manual|review)).*$', 'ORD-1003 logged as declined and parked'], ['(?s)(WARN|retr|redeliver|attempt).*(WARN|retr|redeliver|attempt)', 'two retries logged (as warnings)']]), + "fail-well/circuit-breaker": dict(run_seconds=25, expected_errors=r"(?i)supplier|simulat|down", + log_regex=[['(?i)\\bOPEN\\b', 'the breaker logged as OPEN'], ['(?i)\\bCLOSED\\b', 'the breaker logged as CLOSED'], ['(?i)(fallback|last known|no answer)', 'the fallback answer logged while open']]), + "connect/file-processing": dict(run_seconds=12, expected_errors=r"(?i)Validation|Predicate|Rollback|negative|2003", + require_files=["inbox/note-2005.txt"], output_files=[["**/done/*.json", 3], ["**/failed/*", 1]], + log_regex=[['INV-2001', 'INV-2001 logged as archived'], ['INV-2002', 'INV-2002 logged as archived'], ['INV-2004', 'INV-2004 logged as archived'], ['(?im)^(?=.*(2003|negative))(?=.*(reject|fail|warn|invalid)).*$', 'invoice 2003 (the negative amount) logged as rejected']]), + "connect/stock-api": dict(run_seconds=15, + probe="curl -s localhost:8080/stock/CAMEL-MUG; echo; curl -s -o /dev/null -w '%{http_code}' localhost:8080/stock/CAMEL-SOCKS; echo; curl -s localhost:8080/stock | head -c 400", + probe_regex=r"(?s)CAMEL-MUG" + NL + r"42.*\n404\n.*CAMEL-TSHIRT"), + "connect/http-client": dict(run_seconds=15, + log_regex=[['(?im)^(?=.*ORD-1003)(?=.*CAMEL-CAP)(?=.*(back|short|\\b0\\b)).*$', 'ORD-1003 CAMEL-CAP logged as back-order with the stock level'], ['(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*\\b120\\b).*$', 'ORD-1001 CAMEL-TSHIRT logged as ok with 120 in stock']]), + "contracts/openapi-server": dict(run_seconds=15, + probe="curl -s localhost:8080/api/stock/CAMEL-MUG; echo; " + "curl -s -o /dev/null -w '%{http_code}\\n' -X POST -H 'Content-Type: application/json' -d '{\"orderId\": \"ORD-1001\", \"qty\": 2}' localhost:8080/api/stock/CAMEL-MUG/reserve; " + "curl -s -o /dev/null -w '%{http_code}\\n' -X POST -H 'Content-Type: application/json' -d '{\"orderId\": \"ORD-1003\", \"qty\": 1}' localhost:8080/api/stock/CAMEL-CAP/reserve; " + "curl -s -o /dev/null -w '%{http_code}\\n' -X POST -H 'Content-Type: application/json' localhost:8080/api/stock/CAMEL-MUG/reserve; " + "curl -s localhost:8080/openapi localhost:8080/api/openapi | head -c 300", + probe_regex=r"(?s)CAMEL-MUG" + NL + r"42.*\n200\n409\n400\n.*openapi"), + "contracts/openapi-client": dict(run_seconds=20, expected_errors=r"409|CAMEL-CAP|HttpOperationFailed", + hint="The stock API server (the openapi-server example) is already running at http://localhost:8080/api and its contract is the file stock-api.json.", + pre="./ref-server.sh start", post="./ref-server.sh stop", + log_regex=[['(?im)^(?=.*CAMEL-CAP)(?=.*409).*$', 'the 409 for CAMEL-CAP logged'], ['(?im)^(?=.*ORD-1001)(?=.*CAMEL-TSHIRT)(?=.*reserv).*$', 'the reservation of CAMEL-TSHIRT for ORD-1001 logged']]), +} + + +def seed_dir(src, dst): + """copies the data files of an example (not routes, properties, beans, docs, tests, outputs); returns the list""" + out = [] + for root, dirs, files in os.walk(src): + dirs[:] = [d for d in dirs if d not in NO_SEED and not d.startswith(".")] + for f in files: + if f in NO_SEED or f.endswith(NO_SEED_EXT) or f.startswith("."): + continue + rel = os.path.relpath(os.path.join(root, f), src) + os.makedirs(os.path.dirname(os.path.join(dst, rel)), exist_ok=True) + shutil.copy2(os.path.join(root, f), os.path.join(dst, rel)) + out.append(rel) + return sorted(out) + + +def main(): + cat = json.load(open(os.path.join(REPO, "camel-jbang-example-catalog.json"))) + examples = cat["examples"] if isinstance(cat, dict) else cat + keep = [e for e in examples if e["level"] in LEVELS and not e.get("requiresDocker")] + keep.sort(key=lambda e: (LEVELS.index(e["level"]), e.get("order", 99))) + missing = [e["name"] for e in keep if e["name"] not in CHECKS] + if missing: + sys.exit("no checks for: " + ", ".join(missing)) + seeds = os.path.join(HERE, "seed") + out = [] + for e in keep: + name = e["name"].replace("/", "-") + c = dict(CHECKS[e["name"]]) + entry = {"name": name, "example": e["name"], "level": e["level"], + "prompt": e["description"].rstrip(". "), + "expect": e["description"]} + shutil.rmtree(os.path.join(seeds, name), ignore_errors=True) + seeded = seed_dir(os.path.join(REPO, e["name"]), os.path.join(seeds, name)) + if seeded: + entry["seed"] = True + entry["seed_files"] = seeded + entry.update(c) + out.append(entry) + # the reference stock API server the openapi-client example calls (started by ref-server.sh, port 8080) + ref = os.path.join(seeds, "openapi-server-ref") + shutil.rmtree(ref, ignore_errors=True) + shutil.copytree(os.path.join(REPO, "contracts/openapi-server"), ref, ignore=shutil.ignore_patterns("README.md", "metadata.json", "test")) + json.dump(out, open(os.path.join(HERE, "examples-ladder.json"), "w"), indent=1) + print(f"{len(out)} examples -> examples-ladder.json; seeds under seed/") + for e in out: + print(f" {e['name']:<34} {e['run_seconds']:>3}s seed={len(e.get('seed_files', []))} checks={[k for k in e if k in ('log_regex','log_not_regex','probe','require_files','require_text','output_files','expected_errors')]}") + + +if __name__ == "__main__": + main() diff --git a/ai-benchmark/gen_local.py b/ai-benchmark/gen_local.py index 8d6edb5..8c8a282 100755 --- a/ai-benchmark/gen_local.py +++ b/ai-benchmark/gen_local.py @@ -63,8 +63,13 @@ def write_files(text, folder): f.write(text.strip() + "\n") return ["route.camel.yaml (fallback)"] for i in range(1, len(parts), 2): - name = os.path.basename(parts[i].strip()) + name = os.path.basename(parts[i].strip().rstrip("/")) content = parts[i + 1].strip("\n") + "\n" + # ladder (09-19): a "=== FILE: orders/ ===" entry (a directory the model lists) has no file name; skip it + # instead of opening the attempt folder as a file (the first suite of the full run died on it) + if not name or os.path.isdir(os.path.join(folder, name)): + written.append(name + " (skipped, a directory)") + continue with open(os.path.join(folder, name), "w") as f: f.write(content) written.append(name) diff --git a/ai-benchmark/passk.py b/ai-benchmark/passk.py new file mode 100755 index 0000000..08e7bfc --- /dev/null +++ b/ai-benchmark/passk.py @@ -0,0 +1,38 @@ +#!/usr/bin/env python3 +"""Consistency over k runs of the same examples: per-example passes out of k, pass@k and pass^k. + +Usage: passk.py <tag-1> <tag-2> ... (the tags of k runs made with run-suite.sh <tag> <k>) + +pass@k = share of examples that passed in at least one of the k runs (what a user gets with k tries). +pass^k = share of examples that passed in every run (what a user gets every time). +Both follow the definitions in the MuleSoft integration-skill benchmark post (2026-08) so the two can be compared. +The examples file is BENCH_EXAMPLES (default examples.json), as for the run itself. +""" +import json, os, sys +HERE = os.path.dirname(os.path.abspath(__file__)) +names = [e["name"] for e in json.load(open(os.path.join(HERE, os.environ.get("BENCH_EXAMPLES", "examples.json"))))] +tags = sys.argv[1:] +if len(tags) < 2: + print(__doc__); sys.exit(1) +res = {} +for t in tags: + for n in names: + p = os.path.join(HERE, "oneshot-" + t, n, "result.json") + res[(t, n)] = json.load(open(p)) if os.path.exists(p) else None +k = len(tags) +print("| example | passes of %d | first-round passes | rounds | tokens |" % k) +print("|---|---|---|---|---|") +any_pass = all_pass = 0 +for n in names: + rs = [res[(t, n)] for t in tags if res[(t, n)]] + ok = sum(1 for r in rs if r["ok"]) + first = sum(1 for r in rs if r["ok"] and r["rounds"] == 1) + any_pass += ok > 0 + all_pass += 1 if rs and ok == len(tags) else 0 + print(f"| {n} | {ok} | {first} | {', '.join(str(r['rounds']) for r in rs)} | {', '.join(f'{r['tokens']:,}' for r in rs)} |") +n = len(names) +print() +print(f"pass@{k}: {any_pass} of {n} ({100 * any_pass / n:.0f}%)") +print(f"pass^{k}: {all_pass} of {n} ({100 * all_pass / n:.0f}%)") +per_run = [sum(1 for m in names if res[(t, m)] and res[(t, m)]["ok"]) for t in tags] +print(f"passes per run: {', '.join(map(str, per_run))} (mean {sum(per_run) / k:.1f} of {n})") diff --git a/ai-benchmark/run-suite.sh b/ai-benchmark/run-suite.sh index 77e2506..e285543 100755 --- a/ai-benchmark/run-suite.sh +++ b/ai-benchmark/run-suite.sh @@ -1,24 +1,38 @@ #!/bin/zsh -# Runs one full suite: the one-shot examples, then the stepwise edits. Usage: run-suite.sh <tag> +# Runs one full suite: the one-shot examples, then the stepwise edits. Usage: run-suite.sh <tag> [k] +# With k > 1 the suite runs k times as <tag>-1 .. <tag>-k and passk.py reports pass@k and pass^k at the end. # Results: oneshot-<tag>/, stepwise/<tag>/, <tag>.log (wall clock). Summarise with: summarize_runs.py <tag> +# Environment: BENCH_EXAMPLES=examples-intermediate.json selects set B (services started per example with +# `camel infra`); BENCH_STEPWISE=0 skips the stepwise half (set B has no stepwise project). set -u -TAG="${1:?usage: run-suite.sh <tag>}" +TAG="${1:?usage: run-suite.sh <tag> [k]}" +K="${2:-1}" cd "$(dirname "$0")" export MCP_URL="${MCP_URL:-http://127.0.0.1:9090/mcp}" export BENCH_VALIDATE_PROPS=1 export BENCH_VALIDATE_SOURCE=1 +export BENCH_EXAMPLES="${BENCH_EXAMPLES:-examples.json}" +STEPWISE="${BENCH_STEPWISE:-1}" # the one-shot model gets catalog lookups and validation only: no example catalog (that would hand it the answer), no runtime tools # the shared authoring tools of the catalog and validation kind; after CAMEL-24712 these are the only catalog tools export BENCH_TOOL_ALLOW="${BENCH_TOOL_ALLOW:-^camel_(catalog_(doc|find|sample|docs)|validate_source|component_properties|configuration_validate|error_diagnose|eval_expression)$}" command -v caffeinate > /dev/null && caffeinate -i -s -w $$ & # macOS: keep the machine awake for the hour -echo "[$TAG] one-shot start $(date +%T)" | tee -a "$TAG.log" -BENCH_OUT="oneshot-$TAG" python3 agent_local.py > "oneshot-$TAG.out" 2>&1 -echo "[$TAG] one-shot done $(date +%T)" | tee -a "$TAG.log" -# the stepwise project starts from the timer-log example every time -git checkout -q -- stepwise-project 2>/dev/null || true -echo "[$TAG] stepwise start $(date +%T)" | tee -a "$TAG.log" -BENCH_TAG="$TAG" python3 agent_mcp_stepwise.py > "stepwise-$TAG.out" 2>&1 -echo "[$TAG] stepwise done $(date +%T)" | tee -a "$TAG.log" -git checkout -q -- stepwise-project 2>/dev/null || true -echo "[$TAG] DONE" | tee -a "$TAG.log" -python3 summarize_runs.py "$TAG" +tags=() +for i in $(seq 1 "$K"); do + if (( K > 1 )); then T="$TAG-$i"; else T="$TAG"; fi + tags+=("$T") + echo "[$T] one-shot start $(date +%T) examples=$BENCH_EXAMPLES" | tee -a "$T.log" + BENCH_OUT="oneshot-$T" python3 agent_local.py > "oneshot-$T.out" 2>&1 + echo "[$T] one-shot done $(date +%T)" | tee -a "$T.log" + if [[ "$STEPWISE" == "1" ]]; then + # the stepwise project starts from the timer-log example every time + git checkout -q -- stepwise-project 2>/dev/null || true + echo "[$T] stepwise start $(date +%T)" | tee -a "$T.log" + BENCH_TAG="$T" python3 agent_mcp_stepwise.py > "stepwise-$T.out" 2>&1 + echo "[$T] stepwise done $(date +%T)" | tee -a "$T.log" + git checkout -q -- stepwise-project 2>/dev/null || true + fi + echo "[$T] DONE" | tee -a "$T.log" +done +python3 summarize_runs.py "${tags[@]}" +if (( K > 1 )); then python3 passk.py "${tags[@]}"; fi diff --git a/ai-benchmark/run_one.sh b/ai-benchmark/run_one.sh index 1eff64c..39f3953 100755 --- a/ai-benchmark/run_one.sh +++ b/ai-benchmark/run_one.sh @@ -36,7 +36,13 @@ pid=$! ( sleep $((secs + 45)); pkill -TERM -f -- "--max-seconds=$secs --logging-color=false" 2>/dev/null; kill -TERM $pid 2>/dev/null ) & watchdog=$! if [[ -n "$probe" ]]; then - sleep 7 + # ladder (09-19): wait for the routes to be up before probing (jbang resolution before the first log line took + # longer than the fixed 7 s in the dry run: the stock API answered 000 three times while the log showed it started) + for i in $(seq 1 $((secs > 4 ? secs - 3 : 1))); do + grep -q 'Routes startup\|Started route\|HttpServer started' run.log 2>/dev/null && break + sleep 1 + done + sleep 2 eval "$probe" > probe.log 2>&1 fi wait $pid diff --git a/ai-benchmark/steps.json b/ai-benchmark/steps.json index 3bfa65a..4a8f0ea 100644 --- a/ai-benchmark/steps.json +++ b/ai-benchmark/steps.json @@ -23,8 +23,9 @@ "request": "Set the body to a random number between 0 and 40 instead of the greeting message.", "check": { "file_regex": "random", - "log_regex": "^\\d{1,2}$", - "min_log": 1 + "log_regex": "(^|\\D)\\d{1,2}$", + "min_log": 1, + "note": "round 2: the number may follow a label (run r2-baseline logged 'Random number: 22'); round 1 required the bare number and counted that as a strict failure in runs 4, 6, 20 and r2-baseline" }, "reference": { "timer-log.camel.yaml": "- route:\n id: timer-log\n from:\n uri: timer\n parameters:\n timerName: tick\n period: \"{{timer.period}}\"\n steps:\n - setBody:\n expression:\n simple:\n expression: \"${random(0,40)}\"\n - log:\n message: \"${body}\"\n", @@ -114,4 +115,4 @@ "reference": {} } ] -} +} \ No newline at end of file
