This is an automated email from the ASF dual-hosted git repository. davsclaus pushed a commit to branch main in repository https://gitbox.apache.org/repos/asf/camel-performance-tests.git
commit a9de890be061a44656b3a0122e45d99bdf0cc11a Author: Claus Ibsen <[email protected]> AuthorDate: Mon Sep 28 09:27:16 2026 +0200 The sql rung runs on Postgres with camel infra, as the example ships it The ladder ran connect-service-sql on H2 to avoid needing a service, kept the example's route shape, and rewrote its insert into H2's standard MERGE. The example on main is Postgres and its README teaches ON CONFLICT ("run twice and see the conflict error, then add ON CONFLICT"), so the ladder was scoring the model against an answer the example does not give. Four of eighteen attempts wrote the example's own ON CONFLICT insert and were marked wrong for it; the H2-correct MERGE was never written once, and the step was 0/10 in four runs. The rung now uses the example's datasource and its insert verbatim, and declares infra: ["postgres"]. The harness starts the services an example declares with camel infra run before the app and waits until camel infra get answers, and the runner stops them when the run is over. Each pass stops the service first, so it starts against an empty database: kept running, its rows would carry over and step 3, which sets C-207 to NL, would pass on what the pass before it left. The reset belongs there and not in the route. A DROP in the setup route also runs on every reload, and an UPDATE that fires before the file route's inserts is then wiped -- which cost step 3 six of ten passes before it was taken out. Step 1 goes from 0/10 to 2/10: winnable now, where it was impossible. What holds it back is not the database but CAMEL-25075, the model spending its whole budget on identical camel_catalog_doc(sql) calls in 8 of 10 passes. Co-Authored-By: Claude Opus 5 (1M context) <[email protected]> Claude-Session: https://claude.ai/code/session_01Bp3538HRBPMQkb5ta9xRaj --- ai-benchmark/agent_mcp_stepwise.py | 83 +++++++++++++++++++++++-------------- ai-benchmark/gen_stepwise.py | 24 +++++++---- ai-benchmark/run-stepwise-ladder.sh | 8 ++++ 3 files changed, 75 insertions(+), 40 deletions(-) diff --git a/ai-benchmark/agent_mcp_stepwise.py b/ai-benchmark/agent_mcp_stepwise.py index e1dd732..9eb6b28 100755 --- a/ai-benchmark/agent_mcp_stepwise.py +++ b/ai-benchmark/agent_mcp_stepwise.py @@ -129,46 +129,53 @@ def error_key(l): return (l.get("time") or l.get("timestamp") or "") + "|" + (l.get("message") or l.get("msg") or "")[:120] -def await_reload(mcp, name, seen, tries=12): - """Wait (up to tries*2 s) for a reload record newer than seen; macOS polls the WatchService about every 10 s.""" - for _ in range(tries): - time.sleep(2) - if any(reload_key(l) not in seen for l in log_lines(mcp, name, 150) if isinstance(l, dict) and is_reload(l)): - return True - return False +INFRA_TIMEOUT = int(os.environ.get("BENCH_INFRA_TIMEOUT", "300")) # seconds to wait for `camel infra` services -def apply_reference(project, cfg, reference, mcp, name, before=None): - """Write a step's reference files the way a person does: the properties (and any other file) first, then the route. +def infra_start(services): + """Start the services the example needs with `camel infra run <svc> --background` and wait until each answers. - Not one batch: dev mode reloads the routes before it reloads the properties, so a route written in the same poll - as the property it uses fails to start ("Property with key [shop.currency] not found") and the later properties - reload does not retry it. Writing the other files first, and letting that reload land, is what the model's own - pace does for free. The step's reload baseline is re-taken just before the route is written, so the caller waits - for the route's reload and not for the one the first write already triggered. + Stopped first, so every pass starts against an empty database. A service kept running between passes would carry + its rows over, and then a step that changes a row passes on what the pass before it left rather than on the work + of the model: step 3 sets C-207 to NL, and the next pass would find it already NL. Postgres takes about 80 s to + come up, which is the price of that. The runner stops them when the whole run is over. """ - route_file = cfg["route_file"] - others = [f for f in reference if f != route_file] + for svc in services: + subprocess.run(["camel", "infra", "stop", svc], check=False, + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) + for svc in services: + subprocess.run(["camel", "infra", "run", svc, "--background"], check=False, + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) + data, deadline = {}, time.time() + INFRA_TIMEOUT + for svc in services: + while time.time() < deadline: + out = subprocess.run(["camel", "infra", "get", svc, "--json"], capture_output=True, text=True).stdout + m = re.search(r"\{.*\}", out, re.S) + if m: + try: + data[svc] = json.loads(m.group(0)) + break + except json.JSONDecodeError: + pass + time.sleep(5) + else: + data[svc] = {"error": f"{svc} did not come up within {INFRA_TIMEOUT}s"} + return data + - def write(fname): +def apply_reference(project, cfg, reference, mcp, name, before=None): + """Write a step's reference files, all of them, as one save. + + They used to be written in two goes, the properties first and the route once that reload had landed, because dev + mode reloaded the routes of a change before its properties, so a route written in the same poll as the property it + uses failed to start. CAMEL-25041 reloads one save as one batch with the properties applied first, so one go is + right again -- and better: two writes two seconds apart reloaded the app twice, and an HTTP call in flight across + the second reload was answered 404 by a rest route that was being rebuilt (ref12, connect-http-client step 1). + """ + for fname in reference: os.makedirs(os.path.dirname(os.path.join(project, fname)) or project, exist_ok=True) with open(os.path.join(project, fname), "w") as f: f.write(reference[fname]) - - def reloads(): - return {reload_key(l) for l in log_lines(mcp, name, 150) if isinstance(l, dict) and is_reload(l)} - - if others and route_file in reference: - seen = reloads() - for fname in others: - write(fname) - await_reload(mcp, name, seen) - if before is not None: - before["reload_records"] = reloads() - write(route_file) - else: - for fname in reference: - write(fname) return len(reference) @@ -338,6 +345,18 @@ def main(): proc = None peer = None + if cfg.get("infra"): + # the services the example connects to, as its README starts them: the sql example's postgres. Left running + # when the pass ends, so the next pass of the same example does not pay the startup again (about 80 s); the + # runner stops them when the whole run is over + t0 = time.time() + data = infra_start(cfg["infra"]) + print(f"camel infra run {' '.join(cfg['infra'])} -> {json.dumps(data)[:300]} ({time.time() - t0:.0f}s)", + file=log, flush=True) + for svc, d in data.items(): + if isinstance(d, dict) and d.get("error"): + print(f"ABORT: {d['error']}", file=log, flush=True) + raise SystemExit(f"{svc} did not come up") if cfg.get("peer"): # CAMEL-24886: a second app the example talks to (the stock API behind a client), from its own directory pd = os.path.join(HERE, cfg["peer"]["project"]); pname = cfg["peer"]["name"] diff --git a/ai-benchmark/gen_stepwise.py b/ai-benchmark/gen_stepwise.py index 5e8b3f0..f4a32af 100644 --- a/ai-benchmark/gen_stepwise.py +++ b/ai-benchmark/gen_stepwise.py @@ -335,18 +335,23 @@ EXAMPLES.append({ "reference": {CBR: cbr_s4}}, ]}) -# ---------------------------------------------------------------- connect-service/sql (on H2 instead of Postgres: no infra, same routes; CAMEL-24834 tool-group experiment) +# ---------------------------------------------------------------- connect-service/sql (Postgres as the example ships it, started with camel infra) SQL = "sql.camel.yaml" -SQL_PROPS = ("spring.datasource.url=jdbc:h2:mem:shop;DB_CLOSE_DELAY=-1\nspring.datasource.username=sa\nspring.datasource.password=\n" - "spring.datasource.driverClassName=org.h2.Driver\ncamel.jbang.dependencies=com.h2database:h2:2.3.232\n") +# the example's own datasource: what `camel infra run postgres` prints +SQL_PROPS = ("spring.datasource.url=jdbc:postgresql://localhost:5432/test\nspring.datasource.username=test\n" + "spring.datasource.password=test\nspring.datasource.driverClassName=org.postgresql.Driver\n") TIMER_SETUP = " uri: timer\n parameters:\n timerName: setup\n repeatCount: 1\n delay: 0\n" TIMER_REPORT = " uri: timer\n parameters:\n timerName: report\n delay: 5000\n period: 10000\n" FILE_ORDERS_SQL = FILE_ORDERS + " initialDelay: 2000\n" def sql_to(query, noop=False): return (" - to:\n uri: sql\n parameters:\n query: \"" + query + "\"\n" + (" noop: true\n" if noop else "")) +# exactly as the example has it: no reset in the route. A reset here runs on every reload too, and then an UPDATE +# that fires before the file route's inserts is wiped -- which cost step 3 six of ten passes. The database is made +# fresh between passes instead, by recycling the infra service. CREATE = sql_to("CREATE TABLE IF NOT EXISTS customers (id varchar(10) PRIMARY KEY, country varchar(2), orders integer)") -MERGE = sql_to("MERGE INTO customers USING (VALUES (:#${body[customer]}, :#${body[country]})) AS s(id, country) ON customers.id = s.id " - "WHEN MATCHED THEN UPDATE SET orders = customers.orders + 1 WHEN NOT MATCHED THEN INSERT (id, country, orders) VALUES (s.id, s.country, 1)", noop=True) +# the example's own insert, verbatim +MERGE = sql_to("INSERT INTO customers (id, country, orders) VALUES (:#${body[customer]}, :#${body[country]}, 1) " + "ON CONFLICT (id) DO UPDATE SET orders = customers.orders + 1", noop=True) REGISTERED = log("Customer ${body[customer]} from ${body[country]} registered with order ${body[orderId]}") REPORT = (sql_to("SELECT id, country, orders FROM customers ORDER BY id") + log("${body.size()} customer(s) in the table") + " - split:\n expression:\n simple:\n expression: \"${body}\"\n steps:\n" @@ -358,11 +363,11 @@ FIX_COUNTRY = route("fix-country", " uri: timer\n parameters:\n sql_to("UPDATE customers SET country = 'NL' WHERE id = 'C-207'") + log("Fixed country of C-207")) EXAMPLES.append({ "name": "connect-service-sql", "route_file": SQL, "props_file": "application.properties", "wait_seconds": 16, - "seed": "connect-service-sql", + "seed": "connect-service-sql", "infra": ["postgres"], "initial": {SQL: sql_s1, "application.properties": SQL_PROPS}, "steps": [ - {"request": "Add a route register-customers that reads the JSON files in the orders directory (file endpoint with noop true, sortBy file:name and initialDelay 2000), unmarshals each with Jackson, and registers the customer in the customers table with the sql component using named parameters from the body: a new customer gets a row (id from body[customer], country from body[country], orders 1), a customer that already has a row gets orders + 1; the orders directory is read again e [...] - "check": {"file_regex": "MERGE|MATCHED|CONFLICT|DUPLICATE", "log_regex": ["Customer C-482 from DK registered with order ORD-1001", "Customer C-134 from US registered with order ORD-1003"], "log_not_regex": "Syntax error|Unique index or primary key violation"}, + {"request": "Add a route register-customers that reads the JSON files in the orders directory (file endpoint with noop true, sortBy file:name and initialDelay 2000), unmarshals each with Jackson, and registers the customer in the customers table with the sql component using named parameters from the body: a new customer gets a row (id from body[customer], country from body[country], orders 1), a customer that already has a row gets orders + 1; the orders directory is read again e [...] + "check": {"file_regex": "CONFLICT|MERGE", "log_regex": ["Customer C-482 from DK registered with order ORD-1001", "Customer C-134 from US registered with order ORD-1003"], "log_not_regex": "Syntax error|duplicate key value"}, "reference": {SQL: sql_s3}}, {"request": "Add a route customer-report from a timer (delay 5000, period 10000) that selects id, country and orders from customers ordered by id, logs \"${body.size()} customer(s) in the table\", and splits the list to log \" ${body[id]} (${body[country]}): ${body[orders]} order(s)\" per row.", "check": {"file_regex": "split", "log_regex": ["3 customer\\(s\\) in the table", "C-482 \\(DK\\): \\d+ order\\(s\\)"]}, @@ -802,6 +807,9 @@ def main(): cfg = {"project": "stepwise-ladder/" + ex["name"], "route_file": ex["route_file"], "props_file": ex["props_file"], "wait_seconds": ex["wait_seconds"], "source_dir": True, "seed": ex.get("seed"), "exclude_seeds": ex.get("exclude_seeds", []), "initial": ex["initial"], "steps": steps} + if ex.get("infra"): + # services the example needs, started with `camel infra run` before the app (the sql example's postgres) + cfg["infra"] = ex["infra"] if ex.get("peer"): cfg["peer"] = {"project": ex["peer"]["project"], "name": ex["peer"]["name"]} pd = os.path.join(HERE, ex["peer"]["project"]) diff --git a/ai-benchmark/run-stepwise-ladder.sh b/ai-benchmark/run-stepwise-ladder.sh index 19d6364..ae894aa 100755 --- a/ai-benchmark/run-stepwise-ladder.sh +++ b/ai-benchmark/run-stepwise-ladder.sh @@ -22,6 +22,14 @@ for i in $(seq 1 "$K"); do # safety net: the harness stops its integration, but an interrupted run leaves one behind camel stop "$name" > /dev/null 2>&1 || true done + # the services an example needed stay up across its passes; stop them once the run is over + for svc in $(python3 -c " +import glob, json +out = set() +for f in glob.glob('steps-ladder/*.json'): + try: out.update(json.load(open(f)).get('infra') or []) + except Exception: pass +print(' '.join(sorted(out)))"); do camel infra stop "$svc" > /dev/null 2>&1 || true; done echo "[$T] DONE $(date +%T)" | tee -a "stepwise-$T.log" done python3 summarize_stepwise.py "$TAG" "$K"
