This is an automated email from the ASF dual-hosted git repository. davsclaus pushed a commit to branch main in repository https://gitbox.apache.org/repos/asf/camel-performance-tests.git
commit 8c2ddea9c35e37ecdeb0d1c4be78716bf416a3c2 Author: Claus Ibsen <[email protected]> AuthorDate: Tue Sep 15 23:06:46 2026 +0200 ai-benchmark: the tool allow-list names the shared authoring tools only (CAMEL-24712 removes the older per-kind ones) Co-Authored-By: Claude Fable 5.1 <[email protected]> Claude-Session: https://claude.ai/code/session_01Bp3538HRBPMQkb5ta9xRaj --- ai-benchmark/README.md | 4 +++- ai-benchmark/agent_local.py | 3 +-- ai-benchmark/agent_mcp_stepwise.py | 5 ++--- ai-benchmark/run-suite.sh | 3 ++- 4 files changed, 8 insertions(+), 7 deletions(-) diff --git a/ai-benchmark/README.md b/ai-benchmark/README.md index d0eeae6..bbf4207 100644 --- a/ai-benchmark/README.md +++ b/ai-benchmark/README.md @@ -25,7 +25,9 @@ Two benchmarks, both scored by what actually runs, not by reading the model's ou integration after the reload, and the errors, against a reference checkpoint. The model is never shown the example projects or their catalog tools; the one-shot tool set is restricted to catalog -lookups and validation (`BENCH_TOOL_ALLOW` in `run-suite.sh`). +lookups and validation (`BENCH_TOOL_ALLOW` in `run-suite.sh`): the shared `camel_catalog_doc`, `camel_catalog_find`, +`camel_catalog_sample`, `camel_validate_source` and a few smaller ones. The first series also offered the older per-kind +catalog tools that CAMEL-24712 removes; the allow-list here names the shared ones only. ## Prerequisites diff --git a/ai-benchmark/agent_local.py b/ai-benchmark/agent_local.py index c804329..3b3a651 100755 --- a/ai-benchmark/agent_local.py +++ b/ai-benchmark/agent_local.py @@ -26,8 +26,7 @@ TOOL_RESULT_CAP = 6000 # Tools offered to the model: catalog lookups and validation only. No example catalog # (that would hand the model the answer), no runtime, security, migration or dependency tools. ALLOW = re.compile(os.environ.get("BENCH_TOOL_ALLOW", - r"^camel_(catalog_(components|component_doc|eips|eip_doc|languages|language_doc|dataformats|dataformat_doc|doc_pages|doc_page)" - r"|validate_(yaml|endpoint|route|configuration)|component_(doc|properties)|eip_doc|language_doc|dataformat_doc)")) + r"^camel_(catalog_(doc|find|sample|docs)|validate_source|component_properties|configuration_validate|error_diagnose|eval_expression)$")) SYSTEM = """You are an AI assistant helping a developer build a small Apache Camel integration that runs with the Camel CLI (camel-jbang), written in Camel YAML DSL. diff --git a/ai-benchmark/agent_mcp_stepwise.py b/ai-benchmark/agent_mcp_stepwise.py index 8d20cc9..a543438 100755 --- a/ai-benchmark/agent_mcp_stepwise.py +++ b/ai-benchmark/agent_mcp_stepwise.py @@ -4,7 +4,7 @@ from CAMEL-24695. The model gets the shared authoring set (camel_catalog_doc, camel_catalog_find, camel_validate_source, camel_get_files, camel_write_file, camel_run, camel_control, camel_get_log, camel_get_errors, -camel_eval_expression, camel_error_diagnose) plus the structured catalog tools, and a short neutral system +camel_eval_expression, camel_error_diagnose) plus a few catalog tools, and a short neutral system prompt. The harness starts the integration once with camel_run (dev mode) before step 1, then sends the 8 requests one at a time. Scoring: files on disk, log via camel_get_log, errors via camel_get_errors, diff size, reference checkpoint after each step. @@ -28,8 +28,7 @@ TOOL_RESULT_CAP = 6000 SHARED = ["camel_catalog_doc", "camel_catalog_find", "camel_catalog_sample", "camel_validate_source", "camel_get_files", "camel_write_file", "camel_run", "camel_control", "camel_get_log", "camel_get_errors", "camel_eval_expression", "camel_error_diagnose"] -EXTRA = ["camel_catalog_components", "camel_catalog_component_doc", "camel_catalog_eips", "camel_catalog_eip_doc", - "camel_catalog_languages", "camel_catalog_language_doc", "camel_catalog_dataformats", "camel_catalog_dataformat_doc"] +EXTRA = ["camel_catalog_docs", "camel_component_properties", "camel_configuration_validate"] SYSTEM = ( "You are an Apache Camel assistant helping a developer edit a running Camel integration through the Camel MCP server.\n\n" diff --git a/ai-benchmark/run-suite.sh b/ai-benchmark/run-suite.sh index 5ce2551..77e2506 100755 --- a/ai-benchmark/run-suite.sh +++ b/ai-benchmark/run-suite.sh @@ -8,7 +8,8 @@ export MCP_URL="${MCP_URL:-http://127.0.0.1:9090/mcp}" export BENCH_VALIDATE_PROPS=1 export BENCH_VALIDATE_SOURCE=1 # the one-shot model gets catalog lookups and validation only: no example catalog (that would hand it the answer), no runtime tools -export BENCH_TOOL_ALLOW="${BENCH_TOOL_ALLOW:-^camel_(catalog_(components|component_doc|eips|eip_doc|languages|language_doc|dataformats|dataformat_doc|docs|doc|find|sample)|validate_(yaml_dsl|route|source)|component_properties|configuration_validate|error_diagnose|eval_expression)$}" +# the shared authoring tools of the catalog and validation kind; after CAMEL-24712 these are the only catalog tools +export BENCH_TOOL_ALLOW="${BENCH_TOOL_ALLOW:-^camel_(catalog_(doc|find|sample|docs)|validate_source|component_properties|configuration_validate|error_diagnose|eval_expression)$}" command -v caffeinate > /dev/null && caffeinate -i -s -w $$ & # macOS: keep the machine awake for the hour echo "[$TAG] one-shot start $(date +%T)" | tee -a "$TAG.log" BENCH_OUT="oneshot-$TAG" python3 agent_local.py > "oneshot-$TAG.out" 2>&1
