This is an automated email from the ASF dual-hosted git repository.

davsclaus pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/camel-performance-tests.git

commit 8c2ddea9c35e37ecdeb0d1c4be78716bf416a3c2
Author: Claus Ibsen <[email protected]>
AuthorDate: Tue Sep 15 23:06:46 2026 +0200

    ai-benchmark: the tool allow-list names the shared authoring tools only 
(CAMEL-24712 removes the older per-kind ones)
    
    Co-Authored-By: Claude Fable 5.1 <[email protected]>
    Claude-Session: https://claude.ai/code/session_01Bp3538HRBPMQkb5ta9xRaj
---
 ai-benchmark/README.md             | 4 +++-
 ai-benchmark/agent_local.py        | 3 +--
 ai-benchmark/agent_mcp_stepwise.py | 5 ++---
 ai-benchmark/run-suite.sh          | 3 ++-
 4 files changed, 8 insertions(+), 7 deletions(-)

diff --git a/ai-benchmark/README.md b/ai-benchmark/README.md
index d0eeae6..bbf4207 100644
--- a/ai-benchmark/README.md
+++ b/ai-benchmark/README.md
@@ -25,7 +25,9 @@ Two benchmarks, both scored by what actually runs, not by 
reading the model's ou
    integration after the reload, and the errors, against a reference 
checkpoint.
 
 The model is never shown the example projects or their catalog tools; the 
one-shot tool set is restricted to catalog
-lookups and validation (`BENCH_TOOL_ALLOW` in `run-suite.sh`).
+lookups and validation (`BENCH_TOOL_ALLOW` in `run-suite.sh`): the shared 
`camel_catalog_doc`, `camel_catalog_find`,
+`camel_catalog_sample`, `camel_validate_source` and a few smaller ones. The 
first series also offered the older per-kind
+catalog tools that CAMEL-24712 removes; the allow-list here names the shared 
ones only.
 
 ## Prerequisites
 
diff --git a/ai-benchmark/agent_local.py b/ai-benchmark/agent_local.py
index c804329..3b3a651 100755
--- a/ai-benchmark/agent_local.py
+++ b/ai-benchmark/agent_local.py
@@ -26,8 +26,7 @@ TOOL_RESULT_CAP = 6000
 # Tools offered to the model: catalog lookups and validation only. No example 
catalog
 # (that would hand the model the answer), no runtime, security, migration or 
dependency tools.
 ALLOW = re.compile(os.environ.get("BENCH_TOOL_ALLOW",
-    
r"^camel_(catalog_(components|component_doc|eips|eip_doc|languages|language_doc|dataformats|dataformat_doc|doc_pages|doc_page)"
-    
r"|validate_(yaml|endpoint|route|configuration)|component_(doc|properties)|eip_doc|language_doc|dataformat_doc)"))
+    
r"^camel_(catalog_(doc|find|sample|docs)|validate_source|component_properties|configuration_validate|error_diagnose|eval_expression)$"))
 
 SYSTEM = """You are an AI assistant helping a developer build a small Apache 
Camel integration that runs
 with the Camel CLI (camel-jbang), written in Camel YAML DSL.
diff --git a/ai-benchmark/agent_mcp_stepwise.py 
b/ai-benchmark/agent_mcp_stepwise.py
index 8d20cc9..a543438 100755
--- a/ai-benchmark/agent_mcp_stepwise.py
+++ b/ai-benchmark/agent_mcp_stepwise.py
@@ -4,7 +4,7 @@ from CAMEL-24695.
 
 The model gets the shared authoring set (camel_catalog_doc, 
camel_catalog_find, camel_validate_source,
 camel_get_files, camel_write_file, camel_run, camel_control, camel_get_log, 
camel_get_errors,
-camel_eval_expression, camel_error_diagnose) plus the structured catalog 
tools, and a short neutral system
+camel_eval_expression, camel_error_diagnose) plus a few catalog tools, and a 
short neutral system
 prompt. The harness starts the integration once with camel_run (dev mode) 
before step 1, then sends the
 8 requests one at a time. Scoring: files on disk, log via camel_get_log,
 errors via camel_get_errors, diff size, reference checkpoint after each step.
@@ -28,8 +28,7 @@ TOOL_RESULT_CAP = 6000
 
 SHARED = ["camel_catalog_doc", "camel_catalog_find", "camel_catalog_sample", 
"camel_validate_source", "camel_get_files", "camel_write_file",
           "camel_run", "camel_control", "camel_get_log", "camel_get_errors", 
"camel_eval_expression", "camel_error_diagnose"]
-EXTRA = ["camel_catalog_components", "camel_catalog_component_doc", 
"camel_catalog_eips", "camel_catalog_eip_doc",
-         "camel_catalog_languages", "camel_catalog_language_doc", 
"camel_catalog_dataformats", "camel_catalog_dataformat_doc"]
+EXTRA = ["camel_catalog_docs", "camel_component_properties", 
"camel_configuration_validate"]
 
 SYSTEM = (
     "You are an Apache Camel assistant helping a developer edit a running 
Camel integration through the Camel MCP server.\n\n"
diff --git a/ai-benchmark/run-suite.sh b/ai-benchmark/run-suite.sh
index 5ce2551..77e2506 100755
--- a/ai-benchmark/run-suite.sh
+++ b/ai-benchmark/run-suite.sh
@@ -8,7 +8,8 @@ export MCP_URL="${MCP_URL:-http://127.0.0.1:9090/mcp}";
 export BENCH_VALIDATE_PROPS=1
 export BENCH_VALIDATE_SOURCE=1
 # the one-shot model gets catalog lookups and validation only: no example 
catalog (that would hand it the answer), no runtime tools
-export 
BENCH_TOOL_ALLOW="${BENCH_TOOL_ALLOW:-^camel_(catalog_(components|component_doc|eips|eip_doc|languages|language_doc|dataformats|dataformat_doc|docs|doc|find|sample)|validate_(yaml_dsl|route|source)|component_properties|configuration_validate|error_diagnose|eval_expression)$}"
+# the shared authoring tools of the catalog and validation kind; after 
CAMEL-24712 these are the only catalog tools
+export 
BENCH_TOOL_ALLOW="${BENCH_TOOL_ALLOW:-^camel_(catalog_(doc|find|sample|docs)|validate_source|component_properties|configuration_validate|error_diagnose|eval_expression)$}"
 command -v caffeinate > /dev/null && caffeinate -i -s -w $$ &   # macOS: keep 
the machine awake for the hour
 echo "[$TAG] one-shot start $(date +%T)" | tee -a "$TAG.log"
 BENCH_OUT="oneshot-$TAG" python3 agent_local.py > "oneshot-$TAG.out" 2>&1

Reply via email to