This is an automated email from the ASF dual-hosted git repository.
davsclaus pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/camel-performance-tests.git
The following commit(s) were added to refs/heads/main by this push:
new 4389512 Three checks were failing correct answers, and a step's
errors are now recorded
4389512 is described below
commit 4389512e5bdd52c898d6e64b5f338ff0182f3a58
Author: Claus Ibsen <[email protected]>
AuthorDate: Tue Sep 29 13:40:33 2026 +0200
Three checks were failing correct answers, and a step's errors are now
recorded
Reading the traces of the steps that fail most often, four of the top five
were
the harness marking correct work wrong rather than the model getting it
wrong.
connect-http-client step 3 asked for throwExceptionOnFailure and checked the
file for "throwExceptionOnFailure=false". Seven of ten passes wrote it as a
YAML
parameter instead, which is the same option and arguably the better style,
and
were failed for it although their log output was exactly right. The check
takes
either spelling now, and re-scoring the saved route files flips those seven.
contracts-openapi-client step 1 provokes failures on purpose (errors_ok, 90
to
108 of them) and then looks for one line in the last 150 log records. The
step's
own expected errors fill the window and push the line out of it, while it
sits
in the run log the whole time, byte for byte what the check asks for. A
step is
scored on 400 records now, BENCH_LOG_WINDOW.
The third, route-aggregator step 1, is left failing. Its file, properties
and log
checks all pass and it fails on two errors that are not in the run log, so
they
come from the error registry and nothing records what they were. Tolerating
them
would silence something that cannot be named, so instead the errors of a
step are
now kept in results.json as error_detail, and the next run can settle it.
connect-service-sql (H2 for Postgres), the log_not_regex baseline, two stale
reference files and a reset in a route that ran on every reload were the
earlier
ones. Every count so far has been biased against the model.
Co-Authored-By: Claude Opus 5 (1M context) <[email protected]>
Claude-Session: https://claude.ai/code/session_01Bp3538HRBPMQkb5ta9xRaj
---
ai-benchmark/agent_mcp_stepwise.py | 16 +++++++++++-----
ai-benchmark/gen_stepwise.py | 2 +-
2 files changed, 12 insertions(+), 6 deletions(-)
diff --git a/ai-benchmark/agent_mcp_stepwise.py
b/ai-benchmark/agent_mcp_stepwise.py
index 9eb6b28..14c72bc 100755
--- a/ai-benchmark/agent_mcp_stepwise.py
+++ b/ai-benchmark/agent_mcp_stepwise.py
@@ -24,6 +24,9 @@ HERE = os.path.dirname(os.path.abspath(__file__))
TAG = os.environ.get("BENCH_TAG", "mcp-" + MODEL.replace(":",
"_").replace("/", "_"))
OUT = os.path.join(HERE, "stepwise", TAG)
MAX_TOOL_CALLS = int(os.environ.get("BENCH_TOOL_CALLS", "20"))
+# how many log records a step is scored on. A step that provokes failures on
purpose (errors_ok) fills the window
+# with its own expected errors, and the line the check looks for scrolls out
of it while sitting in the run log
+LOG_WINDOW = int(os.environ.get("BENCH_LOG_WINDOW", "400"))
# round 2: BENCH_REFERENCE=1 skips the model and applies each step's reference
files instead, to check the steps file itself
REFERENCE = os.environ.get("BENCH_REFERENCE") == "1"
TOOL_RESULT_CAP = 6000
@@ -185,7 +188,7 @@ def error_snapshot(mcp, name):
Also the INFO and WARN records already in the window: the window holds 150
records and spans the step before,
so a log_not_regex would otherwise fail a step for output that the
previous step's route produced.
"""
- lines = log_lines(mcp, name, 150)
+ lines = log_lines(mcp, name, LOG_WINDOW)
records = {error_key(l) for l in lines if isinstance(l, dict) and
(l.get("level") or "").upper() == "ERROR"}
reloads = {reload_key(l) for l in lines if isinstance(l, dict) and
is_reload(l)}
before_msgs = {error_key(l) for l in lines
@@ -250,7 +253,7 @@ def score(step, project, cfg, mcp, name, before):
seen_reloads = before.get("reload_records") or set()
reload_at = None
for _ in range(12):
- fresh_reloads = [l for l in log_lines(mcp, name, 150)
+ fresh_reloads = [l for l in log_lines(mcp, name, LOG_WINDOW)
if isinstance(l, dict) and is_reload(l) and
reload_key(l) not in seen_reloads]
if fresh_reloads:
# the reload this step caused: what the step forbids is only
forbidden from here on, whatever the route
@@ -278,7 +281,7 @@ def score(step, project, cfg, mcp, name, before):
break
if not re.search(rx, content):
result["file_ok"] = False
- lines = log_lines(mcp, name, 150)
+ lines = log_lines(mcp, name, LOG_WINDOW)
def lvl(l): return (l.get("level") or "").upper()
# a multi-line message (a pretty-printed body) comes as one record with a
detail block: match on both
def msg(l): return (l.get("message") or l.get("msg") or "") + ("\n" +
l["detail"] if l.get("detail") else "")
@@ -317,7 +320,10 @@ def score(step, project, cfg, mcp, name, before):
result["errors"] = sum(1 for l in lines if isinstance(l, dict) and lvl(l)
== "ERROR" and error_key(l) not in seen)
data, _ = jcall(mcp, "camel_get_errors", {"name": name})
if isinstance(data, dict):
- result["errors"] += max(0, len(data.get("errors", []) or []) -
before.get("error_count", 0))
+ errs = data.get("errors", []) or []
+ result["errors"] += max(0, len(errs) - before.get("error_count", 0))
+ # what they were, not just how many: a step that fails on errors alone
cannot be judged from a count
+ result["error_detail"] = [str(e)[:300] for e in
errs[before.get("error_count", 0):]][:5]
diff = list(difflib.unified_diff(before["route"].splitlines(),
route.splitlines(), lineterm="", n=0))
diffp = list(difflib.unified_diff(before["props"].splitlines(),
props.splitlines(), lineterm="", n=0))
result["changed_lines"] = sum(1 for l in diff + diffp if
(l.startswith("+") or l.startswith("-")) and not l.startswith(("+++", "---")))
@@ -479,7 +485,7 @@ def main():
seen = error_snapshot(mcp, name)[2]
for _ in range(12):
time.sleep(2)
- if any(reload_key(l) not in seen for l in log_lines(mcp,
name, 150) if isinstance(l, dict) and is_reload(l)):
+ if any(reload_key(l) not in seen for l in log_lines(mcp,
name, LOG_WINDOW) if isinstance(l, dict) and is_reload(l)):
break
time.sleep(6)
if not res["ok"]:
diff --git a/ai-benchmark/gen_stepwise.py b/ai-benchmark/gen_stepwise.py
index f4a32af..28213b7 100644
--- a/ai-benchmark/gen_stepwise.py
+++ b/ai-benchmark/gen_stepwise.py
@@ -673,7 +673,7 @@ EXAMPLES.append({
"check": {"file_regex": "toD", "log_regex": ["ORD-1001: CAMEL-TSHIRT
x 2, 120 in stock", "ORD-1003: CAMEL-CAP x 1, 0 in stock"]},
"reference": {HC: hc_s3}},
{"request": "Decide per line: add throwExceptionOnFailure=false to the
HTTP call so a 404 does not throw, then a choice: when the header
CamelHttpResponseCode is not 200 log \"${exchangeProperty.orderId}:
${body[error]} (HTTP ${header.CamelHttpResponseCode})\", when ${body[qty]} >=
${exchangeProperty.needed} log \"${exchangeProperty.orderId}:
${exchangeProperty.sku} x ${exchangeProperty.needed}, ${body[qty]} in stock,
ok\", otherwise log \"${exchangeProperty.orderId}: ${exchangeP [...]
- "check": {"file_regex": "throwExceptionOnFailure=false", "log_regex":
["ORD-1001: CAMEL-TSHIRT x 2, 120 in stock, ok", "ORD-1003: CAMEL-CAP x 1, only
0 in stock, back-order"]},
+ "check": {"file_regex": "throwExceptionOnFailure[=:] *false",
"log_regex": ["ORD-1001: CAMEL-TSHIRT x 2, 120 in stock, ok", "ORD-1003:
CAMEL-CAP x 1, only 0 in stock, back-order"]},
"reference": {HC: hc_s4}},
]})