This is an automated email from the ASF dual-hosted git repository.

davsclaus pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/camel-performance-tests.git


The following commit(s) were added to refs/heads/main by this push:
     new 4389512  Three checks were failing correct answers, and a step's 
errors are now recorded
4389512 is described below

commit 4389512e5bdd52c898d6e64b5f338ff0182f3a58
Author: Claus Ibsen <[email protected]>
AuthorDate: Tue Sep 29 13:40:33 2026 +0200

    Three checks were failing correct answers, and a step's errors are now 
recorded
    
    Reading the traces of the steps that fail most often, four of the top five 
were
    the harness marking correct work wrong rather than the model getting it 
wrong.
    
    connect-http-client step 3 asked for throwExceptionOnFailure and checked the
    file for "throwExceptionOnFailure=false". Seven of ten passes wrote it as a 
YAML
    parameter instead, which is the same option and arguably the better style, 
and
    were failed for it although their log output was exactly right. The check 
takes
    either spelling now, and re-scoring the saved route files flips those seven.
    
    contracts-openapi-client step 1 provokes failures on purpose (errors_ok, 90 
to
    108 of them) and then looks for one line in the last 150 log records. The 
step's
    own expected errors fill the window and push the line out of it, while it 
sits
    in the run log the whole time, byte for byte what the check asks for. A 
step is
    scored on 400 records now, BENCH_LOG_WINDOW.
    
    The third, route-aggregator step 1, is left failing. Its file, properties 
and log
    checks all pass and it fails on two errors that are not in the run log, so 
they
    come from the error registry and nothing records what they were. Tolerating 
them
    would silence something that cannot be named, so instead the errors of a 
step are
    now kept in results.json as error_detail, and the next run can settle it.
    
    connect-service-sql (H2 for Postgres), the log_not_regex baseline, two stale
    reference files and a reset in a route that ran on every reload were the 
earlier
    ones. Every count so far has been biased against the model.
    
    Co-Authored-By: Claude Opus 5 (1M context) <[email protected]>
    Claude-Session: https://claude.ai/code/session_01Bp3538HRBPMQkb5ta9xRaj
---
 ai-benchmark/agent_mcp_stepwise.py | 16 +++++++++++-----
 ai-benchmark/gen_stepwise.py       |  2 +-
 2 files changed, 12 insertions(+), 6 deletions(-)

diff --git a/ai-benchmark/agent_mcp_stepwise.py 
b/ai-benchmark/agent_mcp_stepwise.py
index 9eb6b28..14c72bc 100755
--- a/ai-benchmark/agent_mcp_stepwise.py
+++ b/ai-benchmark/agent_mcp_stepwise.py
@@ -24,6 +24,9 @@ HERE = os.path.dirname(os.path.abspath(__file__))
 TAG = os.environ.get("BENCH_TAG", "mcp-" + MODEL.replace(":", 
"_").replace("/", "_"))
 OUT = os.path.join(HERE, "stepwise", TAG)
 MAX_TOOL_CALLS = int(os.environ.get("BENCH_TOOL_CALLS", "20"))
+# how many log records a step is scored on. A step that provokes failures on 
purpose (errors_ok) fills the window
+# with its own expected errors, and the line the check looks for scrolls out 
of it while sitting in the run log
+LOG_WINDOW = int(os.environ.get("BENCH_LOG_WINDOW", "400"))
 # round 2: BENCH_REFERENCE=1 skips the model and applies each step's reference 
files instead, to check the steps file itself
 REFERENCE = os.environ.get("BENCH_REFERENCE") == "1"
 TOOL_RESULT_CAP = 6000
@@ -185,7 +188,7 @@ def error_snapshot(mcp, name):
     Also the INFO and WARN records already in the window: the window holds 150 
records and spans the step before,
     so a log_not_regex would otherwise fail a step for output that the 
previous step's route produced.
     """
-    lines = log_lines(mcp, name, 150)
+    lines = log_lines(mcp, name, LOG_WINDOW)
     records = {error_key(l) for l in lines if isinstance(l, dict) and 
(l.get("level") or "").upper() == "ERROR"}
     reloads = {reload_key(l) for l in lines if isinstance(l, dict) and 
is_reload(l)}
     before_msgs = {error_key(l) for l in lines
@@ -250,7 +253,7 @@ def score(step, project, cfg, mcp, name, before):
     seen_reloads = before.get("reload_records") or set()
     reload_at = None
     for _ in range(12):
-        fresh_reloads = [l for l in log_lines(mcp, name, 150)
+        fresh_reloads = [l for l in log_lines(mcp, name, LOG_WINDOW)
                          if isinstance(l, dict) and is_reload(l) and 
reload_key(l) not in seen_reloads]
         if fresh_reloads:
             # the reload this step caused: what the step forbids is only 
forbidden from here on, whatever the route
@@ -278,7 +281,7 @@ def score(step, project, cfg, mcp, name, before):
                     break
         if not re.search(rx, content):
             result["file_ok"] = False
-    lines = log_lines(mcp, name, 150)
+    lines = log_lines(mcp, name, LOG_WINDOW)
     def lvl(l): return (l.get("level") or "").upper()
     # a multi-line message (a pretty-printed body) comes as one record with a 
detail block: match on both
     def msg(l): return (l.get("message") or l.get("msg") or "") + ("\n" + 
l["detail"] if l.get("detail") else "")
@@ -317,7 +320,10 @@ def score(step, project, cfg, mcp, name, before):
     result["errors"] = sum(1 for l in lines if isinstance(l, dict) and lvl(l) 
== "ERROR" and error_key(l) not in seen)
     data, _ = jcall(mcp, "camel_get_errors", {"name": name})
     if isinstance(data, dict):
-        result["errors"] += max(0, len(data.get("errors", []) or []) - 
before.get("error_count", 0))
+        errs = data.get("errors", []) or []
+        result["errors"] += max(0, len(errs) - before.get("error_count", 0))
+        # what they were, not just how many: a step that fails on errors alone 
cannot be judged from a count
+        result["error_detail"] = [str(e)[:300] for e in 
errs[before.get("error_count", 0):]][:5]
     diff = list(difflib.unified_diff(before["route"].splitlines(), 
route.splitlines(), lineterm="", n=0))
     diffp = list(difflib.unified_diff(before["props"].splitlines(), 
props.splitlines(), lineterm="", n=0))
     result["changed_lines"] = sum(1 for l in diff + diffp if 
(l.startswith("+") or l.startswith("-")) and not l.startswith(("+++", "---")))
@@ -479,7 +485,7 @@ def main():
                 seen = error_snapshot(mcp, name)[2]
                 for _ in range(12):
                     time.sleep(2)
-                    if any(reload_key(l) not in seen for l in log_lines(mcp, 
name, 150) if isinstance(l, dict) and is_reload(l)):
+                    if any(reload_key(l) not in seen for l in log_lines(mcp, 
name, LOG_WINDOW) if isinstance(l, dict) and is_reload(l)):
                         break
                 time.sleep(6)
                 if not res["ok"]:
diff --git a/ai-benchmark/gen_stepwise.py b/ai-benchmark/gen_stepwise.py
index f4a32af..28213b7 100644
--- a/ai-benchmark/gen_stepwise.py
+++ b/ai-benchmark/gen_stepwise.py
@@ -673,7 +673,7 @@ EXAMPLES.append({
          "check": {"file_regex": "toD", "log_regex": ["ORD-1001: CAMEL-TSHIRT 
x 2, 120 in stock", "ORD-1003: CAMEL-CAP x 1, 0 in stock"]},
          "reference": {HC: hc_s3}},
         {"request": "Decide per line: add throwExceptionOnFailure=false to the 
HTTP call so a 404 does not throw, then a choice: when the header 
CamelHttpResponseCode is not 200 log \"${exchangeProperty.orderId}: 
${body[error]} (HTTP ${header.CamelHttpResponseCode})\", when ${body[qty]} >= 
${exchangeProperty.needed} log \"${exchangeProperty.orderId}: 
${exchangeProperty.sku} x ${exchangeProperty.needed}, ${body[qty]} in stock, 
ok\", otherwise log \"${exchangeProperty.orderId}: ${exchangeP [...]
-         "check": {"file_regex": "throwExceptionOnFailure=false", "log_regex": 
["ORD-1001: CAMEL-TSHIRT x 2, 120 in stock, ok", "ORD-1003: CAMEL-CAP x 1, only 
0 in stock, back-order"]},
+         "check": {"file_regex": "throwExceptionOnFailure[=:] *false", 
"log_regex": ["ORD-1001: CAMEL-TSHIRT x 2, 120 in stock, ok", "ORD-1003: 
CAMEL-CAP x 1, only 0 in stock, back-order"]},
          "reference": {HC: hc_s4}},
     ]})
 

Reply via email to