Loop guard (3 identical pushes), end_reason and activation error records per run; thinking off for the stage 1 baseline

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat
This commit is contained in:
Kral
2026-10-04 07:07:44 +02:00
parent 040b9900fd
commit 0401186a6c
6 changed files with 79 additions and 8 deletions

View File

@@ -61,7 +61,9 @@ def _norm(src):
def trajectory_stats(path):
"""Rebuild agent statistics from a trajectory (same rules as ToolProxy)."""
from .proxy import WRITE_TOOLS
calls = activations = streak = max_streak = 0
from .proxy import activation_messages
calls = activations = streak = max_streak = fails = 0
errors = []
final, t_first, t_last = "", None, None
for line in open(path):
e = json.loads(line)
@@ -79,7 +81,13 @@ def trajectory_stats(path):
failed = e["is_error"] or '"success":false' in e["result"].replace(" ", "")
streak = streak + 1 if failed else 0
max_streak = max(max_streak, streak)
if failed:
fails += 1
for m in activation_messages(e["result"], e["is_error"]):
if m not in errors and len(errors) < 30:
errors.append(m)
return {"tool_calls": calls, "activations": activations, "max_fail_streak": max_streak,
"activation_failures": fails, "activation_error_messages": errors,
"final_report": final, "agent_seconds": round((t_last or 0) - (t_first or 0), 1)}
@@ -207,6 +215,9 @@ class Runner:
rep["agent_seconds"] = round(time.time() - t1, 1)
rep["tool_calls"], rep["activations"] = proxy.calls, proxy.activations
rep["max_fail_streak"] = proxy.max_fail_streak
rep["activation_failures"] = proxy.activation_failures
rep["activation_error_messages"] = proxy.activation_errors
rep["end_reason"] = getattr(agent, "end_reason", None)
if proxy.fallbacks:
rep["adt_fallbacks"] = proxy.fallbacks