Loop guard (3 identical pushes), end_reason and activation error records per run; thinking off for the stage 1 baseline
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat
This commit is contained in:
@@ -61,7 +61,9 @@ def _norm(src):
|
||||
def trajectory_stats(path):
|
||||
"""Rebuild agent statistics from a trajectory (same rules as ToolProxy)."""
|
||||
from .proxy import WRITE_TOOLS
|
||||
calls = activations = streak = max_streak = 0
|
||||
from .proxy import activation_messages
|
||||
calls = activations = streak = max_streak = fails = 0
|
||||
errors = []
|
||||
final, t_first, t_last = "", None, None
|
||||
for line in open(path):
|
||||
e = json.loads(line)
|
||||
@@ -79,7 +81,13 @@ def trajectory_stats(path):
|
||||
failed = e["is_error"] or '"success":false' in e["result"].replace(" ", "")
|
||||
streak = streak + 1 if failed else 0
|
||||
max_streak = max(max_streak, streak)
|
||||
if failed:
|
||||
fails += 1
|
||||
for m in activation_messages(e["result"], e["is_error"]):
|
||||
if m not in errors and len(errors) < 30:
|
||||
errors.append(m)
|
||||
return {"tool_calls": calls, "activations": activations, "max_fail_streak": max_streak,
|
||||
"activation_failures": fails, "activation_error_messages": errors,
|
||||
"final_report": final, "agent_seconds": round((t_last or 0) - (t_first or 0), 1)}
|
||||
|
||||
|
||||
@@ -207,6 +215,9 @@ class Runner:
|
||||
rep["agent_seconds"] = round(time.time() - t1, 1)
|
||||
rep["tool_calls"], rep["activations"] = proxy.calls, proxy.activations
|
||||
rep["max_fail_streak"] = proxy.max_fail_streak
|
||||
rep["activation_failures"] = proxy.activation_failures
|
||||
rep["activation_error_messages"] = proxy.activation_errors
|
||||
rep["end_reason"] = getattr(agent, "end_reason", None)
|
||||
if proxy.fallbacks:
|
||||
rep["adt_fallbacks"] = proxy.fallbacks
|
||||
|
||||
|
||||
Reference in New Issue
Block a user