Loop guard (3 identical pushes), end_reason and activation error records per run; thinking off for the stage 1 baseline

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat
This commit is contained in:
Kral
2026-10-04 07:07:44 +02:00
parent 040b9900fd
commit 0401186a6c
6 changed files with 79 additions and 8 deletions

View File

@@ -70,7 +70,7 @@ class LlmAgent:
"""
def __init__(self, model, base_url=None, api_key=None, max_turns=80, temperature=0.2,
max_seconds=None, max_tokens=None):
max_seconds=None, max_tokens=None, chat_template_kwargs=None, loop_guard=None):
self.model = model
self.name = f"llm:{model}"
self.base_url = (base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")).rstrip("/")
@@ -83,12 +83,17 @@ class LlmAgent:
# run cost, 2026-10-03). Normal turns: p95 17.5k, max 82k. A cut turn is retried (see run).
self.max_tokens = max_tokens or (32000 if ":cloud" in (model or "") else None)
self.empty_retries = 2
self.loop_guard = loop_guard # end the run after this many identical pushes in a row (None = off)
self.end_reason = None
self.chat_template_kwargs = chat_template_kwargs # local server only, e.g. {"enable_thinking": False}
def _chat(self, messages, tools):
body = {"model": self.model, "messages": messages, "tools": tools,
"temperature": self.temperature, "parallel_tool_calls": False}
if self.max_tokens:
body["max_tokens"] = self.max_tokens
if self.chat_template_kwargs:
body["chat_template_kwargs"] = self.chat_template_kwargs
req = urllib.request.Request(f"{self.base_url}/chat/completions", json.dumps(body).encode(),
{"Content-Type": "application/json",
"Authorization": f"Bearer {self.api_key}"})
@@ -117,16 +122,19 @@ class LlmAgent:
check_budget()
final = ""
empty = 0
self.end_reason = "max_turns"
start = time.time()
for _ in range(self.max_turns):
if self.max_seconds and time.time() - start > self.max_seconds:
final = f"Stopped: time budget exceeded ({self.max_seconds} s)."
self.end_reason = "time_budget"
break
try:
msg, usage = self._chat(messages, tools)
add_usage(self.model, usage, kind="run", ref=proxy.prefix)
except RuntimeError as e:
final = f"Stopped: {e}"
self.end_reason = "model_error"
break
proxy.note("assistant", {"content": msg.get("content"),
"tool_calls": msg.get("tool_calls"), "usage": usage})
@@ -138,6 +146,7 @@ class LlmAgent:
continue
if not calls:
final = msg.get("content") or ("Stopped: empty model response." if empty else "")
self.end_reason = "report" if (msg.get("content") or "").strip() else "empty_response"
break
try:
for c in calls:
@@ -147,8 +156,15 @@ class LlmAgent:
err, text = proxy.call(fn["name"], args)
messages.append({"role": "tool", "tool_call_id": c.get("id", ""),
"content": ("ERROR: " if err else "") + text[:12000]})
if self.loop_guard and proxy.same_push_streak >= self.loop_guard:
final = f"Stopped: loop (the same source was pushed {proxy.same_push_streak} times in a row)."
self.end_reason = "loop"
break
if self.end_reason == "loop":
break
except BudgetExceeded as e:
final = f"Stopped: budget exceeded ({e})."
self.end_reason = "tool_budget"
break
proxy.note("final", final)
return final