Loop guard (3 identical pushes), end_reason and activation error records per run; thinking off for the stage 1 baseline
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat
This commit is contained in:
@@ -70,7 +70,7 @@ class LlmAgent:
|
||||
"""
|
||||
|
||||
def __init__(self, model, base_url=None, api_key=None, max_turns=80, temperature=0.2,
|
||||
max_seconds=None, max_tokens=None):
|
||||
max_seconds=None, max_tokens=None, chat_template_kwargs=None, loop_guard=None):
|
||||
self.model = model
|
||||
self.name = f"llm:{model}"
|
||||
self.base_url = (base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")).rstrip("/")
|
||||
@@ -83,12 +83,17 @@ class LlmAgent:
|
||||
# run cost, 2026-10-03). Normal turns: p95 17.5k, max 82k. A cut turn is retried (see run).
|
||||
self.max_tokens = max_tokens or (32000 if ":cloud" in (model or "") else None)
|
||||
self.empty_retries = 2
|
||||
self.loop_guard = loop_guard # end the run after this many identical pushes in a row (None = off)
|
||||
self.end_reason = None
|
||||
self.chat_template_kwargs = chat_template_kwargs # local server only, e.g. {"enable_thinking": False}
|
||||
|
||||
def _chat(self, messages, tools):
|
||||
body = {"model": self.model, "messages": messages, "tools": tools,
|
||||
"temperature": self.temperature, "parallel_tool_calls": False}
|
||||
if self.max_tokens:
|
||||
body["max_tokens"] = self.max_tokens
|
||||
if self.chat_template_kwargs:
|
||||
body["chat_template_kwargs"] = self.chat_template_kwargs
|
||||
req = urllib.request.Request(f"{self.base_url}/chat/completions", json.dumps(body).encode(),
|
||||
{"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {self.api_key}"})
|
||||
@@ -117,16 +122,19 @@ class LlmAgent:
|
||||
check_budget()
|
||||
final = ""
|
||||
empty = 0
|
||||
self.end_reason = "max_turns"
|
||||
start = time.time()
|
||||
for _ in range(self.max_turns):
|
||||
if self.max_seconds and time.time() - start > self.max_seconds:
|
||||
final = f"Stopped: time budget exceeded ({self.max_seconds} s)."
|
||||
self.end_reason = "time_budget"
|
||||
break
|
||||
try:
|
||||
msg, usage = self._chat(messages, tools)
|
||||
add_usage(self.model, usage, kind="run", ref=proxy.prefix)
|
||||
except RuntimeError as e:
|
||||
final = f"Stopped: {e}"
|
||||
self.end_reason = "model_error"
|
||||
break
|
||||
proxy.note("assistant", {"content": msg.get("content"),
|
||||
"tool_calls": msg.get("tool_calls"), "usage": usage})
|
||||
@@ -138,6 +146,7 @@ class LlmAgent:
|
||||
continue
|
||||
if not calls:
|
||||
final = msg.get("content") or ("Stopped: empty model response." if empty else "")
|
||||
self.end_reason = "report" if (msg.get("content") or "").strip() else "empty_response"
|
||||
break
|
||||
try:
|
||||
for c in calls:
|
||||
@@ -147,8 +156,15 @@ class LlmAgent:
|
||||
err, text = proxy.call(fn["name"], args)
|
||||
messages.append({"role": "tool", "tool_call_id": c.get("id", ""),
|
||||
"content": ("ERROR: " if err else "") + text[:12000]})
|
||||
if self.loop_guard and proxy.same_push_streak >= self.loop_guard:
|
||||
final = f"Stopped: loop (the same source was pushed {proxy.same_push_streak} times in a row)."
|
||||
self.end_reason = "loop"
|
||||
break
|
||||
if self.end_reason == "loop":
|
||||
break
|
||||
except BudgetExceeded as e:
|
||||
final = f"Stopped: budget exceeded ({e})."
|
||||
self.end_reason = "tool_budget"
|
||||
break
|
||||
proxy.note("final", final)
|
||||
return final
|
||||
|
||||
Reference in New Issue
Block a user