B: stage 2 data analysis (repair taxonomy, teacher vs Qwen behaviors, duplicates, empty_response); empty_response fix (cap 24000, retry temperature, stream guard)

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-06 05:56:23 +02:00
parent 1ddb9a7c65
commit 542fc8ec31
6 changed files with 406 additions and 9 deletions

View File

@@ -84,7 +84,7 @@ class LlmAgent:
def __init__(self, model, base_url=None, api_key=None, max_turns=80, temperature=0.2,
max_seconds=None, max_tokens=None, chat_template_kwargs=None, loop_guard=None,
deadline=None, watch=False):
deadline=None, watch=False, empty_retries=2, retry_temperature=None, stream_guard=None):
self.model = model
self.name = f"llm:{model}"
self.base_url = (base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")).rstrip("/")
@@ -96,7 +96,12 @@ class LlmAgent:
# Runaway reasoning: 35 of 1965 DeepSeek turns produced 393k output tokens and no content (46 % of the
# run cost, 2026-10-03). Normal turns: p95 17.5k, max 82k. A cut turn is retried (see run).
self.max_tokens = max_tokens or (32000 if ":cloud" in (model or "") else None)
self.empty_retries = 2
self.empty_retries = empty_retries
# empty_response fix (2026-10-06): a retry after an empty turn can use another temperature (the same sample often
# runs away again: 7 of 13 empty runs had two or three capped turns in a row); stream_guard = reasoning tokens
# after which a streamed turn without any content or tool call is cut and counted as an empty turn
self.retry_temperature = retry_temperature
self.stream_guard = stream_guard
self.loop_guard = loop_guard # end the run after this many identical pushes in a row (None = off)
self.deadline = deadline # absolute time (time.time()) of the hard stop, or None
self.watch = watch # remote model: ping the server during a request, end the run when it is gone
@@ -104,9 +109,9 @@ class LlmAgent:
self.messages, self.tools, self.reasoning, self.turn_usage = [], [], [], [] # for the trajectory record
self.chat_template_kwargs = chat_template_kwargs # local server only, e.g. {"enable_thinking": False}
def _chat(self, messages, tools):
def _chat(self, messages, tools, temperature=None):
body = {"model": self.model, "messages": messages, "tools": tools,
"temperature": self.temperature, "parallel_tool_calls": False}
"temperature": temperature if temperature is not None else self.temperature, "parallel_tool_calls": False}
if self.max_tokens:
body["max_tokens"] = self.max_tokens
if self.chat_template_kwargs:
@@ -115,6 +120,11 @@ class LlmAgent:
{"Content-Type": "application/json",
"Authorization": f"Bearer {self.api_key}"})
last = None
if self.stream_guard and not (self.watch or self.deadline):
try:
return self._chat_stream(dict(body), messages)
except Exception as e: # noqa: BLE001 streaming not usable (server, parse): the normal request below
last = e
for attempt in range(4): # model server errors (HTTP 5xx, timeouts): retry with backoff
try:
if self.watch or self.deadline:
@@ -137,6 +147,49 @@ class LlmAgent:
time.sleep(10 * (attempt + 1))
raise RuntimeError(f"model request failed after retries: {last}")
def _chat_stream(self, body, messages):
"""Streamed request. A turn that has produced only reasoning for `stream_guard` tokens (about 3.2 characters per
token) and no content and no tool call is cut and returned as an empty turn (the run loop retries it)."""
body["stream"] = True
body["stream_options"] = {"include_usage": True}
req = urllib.request.Request(f"{self.base_url}/chat/completions", json.dumps(body).encode(),
{"Content-Type": "application/json", "Authorization": f"Bearer {self.api_key}"})
content, reasoning, calls, usage, cut = "", 0, {}, {}, False
with urllib.request.urlopen(req, timeout=self.request_timeout) as r:
for raw in r:
line = raw.decode("utf-8", "replace").strip()
if not line.startswith("data:"):
continue
data = line[5:].strip()
if data == "[DONE]":
break
chunk = json.loads(data)
if chunk.get("usage"):
usage = chunk["usage"]
for ch in chunk.get("choices") or []:
d = ch.get("delta") or {}
content += d.get("content") or ""
reasoning += len(d.get("reasoning") or d.get("reasoning_content") or d.get("thinking") or "")
for tc in d.get("tool_calls") or []:
c = calls.setdefault(tc.get("index", 0), {"id": tc.get("id") or "", "type": "function",
"function": {"name": "", "arguments": ""}})
c["id"] = c["id"] or tc.get("id") or ""
f = tc.get("function") or {}
c["function"]["name"] += f.get("name") or ""
a = f.get("arguments")
c["function"]["arguments"] += a if isinstance(a, str) else json.dumps(a) if a else ""
if not content.strip() and not calls and reasoning / 3.2 >= self.stream_guard:
cut = True
break
if cut:
est = {"prompt_tokens": int(len(json.dumps(messages)) / 3.5), "completion_tokens": int(reasoning / 3.2),
"estimated": True, "cut_by_stream_guard": True}
return {"role": "assistant", "content": ""}, est
msg = {"role": "assistant", "content": content or None}
if calls:
msg["tool_calls"] = [calls[k] for k in sorted(calls)]
return msg, usage
def _ping(self, timeout=10):
try:
urllib.request.urlopen(urllib.request.Request(f"{self.base_url}/models"), timeout=timeout).read()
@@ -195,7 +248,8 @@ class LlmAgent:
self.end_reason = "time_budget"
break
try:
msg, usage = self._chat(messages, tools)
msg, usage = self._chat(messages, tools,
self.retry_temperature if (empty and self.retry_temperature) else None)
add_usage(self.model, usage, kind="run", ref=proxy.prefix)
self.turn_usage.append(usage)
except WindowEnd: