Series A: local Qwen on the MacBook (remote server), 12 h window with hard stop, clean pause on server loss, dashboard card, docs/remote-model.md

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-05 21:57:01 +02:00
parent dc8d99a913
commit eb13453097
6 changed files with 576 additions and 6 deletions

View File

@@ -4,9 +4,20 @@ import os
import time
import urllib.request
import threading
import urllib.error
from .ledger import add_usage, check_budget
from .proxy import BudgetExceeded
class ServerDown(Exception):
"""The model server does not answer (remote local model). The run ends cleanly and is not a result."""
class WindowEnd(Exception):
"""The time window of the series is over (hard stop, also in the middle of a request)."""
SYSTEM_PROMPT = """You are an ABAP developer. You implement a plan on an SAP system with the tools.
Rules:
@@ -72,7 +83,8 @@ class LlmAgent:
"""
def __init__(self, model, base_url=None, api_key=None, max_turns=80, temperature=0.2,
max_seconds=None, max_tokens=None, chat_template_kwargs=None, loop_guard=None):
max_seconds=None, max_tokens=None, chat_template_kwargs=None, loop_guard=None,
deadline=None, watch=False):
self.model = model
self.name = f"llm:{model}"
self.base_url = (base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")).rstrip("/")
@@ -86,6 +98,8 @@ class LlmAgent:
self.max_tokens = max_tokens or (32000 if ":cloud" in (model or "") else None)
self.empty_retries = 2
self.loop_guard = loop_guard # end the run after this many identical pushes in a row (None = off)
self.deadline = deadline # absolute time (time.time()) of the hard stop, or None
self.watch = watch # remote model: ping the server during a request, end the run when it is gone
self.end_reason = None
self.messages, self.tools, self.reasoning, self.turn_usage = [], [], [], [] # for the trajectory record
self.chat_template_kwargs = chat_template_kwargs # local server only, e.g. {"enable_thinking": False}
@@ -103,17 +117,60 @@ class LlmAgent:
last = None
for attempt in range(4): # model server errors (HTTP 5xx, timeouts): retry with backoff
try:
with urllib.request.urlopen(req, timeout=self.request_timeout) as r:
data = json.loads(r.read().decode())
return data["choices"][0]["message"], data.get("usage", {})
if self.watch or self.deadline:
data = self._post_watched(req)
else:
with urllib.request.urlopen(req, timeout=self.request_timeout) as r:
data = json.loads(r.read().decode())
return data["choices"][0]["message"], data.get("usage", {})
except (ServerDown, WindowEnd):
raise
except Exception as e: # noqa: BLE001
last = e
code = getattr(e, "code", None)
if code is not None and code < 500 and code != 429:
raise
if self.deadline and time.time() >= self.deadline:
raise WindowEnd()
if self.watch and not self._ping():
raise ServerDown(str(e)[:200])
time.sleep(10 * (attempt + 1))
raise RuntimeError(f"model request failed after retries: {last}")
def _ping(self, timeout=10):
try:
urllib.request.urlopen(urllib.request.Request(f"{self.base_url}/models"), timeout=timeout).read()
return True
except Exception: # noqa: BLE001
return False
def _post_watched(self, req):
"""The request runs in a thread; this thread watches the deadline and (remote model) the server: a hard stop
or a dead server ends the wait at once, also in the middle of a long generation."""
box = {}
def work():
try:
with urllib.request.urlopen(req, timeout=self.request_timeout) as r:
box["data"] = json.loads(r.read().decode())
except BaseException as e: # noqa: BLE001
box["err"] = e
th = threading.Thread(target=work, daemon=True)
th.start()
last_ping, fails = time.time(), 0
while th.is_alive():
th.join(5)
if self.deadline and time.time() >= self.deadline:
raise WindowEnd()
if self.watch and th.is_alive() and time.time() - last_ping >= 30:
last_ping = time.time()
fails = 0 if self._ping() else fails + 1
if fails >= 3:
raise ServerDown("no answer to 3 pings in a row during a request")
if "err" in box:
raise box["err"]
return box["data"]
def run(self, task, proxy):
tools = [{"type": "function", "function": {"name": t["name"],
"description": t.get("description", ""),
@@ -129,6 +186,10 @@ class LlmAgent:
self.end_reason = "max_turns"
start = time.time()
for _ in range(self.max_turns):
if self.deadline and time.time() >= self.deadline:
final = "Stopped: time window over."
self.end_reason = "window_end"
break
if self.max_seconds and time.time() - start > self.max_seconds:
final = f"Stopped: time budget exceeded ({self.max_seconds} s)."
self.end_reason = "time_budget"
@@ -137,6 +198,14 @@ class LlmAgent:
msg, usage = self._chat(messages, tools)
add_usage(self.model, usage, kind="run", ref=proxy.prefix)
self.turn_usage.append(usage)
except WindowEnd:
final = "Stopped: time window over."
self.end_reason = "window_end"
break
except ServerDown as e:
final = f"Stopped: model server not reachable ({e})."
self.end_reason = "server_down"
break
except RuntimeError as e:
final = f"Stopped: {e}"
self.end_reason = "model_error"