Series A: local Qwen on the MacBook (remote server), 12 h window with hard stop, clean pause on server loss, dashboard card, docs/remote-model.md
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -4,9 +4,20 @@ import os
|
||||
import time
|
||||
import urllib.request
|
||||
|
||||
import threading
|
||||
import urllib.error
|
||||
|
||||
from .ledger import add_usage, check_budget
|
||||
from .proxy import BudgetExceeded
|
||||
|
||||
|
||||
class ServerDown(Exception):
|
||||
"""The model server does not answer (remote local model). The run ends cleanly and is not a result."""
|
||||
|
||||
|
||||
class WindowEnd(Exception):
|
||||
"""The time window of the series is over (hard stop, also in the middle of a request)."""
|
||||
|
||||
SYSTEM_PROMPT = """You are an ABAP developer. You implement a plan on an SAP system with the tools.
|
||||
|
||||
Rules:
|
||||
@@ -72,7 +83,8 @@ class LlmAgent:
|
||||
"""
|
||||
|
||||
def __init__(self, model, base_url=None, api_key=None, max_turns=80, temperature=0.2,
|
||||
max_seconds=None, max_tokens=None, chat_template_kwargs=None, loop_guard=None):
|
||||
max_seconds=None, max_tokens=None, chat_template_kwargs=None, loop_guard=None,
|
||||
deadline=None, watch=False):
|
||||
self.model = model
|
||||
self.name = f"llm:{model}"
|
||||
self.base_url = (base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")).rstrip("/")
|
||||
@@ -86,6 +98,8 @@ class LlmAgent:
|
||||
self.max_tokens = max_tokens or (32000 if ":cloud" in (model or "") else None)
|
||||
self.empty_retries = 2
|
||||
self.loop_guard = loop_guard # end the run after this many identical pushes in a row (None = off)
|
||||
self.deadline = deadline # absolute time (time.time()) of the hard stop, or None
|
||||
self.watch = watch # remote model: ping the server during a request, end the run when it is gone
|
||||
self.end_reason = None
|
||||
self.messages, self.tools, self.reasoning, self.turn_usage = [], [], [], [] # for the trajectory record
|
||||
self.chat_template_kwargs = chat_template_kwargs # local server only, e.g. {"enable_thinking": False}
|
||||
@@ -103,17 +117,60 @@ class LlmAgent:
|
||||
last = None
|
||||
for attempt in range(4): # model server errors (HTTP 5xx, timeouts): retry with backoff
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=self.request_timeout) as r:
|
||||
data = json.loads(r.read().decode())
|
||||
return data["choices"][0]["message"], data.get("usage", {})
|
||||
if self.watch or self.deadline:
|
||||
data = self._post_watched(req)
|
||||
else:
|
||||
with urllib.request.urlopen(req, timeout=self.request_timeout) as r:
|
||||
data = json.loads(r.read().decode())
|
||||
return data["choices"][0]["message"], data.get("usage", {})
|
||||
except (ServerDown, WindowEnd):
|
||||
raise
|
||||
except Exception as e: # noqa: BLE001
|
||||
last = e
|
||||
code = getattr(e, "code", None)
|
||||
if code is not None and code < 500 and code != 429:
|
||||
raise
|
||||
if self.deadline and time.time() >= self.deadline:
|
||||
raise WindowEnd()
|
||||
if self.watch and not self._ping():
|
||||
raise ServerDown(str(e)[:200])
|
||||
time.sleep(10 * (attempt + 1))
|
||||
raise RuntimeError(f"model request failed after retries: {last}")
|
||||
|
||||
def _ping(self, timeout=10):
|
||||
try:
|
||||
urllib.request.urlopen(urllib.request.Request(f"{self.base_url}/models"), timeout=timeout).read()
|
||||
return True
|
||||
except Exception: # noqa: BLE001
|
||||
return False
|
||||
|
||||
def _post_watched(self, req):
|
||||
"""The request runs in a thread; this thread watches the deadline and (remote model) the server: a hard stop
|
||||
or a dead server ends the wait at once, also in the middle of a long generation."""
|
||||
box = {}
|
||||
|
||||
def work():
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=self.request_timeout) as r:
|
||||
box["data"] = json.loads(r.read().decode())
|
||||
except BaseException as e: # noqa: BLE001
|
||||
box["err"] = e
|
||||
th = threading.Thread(target=work, daemon=True)
|
||||
th.start()
|
||||
last_ping, fails = time.time(), 0
|
||||
while th.is_alive():
|
||||
th.join(5)
|
||||
if self.deadline and time.time() >= self.deadline:
|
||||
raise WindowEnd()
|
||||
if self.watch and th.is_alive() and time.time() - last_ping >= 30:
|
||||
last_ping = time.time()
|
||||
fails = 0 if self._ping() else fails + 1
|
||||
if fails >= 3:
|
||||
raise ServerDown("no answer to 3 pings in a row during a request")
|
||||
if "err" in box:
|
||||
raise box["err"]
|
||||
return box["data"]
|
||||
|
||||
def run(self, task, proxy):
|
||||
tools = [{"type": "function", "function": {"name": t["name"],
|
||||
"description": t.get("description", ""),
|
||||
@@ -129,6 +186,10 @@ class LlmAgent:
|
||||
self.end_reason = "max_turns"
|
||||
start = time.time()
|
||||
for _ in range(self.max_turns):
|
||||
if self.deadline and time.time() >= self.deadline:
|
||||
final = "Stopped: time window over."
|
||||
self.end_reason = "window_end"
|
||||
break
|
||||
if self.max_seconds and time.time() - start > self.max_seconds:
|
||||
final = f"Stopped: time budget exceeded ({self.max_seconds} s)."
|
||||
self.end_reason = "time_budget"
|
||||
@@ -137,6 +198,14 @@ class LlmAgent:
|
||||
msg, usage = self._chat(messages, tools)
|
||||
add_usage(self.model, usage, kind="run", ref=proxy.prefix)
|
||||
self.turn_usage.append(usage)
|
||||
except WindowEnd:
|
||||
final = "Stopped: time window over."
|
||||
self.end_reason = "window_end"
|
||||
break
|
||||
except ServerDown as e:
|
||||
final = f"Stopped: model server not reachable ({e})."
|
||||
self.end_reason = "server_down"
|
||||
break
|
||||
except RuntimeError as e:
|
||||
final = f"Stopped: {e}"
|
||||
self.end_reason = "model_error"
|
||||
|
||||
Reference in New Issue
Block a user