From 01e99158a9ce917d830c1845d817a86aba4f7ee7 Mon Sep 17 00:00:00 2001 From: Kral Date: Sat, 3 Oct 2026 17:07:31 +0200 Subject: [PATCH] Agent: max_tokens 32k for cloud models and retry on empty turns (runaway reasoning); proxy: sap_activate fallback for PROG/FUNC Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat --- harness/agents.py | 13 ++++++++++++- harness/proxy.py | 7 +++++++ 2 files changed, 19 insertions(+), 1 deletion(-) diff --git a/harness/agents.py b/harness/agents.py index 5b8d77e..366f454 100644 --- a/harness/agents.py +++ b/harness/agents.py @@ -79,10 +79,16 @@ class LlmAgent: self.temperature = temperature self.max_seconds = max_seconds self.request_timeout = 3600 # local models are slow; a hung request still ends + # Runaway reasoning: 35 of 1965 DeepSeek turns produced 393k output tokens and no content (46 % of the + # run cost, 2026-10-03). Normal turns: p95 17.5k, max 82k. A cut turn is retried (see run). + self.max_tokens = 32000 if ":cloud" in (model or "") else None + self.empty_retries = 2 def _chat(self, messages, tools): body = {"model": self.model, "messages": messages, "tools": tools, "temperature": self.temperature, "parallel_tool_calls": False} + if self.max_tokens: + body["max_tokens"] = self.max_tokens req = urllib.request.Request(f"{self.base_url}/chat/completions", json.dumps(body).encode(), {"Content-Type": "application/json", "Authorization": f"Bearer {self.api_key}"}) @@ -110,6 +116,7 @@ class LlmAgent: if ":cloud" in self.model: check_budget() final = "" + empty = 0 start = time.time() for _ in range(self.max_turns): if self.max_seconds and time.time() - start > self.max_seconds: @@ -125,8 +132,12 @@ class LlmAgent: "tool_calls": msg.get("tool_calls"), "usage": usage}) messages.append({k: v for k, v in msg.items() if k in ("role", "content", "tool_calls")}) calls = msg.get("tool_calls") or [] + if not calls and not (msg.get("content") or "").strip() and empty < self.empty_retries: + empty += 1 # empty turn (output limit reached in reasoning): ask the same question again + messages.pop() + continue if not calls: - final = msg.get("content") or "" + final = msg.get("content") or ("Stopped: empty model response." if empty else "") break try: for c in calls: diff --git a/harness/proxy.py b/harness/proxy.py index 69fba03..44661fd 100644 --- a/harness/proxy.py +++ b/harness/proxy.py @@ -52,6 +52,7 @@ class ToolProxy: self.log = open(log_path, "a") self.adt = None self.fallbacks = [] # writes that needed the ADT activation fallback + self.last_source = {} # last main source pushed per PROG/FUNC, for the sap_activate fallback def schemas(self): tools = [t for t in self.mcp.list_tools() if t["name"] in MODEL_TOOLS] @@ -126,7 +127,13 @@ class ToolProxy: self.activations += 1 err, text = self.mcp.call(tool, args) if tool == "sap_push_source": + if str(args.get("objectType", "")).upper() in ("PROG", "FUNC") and not args.get("includeType"): + self.last_source[str(args.get("objectName", "")).upper()] = args.get("source", "") err, text = self._activation_fallback(args, err, text) + elif tool == "sap_activate": # a write with activate=false, then sap_activate (G0014, G0124) + src = self.last_source.get(str(args.get("objectName", "")).upper()) + if src is not None: + err, text = self._activation_fallback(dict(args, source=src), err, text) result = (err, self._filter(tool, text)) if tool in WRITE_TOOLS: failed = err or '"success":false' in text.replace(" ", "")