diff --git a/harness/agents.py b/harness/agents.py index 1ad5b6e..afb7d4c 100644 --- a/harness/agents.py +++ b/harness/agents.py @@ -70,7 +70,7 @@ class LlmAgent: """ def __init__(self, model, base_url=None, api_key=None, max_turns=80, temperature=0.2, - max_seconds=None, max_tokens=None): + max_seconds=None, max_tokens=None, chat_template_kwargs=None, loop_guard=None): self.model = model self.name = f"llm:{model}" self.base_url = (base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")).rstrip("/") @@ -83,12 +83,17 @@ class LlmAgent: # run cost, 2026-10-03). Normal turns: p95 17.5k, max 82k. A cut turn is retried (see run). self.max_tokens = max_tokens or (32000 if ":cloud" in (model or "") else None) self.empty_retries = 2 + self.loop_guard = loop_guard # end the run after this many identical pushes in a row (None = off) + self.end_reason = None + self.chat_template_kwargs = chat_template_kwargs # local server only, e.g. {"enable_thinking": False} def _chat(self, messages, tools): body = {"model": self.model, "messages": messages, "tools": tools, "temperature": self.temperature, "parallel_tool_calls": False} if self.max_tokens: body["max_tokens"] = self.max_tokens + if self.chat_template_kwargs: + body["chat_template_kwargs"] = self.chat_template_kwargs req = urllib.request.Request(f"{self.base_url}/chat/completions", json.dumps(body).encode(), {"Content-Type": "application/json", "Authorization": f"Bearer {self.api_key}"}) @@ -117,16 +122,19 @@ class LlmAgent: check_budget() final = "" empty = 0 + self.end_reason = "max_turns" start = time.time() for _ in range(self.max_turns): if self.max_seconds and time.time() - start > self.max_seconds: final = f"Stopped: time budget exceeded ({self.max_seconds} s)." + self.end_reason = "time_budget" break try: msg, usage = self._chat(messages, tools) add_usage(self.model, usage, kind="run", ref=proxy.prefix) except RuntimeError as e: final = f"Stopped: {e}" + self.end_reason = "model_error" break proxy.note("assistant", {"content": msg.get("content"), "tool_calls": msg.get("tool_calls"), "usage": usage}) @@ -138,6 +146,7 @@ class LlmAgent: continue if not calls: final = msg.get("content") or ("Stopped: empty model response." if empty else "") + self.end_reason = "report" if (msg.get("content") or "").strip() else "empty_response" break try: for c in calls: @@ -147,8 +156,15 @@ class LlmAgent: err, text = proxy.call(fn["name"], args) messages.append({"role": "tool", "tool_call_id": c.get("id", ""), "content": ("ERROR: " if err else "") + text[:12000]}) + if self.loop_guard and proxy.same_push_streak >= self.loop_guard: + final = f"Stopped: loop (the same source was pushed {proxy.same_push_streak} times in a row)." + self.end_reason = "loop" + break + if self.end_reason == "loop": + break except BudgetExceeded as e: final = f"Stopped: budget exceeded ({e})." + self.end_reason = "tool_budget" break proxy.note("final", final) return final diff --git a/harness/proxy.py b/harness/proxy.py index 44661fd..27fc4e4 100644 --- a/harness/proxy.py +++ b/harness/proxy.py @@ -1,4 +1,5 @@ """Tool proxy between agent and MCP: whitelist, budget, prefix filter, trajectory log.""" +import hashlib import json import re import time @@ -30,6 +31,23 @@ VARIANTS = { "packageName": "package", "includeType": "include", "className": "class_name"}, }, } + + +def activation_messages(text, is_error=False): + """Error messages (severity E) of a write result; the whole text if the call itself failed.""" + try: + obj = json.loads(text) + except (TypeError, ValueError): + return [str(text)[:300]] if is_error else [] + if not isinstance(obj, dict): + return [] + msgs = (obj.get("activation") or {}).get("messages") or obj.get("messages") or [] + out = [str(m.get("message", ""))[:300] for m in msgs if isinstance(m, dict) and m.get("severity") == "E"] + if not out and is_error: + out = [str(obj.get("message") or text)[:300]] + return out + + RUN_PREFIX = re.compile(r"^Z\d[0-9A-Z]{5,6}_", re.I) @@ -49,6 +67,10 @@ class ToolProxy: self.activations = 0 self.fail_streak = 0 self.max_fail_streak = 0 + self.activation_failures = 0 # write calls that failed + self.activation_errors = [] # unique error messages of failed writes + self.same_push_streak = 0 # identical sap_push_source (object + source hash) in a row + self._last_push = None self.log = open(log_path, "a") self.adt = None self.fallbacks = [] # writes that needed the ADT activation fallback @@ -125,6 +147,11 @@ class ToolProxy: else: if tool in WRITE_TOOLS and (tool == "sap_activate" or args.get("activate", True)): self.activations += 1 + if tool == "sap_push_source": + key = (name, str(args.get("includeType", "")), + hashlib.md5(str(args.get("source", "")).encode()).hexdigest()) + self.same_push_streak = self.same_push_streak + 1 if key == self._last_push else 1 + self._last_push = key err, text = self.mcp.call(tool, args) if tool == "sap_push_source": if str(args.get("objectType", "")).upper() in ("PROG", "FUNC") and not args.get("includeType"): @@ -139,6 +166,11 @@ class ToolProxy: failed = err or '"success":false' in text.replace(" ", "") self.fail_streak = self.fail_streak + 1 if failed else 0 self.max_fail_streak = max(self.max_fail_streak, self.fail_streak) + if failed: + self.activation_failures += 1 + for m in activation_messages(text, err): + if m not in self.activation_errors and len(self.activation_errors) < 30: + self.activation_errors.append(m) entry["is_error"], entry["result"] = result[0], result[1][:20000] self.log.write(json.dumps(entry) + "\n") self.log.flush() diff --git a/harness/runner.py b/harness/runner.py index f9a5313..10fafe3 100644 --- a/harness/runner.py +++ b/harness/runner.py @@ -61,7 +61,9 @@ def _norm(src): def trajectory_stats(path): """Rebuild agent statistics from a trajectory (same rules as ToolProxy).""" from .proxy import WRITE_TOOLS - calls = activations = streak = max_streak = 0 + from .proxy import activation_messages + calls = activations = streak = max_streak = fails = 0 + errors = [] final, t_first, t_last = "", None, None for line in open(path): e = json.loads(line) @@ -79,7 +81,13 @@ def trajectory_stats(path): failed = e["is_error"] or '"success":false' in e["result"].replace(" ", "") streak = streak + 1 if failed else 0 max_streak = max(max_streak, streak) + if failed: + fails += 1 + for m in activation_messages(e["result"], e["is_error"]): + if m not in errors and len(errors) < 30: + errors.append(m) return {"tool_calls": calls, "activations": activations, "max_fail_streak": max_streak, + "activation_failures": fails, "activation_error_messages": errors, "final_report": final, "agent_seconds": round((t_last or 0) - (t_first or 0), 1)} @@ -207,6 +215,9 @@ class Runner: rep["agent_seconds"] = round(time.time() - t1, 1) rep["tool_calls"], rep["activations"] = proxy.calls, proxy.activations rep["max_fail_streak"] = proxy.max_fail_streak + rep["activation_failures"] = proxy.activation_failures + rep["activation_error_messages"] = proxy.activation_errors + rep["end_reason"] = getattr(agent, "end_reason", None) if proxy.fallbacks: rep["adt_fallbacks"] = proxy.fallbacks diff --git a/train/README.md b/train/README.md index 91aec30..ec0d268 100644 --- a/train/README.md +++ b/train/README.md @@ -53,15 +53,18 @@ G0139, G0151, G0157, G0174, G0167, G0185. Use the same list before and after tra | Setting | Value | |---|---| | Model | `~/models/Qwen3.8-27B-4bit` (MLX affine 4 bit); after training the same with `--adapter-path` | -| Thinking | on, `reasoning_effort` medium | +| Thinking | **off (`enable_thinking: false`), fixed for baseline and after training.** Sent per request as `chat_template_kwargs` by `train/baseline.py` (the server default stays thinking on). Reason: with thinking on, all 3 baseline runs ended with empty responses at the thinking limit (`runs/stage1/baseline_thinking_on.json`) | | temperature / top_p / top_k / min_p | 0.2 / 0.95 / 20 / 0 | | max_tokens per turn | **16384**, sent in each request by `train/baseline.py` (`MAX_TOKENS`); the server limit stays 32768 | | Tool-call budget per task | 60 calls, 15 activations (T01 too) | +| Loop guard | `loop_guard` 3: the run ends when `sap_push_source` pushes the same source (object + md5) 3 times in a row; final report "Stopped: loop ...", `end_reason` "loop". Scores use the final state, so they do not change. Same after training. T01 of the first run (started before the guard) ran without it | +| Per run record | `end_reason` (report, loop, empty_response, tool_budget, time_budget, model_error, max_turns), `activation_failures`, `activation_error_messages` (unique), also in `baseline.json` | | Empty turn | retried (2 times), then the run stops ("Stopped: empty model response") | +| Docker (A4H) | VM memory 36 GB (`MemoryMiB` 36864), container `--memory 32g --memory-swap 32g` (2026-10-04) | | Prompt cache of the server | `--prompt-cache-size 4 --prompt-cache-bytes 6000000000` | -The settings are also written into `runs/stage1/baseline.json` (`settings`). The old Ollama run (41.7) and the -T01 test with 32768 tokens and budget 40 (`t01_test_budget40`: 40.0) are not comparable. +The settings are also written into `runs/stage1/baseline.json` (`settings`). The old Ollama run (41.7), the +T01 test with 32768 tokens and budget 40 (`t01_test_budget40`: 40.0) and the thinking-on runs are not comparable. ## Baseline diff --git a/train/STATE.md b/train/STATE.md index 3c1a534..5368a8f 100644 --- a/train/STATE.md +++ b/train/STATE.md @@ -36,6 +36,11 @@ Task: `docs/stage1-training-task.md`. Settings and weights: `train/README.md`. U - The earlier night chain (25 tasks) was stopped with its first T01 run (aborted at 33 tool calls after 2 h; A4H objects deleted; directory `runs/stage1/baseline/_aborted_20100_T01`). +- 2026-10-04: A4H memory: Docker VM 36 GB (was 64 GB = all RAM), container 32 GB; swap went from 15 GB to 1.4 GB. + Baseline switched to thinking off (`enable_thinking: false` per request); thinking-on results kept in + `runs/stage1/baseline_thinking_on.json`. Loop guard (3 identical pushes) and per-run activation error records + added (`harness/agents.py`, `harness/proxy.py`, `harness/runner.py`, `train/baseline.py`). + ## Next 0. After the baseline ends: Step 3 training test (20 iterations), Kral stops A4H first and the MLX server is stopped. diff --git a/train/baseline.py b/train/baseline.py index d4d17a6..b4d58fb 100644 --- a/train/baseline.py +++ b/train/baseline.py @@ -17,6 +17,7 @@ from harness.runner import Runner # noqa: E402 MODEL = os.path.expanduser("~/models/Qwen3.8-27B-4bit") BASE_URL = "http://127.0.0.1:8080/v1" +LOOP_GUARD = 3 # end the run after 3 identical pushes in a row (end_reason "loop"); same for after training MAX_TOKENS = 16384 # per turn; sent in each request (the server limit stays 32768). Same for before/after. @@ -31,15 +32,16 @@ def main(): out_path = os.path.join(ROOT, "runs", "stage1", f"{a.label}.json") os.makedirs(os.path.dirname(out_path), exist_ok=True) res = json.load(open(out_path)) if os.path.exists(out_path) else {"model": MODEL, "tasks": {}} - res["settings"] = {"max_tokens": MAX_TOKENS, "reasoning_effort": "medium", "temperature": 0.2, "tool_call_budget": 60, - "subset": "train/subset.json", "date": "2026-10-03"} + res["settings"] = {"max_tokens": MAX_TOKENS, "enable_thinking": False, "temperature": 0.2, "tool_call_budget": 60, "loop_guard": LOOP_GUARD, + "subset": "train/subset.json", "date": "2026-10-04"} runs_root = os.path.join(ROOT, "runs", "stage1", a.label) for i, t in enumerate(subset): tid = t["id"] if (a.only and tid not in a.only) or tid in res["tasks"]: continue pool = os.path.join(ROOT, "tasks") if tid.startswith("T") else os.path.join(ROOT, "tasks_gen", "eval") - agent = LlmAgent(MODEL, BASE_URL, max_tokens=MAX_TOKENS) + agent = LlmAgent(MODEL, BASE_URL, max_tokens=MAX_TOKENS, chat_template_kwargs={"enable_thinking": False}, + loop_guard=LOOP_GUARD) try: rep, run_dir = Runner(pool, runs_root).run(tid, agent, a.run_base + i) except Exception as e: # noqa: BLE001 one broken run must not stop the series @@ -52,6 +54,8 @@ def main(): "score": (rep.get("score") or {}).get("total"), "parts": rep.get("score"), "gates": rep.get("gates"), "hidden": f"{h.get('passed')}/{h.get('total')}", "tool_calls": rep.get("tool_calls"), "seconds": rep.get("seconds"), + "end_reason": rep.get("end_reason"), "activation_failures": rep.get("activation_failures"), + "activation_error_messages": rep.get("activation_error_messages"), "agent_seconds": rep.get("agent_seconds"), "final": (rep.get("final_report") or "")[:300], "run_dir": os.path.relpath(run_dir, ROOT)} json.dump(res, open(out_path, "w"), indent=1)