diff --git a/harness/proxy.py b/harness/proxy.py index 786b601..d3698a2 100644 --- a/harness/proxy.py +++ b/harness/proxy.py @@ -43,8 +43,8 @@ def activation_messages(text, is_error=False): return [] msgs = (obj.get("activation") or {}).get("messages") or obj.get("messages") or [] out = [str(m.get("message", ""))[:300] for m in msgs if isinstance(m, dict) and m.get("severity") == "E"] - if not out and is_error: - out = [str(obj.get("message") or text)[:300]] + if not out and (is_error or obj.get("success") is False): + out = [str(obj.get("error") or obj.get("message") or text)[:300]] # {"success":false,"error":...} of a failed write return out diff --git a/harness/runner.py b/harness/runner.py index 10fafe3..89a9d48 100644 --- a/harness/runner.py +++ b/harness/runner.py @@ -91,6 +91,26 @@ def trajectory_stats(path): "final_report": final, "agent_seconds": round((t_last or 0) - (t_first or 0), 1)} +def repair_stats(path): + """Repair rate: pushes whose source changed after a failed push of the same object / pushes after a failed push.""" + import hashlib + last = None # (object, md5, failed) + after = changed = 0 + for line in open(path): + e = json.loads(line) + if e.get("tool") != "sap_push_source" or (e["result"].startswith("Tool ") and e["is_error"]): + continue + args = e.get("args", {}) + key = (str(args.get("objectName", "")).upper(), str(args.get("includeType", ""))) + md5 = hashlib.md5(str(args.get("source", "")).encode()).hexdigest() + if last and last[2] and last[0] == key: + after += 1 + changed += md5 != last[1] + last = (key, md5, e["is_error"] or '"success":false' in e["result"].replace(" ", "")) + return {"pushes_after_error": after, "pushes_changed_after_error": changed, + "repair_rate": round(changed / after, 3) if after else None} + + class Runner: def __init__(self, tasks_root, runs_root): self.tasks_root = tasks_root diff --git a/train/baseline.py b/train/baseline.py index b4d58fb..924cc59 100644 --- a/train/baseline.py +++ b/train/baseline.py @@ -13,9 +13,9 @@ ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) sys.path.insert(0, ROOT) from harness.adt_client import load_env # noqa: E402 from harness.agents import LlmAgent # noqa: E402 -from harness.runner import Runner # noqa: E402 +from harness.runner import Runner, repair_stats # noqa: E402 -MODEL = os.path.expanduser("~/models/Qwen3.8-27B-4bit") +MODEL = os.path.expanduser("~/models/Devstral-Small-2-24B-4bit") BASE_URL = "http://127.0.0.1:8080/v1" LOOP_GUARD = 3 # end the run after 3 identical pushes in a row (end_reason "loop"); same for after training MAX_TOKENS = 16384 # per turn; sent in each request (the server limit stays 32768). Same for before/after. @@ -32,7 +32,7 @@ def main(): out_path = os.path.join(ROOT, "runs", "stage1", f"{a.label}.json") os.makedirs(os.path.dirname(out_path), exist_ok=True) res = json.load(open(out_path)) if os.path.exists(out_path) else {"model": MODEL, "tasks": {}} - res["settings"] = {"max_tokens": MAX_TOKENS, "enable_thinking": False, "temperature": 0.2, "tool_call_budget": 60, "loop_guard": LOOP_GUARD, + res["settings"] = {"max_tokens": MAX_TOKENS, "enable_thinking": None, "temperature": 0.2, "tool_call_budget": 60, "loop_guard": LOOP_GUARD, "subset": "train/subset.json", "date": "2026-10-04"} runs_root = os.path.join(ROOT, "runs", "stage1", a.label) for i, t in enumerate(subset): @@ -40,8 +40,7 @@ def main(): if (a.only and tid not in a.only) or tid in res["tasks"]: continue pool = os.path.join(ROOT, "tasks") if tid.startswith("T") else os.path.join(ROOT, "tasks_gen", "eval") - agent = LlmAgent(MODEL, BASE_URL, max_tokens=MAX_TOKENS, chat_template_kwargs={"enable_thinking": False}, - loop_guard=LOOP_GUARD) + agent = LlmAgent(MODEL, BASE_URL, max_tokens=MAX_TOKENS, loop_guard=LOOP_GUARD) try: rep, run_dir = Runner(pool, runs_root).run(tid, agent, a.run_base + i) except Exception as e: # noqa: BLE001 one broken run must not stop the series @@ -56,6 +55,7 @@ def main(): "tool_calls": rep.get("tool_calls"), "seconds": rep.get("seconds"), "end_reason": rep.get("end_reason"), "activation_failures": rep.get("activation_failures"), "activation_error_messages": rep.get("activation_error_messages"), + **repair_stats(os.path.join(run_dir, "trajectory.jsonl")), "agent_seconds": rep.get("agent_seconds"), "final": (rep.get("final_report") or "")[:300], "run_dir": os.path.relpath(run_dir, ROOT)} json.dump(res, open(out_path, "w"), indent=1) diff --git a/train/baseline_chain.sh b/train/baseline_chain.sh index f8b8484..4fcdff1 100755 --- a/train/baseline_chain.sh +++ b/train/baseline_chain.sh @@ -1,20 +1,36 @@ #!/bin/sh -# Baseline on the 11-task subset (Kral decision 2026-10-03), detached. Replaces night_chain.sh. -# macOS notification after 2 tasks (time estimate) and at the end. Log: runs/stage1/baseline.log, chain log below. +# Baseline on the 11-task subset, Devstral Small 2 (2026-10-04), detached. +# Stop rule: after the first 4 tasks, if all 4 ended with "loop" and the repair rate is below 20 %, stop the run. +# Log: runs/stage1/baseline.log, chain log runs/stage1/baseline_chain.log cd "$(dirname "$0")/.." log() { echo "$(date '+%F %T') $*"; } -log "start baseline, 11 tasks, thinking off, max_tokens 16384, loop guard 3" -python3 train/baseline.py --label baseline --run-base 20500 > runs/stage1/baseline.log 2>&1 & +log "start baseline, 11 tasks, no thinking mode (Devstral), max_tokens 16384, loop guard 3" +python3 train/baseline.py --label baseline_devstral --run-base 21000 > runs/stage1/baseline.log 2>&1 & BP=$! log "baseline PID $BP" -NOTE=0 +CHECKED=0 while kill -0 $BP 2>/dev/null; do - N=$(python3 -c "import json;print(len(json.load(open('runs/stage1/baseline.json'))['tasks']))" 2>/dev/null || echo 0) - if [ "$NOTE" = 0 ] && [ "$N" -ge 2 ]; then - osascript -e 'display notification "Baseline: first 2 tasks done, ask for the time estimate" with title "stage1"' - log "2 tasks done"; NOTE=1 + if [ "$CHECKED" = 0 ]; then + R=$(python3 - <<'PY' 2>/dev/null +import json +t = json.load(open("runs/stage1/baseline_devstral.json"))["tasks"] +if len(t) < 4: + print("wait") +else: + f = list(t.values())[:4] + after = sum(x.get("pushes_after_error") or 0 for x in f) + ch = sum(x.get("pushes_changed_after_error") or 0 for x in f) + rate = ch / after if after else 0 + loops = sum(x.get("end_reason") == "loop" for x in f) + print("stop" if loops == 4 and rate < 0.2 else "go", loops, round(rate, 3)) +PY +) + case "$R" in + stop*) log "STOP RULE: $R"; kill $BP; osascript -e 'display notification "Stop rule: 4 loops, repair rate below 20 %" with title "stage1"'; exit 0 ;; + go*) log "first 4 tasks checked: $R"; CHECKED=1 ;; + esac fi sleep 60 done -log "baseline ended: $(python3 -c "import json;print(len(json.load(open('runs/stage1/baseline.json'))['tasks']))") tasks" -osascript -e 'display notification "Baseline (11 tasks) ended" with title "stage1"' +log "baseline ended: $(python3 -c "import json;print(len(json.load(open('runs/stage1/baseline_devstral.json'))['tasks']))") tasks" +osascript -e 'display notification "Baseline Devstral (11 tasks) ended" with title "stage1"' diff --git a/train/serve.sh b/train/serve.sh index 148d701..68db2b2 100755 --- a/train/serve.sh +++ b/train/serve.sh @@ -1,13 +1,12 @@ #!/bin/sh -# Serve the base model (no adapter) with mlx_lm.server. Settings: train/README.md. +# Base model: Devstral Small 2 (no thinking mode, no chat-template args). Serve the base model (no adapter) with mlx_lm.server. Settings: train/README.md. # Usage: train/serve.sh [--adapter-path PATH] # Prompt cache limit: without it the cache grew to 7.9 GB in 15 min (2026-10-03); A4H needs 22 GB. cd "$(dirname "$0")/.." exec train/.venv/bin/mlx_lm.server \ - --model "$HOME/models/Qwen3.8-27B-4bit" \ + --model "$HOME/models/Devstral-Small-2-24B-4bit" \ --host 127.0.0.1 --port 8080 \ --temp 0.2 --top-p 0.95 --top-k 20 --min-p 0 \ --max-tokens 32768 \ --prompt-cache-size 4 --prompt-cache-bytes 6000000000 \ - --chat-template-args '{"enable_thinking": true, "reasoning_effort": "medium"}' \ "$@"