Devstral Small 2: serve.sh, baseline.py without thinking args, repair rate, activation error message fix, stop rule in chain (run base 21000)

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat
This commit is contained in:
Kral
2026-10-04 08:57:12 +02:00
parent e84ea43a3d
commit 8d1db9c67c
5 changed files with 56 additions and 21 deletions

View File

@@ -13,9 +13,9 @@ ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, ROOT)
from harness.adt_client import load_env # noqa: E402
from harness.agents import LlmAgent # noqa: E402
from harness.runner import Runner # noqa: E402
from harness.runner import Runner, repair_stats # noqa: E402
MODEL = os.path.expanduser("~/models/Qwen3.8-27B-4bit")
MODEL = os.path.expanduser("~/models/Devstral-Small-2-24B-4bit")
BASE_URL = "http://127.0.0.1:8080/v1"
LOOP_GUARD = 3 # end the run after 3 identical pushes in a row (end_reason "loop"); same for after training
MAX_TOKENS = 16384 # per turn; sent in each request (the server limit stays 32768). Same for before/after.
@@ -32,7 +32,7 @@ def main():
out_path = os.path.join(ROOT, "runs", "stage1", f"{a.label}.json")
os.makedirs(os.path.dirname(out_path), exist_ok=True)
res = json.load(open(out_path)) if os.path.exists(out_path) else {"model": MODEL, "tasks": {}}
res["settings"] = {"max_tokens": MAX_TOKENS, "enable_thinking": False, "temperature": 0.2, "tool_call_budget": 60, "loop_guard": LOOP_GUARD,
res["settings"] = {"max_tokens": MAX_TOKENS, "enable_thinking": None, "temperature": 0.2, "tool_call_budget": 60, "loop_guard": LOOP_GUARD,
"subset": "train/subset.json", "date": "2026-10-04"}
runs_root = os.path.join(ROOT, "runs", "stage1", a.label)
for i, t in enumerate(subset):
@@ -40,8 +40,7 @@ def main():
if (a.only and tid not in a.only) or tid in res["tasks"]:
continue
pool = os.path.join(ROOT, "tasks") if tid.startswith("T") else os.path.join(ROOT, "tasks_gen", "eval")
agent = LlmAgent(MODEL, BASE_URL, max_tokens=MAX_TOKENS, chat_template_kwargs={"enable_thinking": False},
loop_guard=LOOP_GUARD)
agent = LlmAgent(MODEL, BASE_URL, max_tokens=MAX_TOKENS, loop_guard=LOOP_GUARD)
try:
rep, run_dir = Runner(pool, runs_root).run(tid, agent, a.run_base + i)
except Exception as e: # noqa: BLE001 one broken run must not stop the series
@@ -56,6 +55,7 @@ def main():
"tool_calls": rep.get("tool_calls"), "seconds": rep.get("seconds"),
"end_reason": rep.get("end_reason"), "activation_failures": rep.get("activation_failures"),
"activation_error_messages": rep.get("activation_error_messages"),
**repair_stats(os.path.join(run_dir, "trajectory.jsonl")),
"agent_seconds": rep.get("agent_seconds"), "final": (rep.get("final_report") or "")[:300],
"run_dir": os.path.relpath(run_dir, ROOT)}
json.dump(res, open(out_path, "w"), indent=1)

View File

@@ -1,20 +1,36 @@
#!/bin/sh
# Baseline on the 11-task subset (Kral decision 2026-10-03), detached. Replaces night_chain.sh.
# macOS notification after 2 tasks (time estimate) and at the end. Log: runs/stage1/baseline.log, chain log below.
# Baseline on the 11-task subset, Devstral Small 2 (2026-10-04), detached.
# Stop rule: after the first 4 tasks, if all 4 ended with "loop" and the repair rate is below 20 %, stop the run.
# Log: runs/stage1/baseline.log, chain log runs/stage1/baseline_chain.log
cd "$(dirname "$0")/.."
log() { echo "$(date '+%F %T') $*"; }
log "start baseline, 11 tasks, thinking off, max_tokens 16384, loop guard 3"
python3 train/baseline.py --label baseline --run-base 20500 > runs/stage1/baseline.log 2>&1 &
log "start baseline, 11 tasks, no thinking mode (Devstral), max_tokens 16384, loop guard 3"
python3 train/baseline.py --label baseline_devstral --run-base 21000 > runs/stage1/baseline.log 2>&1 &
BP=$!
log "baseline PID $BP"
NOTE=0
CHECKED=0
while kill -0 $BP 2>/dev/null; do
N=$(python3 -c "import json;print(len(json.load(open('runs/stage1/baseline.json'))['tasks']))" 2>/dev/null || echo 0)
if [ "$NOTE" = 0 ] && [ "$N" -ge 2 ]; then
osascript -e 'display notification "Baseline: first 2 tasks done, ask for the time estimate" with title "stage1"'
log "2 tasks done"; NOTE=1
if [ "$CHECKED" = 0 ]; then
R=$(python3 - <<'PY' 2>/dev/null
import json
t = json.load(open("runs/stage1/baseline_devstral.json"))["tasks"]
if len(t) < 4:
print("wait")
else:
f = list(t.values())[:4]
after = sum(x.get("pushes_after_error") or 0 for x in f)
ch = sum(x.get("pushes_changed_after_error") or 0 for x in f)
rate = ch / after if after else 0
loops = sum(x.get("end_reason") == "loop" for x in f)
print("stop" if loops == 4 and rate < 0.2 else "go", loops, round(rate, 3))
PY
)
case "$R" in
stop*) log "STOP RULE: $R"; kill $BP; osascript -e 'display notification "Stop rule: 4 loops, repair rate below 20 %" with title "stage1"'; exit 0 ;;
go*) log "first 4 tasks checked: $R"; CHECKED=1 ;;
esac
fi
sleep 60
done
log "baseline ended: $(python3 -c "import json;print(len(json.load(open('runs/stage1/baseline.json'))['tasks']))") tasks"
osascript -e 'display notification "Baseline (11 tasks) ended" with title "stage1"'
log "baseline ended: $(python3 -c "import json;print(len(json.load(open('runs/stage1/baseline_devstral.json'))['tasks']))") tasks"
osascript -e 'display notification "Baseline Devstral (11 tasks) ended" with title "stage1"'

View File

@@ -1,13 +1,12 @@
#!/bin/sh
# Serve the base model (no adapter) with mlx_lm.server. Settings: train/README.md.
# Base model: Devstral Small 2 (no thinking mode, no chat-template args). Serve the base model (no adapter) with mlx_lm.server. Settings: train/README.md.
# Usage: train/serve.sh [--adapter-path PATH]
# Prompt cache limit: without it the cache grew to 7.9 GB in 15 min (2026-10-03); A4H needs 22 GB.
cd "$(dirname "$0")/.."
exec train/.venv/bin/mlx_lm.server \
--model "$HOME/models/Qwen3.8-27B-4bit" \
--model "$HOME/models/Devstral-Small-2-24B-4bit" \
--host 127.0.0.1 --port 8080 \
--temp 0.2 --top-p 0.95 --top-k 20 --min-p 0 \
--max-tokens 32768 \
--prompt-cache-size 4 --prompt-cache-bytes 6000000000 \
--chat-template-args '{"enable_thinking": true, "reasoning_effort": "medium"}' \
"$@"