From 689820af4dc94a5521c59a2a62c07ca51dc7c782 Mon Sep 17 00:00:00 2001 From: Kral Date: Sat, 3 Oct 2026 22:36:00 +0200 Subject: [PATCH] Baseline shortened: 11-task subset (1 per category + T01), max_tokens 16384, settings in README; baseline_chain.sh Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat --- harness/agents.py | 4 +- train/README.md | 27 ++++++-- train/baseline.py | 5 +- train/baseline_chain.sh | 20 ++++++ train/subset.json | 95 ++------------------------ train/subset_v1_25.json | 146 ++++++++++++++++++++++++++++++++++++++++ 6 files changed, 200 insertions(+), 97 deletions(-) create mode 100755 train/baseline_chain.sh create mode 100644 train/subset_v1_25.json diff --git a/harness/agents.py b/harness/agents.py index 366f454..1ad5b6e 100644 --- a/harness/agents.py +++ b/harness/agents.py @@ -70,7 +70,7 @@ class LlmAgent: """ def __init__(self, model, base_url=None, api_key=None, max_turns=80, temperature=0.2, - max_seconds=None): + max_seconds=None, max_tokens=None): self.model = model self.name = f"llm:{model}" self.base_url = (base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")).rstrip("/") @@ -81,7 +81,7 @@ class LlmAgent: self.request_timeout = 3600 # local models are slow; a hung request still ends # Runaway reasoning: 35 of 1965 DeepSeek turns produced 393k output tokens and no content (46 % of the # run cost, 2026-10-03). Normal turns: p95 17.5k, max 82k. A cut turn is retried (see run). - self.max_tokens = 32000 if ":cloud" in (model or "") else None + self.max_tokens = max_tokens or (32000 if ":cloud" in (model or "") else None) self.empty_retries = 2 def _chat(self, messages, tools): diff --git a/train/README.md b/train/README.md index 79f2c63..91aec30 100644 --- a/train/README.md +++ b/train/README.md @@ -43,12 +43,27 @@ qwen3_coder XML tool format; `mlx_lm` parses it. First request: 86 s (prompt of ## Eval subset -`train/subset.json` (copy: `runs/stage1/subset.json`): 25 tasks, balanced over the categories -(A, B, C, E, F: 3 each; D, G, H, I, K: 2 each), with T01. Use the same list before and after training. +`train/subset.json` (copy: `runs/stage1/subset.json`): **11 tasks** (Kral decision 2026-10-03): one task per +category (A B C D E F G H I K) plus T01. The first subset of 25 tasks (`train/subset_v1_25.json`) was cut because +one task needed 2-3 hours with the MLX 4-bit model (about 12 tokens/s). Tasks: T01, G0105, G0017, G0125, G0128, +G0139, G0151, G0157, G0174, G0167, G0185. Use the same list before and after training. + +## Settings of the stage 1 runs (the same for baseline and after training) + +| Setting | Value | +|---|---| +| Model | `~/models/Qwen3.8-27B-4bit` (MLX affine 4 bit); after training the same with `--adapter-path` | +| Thinking | on, `reasoning_effort` medium | +| temperature / top_p / top_k / min_p | 0.2 / 0.95 / 20 / 0 | +| max_tokens per turn | **16384**, sent in each request by `train/baseline.py` (`MAX_TOKENS`); the server limit stays 32768 | +| Tool-call budget per task | 60 calls, 15 activations (T01 too) | +| Empty turn | retried (2 times), then the run stops ("Stopped: empty model response") | +| Prompt cache of the server | `--prompt-cache-size 4 --prompt-cache-bytes 6000000000` | + +The settings are also written into `runs/stage1/baseline.json` (`settings`). The old Ollama run (41.7) and the +T01 test with 32768 tokens and budget 40 (`t01_test_budget40`: 40.0) are not comparable. ## Baseline -- Runner: `python3 train/baseline.py --label baseline` (one task at a time; results - `runs/stage1/baseline.json`, run directories `runs/stage1/baseline/`). -- The earlier T01 score 41.7 (run 103) used the Ollama NVFP4 weights. The baseline of 2026-10-03 with the - MLX 4-bit weights is the new reference for stage 1. +- Runner: `python3 train/baseline.py --label baseline --run-base 20200` (one task at a time; results + `runs/stage1/baseline.json`, run directories `runs/stage1/baseline/`). Started by `train/baseline_chain.sh`. diff --git a/train/baseline.py b/train/baseline.py index a628bf5..d4d17a6 100644 --- a/train/baseline.py +++ b/train/baseline.py @@ -17,6 +17,7 @@ from harness.runner import Runner # noqa: E402 MODEL = os.path.expanduser("~/models/Qwen3.8-27B-4bit") BASE_URL = "http://127.0.0.1:8080/v1" +MAX_TOKENS = 16384 # per turn; sent in each request (the server limit stays 32768). Same for before/after. def main(): @@ -30,13 +31,15 @@ def main(): out_path = os.path.join(ROOT, "runs", "stage1", f"{a.label}.json") os.makedirs(os.path.dirname(out_path), exist_ok=True) res = json.load(open(out_path)) if os.path.exists(out_path) else {"model": MODEL, "tasks": {}} + res["settings"] = {"max_tokens": MAX_TOKENS, "reasoning_effort": "medium", "temperature": 0.2, "tool_call_budget": 60, + "subset": "train/subset.json", "date": "2026-10-03"} runs_root = os.path.join(ROOT, "runs", "stage1", a.label) for i, t in enumerate(subset): tid = t["id"] if (a.only and tid not in a.only) or tid in res["tasks"]: continue pool = os.path.join(ROOT, "tasks") if tid.startswith("T") else os.path.join(ROOT, "tasks_gen", "eval") - agent = LlmAgent(MODEL, BASE_URL) + agent = LlmAgent(MODEL, BASE_URL, max_tokens=MAX_TOKENS) try: rep, run_dir = Runner(pool, runs_root).run(tid, agent, a.run_base + i) except Exception as e: # noqa: BLE001 one broken run must not stop the series diff --git a/train/baseline_chain.sh b/train/baseline_chain.sh new file mode 100755 index 0000000..108d5ba --- /dev/null +++ b/train/baseline_chain.sh @@ -0,0 +1,20 @@ +#!/bin/sh +# Baseline on the 11-task subset (Kral decision 2026-10-03), detached. Replaces night_chain.sh. +# macOS notification after 2 tasks (time estimate) and at the end. Log: runs/stage1/baseline.log, chain log below. +cd "$(dirname "$0")/.." +log() { echo "$(date '+%F %T') $*"; } +log "start baseline, 11 tasks, max_tokens 16384" +python3 train/baseline.py --label baseline --run-base 20200 > runs/stage1/baseline.log 2>&1 & +BP=$! +log "baseline PID $BP" +NOTE=0 +while kill -0 $BP 2>/dev/null; do + N=$(python3 -c "import json;print(len(json.load(open('runs/stage1/baseline.json'))['tasks']))" 2>/dev/null || echo 0) + if [ "$NOTE" = 0 ] && [ "$N" -ge 2 ]; then + osascript -e 'display notification "Baseline: first 2 tasks done, ask for the time estimate" with title "stage1"' + log "2 tasks done"; NOTE=1 + fi + sleep 60 +done +log "baseline ended: $(python3 -c "import json;print(len(json.load(open('runs/stage1/baseline.json'))['tasks']))") tasks" +osascript -e 'display notification "Baseline (11 tasks) ended" with title "stage1"' diff --git a/train/subset.json b/train/subset.json index 3e4e97d..c3169c7 100644 --- a/train/subset.json +++ b/train/subset.json @@ -1,65 +1,33 @@ { - "name": "stage1-subset", + "name": "stage1-subset-v2", "created": "2026-10-03", "seed": 20261003, - "quota": { - "A": 3, - "B": 3, - "C": 3, - "E": 3, - "F": 3, - "D": 2, - "G": 2, - "H": 2, - "I": 2, - "K": 2 - }, - "rule": "accepted and review=accept; per category a fixed quota; object types spread; T01 added (old Qwen baseline). Use this list unchanged before and after training.", + "rule": "Kral decision 2026-10-03: 1 task per category (A B C D E F G H I K) plus T01; drawn from the 25-task subset (train/subset_v1_25.json), object types spread, no known infrastructure-problem task. Use unchanged before and after training.", "tasks": [ { "id": "T01", "category": "A", "object_type": "CLAS", - "note": "hand-written; old Qwen run 41.7" + "note": "hand-written; extra task" }, { "id": "G0105", "category": "A", "object_type": "FUNC" }, - { - "id": "G0104", - "category": "A", - "object_type": "CLAS" - }, - { - "id": "G0112", - "category": "B", - "object_type": "PROG" - }, { "id": "G0017", "category": "B", "object_type": "DDLS" }, - { - "id": "G0003", - "category": "B", - "object_type": "CLAS" - }, - { - "id": "G0123", - "category": "C", - "object_type": "FUNC" - }, { "id": "G0125", "category": "C", "object_type": "PROG" }, { - "id": "G0010", - "category": "C", + "id": "G0128", + "category": "D", "object_type": "CLAS" }, { @@ -67,65 +35,21 @@ "category": "E", "object_type": "CLAS" }, - { - "id": "G0142", - "category": "E", - "object_type": "PROG" - }, - { - "id": "G0147", - "category": "E", - "object_type": "DDLS" - }, - { - "id": "G0154", - "category": "F", - "object_type": "DDLS" - }, - { - "id": "G0007", - "category": "F", - "object_type": "CLAS" - }, { "id": "G0151", "category": "F", "object_type": "FUNC" }, - { - "id": "G0128", - "category": "D", - "object_type": "CLAS" - }, - { - "id": "G0133", - "category": "D", - "object_type": "FUNC" - }, { "id": "G0157", "category": "G", "object_type": "FUNC" }, - { - "id": "G0162", - "category": "G", - "object_type": "PROG" - }, - { - "id": "G0170", - "category": "H", - "object_type": "CLAS" - }, { "id": "G0174", "category": "H", - "object_type": "PROG" - }, - { - "id": "G0166", - "category": "I", - "object_type": "DDLS" + "object_type": "PROG", + "note": "stop task; gap strengthened by Kral" }, { "id": "G0167", @@ -136,11 +60,6 @@ "id": "G0185", "category": "K", "object_type": "FUNC" - }, - { - "id": "G0181", - "category": "K", - "object_type": "PROG" } ] } \ No newline at end of file diff --git a/train/subset_v1_25.json b/train/subset_v1_25.json new file mode 100644 index 0000000..3e4e97d --- /dev/null +++ b/train/subset_v1_25.json @@ -0,0 +1,146 @@ +{ + "name": "stage1-subset", + "created": "2026-10-03", + "seed": 20261003, + "quota": { + "A": 3, + "B": 3, + "C": 3, + "E": 3, + "F": 3, + "D": 2, + "G": 2, + "H": 2, + "I": 2, + "K": 2 + }, + "rule": "accepted and review=accept; per category a fixed quota; object types spread; T01 added (old Qwen baseline). Use this list unchanged before and after training.", + "tasks": [ + { + "id": "T01", + "category": "A", + "object_type": "CLAS", + "note": "hand-written; old Qwen run 41.7" + }, + { + "id": "G0105", + "category": "A", + "object_type": "FUNC" + }, + { + "id": "G0104", + "category": "A", + "object_type": "CLAS" + }, + { + "id": "G0112", + "category": "B", + "object_type": "PROG" + }, + { + "id": "G0017", + "category": "B", + "object_type": "DDLS" + }, + { + "id": "G0003", + "category": "B", + "object_type": "CLAS" + }, + { + "id": "G0123", + "category": "C", + "object_type": "FUNC" + }, + { + "id": "G0125", + "category": "C", + "object_type": "PROG" + }, + { + "id": "G0010", + "category": "C", + "object_type": "CLAS" + }, + { + "id": "G0139", + "category": "E", + "object_type": "CLAS" + }, + { + "id": "G0142", + "category": "E", + "object_type": "PROG" + }, + { + "id": "G0147", + "category": "E", + "object_type": "DDLS" + }, + { + "id": "G0154", + "category": "F", + "object_type": "DDLS" + }, + { + "id": "G0007", + "category": "F", + "object_type": "CLAS" + }, + { + "id": "G0151", + "category": "F", + "object_type": "FUNC" + }, + { + "id": "G0128", + "category": "D", + "object_type": "CLAS" + }, + { + "id": "G0133", + "category": "D", + "object_type": "FUNC" + }, + { + "id": "G0157", + "category": "G", + "object_type": "FUNC" + }, + { + "id": "G0162", + "category": "G", + "object_type": "PROG" + }, + { + "id": "G0170", + "category": "H", + "object_type": "CLAS" + }, + { + "id": "G0174", + "category": "H", + "object_type": "PROG" + }, + { + "id": "G0166", + "category": "I", + "object_type": "DDLS" + }, + { + "id": "G0167", + "category": "I", + "object_type": "CLAS" + }, + { + "id": "G0185", + "category": "K", + "object_type": "FUNC" + }, + { + "id": "G0181", + "category": "K", + "object_type": "PROG" + } + ] +} \ No newline at end of file