Baseline shortened: 11-task subset (1 per category + T01), max_tokens 16384, settings in README; baseline_chain.sh

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat
This commit is contained in:
Kral
2026-10-03 22:36:00 +02:00
parent 7054d9fd2e
commit 689820af4d
6 changed files with 200 additions and 97 deletions

View File

@@ -70,7 +70,7 @@ class LlmAgent:
""" """
def __init__(self, model, base_url=None, api_key=None, max_turns=80, temperature=0.2, def __init__(self, model, base_url=None, api_key=None, max_turns=80, temperature=0.2,
max_seconds=None): max_seconds=None, max_tokens=None):
self.model = model self.model = model
self.name = f"llm:{model}" self.name = f"llm:{model}"
self.base_url = (base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")).rstrip("/") self.base_url = (base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")).rstrip("/")
@@ -81,7 +81,7 @@ class LlmAgent:
self.request_timeout = 3600 # local models are slow; a hung request still ends self.request_timeout = 3600 # local models are slow; a hung request still ends
# Runaway reasoning: 35 of 1965 DeepSeek turns produced 393k output tokens and no content (46 % of the # Runaway reasoning: 35 of 1965 DeepSeek turns produced 393k output tokens and no content (46 % of the
# run cost, 2026-10-03). Normal turns: p95 17.5k, max 82k. A cut turn is retried (see run). # run cost, 2026-10-03). Normal turns: p95 17.5k, max 82k. A cut turn is retried (see run).
self.max_tokens = 32000 if ":cloud" in (model or "") else None self.max_tokens = max_tokens or (32000 if ":cloud" in (model or "") else None)
self.empty_retries = 2 self.empty_retries = 2
def _chat(self, messages, tools): def _chat(self, messages, tools):

View File

@@ -43,12 +43,27 @@ qwen3_coder XML tool format; `mlx_lm` parses it. First request: 86 s (prompt of
## Eval subset ## Eval subset
`train/subset.json` (copy: `runs/stage1/subset.json`): 25 tasks, balanced over the categories `train/subset.json` (copy: `runs/stage1/subset.json`): **11 tasks** (Kral decision 2026-10-03): one task per
(A, B, C, E, F: 3 each; D, G, H, I, K: 2 each), with T01. Use the same list before and after training. category (A B C D E F G H I K) plus T01. The first subset of 25 tasks (`train/subset_v1_25.json`) was cut because
one task needed 2-3 hours with the MLX 4-bit model (about 12 tokens/s). Tasks: T01, G0105, G0017, G0125, G0128,
G0139, G0151, G0157, G0174, G0167, G0185. Use the same list before and after training.
## Settings of the stage 1 runs (the same for baseline and after training)
| Setting | Value |
|---|---|
| Model | `~/models/Qwen3.8-27B-4bit` (MLX affine 4 bit); after training the same with `--adapter-path` |
| Thinking | on, `reasoning_effort` medium |
| temperature / top_p / top_k / min_p | 0.2 / 0.95 / 20 / 0 |
| max_tokens per turn | **16384**, sent in each request by `train/baseline.py` (`MAX_TOKENS`); the server limit stays 32768 |
| Tool-call budget per task | 60 calls, 15 activations (T01 too) |
| Empty turn | retried (2 times), then the run stops ("Stopped: empty model response") |
| Prompt cache of the server | `--prompt-cache-size 4 --prompt-cache-bytes 6000000000` |
The settings are also written into `runs/stage1/baseline.json` (`settings`). The old Ollama run (41.7) and the
T01 test with 32768 tokens and budget 40 (`t01_test_budget40`: 40.0) are not comparable.
## Baseline ## Baseline
- Runner: `python3 train/baseline.py --label baseline` (one task at a time; results - Runner: `python3 train/baseline.py --label baseline --run-base 20200` (one task at a time; results
`runs/stage1/baseline.json`, run directories `runs/stage1/baseline/`). `runs/stage1/baseline.json`, run directories `runs/stage1/baseline/`). Started by `train/baseline_chain.sh`.
- The earlier T01 score 41.7 (run 103) used the Ollama NVFP4 weights. The baseline of 2026-10-03 with the
MLX 4-bit weights is the new reference for stage 1.

View File

@@ -17,6 +17,7 @@ from harness.runner import Runner # noqa: E402
MODEL = os.path.expanduser("~/models/Qwen3.8-27B-4bit") MODEL = os.path.expanduser("~/models/Qwen3.8-27B-4bit")
BASE_URL = "http://127.0.0.1:8080/v1" BASE_URL = "http://127.0.0.1:8080/v1"
MAX_TOKENS = 16384 # per turn; sent in each request (the server limit stays 32768). Same for before/after.
def main(): def main():
@@ -30,13 +31,15 @@ def main():
out_path = os.path.join(ROOT, "runs", "stage1", f"{a.label}.json") out_path = os.path.join(ROOT, "runs", "stage1", f"{a.label}.json")
os.makedirs(os.path.dirname(out_path), exist_ok=True) os.makedirs(os.path.dirname(out_path), exist_ok=True)
res = json.load(open(out_path)) if os.path.exists(out_path) else {"model": MODEL, "tasks": {}} res = json.load(open(out_path)) if os.path.exists(out_path) else {"model": MODEL, "tasks": {}}
res["settings"] = {"max_tokens": MAX_TOKENS, "reasoning_effort": "medium", "temperature": 0.2, "tool_call_budget": 60,
"subset": "train/subset.json", "date": "2026-10-03"}
runs_root = os.path.join(ROOT, "runs", "stage1", a.label) runs_root = os.path.join(ROOT, "runs", "stage1", a.label)
for i, t in enumerate(subset): for i, t in enumerate(subset):
tid = t["id"] tid = t["id"]
if (a.only and tid not in a.only) or tid in res["tasks"]: if (a.only and tid not in a.only) or tid in res["tasks"]:
continue continue
pool = os.path.join(ROOT, "tasks") if tid.startswith("T") else os.path.join(ROOT, "tasks_gen", "eval") pool = os.path.join(ROOT, "tasks") if tid.startswith("T") else os.path.join(ROOT, "tasks_gen", "eval")
agent = LlmAgent(MODEL, BASE_URL) agent = LlmAgent(MODEL, BASE_URL, max_tokens=MAX_TOKENS)
try: try:
rep, run_dir = Runner(pool, runs_root).run(tid, agent, a.run_base + i) rep, run_dir = Runner(pool, runs_root).run(tid, agent, a.run_base + i)
except Exception as e: # noqa: BLE001 one broken run must not stop the series except Exception as e: # noqa: BLE001 one broken run must not stop the series

20
train/baseline_chain.sh Executable file
View File

@@ -0,0 +1,20 @@
#!/bin/sh
# Baseline on the 11-task subset (Kral decision 2026-10-03), detached. Replaces night_chain.sh.
# macOS notification after 2 tasks (time estimate) and at the end. Log: runs/stage1/baseline.log, chain log below.
cd "$(dirname "$0")/.."
log() { echo "$(date '+%F %T') $*"; }
log "start baseline, 11 tasks, max_tokens 16384"
python3 train/baseline.py --label baseline --run-base 20200 > runs/stage1/baseline.log 2>&1 &
BP=$!
log "baseline PID $BP"
NOTE=0
while kill -0 $BP 2>/dev/null; do
N=$(python3 -c "import json;print(len(json.load(open('runs/stage1/baseline.json'))['tasks']))" 2>/dev/null || echo 0)
if [ "$NOTE" = 0 ] && [ "$N" -ge 2 ]; then
osascript -e 'display notification "Baseline: first 2 tasks done, ask for the time estimate" with title "stage1"'
log "2 tasks done"; NOTE=1
fi
sleep 60
done
log "baseline ended: $(python3 -c "import json;print(len(json.load(open('runs/stage1/baseline.json'))['tasks']))") tasks"
osascript -e 'display notification "Baseline (11 tasks) ended" with title "stage1"'

View File

@@ -1,65 +1,33 @@
{ {
"name": "stage1-subset", "name": "stage1-subset-v2",
"created": "2026-10-03", "created": "2026-10-03",
"seed": 20261003, "seed": 20261003,
"quota": { "rule": "Kral decision 2026-10-03: 1 task per category (A B C D E F G H I K) plus T01; drawn from the 25-task subset (train/subset_v1_25.json), object types spread, no known infrastructure-problem task. Use unchanged before and after training.",
"A": 3,
"B": 3,
"C": 3,
"E": 3,
"F": 3,
"D": 2,
"G": 2,
"H": 2,
"I": 2,
"K": 2
},
"rule": "accepted and review=accept; per category a fixed quota; object types spread; T01 added (old Qwen baseline). Use this list unchanged before and after training.",
"tasks": [ "tasks": [
{ {
"id": "T01", "id": "T01",
"category": "A", "category": "A",
"object_type": "CLAS", "object_type": "CLAS",
"note": "hand-written; old Qwen run 41.7" "note": "hand-written; extra task"
}, },
{ {
"id": "G0105", "id": "G0105",
"category": "A", "category": "A",
"object_type": "FUNC" "object_type": "FUNC"
}, },
{
"id": "G0104",
"category": "A",
"object_type": "CLAS"
},
{
"id": "G0112",
"category": "B",
"object_type": "PROG"
},
{ {
"id": "G0017", "id": "G0017",
"category": "B", "category": "B",
"object_type": "DDLS" "object_type": "DDLS"
}, },
{
"id": "G0003",
"category": "B",
"object_type": "CLAS"
},
{
"id": "G0123",
"category": "C",
"object_type": "FUNC"
},
{ {
"id": "G0125", "id": "G0125",
"category": "C", "category": "C",
"object_type": "PROG" "object_type": "PROG"
}, },
{ {
"id": "G0010", "id": "G0128",
"category": "C", "category": "D",
"object_type": "CLAS" "object_type": "CLAS"
}, },
{ {
@@ -67,65 +35,21 @@
"category": "E", "category": "E",
"object_type": "CLAS" "object_type": "CLAS"
}, },
{
"id": "G0142",
"category": "E",
"object_type": "PROG"
},
{
"id": "G0147",
"category": "E",
"object_type": "DDLS"
},
{
"id": "G0154",
"category": "F",
"object_type": "DDLS"
},
{
"id": "G0007",
"category": "F",
"object_type": "CLAS"
},
{ {
"id": "G0151", "id": "G0151",
"category": "F", "category": "F",
"object_type": "FUNC" "object_type": "FUNC"
}, },
{
"id": "G0128",
"category": "D",
"object_type": "CLAS"
},
{
"id": "G0133",
"category": "D",
"object_type": "FUNC"
},
{ {
"id": "G0157", "id": "G0157",
"category": "G", "category": "G",
"object_type": "FUNC" "object_type": "FUNC"
}, },
{
"id": "G0162",
"category": "G",
"object_type": "PROG"
},
{
"id": "G0170",
"category": "H",
"object_type": "CLAS"
},
{ {
"id": "G0174", "id": "G0174",
"category": "H", "category": "H",
"object_type": "PROG" "object_type": "PROG",
}, "note": "stop task; gap strengthened by Kral"
{
"id": "G0166",
"category": "I",
"object_type": "DDLS"
}, },
{ {
"id": "G0167", "id": "G0167",
@@ -136,11 +60,6 @@
"id": "G0185", "id": "G0185",
"category": "K", "category": "K",
"object_type": "FUNC" "object_type": "FUNC"
},
{
"id": "G0181",
"category": "K",
"object_type": "PROG"
} }
] ]
} }

146
train/subset_v1_25.json Normal file
View File

@@ -0,0 +1,146 @@
{
"name": "stage1-subset",
"created": "2026-10-03",
"seed": 20261003,
"quota": {
"A": 3,
"B": 3,
"C": 3,
"E": 3,
"F": 3,
"D": 2,
"G": 2,
"H": 2,
"I": 2,
"K": 2
},
"rule": "accepted and review=accept; per category a fixed quota; object types spread; T01 added (old Qwen baseline). Use this list unchanged before and after training.",
"tasks": [
{
"id": "T01",
"category": "A",
"object_type": "CLAS",
"note": "hand-written; old Qwen run 41.7"
},
{
"id": "G0105",
"category": "A",
"object_type": "FUNC"
},
{
"id": "G0104",
"category": "A",
"object_type": "CLAS"
},
{
"id": "G0112",
"category": "B",
"object_type": "PROG"
},
{
"id": "G0017",
"category": "B",
"object_type": "DDLS"
},
{
"id": "G0003",
"category": "B",
"object_type": "CLAS"
},
{
"id": "G0123",
"category": "C",
"object_type": "FUNC"
},
{
"id": "G0125",
"category": "C",
"object_type": "PROG"
},
{
"id": "G0010",
"category": "C",
"object_type": "CLAS"
},
{
"id": "G0139",
"category": "E",
"object_type": "CLAS"
},
{
"id": "G0142",
"category": "E",
"object_type": "PROG"
},
{
"id": "G0147",
"category": "E",
"object_type": "DDLS"
},
{
"id": "G0154",
"category": "F",
"object_type": "DDLS"
},
{
"id": "G0007",
"category": "F",
"object_type": "CLAS"
},
{
"id": "G0151",
"category": "F",
"object_type": "FUNC"
},
{
"id": "G0128",
"category": "D",
"object_type": "CLAS"
},
{
"id": "G0133",
"category": "D",
"object_type": "FUNC"
},
{
"id": "G0157",
"category": "G",
"object_type": "FUNC"
},
{
"id": "G0162",
"category": "G",
"object_type": "PROG"
},
{
"id": "G0170",
"category": "H",
"object_type": "CLAS"
},
{
"id": "G0174",
"category": "H",
"object_type": "PROG"
},
{
"id": "G0166",
"category": "I",
"object_type": "DDLS"
},
{
"id": "G0167",
"category": "I",
"object_type": "CLAS"
},
{
"id": "G0185",
"category": "K",
"object_type": "FUNC"
},
{
"id": "G0181",
"category": "K",
"object_type": "PROG"
}
]
}