Baseline shortened: 11-task subset (1 per category + T01), max_tokens 16384, settings in README; baseline_chain.sh
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat
This commit is contained in:
@@ -70,7 +70,7 @@ class LlmAgent:
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(self, model, base_url=None, api_key=None, max_turns=80, temperature=0.2,
|
def __init__(self, model, base_url=None, api_key=None, max_turns=80, temperature=0.2,
|
||||||
max_seconds=None):
|
max_seconds=None, max_tokens=None):
|
||||||
self.model = model
|
self.model = model
|
||||||
self.name = f"llm:{model}"
|
self.name = f"llm:{model}"
|
||||||
self.base_url = (base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")).rstrip("/")
|
self.base_url = (base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")).rstrip("/")
|
||||||
@@ -81,7 +81,7 @@ class LlmAgent:
|
|||||||
self.request_timeout = 3600 # local models are slow; a hung request still ends
|
self.request_timeout = 3600 # local models are slow; a hung request still ends
|
||||||
# Runaway reasoning: 35 of 1965 DeepSeek turns produced 393k output tokens and no content (46 % of the
|
# Runaway reasoning: 35 of 1965 DeepSeek turns produced 393k output tokens and no content (46 % of the
|
||||||
# run cost, 2026-10-03). Normal turns: p95 17.5k, max 82k. A cut turn is retried (see run).
|
# run cost, 2026-10-03). Normal turns: p95 17.5k, max 82k. A cut turn is retried (see run).
|
||||||
self.max_tokens = 32000 if ":cloud" in (model or "") else None
|
self.max_tokens = max_tokens or (32000 if ":cloud" in (model or "") else None)
|
||||||
self.empty_retries = 2
|
self.empty_retries = 2
|
||||||
|
|
||||||
def _chat(self, messages, tools):
|
def _chat(self, messages, tools):
|
||||||
|
|||||||
@@ -43,12 +43,27 @@ qwen3_coder XML tool format; `mlx_lm` parses it. First request: 86 s (prompt of
|
|||||||
|
|
||||||
## Eval subset
|
## Eval subset
|
||||||
|
|
||||||
`train/subset.json` (copy: `runs/stage1/subset.json`): 25 tasks, balanced over the categories
|
`train/subset.json` (copy: `runs/stage1/subset.json`): **11 tasks** (Kral decision 2026-10-03): one task per
|
||||||
(A, B, C, E, F: 3 each; D, G, H, I, K: 2 each), with T01. Use the same list before and after training.
|
category (A B C D E F G H I K) plus T01. The first subset of 25 tasks (`train/subset_v1_25.json`) was cut because
|
||||||
|
one task needed 2-3 hours with the MLX 4-bit model (about 12 tokens/s). Tasks: T01, G0105, G0017, G0125, G0128,
|
||||||
|
G0139, G0151, G0157, G0174, G0167, G0185. Use the same list before and after training.
|
||||||
|
|
||||||
|
## Settings of the stage 1 runs (the same for baseline and after training)
|
||||||
|
|
||||||
|
| Setting | Value |
|
||||||
|
|---|---|
|
||||||
|
| Model | `~/models/Qwen3.8-27B-4bit` (MLX affine 4 bit); after training the same with `--adapter-path` |
|
||||||
|
| Thinking | on, `reasoning_effort` medium |
|
||||||
|
| temperature / top_p / top_k / min_p | 0.2 / 0.95 / 20 / 0 |
|
||||||
|
| max_tokens per turn | **16384**, sent in each request by `train/baseline.py` (`MAX_TOKENS`); the server limit stays 32768 |
|
||||||
|
| Tool-call budget per task | 60 calls, 15 activations (T01 too) |
|
||||||
|
| Empty turn | retried (2 times), then the run stops ("Stopped: empty model response") |
|
||||||
|
| Prompt cache of the server | `--prompt-cache-size 4 --prompt-cache-bytes 6000000000` |
|
||||||
|
|
||||||
|
The settings are also written into `runs/stage1/baseline.json` (`settings`). The old Ollama run (41.7) and the
|
||||||
|
T01 test with 32768 tokens and budget 40 (`t01_test_budget40`: 40.0) are not comparable.
|
||||||
|
|
||||||
## Baseline
|
## Baseline
|
||||||
|
|
||||||
- Runner: `python3 train/baseline.py --label baseline` (one task at a time; results
|
- Runner: `python3 train/baseline.py --label baseline --run-base 20200` (one task at a time; results
|
||||||
`runs/stage1/baseline.json`, run directories `runs/stage1/baseline/`).
|
`runs/stage1/baseline.json`, run directories `runs/stage1/baseline/`). Started by `train/baseline_chain.sh`.
|
||||||
- The earlier T01 score 41.7 (run 103) used the Ollama NVFP4 weights. The baseline of 2026-10-03 with the
|
|
||||||
MLX 4-bit weights is the new reference for stage 1.
|
|
||||||
|
|||||||
@@ -17,6 +17,7 @@ from harness.runner import Runner # noqa: E402
|
|||||||
|
|
||||||
MODEL = os.path.expanduser("~/models/Qwen3.8-27B-4bit")
|
MODEL = os.path.expanduser("~/models/Qwen3.8-27B-4bit")
|
||||||
BASE_URL = "http://127.0.0.1:8080/v1"
|
BASE_URL = "http://127.0.0.1:8080/v1"
|
||||||
|
MAX_TOKENS = 16384 # per turn; sent in each request (the server limit stays 32768). Same for before/after.
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
@@ -30,13 +31,15 @@ def main():
|
|||||||
out_path = os.path.join(ROOT, "runs", "stage1", f"{a.label}.json")
|
out_path = os.path.join(ROOT, "runs", "stage1", f"{a.label}.json")
|
||||||
os.makedirs(os.path.dirname(out_path), exist_ok=True)
|
os.makedirs(os.path.dirname(out_path), exist_ok=True)
|
||||||
res = json.load(open(out_path)) if os.path.exists(out_path) else {"model": MODEL, "tasks": {}}
|
res = json.load(open(out_path)) if os.path.exists(out_path) else {"model": MODEL, "tasks": {}}
|
||||||
|
res["settings"] = {"max_tokens": MAX_TOKENS, "reasoning_effort": "medium", "temperature": 0.2, "tool_call_budget": 60,
|
||||||
|
"subset": "train/subset.json", "date": "2026-10-03"}
|
||||||
runs_root = os.path.join(ROOT, "runs", "stage1", a.label)
|
runs_root = os.path.join(ROOT, "runs", "stage1", a.label)
|
||||||
for i, t in enumerate(subset):
|
for i, t in enumerate(subset):
|
||||||
tid = t["id"]
|
tid = t["id"]
|
||||||
if (a.only and tid not in a.only) or tid in res["tasks"]:
|
if (a.only and tid not in a.only) or tid in res["tasks"]:
|
||||||
continue
|
continue
|
||||||
pool = os.path.join(ROOT, "tasks") if tid.startswith("T") else os.path.join(ROOT, "tasks_gen", "eval")
|
pool = os.path.join(ROOT, "tasks") if tid.startswith("T") else os.path.join(ROOT, "tasks_gen", "eval")
|
||||||
agent = LlmAgent(MODEL, BASE_URL)
|
agent = LlmAgent(MODEL, BASE_URL, max_tokens=MAX_TOKENS)
|
||||||
try:
|
try:
|
||||||
rep, run_dir = Runner(pool, runs_root).run(tid, agent, a.run_base + i)
|
rep, run_dir = Runner(pool, runs_root).run(tid, agent, a.run_base + i)
|
||||||
except Exception as e: # noqa: BLE001 one broken run must not stop the series
|
except Exception as e: # noqa: BLE001 one broken run must not stop the series
|
||||||
|
|||||||
20
train/baseline_chain.sh
Executable file
20
train/baseline_chain.sh
Executable file
@@ -0,0 +1,20 @@
|
|||||||
|
#!/bin/sh
|
||||||
|
# Baseline on the 11-task subset (Kral decision 2026-10-03), detached. Replaces night_chain.sh.
|
||||||
|
# macOS notification after 2 tasks (time estimate) and at the end. Log: runs/stage1/baseline.log, chain log below.
|
||||||
|
cd "$(dirname "$0")/.."
|
||||||
|
log() { echo "$(date '+%F %T') $*"; }
|
||||||
|
log "start baseline, 11 tasks, max_tokens 16384"
|
||||||
|
python3 train/baseline.py --label baseline --run-base 20200 > runs/stage1/baseline.log 2>&1 &
|
||||||
|
BP=$!
|
||||||
|
log "baseline PID $BP"
|
||||||
|
NOTE=0
|
||||||
|
while kill -0 $BP 2>/dev/null; do
|
||||||
|
N=$(python3 -c "import json;print(len(json.load(open('runs/stage1/baseline.json'))['tasks']))" 2>/dev/null || echo 0)
|
||||||
|
if [ "$NOTE" = 0 ] && [ "$N" -ge 2 ]; then
|
||||||
|
osascript -e 'display notification "Baseline: first 2 tasks done, ask for the time estimate" with title "stage1"'
|
||||||
|
log "2 tasks done"; NOTE=1
|
||||||
|
fi
|
||||||
|
sleep 60
|
||||||
|
done
|
||||||
|
log "baseline ended: $(python3 -c "import json;print(len(json.load(open('runs/stage1/baseline.json'))['tasks']))") tasks"
|
||||||
|
osascript -e 'display notification "Baseline (11 tasks) ended" with title "stage1"'
|
||||||
@@ -1,65 +1,33 @@
|
|||||||
{
|
{
|
||||||
"name": "stage1-subset",
|
"name": "stage1-subset-v2",
|
||||||
"created": "2026-10-03",
|
"created": "2026-10-03",
|
||||||
"seed": 20261003,
|
"seed": 20261003,
|
||||||
"quota": {
|
"rule": "Kral decision 2026-10-03: 1 task per category (A B C D E F G H I K) plus T01; drawn from the 25-task subset (train/subset_v1_25.json), object types spread, no known infrastructure-problem task. Use unchanged before and after training.",
|
||||||
"A": 3,
|
|
||||||
"B": 3,
|
|
||||||
"C": 3,
|
|
||||||
"E": 3,
|
|
||||||
"F": 3,
|
|
||||||
"D": 2,
|
|
||||||
"G": 2,
|
|
||||||
"H": 2,
|
|
||||||
"I": 2,
|
|
||||||
"K": 2
|
|
||||||
},
|
|
||||||
"rule": "accepted and review=accept; per category a fixed quota; object types spread; T01 added (old Qwen baseline). Use this list unchanged before and after training.",
|
|
||||||
"tasks": [
|
"tasks": [
|
||||||
{
|
{
|
||||||
"id": "T01",
|
"id": "T01",
|
||||||
"category": "A",
|
"category": "A",
|
||||||
"object_type": "CLAS",
|
"object_type": "CLAS",
|
||||||
"note": "hand-written; old Qwen run 41.7"
|
"note": "hand-written; extra task"
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"id": "G0105",
|
"id": "G0105",
|
||||||
"category": "A",
|
"category": "A",
|
||||||
"object_type": "FUNC"
|
"object_type": "FUNC"
|
||||||
},
|
},
|
||||||
{
|
|
||||||
"id": "G0104",
|
|
||||||
"category": "A",
|
|
||||||
"object_type": "CLAS"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": "G0112",
|
|
||||||
"category": "B",
|
|
||||||
"object_type": "PROG"
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
"id": "G0017",
|
"id": "G0017",
|
||||||
"category": "B",
|
"category": "B",
|
||||||
"object_type": "DDLS"
|
"object_type": "DDLS"
|
||||||
},
|
},
|
||||||
{
|
|
||||||
"id": "G0003",
|
|
||||||
"category": "B",
|
|
||||||
"object_type": "CLAS"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": "G0123",
|
|
||||||
"category": "C",
|
|
||||||
"object_type": "FUNC"
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
"id": "G0125",
|
"id": "G0125",
|
||||||
"category": "C",
|
"category": "C",
|
||||||
"object_type": "PROG"
|
"object_type": "PROG"
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"id": "G0010",
|
"id": "G0128",
|
||||||
"category": "C",
|
"category": "D",
|
||||||
"object_type": "CLAS"
|
"object_type": "CLAS"
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
@@ -67,65 +35,21 @@
|
|||||||
"category": "E",
|
"category": "E",
|
||||||
"object_type": "CLAS"
|
"object_type": "CLAS"
|
||||||
},
|
},
|
||||||
{
|
|
||||||
"id": "G0142",
|
|
||||||
"category": "E",
|
|
||||||
"object_type": "PROG"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": "G0147",
|
|
||||||
"category": "E",
|
|
||||||
"object_type": "DDLS"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": "G0154",
|
|
||||||
"category": "F",
|
|
||||||
"object_type": "DDLS"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": "G0007",
|
|
||||||
"category": "F",
|
|
||||||
"object_type": "CLAS"
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
"id": "G0151",
|
"id": "G0151",
|
||||||
"category": "F",
|
"category": "F",
|
||||||
"object_type": "FUNC"
|
"object_type": "FUNC"
|
||||||
},
|
},
|
||||||
{
|
|
||||||
"id": "G0128",
|
|
||||||
"category": "D",
|
|
||||||
"object_type": "CLAS"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": "G0133",
|
|
||||||
"category": "D",
|
|
||||||
"object_type": "FUNC"
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
"id": "G0157",
|
"id": "G0157",
|
||||||
"category": "G",
|
"category": "G",
|
||||||
"object_type": "FUNC"
|
"object_type": "FUNC"
|
||||||
},
|
},
|
||||||
{
|
|
||||||
"id": "G0162",
|
|
||||||
"category": "G",
|
|
||||||
"object_type": "PROG"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": "G0170",
|
|
||||||
"category": "H",
|
|
||||||
"object_type": "CLAS"
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
"id": "G0174",
|
"id": "G0174",
|
||||||
"category": "H",
|
"category": "H",
|
||||||
"object_type": "PROG"
|
"object_type": "PROG",
|
||||||
},
|
"note": "stop task; gap strengthened by Kral"
|
||||||
{
|
|
||||||
"id": "G0166",
|
|
||||||
"category": "I",
|
|
||||||
"object_type": "DDLS"
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"id": "G0167",
|
"id": "G0167",
|
||||||
@@ -136,11 +60,6 @@
|
|||||||
"id": "G0185",
|
"id": "G0185",
|
||||||
"category": "K",
|
"category": "K",
|
||||||
"object_type": "FUNC"
|
"object_type": "FUNC"
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": "G0181",
|
|
||||||
"category": "K",
|
|
||||||
"object_type": "PROG"
|
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
146
train/subset_v1_25.json
Normal file
146
train/subset_v1_25.json
Normal file
@@ -0,0 +1,146 @@
|
|||||||
|
{
|
||||||
|
"name": "stage1-subset",
|
||||||
|
"created": "2026-10-03",
|
||||||
|
"seed": 20261003,
|
||||||
|
"quota": {
|
||||||
|
"A": 3,
|
||||||
|
"B": 3,
|
||||||
|
"C": 3,
|
||||||
|
"E": 3,
|
||||||
|
"F": 3,
|
||||||
|
"D": 2,
|
||||||
|
"G": 2,
|
||||||
|
"H": 2,
|
||||||
|
"I": 2,
|
||||||
|
"K": 2
|
||||||
|
},
|
||||||
|
"rule": "accepted and review=accept; per category a fixed quota; object types spread; T01 added (old Qwen baseline). Use this list unchanged before and after training.",
|
||||||
|
"tasks": [
|
||||||
|
{
|
||||||
|
"id": "T01",
|
||||||
|
"category": "A",
|
||||||
|
"object_type": "CLAS",
|
||||||
|
"note": "hand-written; old Qwen run 41.7"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0105",
|
||||||
|
"category": "A",
|
||||||
|
"object_type": "FUNC"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0104",
|
||||||
|
"category": "A",
|
||||||
|
"object_type": "CLAS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0112",
|
||||||
|
"category": "B",
|
||||||
|
"object_type": "PROG"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0017",
|
||||||
|
"category": "B",
|
||||||
|
"object_type": "DDLS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0003",
|
||||||
|
"category": "B",
|
||||||
|
"object_type": "CLAS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0123",
|
||||||
|
"category": "C",
|
||||||
|
"object_type": "FUNC"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0125",
|
||||||
|
"category": "C",
|
||||||
|
"object_type": "PROG"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0010",
|
||||||
|
"category": "C",
|
||||||
|
"object_type": "CLAS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0139",
|
||||||
|
"category": "E",
|
||||||
|
"object_type": "CLAS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0142",
|
||||||
|
"category": "E",
|
||||||
|
"object_type": "PROG"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0147",
|
||||||
|
"category": "E",
|
||||||
|
"object_type": "DDLS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0154",
|
||||||
|
"category": "F",
|
||||||
|
"object_type": "DDLS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0007",
|
||||||
|
"category": "F",
|
||||||
|
"object_type": "CLAS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0151",
|
||||||
|
"category": "F",
|
||||||
|
"object_type": "FUNC"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0128",
|
||||||
|
"category": "D",
|
||||||
|
"object_type": "CLAS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0133",
|
||||||
|
"category": "D",
|
||||||
|
"object_type": "FUNC"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0157",
|
||||||
|
"category": "G",
|
||||||
|
"object_type": "FUNC"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0162",
|
||||||
|
"category": "G",
|
||||||
|
"object_type": "PROG"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0170",
|
||||||
|
"category": "H",
|
||||||
|
"object_type": "CLAS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0174",
|
||||||
|
"category": "H",
|
||||||
|
"object_type": "PROG"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0166",
|
||||||
|
"category": "I",
|
||||||
|
"object_type": "DDLS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0167",
|
||||||
|
"category": "I",
|
||||||
|
"object_type": "CLAS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0185",
|
||||||
|
"category": "K",
|
||||||
|
"object_type": "FUNC"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "G0181",
|
||||||
|
"category": "K",
|
||||||
|
"object_type": "PROG"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user