Stage 1 step 0: train/.venv (py3.11, mlx-lm 0.32), MLX 4-bit Qwen3.8-27B, serve script, baseline runner, README
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat
This commit is contained in:
60
train/baseline.py
Normal file
60
train/baseline.py
Normal file
@@ -0,0 +1,60 @@
|
||||
"""Stage 1 eval runs on the fixed subset (train/subset.json) with a local OpenAI-compatible server.
|
||||
|
||||
train/.venv not needed: python3 train/baseline.py --label baseline [--only T01] [--run-base 20000]
|
||||
|
||||
One task at a time. Results: runs/stage1/<label>.json (written after each task).
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
sys.path.insert(0, ROOT)
|
||||
from harness.adt_client import load_env # noqa: E402
|
||||
from harness.agents import LlmAgent # noqa: E402
|
||||
from harness.runner import Runner # noqa: E402
|
||||
|
||||
MODEL = os.path.expanduser("~/models/Qwen3.8-27B-4bit")
|
||||
BASE_URL = "http://127.0.0.1:8080/v1"
|
||||
|
||||
|
||||
def main():
|
||||
load_env(os.path.join(ROOT, ".env"))
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--label", default="baseline")
|
||||
ap.add_argument("--only", nargs="*")
|
||||
ap.add_argument("--run-base", type=int, default=20000)
|
||||
a = ap.parse_args()
|
||||
subset = json.load(open(os.path.join(ROOT, "train", "subset.json")))["tasks"]
|
||||
out_path = os.path.join(ROOT, "runs", "stage1", f"{a.label}.json")
|
||||
os.makedirs(os.path.dirname(out_path), exist_ok=True)
|
||||
res = json.load(open(out_path)) if os.path.exists(out_path) else {"model": MODEL, "tasks": {}}
|
||||
runs_root = os.path.join(ROOT, "runs", "stage1", a.label)
|
||||
for i, t in enumerate(subset):
|
||||
tid = t["id"]
|
||||
if (a.only and tid not in a.only) or tid in res["tasks"]:
|
||||
continue
|
||||
pool = os.path.join(ROOT, "tasks") if tid.startswith("T") else os.path.join(ROOT, "tasks_gen", "eval")
|
||||
agent = LlmAgent(MODEL, BASE_URL)
|
||||
try:
|
||||
rep, run_dir = Runner(pool, runs_root).run(tid, agent, a.run_base + i)
|
||||
except Exception as e: # noqa: BLE001 one broken run must not stop the series
|
||||
res["tasks"][tid] = {"error": str(e)[:300]}
|
||||
json.dump(res, open(out_path, "w"), indent=1)
|
||||
print(json.dumps({"task": tid, "error": str(e)[:300]}), flush=True)
|
||||
continue
|
||||
h = rep.get("hidden_tests") or {}
|
||||
res["tasks"][tid] = {"category": t["category"], "object_type": t["object_type"],
|
||||
"score": (rep.get("score") or {}).get("total"), "parts": rep.get("score"),
|
||||
"gates": rep.get("gates"), "hidden": f"{h.get('passed')}/{h.get('total')}",
|
||||
"tool_calls": rep.get("tool_calls"), "seconds": rep.get("seconds"),
|
||||
"agent_seconds": rep.get("agent_seconds"), "final": (rep.get("final_report") or "")[:300],
|
||||
"run_dir": os.path.relpath(run_dir, ROOT)}
|
||||
json.dump(res, open(out_path, "w"), indent=1)
|
||||
print(json.dumps({"task": tid, "score": res["tasks"][tid]["score"], "hidden": res["tasks"][tid]["hidden"],
|
||||
"tool_calls": rep.get("tool_calls"), "seconds": rep.get("seconds")}), flush=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user