baseline.py: --model, --base-url, --enable-thinking-false (Qwen run on MacBook)
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat
This commit is contained in:
@@ -27,12 +27,16 @@ def main():
|
||||
ap.add_argument("--label", default="baseline")
|
||||
ap.add_argument("--only", nargs="*")
|
||||
ap.add_argument("--run-base", type=int, default=20000)
|
||||
ap.add_argument("--model", default=MODEL, help="path of the served model (the server ignores it, it is recorded)")
|
||||
ap.add_argument("--base-url", default=BASE_URL)
|
||||
ap.add_argument("--enable-thinking-false", action="store_true", help="Qwen: send enable_thinking=false per request")
|
||||
a = ap.parse_args()
|
||||
subset = json.load(open(os.path.join(ROOT, "train", "subset.json")))["tasks"]
|
||||
out_path = os.path.join(ROOT, "runs", "stage1", f"{a.label}.json")
|
||||
os.makedirs(os.path.dirname(out_path), exist_ok=True)
|
||||
res = json.load(open(out_path)) if os.path.exists(out_path) else {"model": MODEL, "tasks": {}}
|
||||
res["settings"] = {"max_tokens": MAX_TOKENS, "enable_thinking": None, "temperature": 0.2, "tool_call_budget": 60, "loop_guard": LOOP_GUARD,
|
||||
res = json.load(open(out_path)) if os.path.exists(out_path) else {"model": a.model, "tasks": {}}
|
||||
kw = {"enable_thinking": False} if a.enable_thinking_false else None
|
||||
res["settings"] = {"max_tokens": MAX_TOKENS, "enable_thinking": False if a.enable_thinking_false else None, "temperature": 0.2, "tool_call_budget": 60, "loop_guard": LOOP_GUARD,
|
||||
"subset": "train/subset.json", "date": "2026-10-04"}
|
||||
runs_root = os.path.join(ROOT, "runs", "stage1", a.label)
|
||||
for i, t in enumerate(subset):
|
||||
@@ -40,7 +44,7 @@ def main():
|
||||
if (a.only and tid not in a.only) or tid in res["tasks"]:
|
||||
continue
|
||||
pool = os.path.join(ROOT, "tasks") if tid.startswith("T") else os.path.join(ROOT, "tasks_gen", "eval")
|
||||
agent = LlmAgent(MODEL, BASE_URL, max_tokens=MAX_TOKENS, loop_guard=LOOP_GUARD)
|
||||
agent = LlmAgent(a.model, a.base_url, max_tokens=MAX_TOKENS, chat_template_kwargs=kw, loop_guard=LOOP_GUARD)
|
||||
try:
|
||||
rep, run_dir = Runner(pool, runs_root).run(tid, agent, a.run_base + i)
|
||||
except Exception as e: # noqa: BLE001 one broken run must not stop the series
|
||||
|
||||
Reference in New Issue
Block a user