K variants for training (free text, EPOD tool names), second attempt only for failed tasks, docs/epod-syntax-hint.md
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -490,11 +490,12 @@ def contract_terms(task):
|
||||
return out
|
||||
|
||||
|
||||
def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema="generic_v0", max_repairs=2):
|
||||
def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema="generic_v0", max_repairs=2,
|
||||
pool="eval"):
|
||||
"""Category K: free-text or incomplete input + other tool schema. Same reference and hidden tests
|
||||
as the source task; only spec.md, category and tool_schema change."""
|
||||
check_budget()
|
||||
pool_root = os.path.join(ROOT, "tasks_gen", "eval")
|
||||
pool_root = os.path.join(ROOT, "tasks_gen", pool)
|
||||
src_dir, task_dir = os.path.join(pool_root, src_id), os.path.join(pool_root, new_id)
|
||||
b = bundle_of(src_dir)
|
||||
b["files"] = {k: v for k, v in b["files"].items()
|
||||
@@ -507,7 +508,7 @@ def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema
|
||||
"ALV). Keep these names exactly as written (the tests use them): "
|
||||
f"{', '.join(terms)}. Keep the placeholder {{{{P}}}} in names. Use Simplified Technical English. "
|
||||
f"Return only the new spec text.\n\nSpec:\n{spec}"}]
|
||||
log = {"id": new_id, "pool": "eval", "category": "K", "base_task": src_id, "style": style,
|
||||
log = {"id": new_id, "pool": pool, "category": "K", "base_task": src_id, "style": style,
|
||||
"tool_schema": tool_schema, "attempts": []}
|
||||
for attempt in range(max_repairs + 1):
|
||||
text = chat(model, messages, base_url).strip()
|
||||
@@ -520,7 +521,9 @@ def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema
|
||||
+ ". Return the full spec again with all names."})
|
||||
continue
|
||||
b["files"]["spec.md"] = text + "\n"
|
||||
b["task"].update(category="K", base_task=src_id, input_style=style, tool_schema=tool_schema)
|
||||
b["task"].update(category="K", base_task=src_id, input_style=style)
|
||||
if tool_schema: # None: EPOD tool names (training: free-text input only)
|
||||
b["task"]["tool_schema"] = tool_schema
|
||||
write_bundle(b, task_dir, new_id)
|
||||
rep_o, rep_n = validate(pool_root, new_id, run_base + RUNS_PER_ATTEMPT * attempt)
|
||||
so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total")
|
||||
|
||||
@@ -16,7 +16,7 @@ import time
|
||||
|
||||
from .adt_client import load_env
|
||||
from .evalset import SLOTS, RELEASES, accepted_goals
|
||||
from .generator import ROOT, generate
|
||||
from .generator import ROOT, generate, make_k_variant
|
||||
from .ledger import BudgetExceeded, spent
|
||||
from . import overlap
|
||||
|
||||
@@ -140,10 +140,58 @@ def run(part, parts, target, stop_ledger):
|
||||
print(json.dumps(log), flush=True)
|
||||
|
||||
|
||||
K_FIRST_ID = 1300
|
||||
K_RUN_BASE = 41800 # 20 per variant; above the trajectory run numbers (41000-41700)
|
||||
K_COUNT = 18 # K share of the eval plan: 10 of 110 (9 %); counted inside the 200 accepted tasks
|
||||
K_STYLES_CYCLE = ["free_text", "incomplete"]
|
||||
|
||||
|
||||
def run_k(count):
|
||||
"""K tasks for training: free-text or incomplete spec of an accepted training task, EPOD tool names
|
||||
(no generic_v0). Same reference and hidden tests as the source task."""
|
||||
base_url = os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")
|
||||
for n in range(count):
|
||||
new_id = f"G{K_FIRST_ID + n}"
|
||||
if os.path.exists(os.path.join(POOL, "_logs", new_id + ".json")):
|
||||
continue
|
||||
used = {json.load(open(f)).get("base_task") for f in glob.glob(os.path.join(POOL, "_logs", "G13*.json"))}
|
||||
cands = [] # accepted, not K, not H (a stop task has no free-text form), not used yet
|
||||
for f in sorted(glob.glob(os.path.join(POOL, "_logs", "G1[0-2]*.json"))):
|
||||
lg = json.load(open(f))
|
||||
if lg.get("accepted") and lg.get("category") not in ("H", "K") and lg["id"] not in used:
|
||||
cands.append(lg)
|
||||
if not cands:
|
||||
print("no source task left", flush=True)
|
||||
return
|
||||
kinds = {}
|
||||
for lg in cands: # spread over object types: take the type with the fewest K variants so far
|
||||
kinds.setdefault(lg["object_type"], []).append(lg)
|
||||
done_types = [json.load(open(f)).get("object_type") for f in glob.glob(os.path.join(POOL, "_logs", "G13*.json"))]
|
||||
otype = min(kinds, key=lambda t: done_types.count(t))
|
||||
src = kinds[otype][0]
|
||||
style = K_STYLES_CYCLE[n % 2]
|
||||
if spent() >= json.load(open(PLAN))["ledger_at_start"] + json.load(open(PLAN))["stop_ledger"]:
|
||||
print("PHASE LIMIT", flush=True)
|
||||
return
|
||||
try:
|
||||
log = make_k_variant(src["id"], new_id, style, "deepseek-v4.1-flash:cloud", base_url,
|
||||
K_RUN_BASE + 20 * n, tool_schema=None, pool="train")
|
||||
except BudgetExceeded as e:
|
||||
print("BUDGET", e, flush=True)
|
||||
return
|
||||
log.update(object_type=otype, spent_total=spent())
|
||||
stray = os.path.join(POOL, "generation.json")
|
||||
if os.path.exists(stray):
|
||||
os.remove(stray)
|
||||
os.makedirs(os.path.join(POOL, "_logs"), exist_ok=True)
|
||||
json.dump(log, open(os.path.join(POOL, "_logs", new_id + ".json"), "w"), indent=1)
|
||||
print(json.dumps(log), flush=True)
|
||||
|
||||
|
||||
def main():
|
||||
load_env(os.path.join(ROOT, ".env"))
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("cmd", choices=["plan", "run"])
|
||||
ap.add_argument("cmd", choices=["plan", "run", "k"])
|
||||
ap.add_argument("--part", type=int, default=0)
|
||||
ap.add_argument("--parts", type=int, default=1)
|
||||
ap.add_argument("--target", type=int, default=200)
|
||||
@@ -155,6 +203,9 @@ def main():
|
||||
print(len(p), "slots", collections.Counter(x["category"] for x in p))
|
||||
print(collections.Counter(x["object_type"] for x in p), collections.Counter(x.get("error_kind") for x in p))
|
||||
return
|
||||
if a.cmd == "k":
|
||||
run_k(K_COUNT)
|
||||
return
|
||||
run(a.part, a.parts, a.target, a.stop_ledger)
|
||||
|
||||
|
||||
|
||||
@@ -47,6 +47,15 @@ def done_keys():
|
||||
return {(r["task"], r["attempt"]) for r in map(json.loads, open(path))}
|
||||
|
||||
|
||||
def failed_first_attempts(min_score):
|
||||
"""Tasks whose attempt 0 did not give an acceptable run (score, end reason, harness error)."""
|
||||
path = os.path.join(OUT, "summary.jsonl")
|
||||
rows = [json.loads(l) for l in open(path)] if os.path.exists(path) else []
|
||||
return sorted(r["task"] for r in rows if r["attempt"] == 0 and (
|
||||
r.get("harness_error") or r.get("setup_failed") or r.get("end_reason") != "report"
|
||||
or (r.get("score") or 0) < min_score))
|
||||
|
||||
|
||||
def append(name, row):
|
||||
with LOCK, open(os.path.join(OUT, name), "a") as f:
|
||||
f.write(json.dumps(row) + "\n")
|
||||
@@ -91,7 +100,10 @@ def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("cmd", choices=["run", "list"])
|
||||
ap.add_argument("--workers", type=int, default=6)
|
||||
ap.add_argument("--attempts", type=int, default=1)
|
||||
ap.add_argument("--attempts", type=int, default=1,
|
||||
help="1 = one attempt per task; with --retry-failed: the second attempt only for failed tasks")
|
||||
ap.add_argument("--retry-failed", action="store_true")
|
||||
ap.add_argument("--min-score", type=float, default=80)
|
||||
ap.add_argument("--limit", type=int, help="maximum number of runs in this call")
|
||||
ap.add_argument("--tasks", nargs="*")
|
||||
a = ap.parse_args()
|
||||
@@ -99,6 +111,8 @@ def main():
|
||||
tasks = a.tasks or accepted_tasks()
|
||||
done = done_keys()
|
||||
todo = [(t, k) for k in range(a.attempts) for t in tasks if (t, k) not in done]
|
||||
if a.retry_failed: # attempt 1 only for tasks whose attempt 0 failed
|
||||
todo = [(t, 1) for t in failed_first_attempts(a.min_score) if t in tasks and (t, 1) not in done]
|
||||
if a.limit:
|
||||
todo = todo[:a.limit]
|
||||
print(len(tasks), "accepted tasks,", len(todo), "runs to do, spent", spent(), flush=True)
|
||||
|
||||
Reference in New Issue
Block a user