Pipeline controller: plan 2 (hard +, error +50 %), backlog throttle, 2 trajectory workers, stop rules, summaries every 50; budget reserve
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -35,7 +35,8 @@ def spent(since=None):
|
||||
|
||||
|
||||
def check_budget():
|
||||
limit = float(os.environ.get("BUDGET_LIMIT_USD", "45"))
|
||||
# BUDGET_RESERVE_USD: ledger USD kept back (second teacher test later); the guard stops at limit - reserve
|
||||
limit = float(os.environ.get("BUDGET_LIMIT_USD", "45")) - float(os.environ.get("BUDGET_RESERVE_USD", "0"))
|
||||
s = spent()
|
||||
if s >= limit:
|
||||
raise BudgetExceeded(f"cycle spend {s} USD >= limit {limit} USD")
|
||||
|
||||
353
harness/pipeline.py
Normal file
353
harness/pipeline.py
Normal file
@@ -0,0 +1,353 @@
|
||||
"""Stage 2 pipeline controller (rules of 2026-10-05, Opus 5.5 + Kral).
|
||||
|
||||
Runs in the background; nobody polls it. It
|
||||
- starts plan 2 generation (3 workers, hard and error share raised, no fixed target) when the first plan has
|
||||
ended, until --gen-deadline or the budget guard; adds K (free-text) variants at about 10 % of the other tasks;
|
||||
- runs trajectories with 2 workers (1 after any A4H error), at most 2 attempts per task: a second attempt only
|
||||
when the first one failed or was accepted without a repair;
|
||||
- writes a summary to train/STATE.md and docs/yol-haritasi.md every 50 accepted trajectories, and commits;
|
||||
- stops and reports (runs/pipeline/STOPPED.txt, STOP flag for the generators) when: acceptance over the last
|
||||
30 runs is below 50 %, one harness error pattern appears 3 times, the budget guard stops, or the deadline.
|
||||
|
||||
python3 -m harness.pipeline [--wait-pid PID] [--workers 2] [--gen-deadline 2026-10-10T18:00]
|
||||
[--traj-deadline 2026-10-11T23:30]
|
||||
Budget: BUDGET_LIMIT_USD and BUDGET_RESERVE_USD in .env (the reserve stays for a second teacher test).
|
||||
"""
|
||||
import argparse
|
||||
import glob
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import statistics
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
|
||||
from .adt_client import load_env
|
||||
from . import trainset, trajectories
|
||||
from .ledger import spent
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
sys.path.insert(0, os.path.join(ROOT, "train"))
|
||||
import accept as acc # noqa: E402 (train/accept.py)
|
||||
|
||||
OUT = trajectories.OUT
|
||||
PIPE = os.path.join(ROOT, "runs", "pipeline")
|
||||
BASE_LEDGER = 66.9665 # ledger when Part B started (2026-10-05 morning)
|
||||
LEDGER_TO_USAGE = 2.7
|
||||
ALLOWANCE_USAGE = 35.0 # until 11 October, without asking (Kral)
|
||||
MARKER = "<!-- stage2-summaries -->"
|
||||
GIT_ENV = dict(os.environ, GIT_AUTHOR_NAME="Kral", GIT_AUTHOR_EMAIL="kral@local",
|
||||
GIT_COMMITTER_NAME="Kral", GIT_COMMITTER_EMAIL="kral@local")
|
||||
HARNESS_MATCH = acc.HARNESS_ERRORS
|
||||
|
||||
|
||||
def log(*a):
|
||||
print(time.strftime("%F %T"), *a, flush=True)
|
||||
|
||||
|
||||
def ts(text):
|
||||
return time.mktime(time.strptime(text, "%Y-%m-%dT%H:%M"))
|
||||
|
||||
|
||||
class Pipeline:
|
||||
def __init__(self, a):
|
||||
self.a = a
|
||||
self.lock = threading.Lock()
|
||||
self.stop = threading.Event()
|
||||
self.reason = None
|
||||
self.max_workers = a.workers
|
||||
self.inflight = set()
|
||||
self.cache = {} # run_dir -> evaluation
|
||||
self.events = {} # harness error key -> count
|
||||
self.task_meta = {}
|
||||
self.gen_deadline, self.traj_deadline = ts(a.gen_deadline), ts(a.traj_deadline)
|
||||
self.children = []
|
||||
rows = self.rows()
|
||||
for r in rows:
|
||||
self.evaluate(r)
|
||||
self.milestone = self.accepted_total() // 50
|
||||
|
||||
# ---- data -----------------------------------------------------------------------------------------------
|
||||
def rows(self):
|
||||
p = os.path.join(OUT, "summary.jsonl")
|
||||
return [json.loads(l) for l in open(p)] if os.path.exists(p) else []
|
||||
|
||||
def meta(self, task_id):
|
||||
if task_id not in self.task_meta:
|
||||
try:
|
||||
t = json.load(open(os.path.join(trajectories.POOL, task_id, "task.json")))
|
||||
except OSError:
|
||||
t = {}
|
||||
self.task_meta[task_id] = {"category": t.get("category"), "object_type": t.get("object_type")}
|
||||
return self.task_meta[task_id]
|
||||
|
||||
def evaluate(self, row):
|
||||
"""{accepted, repair, reasons, event_keys, hints, tokens-free} of one summary row (cached)."""
|
||||
key = row.get("run_dir") or f"{row['task']}_{row['attempt']}"
|
||||
if key in self.cache:
|
||||
return self.cache[key]
|
||||
ev = {"task": row["task"], "attempt": row["attempt"], "accepted": False, "repair": False,
|
||||
"reasons": [], "events": [], "hints": 0}
|
||||
path = os.path.join(OUT, row.get("run_dir") or "-", "record.json")
|
||||
if row.get("harness_error"):
|
||||
ev["reasons"].append("harness_error")
|
||||
ev["events"].append("harness:" + str(row.get("error"))[:60])
|
||||
if row.get("setup_failed"):
|
||||
ev["reasons"].append("setup_failed")
|
||||
ev["events"].append("setup_failed")
|
||||
if os.path.exists(path):
|
||||
rec = json.load(open(path))
|
||||
ok, why = acc.judge(rec, row, self.a.min_score)
|
||||
ev["accepted"], ev["reasons"] = ok, why or ev["reasons"]
|
||||
ev["hints"] = rec["metadata"].get("syntax_hints") or 0
|
||||
if ok:
|
||||
ev["repair"] = acc.has_repair(acc.trim_loops(rec["messages"])[0])
|
||||
for r in rec["tool_results_raw"]:
|
||||
m = HARNESS_MATCH.search(r.get("result") or "")
|
||||
if m and not r.get("tool", "").startswith("sap_run_unit"):
|
||||
ev["events"].append("text:" + m.group(0)[:40])
|
||||
break
|
||||
if row.get("teardown_ok") is False:
|
||||
ev["events"].append("teardown")
|
||||
elif not ev["reasons"]:
|
||||
ev["reasons"].append("no_record")
|
||||
self.cache[key] = ev
|
||||
return ev
|
||||
|
||||
def evaluated(self):
|
||||
return [self.evaluate(r) for r in self.rows()]
|
||||
|
||||
def accepted_total(self):
|
||||
return sum(e["accepted"] for e in self.evaluated())
|
||||
|
||||
# ---- jobs -----------------------------------------------------------------------------------------------
|
||||
def next_job(self):
|
||||
with self.lock:
|
||||
rows = self.rows()
|
||||
done = {(r["task"], r["attempt"]) for r in rows}
|
||||
ev0 = {r["task"]: self.evaluate(r) for r in rows if r["attempt"] == 0}
|
||||
tasks = trajectories.accepted_tasks()
|
||||
for t in tasks: # first attempt for every task first
|
||||
if (t, 0) not in done and (t, 0) not in self.inflight:
|
||||
self.inflight.add((t, 0))
|
||||
return t, 0
|
||||
for t in tasks: # second attempt: first failed, or accepted without a repair
|
||||
e = ev0.get(t)
|
||||
if e and (t, 1) not in done and (t, 1) not in self.inflight and (not e["accepted"] or not e["repair"]):
|
||||
self.inflight.add((t, 1))
|
||||
return t, 1
|
||||
return None
|
||||
|
||||
def worker(self, i):
|
||||
while not self.stop.is_set():
|
||||
if time.time() > self.traj_deadline:
|
||||
self.halt("trajectory deadline reached")
|
||||
return
|
||||
if i >= self.max_workers:
|
||||
time.sleep(30)
|
||||
continue
|
||||
job = self.next_job()
|
||||
if not job:
|
||||
if self.generation_alive() or self.inflight:
|
||||
time.sleep(60)
|
||||
continue
|
||||
self.halt("no job left and generation ended")
|
||||
return
|
||||
trajectories.one(job[0], job[1], self.stop)
|
||||
with self.lock:
|
||||
self.inflight.discard(job)
|
||||
self.after_run(job)
|
||||
if self.stop.is_set() and not self.reason:
|
||||
self.halt("budget guard stopped a run")
|
||||
|
||||
# ---- rules ----------------------------------------------------------------------------------------------
|
||||
def after_run(self, job):
|
||||
rows = self.rows()
|
||||
row = next((r for r in reversed(rows) if (r["task"], r["attempt"]) == job), None)
|
||||
if row is None:
|
||||
return
|
||||
ev = self.evaluate(row)
|
||||
for k in ev["events"]:
|
||||
self.events[k] = self.events.get(k, 0) + 1
|
||||
log("A4H/harness event:", k, "x", self.events[k], "->", row.get("run_dir"))
|
||||
if self.max_workers > 1:
|
||||
self.max_workers = 1
|
||||
log("harness error seen: trajectory workers 2 -> 1")
|
||||
if self.events[k] >= 3:
|
||||
self.halt(f"harness error pattern repeated 3 times: {k}")
|
||||
evs = self.evaluated()
|
||||
if len(evs) >= 30:
|
||||
rate = sum(e["accepted"] for e in evs[-30:]) / 30
|
||||
if rate < 0.5:
|
||||
self.halt(f"acceptance {rate:.0%} over the last 30 runs (below 50 %)")
|
||||
try:
|
||||
from .ledger import check_budget
|
||||
check_budget()
|
||||
except Exception as e: # noqa: BLE001 BudgetExceeded
|
||||
self.halt(f"budget guard: {e}")
|
||||
total = sum(e["accepted"] for e in evs)
|
||||
while total // 50 > self.milestone and not self.reason:
|
||||
self.milestone += 1
|
||||
self.summary(self.milestone * 50)
|
||||
|
||||
def halt(self, reason):
|
||||
if self.stop.is_set() and self.reason:
|
||||
return
|
||||
self.reason = reason
|
||||
self.stop.set()
|
||||
os.makedirs(PIPE, exist_ok=True)
|
||||
open(trainset.STOP_FLAG, "w").write(reason + "\n") # generators end after their current slot
|
||||
log("STOP:", reason)
|
||||
|
||||
# ---- generation side --------------------------------------------------------------------------------------
|
||||
def generation_alive(self):
|
||||
if self.reason:
|
||||
return False
|
||||
out = subprocess.run(["pgrep", "-f", r"harness\.trainset (run|k)"], capture_output=True, text=True).stdout
|
||||
return bool(out.split()) or time.time() < self.gen_deadline
|
||||
|
||||
def old_generators(self):
|
||||
out = subprocess.run(["pgrep", "-fl", r"harness\.trainset run --part"], capture_output=True, text=True).stdout
|
||||
return [l for l in out.splitlines() if "--plan plan2" not in l]
|
||||
|
||||
def supervisor(self):
|
||||
launched = False
|
||||
k_proc = None
|
||||
crashes = 0
|
||||
while not self.stop.is_set():
|
||||
time.sleep(60)
|
||||
if time.time() > self.gen_deadline or os.path.exists(trainset.STOP_FLAG):
|
||||
continue
|
||||
if not launched and not self.old_generators():
|
||||
trainset.ensure_plan("plan2")
|
||||
dl = self.a.gen_deadline
|
||||
for i in range(3):
|
||||
f = open(os.path.join(ROOT, "runs", "gen_train", f"p2_w{i}.log"), "a")
|
||||
self.children.append(subprocess.Popen(
|
||||
[sys.executable, "-m", "harness.trainset", "run", "--plan", "plan2", "--part", str(i),
|
||||
"--parts", "3", "--deadline", dl], cwd=ROOT, stdout=f, stderr=f))
|
||||
launched = True
|
||||
log("plan 2 generators started (3 workers, deadline", dl + ")")
|
||||
elif launched:
|
||||
for i, c in enumerate(self.children):
|
||||
if c.poll() not in (None, 0) and crashes < 3:
|
||||
crashes += 1
|
||||
log("generator", i, "exited with", c.returncode, "- restarted")
|
||||
f = open(os.path.join(ROOT, "runs", "gen_train", f"p2_w{i}.log"), "a")
|
||||
self.children[i] = subprocess.Popen(
|
||||
[sys.executable, "-m", "harness.trainset", "run", "--plan", "plan2", "--part", str(i),
|
||||
"--parts", "3", "--deadline", self.a.gen_deadline], cwd=ROOT, stdout=f, stderr=f)
|
||||
# K variants: free text, about 10 % of the other accepted training tasks
|
||||
if k_proc is None or k_proc.poll() is not None:
|
||||
logs = [json.load(open(f)) for f in glob.glob(os.path.join(trainset.POOL, "_logs", "G*.json"))]
|
||||
base = sum(1 for l in logs if l.get("accepted") and l.get("category") != "K")
|
||||
have = sum(1 for l in logs if l.get("category") == "K")
|
||||
want = min(99, int(0.1 * base))
|
||||
if want > have:
|
||||
f = open(os.path.join(ROOT, "runs", "gen_train", "k.log"), "a")
|
||||
k_proc = subprocess.Popen([sys.executable, "-m", "harness.trainset", "k", "--k-count", str(want)],
|
||||
cwd=ROOT, stdout=f, stderr=f)
|
||||
|
||||
# ---- summary ----------------------------------------------------------------------------------------------
|
||||
def summary(self, n):
|
||||
log("summary at", n, "accepted trajectories")
|
||||
subprocess.run([sys.executable, "train/accept.py", "--min-score", str(self.a.min_score)], cwd=ROOT,
|
||||
capture_output=True)
|
||||
vp = os.path.join(ROOT, "train", ".venv", "bin", "python")
|
||||
subprocess.run([vp, "train/to_qwen.py", os.path.join(OUT, "accepted.jsonl"), os.path.join(OUT, "qwen.jsonl")],
|
||||
cwd=ROOT, capture_output=True)
|
||||
toks = sorted(json.loads(l)["n_tokens"] for l in open(os.path.join(OUT, "qwen.jsonl")))
|
||||
pct = lambda p: toks[min(len(toks) - 1, int(p * len(toks)))] if toks else None # noqa: E731
|
||||
rows, evs = self.rows(), self.evaluated()
|
||||
logs = [json.load(open(f)) for f in glob.glob(os.path.join(trainset.POOL, "_logs", "G*.json"))]
|
||||
acc_t = sum(1 for l in logs if l.get("accepted"))
|
||||
accepted = [e for e in evs if e["accepted"]]
|
||||
repairs = sum(e["repair"] for e in accepted)
|
||||
|
||||
def table(keyname):
|
||||
agg = {}
|
||||
for e in evs:
|
||||
k = self.meta(e["task"])[keyname] or "?"
|
||||
a = agg.setdefault(k, [0, 0, 0])
|
||||
a[0] += 1
|
||||
a[1] += e["accepted"]
|
||||
a[2] += e["accepted"] and e["repair"]
|
||||
return "; ".join(f"{k} {v[1]}/{v[0]} (repair {v[2]})" for k, v in sorted(agg.items()))
|
||||
s = spent()
|
||||
used = (s - BASE_LEDGER) / LEDGER_TO_USAGE
|
||||
guard = float(os.environ.get("BUDGET_LIMIT_USD", "0")) - float(os.environ.get("BUDGET_RESERVE_USD", "0"))
|
||||
text = f"""### Stage 2 summary at {n} accepted trajectories ({time.strftime('%Y-%m-%d %H:%M')})
|
||||
|
||||
- Tasks: {acc_t} accepted of {len(logs)} generated (K variants {sum(1 for l in logs if l.get('category') == 'K')}); \
|
||||
trajectory runs {len(rows)}, accepted trajectories {len(accepted)} (acceptance {len(accepted) / max(len(rows), 1):.0%}).
|
||||
- Budget: used {used:.1f} usage (ledger {s - BASE_LEDGER:.1f}) since 2026-10-05; left to the guard {(guard - s) / LEDGER_TO_USAGE:.1f} usage \
|
||||
(ledger {guard - s:.1f}; reserve {os.environ.get('BUDGET_RESERVE_USD', '0')} ledger kept); allowance until 11 October {ALLOWANCE_USAGE - used:.1f} usage.
|
||||
- Repair share (accepted trajectories with an error followed by a fix): {repairs}/{len(accepted)} = {repairs / max(len(accepted), 1):.0%}.
|
||||
- By category, accepted/runs: {table('category')}.
|
||||
- By object type, accepted/runs: {table('object_type')}.
|
||||
- Tokens of accepted samples (20 tool schemas kept): p50 {pct(0.5)}, p90 {pct(0.9)}, p95 {pct(0.95)}, max {toks[-1] if toks else None}, n {len(toks)}.
|
||||
- Syntax hints (proxy syntaxCheck added): {sum(e['hints'] for e in evs)} in {sum(1 for e in evs if e['hints'])} runs.
|
||||
- Harness events: {sum(self.events.values())} ({self.events or 'none'}); trajectory workers now {self.max_workers}.
|
||||
"""
|
||||
if getattr(self.a, "dry_summary", False):
|
||||
print(text)
|
||||
return
|
||||
with open(os.path.join(ROOT, "train", "STATE.md"), "a") as f:
|
||||
f.write("\n" + text)
|
||||
p = os.path.join(ROOT, "docs", "yol-haritasi.md")
|
||||
d = open(p).read()
|
||||
d = d.replace(MARKER, MARKER + "\n\n" + text, 1) if MARKER in d else d + "\n" + text
|
||||
open(p, "w").write(d)
|
||||
subprocess.run(["git", "add", "train/STATE.md", "docs/yol-haritasi.md", "tasks_gen/train", "harness", "train"],
|
||||
cwd=ROOT, env=GIT_ENV, capture_output=True)
|
||||
subprocess.run(["git", "commit", "-q", "-m", f"Stage 2 summary at {n} accepted trajectories\n\n"
|
||||
"Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>"],
|
||||
cwd=ROOT, env=GIT_ENV, capture_output=True)
|
||||
|
||||
# ---- main -------------------------------------------------------------------------------------------------
|
||||
def run(self):
|
||||
os.makedirs(PIPE, exist_ok=True)
|
||||
if os.path.exists(trainset.STOP_FLAG):
|
||||
os.remove(trainset.STOP_FLAG)
|
||||
log("pipeline start: workers", self.max_workers, "accepted", self.accepted_total(), "spent", spent())
|
||||
threads = [threading.Thread(target=self.worker, args=(i,)) for i in range(self.a.workers)]
|
||||
threading.Thread(target=self.supervisor, daemon=True).start()
|
||||
for t in threads:
|
||||
t.start()
|
||||
for t in threads:
|
||||
t.join()
|
||||
evs = self.evaluated()
|
||||
text = (f"reason: {self.reason}\ntime: {time.strftime('%F %T')}\nruns: {len(evs)}, accepted "
|
||||
f"{sum(e['accepted'] for e in evs)}\nspent ledger: {spent()}\nevents: {self.events}\n")
|
||||
open(os.path.join(PIPE, "STOPPED.txt"), "w").write(text)
|
||||
log("pipeline ended:", text.replace("\n", " | "))
|
||||
|
||||
|
||||
def main():
|
||||
load_env(os.path.join(ROOT, ".env"))
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--wait-pid", type=int, help="start when this process has ended (the first trajectory batch)")
|
||||
ap.add_argument("--workers", type=int, default=2)
|
||||
ap.add_argument("--min-score", type=float, default=80)
|
||||
ap.add_argument("--gen-deadline", default="2026-10-10T18:00")
|
||||
ap.add_argument("--traj-deadline", default="2026-10-11T23:30")
|
||||
ap.add_argument("--dry-summary", action="store_true", help="print the summary text and exit (no file, no commit)")
|
||||
a = ap.parse_args()
|
||||
if a.dry_summary:
|
||||
Pipeline(a).summary(0)
|
||||
return
|
||||
if a.wait_pid:
|
||||
while True:
|
||||
try:
|
||||
os.kill(a.wait_pid, 0)
|
||||
except OSError:
|
||||
break
|
||||
time.sleep(30)
|
||||
os.makedirs(OUT, exist_ok=True)
|
||||
Pipeline(a).run()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -46,13 +46,16 @@ CATEGORY_SHARE = {"A": 10, "B": 15, "C": 15, "D": 10, "E": 15, "F": 10, "G": 10,
|
||||
ERROR_CATEGORY = {"named-type": "A", "reserved-word": "D", "long-names": "C"}
|
||||
|
||||
|
||||
def build_plan():
|
||||
"""Category shares as in the eval plan; object types inside a category as in evalset.SLOTS."""
|
||||
def build_plan(n_slots=N_SLOTS, n_error=N_ERROR, hard_share=0.0, first_id=FIRST_ID, run_base=RUN_BASE):
|
||||
"""Category shares as in the eval plan; object types inside a category as in evalset.SLOTS.
|
||||
n_error slots target the common errors (named-type, reserved-word, long-names); hard_share of the other
|
||||
slots get difficulty 3 (every 10th slot pattern, deterministic)."""
|
||||
share_sum = sum(CATEGORY_SHARE.values())
|
||||
items = []
|
||||
for cat, share in CATEGORY_SHARE.items():
|
||||
k_cat = round(N_SLOTS * share / share_sum)
|
||||
k_err = sum(1 for k in ERROR_CATEGORY if ERROR_CATEGORY[k] == cat) * N_ERROR // 3
|
||||
k_cat = round(n_slots * share / share_sum)
|
||||
per_kind = n_error // 3
|
||||
k_err = sum(1 for k in ERROR_CATEGORY if ERROR_CATEGORY[k] == cat) * per_kind
|
||||
slots = [x for x in SLOTS if x[0] == cat]
|
||||
weight = sum(n for _, _, n, _ in slots)
|
||||
k_rest = k_cat - k_err
|
||||
@@ -69,16 +72,19 @@ def build_plan():
|
||||
for kind, text in ERROR_HINTS:
|
||||
if ERROR_CATEGORY[kind] != cat:
|
||||
continue
|
||||
for i in range(N_ERROR // 3):
|
||||
for i in range(per_kind):
|
||||
t = ERROR_TYPES[i % len(ERROR_TYPES)]
|
||||
pos += 1
|
||||
items.append({"category": cat, "object_type": t, "topic": text, "error_kind": kind,
|
||||
"frac": pos / (k_cat + 1)})
|
||||
items.sort(key=lambda x: (x["frac"], x["category"], x["object_type"])) # interleave: a cut keeps the mix
|
||||
hard_every = round(1 / hard_share) if hard_share else 0
|
||||
for n, it in enumerate(items):
|
||||
it.pop("frac")
|
||||
it["id"] = f"G{FIRST_ID + n:04d}"
|
||||
it["run_base"] = RUN_BASE + 40 * n
|
||||
it["id"] = f"G{first_id + n:04d}"
|
||||
it["run_base"] = run_base + 40 * n
|
||||
if hard_every and it["category"] != "H" and n % hard_every == hard_every // 2:
|
||||
it["difficulty"] = 3
|
||||
return items
|
||||
|
||||
|
||||
@@ -92,12 +98,42 @@ def accepted_count():
|
||||
return n
|
||||
|
||||
|
||||
def run(part, parts, target, stop_ledger):
|
||||
STOP_FLAG = os.path.join(ROOT, "runs", "pipeline", "STOP") # the pipeline writes it: workers end after the current slot
|
||||
BACKLOG_LIMIT = 40 # accepted tasks without a first trajectory run: generation waits above this
|
||||
PLAN2 = {"n_slots": 500, "n_error": 100, "hard_share": 0.3, "first_id": 1400, "run_base": 100000}
|
||||
# plan 2 (2026-10-05, Opus 5.5 rules): error tasks 20 % (plan 1: 13 %, +50 %), hard (difficulty 3) 30 % of the
|
||||
# other slots (plan 1: 20 % assumed from the pilot hard batch), same category shares, ids G1400+, no fixed target
|
||||
|
||||
|
||||
def plan_path(name):
|
||||
return os.path.join(POOL, name + ".json")
|
||||
|
||||
|
||||
def ensure_plan(name, stop_ledger=27.0, target=200):
|
||||
os.makedirs(POOL, exist_ok=True)
|
||||
if not os.path.exists(PLAN):
|
||||
json.dump({"slots": build_plan(), "ledger_at_start": spent(), "stop_ledger": stop_ledger,
|
||||
"target": target, "created": time.strftime("%F %T")}, open(PLAN, "w"), indent=1)
|
||||
plan = json.load(open(PLAN))
|
||||
if not os.path.exists(plan_path(name)):
|
||||
if name == "plan":
|
||||
slots, tgt, stop = build_plan(), target, stop_ledger
|
||||
else:
|
||||
slots, tgt, stop = build_plan(**PLAN2), None, 1e6 # only the budget guard and the deadline stop it
|
||||
json.dump({"slots": slots, "ledger_at_start": spent(), "stop_ledger": stop, "target": tgt,
|
||||
"created": time.strftime("%F %T")}, open(plan_path(name), "w"), indent=1)
|
||||
return json.load(open(plan_path(name)))
|
||||
|
||||
|
||||
def backlog():
|
||||
"""Accepted training tasks that have no first trajectory run yet."""
|
||||
done = set()
|
||||
sp = os.path.join(ROOT, "runs", "traj", "summary.jsonl")
|
||||
if os.path.exists(sp):
|
||||
done = {json.loads(l)["task"] for l in open(sp) if json.loads(l)["attempt"] == 0}
|
||||
acc = {os.path.basename(os.path.dirname(f)) for f in glob.glob(os.path.join(POOL, "G*", "generation.json"))
|
||||
if json.load(open(f)).get("accepted")}
|
||||
return len(acc - done)
|
||||
|
||||
|
||||
def run(part, parts, target, stop_ledger, plan_name="plan", deadline=None):
|
||||
plan = ensure_plan(plan_name, stop_ledger, target)
|
||||
stop_at = plan["ledger_at_start"] + plan["stop_ledger"]
|
||||
base_url = os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")
|
||||
evals = overlap.load_pool("eval")
|
||||
@@ -106,7 +142,16 @@ def run(part, parts, target, stop_ledger):
|
||||
continue
|
||||
if os.path.exists(os.path.join(POOL, "_logs", s["id"] + ".json")):
|
||||
continue # done or failed: not tried again (delete the log to retry)
|
||||
if accepted_count() >= plan["target"]:
|
||||
while backlog() > BACKLOG_LIMIT and not os.path.exists(STOP_FLAG) \
|
||||
and not (deadline and time.time() > deadline):
|
||||
time.sleep(120) # trajectories are the slower side: do not generate tasks that wait for days
|
||||
if os.path.exists(STOP_FLAG):
|
||||
print("STOP flag", flush=True)
|
||||
break
|
||||
if deadline and time.time() > deadline:
|
||||
print("DEADLINE", flush=True)
|
||||
break
|
||||
if plan["target"] and accepted_count() >= plan["target"]:
|
||||
print("TARGET reached", flush=True)
|
||||
break
|
||||
if spent() >= stop_at:
|
||||
@@ -124,14 +169,15 @@ def run(part, parts, target, stop_ledger):
|
||||
f"names {sc['name']:.2f}). Choose a different business topic and different object names."
|
||||
for i, sc in hits[:3]]
|
||||
try:
|
||||
log = generate(s["id"], "train", s["object_type"], s["category"], 2, "deepseek-v4.1-flash:cloud",
|
||||
log = generate(s["id"], "train", s["object_type"], s["category"], s.get("difficulty", 2),
|
||||
"deepseek-v4.1-flash:cloud",
|
||||
base_url, s["run_base"], topic, extra_check=extra)
|
||||
except BudgetExceeded as e:
|
||||
print("BUDGET", e, flush=True)
|
||||
break
|
||||
except Exception as e: # noqa: BLE001
|
||||
log = {"id": s["id"], "error": str(e)[:500]}
|
||||
log.update(error_kind=s.get("error_kind"), spent_total=spent())
|
||||
log.update(error_kind=s.get("error_kind"), difficulty=s.get("difficulty", 2), spent_total=spent())
|
||||
os.makedirs(os.path.join(POOL, "_logs"), exist_ok=True)
|
||||
json.dump(log, open(os.path.join(POOL, "_logs", s["id"] + ".json"), "w"), indent=1)
|
||||
stray = os.path.join(POOL, "generation.json") # generate() writes here when no task dir exists
|
||||
@@ -156,7 +202,7 @@ def run_k(count):
|
||||
continue
|
||||
used = {json.load(open(f)).get("base_task") for f in glob.glob(os.path.join(POOL, "_logs", "G13*.json"))}
|
||||
cands = [] # accepted, not K, not H (a stop task has no free-text form), not used yet
|
||||
for f in sorted(glob.glob(os.path.join(POOL, "_logs", "G1[0-2]*.json"))):
|
||||
for f in sorted(glob.glob(os.path.join(POOL, "_logs", "G1*.json"))):
|
||||
lg = json.load(open(f))
|
||||
if lg.get("accepted") and lg.get("category") not in ("H", "K") and lg["id"] not in used:
|
||||
cands.append(lg)
|
||||
@@ -170,8 +216,7 @@ def run_k(count):
|
||||
otype = min(kinds, key=lambda t: done_types.count(t))
|
||||
src = kinds[otype][0]
|
||||
style = K_STYLES_CYCLE[n % 2]
|
||||
if spent() >= json.load(open(PLAN))["ledger_at_start"] + json.load(open(PLAN))["stop_ledger"]:
|
||||
print("PHASE LIMIT", flush=True)
|
||||
if os.path.exists(STOP_FLAG):
|
||||
return
|
||||
try:
|
||||
log = make_k_variant(src["id"], new_id, style, "deepseek-v4.1-flash:cloud", base_url,
|
||||
@@ -195,6 +240,9 @@ def main():
|
||||
ap.add_argument("--part", type=int, default=0)
|
||||
ap.add_argument("--parts", type=int, default=1)
|
||||
ap.add_argument("--target", type=int, default=200)
|
||||
ap.add_argument("--plan", default="plan", help="plan (first 223 slots) or plan2 (hard and error share raised)")
|
||||
ap.add_argument("--deadline", help="YYYY-MM-DDTHH:MM local time: no new slot after it")
|
||||
ap.add_argument("--k-count", type=int, default=K_COUNT)
|
||||
ap.add_argument("--stop-ledger", type=float, default=27.0, help="ledger USD for this phase (10 USD usage = 27)")
|
||||
a = ap.parse_args()
|
||||
if a.cmd == "plan":
|
||||
@@ -204,9 +252,10 @@ def main():
|
||||
print(collections.Counter(x["object_type"] for x in p), collections.Counter(x.get("error_kind") for x in p))
|
||||
return
|
||||
if a.cmd == "k":
|
||||
run_k(K_COUNT)
|
||||
run_k(a.k_count)
|
||||
return
|
||||
run(a.part, a.parts, a.target, a.stop_ledger)
|
||||
dl = time.mktime(time.strptime(a.deadline, "%Y-%m-%dT%H:%M")) if a.deadline else None
|
||||
run(a.part, a.parts, a.target, a.stop_ledger, a.plan, dl)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -24,7 +24,7 @@ from .runner import Runner
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
POOL = os.path.join(ROOT, "tasks_gen", "train")
|
||||
OUT = os.path.join(ROOT, "runs", "traj")
|
||||
RUN_BASE = 41000 # 41000 + (task number - 1000) * 3 + attempt; below 36**3 * ... (prefix rule)
|
||||
RUN_BASE = 200000 # 200000 + (task number - 1000) * 3 + attempt (a digit must lead the 4-char base36 run: < 466560)
|
||||
MODEL = "deepseek-v4.1-flash:cloud"
|
||||
LOCK = threading.Lock()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user