Pipeline controller: plan 2 (hard +, error +50 %), backlog throttle, 2 trajectory workers, stop rules, summaries every 50; budget reserve

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-05 13:35:54 +02:00
parent 1c4667a0fc
commit 13360292dd
781 changed files with 50211 additions and 22 deletions

View File

@@ -46,13 +46,16 @@ CATEGORY_SHARE = {"A": 10, "B": 15, "C": 15, "D": 10, "E": 15, "F": 10, "G": 10,
ERROR_CATEGORY = {"named-type": "A", "reserved-word": "D", "long-names": "C"}
def build_plan():
"""Category shares as in the eval plan; object types inside a category as in evalset.SLOTS."""
def build_plan(n_slots=N_SLOTS, n_error=N_ERROR, hard_share=0.0, first_id=FIRST_ID, run_base=RUN_BASE):
"""Category shares as in the eval plan; object types inside a category as in evalset.SLOTS.
n_error slots target the common errors (named-type, reserved-word, long-names); hard_share of the other
slots get difficulty 3 (every 10th slot pattern, deterministic)."""
share_sum = sum(CATEGORY_SHARE.values())
items = []
for cat, share in CATEGORY_SHARE.items():
k_cat = round(N_SLOTS * share / share_sum)
k_err = sum(1 for k in ERROR_CATEGORY if ERROR_CATEGORY[k] == cat) * N_ERROR // 3
k_cat = round(n_slots * share / share_sum)
per_kind = n_error // 3
k_err = sum(1 for k in ERROR_CATEGORY if ERROR_CATEGORY[k] == cat) * per_kind
slots = [x for x in SLOTS if x[0] == cat]
weight = sum(n for _, _, n, _ in slots)
k_rest = k_cat - k_err
@@ -69,16 +72,19 @@ def build_plan():
for kind, text in ERROR_HINTS:
if ERROR_CATEGORY[kind] != cat:
continue
for i in range(N_ERROR // 3):
for i in range(per_kind):
t = ERROR_TYPES[i % len(ERROR_TYPES)]
pos += 1
items.append({"category": cat, "object_type": t, "topic": text, "error_kind": kind,
"frac": pos / (k_cat + 1)})
items.sort(key=lambda x: (x["frac"], x["category"], x["object_type"])) # interleave: a cut keeps the mix
hard_every = round(1 / hard_share) if hard_share else 0
for n, it in enumerate(items):
it.pop("frac")
it["id"] = f"G{FIRST_ID + n:04d}"
it["run_base"] = RUN_BASE + 40 * n
it["id"] = f"G{first_id + n:04d}"
it["run_base"] = run_base + 40 * n
if hard_every and it["category"] != "H" and n % hard_every == hard_every // 2:
it["difficulty"] = 3
return items
@@ -92,12 +98,42 @@ def accepted_count():
return n
def run(part, parts, target, stop_ledger):
STOP_FLAG = os.path.join(ROOT, "runs", "pipeline", "STOP") # the pipeline writes it: workers end after the current slot
BACKLOG_LIMIT = 40 # accepted tasks without a first trajectory run: generation waits above this
PLAN2 = {"n_slots": 500, "n_error": 100, "hard_share": 0.3, "first_id": 1400, "run_base": 100000}
# plan 2 (2026-10-05, Opus 5.5 rules): error tasks 20 % (plan 1: 13 %, +50 %), hard (difficulty 3) 30 % of the
# other slots (plan 1: 20 % assumed from the pilot hard batch), same category shares, ids G1400+, no fixed target
def plan_path(name):
return os.path.join(POOL, name + ".json")
def ensure_plan(name, stop_ledger=27.0, target=200):
os.makedirs(POOL, exist_ok=True)
if not os.path.exists(PLAN):
json.dump({"slots": build_plan(), "ledger_at_start": spent(), "stop_ledger": stop_ledger,
"target": target, "created": time.strftime("%F %T")}, open(PLAN, "w"), indent=1)
plan = json.load(open(PLAN))
if not os.path.exists(plan_path(name)):
if name == "plan":
slots, tgt, stop = build_plan(), target, stop_ledger
else:
slots, tgt, stop = build_plan(**PLAN2), None, 1e6 # only the budget guard and the deadline stop it
json.dump({"slots": slots, "ledger_at_start": spent(), "stop_ledger": stop, "target": tgt,
"created": time.strftime("%F %T")}, open(plan_path(name), "w"), indent=1)
return json.load(open(plan_path(name)))
def backlog():
"""Accepted training tasks that have no first trajectory run yet."""
done = set()
sp = os.path.join(ROOT, "runs", "traj", "summary.jsonl")
if os.path.exists(sp):
done = {json.loads(l)["task"] for l in open(sp) if json.loads(l)["attempt"] == 0}
acc = {os.path.basename(os.path.dirname(f)) for f in glob.glob(os.path.join(POOL, "G*", "generation.json"))
if json.load(open(f)).get("accepted")}
return len(acc - done)
def run(part, parts, target, stop_ledger, plan_name="plan", deadline=None):
plan = ensure_plan(plan_name, stop_ledger, target)
stop_at = plan["ledger_at_start"] + plan["stop_ledger"]
base_url = os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")
evals = overlap.load_pool("eval")
@@ -106,7 +142,16 @@ def run(part, parts, target, stop_ledger):
continue
if os.path.exists(os.path.join(POOL, "_logs", s["id"] + ".json")):
continue # done or failed: not tried again (delete the log to retry)
if accepted_count() >= plan["target"]:
while backlog() > BACKLOG_LIMIT and not os.path.exists(STOP_FLAG) \
and not (deadline and time.time() > deadline):
time.sleep(120) # trajectories are the slower side: do not generate tasks that wait for days
if os.path.exists(STOP_FLAG):
print("STOP flag", flush=True)
break
if deadline and time.time() > deadline:
print("DEADLINE", flush=True)
break
if plan["target"] and accepted_count() >= plan["target"]:
print("TARGET reached", flush=True)
break
if spent() >= stop_at:
@@ -124,14 +169,15 @@ def run(part, parts, target, stop_ledger):
f"names {sc['name']:.2f}). Choose a different business topic and different object names."
for i, sc in hits[:3]]
try:
log = generate(s["id"], "train", s["object_type"], s["category"], 2, "deepseek-v4.1-flash:cloud",
log = generate(s["id"], "train", s["object_type"], s["category"], s.get("difficulty", 2),
"deepseek-v4.1-flash:cloud",
base_url, s["run_base"], topic, extra_check=extra)
except BudgetExceeded as e:
print("BUDGET", e, flush=True)
break
except Exception as e: # noqa: BLE001
log = {"id": s["id"], "error": str(e)[:500]}
log.update(error_kind=s.get("error_kind"), spent_total=spent())
log.update(error_kind=s.get("error_kind"), difficulty=s.get("difficulty", 2), spent_total=spent())
os.makedirs(os.path.join(POOL, "_logs"), exist_ok=True)
json.dump(log, open(os.path.join(POOL, "_logs", s["id"] + ".json"), "w"), indent=1)
stray = os.path.join(POOL, "generation.json") # generate() writes here when no task dir exists
@@ -156,7 +202,7 @@ def run_k(count):
continue
used = {json.load(open(f)).get("base_task") for f in glob.glob(os.path.join(POOL, "_logs", "G13*.json"))}
cands = [] # accepted, not K, not H (a stop task has no free-text form), not used yet
for f in sorted(glob.glob(os.path.join(POOL, "_logs", "G1[0-2]*.json"))):
for f in sorted(glob.glob(os.path.join(POOL, "_logs", "G1*.json"))):
lg = json.load(open(f))
if lg.get("accepted") and lg.get("category") not in ("H", "K") and lg["id"] not in used:
cands.append(lg)
@@ -170,8 +216,7 @@ def run_k(count):
otype = min(kinds, key=lambda t: done_types.count(t))
src = kinds[otype][0]
style = K_STYLES_CYCLE[n % 2]
if spent() >= json.load(open(PLAN))["ledger_at_start"] + json.load(open(PLAN))["stop_ledger"]:
print("PHASE LIMIT", flush=True)
if os.path.exists(STOP_FLAG):
return
try:
log = make_k_variant(src["id"], new_id, style, "deepseek-v4.1-flash:cloud", base_url,
@@ -195,6 +240,9 @@ def main():
ap.add_argument("--part", type=int, default=0)
ap.add_argument("--parts", type=int, default=1)
ap.add_argument("--target", type=int, default=200)
ap.add_argument("--plan", default="plan", help="plan (first 223 slots) or plan2 (hard and error share raised)")
ap.add_argument("--deadline", help="YYYY-MM-DDTHH:MM local time: no new slot after it")
ap.add_argument("--k-count", type=int, default=K_COUNT)
ap.add_argument("--stop-ledger", type=float, default=27.0, help="ledger USD for this phase (10 USD usage = 27)")
a = ap.parse_args()
if a.cmd == "plan":
@@ -204,9 +252,10 @@ def main():
print(collections.Counter(x["object_type"] for x in p), collections.Counter(x.get("error_kind") for x in p))
return
if a.cmd == "k":
run_k(K_COUNT)
run_k(a.k_count)
return
run(a.part, a.parts, a.target, a.stop_ledger)
dl = time.mktime(time.strptime(a.deadline, "%Y-%m-%dT%H:%M")) if a.deadline else None
run(a.part, a.parts, a.target, a.stop_ledger, a.plan, dl)
if __name__ == "__main__":