78 lines
3.5 KiB
Python
78 lines
3.5 KiB
Python
"""State and suggested settings for restarting the cloud work after the budget reset (12 October).
|
|
|
|
python3 -m harness.restart_plan [--panel 0.0] [--ratio 1.2]
|
|
|
|
Prints (no cloud call, no file change): ledger, the suggested BUDGET_* values from the panel value, the object
|
|
type deficits, the tasks waiting for a first run per kind, the second attempt candidates, and the commands.
|
|
"""
|
|
import argparse
|
|
import json
|
|
import os
|
|
|
|
from .adt_client import load_env
|
|
from . import mix
|
|
from .ledger import spent
|
|
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
RESERVE_USAGE = 3.0
|
|
LEDGER_RESERVE = 8
|
|
|
|
|
|
def main():
|
|
load_env(os.path.join(ROOT, ".env"))
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--panel", type=float, default=0.0, help="Ollama panel usage after the reset (USD)")
|
|
ap.add_argument("--ratio", type=float, default=1.2, help="ledger per usage, pessimistic (trajectory runs 1.2)")
|
|
a = ap.parse_args()
|
|
s = spent("2026-10-12")
|
|
usable = 60 - a.panel - RESERVE_USAGE
|
|
limit = round(usable * a.ratio + LEDGER_RESERVE)
|
|
print(f"ledger since 2026-10-12: {s:.2f}")
|
|
print(f"panel {a.panel} of 60, reserve {RESERVE_USAGE} usage -> usable {usable:.1f} usage x {a.ratio} = {usable * a.ratio:.0f} ledger")
|
|
print(f"set in .env: BUDGET_CYCLE_START=2026-10-12 BUDGET_LIMIT_USD={limit + int(s)} BUDGET_RESERVE_USD={LEDGER_RESERVE}")
|
|
rows = [json.loads(l) for l in open(os.path.join(ROOT, "runs", "traj", "summary.jsonl"))]
|
|
import sys
|
|
sys.path.insert(0, os.path.join(ROOT, "train"))
|
|
import accept
|
|
acc_traj, runs = {}, {}
|
|
first = {}
|
|
for r in rows:
|
|
k = mix.kind_of_task_dir(r["task"])
|
|
runs[k] = runs.get(k, 0) + 1
|
|
ok = False
|
|
p = os.path.join(ROOT, "runs", "traj", r.get("run_dir") or "-", "record.json")
|
|
if os.path.exists(p):
|
|
ok = accept.judge(json.load(open(p)), r, 80)[0]
|
|
acc_traj[k] = acc_traj.get(k, 0) + ok
|
|
if r["attempt"] == 0:
|
|
first[r["task"]] = ok
|
|
tot_t = max(sum(acc_traj.values()), 1)
|
|
tasks = mix.accepted_task_counts()
|
|
ss = sum(mix.TYPE_SHARE.values())
|
|
print("\nkind target tasks traj acc/runs (share) below target (generate + run)")
|
|
below_t, below_r = mix.below_target(tasks), mix.below_target(acc_traj)
|
|
for k, v in mix.TYPE_SHARE.items():
|
|
print(f"{k:5s} {100 * v / ss:5.1f}% {tasks.get(k, 0):6d} {acc_traj.get(k, 0):4d}/{runs.get(k, 0):<4d} ({100 * acc_traj.get(k, 0) / tot_t:4.1f}%) "
|
|
f"tasks:{'yes' if k in below_t else 'no '} trajectories:{'yes' if k in below_r else 'no'}")
|
|
done0 = {r["task"] for r in rows if r["attempt"] == 0}
|
|
done1 = {r["task"] for r in rows if r["attempt"] == 1}
|
|
waiting, second = {}, {}
|
|
import glob
|
|
for f in glob.glob(os.path.join(ROOT, "tasks_gen", "train", "G*", "generation.json")):
|
|
tid = os.path.basename(os.path.dirname(f))
|
|
if not json.load(open(f)).get("accepted"):
|
|
continue
|
|
k = mix.kind_of_task_dir(tid)
|
|
if tid not in done0:
|
|
waiting[k] = waiting.get(k, 0) + 1
|
|
elif tid not in done1 and not first.get(tid):
|
|
second[k] = second.get(k, 0) + 1
|
|
print("\nwaiting for a first run, per kind:", dict(sorted(waiting.items())))
|
|
print("second attempt candidates (first attempt failed), per kind:", dict(sorted(second.items())))
|
|
print("\nstart: python3 -m harness.pipeline (one controller only; it starts the generators, the K variants and the "
|
|
"trajectory workers; only kinds below target run; second attempts for failed tasks)")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|