' % (cls, E(label), value, sub)
def agg_table(title, evs, rows, key, note=""):
agg = {}
for r, e in zip(rows, evs):
k = task_meta(r["task"])[key]
a = agg.setdefault(k, [0, 0, 0])
a[0] += 1
a[1] += e["accepted"]
a[2] += e["accepted"] and e["repair"]
trs = ""
for k, (n, ok, rep) in sorted(agg.items()):
rate = ok / n if n else 0
cls = "g" if rate >= 0.8 else ("w" if rate >= 0.5 else "r")
trs += "
%s
%d
%d
%d
%s
%.0f%%
" % (
E(str(k)), n, ok, rep, bar(ok, n, cls), 100 * rate)
return ("
%s
%s
runs
accepted
"
"
repair
acceptance
%%
%s
%s
"
% (E(title), E(key.replace("_", " ")), trs, "
%s
" % E(note) if note else ""))
def local_card():
"""Card for series A (local Qwen on the MacBook): state.json of harness.localqwen. Empty string without a series."""
path = os.path.join(ROOT, "runs", "local_qwen", "state.json")
if not os.path.exists(path):
return ""
try:
st = json.load(open(path))
except ValueError:
return ""
now = time.time()
url = st.get("base_url") or ""
up = ping(url + "/models", timeout=4) if url else False # the dashboard asks the MacBook itself
status = st.get("status", "?")
left = st.get("window_start", now) + st.get("window_hours", 12) * 3600 - now
cur = st.get("current")
plan, res = st.get("plan", []), st.get("results", [])
chips = "server %s%s" % (
"ok" if up else "er", "answers" if up else "does not answer",
{"running": "ok", "paused": "wa", "finished": "gr", "window_over": "gr"}.get(status, "gr"), E(status))
rows = ""
per = {}
for r in res:
k = per.setdefault(r["kind"], {"runs": 0, "ok": 0, "scores": [], "fail": {}})
k["runs"] += 1
k["ok"] += r["accepted"]
if r.get("score") is not None:
k["scores"].append(r["score"])
for f in r.get("failure_types", []):
k["fail"][f] = k["fail"].get(f, 0) + 1
planned = {}
for p_ in plan:
planned[p_["kind"]] = planned.get(p_["kind"], 0) + 1
for kind in [x for x in ("INTF", "TABL", "STRU", "MSAG", "EXC", "DDLS") if x in planned or x in per]:
k = per.get(kind, {"runs": 0, "ok": 0, "scores": [], "fail": {}})
mean = "%.0f" % (sum(k["scores"]) / len(k["scores"])) if k["scores"] else "-"
rate = "%.0f%%" % (100.0 * k["ok"] / k["runs"]) if k["runs"] else "-"
fails = ", ".join("%s %d" % (f, n) for f, n in sorted(k["fail"].items(), key=lambda x: -x[1])) or "-"
rows += ("
" % trs)
def gen_table(logs, key, title):
agg = {}
for l in logs:
if key == "object_type":
k = l.get("object_type") or "?"
else:
k = l.get("category") or "?"
a = agg.setdefault(k, [0, 0])
a[0] += 1
a[1] += bool(l.get("accepted"))
trs = "".join("
%s
%d
%d
%s
%.0f%%
"
% (E(str(k)), n, ok, bar(ok, n, "g" if ok / n >= 0.8 else "w"), 100.0 * ok / n) for k, (n, ok) in sorted(agg.items()))
return ("
%s
%s
slots
accepted
rate
%%
%s
"
% (E(title), E(key.replace("_", " ")), trs))
def render(d):
rows, evs, logs = d["rows"], d["evs"], d["logs"]
acc = [e for e in evs if e["accepted"]]
n_acc, n_runs = len(acc), len(rows)
repairs = sum(e["repair"] for e in acc)
tasks_acc = sum(1 for l in logs if l.get("accepted"))
k_n = sum(1 for l in logs if l.get("category") == "K")
err_n = sum(1 for l in logs if l.get("error_kind"))
hard_n = sum(1 for l in logs if (l.get("difficulty") or 2) == 3)
tasks_with_run = {r["task"] for r in rows}
backlog = sum(1 for l in logs if l.get("accepted") and l["id"] not in tasks_with_run)
alive = any("harness.pipeline" in p for p in d["procs"])
if alive: # a STOPPED.txt of an earlier run does not count while a controller runs
d["stopped"] = None
status, scls = ("RUNNING", "ok") if alive else (("STOPPED", "er") if d["stopped"] else ("NOT RUNNING", "er"))
hints = sum(e["hints"] for e in evs)
ctx = sorted(e["ctx"] for e in acc if e["ctx"])
next50 = (n_acc // 50 + 1) * 50
est = d["panel_est"]
pcls = "good" if est < 50 else ("warn" if est < 57 else "bad")
eta = ("%.1f h (%s)" % (d["eta_h"], time.strftime("%a %H:%M", time.localtime(d["now"] + d["eta_h"] * 3600)))) if d["eta_h"] and alive else "n/a"
if not alive: # a stopped pipeline has no burn rate and no speed
d["burn_ledger_h"] = d["rate_runs_h"] = d["rate_acc_h"] = None
a30 = d["acc30"]
a30s = "%.0f %%" % (100 * a30) if a30 is not None else "n/a (<30 runs)"
cost_per = d["run_cost"] / n_acc if n_acc else 0
banner = ""
if d["stopped"]:
banner = "
STOP flag set: generators finish their current slot
"
tiles = "".join([
tile("Accepted trajectories", "%d" % n_acc, "next summary at %d (%d to go)" % (next50, next50 - n_acc) + bar(n_acc % 50, 50), "good"),
tile("Acceptance (all runs)", "%.0f %%" % (100.0 * n_acc / n_runs if n_runs else 0), "%d of %d runs; last 30: %s (stop below 50 %%)" % (n_acc, n_runs, a30s),
"good" if not n_runs or n_acc / n_runs >= 0.7 else "warn"),
tile("Repair share", "%.0f %%" % (100.0 * repairs / n_acc if n_acc else 0), "%d of %d accepted contain error + fix" % (repairs, n_acc)),
tile("Training tasks", "%d" % tasks_acc, "%d slots tried, %d K, %d error-targeted, %d hard; %d wait for a first run" % (len(logs), k_n, err_n, hard_n, backlog)),
tile("Ollama usage (estimated)", "$%.1f" % est, "of $60; ~$%.1f left; last panel $%.2f" % (d["panel_left"], d["panel"][-1][0]), pcls),
tile("Ledger / guard", "%.1f" % d["ledger"], "guard %.1f (limit %.0f - reserve %.0f); %.1f left" % (d["guard"], d["limit"], d["reserve"], d["guard"] - d["ledger"]),
"good" if d["guard"] - d["ledger"] > 10 else "warn"),
tile("Time to guard", eta, "burn %s ledger/h" % ("%.1f" % d["burn_ledger_h"] if d["burn_ledger_h"] else "n/a"), "warn" if d["eta_h"] and d["eta_h"] < 3 else ""),
tile("Speed", "%s runs/h" % ("%.1f" % d["rate_runs_h"] if d["rate_runs_h"] is not None else "n/a"),
"%s accepted/h; %d trajectory workers" % ("%.1f" % d["rate_acc_h"] if d["rate_acc_h"] is not None else "n/a", d["workers"])),
tile("Syntax hints", "%d" % hints, "proxy syntaxCheck added (local abaplint)"),
tile("Cost per accepted", "%.2f" % cost_per, "ledger USD (trajectory runs only); generation %.1f, runs %.1f" % (d["gen_cost"], d["run_cost"])),
])
# budget calibration
pr = d["panel"]
prows = ""
for i, (u, l, lab) in enumerate(pr):
r = ""
if i and u > pr[i - 1][0]:
r = "%.2f" % ((l - pr[i - 1][1]) / (u - pr[i - 1][0]))
prows += "
%s
$%.2f
%.1f
%s
" % (E(lab), u, l, r)
budget = ("
Budget and calibration
panel at
usage
ledger
"
"
ledger per usage
%s
The ledger is a list-price upper bound; real usage per ledger "
"USD varies with the work (generation about 2, trajectory runs about 1.2-1.4). Guard ratio 1.2. Add a value: "
"python3 -m harness.dashboard panel 41.07
"
"
%s
") % (prows, spark([(h["t"], h["ledger"]) for h in d["hist"]], guard=d["guard"], color="#E9730C"))
prog = ("
Progress over time
accepted trajectories
%s"
"
trajectory runs
%s
accepted training tasks
%s
"
% (spark([(h["t"], h["acc"]) for h in d["hist"]], fmt="%d", color="#107E3E"),
spark([(h["t"], h["runs"]) for h in d["hist"]], fmt="%d"),
spark([(h["t"], h["tasks"]) for h in d["hist"]], fmt="%d", color="#5899DA")))
# recent runs
rec = ""
for r, e in list(zip(rows, evs))[-14:][::-1]:
m = task_meta(r["task"])
st = "accepted" if e["accepted"] else "%s" % E((", ".join(e["reasons"]) or "rejected")[:40])
rec += ("