Object type mix: INTF, TABL, STRU, MSAG, exception tasks (harness G2, mutants, generator notes), balanced generator, kind-deficit job order, dashboard mix card
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -25,7 +25,7 @@ import threading
|
||||
import time
|
||||
|
||||
from .adt_client import load_env
|
||||
from . import trainset, trajectories
|
||||
from . import mix, trainset, trajectories
|
||||
from .ledger import spent
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
@@ -129,10 +129,21 @@ class Pipeline:
|
||||
done = {(r["task"], r["attempt"]) for r in rows}
|
||||
ev0 = {r["task"]: self.evaluate(r) for r in rows if r["attempt"] == 0}
|
||||
tasks = trajectories.accepted_tasks()
|
||||
for t in tasks: # first attempt for every task first
|
||||
if (t, 0) not in done and (t, 0) not in self.inflight:
|
||||
self.inflight.add((t, 0))
|
||||
return t, 0
|
||||
cands = [t for t in tasks if (t, 0) not in done and (t, 0) not in self.inflight]
|
||||
if cands: # first attempts: the kind with the biggest deficit against the target mix goes first
|
||||
counts = {}
|
||||
for r in rows:
|
||||
if self.evaluate(r)["accepted"]:
|
||||
k = mix.kind_of_task_dir(r["task"])
|
||||
counts[k] = counts.get(k, 0) + 1
|
||||
for (t, _a) in self.inflight:
|
||||
k = mix.kind_of_task_dir(t)
|
||||
counts[k] = counts.get(k, 0) + 1
|
||||
total, tot_share = sum(counts.values()) + 1, sum(mix.TYPE_SHARE.values())
|
||||
best = max(cands, key=lambda t: (total * mix.TYPE_SHARE.get(mix.kind_of_task_dir(t), 0) / tot_share
|
||||
- counts.get(mix.kind_of_task_dir(t), 0), -int(t[1:])))
|
||||
self.inflight.add((best, 0))
|
||||
return best, 0
|
||||
for t in tasks: # second attempt: first failed, or accepted without a repair
|
||||
e = ev0.get(t)
|
||||
if e and (t, 1) not in done and (t, 1) not in self.inflight and (not e["accepted"] or not e["repair"]):
|
||||
@@ -275,6 +286,25 @@ class Pipeline:
|
||||
a[1] += e["accepted"]
|
||||
a[2] += e["accepted"] and e["repair"]
|
||||
return "; ".join(f"{k} {v[1]}/{v[0]} (repair {v[2]})" for k, v in sorted(agg.items()))
|
||||
kinds_t, kinds_r = mix.accepted_task_counts(), {}
|
||||
for r, e in zip(rows, evs):
|
||||
k = mix.kind_of_task_dir(r["task"])
|
||||
a = kinds_r.setdefault(k, [0, 0])
|
||||
a[0] += 1
|
||||
a[1] += e["accepted"]
|
||||
tt, tr_ = max(sum(kinds_t.values()), 1), max(sum(v[1] for v in kinds_r.values()), 1)
|
||||
share_sum = sum(mix.TYPE_SHARE.values())
|
||||
kind_line = "; ".join("%s tasks %d (%.0f%%) traj %d/%d (%.0f%%) target %.0f%%" % (
|
||||
k, kinds_t.get(k, 0), 100.0 * kinds_t.get(k, 0) / tt, kinds_r.get(k, [0, 0])[1], kinds_r.get(k, [0, 0])[0],
|
||||
100.0 * kinds_r.get(k, [0, 0])[1] / tr_, 100.0 * v / share_sum) for k, v in mix.TYPE_SHARE.items())
|
||||
ddls = {}
|
||||
for r, e in zip(rows, evs):
|
||||
if mix.kind_of_task_dir(r["task"]) == "DDLS" and not e["accepted"]:
|
||||
for w in (e["reasons"] or ["?"]):
|
||||
w = w.split("=")[0] if w.startswith(("end_reason", "score")) else w
|
||||
ddls[w] = ddls.get(w, 0) + 1
|
||||
ddls_n = sum(1 for r in rows if mix.kind_of_task_dir(r["task"]) == "DDLS")
|
||||
ddls_acc = sum(1 for r, e in zip(rows, evs) if mix.kind_of_task_dir(r["task"]) == "DDLS" and e["accepted"])
|
||||
s = spent()
|
||||
used = (s - BASE_LEDGER) / LEDGER_TO_USAGE
|
||||
from .ledger import _env_budget
|
||||
@@ -290,6 +320,8 @@ trajectory runs {len(rows)}, accepted trajectories {len(accepted)} (acceptance {
|
||||
- By category, accepted/runs: {table('category')}.
|
||||
- By object type, accepted/runs: {table('object_type')}.
|
||||
- Tokens of accepted samples (20 tool schemas kept): p50 {pct(0.5)}, p90 {pct(0.9)}, p95 {pct(0.95)}, max {toks[-1] if toks else None}, n {len(toks)}.
|
||||
- Object type mix (accepted tasks; accepted/runs trajectories; target): {kind_line}.
|
||||
- DDLS (CDS) runs: {ddls_acc} accepted of {ddls_n}; reject reasons {ddls or 'none'} (activation, hidden tests and ATC rejections are separate gate failures, they show as score reasons).
|
||||
- Syntax hints (proxy syntaxCheck added): {sum(e['hints'] for e in evs)} in {sum(1 for e in evs if e['hints'])} runs.
|
||||
- Harness events: {sum(self.events.values())} ({self.events or 'none'}); trajectory workers now {self.max_workers}.
|
||||
"""
|
||||
|
||||
Reference in New Issue
Block a user