F: eval slots for INTF/TABL/STRU/MSAG/exception (+K), Kral spot-check sheet, step 0 in the restart plan; D: own-test mutation scoring (running)
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -16,6 +16,7 @@ import sys
|
||||
|
||||
from .adt_client import load_env
|
||||
from .generator import ROOT, generate, make_k_variant
|
||||
from . import mix, overlap
|
||||
from .ledger import BudgetExceeded, spent
|
||||
|
||||
FIRST_ID = 100
|
||||
@@ -52,6 +53,73 @@ K_SOURCES = [("G0004", "free_text"), ("G0007", "incomplete"), ("G0026", "free_te
|
||||
RELEASES = ["v702", "v740sp05"]
|
||||
|
||||
|
||||
# New kinds for the eval set (Opus item F, 2026-10-06): the 130 eval candidates have only CLAS, FUNC, DDLS and PROG, so INTF, TABL, STRU, MSAG and
|
||||
# exception tasks cannot be measured after training. Shares as in the training mix (110 tasks: INTF 8, TABL 9, STRU 2, MSAG 3, exception 3); about
|
||||
# 30 percent more slots than the target because of rejections and the empirical filter. Generation after the reset, FIRST (before training tasks).
|
||||
NEW_FIRST_ID = 200
|
||||
NEW_RUN_BASE = 440000 # 40 per slot, below 466560
|
||||
NEW_KIND_SLOTS = [("INTF", "A", 10), ("TABL", "B", 11), ("STRU", "B", 3), ("MSAG", "D", 4), ("EXC", "D", 4)]
|
||||
NEW_K = [("INTF", "free_text"), ("TABL", "incomplete"), ("TABL", "free_text"), ("MSAG", "free_text"), ("EXC", "incomplete")]
|
||||
EXC_TOPICS = ["exception class (CX_...): a domain exception with context attributes and message texts",
|
||||
"exception class (CX_...): an exception hierarchy with a common super class",
|
||||
"exception class (CX_...): an exception that wraps a previous exception",
|
||||
"exception class (CX_...): an exception with a message class and parameters in the text"]
|
||||
|
||||
|
||||
def plan_new():
|
||||
out, n = [], 0
|
||||
for kind, cat, count in NEW_KIND_SLOTS:
|
||||
for i in range(count):
|
||||
out.append({"id": f"G{NEW_FIRST_ID + n:04d}", "kind": kind, "category": cat, "object_type": "CLAS" if kind == "EXC" else kind,
|
||||
"topic": EXC_TOPICS[i % len(EXC_TOPICS)] if kind == "EXC" else None,
|
||||
"difficulty": 3 if i % 3 == 2 else 2, "run_base": NEW_RUN_BASE + 40 * n})
|
||||
n += 1
|
||||
return out
|
||||
|
||||
|
||||
def run_new(only, model, base_url):
|
||||
"""Generate the new-kind eval candidates. The bundle is also checked against the training pool (no near duplicate of a training task)."""
|
||||
train = overlap.load_pool("train")
|
||||
for s in plan_new():
|
||||
if only and s["id"] not in only:
|
||||
continue
|
||||
if os.path.exists(os.path.join(ROOT, "tasks_gen", "eval", s["id"], "generation.json")):
|
||||
continue
|
||||
avoid = [g for g in accepted_goals() if g][-170:]
|
||||
topic = ((s["topic"] + ". ") if s["topic"] else "Choose a new, realistic business topic. ") + "Do not repeat these existing topics: " + "; ".join(avoid)
|
||||
|
||||
def extra(b, _t=train):
|
||||
return [f"Too close to task {i} (spec {sc['spec']:.2f}, rules {sc['core']:.2f}, names {sc['name']:.2f}). Choose another topic and other object names."
|
||||
for i, sc in overlap.check(overlap.load_bundle(b), _t)[:3]]
|
||||
try:
|
||||
log = generate(s["id"], "eval", s["object_type"], s["category"], s["difficulty"], model, base_url, s["run_base"], topic, extra_check=extra)
|
||||
except BudgetExceeded as e:
|
||||
print("BUDGET", e, flush=True)
|
||||
break
|
||||
except Exception as e: # noqa: BLE001
|
||||
log = {"id": s["id"], "error": str(e)[:500]}
|
||||
log.update(kind=s["kind"], spent_total=spent())
|
||||
print(json.dumps(log), flush=True)
|
||||
|
||||
|
||||
def run_new_k(model, base_url):
|
||||
"""K variants (free text or incomplete spec, EPOD tool names) of the first accepted eval task of each new kind."""
|
||||
plan = plan_new()
|
||||
first_k = NEW_FIRST_ID + len(plan)
|
||||
for j, (kind, style) in enumerate(NEW_K):
|
||||
src = next((s["id"] for s in plan if s["kind"] == kind and os.path.exists(os.path.join(ROOT, "tasks_gen", "eval", s["id"], "generation.json"))
|
||||
and json.load(open(os.path.join(ROOT, "tasks_gen", "eval", s["id"], "generation.json"))).get("accepted")), None)
|
||||
new_id = f"G{first_k + j:04d}"
|
||||
if not src or os.path.exists(os.path.join(ROOT, "tasks_gen", "eval", new_id, "generation.json")):
|
||||
continue
|
||||
try:
|
||||
log = make_k_variant(src, new_id, style, model, base_url, NEW_RUN_BASE + 40 * (len(plan) + j), tool_schema=None, pool="eval")
|
||||
except BudgetExceeded as e:
|
||||
print("BUDGET", e, flush=True)
|
||||
break
|
||||
print(json.dumps(log), flush=True)
|
||||
|
||||
|
||||
def accepted_goals():
|
||||
"""Goal lines of all generated tasks (eval and train), so the model does not repeat a topic."""
|
||||
out = []
|
||||
@@ -90,6 +158,18 @@ def plan():
|
||||
def main():
|
||||
load_env(os.path.join(ROOT, ".env"))
|
||||
cmd = sys.argv[1] if len(sys.argv) > 1 else "plan"
|
||||
model, base_url = "deepseek-v4.1-flash:cloud", os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")
|
||||
if cmd == "plan-new":
|
||||
for x in plan_new():
|
||||
print(x)
|
||||
print(len(plan_new()), "slots,", len(NEW_K), "K variants after them")
|
||||
return
|
||||
if cmd == "run-new":
|
||||
run_new(set(sys.argv[2:]), model, base_url)
|
||||
return
|
||||
if cmd == "run-new-k":
|
||||
run_new_k(model, base_url)
|
||||
return
|
||||
slots = plan()
|
||||
if cmd == "plan":
|
||||
for s in slots:
|
||||
|
||||
Reference in New Issue
Block a user