F: eval slots for INTF/TABL/STRU/MSAG/exception (+K), Kral spot-check sheet, step 0 in the restart plan; D: own-test mutation scoring (running)

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-06 06:06:41 +02:00
parent 7f03849a86
commit 0c0e07fe96
4 changed files with 331 additions and 0 deletions

View File

@@ -16,6 +16,7 @@ import sys
from .adt_client import load_env
from .generator import ROOT, generate, make_k_variant
from . import mix, overlap
from .ledger import BudgetExceeded, spent
FIRST_ID = 100
@@ -52,6 +53,73 @@ K_SOURCES = [("G0004", "free_text"), ("G0007", "incomplete"), ("G0026", "free_te
RELEASES = ["v702", "v740sp05"]
# New kinds for the eval set (Opus item F, 2026-10-06): the 130 eval candidates have only CLAS, FUNC, DDLS and PROG, so INTF, TABL, STRU, MSAG and
# exception tasks cannot be measured after training. Shares as in the training mix (110 tasks: INTF 8, TABL 9, STRU 2, MSAG 3, exception 3); about
# 30 percent more slots than the target because of rejections and the empirical filter. Generation after the reset, FIRST (before training tasks).
NEW_FIRST_ID = 200
NEW_RUN_BASE = 440000 # 40 per slot, below 466560
NEW_KIND_SLOTS = [("INTF", "A", 10), ("TABL", "B", 11), ("STRU", "B", 3), ("MSAG", "D", 4), ("EXC", "D", 4)]
NEW_K = [("INTF", "free_text"), ("TABL", "incomplete"), ("TABL", "free_text"), ("MSAG", "free_text"), ("EXC", "incomplete")]
EXC_TOPICS = ["exception class (CX_...): a domain exception with context attributes and message texts",
"exception class (CX_...): an exception hierarchy with a common super class",
"exception class (CX_...): an exception that wraps a previous exception",
"exception class (CX_...): an exception with a message class and parameters in the text"]
def plan_new():
out, n = [], 0
for kind, cat, count in NEW_KIND_SLOTS:
for i in range(count):
out.append({"id": f"G{NEW_FIRST_ID + n:04d}", "kind": kind, "category": cat, "object_type": "CLAS" if kind == "EXC" else kind,
"topic": EXC_TOPICS[i % len(EXC_TOPICS)] if kind == "EXC" else None,
"difficulty": 3 if i % 3 == 2 else 2, "run_base": NEW_RUN_BASE + 40 * n})
n += 1
return out
def run_new(only, model, base_url):
"""Generate the new-kind eval candidates. The bundle is also checked against the training pool (no near duplicate of a training task)."""
train = overlap.load_pool("train")
for s in plan_new():
if only and s["id"] not in only:
continue
if os.path.exists(os.path.join(ROOT, "tasks_gen", "eval", s["id"], "generation.json")):
continue
avoid = [g for g in accepted_goals() if g][-170:]
topic = ((s["topic"] + ". ") if s["topic"] else "Choose a new, realistic business topic. ") + "Do not repeat these existing topics: " + "; ".join(avoid)
def extra(b, _t=train):
return [f"Too close to task {i} (spec {sc['spec']:.2f}, rules {sc['core']:.2f}, names {sc['name']:.2f}). Choose another topic and other object names."
for i, sc in overlap.check(overlap.load_bundle(b), _t)[:3]]
try:
log = generate(s["id"], "eval", s["object_type"], s["category"], s["difficulty"], model, base_url, s["run_base"], topic, extra_check=extra)
except BudgetExceeded as e:
print("BUDGET", e, flush=True)
break
except Exception as e: # noqa: BLE001
log = {"id": s["id"], "error": str(e)[:500]}
log.update(kind=s["kind"], spent_total=spent())
print(json.dumps(log), flush=True)
def run_new_k(model, base_url):
"""K variants (free text or incomplete spec, EPOD tool names) of the first accepted eval task of each new kind."""
plan = plan_new()
first_k = NEW_FIRST_ID + len(plan)
for j, (kind, style) in enumerate(NEW_K):
src = next((s["id"] for s in plan if s["kind"] == kind and os.path.exists(os.path.join(ROOT, "tasks_gen", "eval", s["id"], "generation.json"))
and json.load(open(os.path.join(ROOT, "tasks_gen", "eval", s["id"], "generation.json"))).get("accepted")), None)
new_id = f"G{first_k + j:04d}"
if not src or os.path.exists(os.path.join(ROOT, "tasks_gen", "eval", new_id, "generation.json")):
continue
try:
log = make_k_variant(src, new_id, style, model, base_url, NEW_RUN_BASE + 40 * (len(plan) + j), tool_schema=None, pool="eval")
except BudgetExceeded as e:
print("BUDGET", e, flush=True)
break
print(json.dumps(log), flush=True)
def accepted_goals():
"""Goal lines of all generated tasks (eval and train), so the model does not repeat a topic."""
out = []
@@ -90,6 +158,18 @@ def plan():
def main():
load_env(os.path.join(ROOT, ".env"))
cmd = sys.argv[1] if len(sys.argv) > 1 else "plan"
model, base_url = "deepseek-v4.1-flash:cloud", os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")
if cmd == "plan-new":
for x in plan_new():
print(x)
print(len(plan_new()), "slots,", len(NEW_K), "K variants after them")
return
if cmd == "run-new":
run_new(set(sys.argv[2:]), model, base_url)
return
if cmd == "run-new-k":
run_new_k(model, base_url)
return
slots = plan()
if cmd == "plan":
for s in slots:

200
harness/owntests.py Normal file
View File

@@ -0,0 +1,200 @@
"""Own-test mutation score of an accepted trajectory (Opus item D, 2026-10-06): metadata only, the acceptance filter does not change.
The model's own unit tests (a testclasses include of the contract class, or global test classes) are run against the faulty references
of the task (`faulty/`: mutants of the reference that the hidden tests kill). Per trajectory: one run with the correct reference (the tests must
pass there, otherwise they encode model specific behavior and a kill proves nothing), then one run per mutant. Status per mutant:
killed (an own test fails), survived (all own tests pass), invalid (the mutant or the tests do not activate, or no own test ran).
score = killed / (killed + survived) over the valid mutants. PROG tasks are not supported (the tests live inside the program source).
python3 -m harness.owntests [--tasks G1000 ...] [--limit N] [--workers 1] writes runs/traj/<run>/own_test_mutation.json
"""
import argparse
import glob
import json
import os
import re
import shutil
import sys
import time
from .adt_client import load_env
from .agents import OracleAgent
from . import mix
from .runner import Runner
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, os.path.join(ROOT, "train"))
import accept as acc # noqa: E402
POOL = os.path.join(ROOT, "tasks_gen", "train")
WORK = os.path.join(ROOT, "runs", "owntests")
RUN_BASE = 480000 # + sequence number; below 466560 is NOT needed here: the run numbers stay under 466560 with 4-char base36
RUN_BASE = 420000
MAX_MUTANTS = 5
def calls_of(messages):
res = {m.get("tool_call_id"): m["content"] for m in messages if m["role"] == "tool"}
out = []
for m in messages:
if m["role"] != "assistant":
continue
for c in m.get("tool_calls") or []:
a = c["function"].get("arguments") or "{}"
try:
a = json.loads(a) if isinstance(a, str) else a
except ValueError:
a = {}
out.append((c["function"]["name"], a, res.get(c.get("id"), "")))
return out
def model_tests(rec, task_meta):
"""{'include': {OBJECT: source}, 'global': {NAME: source}} of the model's last successful pushes."""
contract = {c["name"].upper() for c in task_meta.get("contract", [])}
last = {}
for tool, a, res in calls_of(rec["messages"]):
if tool == "sap_push_source" and a.get("source") and '"success":true' in (res or "").replace(" ", ""):
last[(str(a.get("objectName", "")).upper(), str(a.get("includeType") or "").lower(), a.get("objectType"))] = a["source"]
pre = rec["prefix"].upper()
seed_hidden = {o["name"].replace("{{P}}", pre).upper() for k in ("seed", "hidden_tests") for o in task_meta.get(k, [])}
out = {"include": {}, "global": {}}
for (name, inc, otype), src in last.items():
contract_names = {c.replace("{{P}}", pre).upper() for c in contract}
if inc == "testclasses" and name in contract_names:
out["include"][name] = src
elif otype == "CLAS" and not inc and name not in contract_names and name not in seed_hidden \
and re.search(r"FOR\s+TESTING", src, re.I) and re.search(r"^\s*CLASS\s+\S+\s+DEFINITION[^.]*FOR\s+TESTING", src, re.I | re.M):
out["global"][name] = src
return out
def placeholder(src, prefix):
return re.sub(re.escape(prefix), "{{P}}", re.sub(re.escape(prefix.lower()), "{{p}}", src), flags=re.I) if False else \
src.replace(prefix.upper(), "{{P}}").replace(prefix.lower(), "{{p}}")
def mutant_files(task_id):
"""[(reference object name, path)] of faulty/ (a K variant has none: the base task's)."""
meta = json.load(open(os.path.join(POOL, task_id, "task.json")))
d = os.path.join(POOL, meta.get("base_task") or task_id, "faulty")
return sorted(glob.glob(os.path.join(d, "m*_*")))
def derive(task_id, run_label, tests, prefix, mutant_path=None):
"""Build a temporary task: the reference (optionally with one mutated object) plus the model's own tests only."""
src_dir = os.path.join(POOL, task_id)
pool = os.path.join(WORK, "pool")
dst = os.path.join(pool, task_id)
shutil.rmtree(dst, ignore_errors=True)
shutil.copytree(src_dir, dst, ignore=shutil.ignore_patterns("faulty", "generation.json", "review.json", "mutation.json", "empirical.json"))
meta = json.load(open(os.path.join(dst, "task.json")))
pre = prefix.upper()
mutated = None
if mutant_path:
base = re.sub(r"^m\d+_", "", os.path.basename(mutant_path))
refs = []
for o in meta["reference"]:
f = o.get("file")
is_test = o["type"] == "CLAS" and f and re.search(r"^\s*CLASS\s+\S+\s+DEFINITION[^.]*FOR\s+TESTING",
open(os.path.join(dst, f)).read(), re.I | re.M)
if is_test:
continue # the reference's own global test class: not the model's
o = dict(o)
o.pop("testclasses_file", None) # the reference's local tests are out; the model's go in
if mutant_path and f and os.path.basename(f) == base:
shutil.copy(mutant_path, os.path.join(dst, f))
mutated = o["name"]
name_up = o["name"].replace("{{P}}", pre).upper()
if name_up in tests["include"]:
rel = "reference/_own_include_%s.abap" % re.sub(r"\W", "_", name_up)
open(os.path.join(dst, rel), "w").write(placeholder(tests["include"][name_up], prefix))
o["testclasses_file"] = rel
refs.append(o)
for i, (name, src) in enumerate(tests["global"].items()):
rel = "reference/_own_global_%d.clas.abap" % i
open(os.path.join(dst, rel), "w").write(placeholder(src, prefix))
refs.append({"type": "CLAS", "name": placeholder(name, prefix), "file": rel, "description": "own test class"})
meta["reference"] = refs
meta["budget"] = dict(meta.get("budget", {}), max_tool_calls=200)
json.dump(meta, open(os.path.join(dst, "task.json"), "w"), indent=1)
return pool, mutated
def run_one(pool, task_id, run_no):
runner = Runner(pool, os.path.join(WORK, "runs"))
rep, _ = runner.run(task_id, OracleAgent(), run_no, teardown=True)
own = rep.get("own_tests") or {}
g = rep.get("gates") or {}
return {"tests": own.get("tests", 0), "failures": own.get("failures", 0), "active": bool(g.get("G1_active")), "run": run_no}
def score_run(run_dir, seq):
rec = json.load(open(os.path.join(run_dir, "record.json")))
tid = rec["task"]["id"]
meta = json.load(open(os.path.join(POOL, tid, "task.json")))
out = {"task": tid, "run": rec["run"], "kind": mix.kind_of_task_dir(tid), "time": time.strftime("%F %T")}
if any(c.get("type") == "PROG" for c in meta.get("contract", [])):
return dict(out, status="not_supported", reason="PROG: the tests are inside the program source")
tests = model_tests(rec, meta)
if not tests["include"] and not tests["global"]:
return dict(out, status="no_own_tests", score=None, mutants=[])
muts = mutant_files(tid)[:MAX_MUTANTS]
if not muts:
return dict(out, status="no_mutants", score=None, mutants=[])
pre = rec["prefix"]
pool, _ = derive(tid, "base", tests, pre)
base = run_one(pool, tid, RUN_BASE + seq * 10)
out["reference_run"] = base
out["tests_pass_on_reference"] = base["active"] and base["tests"] > 0 and base["failures"] == 0
res = []
for k, mp in enumerate(muts):
pool, mutated = derive(tid, "m%d" % k, tests, pre, mp)
r = run_one(pool, tid, RUN_BASE + seq * 10 + 1 + k)
status = "invalid" if (not r["active"] or r["tests"] == 0) else ("killed" if r["failures"] > 0 else "survived")
res.append({"mutant": os.path.basename(mp), "object": mutated, "status": status, "tests": r["tests"], "failures": r["failures"]})
valid = [x for x in res if x["status"] != "invalid"]
killed = [x for x in res if x["status"] == "killed"]
out.update(status="scored", mutants=res, valid=len(valid), killed=len(killed),
score=round(len(killed) / len(valid), 2) if valid else None)
shutil.rmtree(os.path.join(WORK, "pool", tid), ignore_errors=True)
return out
def main():
load_env(os.path.join(ROOT, ".env"))
ap = argparse.ArgumentParser()
ap.add_argument("--tasks", nargs="*")
ap.add_argument("--limit", type=int)
ap.add_argument("--redo", action="store_true")
a = ap.parse_args()
os.makedirs(WORK, exist_ok=True)
rows = [json.loads(l) for l in open(os.path.join(ROOT, "runs", "traj", "summary.jsonl"))]
todo = []
for r in rows:
p = os.path.join(ROOT, "runs", "traj", r.get("run_dir") or "-")
if not os.path.exists(os.path.join(p, "record.json")):
continue
if a.tasks and r["task"] not in a.tasks:
continue
if os.path.exists(os.path.join(p, "own_test_mutation.json")) and not a.redo:
continue
rec = json.load(open(os.path.join(p, "record.json")))
if acc.judge(rec, r, 80)[0]:
todo.append((r, p))
if a.limit:
todo = todo[:a.limit]
print(len(todo), "accepted trajectories to score", flush=True)
for i, (r, p) in enumerate(todo):
t0 = time.time()
try:
res = score_run(p, i)
except Exception as e: # noqa: BLE001
res = {"task": r["task"], "status": "error", "error": repr(e)[:300]}
json.dump(res, open(os.path.join(p, "own_test_mutation.json"), "w"), indent=1)
print(r["task"], res.get("status"), res.get("score"), "valid", res.get("valid"), "killed", res.get("killed"),
"ref_ok", res.get("tests_pass_on_reference"), "%.0fs" % (time.time() - t0), flush=True)
if __name__ == "__main__":
main()