Mutation in generator; category H (stop) scoring and judge; category K variants and tool schema variant; evalset plan

- mutation.py: negation mutants, skip WHILE/DO blocks (endless loop blocked RFC ~8 min)
- generator: accept only after mutation check; survivors go back as repair feedback
- runner/judge.py: stop tasks scored 100/30/0; gap by keywords, else judge model
- proxy: tool schema variant generic_v0 (draft); generator make_k_variant
- evalset.py: 88 slots for step F (A-I, H, K)

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-03 05:37:20 +02:00
parent 32f03fcad4
commit 0ba25faeea
20 changed files with 1303 additions and 22 deletions

View File

@@ -15,12 +15,14 @@ import urllib.request
from .adt_client import load_env
from .agents import NullAgent, OracleAgent
from .ledger import add_usage, check_budget
from .mutation import check_task
from .runner import Runner
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
ABAPLINT = os.path.join(ROOT, "node_modules", ".bin", "abaplint")
# pilot: accepted final calls used 4k-76k output tokens (p90 57k); one runaway call used 393k
MAX_OUT_TOKENS = 80000
RUNS_PER_ATTEMPT = 10
ABAPLINT_VERSIONS = {"v702", "v740sp05", "v750", "v758"}
EXAMPLE_FOR = {"CLAS": "T01", "INTF": "T01", "FUNC": "T13", "PROG": "T14", "DDLS": "T15", "TABL": "T15"}
CATEGORIES = {
@@ -32,6 +34,13 @@ CATEGORIES = {
"F": "use an unknown object: the spec names a seed object but not its signature; the model must read it",
"G": "old release target (v702 or v740sp05): no newer syntax",
"I": "bug fix: the seed contains a faulty object; the spec describes the symptom",
"H": "stop task: the spec looks complete, but one business rule is missing or two rules contradict, and "
"a correct solution is not possible without an answer. The model must stop, create no object and "
"name the gap. task.json: expected_outcome \"stop\", expected_gap {\"description\": \"<the gap in one "
"sentence>\", \"keywords\": [[\"word\", \"word\"], ...]} (2 to 4 groups; a group matches when all its "
"words occur in a report that names the gap). Keep the contract, seed and a reference of the "
"intended objects (the reference shows the task is otherwise clear). No hidden tests: "
"hidden_tests is []. Do not write the gap in Open questions (write \"None.\")",
}
SYSTEM = """You write evaluation tasks for an ABAP developer model. Each task is a bundle of files.
@@ -247,9 +256,14 @@ def abaplint_files(b):
def check_bundle(b):
errs = []
t, files = b.get("task", {}), b.get("files", {})
stop = t.get("expected_outcome") == "stop"
for k in ("seed", "contract", "hidden_tests", "reference"):
if k not in t:
errs.append(f"task.json misses '{k}'")
if stop:
gap = t.get("expected_gap") or {}
if not gap.get("description") or not gap.get("keywords"):
errs.append("stop task: expected_gap needs description and keywords")
if "spec.md" not in files:
errs.append("spec.md missing")
for k in ("seed", "hidden_tests", "reference"):
@@ -262,7 +276,7 @@ def check_bundle(b):
limit = 16 if o.get("type") == "TABL" else 26 if o.get("type") == "FUGR" else 30
if len(o.get("name", "").replace("{{P}}", "Z0000000_")) > limit:
errs.append(f"{k}: name {o.get('name')} too long (max {limit})")
if not t.get("hidden_tests"):
if not t.get("hidden_tests") and not stop:
errs.append("no hidden test class")
return errs
@@ -377,12 +391,37 @@ def generate(task_id, pool, object_type, category, difficulty, model, base_url,
b["task"].setdefault("object_type", object_type)
b["task"].setdefault("category", category)
write_bundle(b, task_dir, task_id)
rep_o, rep_n = validate(pool_root, task_id, run_base + 2 * attempt)
# run numbers per attempt: +0 oracle, +1 null, +2..+6 mutants (RUNS_PER_ATTEMPT)
base = run_base + RUNS_PER_ATTEMPT * attempt
rep_o, rep_n = validate(pool_root, task_id, base)
so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total")
log["attempts"].append({"stage": "validate", "oracle": so, "null": sn})
if category == "H": # oracle stops with the gap: 100; null stops without a report: 30
if so == 100 and sn == 30:
log["accepted"] = True
break
messages.append({"role": "user", "content":
f"Stop task check: required oracle 100 and null 30. Result: oracle {so}, null {sn}. "
f"Oracle report: {failure_summary(rep_o)}\nFix expected_gap (description, keywords) "
"and return the full bundle again."})
continue
if so == 100 and sn == 0:
log["accepted"] = True
break
mut = check_task(pool_root, task_id, base + 2, keep=True)
log["attempts"][-1]["mutation"] = {k: mut[k] for k in ("valid", "killed", "ok")}
if mut["ok"]:
log["accepted"] = True
break
survived = [f"{m['object']} {m['mutant']}" for m in mut["mutants"] if m["status"] == "survived"]
shutil.rmtree(os.path.join(task_dir, "faulty"), ignore_errors=True)
messages.append({"role": "user", "content":
"Oracle 100 and null 0: good. But the hidden tests are too weak. The harness changed "
"the reference (mutation check) and all hidden tests still passed for these changes:\n"
+ "\n".join(survived or ["(fewer than 2 changes possible: add more business logic "
"checks to the hidden tests)"])
+ "\nAdd or improve hidden tests so that each of these changes makes a test fail. "
"If a change does not change the behavior, ignore it. Do not change the behavior "
"of the reference. Return the full bundle again."})
continue
messages.append({"role": "user", "content":
"The harness ran your reference solution (oracle) and an empty solution (null). "
f"Required: oracle 100, null 0. Result: oracle {so}, null {sn}.\n"
@@ -396,19 +435,85 @@ def generate(task_id, pool, object_type, category, difficulty, model, base_url,
return log
K_STYLES = {
"free_text": "Rewrite the spec as a short free-text request from a functional consultant: no section "
"headings, no numbered rules, plain sentences, as in an e-mail. Keep every business rule.",
"incomplete": "Rewrite the spec as a short free-text request. Leave out the craft hints and the context "
"that a good ABAP developer can find in the system (for example how the seed table looks). "
"Keep every business rule; the task must stay solvable without questions.",
}
def contract_terms(task):
"""Names that a K spec must keep: object names, FM parameters, report parameters, CDS elements."""
out = []
for c in task.get("contract", []):
out.append(c["name"])
out += [p["name"] if isinstance(p, dict) else p for p in c.get("params", []) + c.get("parameters", [])]
out += list(c.get("fields", []))
return out
def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema="generic_v0", max_repairs=2):
"""Category K: free-text or incomplete input + other tool schema. Same reference and hidden tests
as the source task; only spec.md, category and tool_schema change."""
check_budget()
pool_root = os.path.join(ROOT, "tasks_gen", "eval")
src_dir, task_dir = os.path.join(pool_root, src_id), os.path.join(pool_root, new_id)
b = bundle_of(src_dir)
b["files"] = {k: v for k, v in b["files"].items()
if not (k.startswith("faulty/") or k in ("mutation.json",))}
spec = b["files"]["spec.md"]
terms = contract_terms(b["task"])
messages = [{"role": "user", "content":
f"{K_STYLES[style]}\nKeep these names exactly as written (the tests use them): "
f"{', '.join(terms)}. Keep the placeholder {{{{P}}}} in names. Use Simplified Technical English. "
f"Return only the new spec text.\n\nSpec:\n{spec}"}]
log = {"id": new_id, "pool": "eval", "category": "K", "base_task": src_id, "style": style,
"tool_schema": tool_schema, "attempts": []}
for attempt in range(max_repairs + 1):
text = chat(model, messages, base_url).strip()
text = re.sub(r"^```\w*\s*|\s*```$", "", text)
messages.append({"role": "assistant", "content": text})
missing = [t for t in terms if t.upper() not in text.upper()]
if missing:
log["attempts"].append({"stage": "spec", "missing_names": missing})
messages.append({"role": "user", "content": "These names are missing: " + ", ".join(missing)
+ ". Return the full spec again with all names."})
continue
b["files"]["spec.md"] = text + "\n"
b["task"].update(category="K", base_task=src_id, input_style=style, tool_schema=tool_schema)
write_bundle(b, task_dir, new_id)
rep_o, rep_n = validate(pool_root, new_id, run_base + RUNS_PER_ATTEMPT * attempt)
so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total")
log["attempts"].append({"stage": "validate", "oracle": so, "null": sn})
log["accepted"] = so == 100 and sn == 0
break
else:
log["accepted"] = False
json.dump(log, open(os.path.join(task_dir if os.path.isdir(task_dir) else pool_root, "generation.json"), "w"),
indent=1)
return log
def main():
load_env(os.path.join(ROOT, ".env"))
ap = argparse.ArgumentParser()
ap.add_argument("--id", required=True)
ap.add_argument("--pool", default="eval", choices=["eval", "train"])
ap.add_argument("--object-type", required=True, choices=sorted(EXAMPLE_FOR))
ap.add_argument("--category", required=True, choices=sorted(CATEGORIES))
ap.add_argument("--object-type", choices=sorted(EXAMPLE_FOR))
ap.add_argument("--category", choices=sorted(CATEGORIES))
ap.add_argument("--difficulty", type=int, default=2)
ap.add_argument("--topic")
ap.add_argument("--model", default="deepseek-v4.1-flash:cloud")
ap.add_argument("--base-url", default=os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1"))
ap.add_argument("--run-base", type=int, required=True)
ap.add_argument("--k-from", help="category K: make a variant of this accepted eval task")
ap.add_argument("--k-style", default="free_text", choices=sorted(K_STYLES))
a = ap.parse_args()
if a.k_from:
print(json.dumps(make_k_variant(a.k_from, a.id, a.k_style, a.model, a.base_url, a.run_base)))
return
os.makedirs(os.path.join(ROOT, "tasks_gen", a.pool), exist_ok=True)
log = generate(a.id, a.pool, a.object_type, a.category, a.difficulty, a.model, a.base_url,
a.run_base, a.topic)