Mutation in generator; category H (stop) scoring and judge; category K variants and tool schema variant; evalset plan
- mutation.py: negation mutants, skip WHILE/DO blocks (endless loop blocked RFC ~8 min) - generator: accept only after mutation check; survivors go back as repair feedback - runner/judge.py: stop tasks scored 100/30/0; gap by keywords, else judge model - proxy: tool schema variant generic_v0 (draft); generator make_k_variant - evalset.py: 88 slots for step F (A-I, H, K) Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -15,12 +15,14 @@ import urllib.request
|
||||
from .adt_client import load_env
|
||||
from .agents import NullAgent, OracleAgent
|
||||
from .ledger import add_usage, check_budget
|
||||
from .mutation import check_task
|
||||
from .runner import Runner
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
ABAPLINT = os.path.join(ROOT, "node_modules", ".bin", "abaplint")
|
||||
# pilot: accepted final calls used 4k-76k output tokens (p90 57k); one runaway call used 393k
|
||||
MAX_OUT_TOKENS = 80000
|
||||
RUNS_PER_ATTEMPT = 10
|
||||
ABAPLINT_VERSIONS = {"v702", "v740sp05", "v750", "v758"}
|
||||
EXAMPLE_FOR = {"CLAS": "T01", "INTF": "T01", "FUNC": "T13", "PROG": "T14", "DDLS": "T15", "TABL": "T15"}
|
||||
CATEGORIES = {
|
||||
@@ -32,6 +34,13 @@ CATEGORIES = {
|
||||
"F": "use an unknown object: the spec names a seed object but not its signature; the model must read it",
|
||||
"G": "old release target (v702 or v740sp05): no newer syntax",
|
||||
"I": "bug fix: the seed contains a faulty object; the spec describes the symptom",
|
||||
"H": "stop task: the spec looks complete, but one business rule is missing or two rules contradict, and "
|
||||
"a correct solution is not possible without an answer. The model must stop, create no object and "
|
||||
"name the gap. task.json: expected_outcome \"stop\", expected_gap {\"description\": \"<the gap in one "
|
||||
"sentence>\", \"keywords\": [[\"word\", \"word\"], ...]} (2 to 4 groups; a group matches when all its "
|
||||
"words occur in a report that names the gap). Keep the contract, seed and a reference of the "
|
||||
"intended objects (the reference shows the task is otherwise clear). No hidden tests: "
|
||||
"hidden_tests is []. Do not write the gap in Open questions (write \"None.\")",
|
||||
}
|
||||
|
||||
SYSTEM = """You write evaluation tasks for an ABAP developer model. Each task is a bundle of files.
|
||||
@@ -247,9 +256,14 @@ def abaplint_files(b):
|
||||
def check_bundle(b):
|
||||
errs = []
|
||||
t, files = b.get("task", {}), b.get("files", {})
|
||||
stop = t.get("expected_outcome") == "stop"
|
||||
for k in ("seed", "contract", "hidden_tests", "reference"):
|
||||
if k not in t:
|
||||
errs.append(f"task.json misses '{k}'")
|
||||
if stop:
|
||||
gap = t.get("expected_gap") or {}
|
||||
if not gap.get("description") or not gap.get("keywords"):
|
||||
errs.append("stop task: expected_gap needs description and keywords")
|
||||
if "spec.md" not in files:
|
||||
errs.append("spec.md missing")
|
||||
for k in ("seed", "hidden_tests", "reference"):
|
||||
@@ -262,7 +276,7 @@ def check_bundle(b):
|
||||
limit = 16 if o.get("type") == "TABL" else 26 if o.get("type") == "FUGR" else 30
|
||||
if len(o.get("name", "").replace("{{P}}", "Z0000000_")) > limit:
|
||||
errs.append(f"{k}: name {o.get('name')} too long (max {limit})")
|
||||
if not t.get("hidden_tests"):
|
||||
if not t.get("hidden_tests") and not stop:
|
||||
errs.append("no hidden test class")
|
||||
return errs
|
||||
|
||||
@@ -377,12 +391,37 @@ def generate(task_id, pool, object_type, category, difficulty, model, base_url,
|
||||
b["task"].setdefault("object_type", object_type)
|
||||
b["task"].setdefault("category", category)
|
||||
write_bundle(b, task_dir, task_id)
|
||||
rep_o, rep_n = validate(pool_root, task_id, run_base + 2 * attempt)
|
||||
# run numbers per attempt: +0 oracle, +1 null, +2..+6 mutants (RUNS_PER_ATTEMPT)
|
||||
base = run_base + RUNS_PER_ATTEMPT * attempt
|
||||
rep_o, rep_n = validate(pool_root, task_id, base)
|
||||
so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total")
|
||||
log["attempts"].append({"stage": "validate", "oracle": so, "null": sn})
|
||||
if category == "H": # oracle stops with the gap: 100; null stops without a report: 30
|
||||
if so == 100 and sn == 30:
|
||||
log["accepted"] = True
|
||||
break
|
||||
messages.append({"role": "user", "content":
|
||||
f"Stop task check: required oracle 100 and null 30. Result: oracle {so}, null {sn}. "
|
||||
f"Oracle report: {failure_summary(rep_o)}\nFix expected_gap (description, keywords) "
|
||||
"and return the full bundle again."})
|
||||
continue
|
||||
if so == 100 and sn == 0:
|
||||
log["accepted"] = True
|
||||
break
|
||||
mut = check_task(pool_root, task_id, base + 2, keep=True)
|
||||
log["attempts"][-1]["mutation"] = {k: mut[k] for k in ("valid", "killed", "ok")}
|
||||
if mut["ok"]:
|
||||
log["accepted"] = True
|
||||
break
|
||||
survived = [f"{m['object']} {m['mutant']}" for m in mut["mutants"] if m["status"] == "survived"]
|
||||
shutil.rmtree(os.path.join(task_dir, "faulty"), ignore_errors=True)
|
||||
messages.append({"role": "user", "content":
|
||||
"Oracle 100 and null 0: good. But the hidden tests are too weak. The harness changed "
|
||||
"the reference (mutation check) and all hidden tests still passed for these changes:\n"
|
||||
+ "\n".join(survived or ["(fewer than 2 changes possible: add more business logic "
|
||||
"checks to the hidden tests)"])
|
||||
+ "\nAdd or improve hidden tests so that each of these changes makes a test fail. "
|
||||
"If a change does not change the behavior, ignore it. Do not change the behavior "
|
||||
"of the reference. Return the full bundle again."})
|
||||
continue
|
||||
messages.append({"role": "user", "content":
|
||||
"The harness ran your reference solution (oracle) and an empty solution (null). "
|
||||
f"Required: oracle 100, null 0. Result: oracle {so}, null {sn}.\n"
|
||||
@@ -396,19 +435,85 @@ def generate(task_id, pool, object_type, category, difficulty, model, base_url,
|
||||
return log
|
||||
|
||||
|
||||
K_STYLES = {
|
||||
"free_text": "Rewrite the spec as a short free-text request from a functional consultant: no section "
|
||||
"headings, no numbered rules, plain sentences, as in an e-mail. Keep every business rule.",
|
||||
"incomplete": "Rewrite the spec as a short free-text request. Leave out the craft hints and the context "
|
||||
"that a good ABAP developer can find in the system (for example how the seed table looks). "
|
||||
"Keep every business rule; the task must stay solvable without questions.",
|
||||
}
|
||||
|
||||
|
||||
def contract_terms(task):
|
||||
"""Names that a K spec must keep: object names, FM parameters, report parameters, CDS elements."""
|
||||
out = []
|
||||
for c in task.get("contract", []):
|
||||
out.append(c["name"])
|
||||
out += [p["name"] if isinstance(p, dict) else p for p in c.get("params", []) + c.get("parameters", [])]
|
||||
out += list(c.get("fields", []))
|
||||
return out
|
||||
|
||||
|
||||
def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema="generic_v0", max_repairs=2):
|
||||
"""Category K: free-text or incomplete input + other tool schema. Same reference and hidden tests
|
||||
as the source task; only spec.md, category and tool_schema change."""
|
||||
check_budget()
|
||||
pool_root = os.path.join(ROOT, "tasks_gen", "eval")
|
||||
src_dir, task_dir = os.path.join(pool_root, src_id), os.path.join(pool_root, new_id)
|
||||
b = bundle_of(src_dir)
|
||||
b["files"] = {k: v for k, v in b["files"].items()
|
||||
if not (k.startswith("faulty/") or k in ("mutation.json",))}
|
||||
spec = b["files"]["spec.md"]
|
||||
terms = contract_terms(b["task"])
|
||||
messages = [{"role": "user", "content":
|
||||
f"{K_STYLES[style]}\nKeep these names exactly as written (the tests use them): "
|
||||
f"{', '.join(terms)}. Keep the placeholder {{{{P}}}} in names. Use Simplified Technical English. "
|
||||
f"Return only the new spec text.\n\nSpec:\n{spec}"}]
|
||||
log = {"id": new_id, "pool": "eval", "category": "K", "base_task": src_id, "style": style,
|
||||
"tool_schema": tool_schema, "attempts": []}
|
||||
for attempt in range(max_repairs + 1):
|
||||
text = chat(model, messages, base_url).strip()
|
||||
text = re.sub(r"^```\w*\s*|\s*```$", "", text)
|
||||
messages.append({"role": "assistant", "content": text})
|
||||
missing = [t for t in terms if t.upper() not in text.upper()]
|
||||
if missing:
|
||||
log["attempts"].append({"stage": "spec", "missing_names": missing})
|
||||
messages.append({"role": "user", "content": "These names are missing: " + ", ".join(missing)
|
||||
+ ". Return the full spec again with all names."})
|
||||
continue
|
||||
b["files"]["spec.md"] = text + "\n"
|
||||
b["task"].update(category="K", base_task=src_id, input_style=style, tool_schema=tool_schema)
|
||||
write_bundle(b, task_dir, new_id)
|
||||
rep_o, rep_n = validate(pool_root, new_id, run_base + RUNS_PER_ATTEMPT * attempt)
|
||||
so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total")
|
||||
log["attempts"].append({"stage": "validate", "oracle": so, "null": sn})
|
||||
log["accepted"] = so == 100 and sn == 0
|
||||
break
|
||||
else:
|
||||
log["accepted"] = False
|
||||
json.dump(log, open(os.path.join(task_dir if os.path.isdir(task_dir) else pool_root, "generation.json"), "w"),
|
||||
indent=1)
|
||||
return log
|
||||
|
||||
|
||||
def main():
|
||||
load_env(os.path.join(ROOT, ".env"))
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--id", required=True)
|
||||
ap.add_argument("--pool", default="eval", choices=["eval", "train"])
|
||||
ap.add_argument("--object-type", required=True, choices=sorted(EXAMPLE_FOR))
|
||||
ap.add_argument("--category", required=True, choices=sorted(CATEGORIES))
|
||||
ap.add_argument("--object-type", choices=sorted(EXAMPLE_FOR))
|
||||
ap.add_argument("--category", choices=sorted(CATEGORIES))
|
||||
ap.add_argument("--difficulty", type=int, default=2)
|
||||
ap.add_argument("--topic")
|
||||
ap.add_argument("--model", default="deepseek-v4.1-flash:cloud")
|
||||
ap.add_argument("--base-url", default=os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1"))
|
||||
ap.add_argument("--run-base", type=int, required=True)
|
||||
ap.add_argument("--k-from", help="category K: make a variant of this accepted eval task")
|
||||
ap.add_argument("--k-style", default="free_text", choices=sorted(K_STYLES))
|
||||
a = ap.parse_args()
|
||||
if a.k_from:
|
||||
print(json.dumps(make_k_variant(a.k_from, a.id, a.k_style, a.model, a.base_url, a.run_base)))
|
||||
return
|
||||
os.makedirs(os.path.join(ROOT, "tasks_gen", a.pool), exist_ok=True)
|
||||
log = generate(a.id, a.pool, a.object_type, a.category, a.difficulty, a.model, a.base_url,
|
||||
a.run_base, a.topic)
|
||||
|
||||
Reference in New Issue
Block a user