Files
abap-llm/harness/mutation.py
Kral 0ba25faeea Mutation in generator; category H (stop) scoring and judge; category K variants and tool schema variant; evalset plan
- mutation.py: negation mutants, skip WHILE/DO blocks (endless loop blocked RFC ~8 min)
- generator: accept only after mutation check; survivors go back as repair feedback
- runner/judge.py: stop tasks scored 100/30/0; gap by keywords, else judge model
- proxy: tool schema variant generic_v0 (draft); generator make_k_variant
- evalset.py: 88 slots for step F (A-I, H, K)

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-03 05:37:20 +02:00

211 lines
10 KiB
Python

"""Mutation check (step D): the hidden tests must fail on a broken reference.
Small deterministic changes (mutants) go into the main source of the contract objects of the reference.
For each mutant the oracle writes the mutated reference and the harness runs the hidden tests.
killed = reference active and at least one hidden test fails
survived = reference active and all hidden tests pass (a test gap or an equivalent mutant)
invalid = the mutant does not activate (not counted)
python3 -m harness.mutation G0002 G0003 --pool eval --run-base 5000
"""
import argparse
import json
import os
import random
import re
import shutil
from .adt_client import load_env
from .agents import OracleAgent
from .runner import Runner
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
MAX_MUTANTS = 5
MIN_VALID = 2 # small references (T14, T15) have only a few mutation sites
MIN_KILL_RATE = 0.75
# (kind, pattern, replacement); patterns match only outside comments and literals
# Relational operators are negated, not shifted: a shifted boundary (< to <=) is often equivalent in
# clamp code (T13: "IF fee < 1. fee = 1."), so it says nothing about the tests.
ABAP_OPS = [
("rel", r"(?<=\s)>=(?=\s)", "<"), ("rel", r"(?<=\s)<=(?=\s)", ">"),
("rel", r"(?<=\s)>(?=\s)", "<="), ("rel", r"(?<=\s)<(?=\s)", ">="),
("rel", r"(?<=\s)<>(?=\s)", "="), ("eq", r"(?<=\s)=(?=\s)", "<>"),
("rel", r"\bGE\b", "LT"), ("rel", r"\bLE\b", "GT"), ("rel", r"\bGT\b", "LE"), ("rel", r"\bLT\b", "GE"),
("rel", r"\bNE\b", "EQ"), ("rel", r"\bEQ\b", "NE"),
("logic", r"(?<=\s)AND(?=\s)", "OR"), ("logic", r"(?<=\s)OR(?=\s)", "AND"),
("arith", r"(?<=\s)\+(?=\s)", "-"), ("arith", r"(?<=\s)-(?=\s)", "+"), ("arith", r"(?<=\s)\*(?=\s)", "/"),
("bool", r"\babap_true\b", "abap_false"), ("bool", r"\babap_false\b", "abap_true"),
("const", r"(?<![\w.'-])([1-9]\d{0,5})(?![\w.'])", None), # integer literal + 1
]
CDS_OPS = [
("rel", r"(?<=\s)>=(?=\s)", "<"), ("rel", r"(?<=\s)<=(?=\s)", ">"),
("rel", r"(?<=\s)>(?=\s)", "<="), ("rel", r"(?<=\s)<(?=\s)", ">="), ("rel", r"(?<=\s)<>(?=\s)", "="),
("eq", r"(?<=\s)=(?=\s)", "<>"),
("logic", r"(?<=\s)and(?=\s)", "or"),
("arith", r"(?<=\s)\+(?=\s)", "-"), ("arith", r"(?<=\s)-(?=\s)", "+"),
("agg", r"\bsum\s*\(", "max("), ("agg", r"\bavg\s*\(", "max("),
("join", r"\binner\s+join\b", "left outer join"),
("const", r"(?<![\w.'-])([1-9]\d{0,5})(?![\w.'])", None),
]
# "=" is an assignment in ABAP; it is a comparison only in these lines
ABAP_CONDITION = re.compile(r"\s*(IF|ELSEIF|CHECK|WHILE|WHERE|AND|OR|ON)\b", re.I)
def _mask(src, cds):
"""True for each character inside a comment or a literal."""
m = [False] * len(src)
pats = ([r"//[^\n]*", r"/\*.*?\*/", r"'[^']*'", r"@[^\n]*"] if cds else
[r'"[^\n]*', r"^\*[^\n]*", r"'(?:[^']|'')*'", r"`[^`]*`", r"\|(?:[^|\\]|\\.)*\|"])
for p in pats:
for x in re.finditer(p, src, re.M | re.S):
for i in range(x.start(), x.end()):
m[i] = True
return m
def _region(src, otype):
"""Start and end of the part that may change: the implementation, not declarations or test classes."""
start, end = 0, len(src)
if otype == "CLAS":
x = re.search(r"^\s*CLASS\s+\S+\s+IMPLEMENTATION", src, re.I | re.M)
start = x.end() if x else len(src)
if otype == "PROG":
x = re.search(r"^\s*CLASS\s+\S+\s+DEFINITION[^.]*FOR\s+TESTING", src, re.I | re.M)
end = x.start() if x else len(src)
if otype == "DDLS":
x = re.search(r"\bas\s+select\b|\bas\s+projection\b", src, re.I)
start = x.start() if x else 0
return start, end
def _loop_blocks(src):
"""Ranges of WHILE ... ENDWHILE and DO ... ENDDO. A mutant there can make an endless loop; the hidden
test then blocks the one RFC connection of the server (G0002: 'remaining - step' to '+')."""
out, stack = [], []
for x in re.finditer(r"^\s*(WHILE|DO|ENDWHILE|ENDDO)\b", src, re.I | re.M):
if x.group(1).upper() in ("WHILE", "DO"):
stack.append(x.start())
elif stack:
out.append((stack.pop(), x.end()))
return out
def _skip_line(line):
"""Declarations and signatures: a change there gives syntax errors or no behavior change."""
return re.match(r"\s*(DATA|TYPES|CONSTANTS|METHODS|CLASS-METHODS|PARAMETERS|SELECT-OPTIONS|"
r"IMPORTING|EXPORTING|RETURNING|RAISING|TABLES|FIELD-SYMBOLS)\b", line, re.I)
def mutants(src, otype, seed, n=MAX_MUTANTS):
"""Up to n mutants as (description, source). Different kinds first, spread over the source."""
cds = otype == "DDLS"
mask = _mask(src, cds)
lo, hi = _region(src, otype)
loops = [] if cds else _loop_blocks(src)
sites = []
for kind, pat, rep in (CDS_OPS if cds else ABAP_OPS):
for x in re.finditer(pat, src, re.I):
if not lo <= x.start() < hi or mask[x.start()] or any(a <= x.start() < b for a, b in loops):
continue
line_start = src.rfind("\n", 0, x.start()) + 1
line_end = src.find("\n", x.start())
line = src[line_start:line_end if line_end >= 0 else len(src)]
if kind == "eq" and not cds and not ABAP_CONDITION.match(line):
continue
if kind == "arith" and re.search(r"(\(|\bSELECT)\s*$", src[line_start:x.start()], re.I):
continue # COUNT( * ), SELECT *
if _skip_line(line) or (kind == "const" and re.search(r"\bLENGTH\b|\bDECIMALS\b|\(\s*\d+\s*,", line, re.I)):
continue
new = str(int(x.group(1)) + 1) if rep is None else rep
sites.append((kind, x.start(), x.end(), new, src.count("\n", 0, x.start()) + 1, x.group(0)))
rnd = random.Random(seed)
rnd.shuffle(sites)
chosen, kinds, lines = [], set(), set()
for prefer_new in (True, False):
for s in sites:
if len(chosen) >= n:
break
if s in chosen or (prefer_new and (s[4] in lines or s[0] in kinds)):
continue
chosen.append(s)
kinds.add(s[0])
lines.add(s[4])
out = []
for kind, a, b, new, line, old in sorted(chosen, key=lambda s: s[1]):
out.append((f"line {line}: {old.strip()} -> {new} ({kind})", src[:a] + new + src[b:]))
return out
def check_task(pool_root, task_id, run_base, n=MAX_MUTANTS, keep=False):
"""Run the mutants of one task. Returns a summary dict; writes it to <task>/mutation.json."""
task_dir = os.path.join(pool_root, task_id)
meta = json.load(open(os.path.join(task_dir, "task.json")))
contract = {c["name"].upper() for c in meta["contract"]}
targets = [o for o in meta["reference"] if o["name"].upper() in contract and o.get("file")
and o["type"] in ("CLAS", "FUNC", "PROG", "DDLS")]
work = os.path.join(ROOT, "runs", "gen", "mut")
os.makedirs(work, exist_ok=True)
runner = Runner(os.path.join(work, "pool"), os.path.join(work, "runs"))
# candidates per object, then round robin: objects without mutation sites (exception classes) give their share
cand = [[(o, d, m) for d, m in mutants(open(os.path.join(task_dir, o["file"])).read(), o["type"],
f"{task_id}:{o['name']}", n)] for o in targets]
plan = []
while len(plan) < n and any(cand):
for c in cand:
if c and len(plan) < n:
plan.append(c.pop(0))
results = []
for k, (o, desc, msrc) in enumerate(plan):
mdir = os.path.join(work, "pool", task_id)
shutil.rmtree(mdir, ignore_errors=True)
shutil.copytree(task_dir, mdir)
open(os.path.join(mdir, o["file"]), "w").write(msrc)
rep, _ = runner.run(task_id, OracleAgent(), run_base + k)
h = rep.get("hidden_tests") or {}
g = rep.get("gates") or {}
if not g.get("G1_active") or not h.get("total"):
status = "invalid"
elif h["passed"] < h["total"]:
status = "killed"
else:
status = "survived"
results.append({"object": o["name"], "mutant": desc, "status": status,
"hidden": f"{h.get('passed')}/{h.get('total')}",
"failed_tests": [d["method"] for d in h.get("detail", []) if not d["ok"]]})
if keep and status == "killed": # candidate faulty reference for own-test scoring
fdir = os.path.join(task_dir, "faulty")
os.makedirs(fdir, exist_ok=True)
open(os.path.join(fdir, f"m{k}_{os.path.basename(o['file'])}"), "w").write(msrc)
shutil.rmtree(os.path.join(work, "pool", task_id), ignore_errors=True)
valid = [r for r in results if r["status"] != "invalid"]
killed = [r for r in valid if r["status"] == "killed"]
summary = {"task": task_id, "mutants": results, "valid": len(valid), "killed": len(killed),
"kill_rate": round(len(killed) / len(valid), 2) if valid else None,
"ok": len(valid) >= MIN_VALID and len(killed) / max(1, len(valid)) >= MIN_KILL_RATE}
if len(valid) < MIN_VALID:
summary["note"] = f"fewer than {MIN_VALID} valid mutants; check by review"
json.dump(summary, open(os.path.join(task_dir, "mutation.json"), "w"), indent=1)
return summary
def main():
load_env(os.path.join(ROOT, ".env"))
ap = argparse.ArgumentParser()
ap.add_argument("tasks", nargs="+")
ap.add_argument("--pool", default="eval")
ap.add_argument("--run-base", type=int, required=True)
ap.add_argument("-n", type=int, default=MAX_MUTANTS)
ap.add_argument("--keep", action="store_true", help="store killed mutants in <task>/faulty/")
a = ap.parse_args()
root = os.path.join(ROOT, "tasks_gen", a.pool) if a.pool != "tasks" else os.path.join(ROOT, "tasks")
for i, t in enumerate(a.tasks):
s = check_task(root, t, a.run_base + 10 * i, a.n, a.keep)
print(json.dumps({k: v for k, v in s.items() if k != "mutants"}), flush=True)
for m in s["mutants"]:
print(" ", m["status"], m["object"], m["mutant"], m["hidden"], flush=True)
if __name__ == "__main__":
main()