Eval review v1: checklist, review helper, 24 tasks reviewed; G0002/G0022 tests fixed; I/E generator definitions

- docs/eval-inceleme.md checklist; harness/review.py
- 19 accept, 1 fix pending (G0019), 4 flagged (I/E tasks solvable without legacy code)
- generator: categories I and E fix/refactor the seed object in place

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat
This commit is contained in:
Kral
2026-10-03 06:23:38 +02:00
parent a12fa4dff6
commit 3550f0b651
38 changed files with 607 additions and 40 deletions

View File

@@ -30,10 +30,17 @@ CATEGORIES = {
"B": "database access (ABAP SQL, CDS) with test doubles",
"C": "internal table craft (table kind, keys, access path)",
"D": "exception design (class-based exceptions, messages)",
"E": "refactoring legacy code (seed contains legacy code; behavior must stay the same)",
"E": "refactoring legacy code: the seed contains legacy code, and the contract object IS this seed object "
"(refactor it in place; same name and same public interface; it is not in out_of_scope). The spec "
"gives the quality goals and says 'the behavior stays the same'. It does not list the business "
"rules: the model must read the legacy code. Hidden tests check the unchanged behavior",
"F": "use an unknown object: the spec names a seed object but not its signature; the model must read it",
"G": "old release target (v702 or v740sp05): no newer syntax",
"I": "bug fix: the seed contains a faulty object; the spec describes the symptom",
"I": "bug fix: the seed contains a faulty object, and the contract object IS this seed object (fix it in "
"place; it is not in out_of_scope). The spec describes the symptoms and the expected behavior for "
"them, plus 'all other behavior stays the same'. The spec does not list all rules: the model must "
"read the existing code to keep the other behavior. Hidden tests check the fixed cases and the "
"unchanged behavior",
"H": "stop task: the spec looks complete, but one business rule is missing or two rules contradict, and "
"a correct solution is not possible without an answer. The model must stop, create no object and "
"name the gap. task.json: expected_outcome \"stop\", expected_gap {\"description\": \"<the gap in one "

88
harness/review.py Normal file
View File

@@ -0,0 +1,88 @@
"""Review helper for eval tasks (docs/eval-inceleme.md).
python3 -m harness.review G0100 # print one task in a compact form
python3 -m harness.review --list # accepted tasks and their review state
python3 -m harness.review G0100 --set accept|fix|flag|reject --note "..." [--checks 2=eksik_test,8=sizinti]
"""
import argparse
import glob
import json
import os
import re
import time
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
POOL = os.path.join(ROOT, "tasks_gen", "eval")
def _load(tid, name, default=None):
try:
return json.load(open(os.path.join(POOL, tid, name)))
except (OSError, ValueError):
return default
def show(tid):
d = os.path.join(POOL, tid)
t = _load(tid, "task.json", {})
g = _load(tid, "generation.json", {})
m = _load(tid, "mutation.json")
print(f"== {tid} category {t.get('category')} type {t.get('object_type')} release {t.get('release_target')}"
f" outcome {t.get('expected_outcome', 'implement')}")
if t.get("base_task"):
print(f"K variant of {t['base_task']} ({t.get('input_style')}, tool schema {t.get('tool_schema')})")
if t.get("expected_gap"):
print("expected_gap:", json.dumps(t["expected_gap"]))
print("\n-- spec.md")
print(open(os.path.join(d, "spec.md")).read())
print("-- contract:", json.dumps(t.get("contract")))
print("-- seed:", [(o["type"], o["name"]) for o in t.get("seed", [])])
print("-- reference:", [(o["type"], o["name"]) for o in t.get("reference", [])])
for h in t.get("hidden_tests", []):
src = open(os.path.join(d, h["file"])).read()
names = re.findall(r"^\s*METHODS\s+(\w+)\s+FOR\s+TESTING", src, re.I | re.M)
print(f"-- hidden tests {h['name']} ({len(names)}):", ", ".join(names))
print("-- generation:", [(a.get("stage"), a.get("oracle"), a.get("null"), a.get("mutation")) for a in
g.get("attempts", [])], "accepted:", g.get("accepted"))
if m:
print(f"-- mutation: {m['killed']}/{m['valid']} ok={m['ok']}",
[x["mutant"] for x in m["mutants"] if x["status"] == "survived"])
r = _load(tid, "review.json")
if r:
print("-- review:", json.dumps(r))
def accepted():
out = []
for d in sorted(glob.glob(os.path.join(POOL, "G*"))):
tid = os.path.basename(d)
if (_load(tid, "generation.json") or {}).get("accepted"):
out.append(tid)
return out
def main():
ap = argparse.ArgumentParser()
ap.add_argument("task", nargs="?")
ap.add_argument("--list", action="store_true")
ap.add_argument("--set", choices=["accept", "fix", "flag", "reject"])
ap.add_argument("--note", default="")
ap.add_argument("--checks", default="", help="failed checks, for example 2=eksik_test,8=sizinti")
a = ap.parse_args()
if a.list:
for tid in accepted():
t = _load(tid, "task.json", {})
r = _load(tid, "review.json") or {}
print(tid, t.get("category"), t.get("object_type"), r.get("decision", "-"), r.get("note", "")[:80])
return
if a.set:
rev = {"decision": a.set, "note": a.note, "by": "Claude", "date": time.strftime("%Y-%m-%d"),
"failed_checks": dict(x.split("=", 1) for x in a.checks.split(",") if "=" in x)}
json.dump(rev, open(os.path.join(POOL, a.task, "review.json"), "w"), indent=1)
print(json.dumps(rev))
return
show(a.task)
if __name__ == "__main__":
main()