- mutation.py: negation mutants, skip WHILE/DO blocks (endless loop blocked RFC ~8 min) - generator: accept only after mutation check; survivors go back as repair feedback - runner/judge.py: stop tasks scored 100/30/0; gap by keywords, else judge model - proxy: tool schema variant generic_v0 (draft); generator make_k_variant - evalset.py: 88 slots for step F (A-I, H, K) Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
525 lines
29 KiB
Python
525 lines
29 KiB
Python
"""Task generator: a cloud model writes a task bundle; the harness validates it (oracle = 100, null = 0).
|
|
|
|
python3 -m harness.generator --id G0001 --pool eval --object-type CLAS --category C --difficulty 2
|
|
"""
|
|
import argparse
|
|
import json
|
|
import os
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import tempfile
|
|
import time
|
|
import urllib.request
|
|
|
|
from .adt_client import load_env
|
|
from .agents import NullAgent, OracleAgent
|
|
from .ledger import add_usage, check_budget
|
|
from .mutation import check_task
|
|
from .runner import Runner
|
|
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
ABAPLINT = os.path.join(ROOT, "node_modules", ".bin", "abaplint")
|
|
# pilot: accepted final calls used 4k-76k output tokens (p90 57k); one runaway call used 393k
|
|
MAX_OUT_TOKENS = 80000
|
|
RUNS_PER_ATTEMPT = 10
|
|
ABAPLINT_VERSIONS = {"v702", "v740sp05", "v750", "v758"}
|
|
EXAMPLE_FOR = {"CLAS": "T01", "INTF": "T01", "FUNC": "T13", "PROG": "T14", "DDLS": "T15", "TABL": "T15"}
|
|
CATEGORIES = {
|
|
"A": "pure logic in a new class (language, OO design, Clean ABAP)",
|
|
"B": "database access (ABAP SQL, CDS) with test doubles",
|
|
"C": "internal table craft (table kind, keys, access path)",
|
|
"D": "exception design (class-based exceptions, messages)",
|
|
"E": "refactoring legacy code (seed contains legacy code; behavior must stay the same)",
|
|
"F": "use an unknown object: the spec names a seed object but not its signature; the model must read it",
|
|
"G": "old release target (v702 or v740sp05): no newer syntax",
|
|
"I": "bug fix: the seed contains a faulty object; the spec describes the symptom",
|
|
"H": "stop task: the spec looks complete, but one business rule is missing or two rules contradict, and "
|
|
"a correct solution is not possible without an answer. The model must stop, create no object and "
|
|
"name the gap. task.json: expected_outcome \"stop\", expected_gap {\"description\": \"<the gap in one "
|
|
"sentence>\", \"keywords\": [[\"word\", \"word\"], ...]} (2 to 4 groups; a group matches when all its "
|
|
"words occur in a report that names the gap). Keep the contract, seed and a reference of the "
|
|
"intended objects (the reference shows the task is otherwise clear). No hidden tests: "
|
|
"hidden_tests is []. Do not write the gap in Open questions (write \"None.\")",
|
|
}
|
|
|
|
SYSTEM = """You write evaluation tasks for an ABAP developer model. Each task is a bundle of files.
|
|
The model gets only spec.md and works on an SAP ABAP Platform 2025 system (SAP_BASIS 816, client 001)
|
|
through ADT tools. The harness installs the seed objects, runs the model, then checks the result with
|
|
hidden ABAP Unit tests, ATC, and abaplint.
|
|
|
|
Rules for the bundle:
|
|
- Use the placeholder {{P}} (upper case) and {{p}} (lower case) at the start of EVERY object name.
|
|
The harness replaces it with a run prefix of 9 characters (for example Z005P001_).
|
|
Name length after replacement: classes, programs, function modules, CDS entities max 30;
|
|
database tables max 16; function groups max 26.
|
|
- Every other ABAP name (methods, test methods, local classes, data, types, constants) has max 30
|
|
characters. Keep test method names short, for example "rejects_zero_qty".
|
|
- Method parameters need a complete type: "TYPE c LENGTH 4" is not allowed in a signature.
|
|
Declare a type first (TYPES ty_zone TYPE c LENGTH 4) and use it.
|
|
- Field names in tables and CDS views are not SQL reserved words (no HOURS, MODE, ORDER, DATE, COUNT ...).
|
|
- CDS: use "define view entity". Parameters have no default value. Use a parameter as
|
|
$parameters.p_name (not :p_name). In a UNION, all branches have the same
|
|
key elements and element names, and the view
|
|
has the annotation @Metadata.ignorePropagatedAnnotations: true. Hidden tests for a view with parameters pass all parameters.
|
|
- Legacy seed code (categories E, I) must activate on SAP_BASIS 816: do not mix old and new syntax in one
|
|
SQL statement (with new syntax, every host variable needs "@").
|
|
- The contract lists only the objects that the hidden tests call. Test classes are never in the contract.
|
|
Every reference class (except exception classes) has a "testclasses_file".
|
|
- A database table seed has type "TABL" and DDL source "define table ...". Type "DDLS" is only for CDS views.
|
|
- All objects are in package $TMP. Do not use transports.
|
|
- Do not use SAP application module data (no FI, SD, MM tables). Use generic business domains and
|
|
only objects that exist in every ABAP Platform system (language, ABAP SQL, CDS, CL_ABAP_*, CL_SALV_TABLE,
|
|
CL_OSQL_TEST_ENVIRONMENT, CL_CDS_TEST_ENVIRONMENT). Seed your own tables and data if needed.
|
|
- No dynpro (CALL SCREEN), no SmartForms, no BAdI, no RAP behavior definitions.
|
|
- spec.md uses Simplified Technical English and these sections in this order:
|
|
1. Goal, 2. Open questions (write "None."), 3. Context, 4. Contract, 5. Business rules,
|
|
6. Constraints (release target, coding standards, out of scope), 7. Acceptance.
|
|
The Contract fixes every public name the hidden tests use: object names, method signatures,
|
|
function module parameters, report parameters, ALV column names, CDS element names.
|
|
The Business rules are complete and unambiguous. Every rule is checked by at least one hidden test.
|
|
Do not tell the model HOW to implement (no table kinds, no SQL). Craft decisions belong to the model.
|
|
- Seed objects: TABL as DDL source ("define table ..."), FUGR without source, FUNC with "functionGroup"
|
|
and full source including the signature in the FUNCTION statement, classes with full source.
|
|
Seed table data: a seed class that implements IF_OO_ADT_CLASSRUN, with "run": true.
|
|
- Hidden tests: one global class, "FOR TESTING DURATION SHORT RISK LEVEL HARMLESS", 5 to 12 test
|
|
methods. Use only the public contract. Function modules: CALL FUNCTION with EXCEPTIONS.
|
|
Reports: SUBMIT ... AND RETURN with cl_salv_bs_runtime_info=>set( display = abap_false
|
|
metadata = abap_false data = abap_true ) and get_data_ref. CDS: cl_cds_test_environment.
|
|
- Reference solution: correct, Clean ABAP, methods below 40 statements, passes all hidden tests,
|
|
no ATC priority 1 or 2 findings (for example: pass large parameters by reference).
|
|
Include the model's expected own tests: for classes a "testclasses_file" (local test classes);
|
|
for reports local test classes inside the program; for function modules and CDS a global test class.
|
|
- task.json keys: id, category, object_type, difficulty, release_target, expected_outcome ("implement"),
|
|
budget {max_tool_calls, max_activations}, seed[], contract[], out_of_scope[], hidden_tests[],
|
|
reference[], craft_checks[]. Contract entries: CLAS {"implements"} optional; FUNC {"functionGroup",
|
|
"params":[{"name","type"}]}; PROG {"parameters":[...]}; DDLS {"fields":[...]}.
|
|
|
|
Return ONLY one JSON object, no markdown fence:
|
|
{"task": <task.json object>, "files": {"<relative path>": "<file content>", ...}}
|
|
"""
|
|
|
|
|
|
def bundle_of(task_dir):
|
|
files = {}
|
|
for base, _, names in os.walk(task_dir):
|
|
for n in names:
|
|
if n.startswith(".") or n == "generation.json":
|
|
continue
|
|
path = os.path.join(base, n)
|
|
rel = os.path.relpath(path, task_dir)
|
|
if rel != "task.json":
|
|
files[rel] = open(path).read()
|
|
return {"task": json.load(open(os.path.join(task_dir, "task.json"))), "files": files}
|
|
|
|
|
|
def chat(model, messages, base_url):
|
|
# max_tokens: one pilot call ran to 393k output tokens (reasoning) and returned no content
|
|
body = {"model": model, "messages": messages, "temperature": 0.7, "max_tokens": MAX_OUT_TOKENS}
|
|
req = urllib.request.Request(f"{base_url}/chat/completions", json.dumps(body).encode(),
|
|
{"Content-Type": "application/json", "Authorization": "Bearer none"})
|
|
for attempt in range(4):
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=1800) as r:
|
|
data = json.loads(r.read().decode())
|
|
add_usage(model, data.get("usage", {}), kind="generate")
|
|
content = data["choices"][0]["message"].get("content") or ""
|
|
if content.strip():
|
|
return content
|
|
# empty content (output limit reached in reasoning): retry once more with the next attempt
|
|
if attempt == 3:
|
|
return ""
|
|
continue
|
|
except Exception as e: # noqa: BLE001
|
|
if getattr(e, "code", 500) < 500 and getattr(e, "code", 500) != 429:
|
|
raise
|
|
time.sleep(15 * (attempt + 1))
|
|
raise RuntimeError("generation request failed")
|
|
|
|
|
|
def parse_bundle(text):
|
|
text = text.strip()
|
|
text = re.sub(r"^```(json)?\s*|\s*```$", "", text)
|
|
# raw_decode stops after the first complete object; text after it (a note, a second object) is ignored
|
|
return json.JSONDecoder().raw_decode(text[text.find("{"):])[0]
|
|
|
|
|
|
NAME_RE = re.compile(r"^\s*(?:CLASS-)?(?:METHODS|METHOD|DATA|TYPES|CONSTANTS|FORM|CLASS|INTERFACE)\b:?\s+(\w+)",
|
|
re.I | re.M)
|
|
# reserved words seen in activation errors, plus common SQL keywords
|
|
RESERVED = {"HOURS", "MODE", "ORDER", "GROUP", "DATE", "TIME", "VALUE", "USER", "KEY", "COUNT", "SUM", "MIN",
|
|
"MAX", "AVG", "SELECT", "FROM", "WHERE", "TABLE", "VIEW", "UNION", "JOIN", "LEVEL", "SIZE", "TYPE",
|
|
"INDEX", "CASE", "WHEN", "THEN", "ELSE", "END", "AS", "ON", "BY", "DESC", "ASC", "DAYS", "MINUTES",
|
|
"SECONDS", "YEAR", "MONTH", "DAY", "LIMIT", "OFFSET", "PARAMETERS"}
|
|
GLOBAL_TEST_CLASS = re.compile(r"\A\s*(?:\*[^\n]*\n|\"[^\n]*\n|\s)*CLASS\s+\S+\s+DEFINITION[^.]*FOR\s+TESTING",
|
|
re.I)
|
|
DDL_KIND = re.compile(r"^\s*define\s+(?:root\s+)?(table|view|structure)\b", re.I | re.M)
|
|
|
|
|
|
def lint_files(b):
|
|
"""Static checks that need no SAP system: identifier length, seed type against DDL source."""
|
|
errs = []
|
|
for rel, src in b.get("files", {}).items():
|
|
if not rel.endswith(".abap"):
|
|
continue
|
|
for name in set(NAME_RE.findall(src.replace("{{P}}", "Z0000000_").replace("{{p}}", "z0000000_"))):
|
|
if len(name) > 30:
|
|
errs.append(f"{rel}: name {name} has {len(name)} characters (ABAP max 30)")
|
|
t = b.get("task", {})
|
|
oos = {n.upper() for n in t.get("out_of_scope", [])}
|
|
test_names = {o.get("name", "").upper() for o in t.get("hidden_tests", [])}
|
|
for o in t.get("reference", []):
|
|
name = o.get("name", "")
|
|
if name.upper() in oos:
|
|
errs.append(f"reference object {name} is also in out_of_scope; the solution must change it")
|
|
if o.get("type") == "CLAS" and GLOBAL_TEST_CLASS.search(b.get("files", {}).get(o.get("file", ""), "")):
|
|
test_names.add(name.upper())
|
|
for c in t.get("contract", []):
|
|
if c.get("name", "").upper() in test_names:
|
|
errs.append(f"contract contains the test class {c.get('name')}; only objects that the hidden "
|
|
"tests call belong in the contract")
|
|
for o in t.get("reference", []):
|
|
if (o.get("type") == "CLAS" and o.get("name", "").upper() not in test_names
|
|
and not o.get("testclasses_file")
|
|
and not re.search(r"INHERITING\s+FROM\s+\S*CX_", b.get("files", {}).get(o.get("file", ""), ""),
|
|
re.I)):
|
|
errs.append(f"reference class {o.get('name')} has no testclasses_file (local unit tests)")
|
|
for rel, src in b.get("files", {}).items():
|
|
if DDL_KIND.search(src):
|
|
for f in re.findall(r"^\s*(?:key\s+)?(\w+)\s*:", src, re.I | re.M):
|
|
if f.upper() in RESERVED:
|
|
errs.append(f"{rel}: field name {f} is a reserved word; choose another name")
|
|
for rel, src in b.get("files", {}).items():
|
|
if rel.endswith(".asddls") and re.search(r"define\s+(root\s+)?view\s+entity", src, re.I):
|
|
if re.search(r"\bunion\b", src, re.I) and not re.search(r"@Metadata\.ignorePropagatedAnnotations\s*:\s*true",
|
|
src, re.I):
|
|
errs.append(f"{rel}: a view entity with UNION needs @Metadata.ignorePropagatedAnnotations: true")
|
|
for p in sorted(set(re.findall(r"[=<>(,]\s*:(\w+)|\bbetween\s+:(\w+)|\band\s+:(\w+)", src, re.I))):
|
|
name = next(x for x in p if x)
|
|
errs.append(f"{rel}: parameter :{name}; a view entity needs $parameters.{name}")
|
|
for k in ("seed", "reference"):
|
|
for o in t.get(k, []):
|
|
m = DDL_KIND.search(b.get("files", {}).get(o.get("file", ""), ""))
|
|
if not m:
|
|
continue
|
|
want = {"table": "TABL", "structure": "TABL", "view": "DDLS"}[m.group(1).lower()]
|
|
if o.get("type") != want:
|
|
errs.append(f"{k}: {o.get('name')} has type {o.get('type')}, but its source is "
|
|
f"'define {m.group(1)}'; use type {want}")
|
|
return errs
|
|
|
|
|
|
def abaplint_files(b):
|
|
"""Parser errors of ABAP sources (classes, interfaces, programs) found by local abaplint.
|
|
SAP often reports only "save operation failed" for these; abaplint gives the line.
|
|
Only parser_error: check_syntax gives false errors for standard objects abaplint does not know."""
|
|
t, files = b.get("task", {}), b.get("files", {})
|
|
ext = {"CLAS": "clas", "INTF": "intf", "PROG": "prog"}
|
|
version = t.get("release_target") if t.get("release_target") in ABAPLINT_VERSIONS else "v758"
|
|
d = tempfile.mkdtemp(prefix="genlint_")
|
|
origin = {}
|
|
try:
|
|
os.makedirs(os.path.join(d, "src"))
|
|
for k in ("seed", "reference", "hidden_tests"):
|
|
for o in t.get(k, []):
|
|
if o.get("type") not in ext:
|
|
continue
|
|
name = re.sub(r"\W", "_", o.get("name", "").replace("{{P}}", "Z0000000_")).lower()
|
|
for fk, suffix in (("file", ""), ("testclasses_file", ".testclasses")):
|
|
if o.get(fk) in files:
|
|
fn = f"{name}.{ext[o['type']]}{suffix}.abap"
|
|
origin[fn] = o[fk]
|
|
src = files[o[fk]].replace("{{P}}", "Z0000000_").replace("{{p}}", "z0000000_")
|
|
open(os.path.join(d, "src", fn), "w").write(src)
|
|
if not origin:
|
|
return []
|
|
cfg = {"global": {"files": "/src/**/*.*"}, "dependencies": [],
|
|
"syntax": {"version": version, "errorNamespace": "^(Z|Y)"}, "rules": {"parser_error": True}}
|
|
json.dump(cfg, open(os.path.join(d, "abaplint.json"), "w"))
|
|
p = subprocess.run([ABAPLINT, "abaplint.json", "-f", "json"], cwd=d, capture_output=True, text=True,
|
|
timeout=300)
|
|
issues = json.loads(p.stdout or "[]")
|
|
except Exception: # noqa: BLE001 the pre-check is optional; SAP validation follows
|
|
return []
|
|
finally:
|
|
shutil.rmtree(d, ignore_errors=True)
|
|
out = []
|
|
for i in issues:
|
|
f = i.get("file", "")
|
|
fn = os.path.basename(f.get("filename", "") if isinstance(f, dict) else str(f))
|
|
if i.get("key") == "parser_error" and fn in origin:
|
|
out.append(f"{origin[fn]} line {i.get('start', {}).get('row')}: syntax error for release {version} "
|
|
f"(abaplint): {i.get('description')}")
|
|
return out[:20]
|
|
|
|
|
|
def check_bundle(b):
|
|
errs = []
|
|
t, files = b.get("task", {}), b.get("files", {})
|
|
stop = t.get("expected_outcome") == "stop"
|
|
for k in ("seed", "contract", "hidden_tests", "reference"):
|
|
if k not in t:
|
|
errs.append(f"task.json misses '{k}'")
|
|
if stop:
|
|
gap = t.get("expected_gap") or {}
|
|
if not gap.get("description") or not gap.get("keywords"):
|
|
errs.append("stop task: expected_gap needs description and keywords")
|
|
if "spec.md" not in files:
|
|
errs.append("spec.md missing")
|
|
for k in ("seed", "hidden_tests", "reference"):
|
|
for o in t.get(k, []):
|
|
for fk in ("file", "testclasses_file"):
|
|
if fk in o and o[fk] not in files:
|
|
errs.append(f"{k}: file {o[fk]} missing")
|
|
if not o.get("name", "").startswith("{{P}}"):
|
|
errs.append(f"{k}: name {o.get('name')} does not start with {{{{P}}}}")
|
|
limit = 16 if o.get("type") == "TABL" else 26 if o.get("type") == "FUGR" else 30
|
|
if len(o.get("name", "").replace("{{P}}", "Z0000000_")) > limit:
|
|
errs.append(f"{k}: name {o.get('name')} too long (max {limit})")
|
|
if not t.get("hidden_tests") and not stop:
|
|
errs.append("no hidden test class")
|
|
return errs
|
|
|
|
|
|
def dependency_order(objs, files):
|
|
"""Stable order in which every object comes after the objects that its sources name
|
|
(for example exception classes before the class that raises them). A cycle keeps the given order."""
|
|
def text(o):
|
|
return " ".join(files.get(o.get(k), "") for k in ("file", "testclasses_file")).upper()
|
|
names = [o.get("name", "").upper() for o in objs]
|
|
needs = []
|
|
for o in objs:
|
|
src, own = text(o), o.get("name", "").upper()
|
|
dep = {n for n in names if n != own and re.search(rf"(?<![\w{{}}]){re.escape(n)}(?!\w)", src)}
|
|
if o.get("functionGroup"):
|
|
dep.add(o["functionGroup"].upper())
|
|
needs.append(dep & set(names))
|
|
out, done = [], set()
|
|
while len(out) < len(objs):
|
|
ready = [i for i, o in enumerate(objs) if i not in done and needs[i] <= {names[j] for j in done}]
|
|
i = ready[0] if ready else min(set(range(len(objs))) - done)
|
|
done.add(i)
|
|
out.append(objs[i])
|
|
return out
|
|
|
|
|
|
def floor_budget(task):
|
|
"""The model needs more calls than the oracle (reads, repairs). G0022: budget 12 activations,
|
|
reference alone 13. Minimum: 2 x oracle activations, 3 x oracle tool calls."""
|
|
ref = task.get("reference", [])
|
|
pushes = sum(bool(o.get("file")) + bool(o.get("testclasses_file")) for o in ref)
|
|
calls = len(ref) + pushes
|
|
bud = task.setdefault("budget", {})
|
|
bud["max_activations"] = max(int(bud.get("max_activations", 0)), 2 * pushes)
|
|
bud["max_tool_calls"] = max(int(bud.get("max_tool_calls", 0)), 3 * calls)
|
|
|
|
|
|
def write_bundle(b, task_dir, task_id):
|
|
os.makedirs(task_dir, exist_ok=True)
|
|
b["task"]["id"] = task_id
|
|
for k in ("seed", "reference"):
|
|
b["task"][k] = dependency_order(b["task"].get(k, []), b["files"])
|
|
floor_budget(b["task"])
|
|
json.dump(b["task"], open(os.path.join(task_dir, "task.json"), "w"), indent=2)
|
|
for rel, content in b["files"].items():
|
|
path = os.path.join(task_dir, rel)
|
|
os.makedirs(os.path.dirname(path), exist_ok=True)
|
|
open(path, "w").write(content)
|
|
|
|
|
|
def validate(pool_root, task_id, run_base):
|
|
r = Runner(pool_root, os.path.join(ROOT, "runs", "gen"))
|
|
os.makedirs(os.path.join(ROOT, "runs", "gen"), exist_ok=True)
|
|
rep_o, dir_o = r.run(task_id, OracleAgent(), run_base)
|
|
rep_o["_dir"] = dir_o
|
|
rep_n, _ = r.run(task_id, NullAgent(), run_base + 1)
|
|
return rep_o, rep_n
|
|
|
|
|
|
def write_errors(run_dir):
|
|
"""Failed writes and activation messages of the oracle run."""
|
|
out = []
|
|
path = os.path.join(run_dir or "", "trajectory.jsonl")
|
|
if not os.path.exists(path):
|
|
return out
|
|
for line in open(path):
|
|
e = json.loads(line)
|
|
if e.get("tool") in ("sap_create_object", "sap_push_source") and (
|
|
e["is_error"] or '"success":false' in e["result"].replace(" ", "")):
|
|
out.append({"tool": e["tool"], "object": e["args"].get("objectName"),
|
|
"include": e["args"].get("includeType", "main"), "result": e["result"][:1500]})
|
|
return out
|
|
|
|
|
|
def failure_summary(rep):
|
|
out = {"reference_write_errors": write_errors(rep.get("_dir")), "gates": rep.get("gates"),
|
|
"contract_check": rep.get("contract_check"), "score": rep.get("score"), "setup": rep.get("setup"),
|
|
"atc": rep.get("atc"), "abaplint": rep.get("abaplint"), "own_tests": rep.get("own_tests"),
|
|
"hidden_failed": [d for d in rep.get("hidden_tests", {}).get("detail", []) if not d["ok"]],
|
|
"hidden_install": rep.get("hidden_tests", {}).get("install")}
|
|
return json.dumps(out)[:6000]
|
|
|
|
|
|
def generate(task_id, pool, object_type, category, difficulty, model, base_url, run_base, topic=None,
|
|
max_repairs=3):
|
|
check_budget()
|
|
pool_root = os.path.join(ROOT, "tasks_gen", pool)
|
|
task_dir = os.path.join(pool_root, task_id)
|
|
example = bundle_of(os.path.join(ROOT, "tasks", EXAMPLE_FOR[object_type]))
|
|
ask = (f"Write one new task.\nObject type of the main contract object: {object_type}.\n"
|
|
f"Skill category {category}: {CATEGORIES[category]}.\nDifficulty {difficulty} of 3.\n"
|
|
+ (f"Topic idea: {topic}\n" if topic else "Choose a new, realistic business topic.\n")
|
|
+ "Here is an example bundle of a different task (same format):\n" + json.dumps(example))
|
|
messages = [{"role": "system", "content": SYSTEM}, {"role": "user", "content": ask}]
|
|
log = {"id": task_id, "pool": pool, "object_type": object_type, "category": category, "attempts": []}
|
|
for attempt in range(max_repairs + 1):
|
|
text = chat(model, messages, base_url)
|
|
messages.append({"role": "assistant", "content": text})
|
|
try:
|
|
b = parse_bundle(text)
|
|
found = {"structure": check_bundle(b), "static": lint_files(b), "abaplint": abaplint_files(b)}
|
|
except Exception as e: # noqa: BLE001
|
|
b, found = None, {"json": [f"invalid JSON: {e}"]}
|
|
errs = [e for v in found.values() for e in v]
|
|
if errs:
|
|
# one entry per source, so a batch shows what the checks found before SAP
|
|
log["attempts"].append({"stage": "bundle", "errors": errs,
|
|
"by_check": {k: len(v) for k, v in found.items() if v}})
|
|
messages.append({"role": "user", "content": "Fix these problems and return the full bundle again:\n"
|
|
+ "\n".join(errs)})
|
|
continue
|
|
b["task"].setdefault("object_type", object_type)
|
|
b["task"].setdefault("category", category)
|
|
write_bundle(b, task_dir, task_id)
|
|
# run numbers per attempt: +0 oracle, +1 null, +2..+6 mutants (RUNS_PER_ATTEMPT)
|
|
base = run_base + RUNS_PER_ATTEMPT * attempt
|
|
rep_o, rep_n = validate(pool_root, task_id, base)
|
|
so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total")
|
|
log["attempts"].append({"stage": "validate", "oracle": so, "null": sn})
|
|
if category == "H": # oracle stops with the gap: 100; null stops without a report: 30
|
|
if so == 100 and sn == 30:
|
|
log["accepted"] = True
|
|
break
|
|
messages.append({"role": "user", "content":
|
|
f"Stop task check: required oracle 100 and null 30. Result: oracle {so}, null {sn}. "
|
|
f"Oracle report: {failure_summary(rep_o)}\nFix expected_gap (description, keywords) "
|
|
"and return the full bundle again."})
|
|
continue
|
|
if so == 100 and sn == 0:
|
|
mut = check_task(pool_root, task_id, base + 2, keep=True)
|
|
log["attempts"][-1]["mutation"] = {k: mut[k] for k in ("valid", "killed", "ok")}
|
|
if mut["ok"]:
|
|
log["accepted"] = True
|
|
break
|
|
survived = [f"{m['object']} {m['mutant']}" for m in mut["mutants"] if m["status"] == "survived"]
|
|
shutil.rmtree(os.path.join(task_dir, "faulty"), ignore_errors=True)
|
|
messages.append({"role": "user", "content":
|
|
"Oracle 100 and null 0: good. But the hidden tests are too weak. The harness changed "
|
|
"the reference (mutation check) and all hidden tests still passed for these changes:\n"
|
|
+ "\n".join(survived or ["(fewer than 2 changes possible: add more business logic "
|
|
"checks to the hidden tests)"])
|
|
+ "\nAdd or improve hidden tests so that each of these changes makes a test fail. "
|
|
"If a change does not change the behavior, ignore it. Do not change the behavior "
|
|
"of the reference. Return the full bundle again."})
|
|
continue
|
|
messages.append({"role": "user", "content":
|
|
"The harness ran your reference solution (oracle) and an empty solution (null). "
|
|
f"Required: oracle 100, null 0. Result: oracle {so}, null {sn}.\n"
|
|
f"Oracle report: {failure_summary(rep_o)}\n"
|
|
"Fix the bundle (reference, hidden tests, seed, or contract) and return the full "
|
|
"bundle again."})
|
|
else:
|
|
log["accepted"] = False
|
|
json.dump(log, open(os.path.join(task_dir if os.path.isdir(task_dir) else pool_root, "generation.json"), "w"),
|
|
indent=1)
|
|
return log
|
|
|
|
|
|
K_STYLES = {
|
|
"free_text": "Rewrite the spec as a short free-text request from a functional consultant: no section "
|
|
"headings, no numbered rules, plain sentences, as in an e-mail. Keep every business rule.",
|
|
"incomplete": "Rewrite the spec as a short free-text request. Leave out the craft hints and the context "
|
|
"that a good ABAP developer can find in the system (for example how the seed table looks). "
|
|
"Keep every business rule; the task must stay solvable without questions.",
|
|
}
|
|
|
|
|
|
def contract_terms(task):
|
|
"""Names that a K spec must keep: object names, FM parameters, report parameters, CDS elements."""
|
|
out = []
|
|
for c in task.get("contract", []):
|
|
out.append(c["name"])
|
|
out += [p["name"] if isinstance(p, dict) else p for p in c.get("params", []) + c.get("parameters", [])]
|
|
out += list(c.get("fields", []))
|
|
return out
|
|
|
|
|
|
def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema="generic_v0", max_repairs=2):
|
|
"""Category K: free-text or incomplete input + other tool schema. Same reference and hidden tests
|
|
as the source task; only spec.md, category and tool_schema change."""
|
|
check_budget()
|
|
pool_root = os.path.join(ROOT, "tasks_gen", "eval")
|
|
src_dir, task_dir = os.path.join(pool_root, src_id), os.path.join(pool_root, new_id)
|
|
b = bundle_of(src_dir)
|
|
b["files"] = {k: v for k, v in b["files"].items()
|
|
if not (k.startswith("faulty/") or k in ("mutation.json",))}
|
|
spec = b["files"]["spec.md"]
|
|
terms = contract_terms(b["task"])
|
|
messages = [{"role": "user", "content":
|
|
f"{K_STYLES[style]}\nKeep these names exactly as written (the tests use them): "
|
|
f"{', '.join(terms)}. Keep the placeholder {{{{P}}}} in names. Use Simplified Technical English. "
|
|
f"Return only the new spec text.\n\nSpec:\n{spec}"}]
|
|
log = {"id": new_id, "pool": "eval", "category": "K", "base_task": src_id, "style": style,
|
|
"tool_schema": tool_schema, "attempts": []}
|
|
for attempt in range(max_repairs + 1):
|
|
text = chat(model, messages, base_url).strip()
|
|
text = re.sub(r"^```\w*\s*|\s*```$", "", text)
|
|
messages.append({"role": "assistant", "content": text})
|
|
missing = [t for t in terms if t.upper() not in text.upper()]
|
|
if missing:
|
|
log["attempts"].append({"stage": "spec", "missing_names": missing})
|
|
messages.append({"role": "user", "content": "These names are missing: " + ", ".join(missing)
|
|
+ ". Return the full spec again with all names."})
|
|
continue
|
|
b["files"]["spec.md"] = text + "\n"
|
|
b["task"].update(category="K", base_task=src_id, input_style=style, tool_schema=tool_schema)
|
|
write_bundle(b, task_dir, new_id)
|
|
rep_o, rep_n = validate(pool_root, new_id, run_base + RUNS_PER_ATTEMPT * attempt)
|
|
so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total")
|
|
log["attempts"].append({"stage": "validate", "oracle": so, "null": sn})
|
|
log["accepted"] = so == 100 and sn == 0
|
|
break
|
|
else:
|
|
log["accepted"] = False
|
|
json.dump(log, open(os.path.join(task_dir if os.path.isdir(task_dir) else pool_root, "generation.json"), "w"),
|
|
indent=1)
|
|
return log
|
|
|
|
|
|
def main():
|
|
load_env(os.path.join(ROOT, ".env"))
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--id", required=True)
|
|
ap.add_argument("--pool", default="eval", choices=["eval", "train"])
|
|
ap.add_argument("--object-type", choices=sorted(EXAMPLE_FOR))
|
|
ap.add_argument("--category", choices=sorted(CATEGORIES))
|
|
ap.add_argument("--difficulty", type=int, default=2)
|
|
ap.add_argument("--topic")
|
|
ap.add_argument("--model", default="deepseek-v4.1-flash:cloud")
|
|
ap.add_argument("--base-url", default=os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1"))
|
|
ap.add_argument("--run-base", type=int, required=True)
|
|
ap.add_argument("--k-from", help="category K: make a variant of this accepted eval task")
|
|
ap.add_argument("--k-style", default="free_text", choices=sorted(K_STYLES))
|
|
a = ap.parse_args()
|
|
if a.k_from:
|
|
print(json.dumps(make_k_variant(a.k_from, a.id, a.k_style, a.model, a.base_url, a.run_base)))
|
|
return
|
|
os.makedirs(os.path.join(ROOT, "tasks_gen", a.pool), exist_ok=True)
|
|
log = generate(a.id, a.pool, a.object_type, a.category, a.difficulty, a.model, a.base_url,
|
|
a.run_base, a.topic)
|
|
print(json.dumps(log))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|