Files
abap-llm/harness/evalset.py

212 lines
10 KiB
Python

"""Step F: generate eval candidates (A-I, K).
Target per category (110 set, docs/faz1-tasarim.md 8): A 10, B 15, C 15, D 10, E 15, F 10, G 10, I 5.
H: stop tasks (runner scores the stop, judge.py checks the gap). K: variants of accepted tasks.
DDIC and MSAG contracts need a G2 check first.
The plan fills the gap between the target and the accepted eval tasks. Resumable: a slot with a
generation.json is skipped.
python3 -m harness.evalset plan # show the slots
python3 -m harness.evalset run [IDS] # generate (sequential); IDS: only these slots
"""
import glob
import json
import os
import sys
from .adt_client import load_env
from .generator import ROOT, generate, make_k_variant
from . import mix, overlap
from .ledger import BudgetExceeded, spent
FIRST_ID = 100
RUN_BASE = 6000
# (category, object type, count, topic hints); CDS hints follow docs/faz1-tasarim.md 8.1
SLOTS = [
("A", "CLAS", 5, None), ("A", "FUNC", 3, None),
("B", "DDLS", 3, ["CASE, cast and built-in functions", "association used from ABAP SQL with a path expression",
"aggregation with HAVING"]),
("B", "PROG", 2, None),
("C", "CLAS", 8, None), ("C", "FUNC", 3, None), ("C", "PROG", 2, None),
("D", "CLAS", 7, None), ("D", "FUNC", 2, None),
("E", "CLAS", 5, None), ("E", "PROG", 3, None), ("E", "FUNC", 2, None),
("E", "DDLS", 3, ["old view with nested CASE to clean view entity", "duplicate join logic to one association",
"literal values to a parameter"]),
("F", "CLAS", 3, None), ("F", "FUNC", 2, None),
("F", "DDLS", 3, ["view on a seed table with unknown field names", "view that reuses a seed view",
"view that uses a seed table function result table"]),
("G", "FUNC", 4, None), ("G", "PROG", 3, None), ("G", "CLAS", 2, None),
("I", "DDLS", 2, ["wrong join condition", "wrong aggregation level"]), ("I", "CLAS", 1, None),
("H", "CLAS", 4, ["missing rule for a boundary value", "two rules contradict for one case",
"missing rounding rule", "missing rule for an empty input"]),
("H", "FUNC", 2, ["missing rule for an unknown code", "contradicting priority of two rules"]),
("H", "PROG", 2, ["missing sort order or grouping rule", "contradicting filter rules"]),
("H", "DDLS", 2, ["missing rule for which records count", "contradicting join or filter rule"]),
]
# Replacements for G0009, G0013 (I) and G0016, G0023 (E): those were solvable without the legacy code
# (review 2026-10-03). Same category and object type, new definition (in place).
REGEN = [("I", "CLAS", "G0009"), ("I", "FUNC", "G0013"), ("E", "PROG", "G0016"), ("E", "CLAS", "G0023")]
# Category K: variants of accepted tasks (free text or incomplete input + tool schema generic_v0)
K_SOURCES = [("G0004", "free_text"), ("G0007", "incomplete"), ("G0026", "free_text"), ("G0014", "incomplete"),
("G0017", "free_text"), ("G0020", "incomplete"), ("G0022", "free_text"), ("G0012", "incomplete"),
("G0003", "free_text"), ("G0010", "incomplete")]
RELEASES = ["v702", "v740sp05"]
# New kinds for the eval set (Opus item F, 2026-10-06): the 130 eval candidates have only CLAS, FUNC, DDLS and PROG, so INTF, TABL, STRU, MSAG and
# exception tasks cannot be measured after training. Shares as in the training mix (110 tasks: INTF 8, TABL 9, STRU 2, MSAG 3, exception 3); about
# 30 percent more slots than the target because of rejections and the empirical filter. Generation after the reset, FIRST (before training tasks).
NEW_FIRST_ID = 200
NEW_RUN_BASE = 440000 # 40 per slot, below 466560
NEW_KIND_SLOTS = [("INTF", "A", 10), ("TABL", "B", 11), ("STRU", "B", 3), ("MSAG", "D", 4), ("EXC", "D", 4)]
NEW_K = [("INTF", "free_text"), ("TABL", "incomplete"), ("TABL", "free_text"), ("MSAG", "free_text"), ("EXC", "incomplete")]
EXC_TOPICS = ["exception class (CX_...): a domain exception with context attributes and message texts",
"exception class (CX_...): an exception hierarchy with a common super class",
"exception class (CX_...): an exception that wraps a previous exception",
"exception class (CX_...): an exception with a message class and parameters in the text"]
def plan_new():
out, n = [], 0
for kind, cat, count in NEW_KIND_SLOTS:
for i in range(count):
out.append({"id": f"G{NEW_FIRST_ID + n:04d}", "kind": kind, "category": cat, "object_type": "CLAS" if kind == "EXC" else kind,
"topic": EXC_TOPICS[i % len(EXC_TOPICS)] if kind == "EXC" else None,
"difficulty": 3 if i % 3 == 2 else 2, "run_base": NEW_RUN_BASE + 40 * n})
n += 1
return out
def run_new(only, model, base_url):
"""Generate the new-kind eval candidates. The bundle is also checked against the training pool (no near duplicate of a training task)."""
train = overlap.load_pool("train")
for s in plan_new():
if only and s["id"] not in only:
continue
if os.path.exists(os.path.join(ROOT, "tasks_gen", "eval", s["id"], "generation.json")):
continue
avoid = [g for g in accepted_goals() if g][-170:]
topic = ((s["topic"] + ". ") if s["topic"] else "Choose a new, realistic business topic. ") + "Do not repeat these existing topics: " + "; ".join(avoid)
def extra(b, _t=train):
return [f"Too close to task {i} (spec {sc['spec']:.2f}, rules {sc['core']:.2f}, names {sc['name']:.2f}). Choose another topic and other object names."
for i, sc in overlap.check(overlap.load_bundle(b), _t)[:3]]
try:
log = generate(s["id"], "eval", s["object_type"], s["category"], s["difficulty"], model, base_url, s["run_base"], topic, extra_check=extra)
except BudgetExceeded as e:
print("BUDGET", e, flush=True)
break
except Exception as e: # noqa: BLE001
log = {"id": s["id"], "error": str(e)[:500]}
log.update(kind=s["kind"], spent_total=spent())
print(json.dumps(log), flush=True)
def run_new_k(model, base_url):
"""K variants (free text or incomplete spec, EPOD tool names) of the first accepted eval task of each new kind."""
plan = plan_new()
first_k = NEW_FIRST_ID + len(plan)
for j, (kind, style) in enumerate(NEW_K):
src = next((s["id"] for s in plan if s["kind"] == kind and os.path.exists(os.path.join(ROOT, "tasks_gen", "eval", s["id"], "generation.json"))
and json.load(open(os.path.join(ROOT, "tasks_gen", "eval", s["id"], "generation.json"))).get("accepted")), None)
new_id = f"G{first_k + j:04d}"
if not src or os.path.exists(os.path.join(ROOT, "tasks_gen", "eval", new_id, "generation.json")):
continue
try:
log = make_k_variant(src, new_id, style, model, base_url, NEW_RUN_BASE + 40 * (len(plan) + j), tool_schema=None, pool="eval")
except BudgetExceeded as e:
print("BUDGET", e, flush=True)
break
print(json.dumps(log), flush=True)
def accepted_goals():
"""Goal lines of all generated tasks (eval and train), so the model does not repeat a topic."""
out = []
for d in sorted(glob.glob(os.path.join(ROOT, "tasks_gen", "*", "G*"))):
try:
spec = open(os.path.join(d, "spec.md")).read()
except OSError:
continue
lines = [ln.strip() for ln in spec.split("Goal", 1)[-1].splitlines() if ln.strip()]
if lines:
out.append(lines[0][:120])
return out
def plan():
out, n = [], 0
for cat, otype, count, hints in SLOTS:
for i in range(count):
topic = hints[i] if hints and i < len(hints) else None
if cat == "G":
topic = (topic + "; " if topic else "") + f"release target {RELEASES[i % 2]}"
out.append({"id": f"G{FIRST_ID + n:04d}", "category": cat, "object_type": otype, "topic": topic,
"run_base": RUN_BASE + 40 * n})
n += 1
for src, style in K_SOURCES:
out.append({"id": f"G{FIRST_ID + n:04d}", "category": "K", "k_from": src, "style": style,
"run_base": RUN_BASE + 40 * n})
n += 1
for cat, otype, old in REGEN:
out.append({"id": f"G{FIRST_ID + n:04d}", "category": cat, "object_type": otype, "topic": None,
"replaces": old, "run_base": RUN_BASE + 40 * n})
n += 1
return out
def main():
load_env(os.path.join(ROOT, ".env"))
cmd = sys.argv[1] if len(sys.argv) > 1 else "plan"
model, base_url = "deepseek-v4.1-flash:cloud", os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")
if cmd == "plan-new":
for x in plan_new():
print(x)
print(len(plan_new()), "slots,", len(NEW_K), "K variants after them")
return
if cmd == "run-new":
run_new(set(sys.argv[2:]), model, base_url)
return
if cmd == "run-new-k":
run_new_k(model, base_url)
return
slots = plan()
if cmd == "plan":
for s in slots:
print(s)
print(len(slots), "slots")
return
pool = os.path.join(ROOT, "tasks_gen", "eval")
only = set(sys.argv[2:]) # run G0158 G0178: only these slots
for s in slots:
if only and s["id"] not in only:
continue
if os.path.exists(os.path.join(pool, s["id"], "generation.json")):
continue
if s["category"] == "K":
try:
log = make_k_variant(s["k_from"], s["id"], s["style"], "deepseek-v4.1-flash:cloud",
os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1"), s["run_base"])
except BudgetExceeded as e:
print("BUDGET", e, flush=True)
break
log["spent_total"] = spent()
print(json.dumps(log), flush=True)
continue
avoid = accepted_goals()
topic = (s["topic"] + ". " if s["topic"] else "Choose a new, realistic business topic. ") + "Do not repeat these existing topics: " + "; ".join(avoid)
try:
log = generate(s["id"], "eval", s["object_type"], s["category"], 2, "deepseek-v4.1-flash:cloud",
os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1"), s["run_base"], topic)
except BudgetExceeded as e:
print("BUDGET", e, flush=True)
break
except Exception as e: # noqa: BLE001
log = {"id": s["id"], "error": str(e)[:500]} # no generation.json: the next run tries again
log["spent_total"] = spent()
print(json.dumps(log), flush=True)
if __name__ == "__main__":
main()