"""Step F: generate eval candidates (A-I, K). Target per category (110 set, docs/faz1-tasarim.md 8): A 10, B 15, C 15, D 10, E 15, F 10, G 10, I 5. H: stop tasks (runner scores the stop, judge.py checks the gap). K: variants of accepted tasks. DDIC and MSAG contracts need a G2 check first. The plan fills the gap between the target and the accepted eval tasks. Resumable: a slot with a generation.json is skipped. python3 -m harness.evalset plan # show the slots python3 -m harness.evalset run [IDS] # generate (sequential); IDS: only these slots """ import glob import json import os import sys from .adt_client import load_env from .generator import ROOT, generate, make_k_variant from .ledger import BudgetExceeded, spent FIRST_ID = 100 RUN_BASE = 6000 # (category, object type, count, topic hints); CDS hints follow docs/faz1-tasarim.md 8.1 SLOTS = [ ("A", "CLAS", 5, None), ("A", "FUNC", 3, None), ("B", "DDLS", 3, ["CASE, cast and built-in functions", "association used from ABAP SQL with a path expression", "aggregation with HAVING"]), ("B", "PROG", 2, None), ("C", "CLAS", 8, None), ("C", "FUNC", 3, None), ("C", "PROG", 2, None), ("D", "CLAS", 7, None), ("D", "FUNC", 2, None), ("E", "CLAS", 5, None), ("E", "PROG", 3, None), ("E", "FUNC", 2, None), ("E", "DDLS", 3, ["old view with nested CASE to clean view entity", "duplicate join logic to one association", "literal values to a parameter"]), ("F", "CLAS", 3, None), ("F", "FUNC", 2, None), ("F", "DDLS", 3, ["view on a seed table with unknown field names", "view that reuses a seed view", "view that uses a seed table function result table"]), ("G", "FUNC", 4, None), ("G", "PROG", 3, None), ("G", "CLAS", 2, None), ("I", "DDLS", 2, ["wrong join condition", "wrong aggregation level"]), ("I", "CLAS", 1, None), ("H", "CLAS", 4, ["missing rule for a boundary value", "two rules contradict for one case", "missing rounding rule", "missing rule for an empty input"]), ("H", "FUNC", 2, ["missing rule for an unknown code", "contradicting priority of two rules"]), ("H", "PROG", 2, ["missing sort order or grouping rule", "contradicting filter rules"]), ("H", "DDLS", 2, ["missing rule for which records count", "contradicting join or filter rule"]), ] # Category K: variants of accepted tasks (free text or incomplete input + tool schema generic_v0) K_SOURCES = [("G0004", "free_text"), ("G0007", "incomplete"), ("G0013", "free_text"), ("G0014", "incomplete"), ("G0017", "free_text"), ("G0020", "incomplete"), ("G0022", "free_text"), ("G0012", "incomplete"), ("G0016", "free_text"), ("G0023", "incomplete")] RELEASES = ["v702", "v740sp05"] def accepted_goals(): """Goal lines of all generated tasks (eval and train), so the model does not repeat a topic.""" out = [] for d in sorted(glob.glob(os.path.join(ROOT, "tasks_gen", "*", "G*"))): try: spec = open(os.path.join(d, "spec.md")).read() except OSError: continue lines = [ln.strip() for ln in spec.split("Goal", 1)[-1].splitlines() if ln.strip()] if lines: out.append(lines[0][:120]) return out def plan(): out, n = [], 0 for cat, otype, count, hints in SLOTS: for i in range(count): topic = hints[i] if hints and i < len(hints) else None if cat == "G": topic = (topic + "; " if topic else "") + f"release target {RELEASES[i % 2]}" out.append({"id": f"G{FIRST_ID + n:04d}", "category": cat, "object_type": otype, "topic": topic, "run_base": RUN_BASE + 40 * n}) n += 1 for src, style in K_SOURCES: out.append({"id": f"G{FIRST_ID + n:04d}", "category": "K", "k_from": src, "style": style, "run_base": RUN_BASE + 40 * n}) n += 1 return out def main(): load_env(os.path.join(ROOT, ".env")) cmd = sys.argv[1] if len(sys.argv) > 1 else "plan" slots = plan() if cmd == "plan": for s in slots: print(s) print(len(slots), "slots") return pool = os.path.join(ROOT, "tasks_gen", "eval") only = set(sys.argv[2:]) # run G0158 G0178: only these slots for s in slots: if only and s["id"] not in only: continue if os.path.exists(os.path.join(pool, s["id"], "generation.json")): continue if s["category"] == "K": try: log = make_k_variant(s["k_from"], s["id"], s["style"], "deepseek-v4.1-flash:cloud", os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1"), s["run_base"]) except BudgetExceeded as e: print("BUDGET", e, flush=True) break log["spent_total"] = spent() print(json.dumps(log), flush=True) continue avoid = accepted_goals() topic = (s["topic"] + ". " if s["topic"] else "Choose a new, realistic business topic. ") + "Do not repeat these existing topics: " + "; ".join(avoid) try: log = generate(s["id"], "eval", s["object_type"], s["category"], 2, "deepseek-v4.1-flash:cloud", os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1"), s["run_base"], topic) except BudgetExceeded as e: print("BUDGET", e, flush=True) break except Exception as e: # noqa: BLE001 log = {"id": s["id"], "error": str(e)[:500]} # no generation.json: the next run tries again log["spent_total"] = spent() print(json.dumps(log), flush=True) if __name__ == "__main__": main()