diff --git a/CLAUDE.md b/CLAUDE.md index 3c7521c..915e0bb 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -112,5 +112,7 @@ python3 -c "from harness.ledger import spent; print(spent())" - When a write fails, run a syntax check and return its messages. Now the server says only "save failed" (for example for `TYPE c LENGTH n` in a method signature). The model cannot see the cause and cannot repair; this hurts eval and training. Workaround in the generator: local abaplint parser check. +- Unit test timeout: an endless loop in tested code blocks the one RFC connection for ~8 min + (mutation run 5201, G0002). Request: stop a unit test run after N seconds (DURATION SHORT = 60 s). - Concurrency: queue calls per RFC connection, or use a connection pool. - BDEF creation (needed for RAP tasks). diff --git a/docs/faz1-tasarim.md b/docs/faz1-tasarim.md index 8805059..481c5b1 100644 --- a/docs/faz1-tasarim.md +++ b/docs/faz1-tasarim.md @@ -615,7 +615,7 @@ Cevaplananlar (Adım 1.2): - ~~abaplint ve seed tipleri~~ → çalışıyor, bölüm 6.5. Açık olanlar: -- EPOD server istekleri (Kral): (1) yazma başarısız olursa syntax check çalıştırıp mesajlarını döndürsün; şu an sadece "save failed" diyor, model nedeni göremez (2026-10-02). (2) RFC bağlantısı başına kuyruk veya bağlantı havuzu ("Concurrent call detected"). (3) BDEF oluşturma (RAP). +- EPOD server istekleri (Kral): (1) yazma başarısız olursa syntax check çalıştırıp mesajlarını döndürsün; şu an sadece "save failed" diyor, model nedeni göremez (2026-10-02). (2) RFC bağlantısı başına kuyruk veya bağlantı havuzu ("Concurrent call detected"). (3) BDEF oluşturma (RAP). (4) Unit test zaman aşımı: test edilen kodda sonsuz döngü RFC bağlantısını ~8 dk kilitler (mutasyon koşusu 5201, 2026-10-03); N saniye sonra durdurulmalı. - Genel ABAP MCP arayüzünün spec'i (`abap-mcp-arayuz.md`): tool isimleri, parametre şemaları, çekirdek dışı tool'lar, hata formatı. Sonra yazılacak. 1. ~~Görevleri kim yazar?~~ → Öneri (2026-10-02): DeepSeek V4.1 Flash yazar. İnceleme üç katmanlı, Kral'ın tam incelemesi yok: - **Otomatik kapılar:** oracle = 100, null = 0, gizli testler bozuk referansta (mutasyon) düşer, her iş kuralı en az bir gizli testle örtülü. diff --git a/harness/agents.py b/harness/agents.py index ca5ff56..5b8d77e 100644 --- a/harness/agents.py +++ b/harness/agents.py @@ -43,6 +43,10 @@ class OracleAgent: name = "oracle" def run(self, task, proxy): + if task.meta.get("expected_outcome") == "stop": # the right answer is to stop and name the gap + report = "Stopped. Gap: " + task.meta.get("expected_gap", {}).get("description", "") + proxy.note("final", report) + return report for o in task.objects("reference"): ident = {"objectType": o["type"], "objectName": o["name"]} if o.get("functionGroup"): diff --git a/harness/evalset.py b/harness/evalset.py new file mode 100644 index 0000000..a2d7d3b --- /dev/null +++ b/harness/evalset.py @@ -0,0 +1,124 @@ +"""Step F: generate eval candidates (A-I, K). + +Target per category (110 set, docs/faz1-tasarim.md 8): A 10, B 15, C 15, D 10, E 15, F 10, G 10, I 5. +H: stop tasks (runner scores the stop, judge.py checks the gap). K: variants of accepted tasks. +DDIC and MSAG contracts need a G2 check first. +The plan fills the gap between the target and the accepted eval tasks. Resumable: a slot with a +generation.json is skipped. + + python3 -m harness.evalset plan # show the slots + python3 -m harness.evalset run [IDS] # generate (sequential); IDS: only these slots +""" +import glob +import json +import os +import sys + +from .adt_client import load_env +from .generator import ROOT, generate, make_k_variant +from .ledger import BudgetExceeded, spent + +FIRST_ID = 100 +RUN_BASE = 6000 +# (category, object type, count, topic hints); CDS hints follow docs/faz1-tasarim.md 8.1 +SLOTS = [ + ("A", "CLAS", 5, None), ("A", "FUNC", 3, None), + ("B", "DDLS", 3, ["CASE, cast and built-in functions", "association used from ABAP SQL with a path expression", + "aggregation with HAVING"]), + ("B", "PROG", 2, None), + ("C", "CLAS", 8, None), ("C", "FUNC", 3, None), ("C", "PROG", 2, None), + ("D", "CLAS", 7, None), ("D", "FUNC", 2, None), + ("E", "CLAS", 5, None), ("E", "PROG", 3, None), ("E", "FUNC", 2, None), + ("E", "DDLS", 3, ["old view with nested CASE to clean view entity", "duplicate join logic to one association", + "literal values to a parameter"]), + ("F", "CLAS", 3, None), ("F", "FUNC", 2, None), + ("F", "DDLS", 3, ["view on a seed table with unknown field names", "view that reuses a seed view", + "view that uses a seed table function result table"]), + ("G", "FUNC", 4, None), ("G", "PROG", 3, None), ("G", "CLAS", 2, None), + ("I", "DDLS", 2, ["wrong join condition", "wrong aggregation level"]), ("I", "CLAS", 1, None), + ("H", "CLAS", 4, ["missing rule for a boundary value", "two rules contradict for one case", + "missing rounding rule", "missing rule for an empty input"]), + ("H", "FUNC", 2, ["missing rule for an unknown code", "contradicting priority of two rules"]), + ("H", "PROG", 2, ["missing sort order or grouping rule", "contradicting filter rules"]), + ("H", "DDLS", 2, ["missing rule for which records count", "contradicting join or filter rule"]), +] +# Category K: variants of accepted tasks (free text or incomplete input + tool schema generic_v0) +K_SOURCES = [("G0004", "free_text"), ("G0007", "incomplete"), ("G0013", "free_text"), ("G0014", "incomplete"), + ("G0017", "free_text"), ("G0020", "incomplete"), ("G0022", "free_text"), ("G0012", "incomplete"), + ("G0016", "free_text"), ("G0023", "incomplete")] +RELEASES = ["v702", "v740sp05"] + + +def accepted_goals(): + """Goal lines of all generated tasks (eval and train), so the model does not repeat a topic.""" + out = [] + for d in sorted(glob.glob(os.path.join(ROOT, "tasks_gen", "*", "G*"))): + try: + spec = open(os.path.join(d, "spec.md")).read() + except OSError: + continue + lines = [ln.strip() for ln in spec.split("Goal", 1)[-1].splitlines() if ln.strip()] + if lines: + out.append(lines[0][:120]) + return out + + +def plan(): + out, n = [], 0 + for cat, otype, count, hints in SLOTS: + for i in range(count): + topic = hints[i] if hints and i < len(hints) else None + if cat == "G": + topic = (topic + "; " if topic else "") + f"release target {RELEASES[i % 2]}" + out.append({"id": f"G{FIRST_ID + n:04d}", "category": cat, "object_type": otype, "topic": topic, + "run_base": RUN_BASE + 40 * n}) + n += 1 + for src, style in K_SOURCES: + out.append({"id": f"G{FIRST_ID + n:04d}", "category": "K", "k_from": src, "style": style, + "run_base": RUN_BASE + 40 * n}) + n += 1 + return out + + +def main(): + load_env(os.path.join(ROOT, ".env")) + cmd = sys.argv[1] if len(sys.argv) > 1 else "plan" + slots = plan() + if cmd == "plan": + for s in slots: + print(s) + print(len(slots), "slots") + return + pool = os.path.join(ROOT, "tasks_gen", "eval") + only = set(sys.argv[2:]) # run G0158 G0178: only these slots + for s in slots: + if only and s["id"] not in only: + continue + if os.path.exists(os.path.join(pool, s["id"], "generation.json")): + continue + if s["category"] == "K": + try: + log = make_k_variant(s["k_from"], s["id"], s["style"], "deepseek-v4.1-flash:cloud", + os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1"), s["run_base"]) + except BudgetExceeded as e: + print("BUDGET", e, flush=True) + break + log["spent_total"] = spent() + print(json.dumps(log), flush=True) + continue + avoid = accepted_goals() + topic = (s["topic"] + ". " if s["topic"] else "Choose a new, realistic business topic. ") + "Do not repeat these existing topics: " + "; ".join(avoid) + try: + log = generate(s["id"], "eval", s["object_type"], s["category"], 2, "deepseek-v4.1-flash:cloud", + os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1"), s["run_base"], topic) + except BudgetExceeded as e: + print("BUDGET", e, flush=True) + break + except Exception as e: # noqa: BLE001 + log = {"id": s["id"], "error": str(e)[:500]} # no generation.json: the next run tries again + log["spent_total"] = spent() + print(json.dumps(log), flush=True) + + +if __name__ == "__main__": + main() diff --git a/harness/generator.py b/harness/generator.py index 265a6e2..16c0e9e 100644 --- a/harness/generator.py +++ b/harness/generator.py @@ -15,12 +15,14 @@ import urllib.request from .adt_client import load_env from .agents import NullAgent, OracleAgent from .ledger import add_usage, check_budget +from .mutation import check_task from .runner import Runner ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) ABAPLINT = os.path.join(ROOT, "node_modules", ".bin", "abaplint") # pilot: accepted final calls used 4k-76k output tokens (p90 57k); one runaway call used 393k MAX_OUT_TOKENS = 80000 +RUNS_PER_ATTEMPT = 10 ABAPLINT_VERSIONS = {"v702", "v740sp05", "v750", "v758"} EXAMPLE_FOR = {"CLAS": "T01", "INTF": "T01", "FUNC": "T13", "PROG": "T14", "DDLS": "T15", "TABL": "T15"} CATEGORIES = { @@ -32,6 +34,13 @@ CATEGORIES = { "F": "use an unknown object: the spec names a seed object but not its signature; the model must read it", "G": "old release target (v702 or v740sp05): no newer syntax", "I": "bug fix: the seed contains a faulty object; the spec describes the symptom", + "H": "stop task: the spec looks complete, but one business rule is missing or two rules contradict, and " + "a correct solution is not possible without an answer. The model must stop, create no object and " + "name the gap. task.json: expected_outcome \"stop\", expected_gap {\"description\": \"\", \"keywords\": [[\"word\", \"word\"], ...]} (2 to 4 groups; a group matches when all its " + "words occur in a report that names the gap). Keep the contract, seed and a reference of the " + "intended objects (the reference shows the task is otherwise clear). No hidden tests: " + "hidden_tests is []. Do not write the gap in Open questions (write \"None.\")", } SYSTEM = """You write evaluation tasks for an ABAP developer model. Each task is a bundle of files. @@ -247,9 +256,14 @@ def abaplint_files(b): def check_bundle(b): errs = [] t, files = b.get("task", {}), b.get("files", {}) + stop = t.get("expected_outcome") == "stop" for k in ("seed", "contract", "hidden_tests", "reference"): if k not in t: errs.append(f"task.json misses '{k}'") + if stop: + gap = t.get("expected_gap") or {} + if not gap.get("description") or not gap.get("keywords"): + errs.append("stop task: expected_gap needs description and keywords") if "spec.md" not in files: errs.append("spec.md missing") for k in ("seed", "hidden_tests", "reference"): @@ -262,7 +276,7 @@ def check_bundle(b): limit = 16 if o.get("type") == "TABL" else 26 if o.get("type") == "FUGR" else 30 if len(o.get("name", "").replace("{{P}}", "Z0000000_")) > limit: errs.append(f"{k}: name {o.get('name')} too long (max {limit})") - if not t.get("hidden_tests"): + if not t.get("hidden_tests") and not stop: errs.append("no hidden test class") return errs @@ -377,12 +391,37 @@ def generate(task_id, pool, object_type, category, difficulty, model, base_url, b["task"].setdefault("object_type", object_type) b["task"].setdefault("category", category) write_bundle(b, task_dir, task_id) - rep_o, rep_n = validate(pool_root, task_id, run_base + 2 * attempt) + # run numbers per attempt: +0 oracle, +1 null, +2..+6 mutants (RUNS_PER_ATTEMPT) + base = run_base + RUNS_PER_ATTEMPT * attempt + rep_o, rep_n = validate(pool_root, task_id, base) so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total") log["attempts"].append({"stage": "validate", "oracle": so, "null": sn}) + if category == "H": # oracle stops with the gap: 100; null stops without a report: 30 + if so == 100 and sn == 30: + log["accepted"] = True + break + messages.append({"role": "user", "content": + f"Stop task check: required oracle 100 and null 30. Result: oracle {so}, null {sn}. " + f"Oracle report: {failure_summary(rep_o)}\nFix expected_gap (description, keywords) " + "and return the full bundle again."}) + continue if so == 100 and sn == 0: - log["accepted"] = True - break + mut = check_task(pool_root, task_id, base + 2, keep=True) + log["attempts"][-1]["mutation"] = {k: mut[k] for k in ("valid", "killed", "ok")} + if mut["ok"]: + log["accepted"] = True + break + survived = [f"{m['object']} {m['mutant']}" for m in mut["mutants"] if m["status"] == "survived"] + shutil.rmtree(os.path.join(task_dir, "faulty"), ignore_errors=True) + messages.append({"role": "user", "content": + "Oracle 100 and null 0: good. But the hidden tests are too weak. The harness changed " + "the reference (mutation check) and all hidden tests still passed for these changes:\n" + + "\n".join(survived or ["(fewer than 2 changes possible: add more business logic " + "checks to the hidden tests)"]) + + "\nAdd or improve hidden tests so that each of these changes makes a test fail. " + "If a change does not change the behavior, ignore it. Do not change the behavior " + "of the reference. Return the full bundle again."}) + continue messages.append({"role": "user", "content": "The harness ran your reference solution (oracle) and an empty solution (null). " f"Required: oracle 100, null 0. Result: oracle {so}, null {sn}.\n" @@ -396,19 +435,85 @@ def generate(task_id, pool, object_type, category, difficulty, model, base_url, return log +K_STYLES = { + "free_text": "Rewrite the spec as a short free-text request from a functional consultant: no section " + "headings, no numbered rules, plain sentences, as in an e-mail. Keep every business rule.", + "incomplete": "Rewrite the spec as a short free-text request. Leave out the craft hints and the context " + "that a good ABAP developer can find in the system (for example how the seed table looks). " + "Keep every business rule; the task must stay solvable without questions.", +} + + +def contract_terms(task): + """Names that a K spec must keep: object names, FM parameters, report parameters, CDS elements.""" + out = [] + for c in task.get("contract", []): + out.append(c["name"]) + out += [p["name"] if isinstance(p, dict) else p for p in c.get("params", []) + c.get("parameters", [])] + out += list(c.get("fields", [])) + return out + + +def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema="generic_v0", max_repairs=2): + """Category K: free-text or incomplete input + other tool schema. Same reference and hidden tests + as the source task; only spec.md, category and tool_schema change.""" + check_budget() + pool_root = os.path.join(ROOT, "tasks_gen", "eval") + src_dir, task_dir = os.path.join(pool_root, src_id), os.path.join(pool_root, new_id) + b = bundle_of(src_dir) + b["files"] = {k: v for k, v in b["files"].items() + if not (k.startswith("faulty/") or k in ("mutation.json",))} + spec = b["files"]["spec.md"] + terms = contract_terms(b["task"]) + messages = [{"role": "user", "content": + f"{K_STYLES[style]}\nKeep these names exactly as written (the tests use them): " + f"{', '.join(terms)}. Keep the placeholder {{{{P}}}} in names. Use Simplified Technical English. " + f"Return only the new spec text.\n\nSpec:\n{spec}"}] + log = {"id": new_id, "pool": "eval", "category": "K", "base_task": src_id, "style": style, + "tool_schema": tool_schema, "attempts": []} + for attempt in range(max_repairs + 1): + text = chat(model, messages, base_url).strip() + text = re.sub(r"^```\w*\s*|\s*```$", "", text) + messages.append({"role": "assistant", "content": text}) + missing = [t for t in terms if t.upper() not in text.upper()] + if missing: + log["attempts"].append({"stage": "spec", "missing_names": missing}) + messages.append({"role": "user", "content": "These names are missing: " + ", ".join(missing) + + ". Return the full spec again with all names."}) + continue + b["files"]["spec.md"] = text + "\n" + b["task"].update(category="K", base_task=src_id, input_style=style, tool_schema=tool_schema) + write_bundle(b, task_dir, new_id) + rep_o, rep_n = validate(pool_root, new_id, run_base + RUNS_PER_ATTEMPT * attempt) + so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total") + log["attempts"].append({"stage": "validate", "oracle": so, "null": sn}) + log["accepted"] = so == 100 and sn == 0 + break + else: + log["accepted"] = False + json.dump(log, open(os.path.join(task_dir if os.path.isdir(task_dir) else pool_root, "generation.json"), "w"), + indent=1) + return log + + def main(): load_env(os.path.join(ROOT, ".env")) ap = argparse.ArgumentParser() ap.add_argument("--id", required=True) ap.add_argument("--pool", default="eval", choices=["eval", "train"]) - ap.add_argument("--object-type", required=True, choices=sorted(EXAMPLE_FOR)) - ap.add_argument("--category", required=True, choices=sorted(CATEGORIES)) + ap.add_argument("--object-type", choices=sorted(EXAMPLE_FOR)) + ap.add_argument("--category", choices=sorted(CATEGORIES)) ap.add_argument("--difficulty", type=int, default=2) ap.add_argument("--topic") ap.add_argument("--model", default="deepseek-v4.1-flash:cloud") ap.add_argument("--base-url", default=os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")) ap.add_argument("--run-base", type=int, required=True) + ap.add_argument("--k-from", help="category K: make a variant of this accepted eval task") + ap.add_argument("--k-style", default="free_text", choices=sorted(K_STYLES)) a = ap.parse_args() + if a.k_from: + print(json.dumps(make_k_variant(a.k_from, a.id, a.k_style, a.model, a.base_url, a.run_base))) + return os.makedirs(os.path.join(ROOT, "tasks_gen", a.pool), exist_ok=True) log = generate(a.id, a.pool, a.object_type, a.category, a.difficulty, a.model, a.base_url, a.run_base, a.topic) diff --git a/harness/judge.py b/harness/judge.py new file mode 100644 index 0000000..889d6c5 --- /dev/null +++ b/harness/judge.py @@ -0,0 +1,64 @@ +"""Stop tasks (category H): did the model name the right gap? + +1. Keywords: task.json "expected_gap": {"description": "...", "keywords": [["discount", "rate"], ...]}. + A group matches when all its words occur in the final report (case-insensitive). +2. If no group matches: a judge model compares the report with the description (one short call). +""" +import json +import os +import re +import urllib.request + +from .ledger import add_usage, check_budget + +JUDGE_MODEL = os.environ.get("JUDGE_MODEL", "deepseek-v4.1-flash:cloud") +JUDGE_PROMPT = """You check the report of an ABAP developer model. The task had a gap: a missing or +contradictory business rule. The model had to stop and name this gap. + +Expected gap: +{gap} + +Report of the model: +{report} + +Does the report name the expected gap (the same missing or contradictory rule)? A report that names +only other, unrelated points does not name it. Answer with one JSON object and nothing else: +{{"names_gap": true or false, "reason": ""}}""" + + +def keyword_hit(report, groups): + text = (report or "").lower() + for g in groups or []: + if all(re.search(rf"\b{re.escape(w.lower())}", text) for w in g): + return g + return None + + +def judge(report, gap_description, base_url=None): + check_budget() + base_url = base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1") + body = {"model": JUDGE_MODEL, "temperature": 0, "max_tokens": 4000, + "messages": [{"role": "user", "content": JUDGE_PROMPT.format(gap=gap_description, + report=(report or "")[:6000])}]} + req = urllib.request.Request(f"{base_url}/chat/completions", json.dumps(body).encode(), + {"Content-Type": "application/json", "Authorization": "Bearer none"}) + with urllib.request.urlopen(req, timeout=600) as r: + data = json.loads(r.read().decode()) + add_usage(JUDGE_MODEL, data.get("usage", {}), kind="judge") + text = data["choices"][0]["message"].get("content") or "" + m = re.search(r"\{.*\}", text, re.S) + try: + return json.loads(m.group(0)) if m else {"names_gap": False, "reason": "no JSON: " + text[:200]} + except ValueError: + return {"names_gap": False, "reason": "bad JSON: " + text[:200]} + + +def gap_named(report, expected_gap): + """Returns (named, detail).""" + if not (report or "").strip() or report.strip() == "No action.": + return False, {"how": "empty report"} + g = keyword_hit(report, expected_gap.get("keywords")) + if g: + return True, {"how": "keywords", "group": g} + v = judge(report, expected_gap.get("description", "")) + return bool(v.get("names_gap")), {"how": "judge", "model": JUDGE_MODEL, **v} diff --git a/harness/mutation.py b/harness/mutation.py index 08437a3..940a7bf 100644 --- a/harness/mutation.py +++ b/harness/mutation.py @@ -21,16 +21,18 @@ from .runner import Runner ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) MAX_MUTANTS = 5 -MIN_VALID = 3 +MIN_VALID = 2 # small references (T14, T15) have only a few mutation sites MIN_KILL_RATE = 0.75 # (kind, pattern, replacement); patterns match only outside comments and literals +# Relational operators are negated, not shifted: a shifted boundary (< to <=) is often equivalent in +# clamp code (T13: "IF fee < 1. fee = 1."), so it says nothing about the tests. ABAP_OPS = [ ("rel", r"(?<=\s)>=(?=\s)", "<"), ("rel", r"(?<=\s)<=(?=\s)", ">"), - ("rel", r"(?<=\s)>(?=\s)", ">="), ("rel", r"(?<=\s)<(?=\s)", "<="), - ("rel", r"(?<=\s)<>(?=\s)", "="), - ("rel", r"\bGE\b", "LT"), ("rel", r"\bLE\b", "GT"), ("rel", r"\bGT\b", "GE"), ("rel", r"\bLT\b", "LE"), - ("rel", r"\bNE\b", "EQ"), + ("rel", r"(?<=\s)>(?=\s)", "<="), ("rel", r"(?<=\s)<(?=\s)", ">="), + ("rel", r"(?<=\s)<>(?=\s)", "="), ("eq", r"(?<=\s)=(?=\s)", "<>"), + ("rel", r"\bGE\b", "LT"), ("rel", r"\bLE\b", "GT"), ("rel", r"\bGT\b", "LE"), ("rel", r"\bLT\b", "GE"), + ("rel", r"\bNE\b", "EQ"), ("rel", r"\bEQ\b", "NE"), ("logic", r"(?<=\s)AND(?=\s)", "OR"), ("logic", r"(?<=\s)OR(?=\s)", "AND"), ("arith", r"(?<=\s)\+(?=\s)", "-"), ("arith", r"(?<=\s)-(?=\s)", "+"), ("arith", r"(?<=\s)\*(?=\s)", "/"), ("bool", r"\babap_true\b", "abap_false"), ("bool", r"\babap_false\b", "abap_true"), @@ -38,13 +40,16 @@ ABAP_OPS = [ ] CDS_OPS = [ ("rel", r"(?<=\s)>=(?=\s)", "<"), ("rel", r"(?<=\s)<=(?=\s)", ">"), - ("rel", r"(?<=\s)>(?=\s)", ">="), ("rel", r"(?<=\s)<(?=\s)", "<="), ("rel", r"(?<=\s)<>(?=\s)", "="), + ("rel", r"(?<=\s)>(?=\s)", "<="), ("rel", r"(?<=\s)<(?=\s)", ">="), ("rel", r"(?<=\s)<>(?=\s)", "="), + ("eq", r"(?<=\s)=(?=\s)", "<>"), ("logic", r"(?<=\s)and(?=\s)", "or"), ("arith", r"(?<=\s)\+(?=\s)", "-"), ("arith", r"(?<=\s)-(?=\s)", "+"), - ("agg", r"\bsum\s*\(", "max("), ("agg", r"\bcount\s*\(", "max("), ("agg", r"\bavg\s*\(", "max("), + ("agg", r"\bsum\s*\(", "max("), ("agg", r"\bavg\s*\(", "max("), ("join", r"\binner\s+join\b", "left outer join"), ("const", r"(?= 0 else len(src)] + if kind == "eq" and not cds and not ABAP_CONDITION.match(line): + continue + if kind == "arith" and re.search(r"(\(|\bSELECT)\s*$", src[line_start:x.start()], re.I): + continue # COUNT( * ), SELECT * if _skip_line(line) or (kind == "const" and re.search(r"\bLENGTH\b|\bDECIMALS\b|\(\s*\d+\s*,", line, re.I)): continue new = str(int(x.group(1)) + 1) if rep is None else rep @@ -104,7 +126,7 @@ def mutants(src, otype, seed, n=MAX_MUTANTS): for s in sites: if len(chosen) >= n: break - if s in chosen or s[4] in lines or (prefer_new and s[0] in kinds): + if s in chosen or (prefer_new and (s[4] in lines or s[0] in kinds)): continue chosen.append(s) kinds.add(s[0]) diff --git a/harness/pilot.py b/harness/pilot.py index 883fb08..0159ea2 100644 --- a/harness/pilot.py +++ b/harness/pilot.py @@ -40,7 +40,7 @@ def main(): try: log = generate(tid, "eval", otype, cat, 2, "deepseek-v4.1-flash:cloud", os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1"), - run_base + 10 * k, topic) + run_base + 40 * k, topic) except BudgetExceeded as e: print("BUDGET", e, flush=True) break diff --git a/harness/proxy.py b/harness/proxy.py index 5f9fb0a..b0b13d8 100644 --- a/harness/proxy.py +++ b/harness/proxy.py @@ -12,6 +12,24 @@ MODEL_TOOLS = { } WRITE_TOOLS = {"sap_create_object", "sap_push_source", "sap_push_element", "sap_push_message", "sap_activate"} +# Tool schema variants (category K; small variation in training). Draft until abap-mcp-arayuz.md exists. +# The model sees the variant names and argument names; the proxy translates them to EPOD. +VARIANTS = { + "generic_v0": { + "tools": {"search_object": "sap_search_object", "read_source": "sap_pull_source", + "object_structure": "sap_object_structure", "where_used": "sap_usage_references", + "create_object": "sap_create_object", "write_source": "sap_push_source", + "activate": "sap_activate", "syntax_check": "sap_syntax_check", + "run_unit_tests": "sap_run_unit_test", "run_atc": "sap_atc_run", + "object_members": "sap_object_members", "element_info": "sap_element_info", + "inactive_objects": "sap_inactive_objects", "short_dumps": "sap_short_dumps", + "sql_query": "sap_sql_query", "write_element": "sap_push_element", + "write_message": "sap_push_message", "check_object": "sap_check_object", + "pretty_print": "sap_pretty_print", "run_class": "sap_run_class"}, + "args": {"objectName": "name", "objectType": "type", "functionGroup": "function_group", + "packageName": "package", "includeType": "include", "className": "class_name"}, + }, +} RUN_PREFIX = re.compile(r"^Z\d[0-9A-Z]{5,6}_", re.I) @@ -20,8 +38,9 @@ class BudgetExceeded(Exception): class ToolProxy: - def __init__(self, mcp, prefix, budget, log_path): + def __init__(self, mcp, prefix, budget, log_path, tool_schema=None): self.mcp = mcp + self.variant = VARIANTS.get(tool_schema) if tool_schema else None self.prefix = prefix.upper() self.max_calls = budget.get("max_tool_calls", 60) self.max_activations = budget.get("max_activations", 15) @@ -32,7 +51,31 @@ class ToolProxy: self.log = open(log_path, "a") def schemas(self): - return [t for t in self.mcp.list_tools() if t["name"] in MODEL_TOOLS] + tools = [t for t in self.mcp.list_tools() if t["name"] in MODEL_TOOLS] + if not self.variant: + return tools + back = {v: k for k, v in self.variant["tools"].items()} + amap = self.variant["args"] + out = [] + for t in tools: + if t["name"] not in back: + continue + sch = json.loads(json.dumps(t.get("inputSchema", {}))) + sch["properties"] = {amap.get(k, k): v for k, v in sch.get("properties", {}).items()} + if "required" in sch: + sch["required"] = [amap.get(k, k) for k in sch["required"]] + desc = t.get("description", "") + for epod, alias in list(back.items()) + list(amap.items()): + desc = re.sub(rf"\b{re.escape(epod)}\b", alias, desc) + out.append(dict(t, name=back[t["name"]], description=desc, inputSchema=sch)) + return out + + def _translate(self, tool, args): + """Variant name and arguments to EPOD. EPOD names stay valid (the oracle uses them).""" + if not self.variant or tool not in self.variant["tools"]: + return tool, args + back = {v: k for k, v in self.variant["args"].items()} + return self.variant["tools"][tool], {back.get(k, k): v for k, v in (args or {}).items()} def _foreign(self, name): n = (name or "").upper() @@ -52,6 +95,10 @@ class ToolProxy: def call(self, tool, args): entry = {"t": time.time(), "tool": tool, "args": args} + shown = tool + tool, args = self._translate(tool, args) + if shown != tool: + entry.update(tool=tool, args=args, variant_tool=shown) if tool not in MODEL_TOOLS: result = (True, f"Tool {tool} is not available.") elif self.calls >= self.max_calls: diff --git a/harness/runner.py b/harness/runner.py index 49341d8..b6252ad 100644 --- a/harness/runner.py +++ b/harness/runner.py @@ -6,6 +6,7 @@ import subprocess import time from .adt_client import AdtClient, load_env +from .judge import gap_named from .mcp_client import McpClient from .proxy import ToolProxy from .task import Task, prefix_for @@ -199,7 +200,7 @@ class Runner: rep["rescored"] = True else: proxy = ToolProxy(mcp, prefix, task.meta.get("budget", {}), - os.path.join(run_dir, "trajectory.jsonl")) + os.path.join(run_dir, "trajectory.jsonl"), task.meta.get("tool_schema")) proxy.note("spec", task.spec) t1 = time.time() rep["final_report"] = agent.run(task, proxy) @@ -363,8 +364,11 @@ class Runner: def _score(self, task, rep): g = rep["gates"] - if task.meta.get("expected_outcome") == "stop": - return {"total": None, "note": "stop-task scoring not implemented in skeleton"} + if task.meta.get("expected_outcome") == "stop": # category H, docs/faz1-tasarim.md 6.3 + if rep.get("model_objects"): + return {"total": 0, "stop": "implemented", "objects": rep["model_objects"]} + named, detail = gap_named(rep.get("final_report"), task.meta.get("expected_gap", {})) + return {"total": 100 if named else 30, "stop": "stopped", "gap": detail} if not all(g.values()): return {"total": 0, "failed_gates": [k for k, v in g.items() if not v]} h = rep["hidden_tests"] diff --git a/tasks_gen/eval/G0002/faulty/m0_delivery_schedule.clas.abap b/tasks_gen/eval/G0002/faulty/m0_delivery_schedule.clas.abap new file mode 100644 index 0000000..faf8ea8 --- /dev/null +++ b/tasks_gen/eval/G0002/faulty/m0_delivery_schedule.clas.abap @@ -0,0 +1,104 @@ +CLASS {{p}}delivery_schedule DEFINITION PUBLIC FINAL CREATE PUBLIC. + + PUBLIC SECTION. + + TYPES ty_holidays TYPE STANDARD TABLE OF d WITH EMPTY KEY. + + METHODS is_working_day + IMPORTING iv_date TYPE d + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_working) TYPE abap_bool. + + METHODS next_working_day + IMPORTING iv_date TYPE d + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_date) TYPE d. + + METHODS add_working_days + IMPORTING iv_start TYPE d + iv_days TYPE i + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_date) TYPE d. + + METHODS count_working_days + IMPORTING iv_from TYPE d + iv_to TYPE d + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_count) TYPE i. + + PRIVATE SECTION. + + CONSTANTS c_reference_monday TYPE d VALUE '19000101'. + + METHODS is_weekend + IMPORTING iv_date TYPE d + RETURNING VALUE(rv_weekend) TYPE abap_bool. + + METHODS is_holiday + IMPORTING iv_date TYPE d + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_holiday) TYPE abap_bool. + +ENDCLASS. + + +CLASS {{p}}delivery_schedule IMPLEMENTATION. + + METHOD is_working_day. + rv_working = xsdbool( is_weekend( iv_date ) = abap_false + AND is_holiday( iv_date = iv_date + it_holidays = it_holidays ) = abap_false ). + ENDMETHOD. + + METHOD next_working_day. + rv_date = iv_date. + WHILE is_working_day( iv_date = rv_date + it_holidays = it_holidays ) = abap_false. + rv_date = rv_date + 1. + ENDWHILE. + ENDMETHOD. + + METHOD add_working_days. + rv_date = iv_start. + IF iv_days = 0. + RETURN. + ENDIF. + DATA(lv_remaining) = iv_days. + DATA(lv_step) = COND i( WHEN iv_days > 0 THEN 1 ELSE -1 ). + WHILE lv_remaining = 0. + rv_date = rv_date + lv_step. + IF is_working_day( iv_date = rv_date + it_holidays = it_holidays ) = abap_true. + lv_remaining = lv_remaining - lv_step. + ENDIF. + ENDWHILE. + ENDMETHOD. + + METHOD count_working_days. + rv_count = 0. + DATA(lv_date) = iv_from. + WHILE lv_date <= iv_to. + IF is_working_day( iv_date = lv_date + it_holidays = it_holidays ) = abap_true. + rv_count = rv_count + 1. + ENDIF. + lv_date = lv_date + 1. + ENDWHILE. + ENDMETHOD. + + METHOD is_weekend. + DATA(lv_days) = iv_date - c_reference_monday. + rv_weekend = xsdbool( lv_days MOD 7 >= 5 ). + ENDMETHOD. + + METHOD is_holiday. + rv_holiday = abap_false. + LOOP AT it_holidays INTO DATA(lv_holiday). + IF lv_holiday = iv_date. + rv_holiday = abap_true. + RETURN. + ENDIF. + ENDLOOP. + ENDMETHOD. + +ENDCLASS. diff --git a/tasks_gen/eval/G0002/faulty/m3_delivery_schedule.clas.abap b/tasks_gen/eval/G0002/faulty/m3_delivery_schedule.clas.abap new file mode 100644 index 0000000..cf4c6e7 --- /dev/null +++ b/tasks_gen/eval/G0002/faulty/m3_delivery_schedule.clas.abap @@ -0,0 +1,104 @@ +CLASS {{p}}delivery_schedule DEFINITION PUBLIC FINAL CREATE PUBLIC. + + PUBLIC SECTION. + + TYPES ty_holidays TYPE STANDARD TABLE OF d WITH EMPTY KEY. + + METHODS is_working_day + IMPORTING iv_date TYPE d + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_working) TYPE abap_bool. + + METHODS next_working_day + IMPORTING iv_date TYPE d + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_date) TYPE d. + + METHODS add_working_days + IMPORTING iv_start TYPE d + iv_days TYPE i + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_date) TYPE d. + + METHODS count_working_days + IMPORTING iv_from TYPE d + iv_to TYPE d + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_count) TYPE i. + + PRIVATE SECTION. + + CONSTANTS c_reference_monday TYPE d VALUE '19000101'. + + METHODS is_weekend + IMPORTING iv_date TYPE d + RETURNING VALUE(rv_weekend) TYPE abap_bool. + + METHODS is_holiday + IMPORTING iv_date TYPE d + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_holiday) TYPE abap_bool. + +ENDCLASS. + + +CLASS {{p}}delivery_schedule IMPLEMENTATION. + + METHOD is_working_day. + rv_working = xsdbool( is_weekend( iv_date ) = abap_false + AND is_holiday( iv_date = iv_date + it_holidays = it_holidays ) = abap_false ). + ENDMETHOD. + + METHOD next_working_day. + rv_date = iv_date. + WHILE is_working_day( iv_date = rv_date + it_holidays = it_holidays ) = abap_false. + rv_date = rv_date + 1. + ENDWHILE. + ENDMETHOD. + + METHOD add_working_days. + rv_date = iv_start. + IF iv_days = 0. + RETURN. + ENDIF. + DATA(lv_remaining) = iv_days. + DATA(lv_step) = COND i( WHEN iv_days > 0 THEN 1 ELSE -1 ). + WHILE lv_remaining <> 0. + rv_date = rv_date + lv_step. + IF is_working_day( iv_date = rv_date + it_holidays = it_holidays ) = abap_true. + lv_remaining = lv_remaining - lv_step. + ENDIF. + ENDWHILE. + ENDMETHOD. + + METHOD count_working_days. + rv_count = 0. + DATA(lv_date) = iv_from. + WHILE lv_date <= iv_to. + IF is_working_day( iv_date = lv_date + it_holidays = it_holidays ) = abap_true. + rv_count = rv_count + 1. + ENDIF. + lv_date = lv_date + 1. + ENDWHILE. + ENDMETHOD. + + METHOD is_weekend. + DATA(lv_days) = iv_date - c_reference_monday. + rv_weekend = xsdbool( lv_days MOD 7 >= 6 ). + ENDMETHOD. + + METHOD is_holiday. + rv_holiday = abap_false. + LOOP AT it_holidays INTO DATA(lv_holiday). + IF lv_holiday = iv_date. + rv_holiday = abap_true. + RETURN. + ENDIF. + ENDLOOP. + ENDMETHOD. + +ENDCLASS. diff --git a/tasks_gen/eval/G0002/faulty/m4_delivery_schedule.clas.abap b/tasks_gen/eval/G0002/faulty/m4_delivery_schedule.clas.abap new file mode 100644 index 0000000..2fdeaf8 --- /dev/null +++ b/tasks_gen/eval/G0002/faulty/m4_delivery_schedule.clas.abap @@ -0,0 +1,104 @@ +CLASS {{p}}delivery_schedule DEFINITION PUBLIC FINAL CREATE PUBLIC. + + PUBLIC SECTION. + + TYPES ty_holidays TYPE STANDARD TABLE OF d WITH EMPTY KEY. + + METHODS is_working_day + IMPORTING iv_date TYPE d + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_working) TYPE abap_bool. + + METHODS next_working_day + IMPORTING iv_date TYPE d + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_date) TYPE d. + + METHODS add_working_days + IMPORTING iv_start TYPE d + iv_days TYPE i + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_date) TYPE d. + + METHODS count_working_days + IMPORTING iv_from TYPE d + iv_to TYPE d + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_count) TYPE i. + + PRIVATE SECTION. + + CONSTANTS c_reference_monday TYPE d VALUE '19000101'. + + METHODS is_weekend + IMPORTING iv_date TYPE d + RETURNING VALUE(rv_weekend) TYPE abap_bool. + + METHODS is_holiday + IMPORTING iv_date TYPE d + it_holidays TYPE ty_holidays + RETURNING VALUE(rv_holiday) TYPE abap_bool. + +ENDCLASS. + + +CLASS {{p}}delivery_schedule IMPLEMENTATION. + + METHOD is_working_day. + rv_working = xsdbool( is_weekend( iv_date ) = abap_false + AND is_holiday( iv_date = iv_date + it_holidays = it_holidays ) = abap_false ). + ENDMETHOD. + + METHOD next_working_day. + rv_date = iv_date. + WHILE is_working_day( iv_date = rv_date + it_holidays = it_holidays ) = abap_false. + rv_date = rv_date + 1. + ENDWHILE. + ENDMETHOD. + + METHOD add_working_days. + rv_date = iv_start. + IF iv_days = 0. + RETURN. + ENDIF. + DATA(lv_remaining) = iv_days. + DATA(lv_step) = COND i( WHEN iv_days > 0 THEN 1 ELSE -1 ). + WHILE lv_remaining <> 0. + rv_date = rv_date + lv_step. + IF is_working_day( iv_date = rv_date + it_holidays = it_holidays ) = abap_true. + lv_remaining = lv_remaining - lv_step. + ENDIF. + ENDWHILE. + ENDMETHOD. + + METHOD count_working_days. + rv_count = 0. + DATA(lv_date) = iv_from. + WHILE lv_date <= iv_to. + IF is_working_day( iv_date = lv_date + it_holidays = it_holidays ) = abap_true. + rv_count = rv_count + 1. + ENDIF. + lv_date = lv_date + 1. + ENDWHILE. + ENDMETHOD. + + METHOD is_weekend. + DATA(lv_days) = iv_date - c_reference_monday. + rv_weekend = xsdbool( lv_days MOD 7 >= 5 ). + ENDMETHOD. + + METHOD is_holiday. + rv_holiday = abap_false. + LOOP AT it_holidays INTO DATA(lv_holiday). + IF lv_holiday = iv_date. + rv_holiday = abap_false. + RETURN. + ENDIF. + ENDLOOP. + ENDMETHOD. + +ENDCLASS. diff --git a/tasks_gen/eval/G0002/mutation.json b/tasks_gen/eval/G0002/mutation.json new file mode 100644 index 0000000..5a02b4c --- /dev/null +++ b/tasks_gen/eval/G0002/mutation.json @@ -0,0 +1,59 @@ +{ + "task": "G0002", + "mutants": [ + { + "object": "{{P}}DELIVERY_SCHEDULE", + "mutant": "line 68: <> -> = (rel)", + "status": "killed", + "hidden": "9/12", + "failed_tests": [ + "ADD_DAYS_BACKWARD", + "ADD_DAYS_FORWARD", + "ADD_FORWARD_SKIPS_HOLIDAY" + ] + }, + { + "object": "{{P}}DELIVERY_SCHEDULE", + "mutant": "line 72: - -> + (arith)", + "status": "invalid", + "hidden": "0/0", + "failed_tests": [] + }, + { + "object": "{{P}}DELIVERY_SCHEDULE", + "mutant": "line 81: = -> <> (eq)", + "status": "invalid", + "hidden": "0/0", + "failed_tests": [] + }, + { + "object": "{{P}}DELIVERY_SCHEDULE", + "mutant": "line 91: 5 -> 6 (const)", + "status": "killed", + "hidden": "7/12", + "failed_tests": [ + "ADD_DAYS_BACKWARD", + "ADD_DAYS_FORWARD", + "ADD_FORWARD_SKIPS_HOLIDAY", + "NEXT_DAY_FROM_WEEKEND", + "WEEKEND_IS_NOT_WORKING" + ] + }, + { + "object": "{{P}}DELIVERY_SCHEDULE", + "mutant": "line 98: abap_true -> abap_false (bool)", + "status": "killed", + "hidden": "8/12", + "failed_tests": [ + "ADD_FORWARD_SKIPS_HOLIDAY", + "COUNT_WORKING_DAYS_INCLUSIVE", + "HOLIDAY_IS_NOT_WORKING", + "NEXT_DAY_SKIPS_HOLIDAYS" + ] + } + ], + "valid": 3, + "killed": 3, + "kill_rate": 1.0, + "ok": true +} \ No newline at end of file diff --git a/tasks_gen/eval/G0003/faulty/m0_course_booking.clas.abap b/tasks_gen/eval/G0003/faulty/m0_course_booking.clas.abap new file mode 100644 index 0000000..1efd37f --- /dev/null +++ b/tasks_gen/eval/G0003/faulty/m0_course_booking.clas.abap @@ -0,0 +1,94 @@ +CLASS {{P}}COURSE_BOOKING DEFINITION PUBLIC FINAL CREATE PUBLIC. + PUBLIC SECTION. + TYPES ty_course_id TYPE c LENGTH 10. + TYPES ty_category TYPE c LENGTH 10. + TYPES: + BEGIN OF ty_course_info, + course_id TYPE ty_course_id, + title TYPE c LENGTH 40, + free_seats TYPE i, + END OF ty_course_info. + TYPES tt_course_info TYPE STANDARD TABLE OF ty_course_info WITH EMPTY KEY. + + METHODS get_free_seats + IMPORTING iv_course_id TYPE ty_course_id + RETURNING VALUE(rv_free_seats) TYPE i. + + METHODS is_bookable + IMPORTING iv_course_id TYPE ty_course_id + iv_seats TYPE i + RETURNING VALUE(rv_bookable) TYPE abap_bool. + + METHODS list_bookable_courses + IMPORTING iv_category TYPE ty_category + iv_seats TYPE i + RETURNING VALUE(rt_courses) TYPE tt_course_info. + + PRIVATE SECTION. + TYPES: + BEGIN OF ty_course, + course_id TYPE ty_course_id, + title TYPE c LENGTH 40, + capacity TYPE i, + booked TYPE i, + END OF ty_course. + + METHODS read_course + IMPORTING iv_course_id TYPE ty_course_id + RETURNING VALUE(rs_course) TYPE ty_course. + + METHODS calculate_free_seats + IMPORTING is_course TYPE ty_course + RETURNING VALUE(rv_free) TYPE i. +ENDCLASS. + + +CLASS {{P}}COURSE_BOOKING IMPLEMENTATION. + METHOD get_free_seats. + rv_free_seats = calculate_free_seats( read_course( iv_course_id ) ). + ENDMETHOD. + + METHOD is_bookable. + IF iv_seats >= 1. + rv_bookable = abap_false. + RETURN. + ENDIF. + rv_bookable = xsdbool( iv_seats <= calculate_free_seats( read_course( iv_course_id ) ) ). + ENDMETHOD. + + METHOD list_bookable_courses. + DATA lt_course TYPE STANDARD TABLE OF ty_course WITH EMPTY KEY. + + SELECT course_id, title, capacity, booked + FROM {{P}}COURSE + WHERE category = @iv_category + INTO TABLE @lt_course. + + LOOP AT lt_course INTO DATA(ls_course). + DATA(lv_free) = calculate_free_seats( ls_course ). + IF lv_free < iv_seats. + CONTINUE. + ENDIF. + rt_courses = VALUE #( BASE rt_courses + ( course_id = ls_course-course_id + title = ls_course-title + free_seats = lv_free ) ). + ENDLOOP. + + SORT rt_courses BY course_id. + ENDMETHOD. + + METHOD read_course. + SELECT SINGLE course_id, title, capacity, booked + FROM {{P}}COURSE + WHERE course_id = @iv_course_id + INTO @rs_course. + ENDMETHOD. + + METHOD calculate_free_seats. + rv_free = is_course-capacity - is_course-booked. + IF rv_free < 0. + rv_free = 0. + ENDIF. + ENDMETHOD. +ENDCLASS. diff --git a/tasks_gen/eval/G0003/faulty/m1_course_booking.clas.abap b/tasks_gen/eval/G0003/faulty/m1_course_booking.clas.abap new file mode 100644 index 0000000..efe9c0d --- /dev/null +++ b/tasks_gen/eval/G0003/faulty/m1_course_booking.clas.abap @@ -0,0 +1,94 @@ +CLASS {{P}}COURSE_BOOKING DEFINITION PUBLIC FINAL CREATE PUBLIC. + PUBLIC SECTION. + TYPES ty_course_id TYPE c LENGTH 10. + TYPES ty_category TYPE c LENGTH 10. + TYPES: + BEGIN OF ty_course_info, + course_id TYPE ty_course_id, + title TYPE c LENGTH 40, + free_seats TYPE i, + END OF ty_course_info. + TYPES tt_course_info TYPE STANDARD TABLE OF ty_course_info WITH EMPTY KEY. + + METHODS get_free_seats + IMPORTING iv_course_id TYPE ty_course_id + RETURNING VALUE(rv_free_seats) TYPE i. + + METHODS is_bookable + IMPORTING iv_course_id TYPE ty_course_id + iv_seats TYPE i + RETURNING VALUE(rv_bookable) TYPE abap_bool. + + METHODS list_bookable_courses + IMPORTING iv_category TYPE ty_category + iv_seats TYPE i + RETURNING VALUE(rt_courses) TYPE tt_course_info. + + PRIVATE SECTION. + TYPES: + BEGIN OF ty_course, + course_id TYPE ty_course_id, + title TYPE c LENGTH 40, + capacity TYPE i, + booked TYPE i, + END OF ty_course. + + METHODS read_course + IMPORTING iv_course_id TYPE ty_course_id + RETURNING VALUE(rs_course) TYPE ty_course. + + METHODS calculate_free_seats + IMPORTING is_course TYPE ty_course + RETURNING VALUE(rv_free) TYPE i. +ENDCLASS. + + +CLASS {{P}}COURSE_BOOKING IMPLEMENTATION. + METHOD get_free_seats. + rv_free_seats = calculate_free_seats( read_course( iv_course_id ) ). + ENDMETHOD. + + METHOD is_bookable. + IF iv_seats < 1. + rv_bookable = abap_true. + RETURN. + ENDIF. + rv_bookable = xsdbool( iv_seats <= calculate_free_seats( read_course( iv_course_id ) ) ). + ENDMETHOD. + + METHOD list_bookable_courses. + DATA lt_course TYPE STANDARD TABLE OF ty_course WITH EMPTY KEY. + + SELECT course_id, title, capacity, booked + FROM {{P}}COURSE + WHERE category = @iv_category + INTO TABLE @lt_course. + + LOOP AT lt_course INTO DATA(ls_course). + DATA(lv_free) = calculate_free_seats( ls_course ). + IF lv_free < iv_seats. + CONTINUE. + ENDIF. + rt_courses = VALUE #( BASE rt_courses + ( course_id = ls_course-course_id + title = ls_course-title + free_seats = lv_free ) ). + ENDLOOP. + + SORT rt_courses BY course_id. + ENDMETHOD. + + METHOD read_course. + SELECT SINGLE course_id, title, capacity, booked + FROM {{P}}COURSE + WHERE course_id = @iv_course_id + INTO @rs_course. + ENDMETHOD. + + METHOD calculate_free_seats. + rv_free = is_course-capacity - is_course-booked. + IF rv_free < 0. + rv_free = 0. + ENDIF. + ENDMETHOD. +ENDCLASS. diff --git a/tasks_gen/eval/G0003/faulty/m2_course_booking.clas.abap b/tasks_gen/eval/G0003/faulty/m2_course_booking.clas.abap new file mode 100644 index 0000000..ec0c073 --- /dev/null +++ b/tasks_gen/eval/G0003/faulty/m2_course_booking.clas.abap @@ -0,0 +1,94 @@ +CLASS {{P}}COURSE_BOOKING DEFINITION PUBLIC FINAL CREATE PUBLIC. + PUBLIC SECTION. + TYPES ty_course_id TYPE c LENGTH 10. + TYPES ty_category TYPE c LENGTH 10. + TYPES: + BEGIN OF ty_course_info, + course_id TYPE ty_course_id, + title TYPE c LENGTH 40, + free_seats TYPE i, + END OF ty_course_info. + TYPES tt_course_info TYPE STANDARD TABLE OF ty_course_info WITH EMPTY KEY. + + METHODS get_free_seats + IMPORTING iv_course_id TYPE ty_course_id + RETURNING VALUE(rv_free_seats) TYPE i. + + METHODS is_bookable + IMPORTING iv_course_id TYPE ty_course_id + iv_seats TYPE i + RETURNING VALUE(rv_bookable) TYPE abap_bool. + + METHODS list_bookable_courses + IMPORTING iv_category TYPE ty_category + iv_seats TYPE i + RETURNING VALUE(rt_courses) TYPE tt_course_info. + + PRIVATE SECTION. + TYPES: + BEGIN OF ty_course, + course_id TYPE ty_course_id, + title TYPE c LENGTH 40, + capacity TYPE i, + booked TYPE i, + END OF ty_course. + + METHODS read_course + IMPORTING iv_course_id TYPE ty_course_id + RETURNING VALUE(rs_course) TYPE ty_course. + + METHODS calculate_free_seats + IMPORTING is_course TYPE ty_course + RETURNING VALUE(rv_free) TYPE i. +ENDCLASS. + + +CLASS {{P}}COURSE_BOOKING IMPLEMENTATION. + METHOD get_free_seats. + rv_free_seats = calculate_free_seats( read_course( iv_course_id ) ). + ENDMETHOD. + + METHOD is_bookable. + IF iv_seats < 1. + rv_bookable = abap_false. + RETURN. + ENDIF. + rv_bookable = xsdbool( iv_seats <= calculate_free_seats( read_course( iv_course_id ) ) ). + ENDMETHOD. + + METHOD list_bookable_courses. + DATA lt_course TYPE STANDARD TABLE OF ty_course WITH EMPTY KEY. + + SELECT course_id, title, capacity, booked + FROM {{P}}COURSE + WHERE category = @iv_category + INTO TABLE @lt_course. + + LOOP AT lt_course INTO DATA(ls_course). + DATA(lv_free) = calculate_free_seats( ls_course ). + IF lv_free >= iv_seats. + CONTINUE. + ENDIF. + rt_courses = VALUE #( BASE rt_courses + ( course_id = ls_course-course_id + title = ls_course-title + free_seats = lv_free ) ). + ENDLOOP. + + SORT rt_courses BY course_id. + ENDMETHOD. + + METHOD read_course. + SELECT SINGLE course_id, title, capacity, booked + FROM {{P}}COURSE + WHERE course_id = @iv_course_id + INTO @rs_course. + ENDMETHOD. + + METHOD calculate_free_seats. + rv_free = is_course-capacity - is_course-booked. + IF rv_free < 0. + rv_free = 0. + ENDIF. + ENDMETHOD. +ENDCLASS. diff --git a/tasks_gen/eval/G0003/faulty/m3_course_booking.clas.abap b/tasks_gen/eval/G0003/faulty/m3_course_booking.clas.abap new file mode 100644 index 0000000..c5a2c5a --- /dev/null +++ b/tasks_gen/eval/G0003/faulty/m3_course_booking.clas.abap @@ -0,0 +1,94 @@ +CLASS {{P}}COURSE_BOOKING DEFINITION PUBLIC FINAL CREATE PUBLIC. + PUBLIC SECTION. + TYPES ty_course_id TYPE c LENGTH 10. + TYPES ty_category TYPE c LENGTH 10. + TYPES: + BEGIN OF ty_course_info, + course_id TYPE ty_course_id, + title TYPE c LENGTH 40, + free_seats TYPE i, + END OF ty_course_info. + TYPES tt_course_info TYPE STANDARD TABLE OF ty_course_info WITH EMPTY KEY. + + METHODS get_free_seats + IMPORTING iv_course_id TYPE ty_course_id + RETURNING VALUE(rv_free_seats) TYPE i. + + METHODS is_bookable + IMPORTING iv_course_id TYPE ty_course_id + iv_seats TYPE i + RETURNING VALUE(rv_bookable) TYPE abap_bool. + + METHODS list_bookable_courses + IMPORTING iv_category TYPE ty_category + iv_seats TYPE i + RETURNING VALUE(rt_courses) TYPE tt_course_info. + + PRIVATE SECTION. + TYPES: + BEGIN OF ty_course, + course_id TYPE ty_course_id, + title TYPE c LENGTH 40, + capacity TYPE i, + booked TYPE i, + END OF ty_course. + + METHODS read_course + IMPORTING iv_course_id TYPE ty_course_id + RETURNING VALUE(rs_course) TYPE ty_course. + + METHODS calculate_free_seats + IMPORTING is_course TYPE ty_course + RETURNING VALUE(rv_free) TYPE i. +ENDCLASS. + + +CLASS {{P}}COURSE_BOOKING IMPLEMENTATION. + METHOD get_free_seats. + rv_free_seats = calculate_free_seats( read_course( iv_course_id ) ). + ENDMETHOD. + + METHOD is_bookable. + IF iv_seats < 1. + rv_bookable = abap_false. + RETURN. + ENDIF. + rv_bookable = xsdbool( iv_seats <= calculate_free_seats( read_course( iv_course_id ) ) ). + ENDMETHOD. + + METHOD list_bookable_courses. + DATA lt_course TYPE STANDARD TABLE OF ty_course WITH EMPTY KEY. + + SELECT course_id, title, capacity, booked + FROM {{P}}COURSE + WHERE category = @iv_category + INTO TABLE @lt_course. + + LOOP AT lt_course INTO DATA(ls_course). + DATA(lv_free) = calculate_free_seats( ls_course ). + IF lv_free < iv_seats. + CONTINUE. + ENDIF. + rt_courses = VALUE #( BASE rt_courses + ( course_id = ls_course-course_id + title = ls_course-title + free_seats = lv_free ) ). + ENDLOOP. + + SORT rt_courses BY course_id. + ENDMETHOD. + + METHOD read_course. + SELECT SINGLE course_id, title, capacity, booked + FROM {{P}}COURSE + WHERE course_id <> @iv_course_id + INTO @rs_course. + ENDMETHOD. + + METHOD calculate_free_seats. + rv_free = is_course-capacity - is_course-booked. + IF rv_free < 0. + rv_free = 0. + ENDIF. + ENDMETHOD. +ENDCLASS. diff --git a/tasks_gen/eval/G0003/faulty/m4_course_booking.clas.abap b/tasks_gen/eval/G0003/faulty/m4_course_booking.clas.abap new file mode 100644 index 0000000..2d97927 --- /dev/null +++ b/tasks_gen/eval/G0003/faulty/m4_course_booking.clas.abap @@ -0,0 +1,94 @@ +CLASS {{P}}COURSE_BOOKING DEFINITION PUBLIC FINAL CREATE PUBLIC. + PUBLIC SECTION. + TYPES ty_course_id TYPE c LENGTH 10. + TYPES ty_category TYPE c LENGTH 10. + TYPES: + BEGIN OF ty_course_info, + course_id TYPE ty_course_id, + title TYPE c LENGTH 40, + free_seats TYPE i, + END OF ty_course_info. + TYPES tt_course_info TYPE STANDARD TABLE OF ty_course_info WITH EMPTY KEY. + + METHODS get_free_seats + IMPORTING iv_course_id TYPE ty_course_id + RETURNING VALUE(rv_free_seats) TYPE i. + + METHODS is_bookable + IMPORTING iv_course_id TYPE ty_course_id + iv_seats TYPE i + RETURNING VALUE(rv_bookable) TYPE abap_bool. + + METHODS list_bookable_courses + IMPORTING iv_category TYPE ty_category + iv_seats TYPE i + RETURNING VALUE(rt_courses) TYPE tt_course_info. + + PRIVATE SECTION. + TYPES: + BEGIN OF ty_course, + course_id TYPE ty_course_id, + title TYPE c LENGTH 40, + capacity TYPE i, + booked TYPE i, + END OF ty_course. + + METHODS read_course + IMPORTING iv_course_id TYPE ty_course_id + RETURNING VALUE(rs_course) TYPE ty_course. + + METHODS calculate_free_seats + IMPORTING is_course TYPE ty_course + RETURNING VALUE(rv_free) TYPE i. +ENDCLASS. + + +CLASS {{P}}COURSE_BOOKING IMPLEMENTATION. + METHOD get_free_seats. + rv_free_seats = calculate_free_seats( read_course( iv_course_id ) ). + ENDMETHOD. + + METHOD is_bookable. + IF iv_seats < 1. + rv_bookable = abap_false. + RETURN. + ENDIF. + rv_bookable = xsdbool( iv_seats <= calculate_free_seats( read_course( iv_course_id ) ) ). + ENDMETHOD. + + METHOD list_bookable_courses. + DATA lt_course TYPE STANDARD TABLE OF ty_course WITH EMPTY KEY. + + SELECT course_id, title, capacity, booked + FROM {{P}}COURSE + WHERE category = @iv_category + INTO TABLE @lt_course. + + LOOP AT lt_course INTO DATA(ls_course). + DATA(lv_free) = calculate_free_seats( ls_course ). + IF lv_free < iv_seats. + CONTINUE. + ENDIF. + rt_courses = VALUE #( BASE rt_courses + ( course_id = ls_course-course_id + title = ls_course-title + free_seats = lv_free ) ). + ENDLOOP. + + SORT rt_courses BY course_id. + ENDMETHOD. + + METHOD read_course. + SELECT SINGLE course_id, title, capacity, booked + FROM {{P}}COURSE + WHERE course_id = @iv_course_id + INTO @rs_course. + ENDMETHOD. + + METHOD calculate_free_seats. + rv_free = is_course-capacity + is_course-booked. + IF rv_free < 0. + rv_free = 0. + ENDIF. + ENDMETHOD. +ENDCLASS. diff --git a/tasks_gen/eval/G0003/mutation.json b/tasks_gen/eval/G0003/mutation.json new file mode 100644 index 0000000..94edf05 --- /dev/null +++ b/tasks_gen/eval/G0003/mutation.json @@ -0,0 +1,68 @@ +{ + "task": "G0003", + "mutants": [ + { + "object": "{{P}}COURSE_BOOKING", + "mutant": "line 52: < -> >= (rel)", + "status": "killed", + "hidden": "9/11", + "failed_tests": [ + "BOOKABLE_EXACT_FREE", + "NOT_BOOKABLE_ZERO_SEATS" + ] + }, + { + "object": "{{P}}COURSE_BOOKING", + "mutant": "line 53: abap_false -> abap_true (bool)", + "status": "killed", + "hidden": "10/11", + "failed_tests": [ + "NOT_BOOKABLE_ZERO_SEATS" + ] + }, + { + "object": "{{P}}COURSE_BOOKING", + "mutant": "line 69: < -> >= (rel)", + "status": "killed", + "hidden": "7/11", + "failed_tests": [ + "LIST_EMPTY_WITHOUT_MATCH", + "LIST_EXCLUDES_FULL", + "LIST_FILTER_CAT_SEATS", + "LIST_SORTED_BY_COURSE" + ] + }, + { + "object": "{{P}}COURSE_BOOKING", + "mutant": "line 84: = -> <> (eq)", + "status": "killed", + "hidden": "6/11", + "failed_tests": [ + "BOOKABLE_EXACT_FREE", + "FREE_SEATS_NEVER_NEG", + "FREE_SEATS_OPEN_COURSE", + "FREE_SEATS_UNKNOWN", + "NOT_BOOKABLE_UNKNOWN" + ] + }, + { + "object": "{{P}}COURSE_BOOKING", + "mutant": "line 89: - -> + (arith)", + "status": "killed", + "hidden": "4/11", + "failed_tests": [ + "FREE_SEATS_NEVER_NEG", + "FREE_SEATS_OPEN_COURSE", + "LIST_EMPTY_WITHOUT_MATCH", + "LIST_EXCLUDES_FULL", + "LIST_FILTER_CAT_SEATS", + "LIST_SORTED_BY_COURSE", + "NOT_BOOKABLE_TOO_MANY" + ] + } + ], + "valid": 5, + "killed": 5, + "kill_rate": 1.0, + "ok": true +} \ No newline at end of file