Files
abap-llm/harness/judge.py
Kral 0ba25faeea Mutation in generator; category H (stop) scoring and judge; category K variants and tool schema variant; evalset plan
- mutation.py: negation mutants, skip WHILE/DO blocks (endless loop blocked RFC ~8 min)
- generator: accept only after mutation check; survivors go back as repair feedback
- runner/judge.py: stop tasks scored 100/30/0; gap by keywords, else judge model
- proxy: tool schema variant generic_v0 (draft); generator make_k_variant
- evalset.py: 88 slots for step F (A-I, H, K)

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-03 05:37:20 +02:00

65 lines
2.7 KiB
Python

"""Stop tasks (category H): did the model name the right gap?
1. Keywords: task.json "expected_gap": {"description": "...", "keywords": [["discount", "rate"], ...]}.
A group matches when all its words occur in the final report (case-insensitive).
2. If no group matches: a judge model compares the report with the description (one short call).
"""
import json
import os
import re
import urllib.request
from .ledger import add_usage, check_budget
JUDGE_MODEL = os.environ.get("JUDGE_MODEL", "deepseek-v4.1-flash:cloud")
JUDGE_PROMPT = """You check the report of an ABAP developer model. The task had a gap: a missing or
contradictory business rule. The model had to stop and name this gap.
Expected gap:
{gap}
Report of the model:
{report}
Does the report name the expected gap (the same missing or contradictory rule)? A report that names
only other, unrelated points does not name it. Answer with one JSON object and nothing else:
{{"names_gap": true or false, "reason": "<one sentence>"}}"""
def keyword_hit(report, groups):
text = (report or "").lower()
for g in groups or []:
if all(re.search(rf"\b{re.escape(w.lower())}", text) for w in g):
return g
return None
def judge(report, gap_description, base_url=None):
check_budget()
base_url = base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")
body = {"model": JUDGE_MODEL, "temperature": 0, "max_tokens": 4000,
"messages": [{"role": "user", "content": JUDGE_PROMPT.format(gap=gap_description,
report=(report or "")[:6000])}]}
req = urllib.request.Request(f"{base_url}/chat/completions", json.dumps(body).encode(),
{"Content-Type": "application/json", "Authorization": "Bearer none"})
with urllib.request.urlopen(req, timeout=600) as r:
data = json.loads(r.read().decode())
add_usage(JUDGE_MODEL, data.get("usage", {}), kind="judge")
text = data["choices"][0]["message"].get("content") or ""
m = re.search(r"\{.*\}", text, re.S)
try:
return json.loads(m.group(0)) if m else {"names_gap": False, "reason": "no JSON: " + text[:200]}
except ValueError:
return {"names_gap": False, "reason": "bad JSON: " + text[:200]}
def gap_named(report, expected_gap):
"""Returns (named, detail)."""
if not (report or "").strip() or report.strip() == "No action.":
return False, {"how": "empty report"}
g = keyword_hit(report, expected_gap.get("keywords"))
if g:
return True, {"how": "keywords", "group": g}
v = judge(report, expected_gap.get("description", ""))
return bool(v.get("names_gap")), {"how": "judge", "model": JUDGE_MODEL, **v}