Mutation in generator; category H (stop) scoring and judge; category K variants and tool schema variant; evalset plan
- mutation.py: negation mutants, skip WHILE/DO blocks (endless loop blocked RFC ~8 min) - generator: accept only after mutation check; survivors go back as repair feedback - runner/judge.py: stop tasks scored 100/30/0; gap by keywords, else judge model - proxy: tool schema variant generic_v0 (draft); generator make_k_variant - evalset.py: 88 slots for step F (A-I, H, K) Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
64
harness/judge.py
Normal file
64
harness/judge.py
Normal file
@@ -0,0 +1,64 @@
|
||||
"""Stop tasks (category H): did the model name the right gap?
|
||||
|
||||
1. Keywords: task.json "expected_gap": {"description": "...", "keywords": [["discount", "rate"], ...]}.
|
||||
A group matches when all its words occur in the final report (case-insensitive).
|
||||
2. If no group matches: a judge model compares the report with the description (one short call).
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import urllib.request
|
||||
|
||||
from .ledger import add_usage, check_budget
|
||||
|
||||
JUDGE_MODEL = os.environ.get("JUDGE_MODEL", "deepseek-v4.1-flash:cloud")
|
||||
JUDGE_PROMPT = """You check the report of an ABAP developer model. The task had a gap: a missing or
|
||||
contradictory business rule. The model had to stop and name this gap.
|
||||
|
||||
Expected gap:
|
||||
{gap}
|
||||
|
||||
Report of the model:
|
||||
{report}
|
||||
|
||||
Does the report name the expected gap (the same missing or contradictory rule)? A report that names
|
||||
only other, unrelated points does not name it. Answer with one JSON object and nothing else:
|
||||
{{"names_gap": true or false, "reason": "<one sentence>"}}"""
|
||||
|
||||
|
||||
def keyword_hit(report, groups):
|
||||
text = (report or "").lower()
|
||||
for g in groups or []:
|
||||
if all(re.search(rf"\b{re.escape(w.lower())}", text) for w in g):
|
||||
return g
|
||||
return None
|
||||
|
||||
|
||||
def judge(report, gap_description, base_url=None):
|
||||
check_budget()
|
||||
base_url = base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")
|
||||
body = {"model": JUDGE_MODEL, "temperature": 0, "max_tokens": 4000,
|
||||
"messages": [{"role": "user", "content": JUDGE_PROMPT.format(gap=gap_description,
|
||||
report=(report or "")[:6000])}]}
|
||||
req = urllib.request.Request(f"{base_url}/chat/completions", json.dumps(body).encode(),
|
||||
{"Content-Type": "application/json", "Authorization": "Bearer none"})
|
||||
with urllib.request.urlopen(req, timeout=600) as r:
|
||||
data = json.loads(r.read().decode())
|
||||
add_usage(JUDGE_MODEL, data.get("usage", {}), kind="judge")
|
||||
text = data["choices"][0]["message"].get("content") or ""
|
||||
m = re.search(r"\{.*\}", text, re.S)
|
||||
try:
|
||||
return json.loads(m.group(0)) if m else {"names_gap": False, "reason": "no JSON: " + text[:200]}
|
||||
except ValueError:
|
||||
return {"names_gap": False, "reason": "bad JSON: " + text[:200]}
|
||||
|
||||
|
||||
def gap_named(report, expected_gap):
|
||||
"""Returns (named, detail)."""
|
||||
if not (report or "").strip() or report.strip() == "No action.":
|
||||
return False, {"how": "empty report"}
|
||||
g = keyword_hit(report, expected_gap.get("keywords"))
|
||||
if g:
|
||||
return True, {"how": "keywords", "group": g}
|
||||
v = judge(report, expected_gap.get("description", ""))
|
||||
return bool(v.get("names_gap")), {"how": "judge", "model": JUDGE_MODEL, **v}
|
||||
Reference in New Issue
Block a user