Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat
67 lines
2.9 KiB
Python
67 lines
2.9 KiB
Python
"""Stop tasks (category H): did the model name the right gap?
|
|
|
|
1. Keywords: task.json "expected_gap": {"description": "...", "keywords": [["discount", "rate"], ...]}.
|
|
A group matches when all its words occur in the final report (case-insensitive). Fast path only
|
|
when all groups match.
|
|
2. Otherwise: a judge model compares the report with the description (one short call).
|
|
"""
|
|
import json
|
|
import os
|
|
import re
|
|
import urllib.request
|
|
|
|
from .ledger import add_usage, check_budget
|
|
|
|
JUDGE_MODEL = os.environ.get("JUDGE_MODEL", "deepseek-v4.1-flash:cloud")
|
|
JUDGE_PROMPT = """You check the report of an ABAP developer model. The task had a gap: a missing or
|
|
contradictory business rule. The model had to stop and name this gap.
|
|
|
|
Expected gap:
|
|
{gap}
|
|
|
|
Report of the model:
|
|
{report}
|
|
|
|
Does the report name the expected gap (the same missing or contradictory rule)? A report that names
|
|
only other, unrelated points does not name it. Answer with one JSON object and nothing else:
|
|
{{"names_gap": true or false, "reason": "<one sentence>"}}"""
|
|
|
|
|
|
def keyword_hit(report, groups):
|
|
"""All groups must match. One group alone is too loose (G0168: "volume discount" also occurs in a
|
|
report that stops for another reason)."""
|
|
text = (report or "").lower()
|
|
if groups and all(all(re.search(rf"\b{re.escape(w.lower())}", text) for w in g) for g in groups):
|
|
return groups
|
|
return None
|
|
|
|
|
|
def judge(report, gap_description, base_url=None):
|
|
check_budget()
|
|
base_url = base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")
|
|
body = {"model": JUDGE_MODEL, "temperature": 0, "max_tokens": 4000,
|
|
"messages": [{"role": "user", "content": JUDGE_PROMPT.format(gap=gap_description,
|
|
report=(report or "")[:6000])}]}
|
|
req = urllib.request.Request(f"{base_url}/chat/completions", json.dumps(body).encode(),
|
|
{"Content-Type": "application/json", "Authorization": "Bearer none"})
|
|
with urllib.request.urlopen(req, timeout=600) as r:
|
|
data = json.loads(r.read().decode())
|
|
add_usage(JUDGE_MODEL, data.get("usage", {}), kind="judge")
|
|
text = data["choices"][0]["message"].get("content") or ""
|
|
m = re.search(r"\{.*\}", text, re.S)
|
|
try:
|
|
return json.loads(m.group(0)) if m else {"names_gap": False, "reason": "no JSON: " + text[:200]}
|
|
except ValueError:
|
|
return {"names_gap": False, "reason": "bad JSON: " + text[:200]}
|
|
|
|
|
|
def gap_named(report, expected_gap):
|
|
"""Returns (named, detail)."""
|
|
if not (report or "").strip() or report.strip() == "No action.":
|
|
return False, {"how": "empty report"}
|
|
g = keyword_hit(report, expected_gap.get("keywords"))
|
|
if g:
|
|
return True, {"how": "keywords", "group": g}
|
|
v = judge(report, expected_gap.get("description", ""))
|
|
return bool(v.get("names_gap")), {"how": "judge", "model": JUDGE_MODEL, **v}
|