From 542fc8ec31832584d72d44e44ffc88a15a6988dd Mon Sep 17 00:00:00 2001 From: Kral Date: Tue, 6 Oct 2026 05:56:23 +0200 Subject: [PATCH] B: stage 2 data analysis (repair taxonomy, teacher vs Qwen behaviors, duplicates, empty_response); empty_response fix (cap 24000, retry temperature, stream guard) Co-Authored-By: Claude Sonnet 5.5 --- docs/data-analysis-stage2.md | 55 ++++++++ docs/empty-response.md | 23 ++++ harness/agents.py | 64 +++++++++- harness/dashboard.py | 18 ++- harness/trajectories.py | 19 ++- train/analyze.py | 236 +++++++++++++++++++++++++++++++++++ 6 files changed, 406 insertions(+), 9 deletions(-) create mode 100644 docs/data-analysis-stage2.md create mode 100644 docs/empty-response.md create mode 100644 train/analyze.py diff --git a/docs/data-analysis-stage2.md b/docs/data-analysis-stage2.md new file mode 100644 index 0000000..990fe5f --- /dev/null +++ b/docs/data-analysis-stage2.md @@ -0,0 +1,55 @@ +# Stage 2 data analysis (accepted DeepSeek trajectories), 2026-10-06 + +Script `train/analyze.py`, numbers in `runs/analysis/analysis.json`. Base: 90 accepted trajectories of 119 runs (acceptance 76 %). + +## 1. Repair taxonomy: are the three step-3 errors covered? + +133 failed writes occurred inside accepted trajectories; each of the three common errors appears and is repaired: + +| error (roadmap step 3) | failed writes | trajectories | repaired in the same trajectory | example | +|---|---|---|---|---| +| `TYPE c LENGTH n` in a method signature | 10 | 3 | 3 | Unable to interpret "4". Possible causes of error include incorrect spellings or comma errors. | +| reserved word as a field or parameter name | 18 | 16 | 16 | Field "VALUE" is unknown. | +| name longer than 30 characters | 25 | 23 | 23 | The name "SORTS_BY_TOTAL_DESC_THEN_VARIETY" is longer than the allowed 30 characters. | + +Reading: the long names (23 trajectories) and the reserved words (16, by a text match: the label is generous, it also counts "Field VALUE is unknown") are well covered. +**`TYPE c LENGTH` in a signature is thin: 3 trajectories** (10 failed writes, all repaired). The SAP message is only "save operation failed" or, with the proxy hint, +`Statement does not exist ... METHODS`. This is the error that the proxy syntaxCheck exists for; after the reset the error-targeted slots ("named-type") should be +run first for CLAS and FUNC to get more of these. Other frequent errors (top list in the json): the testclasses include is not reachable with `sap_push_element` (9 times), +"save operation failed" without detail (30 with the hint), "Field X is unknown", "Test method can be defined only in test classes". + +## 2. Behaviors that Qwen lacks (series A, all 20 runs) versus the teacher (accepted trajectories) + +| behavior | DeepSeek (accepted) | Qwen series A | +|---|---|---| +| runs with at least one failed write | 60 of 90 | 14 of 20 | +| **repair after the first error** (a different source that is written successfully) | **59 of 60 (98 %)** | **6 of 14 (43 %)** | +| median calls from the first error to the repair | 1 | 2.0 | +| searches/reads before the first write (median / p90) | 3.0 / 19 | 4.0 / 15 | +| longest series of `sap_search_object` calls | 7 | **61** | +| runs with the same source pushed again | 2 of 90 | **11 of 20** | +| runs without any write | 4 | 4 | + +The teacher repairs after the first error in almost every case, one call later. It writes after a median of 3 reads. The longest search run of the teacher is 7, of Qwen 61. +These are exactly the two behaviors the SFT must teach; both are in the data (repair share 60 to 65 % of accepted trajectories). +Note: the Qwen figure includes failed runs, the teacher figure only accepted ones (a teacher run that never repaired is not accepted), so part of the gap is selection. +For a fair view the rejected teacher runs would be added; they are in `runs/traj/` (30 runs), not analysed here. + +## 3. Near duplicates + +Pairwise comparison of the code that each accepted trajectory wrote (5-word shingles, prefix removed) and of the final reports: **no pair above 0.8 (code) or 0.85 (report)**, also none +between the two trajectories of one task. The data is not repetitive at this level. + +## 4. empty_response (13 of 119 runs, 11 %) + +By kind: {'CLAS': 2, 'DDLS': 6, 'PROG': 3, 'TABL': 2} (CDS 6 of 13). The last call before the empty turn was a read in 10 of 13 cases ({'sap_push_source': 1, 'sap_syntax_check': 1, 'sap_pull_source': 4, 'sap_object_structure': 1, 'sap_search_object': 1, 'sap_sql_query': 4, 'sap_check_object': 1}), with small results (50 to 6000 characters) and +contexts from 13k to 63k tokens: it is not a big tool result and not the context size. In every case the turn used the whole output limit (32000 tokens) on reasoning with no content and no tool call; +in 7 of 13 runs two or three turns in a row did that (the retry with the same prompt and temperature reproduces it). Over all 1997 turns: p50 1109 tokens, p90 7862, +the legitimate long turns (accepted runs, content produced) reach up to 30078; only 3 legitimate turns were longer than 24000, 12 longer than 20000, 22 longer than 16000. +Fix: `docs/empty-response.md`. + +## 5. What it means for the data + +- Keep the repair behavior and the "write soon" behavior: they are present. Weight the error-targeted slots (named-type) up after the reset. +- Do not rely on `TYPE c LENGTH` repairs: only 3 examples. +- The DDLS share of empty runs (6 of 13) explains much of the CDS rejection. diff --git a/docs/empty-response.md b/docs/empty-response.md new file mode 100644 index 0000000..93ea526 --- /dev/null +++ b/docs/empty-response.md @@ -0,0 +1,23 @@ +# empty_response fix (2026-10-06) + +**Problem.** 13 of 119 DeepSeek runs (11 %) ended with `empty_response`: one turn used the whole output limit (32000 tokens) for reasoning, +with no content and no tool call, and the retry (same prompt, same temperature, up to two) did it again in 7 of 13 runs. Each such run costs up to +3 x 32000 output tokens and is lost (DDLS 6, PROG 3, CLAS 2, TABL 2). Analysis: `docs/data-analysis-stage2.md`, section 4. + +**What the data says.** The empty turn comes after a small read result (50 to 6000 characters), in contexts of 13k to 63k tokens, not after a large result. Turns +over all runs: p50 1.1k, p90 7.9k tokens; legitimate long turns (large test classes) reach 30k; only 3 of 1997 turns longer than 24k were legitimate. + +**Implemented (harness side only, `harness/agents.py`, `harness/trajectories.py`)** +1. Output cap per turn **24000** (was 32000; saves a quarter of every runaway turn, costs 3 of 1997 legitimate turns). +2. **One retry at temperature 0.8** (was two retries at 0.2): another sample instead of the same runaway. +3. **Stream guard** (off until verified): the request is streamed; when a turn has produced only reasoning for `STREAM_GUARD` tokens (suggested 9000) and no content and + no tool call, the stream is cut and the turn counts as empty (retry as in 2). Saves most of the cost of a runaway turn. Tested with a fake streaming server (text, tool call, + runaway: cut after 1500 estimated tokens in 0.01 s). **Not tested on the cloud model**: it needs the Ollama cloud stream to carry the reasoning in `delta.reasoning` + (or `reasoning_content` / `thinking`) and tool calls as streamed deltas; if the stream cannot be parsed the agent falls back to the normal request. Usage of a cut turn is estimated + (characters / 3.2) and marked `estimated` in the record. +4. Not implemented: "one object per write call" rule and splitting large test classes. Both change what the model sees (system prompt or task) and so the training distribution and + the comparison with the baseline (same system prompt). Option if 1 to 3 are not enough: a hint in the task spec of CDS tasks only. + +**Verify after the reset (5 runs, DDLS and PROG first because they had most empty runs).** Run with `STREAM_GUARD=9000 python3 -m harness.pipeline` or set the constant: +check that `empty_response` stays below 5 % over 40 runs, that no accepted run was cut wrongly (`cut_by_stream_guard` in `metadata.turn_usage` followed by a good turn is fine), +and that the cost per run does not rise. If the stream breaks (tool calls missing), unset `STREAM_GUARD`: items 1 and 2 stay. diff --git a/harness/agents.py b/harness/agents.py index e92dbc6..cfd257e 100644 --- a/harness/agents.py +++ b/harness/agents.py @@ -84,7 +84,7 @@ class LlmAgent: def __init__(self, model, base_url=None, api_key=None, max_turns=80, temperature=0.2, max_seconds=None, max_tokens=None, chat_template_kwargs=None, loop_guard=None, - deadline=None, watch=False): + deadline=None, watch=False, empty_retries=2, retry_temperature=None, stream_guard=None): self.model = model self.name = f"llm:{model}" self.base_url = (base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")).rstrip("/") @@ -96,7 +96,12 @@ class LlmAgent: # Runaway reasoning: 35 of 1965 DeepSeek turns produced 393k output tokens and no content (46 % of the # run cost, 2026-10-03). Normal turns: p95 17.5k, max 82k. A cut turn is retried (see run). self.max_tokens = max_tokens or (32000 if ":cloud" in (model or "") else None) - self.empty_retries = 2 + self.empty_retries = empty_retries + # empty_response fix (2026-10-06): a retry after an empty turn can use another temperature (the same sample often + # runs away again: 7 of 13 empty runs had two or three capped turns in a row); stream_guard = reasoning tokens + # after which a streamed turn without any content or tool call is cut and counted as an empty turn + self.retry_temperature = retry_temperature + self.stream_guard = stream_guard self.loop_guard = loop_guard # end the run after this many identical pushes in a row (None = off) self.deadline = deadline # absolute time (time.time()) of the hard stop, or None self.watch = watch # remote model: ping the server during a request, end the run when it is gone @@ -104,9 +109,9 @@ class LlmAgent: self.messages, self.tools, self.reasoning, self.turn_usage = [], [], [], [] # for the trajectory record self.chat_template_kwargs = chat_template_kwargs # local server only, e.g. {"enable_thinking": False} - def _chat(self, messages, tools): + def _chat(self, messages, tools, temperature=None): body = {"model": self.model, "messages": messages, "tools": tools, - "temperature": self.temperature, "parallel_tool_calls": False} + "temperature": temperature if temperature is not None else self.temperature, "parallel_tool_calls": False} if self.max_tokens: body["max_tokens"] = self.max_tokens if self.chat_template_kwargs: @@ -115,6 +120,11 @@ class LlmAgent: {"Content-Type": "application/json", "Authorization": f"Bearer {self.api_key}"}) last = None + if self.stream_guard and not (self.watch or self.deadline): + try: + return self._chat_stream(dict(body), messages) + except Exception as e: # noqa: BLE001 streaming not usable (server, parse): the normal request below + last = e for attempt in range(4): # model server errors (HTTP 5xx, timeouts): retry with backoff try: if self.watch or self.deadline: @@ -137,6 +147,49 @@ class LlmAgent: time.sleep(10 * (attempt + 1)) raise RuntimeError(f"model request failed after retries: {last}") + def _chat_stream(self, body, messages): + """Streamed request. A turn that has produced only reasoning for `stream_guard` tokens (about 3.2 characters per + token) and no content and no tool call is cut and returned as an empty turn (the run loop retries it).""" + body["stream"] = True + body["stream_options"] = {"include_usage": True} + req = urllib.request.Request(f"{self.base_url}/chat/completions", json.dumps(body).encode(), + {"Content-Type": "application/json", "Authorization": f"Bearer {self.api_key}"}) + content, reasoning, calls, usage, cut = "", 0, {}, {}, False + with urllib.request.urlopen(req, timeout=self.request_timeout) as r: + for raw in r: + line = raw.decode("utf-8", "replace").strip() + if not line.startswith("data:"): + continue + data = line[5:].strip() + if data == "[DONE]": + break + chunk = json.loads(data) + if chunk.get("usage"): + usage = chunk["usage"] + for ch in chunk.get("choices") or []: + d = ch.get("delta") or {} + content += d.get("content") or "" + reasoning += len(d.get("reasoning") or d.get("reasoning_content") or d.get("thinking") or "") + for tc in d.get("tool_calls") or []: + c = calls.setdefault(tc.get("index", 0), {"id": tc.get("id") or "", "type": "function", + "function": {"name": "", "arguments": ""}}) + c["id"] = c["id"] or tc.get("id") or "" + f = tc.get("function") or {} + c["function"]["name"] += f.get("name") or "" + a = f.get("arguments") + c["function"]["arguments"] += a if isinstance(a, str) else json.dumps(a) if a else "" + if not content.strip() and not calls and reasoning / 3.2 >= self.stream_guard: + cut = True + break + if cut: + est = {"prompt_tokens": int(len(json.dumps(messages)) / 3.5), "completion_tokens": int(reasoning / 3.2), + "estimated": True, "cut_by_stream_guard": True} + return {"role": "assistant", "content": ""}, est + msg = {"role": "assistant", "content": content or None} + if calls: + msg["tool_calls"] = [calls[k] for k in sorted(calls)] + return msg, usage + def _ping(self, timeout=10): try: urllib.request.urlopen(urllib.request.Request(f"{self.base_url}/models"), timeout=timeout).read() @@ -195,7 +248,8 @@ class LlmAgent: self.end_reason = "time_budget" break try: - msg, usage = self._chat(messages, tools) + msg, usage = self._chat(messages, tools, + self.retry_temperature if (empty and self.retry_temperature) else None) add_usage(self.model, usage, kind="run", ref=proxy.prefix) self.turn_usage.append(usage) except WindowEnd: diff --git a/harness/dashboard.py b/harness/dashboard.py index 2ab1bc2..496a393 100644 --- a/harness/dashboard.py +++ b/harness/dashboard.py @@ -302,6 +302,22 @@ def local_card(): return hdr + tbl +def work_card(): + """Progress of the no-cloud work packages (runs/dashboard/work.json, updated by Claude after each item).""" + path = os.path.join(OUT_DIR, "work.json") + if not os.path.exists(path): + return "" + try: + w = json.load(open(path)) + except ValueError: + return "" + cls = {"done": "ok", "in progress": "wa", "waiting": "gr", "parked": "gr", "blocked": "er"} + rows = "".join("%s%s%s%s" % ( + E(i["id"]), E(i["text"]), cls.get(i["status"], "gr"), E(i["status"]), E(i.get("note", ""))) for i in w.get("items", [])) + return ("

%s

%s
itemworkstatusresult
" + "
updated %s
" % (E(w.get("title", "")), rows, time.strftime("%H:%M", time.localtime(w.get("updated", 0))))) + + def mix_table(rows, evs): kt = mix.accepted_task_counts() kr = {} @@ -437,7 +453,7 @@ def render(d): % ("ok" if d["a4h"] else "er", "up" if d["a4h"] else "down", "ok" if d["mcp"] else "er", "up" if d["mcp"] else "down", scls, status, E("\n".join(d["procs"])))) logc = "

Pipeline log

%s
" % E("\n".join(d["log_tail"])) - body = (banner + "
" + tiles + "
" + local_card() + budget + prog + stops + sysc + body = (banner + "
" + tiles + "
" + work_card() + local_card() + budget + prog + stops + sysc + mix_table(rows, evs) + agg_table("Trajectories by category", evs, rows, "category") + agg_table("Trajectories by object type", evs, rows, "object_type") + gen_table(logs, "category", "Task generation by category") + gen_table(logs, "object_type", "Task generation by object type") + tok + recent + logc + "
") diff --git a/harness/trajectories.py b/harness/trajectories.py index 231ecdc..0268897 100644 --- a/harness/trajectories.py +++ b/harness/trajectories.py @@ -28,7 +28,20 @@ POOL = os.path.join(ROOT, "tasks_gen", "train") OUT = os.path.join(ROOT, "runs", "traj") RUN_BASE = 200000 # 200000 + (task number - 1000) * 3 + attempt (a digit must lead the 4-char base36 run: < 466560) MODEL = "deepseek-v4.1-flash:cloud" -CDS_CALLS = 100 # tool-call budget for tasks with a CDS contract object (eval keeps 60; Kral 2026-10-05) +# empty_response fix (2026-10-06, docs/empty-response.md): 13 of 119 runs ended with one turn that used the whole output limit on +# reasoning. Cap per turn 24000 (only 3 of 1997 turns were legitimately longer), one retry at another temperature (the same +# sample ran away again in 7 of 13 runs), and the stream guard (a streamed turn with only reasoning is cut after that many reasoning +# tokens) which stays off until it is verified on the cloud model (env STREAM_GUARD, for example 9000). +TEACHER_MAX_TOKENS = 24000 +EMPTY_RETRIES = 1 +RETRY_TEMPERATURE = 0.8 +STREAM_GUARD = int(os.environ["STREAM_GUARD"]) if os.environ.get("STREAM_GUARD") else None +CDS_CALLS = 100 + + +def new_agent(): + return LlmAgent(MODEL, loop_guard=3, max_tokens=TEACHER_MAX_TOKENS, empty_retries=EMPTY_RETRIES, + retry_temperature=RETRY_TEMPERATURE, stream_guard=STREAM_GUARD) # tool-call budget for tasks with a CDS contract object (eval keeps 60; Kral 2026-10-05) LOCK = threading.Lock() @@ -74,7 +87,7 @@ def one(task_id, attempt, stop): print("BUDGET", e, flush=True) return run_no = RUN_BASE + (int("".join(c for c in task_id if c.isdigit())) - 1000) * 3 + attempt - agent = LlmAgent(MODEL, loop_guard=3) + agent = new_agent() runner = Runner(POOL, OUT) t0 = time.time() row = {"task": task_id, "attempt": attempt, "run": run_no, "model": MODEL} @@ -97,7 +110,7 @@ def one(task_id, attempt, stop): for d in glob.glob(os.path.join(OUT, f"{run_no}_{task_id}_*")): os.makedirs(os.path.join(OUT, "_aborted"), exist_ok=True) os.rename(d, os.path.join(OUT, "_aborted", os.path.basename(d) + "_" + str(int(time.time())))) - agent = LlmAgent(MODEL, loop_guard=3) + agent = new_agent() score = (rep.get("score") or {}).get("total") rec = os.path.exists(os.path.join(run_dir, "record.json")) row.update(score=score, setup_failed=bool(rep.get("setup_failed")), end_reason=rep.get("end_reason"), diff --git a/train/analyze.py b/train/analyze.py new file mode 100644 index 0000000..b78a23f --- /dev/null +++ b/train/analyze.py @@ -0,0 +1,236 @@ +"""Analysis of the accepted DeepSeek trajectories (Opus item B, 2026-10-06): repair taxonomy, near duplicates, +empty_response, and the behaviors the teacher shows and Qwen lacks. + + python3 train/analyze.py writes runs/analysis/analysis.json and prints the numbers +""" +import glob +import json +import os +import re +import statistics +import sys +from collections import Counter + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, ROOT) +sys.path.insert(0, os.path.join(ROOT, "train")) +import accept as acc # noqa: E402 +from harness import mix # noqa: E402 +from harness.generator import RESERVED # noqa: E402 + +WRITE = ("sap_push_source", "sap_push_element", "sap_push_message", "sap_activate", "sap_create_object") +SEARCH_LIKE = ("sap_search_object", "sap_pull_source", "sap_object_members", "sap_object_structure", "sap_element_info", + "sap_usage_references", "sap_sql_query", "sap_inactive_objects") +SIG = re.compile(r"(?:IMPORTING|EXPORTING|RETURNING|CHANGING)[^.]*?\bTYPE\s+(?:c|n|p|x)\s+LENGTH\b|\bVALUE\([^)]*\)\s+TYPE\s+(?:c|n|p|x)\s+LENGTH\b", re.I | re.S) + + +def pct(v, p): + v = sorted(v) + return v[min(len(v) - 1, int(p * len(v)))] if v else None + + +def calls_of(messages): + """[(index, tool name, args dict, result text)] in order.""" + res = {m.get("tool_call_id"): m["content"] for m in messages if m["role"] == "tool"} + out = [] + for i, m in enumerate(messages): + if m["role"] != "assistant": + continue + for c in m.get("tool_calls") or []: + a = c["function"].get("arguments") or "{}" + try: + a = json.loads(a) if isinstance(a, str) else a + except ValueError: + a = {} + out.append((i, c["function"]["name"], a, res.get(c.get("id"), ""))) + return out + + +def failed(tool, text): + t = (text or "").replace(" ", "") + return tool in WRITE and (t.startswith("ERROR:") or '"success":false' in t) + + +def error_text(text): + try: + o = json.loads(text.replace("ERROR: ", "", 1) if text.startswith("ERROR:") else text) + except ValueError: + return text[:200] + if not isinstance(o, dict): + return text[:200] + msgs = [m.get("message", "") for m in (o.get("activation") or {}).get("messages", []) if m.get("severity") == "E"] + out = " | ".join(msgs) or str(o.get("error") or o.get("message") or "") + sc = o.get("syntaxCheck") + if sc: + out += " | syntaxCheck: " + "; ".join(m.get("message", "") for m in sc.get("messages", [])) + return out[:400] + + +def classify(src, err): + """Labels of a failed write (a write can match more than one).""" + labels = [] + if src and SIG.search(src): + labels.append("type_c_length_in_signature") + low = (err or "").lower() + if "longer than the allowed 30 characters" in low or "30 characters" in low: + labels.append("name_over_30") + words = {w for w in re.findall(r"[A-Za-z_]+", err or "")} + if "is not valid" in low or "reserved" in low or any(w.upper() in RESERVED and w.upper() in (err or "").upper() for w in words if len(w) > 3 and w.isupper()): + if re.search(r"\b(?:" + "|".join(sorted(RESERVED, key=len, reverse=True)) + r")\b", err or "", re.I) or "reserved" in low: + labels.append("reserved_word") + return labels + + +def norm(err): + e = re.sub(r"\bZ[0-9A-Z]{7}_\w+", "", err or "") + e = re.sub(r"\b[A-Z][A-Z0-9_]{5,}\b", "", e) + e = re.sub(r"\d+", "N", e) + return e[:110] + + +def behaviors(messages): + cs = calls_of(messages) + out = {"first_write": None, "searches_before_first_write": 0, "max_consecutive_search_object": 0, "calls": len(cs), + "failed_writes": 0, "repaired": False, "calls_to_repair": None, "same_push_repeats": 0} + run_so = 0 + first_fail = None + last_src = {} + for n, (i, tool, a, res) in enumerate(cs): + if tool == "sap_search_object": + run_so += 1 + out["max_consecutive_search_object"] = max(out["max_consecutive_search_object"], run_so) + else: + run_so = 0 + if tool in ("sap_push_source", "sap_push_element", "sap_push_message") and out["first_write"] is None: + out["first_write"] = n + if out["first_write"] is None and tool in SEARCH_LIKE: + out["searches_before_first_write"] += 1 + if tool == "sap_push_source": + key = (a.get("objectName"), a.get("includeType")) + if last_src.get(key) == a.get("source") and a.get("source"): + out["same_push_repeats"] += 1 + last_src[key] = a.get("source") + if failed(tool, res): + out["failed_writes"] += 1 + if first_fail is None: + first_fail = (n, a.get("objectName"), a.get("source")) + elif first_fail and tool in ("sap_push_source", "sap_push_element") and '"success":true' in (res or "").replace(" ", "") \ + and a.get("objectName") == first_fail[1] and a.get("source") != first_fail[2] and not out["repaired"]: + out["repaired"] = True + out["calls_to_repair"] = n - first_fail[0] + return out + + +def shingles(text, k=5): + t = re.sub(r"\bz[0-9a-z]{7}_", "", text.lower()) + w = re.findall(r"\w+", t) + return {" ".join(w[i:i + k]) for i in range(max(len(w) - k + 1, 0))} + + +def final_sources(messages): + """Last pushed source per (object, include): the trajectory's own code.""" + last = {} + for i, tool, a, res in calls_of(messages): + if tool == "sap_push_source" and a.get("source") and '"success":true' in (res or "").replace(" ", ""): + last[(a.get("objectName"), a.get("includeType"))] = a["source"] + return "\n".join(last.values()) + + +def main(): + rows = [json.loads(l) for l in open(os.path.join(ROOT, "runs", "traj", "summary.jsonl"))] + acc_recs, empty_runs, all_turn_out = [], [], [] + for r in rows: + p = os.path.join(ROOT, "runs", "traj", r.get("run_dir") or "-", "record.json") + if not os.path.exists(p): + continue + rec = json.load(open(p)) + ok, why = acc.judge(rec, r, 80) + md = rec["metadata"] + if ok: + acc_recs.append((r, rec)) + all_turn_out += [u.get("completion_tokens", 0) for u in md.get("turn_usage", [])] + if md.get("end_reason") == "empty_response": + empty_runs.append((r, rec)) + out = {"accepted": len(acc_recs), "runs": len(rows)} + + # ---- repair taxonomy + fail_ct, labels_ct, traj_with, repaired_with, top = 0, Counter(), Counter(), Counter(), Counter() + per_label_examples = {} + kinds = Counter() + for r, rec in acc_recs: + seen = {} + for i, tool, a, res in calls_of(rec["messages"]): + if failed(tool, res): + fail_ct += 1 + err = error_text(res) + top[norm(err)] += 1 + for l in classify(a.get("source"), err): + labels_ct[l] += 1 + seen[l] = True + per_label_examples.setdefault(l, (r["task"], err[:160])) + b = behaviors(rec["messages"]) + for l in seen: + traj_with[l] += 1 + if b["repaired"]: + repaired_with[l] += 1 + out["repair_taxonomy"] = {"failed_writes_in_accepted": fail_ct, "step3_errors": { + l: {"failed_writes": labels_ct[l], "trajectories": traj_with[l], "trajectories_with_repair": repaired_with[l], + "example": per_label_examples.get(l)} for l in ("type_c_length_in_signature", "reserved_word", "name_over_30")}, + "top_error_messages": top.most_common(15)} + + # ---- teacher behaviors vs Qwen + def stats(recs): + b = [behaviors(x["messages"]) for x in recs] + had_fail = [x for x in b if x["failed_writes"]] + sb = [x["searches_before_first_write"] for x in b if x["first_write"] is not None] + return {"n": len(b), "with_a_failed_write": len(had_fail), + "repair_after_first_error": sum(x["repaired"] for x in had_fail), "median_calls_to_repair": + statistics.median([x["calls_to_repair"] for x in had_fail if x["calls_to_repair"] is not None] or [None]), + "median_searches_before_first_write": statistics.median(sb) if sb else None, "p90_searches_before_first_write": pct(sb, .9), + "max_consecutive_search_object": max([x["max_consecutive_search_object"] for x in b] or [0]), + "runs_with_no_write": sum(1 for x in b if x["first_write"] is None), + "runs_with_same_push_repeat": sum(1 for x in b if x["same_push_repeats"] > 0)} + out["teacher"] = stats([rec for _, rec in acc_recs]) + qrecs = [] + for p in glob.glob(os.path.join(ROOT, "runs", "local_qwen", "runs", "*", "record.json")): + qrecs.append(json.load(open(p))) + out["qwen_series_A"] = stats(qrecs) + out["qwen_series_A"]["note"] = "all 20 runs of series A (accepted and failed)" + + # ---- near duplicates + sh = [(r["task"], r["run"], shingles(final_sources(rec["messages"])), shingles(rec["messages"][-1].get("content") or "")) for r, rec in acc_recs] + dup = [] + for i in range(len(sh)): + for j in range(i + 1, len(sh)): + a, b = sh[i], sh[j] + for k, nm in ((2, "code"), (3, "report")): + if a[k] and b[k]: + jac = len(a[k] & b[k]) / len(a[k] | b[k]) + if jac >= (0.8 if nm == "code" else 0.85): + dup.append({"a": f"{a[0]}_r{a[1]}", "b": f"{b[0]}_r{b[1]}", "what": nm, "jaccard": round(jac, 2), "same_task": a[0] == b[0]}) + out["near_duplicates"] = {"pairs_flagged": len(dup), "same_task_pairs": sum(1 for d in dup if d["same_task"]), "examples": dup[:12]} + + # ---- empty_response + er = [] + for r, rec in empty_runs: + md = rec["metadata"] + tu = md.get("turn_usage", []) + last = [u.get("completion_tokens", 0) for u in tu[-3:]] + cs = calls_of(rec["messages"]) + lastcall = cs[-1] if cs else None + er.append({"task": r["task"], "kind": mix.kind_of_task_dir(r["task"]), "category": rec["task"]["category"], "calls": md.get("tool_calls"), + "turns": len(tu), "last3_output_tokens": last, "last_tool": lastcall[1] if lastcall else None, + "last_tool_result_chars": len(lastcall[3]) if lastcall else None, + "context_tokens_last": (tu[-1].get("prompt_tokens") if tu else None)}) + out["empty_response"] = {"runs": len(er), "of": len(rows), "by_kind": Counter(e["kind"] for e in er), + "by_last_tool": Counter(e["last_tool"] for e in er), "detail": er, + "output_tokens_of_normal_turns": {"p50": pct(all_turn_out, .5), "p90": pct(all_turn_out, .9), + "p99": pct(all_turn_out, .99), "p999": pct(all_turn_out, .999), "max": max(all_turn_out or [0]), + "turns": len(all_turn_out), "turns_over_16k": sum(1 for x in all_turn_out if x > 16000), + "turns_over_20k": sum(1 for x in all_turn_out if x > 20000)}} + json.dump(out, open(os.path.join(ROOT, "runs", "analysis", "analysis.json"), "w"), indent=1, default=dict) + print(json.dumps(out, indent=1, default=dict)[:6500]) + + +if __name__ == "__main__": + main()