Decisions of 2026-10-06: 60 % rule in the training script, own-test weights (fractional), 64k in the sweep, foreign-read trajectories back to the pending pool, memory test waits, foreign object scan of baselines and eval runs, 11 October check list

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-06 09:04:09 +02:00
parent a465c33e3f
commit 8ef85c2713
10 changed files with 402 additions and 41 deletions

79
train/foreign_scan.py Normal file
View File

@@ -0,0 +1,79 @@
"""Read-only scan: which runs saw other runs' objects in tool results, or read them (teardown/proxy bug of 2026-10-06).
python3 train/foreign_scan.py -> prints a summary and writes runs/analysis/foreign_scan.json
"""
import glob
import json
import os
import re
import sys
from collections import Counter
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, ROOT)
from harness.task import prefix_for # noqa: E402
START = re.compile(r"Z\d[0-9A-Z]{6}_")
MID_NAME = re.compile(r"^[A-Z]{1,5}_(Z\d[0-9A-Z]{6}_)", re.I)
READ = ("sap_pull_source", "sap_object_structure", "sap_object_members", "sap_element_info", "sap_run_unit_test", "sap_check_object",
"sap_syntax_check", "sap_atc_run")
GROUPS = [("official baseline Qwen (MacBook, 11 tasks)", "runs/stage1/baseline_qwen/*"),
("Devstral baseline (dropped)", "runs/archive/devstral/baseline_devstral/*"),
("older Qwen baselines (archive)", "runs/archive/qwen38/*/*"),
("smoke", "runs/stage1/smoke/*"),
("eval empirical filter (DeepSeek on eval candidates)", "runs/emp/*"),
("eval generation validation (oracle, null, mutants)", "runs/gen/*"),
("early pilot runs", "runs/2*_*"),
("training trajectories (DeepSeek)", "runs/traj/*_G*"),
("series A (local Qwen)", "runs/local_qwen/runs/*")]
def own_prefix(d):
m = re.match(r"^(\d+)_([GT]\d+)_", os.path.basename(d))
return prefix_for(int(m.group(1)), m.group(2)).upper() if m else None
def scan(d):
own = own_prefix(d)
p = os.path.join(d, "trajectory.jsonl")
if not own or not os.path.exists(p):
return None
seen, reads, tools = set(), 0, Counter()
for l in open(p):
try:
e = json.loads(l)
except ValueError:
continue
if "tool" not in e:
continue
text = (e.get("result") or "").upper()
found = {x for x in START.findall(text) if x != own and x[1].isdigit()}
found |= {m.group(1).upper() for m in re.finditer(r"\b[A-Z]{1,5}_(Z\d[0-9A-Z]{6}_)", text) if m.group(1).upper() != own}
if found:
seen |= found
tools[e["tool"]] += 1
name = str((e.get("args") or {}).get("objectName", "")).upper()
m = START.match(name) or MID_NAME.match(name)
pre = (m.group(1) if m and m.re is MID_NAME else (m.group(0) if m else "")).upper()
if e["tool"] in READ and pre and pre != own:
reads += 1
return {"run": os.path.basename(d), "foreign_prefixes_seen": len(seen), "foreign_reads": reads, "tools": dict(tools)}
def main():
out = {}
for label, pat in GROUPS:
dirs = [d for d in sorted(glob.glob(os.path.join(ROOT, pat))) if os.path.isdir(d)]
rows = [r for r in (scan(d) for d in dirs) if r]
seen = [r for r in rows if r["foreign_prefixes_seen"]]
reads = [r for r in rows if r["foreign_reads"]]
out[label] = {"runs": len(rows), "runs_with_foreign_names": len(seen), "runs_with_foreign_reads": len(reads),
"tools": dict(sum((Counter(r["tools"]) for r in seen), Counter())),
"affected": [r["run"] for r in seen][:40], "read_runs": [r["run"] for r in reads]}
print(f"{label}: {len(rows)} runs | foreign names in tool results: {len(seen)} | foreign reads: {len(reads)} | tools {out[label]['tools']}")
os.makedirs(os.path.join(ROOT, "runs", "analysis"), exist_ok=True)
json.dump(out, open(os.path.join(ROOT, "runs", "analysis", "foreign_scan.json"), "w"), indent=1)
if __name__ == "__main__":
main()