"""Summary of the own-test mutation scores (item D): docs/own-test-mutation.md. Run after `python3 -m harness.owntests`.""" import glob import json import os import statistics from collections import Counter, defaultdict ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) res = [json.load(open(p)) for p in glob.glob(os.path.join(ROOT, "runs", "traj", "*", "own_test_mutation.json"))] by = Counter(r.get("status") for r in res) scored = [r for r in res if r.get("status") == "scored"] ref_ok = [r for r in scored if r.get("tests_pass_on_reference")] per_kind = defaultdict(list) for r in ref_ok: if r.get("score") is not None: per_kind[r["kind"]].append(r["score"]) sc = [r["score"] for r in ref_ok if r.get("score") is not None] bins = Counter("1.0" if s == 1 else ">=0.75" if s >= 0.75 else ">=0.5" if s >= 0.5 else "<0.5" for s in sc) low = sorted([r for r in ref_ok if r.get("score") is not None and r["score"] < 0.75], key=lambda r: r["score"]) doc = f"""# Own-test mutation score of the accepted trajectories (item D, metadata only) Module `harness/owntests.py`, results `runs/traj//own_test_mutation.json` (not in git), report by `train/own_test_report.py`. The model's own unit tests (testclasses include or global test classes) are run against the faulty references of the task (`faulty/`, mutants that the hidden tests kill). The correct reference must pass the model's tests first (otherwise the tests encode model specific behavior). score = killed / (killed + survived). **Metadata only: the acceptance filter is not changed.** The builder takes the score through `--hook hooks_example:own_test_weight` (field `own_test_mutation`). ## Result {len(res)} accepted trajectories looked at: {dict(by)}. - Scored: {len(scored)}; the model's tests also passed on the correct reference: {len(ref_ok)} (the others are not reliable: a test that fails on the correct solution kills every mutant). - Mean score over the reliable ones: {statistics.mean(sc):.2f}; median {statistics.median(sc):.2f}; distribution {dict(bins)} (n = {len(sc)}). - By kind (mean, n): {{{", ".join(f"{k}: {statistics.mean(v):.2f} ({len(v)})" for k, v in sorted(per_kind.items()))}}} - `no_own_tests`: {by.get('no_own_tests', 0)} trajectories were accepted without any own test (at most 85 points); `no_mutants`: {by.get('no_mutants', 0)}; `not_supported` (PROG: tests are inside the program): {by.get('not_supported', 0)}. ## Weakest (score under 0.75, tests pass on the reference) """ + "\n".join(f"- {r['task']} run {r['run']} {r['kind']}: score {r['score']} ({r['killed']} of {r['valid']} valid mutants killed)" for r in low[:15]) + """ ## Use Not used for filtering yet. Candidates for a later rule (Kral + Opus decide): drop or down-weight trajectories with a reliable score under 0.5 or with `tests_pass_on_reference` false; prefer trajectories without own tests last. The hook example shows where the weight goes (`train/hooks_example.py`). """ open(os.path.join(ROOT, "docs", "own-test-mutation.md"), "w").write(doc) print(doc[:1500])