43 lines
3.0 KiB
Python
43 lines
3.0 KiB
Python
"""Summary of the own-test mutation scores (item D): docs/own-test-mutation.md. Run after `python3 -m harness.owntests`."""
|
|
import glob
|
|
import json
|
|
import os
|
|
import statistics
|
|
from collections import Counter, defaultdict
|
|
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
res = [json.load(open(p)) for p in glob.glob(os.path.join(ROOT, "runs", "traj", "*", "own_test_mutation.json"))]
|
|
by = Counter(r.get("status") for r in res)
|
|
scored = [r for r in res if r.get("status") == "scored"]
|
|
ref_ok = [r for r in scored if r.get("tests_pass_on_reference")]
|
|
per_kind = defaultdict(list)
|
|
for r in ref_ok:
|
|
if r.get("score") is not None:
|
|
per_kind[r["kind"]].append(r["score"])
|
|
sc = [r["score"] for r in ref_ok if r.get("score") is not None]
|
|
bins = Counter("1.0" if s == 1 else ">=0.75" if s >= 0.75 else ">=0.5" if s >= 0.5 else "<0.5" for s in sc)
|
|
low = sorted([r for r in ref_ok if r.get("score") is not None and r["score"] < 0.75], key=lambda r: r["score"])
|
|
doc = f"""# Own-test mutation score of the accepted trajectories (item D, metadata only)
|
|
|
|
Module `harness/owntests.py`, results `runs/traj/<run>/own_test_mutation.json` (not in git), report by `train/own_test_report.py`.
|
|
The model's own unit tests (testclasses include or global test classes) are run against the faulty references of the task (`faulty/`, mutants that the hidden tests kill).
|
|
The correct reference must pass the model's tests first (otherwise the tests encode model specific behavior). score = killed / (killed + survived).
|
|
**Metadata only: the acceptance filter is not changed.** The builder takes the score through `--hook hooks_example:own_test_weight` (field `own_test_mutation`).
|
|
|
|
## Result
|
|
{len(res)} accepted trajectories looked at: {dict(by)}.
|
|
- Scored: {len(scored)}; the model's tests also passed on the correct reference: {len(ref_ok)} (the others are not reliable: a test that fails on the correct solution kills every mutant).
|
|
- Mean score over the reliable ones: {statistics.mean(sc):.2f}; median {statistics.median(sc):.2f}; distribution {dict(bins)} (n = {len(sc)}).
|
|
- By kind (mean, n): {{{", ".join(f"{k}: {statistics.mean(v):.2f} ({len(v)})" for k, v in sorted(per_kind.items()))}}}
|
|
- `no_own_tests`: {by.get('no_own_tests', 0)} trajectories were accepted without any own test (at most 85 points); `no_mutants`: {by.get('no_mutants', 0)}; `not_supported` (PROG: tests are inside the program): {by.get('not_supported', 0)}.
|
|
|
|
## Weakest (score under 0.75, tests pass on the reference)
|
|
""" + "\n".join(f"- {r['task']} run {r['run']} {r['kind']}: score {r['score']} ({r['killed']} of {r['valid']} valid mutants killed)" for r in low[:15]) + """
|
|
|
|
## Use
|
|
Not used for filtering yet. Candidates for a later rule (Kral + Opus decide): drop or down-weight trajectories with a reliable score under 0.5 or with `tests_pass_on_reference` false;
|
|
prefer trajectories without own tests last. The hook example shows where the weight goes (`train/hooks_example.py`).
|
|
"""
|
|
open(os.path.join(ROOT, "docs", "own-test-mutation.md"), "w").write(doc)
|
|
print(doc[:1500])
|