MachineSex/tests/test_neural_torch.py
Giorgio Gilestro ab3dc10587 Restructure: descriptive tier and experiment names, paper/manuscript
- paper/pnas -> paper/manuscript (venue-neutral)
- configs/layer1 -> configs/inheritance, src/knowledge -> src/inheritance
  (imported as `inheritance`), make layer1 -> make inheritance; layer2 alias dropped
- inheritance and trained-network bundles named after the manuscript figure
  they feed (fig2_grounding_sweep, figS3_rebaselining, ...), or descriptively
  where they feed none; configs keep their `experiment:` value so parquet
  hashes are unchanged, only output.dir moves
- figure scripts, SI figure sources, notebooks, REPRODUCING.md, README and the
  SI Methods/tables updated; make clean no longer deletes tracked manifests;
  reproduce.sh hashes the s{seed}/ layouts too

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Y64o8FKP7rCuXzC48pxpMm
2026-09-13 17:00:40 +01:00

83 lines
4 KiB
Python

"""Stage C torch-model tests (skipped when torch is absent).
Small, fast sign checks — the neural tiers are statistically reproducible and directional,
not exact, so these assert the *sign* of each effect (blueprint 3.5): gen-0 fidelity, dry
collapse, and grounding arresting it. They gate the RNN before the N-series experiments.
"""
from __future__ import annotations
import numpy as np
import pytest
pytest.importorskip("torch")
from inheritance.metrics import forward_kl, heterozygosity # noqa: E402
from neural.config import ModelCfg, SyntheticCfg # noqa: E402
from neural.generation_loop import run_generative_lineage # noqa: E402
from neural.models import make_model # noqa: E402
from neural.oracle import ExactOracle # noqa: E402
from neural.synthetic import make_mode_truth # noqa: E402
_SYN = dict(K=256, R=1, zipf_s=1.3, init="truth", style_len=3, style_vocab=5, id_base=2,
tail_threshold=1e-3)
# hidden/epochs high enough that the RNN sharpens (an underfit RNN smooths and resists
# collapse); with n=200 K=256 the dry lineage collapses robustly across seeds.
_MODEL = dict(kind="rnn", hidden=128, embed=24, epochs=25, lr=2e-3, batch_size=256, n_eval=10000)
def _lineage_cfg(g: float, n: int, gens: int) -> dict:
m = 0 if g == 0 else round(n * g / (1 - g))
return {
"synthetic": dict(_SYN),
"model": dict(_MODEL),
"dynamics": {"n": n, "grounding": {"m": m, "policy": "proportional"}},
"generations": gens,
}
@pytest.mark.parametrize("kind", ["rnn", "mlp"])
def test_gen0_fidelity(kind):
# A trained gen-0 model must recover p* (else "collapse" would be underfitting). Checked
# for the RNN and MLP; the VAE does not clear this gate on the codeword task (see todo).
syn = SyntheticCfg(**_SYN)
td = make_mode_truth(syn)
model = make_model(ModelCfg(**{**_MODEL, "kind": kind}), syn, ExactOracle(syn))
model.initialise(td.p_star, np.random.default_rng(0))
p_hat = model.mode_distribution(np.random.default_rng(1))
assert forward_kl(td.p_star, p_hat, 1e-9) < 0.25 # close to truth
assert (p_hat > 1e-9).sum() >= 0.9 * syn.K # most modes represented
def test_rnn_dry_collapses_grounded_holds():
# gens=20 gives clean dry-vs-grounded separation (KL ~2+ vs ~0.3); big margins survive
# GPU non-determinism. Directional per blueprint 3.5.
dry = run_generative_lineage(_lineage_cfg(0.0, 200, 25), seed=0)
grd = run_generative_lineage(_lineage_cfg(0.05, 200, 25), seed=0)
assert dry["heterozygosity"].iloc[-1] < dry["heterozygosity"].iloc[0] - 0.10
assert dry["forward_kl"].iloc[-1] > 1.5 # tail forgotten
assert grd["forward_kl"].iloc[-1] < dry["forward_kl"].iloc[-1] # grounding closer to truth
assert grd["heterozygosity"].iloc[-1] > dry["heterozygosity"].iloc[-1]
def test_recombination_schema_and_union_supply():
# run_recombination trains K_T specialists and reports mean vs max-merge coverage.
# Cheap check: schema is right and the construction-level union rises with K_T (the
# recombination *supply*; magnitudes of surviving coverage need the full multi-rep run).
from neural.recombine import run_recombination
cfg = {
"experiment": "recomb_test", "seed": 20260704, "n_replicates": 1,
"synthetic": {"K": 128, "R": 1, "zipf_s": 1.3, "tail_threshold": 1e-3,
"style_len": 3, "style_vocab": 5, "id_base": 2},
"model": {"kind": "rnn", "hidden": 96, "embed": 20, "epochs": 8, "lr": 2e-3,
"batch_size": 256, "n_eval": 5000},
"coverage": {"n": 200, "q": 0.5, "retain_thresh": 1e-3},
"sweep": [{"param": "K_T", "values": [1, 3]}, {"param": "rho", "values": [0.0]}],
}
df = run_recombination(cfg)
for col in ("union_coverage", "surviving_mean", "surviving_max",
"surviving_mean_target", "surviving_max_target"):
assert col in df.columns
u = df.groupby("K_T")["union_coverage"].mean()
assert u.loc[3] > u.loc[1] + 0.1 # union supply rises with teacher count