MachineSex/tests/test_neural_torch.py
Giorgio Gilestro d22dd9d535 recombination: reproduce the E4 "merge, don't average" finding in real weights
src/neural/recombine.py mirrors Layer-1 run_coverage but trains K_T specialist RNNs on
assignments from the exact shared-switch retention construction (K_T/rho/q clean; union
matches the closed form), then recombines the measured teacher distributions two ways:
mean (naive pooling) vs oracle-guided max-merge (per-mode strongest teacher, M2N2-style),
each followed by size-n resampling.

Result (8 reps): at rho=0, union rises 0.49->0.96 (supply matches closed form); analytic
surviving_max rises 0.043->0.087 while surviving_mean stays flat ~0.045 — the conservation
law (averaging cancels the union gain, max-merge realises it). At rho=1 (identical
teachers) union and max are flat. The lesson holds in the neural setting; trained-weight
columns show the same signs but noisier (smoothing inflates baseline; deep tail barely
clears n=200 resampling). torch-gated test added. 93 tests green.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-04 21:49:44 +01:00

83 lines
4 KiB
Python

"""Stage C torch-model tests (skipped when torch is absent).
Small, fast sign checks — the neural tiers are statistically reproducible and directional,
not exact, so these assert the *sign* of each effect (blueprint 3.5): gen-0 fidelity, dry
collapse, and grounding arresting it. They gate the RNN before the N-series experiments.
"""
from __future__ import annotations
import numpy as np
import pytest
pytest.importorskip("torch")
from knowledge.metrics import forward_kl, heterozygosity # noqa: E402
from neural.config import ModelCfg, SyntheticCfg # noqa: E402
from neural.generation_loop import run_generative_lineage # noqa: E402
from neural.models import make_model # noqa: E402
from neural.oracle import ExactOracle # noqa: E402
from neural.synthetic import make_mode_truth # noqa: E402
_SYN = dict(K=256, R=1, zipf_s=1.3, init="truth", style_len=3, style_vocab=5, id_base=2,
tail_threshold=1e-3)
# hidden/epochs high enough that the RNN sharpens (an underfit RNN smooths and resists
# collapse); with n=200 K=256 the dry lineage collapses robustly across seeds.
_MODEL = dict(kind="rnn", hidden=128, embed=24, epochs=25, lr=2e-3, batch_size=256, n_eval=10000)
def _lineage_cfg(g: float, n: int, gens: int) -> dict:
m = 0 if g == 0 else round(n * g / (1 - g))
return {
"synthetic": dict(_SYN),
"model": dict(_MODEL),
"dynamics": {"n": n, "grounding": {"m": m, "policy": "proportional"}},
"generations": gens,
}
@pytest.mark.parametrize("kind", ["rnn", "mlp"])
def test_gen0_fidelity(kind):
# A trained gen-0 model must recover p* (else "collapse" would be underfitting). Checked
# for the RNN and MLP; the VAE does not clear this gate on the codeword task (see todo).
syn = SyntheticCfg(**_SYN)
td = make_mode_truth(syn)
model = make_model(ModelCfg(**{**_MODEL, "kind": kind}), syn, ExactOracle(syn))
model.initialise(td.p_star, np.random.default_rng(0))
p_hat = model.mode_distribution(np.random.default_rng(1))
assert forward_kl(td.p_star, p_hat, 1e-9) < 0.25 # close to truth
assert (p_hat > 1e-9).sum() >= 0.9 * syn.K # most modes represented
def test_rnn_dry_collapses_grounded_holds():
# gens=20 gives clean dry-vs-grounded separation (KL ~2+ vs ~0.3); big margins survive
# GPU non-determinism. Directional per blueprint 3.5.
dry = run_generative_lineage(_lineage_cfg(0.0, 200, 25), seed=0)
grd = run_generative_lineage(_lineage_cfg(0.05, 200, 25), seed=0)
assert dry["heterozygosity"].iloc[-1] < dry["heterozygosity"].iloc[0] - 0.10
assert dry["forward_kl"].iloc[-1] > 1.5 # tail forgotten
assert grd["forward_kl"].iloc[-1] < dry["forward_kl"].iloc[-1] # grounding closer to truth
assert grd["heterozygosity"].iloc[-1] > dry["heterozygosity"].iloc[-1]
def test_recombination_schema_and_union_supply():
# run_recombination trains K_T specialists and reports mean vs max-merge coverage.
# Cheap check: schema is right and the construction-level union rises with K_T (the
# recombination *supply*; magnitudes of surviving coverage need the full multi-rep run).
from neural.recombine import run_recombination
cfg = {
"experiment": "recomb_test", "seed": 20260704, "n_replicates": 1,
"synthetic": {"K": 128, "R": 1, "zipf_s": 1.3, "tail_threshold": 1e-3,
"style_len": 3, "style_vocab": 5, "id_base": 2},
"model": {"kind": "rnn", "hidden": 96, "embed": 20, "epochs": 8, "lr": 2e-3,
"batch_size": 256, "n_eval": 5000},
"coverage": {"n": 200, "q": 0.5, "retain_thresh": 1e-3},
"sweep": [{"param": "K_T", "values": [1, 3]}, {"param": "rho", "values": [0.0]}],
}
df = run_recombination(cfg)
for col in ("union_coverage", "surviving_mean", "surviving_max",
"surviving_mean_target", "surviving_max_target"):
assert col in df.columns
u = df.groupby("K_T")["union_coverage"].mean()
assert u.loc[3] > u.loc[1] + 0.1 # union supply rises with teacher count