- paper/pnas -> paper/manuscript (venue-neutral)
- configs/layer1 -> configs/inheritance, src/knowledge -> src/inheritance
(imported as `inheritance`), make layer1 -> make inheritance; layer2 alias dropped
- inheritance and trained-network bundles named after the manuscript figure
they feed (fig2_grounding_sweep, figS3_rebaselining, ...), or descriptively
where they feed none; configs keep their `experiment:` value so parquet
hashes are unchanged, only output.dir moves
- figure scripts, SI figure sources, notebooks, REPRODUCING.md, README and the
SI Methods/tables updated; make clean no longer deletes tracked manifests;
reproduce.sh hashes the s{seed}/ layouts too
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Y64o8FKP7rCuXzC48pxpMm
169 lines
6.8 KiB
Python
169 lines
6.8 KiB
Python
"""Correctness tests (blueprint 4): shapes, normalisation, determinism.
|
|
|
|
These check that each module *does what it says* — distinct from the scientific-validation
|
|
suite, which checks that the dynamics reproduce the analytic targets. Cheap and fast.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import numpy as np
|
|
import pytest
|
|
|
|
from inheritance.config import LineageCfg
|
|
from inheritance.metrics import forward_kl, heterozygosity, support_size, tail_mass
|
|
from inheritance.step import allocate_m, apply_selection, generation_step, StepCtx, \
|
|
structured_multinomial
|
|
from inheritance.teachers import make_correlated_teachers, make_retention_matrix
|
|
from inheritance.truth import make_true_distribution, uniform_init
|
|
from inheritance.lineage import run_lineage
|
|
|
|
|
|
# ---- truth --------------------------------------------------------------------------
|
|
|
|
def test_true_distribution_normalised_and_shaped():
|
|
td = make_true_distribution(K=100, R=10, tail="zipf", tail_frac=0.5, zipf_s=1.1, seed=0)
|
|
assert td.p_star.shape == (100,)
|
|
assert td.p_star.sum() == pytest.approx(1.0)
|
|
assert np.all(td.p_star > 0)
|
|
# 10 equal regions of 10 items each
|
|
assert np.bincount(td.regions).tolist() == [10] * 10
|
|
assert td.tail_mask.dtype == bool
|
|
|
|
def test_regions_equal_mass():
|
|
td = make_true_distribution(K=100, R=10, tail="zipf", tail_frac=0.5, zipf_s=1.1, seed=0)
|
|
masses = [td.p_star[td.regions == r].sum() for r in range(10)]
|
|
assert np.allclose(masses, 0.1) # each region carries 1/R
|
|
|
|
def test_true_distribution_requires_divisible():
|
|
with pytest.raises(ValueError):
|
|
make_true_distribution(K=100, R=7, tail="zipf", tail_frac=0.5, zipf_s=1.1, seed=0)
|
|
|
|
def test_twocomponent_tail_below_threshold():
|
|
td = make_true_distribution(K=100, R=1, tail="twocomponent", tail_frac=0.5,
|
|
zipf_s=1.1, seed=0, tail_threshold=1e-3)
|
|
assert td.tail_mask.sum() > 0
|
|
assert np.all(td.p_star[td.tail_mask] < 1e-3)
|
|
|
|
|
|
# ---- metrics ------------------------------------------------------------------------
|
|
|
|
def test_heterozygosity_extremes():
|
|
assert heterozygosity(np.array([1.0, 0.0, 0.0])) == pytest.approx(0.0)
|
|
assert heterozygosity(np.full(10, 0.1)) == pytest.approx(1 - 1 / 10)
|
|
|
|
def test_forward_kl_zero_when_equal_and_positive_otherwise():
|
|
p = np.array([0.5, 0.3, 0.2])
|
|
assert forward_kl(p, p, eps=1e-9) == pytest.approx(0.0, abs=1e-12)
|
|
q = np.array([0.6, 0.3, 0.1])
|
|
assert forward_kl(p, q, eps=1e-9) > 0
|
|
|
|
def test_tail_mass_and_support():
|
|
p = np.array([0.7, 0.2, 0.1, 0.0])
|
|
mask = np.array([False, False, True, True])
|
|
assert tail_mass(p, mask) == pytest.approx(0.1)
|
|
assert support_size(p, eps=1e-9) == 3
|
|
|
|
|
|
# ---- teachers -----------------------------------------------------------------------
|
|
|
|
def test_retention_matrix_shape_and_dtype():
|
|
rng = np.random.default_rng(0)
|
|
R = make_retention_matrix(T=1000, K_T=3, rho=0.5, q=0.3, rng=rng)
|
|
assert R.shape == (3, 1000)
|
|
assert set(np.unique(R)).issubset({0, 1})
|
|
|
|
def test_correlated_teachers_normalised():
|
|
td = make_true_distribution(K=200, R=4, tail="zipf", tail_frac=0.5, zipf_s=1.1, seed=1)
|
|
teachers = make_correlated_teachers(td.p_star, td.tail_mask, K_T=3, rho=0.0, q=0.5, seed=2)
|
|
assert len(teachers) == 3
|
|
for p in teachers:
|
|
assert p.sum() == pytest.approx(1.0)
|
|
assert np.all(p > 0)
|
|
|
|
def test_region_specialisation_retains_home_tails():
|
|
td = make_true_distribution(K=200, R=4, tail="zipf", tail_frac=0.5, zipf_s=1.1, seed=1)
|
|
teachers = make_correlated_teachers(
|
|
td.p_star, td.tail_mask, K_T=4, rho=0.0, q=0.0,
|
|
region_assignment=td.regions, region_specialisation=True, seed=3)
|
|
# teacher k fully retains its home region's tails even at q=0
|
|
tail_idx = np.flatnonzero(td.tail_mask)
|
|
home0_tail = tail_idx[td.regions[tail_idx] == 0]
|
|
assert np.all(teachers[0][home0_tail] > 1e-6) # kept at p_star, not floored
|
|
|
|
|
|
# ---- step ---------------------------------------------------------------------------
|
|
|
|
def test_generation_step_sums_to_one():
|
|
td = make_true_distribution(K=50, R=1, tail="zipf", tail_frac=0.5, zipf_s=1.1, seed=0)
|
|
rng = np.random.default_rng(0)
|
|
ctx = StepCtx(n=200, m_vector=allocate_m(20, 1, "uniform"), policy="uniform",
|
|
regions=td.regions)
|
|
p = generation_step([uniform_init(50)], td.p_star, ctx, rng)
|
|
assert p.sum() == pytest.approx(1.0)
|
|
assert np.all(p >= 0)
|
|
|
|
def test_structured_multinomial_counts_sum_to_budget():
|
|
td = make_true_distribution(K=100, R=10, tail="zipf", tail_frac=0.5, zipf_s=1.1, seed=0)
|
|
rng = np.random.default_rng(0)
|
|
m_vec = allocate_m(50, 10, "uniform")
|
|
counts = structured_multinomial(m_vec, td.p_star, td.regions, "uniform", rng)
|
|
assert counts.sum() == 50
|
|
# each region gets exactly its budget
|
|
for r in range(10):
|
|
assert counts[td.regions == r].sum() == m_vec[r]
|
|
|
|
def test_allocate_m_splits_remainder():
|
|
m = allocate_m(23, 10, "uniform")
|
|
assert m.sum() == 23
|
|
assert m.max() - m.min() <= 1 # as even as possible
|
|
|
|
def test_selection_none_is_identity():
|
|
p = np.array([0.5, 0.3, 0.2])
|
|
assert np.allclose(apply_selection(p, p, "none", 0.0), p)
|
|
|
|
def test_greedy_selection_concentrates_on_high_fitness():
|
|
p = np.array([0.4, 0.4, 0.2])
|
|
f = np.array([0.1, 0.1, 0.8]) # item 2 is fittest
|
|
out = apply_selection(p, f, "greedy", 0.0)
|
|
assert out[2] > p[2] # fitness-proportional shifts mass toward the fit item
|
|
|
|
|
|
# ---- lineage ------------------------------------------------------------------------
|
|
|
|
def _cfg(**over):
|
|
base = {
|
|
"truth": {"K": 50, "R": 1, "tail": "zipf", "zipf_s": 1.1,
|
|
"tail_frac": 0.5, "tail_threshold": 1e-3},
|
|
"dynamics": {"n": 100, "grounding": {"m": 0, "policy": "uniform"}},
|
|
"generations": 20,
|
|
"metrics": {"kl_floor": 1e-9},
|
|
}
|
|
base.update(over)
|
|
return base
|
|
|
|
def test_run_lineage_rows_and_columns():
|
|
df = run_lineage(_cfg(), seed=0)
|
|
assert len(df) == 21 # generations 0..20
|
|
for col in ("generation", "heterozygosity", "forward_kl", "tail_mass", "support_size"):
|
|
assert col in df.columns
|
|
assert df["generation"].tolist() == list(range(21))
|
|
|
|
def test_run_lineage_is_deterministic():
|
|
a = run_lineage(_cfg(), seed=42)
|
|
b = run_lineage(_cfg(), seed=42)
|
|
assert a.equals(b)
|
|
|
|
def test_run_lineage_different_seeds_differ():
|
|
a = run_lineage(_cfg(), seed=1)
|
|
b = run_lineage(_cfg(), seed=2)
|
|
assert not a["heterozygosity"].equals(b["heterozygosity"])
|
|
|
|
def test_per_region_columns_present_when_multiregion():
|
|
df = run_lineage(_cfg(truth={"K": 100, "R": 5, "tail": "zipf", "zipf_s": 1.1,
|
|
"tail_frac": 0.5, "tail_threshold": 1e-3}), seed=0)
|
|
assert "H_region_0" in df.columns
|
|
assert "tail_region_4" in df.columns
|
|
|
|
def test_config_rejects_unknown_keys():
|
|
with pytest.raises(ValueError):
|
|
LineageCfg.from_dict({"truth": {"K": 10, "bogus": 1}})
|