# Calibration stage B (prereg §4, gates C2 / C3 / C5) over the 12 families chosen from stage A. # Run three times with stage: transmission | cross | consensus (see Makefile `llm-society-calib`). # Families = stage-A pass-2 option 1 (L=9, gate 0.41; prereg §4a) — pending GG's go. experiment: llm_society_v2_calib_b kind: llm_society_calib stage: transmission base_model: Qwen/Qwen2.5-0.5B-Instruct seed: 1 families: [strings, setops, numtheory, mixedtoken, digits, alphabet, prime, wordlen, roman] # C2: examples-per-family k and epochs to sweep; probe_families are the three whose retention is # measured (spread across answer types: list / word / int). probe_families: [setops, alphabet, digits] # list / letter / integer answers ks: [25, 50, 100, 150] epochs_grid: [2, 3] # C3: the two-founder cross (union-distil vs best-of-6 linear-merge-distil) cross: [setops, alphabet] k_inherit: 100 epochs: 3 n_candidates: 6 n_test: 100 n_probe: 10 spec_train: 1200 spec_epochs: 3 lora: {r: 16, alpha: 32} output: {dir: results/llm_society_v2_calib_b}