Manuscript revision and pending experiment work, snapshot before restructuring

Clarity pass over the main text (36-item audit), Discussion rewrite and cut,
acknowledgements, Souly et al. as ref 62, lettered SI panels, model section
moved under Results; plus the untracked curriculum/society/compose/smol
configs, runners, figures, stats and tests that the SI already cites.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Y64o8FKP7rCuXzC48pxpMm
This commit is contained in:
Giorgio Gilestro 2026-09-13 16:54:09 +01:00
parent e4804adabc
commit 84124de143
450 changed files with 52813 additions and 1202 deletions

View file

@ -0,0 +1,28 @@
{
"experiment": "llm_moe_hard_hpc",
"master_seed": 3,
"git_commit": null,
"python": "3.11.7",
"libraries": {
"numpy": "2.4.6",
"scipy": "1.17.1",
"pandas": "3.0.3",
"pyarrow": "24.0.0",
"torch": "2.12.1",
"transformers": "5.16.1",
"peft": "0.20.0"
},
"rows": 47,
"results_sha256": "1a9962f95ed0b0458676682d415fafcaa530a11eefc3833ba46a144e22ee90d5",
"layer": "2",
"tier": "llm",
"base_model": "Qwen/Qwen2.5-7B-Instruct",
"hard": true,
"operators": [
"soup",
"ties",
"moe_oracle",
"moe_learned",
"max_merge"
]
}

View file

@ -0,0 +1,29 @@
experiment: llm_moe_hard_hpc
seed: 3
n_replicates: 1
source_config:
experiment: llm_moe_hard_hpc
kind: llm_moe
seed: 3
n_replicates: 1
base_model: Qwen/Qwen2.5-7B-Instruct
hard: true
families:
- lists
- strings
- arith
n_train: 800
n_test: 200
n_route: 48
epochs: 3
lora:
r: 16
alpha: 32
operators:
- soup
- ties
- moe_oracle
- moe_learned
- max_merge
output:
dir: results/llm_moe_hard_hpc/s3