Manuscript revision and pending experiment work, snapshot before restructuring

Clarity pass over the main text (36-item audit), Discussion rewrite and cut,
acknowledgements, Souly et al. as ref 62, lettered SI panels, model section
moved under Results; plus the untracked curriculum/society/compose/smol
configs, runners, figures, stats and tests that the SI already cites.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Y64o8FKP7rCuXzC48pxpMm
This commit is contained in:
Giorgio Gilestro 2026-09-13 16:54:09 +01:00
parent e4804adabc
commit 84124de143
450 changed files with 52813 additions and 1202 deletions

View file

@ -0,0 +1,52 @@
{
"experiment": "llm_curriculum_v5",
"master_seed": 1,
"git_commit": "e4804adabcdce6d928c5e6b1e85429b2a6acf2fe",
"python": "3.14.7",
"libraries": {
"numpy": "2.5.0",
"scipy": "1.18.0",
"pandas": "3.0.3",
"pyarrow": "24.0.0",
"torch": "2.12.1",
"transformers": "5.16.1",
"peft": "0.20.0"
},
"rows": 829,
"results_sha256": "8c33f8cde08890036bf203fbd501f10d9a0aaba6c6e3d413c5eaed7213a4aade",
"layer": "2",
"tier": "llm",
"base_model": "Qwen/Qwen2.5-1.5B",
"hard": false,
"curriculum": {
"families": [
"mnli",
"arc",
"hellaswag",
"squad",
"boolq",
"winogrande"
],
"lineages": 3,
"generations": 6,
"arms": [
"isolated",
"society",
"society_dry",
"seed_bank"
],
"baselines": [
"sequential",
"single_shot_merge",
"joint"
],
"n_new": 300,
"n_replay": 150,
"operator": "linear",
"ancestor_depth": 3,
"lora": {
"r": 16,
"alpha": 32
}
}
}

View file

@ -0,0 +1,65 @@
experiment: llm_curriculum_v5
seed: 1
n_replicates: 1
source_config:
experiment: llm_curriculum_v5
kind: llm_curriculum
base_model: Qwen/Qwen2.5-1.5B
seed: 1
families:
- mnli
- arc
- hellaswag
- squad
- boolq
- winogrande
lineages: 3
generations: 6
arms:
- isolated
- society
- society_dry
- seed_bank
baselines:
- sequential
- single_shot_merge
- joint
n_new: 300
n_replay: 150
n_test: 60
n_val: 20
epochs: 3
lr: 0.0001
ancestor_depth: 3
operator: linear
merge_weights:
- - 0.5
- 0.5
- - 0.3
- 0.7
- - 0.7
- 0.3
baseline_weights:
- - 0.333
- 0.333
- 0.334
- - 0.5
- 0.25
- 0.25
- - 0.25
- 0.5
- 0.25
- - 0.25
- 0.25
- 0.5
max_new_tokens: 48
batch_size: 24
train_batch_size: 2
train_max_len: 512
resume: true
lora:
r: 16
alpha: 32
output:
dir: results/llm_curriculum_v5/s1
n_replicates: 1

View file

@ -0,0 +1,49 @@
{
"experiment": "llm_curriculum_v5",
"master_seed": 2,
"git_commit": null,
"python": "3.11.7",
"libraries": {
"numpy": "2.4.6",
"scipy": "1.17.1",
"pandas": "3.0.3",
"pyarrow": "24.0.0",
"torch": "2.12.1",
"transformers": "5.16.1",
"peft": "0.20.0"
},
"rows": 235,
"results_sha256": "b6b9a506e47066ec93ec262c5d499f775b75f1669b021a651cd620a7af899cbc",
"layer": "2",
"tier": "llm",
"base_model": "Qwen/Qwen2.5-1.5B",
"hard": false,
"curriculum": {
"families": [
"mnli",
"arc",
"hellaswag",
"squad",
"boolq",
"winogrande"
],
"lineages": 3,
"generations": 6,
"arms": [
"isolated"
],
"baselines": [
"sequential",
"single_shot_merge",
"joint"
],
"n_new": 300,
"n_replay": 150,
"operator": "linear",
"ancestor_depth": 3,
"lora": {
"r": 16,
"alpha": 32
}
}
}

View file

@ -0,0 +1,62 @@
experiment: llm_curriculum_v5
seed: 2
n_replicates: 1
source_config:
experiment: llm_curriculum_v5
kind: llm_curriculum
base_model: Qwen/Qwen2.5-1.5B
seed: 2
families:
- mnli
- arc
- hellaswag
- squad
- boolq
- winogrande
lineages: 3
generations: 6
arms:
- isolated
baselines:
- sequential
- single_shot_merge
- joint
n_new: 300
n_replay: 150
n_test: 60
n_val: 20
epochs: 3
lr: 0.0001
ancestor_depth: 3
operator: linear
merge_weights:
- - 0.5
- 0.5
- - 0.3
- 0.7
- - 0.7
- 0.3
baseline_weights:
- - 0.333
- 0.333
- 0.334
- - 0.5
- 0.25
- 0.25
- - 0.25
- 0.5
- 0.25
- - 0.25
- 0.25
- 0.5
max_new_tokens: 48
batch_size: 48
train_batch_size: 4
train_max_len: 512
resume: true
lora:
r: 16
alpha: 32
output:
dir: results/llm_curriculum_v5/s2_isolated
n_replicates: 1

View file

@ -0,0 +1,45 @@
{
"experiment": "llm_curriculum_v5",
"master_seed": 2,
"git_commit": null,
"python": "3.11.7",
"libraries": {
"numpy": "2.4.6",
"scipy": "1.17.1",
"pandas": "3.0.3",
"pyarrow": "24.0.0",
"torch": "2.12.1",
"transformers": "5.16.1",
"peft": "0.20.0"
},
"rows": 205,
"results_sha256": "ee5e10d1da4a83ad4e499e09b00a6d70e66517d4915a32517b65514d462cc261",
"layer": "2",
"tier": "llm",
"base_model": "Qwen/Qwen2.5-1.5B",
"hard": false,
"curriculum": {
"families": [
"mnli",
"arc",
"hellaswag",
"squad",
"boolq",
"winogrande"
],
"lineages": 3,
"generations": 6,
"arms": [
"seed_bank"
],
"baselines": [],
"n_new": 300,
"n_replay": 150,
"operator": "linear",
"ancestor_depth": 3,
"lora": {
"r": 16,
"alpha": 32
}
}
}

View file

@ -0,0 +1,59 @@
experiment: llm_curriculum_v5
seed: 2
n_replicates: 1
source_config:
experiment: llm_curriculum_v5
kind: llm_curriculum
base_model: Qwen/Qwen2.5-1.5B
seed: 2
families:
- mnli
- arc
- hellaswag
- squad
- boolq
- winogrande
lineages: 3
generations: 6
arms:
- seed_bank
baselines: []
n_new: 300
n_replay: 150
n_test: 60
n_val: 20
epochs: 3
lr: 0.0001
ancestor_depth: 3
operator: linear
merge_weights:
- - 0.5
- 0.5
- - 0.3
- 0.7
- - 0.7
- 0.3
baseline_weights:
- - 0.333
- 0.333
- 0.334
- - 0.5
- 0.25
- 0.25
- - 0.25
- 0.5
- 0.25
- - 0.25
- 0.25
- 0.5
max_new_tokens: 48
batch_size: 48
train_batch_size: 4
train_max_len: 512
resume: true
lora:
r: 16
alpha: 32
output:
dir: results/llm_curriculum_v5/s2_seed_bank
n_replicates: 1

View file

@ -0,0 +1,45 @@
{
"experiment": "llm_curriculum_v5",
"master_seed": 2,
"git_commit": null,
"python": "3.11.7",
"libraries": {
"numpy": "2.4.6",
"scipy": "1.17.1",
"pandas": "3.0.3",
"pyarrow": "24.0.0",
"torch": "2.12.1",
"transformers": "5.16.1",
"peft": "0.20.0"
},
"rows": 205,
"results_sha256": "67c63a97ee14a686e3395e927a17594e65ead722415c60636915f9eba30c7426",
"layer": "2",
"tier": "llm",
"base_model": "Qwen/Qwen2.5-1.5B",
"hard": false,
"curriculum": {
"families": [
"mnli",
"arc",
"hellaswag",
"squad",
"boolq",
"winogrande"
],
"lineages": 3,
"generations": 6,
"arms": [
"society"
],
"baselines": [],
"n_new": 300,
"n_replay": 150,
"operator": "linear",
"ancestor_depth": 3,
"lora": {
"r": 16,
"alpha": 32
}
}
}

View file

@ -0,0 +1,59 @@
experiment: llm_curriculum_v5
seed: 2
n_replicates: 1
source_config:
experiment: llm_curriculum_v5
kind: llm_curriculum
base_model: Qwen/Qwen2.5-1.5B
seed: 2
families:
- mnli
- arc
- hellaswag
- squad
- boolq
- winogrande
lineages: 3
generations: 6
arms:
- society
baselines: []
n_new: 300
n_replay: 150
n_test: 60
n_val: 20
epochs: 3
lr: 0.0001
ancestor_depth: 3
operator: linear
merge_weights:
- - 0.5
- 0.5
- - 0.3
- 0.7
- - 0.7
- 0.3
baseline_weights:
- - 0.333
- 0.333
- 0.334
- - 0.5
- 0.25
- 0.25
- - 0.25
- 0.5
- 0.25
- - 0.25
- 0.25
- 0.5
max_new_tokens: 48
batch_size: 48
train_batch_size: 4
train_max_len: 512
resume: true
lora:
r: 16
alpha: 32
output:
dir: results/llm_curriculum_v5/s2_society
n_replicates: 1

View file

@ -0,0 +1,45 @@
{
"experiment": "llm_curriculum_v5",
"master_seed": 2,
"git_commit": null,
"python": "3.11.7",
"libraries": {
"numpy": "2.4.6",
"scipy": "1.17.1",
"pandas": "3.0.3",
"pyarrow": "24.0.0",
"torch": "2.12.1",
"transformers": "5.16.1",
"peft": "0.20.0"
},
"rows": 205,
"results_sha256": "226126c5e9799d4cdf2589068e3aaf6bfa1b9a71fcda03777f908b754e3f4bbf",
"layer": "2",
"tier": "llm",
"base_model": "Qwen/Qwen2.5-1.5B",
"hard": false,
"curriculum": {
"families": [
"mnli",
"arc",
"hellaswag",
"squad",
"boolq",
"winogrande"
],
"lineages": 3,
"generations": 6,
"arms": [
"society_dry"
],
"baselines": [],
"n_new": 300,
"n_replay": 150,
"operator": "linear",
"ancestor_depth": 3,
"lora": {
"r": 16,
"alpha": 32
}
}
}

View file

@ -0,0 +1,59 @@
experiment: llm_curriculum_v5
seed: 2
n_replicates: 1
source_config:
experiment: llm_curriculum_v5
kind: llm_curriculum
base_model: Qwen/Qwen2.5-1.5B
seed: 2
families:
- mnli
- arc
- hellaswag
- squad
- boolq
- winogrande
lineages: 3
generations: 6
arms:
- society_dry
baselines: []
n_new: 300
n_replay: 150
n_test: 60
n_val: 20
epochs: 3
lr: 0.0001
ancestor_depth: 3
operator: linear
merge_weights:
- - 0.5
- 0.5
- - 0.3
- 0.7
- - 0.7
- 0.3
baseline_weights:
- - 0.333
- 0.333
- 0.334
- - 0.5
- 0.25
- 0.25
- - 0.25
- 0.5
- 0.25
- - 0.25
- 0.25
- 0.5
max_new_tokens: 48
batch_size: 48
train_batch_size: 4
train_max_len: 512
resume: true
lora:
r: 16
alpha: 32
output:
dir: results/llm_curriculum_v5/s2_society_dry
n_replicates: 1

View file

@ -0,0 +1,49 @@
{
"experiment": "llm_curriculum_v5",
"master_seed": 3,
"git_commit": null,
"python": "3.11.7",
"libraries": {
"numpy": "2.4.6",
"scipy": "1.17.1",
"pandas": "3.0.3",
"pyarrow": "24.0.0",
"torch": "2.12.1",
"transformers": "5.16.1",
"peft": "0.20.0"
},
"rows": 235,
"results_sha256": "291a074fe83cc1c282346197177afa78a69d48d8e4dbfcff29f165b4e1720744",
"layer": "2",
"tier": "llm",
"base_model": "Qwen/Qwen2.5-1.5B",
"hard": false,
"curriculum": {
"families": [
"mnli",
"arc",
"hellaswag",
"squad",
"boolq",
"winogrande"
],
"lineages": 3,
"generations": 6,
"arms": [
"isolated"
],
"baselines": [
"sequential",
"single_shot_merge",
"joint"
],
"n_new": 300,
"n_replay": 150,
"operator": "linear",
"ancestor_depth": 3,
"lora": {
"r": 16,
"alpha": 32
}
}
}

View file

@ -0,0 +1,62 @@
experiment: llm_curriculum_v5
seed: 3
n_replicates: 1
source_config:
experiment: llm_curriculum_v5
kind: llm_curriculum
base_model: Qwen/Qwen2.5-1.5B
seed: 3
families:
- mnli
- arc
- hellaswag
- squad
- boolq
- winogrande
lineages: 3
generations: 6
arms:
- isolated
baselines:
- sequential
- single_shot_merge
- joint
n_new: 300
n_replay: 150
n_test: 60
n_val: 20
epochs: 3
lr: 0.0001
ancestor_depth: 3
operator: linear
merge_weights:
- - 0.5
- 0.5
- - 0.3
- 0.7
- - 0.7
- 0.3
baseline_weights:
- - 0.333
- 0.333
- 0.334
- - 0.5
- 0.25
- 0.25
- - 0.25
- 0.5
- 0.25
- - 0.25
- 0.25
- 0.5
max_new_tokens: 48
batch_size: 48
train_batch_size: 4
train_max_len: 512
resume: true
lora:
r: 16
alpha: 32
output:
dir: results/llm_curriculum_v5/s3_isolated
n_replicates: 1

View file

@ -0,0 +1,45 @@
{
"experiment": "llm_curriculum_v5",
"master_seed": 3,
"git_commit": null,
"python": "3.11.7",
"libraries": {
"numpy": "2.4.6",
"scipy": "1.17.1",
"pandas": "3.0.3",
"pyarrow": "24.0.0",
"torch": "2.12.1",
"transformers": "5.16.1",
"peft": "0.20.0"
},
"rows": 205,
"results_sha256": "f41adbd818646624e419c1448d5abe68c2005d810e709e740171e62bfc4a469d",
"layer": "2",
"tier": "llm",
"base_model": "Qwen/Qwen2.5-1.5B",
"hard": false,
"curriculum": {
"families": [
"mnli",
"arc",
"hellaswag",
"squad",
"boolq",
"winogrande"
],
"lineages": 3,
"generations": 6,
"arms": [
"seed_bank"
],
"baselines": [],
"n_new": 300,
"n_replay": 150,
"operator": "linear",
"ancestor_depth": 3,
"lora": {
"r": 16,
"alpha": 32
}
}
}

View file

@ -0,0 +1,59 @@
experiment: llm_curriculum_v5
seed: 3
n_replicates: 1
source_config:
experiment: llm_curriculum_v5
kind: llm_curriculum
base_model: Qwen/Qwen2.5-1.5B
seed: 3
families:
- mnli
- arc
- hellaswag
- squad
- boolq
- winogrande
lineages: 3
generations: 6
arms:
- seed_bank
baselines: []
n_new: 300
n_replay: 150
n_test: 60
n_val: 20
epochs: 3
lr: 0.0001
ancestor_depth: 3
operator: linear
merge_weights:
- - 0.5
- 0.5
- - 0.3
- 0.7
- - 0.7
- 0.3
baseline_weights:
- - 0.333
- 0.333
- 0.334
- - 0.5
- 0.25
- 0.25
- - 0.25
- 0.5
- 0.25
- - 0.25
- 0.25
- 0.5
max_new_tokens: 48
batch_size: 48
train_batch_size: 4
train_max_len: 512
resume: true
lora:
r: 16
alpha: 32
output:
dir: results/llm_curriculum_v5/s3_seed_bank
n_replicates: 1

View file

@ -0,0 +1,45 @@
{
"experiment": "llm_curriculum_v5",
"master_seed": 3,
"git_commit": null,
"python": "3.11.7",
"libraries": {
"numpy": "2.4.6",
"scipy": "1.17.1",
"pandas": "3.0.3",
"pyarrow": "24.0.0",
"torch": "2.12.1",
"transformers": "5.16.1",
"peft": "0.20.0"
},
"rows": 205,
"results_sha256": "e4624c2f2db818a23db13cf9fde14cac60214fad9b109b4646a2604788ea1019",
"layer": "2",
"tier": "llm",
"base_model": "Qwen/Qwen2.5-1.5B",
"hard": false,
"curriculum": {
"families": [
"mnli",
"arc",
"hellaswag",
"squad",
"boolq",
"winogrande"
],
"lineages": 3,
"generations": 6,
"arms": [
"society"
],
"baselines": [],
"n_new": 300,
"n_replay": 150,
"operator": "linear",
"ancestor_depth": 3,
"lora": {
"r": 16,
"alpha": 32
}
}
}

View file

@ -0,0 +1,59 @@
experiment: llm_curriculum_v5
seed: 3
n_replicates: 1
source_config:
experiment: llm_curriculum_v5
kind: llm_curriculum
base_model: Qwen/Qwen2.5-1.5B
seed: 3
families:
- mnli
- arc
- hellaswag
- squad
- boolq
- winogrande
lineages: 3
generations: 6
arms:
- society
baselines: []
n_new: 300
n_replay: 150
n_test: 60
n_val: 20
epochs: 3
lr: 0.0001
ancestor_depth: 3
operator: linear
merge_weights:
- - 0.5
- 0.5
- - 0.3
- 0.7
- - 0.7
- 0.3
baseline_weights:
- - 0.333
- 0.333
- 0.334
- - 0.5
- 0.25
- 0.25
- - 0.25
- 0.5
- 0.25
- - 0.25
- 0.25
- 0.5
max_new_tokens: 48
batch_size: 48
train_batch_size: 4
train_max_len: 512
resume: true
lora:
r: 16
alpha: 32
output:
dir: results/llm_curriculum_v5/s3_society
n_replicates: 1

View file

@ -0,0 +1,45 @@
{
"experiment": "llm_curriculum_v5",
"master_seed": 3,
"git_commit": null,
"python": "3.11.7",
"libraries": {
"numpy": "2.4.6",
"scipy": "1.17.1",
"pandas": "3.0.3",
"pyarrow": "24.0.0",
"torch": "2.12.1",
"transformers": "5.16.1",
"peft": "0.20.0"
},
"rows": 205,
"results_sha256": "14ef9eeaa96b4d9345e14094cc9015b5797d2c08e78ead3c0f014ac6b39e1846",
"layer": "2",
"tier": "llm",
"base_model": "Qwen/Qwen2.5-1.5B",
"hard": false,
"curriculum": {
"families": [
"mnli",
"arc",
"hellaswag",
"squad",
"boolq",
"winogrande"
],
"lineages": 3,
"generations": 6,
"arms": [
"society_dry"
],
"baselines": [],
"n_new": 300,
"n_replay": 150,
"operator": "linear",
"ancestor_depth": 3,
"lora": {
"r": 16,
"alpha": 32
}
}
}

View file

@ -0,0 +1,59 @@
experiment: llm_curriculum_v5
seed: 3
n_replicates: 1
source_config:
experiment: llm_curriculum_v5
kind: llm_curriculum
base_model: Qwen/Qwen2.5-1.5B
seed: 3
families:
- mnli
- arc
- hellaswag
- squad
- boolq
- winogrande
lineages: 3
generations: 6
arms:
- society_dry
baselines: []
n_new: 300
n_replay: 150
n_test: 60
n_val: 20
epochs: 3
lr: 0.0001
ancestor_depth: 3
operator: linear
merge_weights:
- - 0.5
- 0.5
- - 0.3
- 0.7
- - 0.7
- 0.3
baseline_weights:
- - 0.333
- 0.333
- 0.334
- - 0.5
- 0.25
- 0.25
- - 0.25
- 0.5
- 0.25
- - 0.25
- 0.25
- 0.5
max_new_tokens: 48
batch_size: 48
train_batch_size: 4
train_max_len: 512
resume: true
lora:
r: 16
alpha: 32
output:
dir: results/llm_curriculum_v5/s3_society_dry
n_replicates: 1