diff --git a/configs/llm/merge_hpc.yaml b/configs/llm/merge_hpc.yaml new file mode 100644 index 0000000..50fd22e --- /dev/null +++ b/configs/llm/merge_hpc.yaml @@ -0,0 +1,20 @@ +experiment: llm_merge_hpc +kind: llm_merge +seed: 1 +n_replicates: 1 + +# Scaled version of configs/llm/merge.yaml for an L40S (48 GB): a bigger, more capable base so the +# specialists decorrelate cleanly and the "recombined model exceeds any single specialist" sign is +# less noisy than the 0.5B local prototype (where it was only marginal on overall accuracy). +# NOTE (code TODOs for the *definitive* run, not yet built): loop several seeds and report mean±CI; +# add more task families; and add a dilution-resistant / offspring-selected ("directed sex") merge. + +base_model: Qwen/Qwen2.5-7B-Instruct # ~15 GB bf16 + LoRA fits the L40S; fallback: 3B-Instruct +families: [lists, strings, arith] +n_train: 800 +n_test: 200 +epochs: 3 +lora: {r: 16, alpha: 32} +merges: [soup, ties] + +output: {dir: results/llm_merge_hpc} diff --git a/hpc/README.md b/hpc/README.md new file mode 100644 index 0000000..e9817cd --- /dev/null +++ b/hpc/README.md @@ -0,0 +1,64 @@ +# Running the Lamarckian Society on Imperial's HPC (CX3, PBS Pro) + +The LLM experiments are the only part that wants more than a laptop GPU. This directory holds the +PBS job scripts for Imperial's **CX3** cluster (scheduler: **PBS Pro** — `qsub`, not Slurm). The +analytic (Layer 1) and small-neural (Layer 1.5) tiers all run locally and need nothing here. + +## Confirmed facts (Imperial RCS user guide) + +- **Submit / monitor / cancel:** `qsub