summaryrefslogtreecommitdiff
path: root/src/config/train_gpt2.py
diff options
context:
space:
mode:
authorVoid Agent <void@jayrup.hermes>2026-07-29 17:47:48 +0100
committerVoid Agent <void@jayrup.hermes>2026-07-29 17:47:48 +0100
commitdb4ec5bb3839bc5cc50d82e427848595d14b3070 (patch)
treed2d9fb11e11392077a79e6a10c38d44d1d9f5a38 /src/config/train_gpt2.py
parent66f99ee30087a5f28ad852e581a0334c7f556091 (diff)
Restructure: nanoGPT at root, custom code in src/
- Move model.py, train.py, configurator.py to root for nanoGPT compatibility - data/ and config/ directories at root with Shakespeare dataset prep scripts - src/jlens.py updated to import model from project root - Cleaned up stale src/config/ and duplicate src/ files - Fixed .gitignore: exclude out-shakespeare-char/ instead of raw data dirs
Diffstat (limited to 'src/config/train_gpt2.py')
-rw-r--r--src/config/train_gpt2.py25
1 files changed, 0 insertions, 25 deletions
diff --git a/src/config/train_gpt2.py b/src/config/train_gpt2.py
deleted file mode 100644
index 8f19273..0000000
--- a/src/config/train_gpt2.py
+++ /dev/null
@@ -1,25 +0,0 @@
-# config for training GPT-2 (124M) down to very nice loss of ~2.85 on 1 node of 8X A100 40GB
-# launch as the following (e.g. in a screen session) and wait ~5 days:
-# $ torchrun --standalone --nproc_per_node=8 train.py config/train_gpt2.py
-
-wandb_log = True
-wandb_project = 'owt'
-wandb_run_name='gpt2-124M'
-
-# these make the total batch size be ~0.5M
-# 12 batch size * 1024 block size * 5 gradaccum * 8 GPUs = 491,520
-batch_size = 12
-block_size = 1024
-gradient_accumulation_steps = 5 * 8
-
-# this makes total number of tokens be 300B
-max_iters = 600000
-lr_decay_iters = 600000
-
-# eval stuff
-eval_interval = 1000
-eval_iters = 200
-log_interval = 10
-
-# weight decay
-weight_decay = 1e-1