From db4ec5bb3839bc5cc50d82e427848595d14b3070 Mon Sep 17 00:00:00 2001 From: Void Agent Date: Wed, 29 Jul 2026 17:47:48 +0100 Subject: Restructure: nanoGPT at root, custom code in src/ - Move model.py, train.py, configurator.py to root for nanoGPT compatibility - data/ and config/ directories at root with Shakespeare dataset prep scripts - src/jlens.py updated to import model from project root - Cleaned up stale src/config/ and duplicate src/ files - Fixed .gitignore: exclude out-shakespeare-char/ instead of raw data dirs --- config/finetune_shakespeare.py | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) create mode 100644 config/finetune_shakespeare.py (limited to 'config/finetune_shakespeare.py') diff --git a/config/finetune_shakespeare.py b/config/finetune_shakespeare.py new file mode 100644 index 0000000..148a4c4 --- /dev/null +++ b/config/finetune_shakespeare.py @@ -0,0 +1,25 @@ +import time + +out_dir = 'out-shakespeare' +eval_interval = 5 +eval_iters = 40 +wandb_log = False # feel free to turn on +wandb_project = 'shakespeare' +wandb_run_name = 'ft-' + str(time.time()) + +dataset = 'shakespeare' +init_from = 'gpt2-xl' # this is the largest GPT-2 model + +# only save checkpoints if the validation loss improves +always_save_checkpoint = False + +# the number of examples per iter: +# 1 batch_size * 32 grad_accum * 1024 tokens = 32,768 tokens/iter +# shakespeare has 301,966 tokens, so 1 epoch ~= 9.2 iters +batch_size = 1 +gradient_accumulation_steps = 32 +max_iters = 20 + +# finetune at constant LR +learning_rate = 3e-5 +decay_lr = False -- cgit v1.2.3