From 11263aaf0a6abf035f7b1d48f3cc8bf38ad41ccf Mon Sep 17 00:00:00 2001 From: Void Agent Date: Wed, 29 Jul 2026 17:49:17 +0100 Subject: Maxwell GPU fixes: disable bf16 SDPA, force fp32 - model.py: honor config.flash flag (defaults True, False disables SDPA) - config: force dtype=float32 and flash=False for K2200 (compute 5.0) - Maxwell GPUs don't support bf16; SDPA internally uses bf16 operations --- config/train_shakespeare_char.py | 4 ++++ model.py | 3 ++- 2 files changed, 6 insertions(+), 1 deletion(-) diff --git a/config/train_shakespeare_char.py b/config/train_shakespeare_char.py index 41c81df..ec4e5b5 100644 --- a/config/train_shakespeare_char.py +++ b/config/train_shakespeare_char.py @@ -35,3 +35,7 @@ warmup_iters = 100 # not super necessary potentially # on macbook also add # device = 'cpu' # run on cpu only # compile = False # do not torch compile the model + +# Maxwell GPU fix (K2200 — no bf16 support) +dtype = 'float32' # bfloat16 requires compute 8.0+; Maxwell is 5.0 +flash = False # disable SDPA — uses bf16 internally on CUDA 11.8 diff --git a/model.py b/model.py index c698f8b..7071571 100644 --- a/model.py +++ b/model.py @@ -42,7 +42,8 @@ class CausalSelfAttention(nn.Module): self.n_embd = config.n_embd self.dropout = config.dropout # flash attention make GPU go brrrrr but support is only in PyTorch >= 2.0 - self.flash = hasattr(torch.nn.functional, 'scaled_dot_product_attention') + # Maxwell GPUs (compute 5.0) don't support bf16 ops used by SDPA internals + self.flash = hasattr(torch.nn.functional, 'scaled_dot_product_attention') and getattr(config, 'flash', True) if not self.flash: print("WARNING: using slow attention. Flash Attention requires PyTorch >= 2.0") # causal mask to ensure that attention is only applied to the left in the input sequence -- cgit v1.2.3