nanoGPT (Tiny Shakespeare)
A 10.7M parameter character-level GPT trained from scratch on the Tiny Shakespeare dataset.
Training Details
- Architecture: 6 layers, 6 heads, 384 embedding dimensions
- Context Length: 256 tokens
- Training Steps: 5,000 iterations
- Best Validation Loss: 1.4646 (reached at step 1500)
- Precision: Mixed Precision (FP16) on Tesla T4 with PyTorch 2.0 compile
How to Use
import os
import sys
import json
import torch
from safetensors.torch import load_model
from huggingface_hub import hf_hub_download
# 1. Download model files
repo_id = "codeby-hp/nanogpt-shakespeare"
weights_file = hf_hub_download(repo_id=repo_id, filename="model.safetensors")
config_file = hf_hub_download(repo_id=repo_id, filename="config.json")
vocab_file = hf_hub_download(repo_id=repo_id, filename="vocab.json")
# Download model.py and add its folder to sys.path so we can import it!
model_code_file = hf_hub_download(repo_id=repo_id, filename="model.py")
sys.path.append(os.path.dirname(model_code_file))
from model import GPTConfig, GPT
# 2. Load config and vocabulary
with open(config_file, "r") as f:
cfg_dict = json.load(f)
with open(vocab_file, "r") as f:
vocab = json.load(f)
itos = {int(k): v for k, v in vocab["itos"].items()}
decode = lambda l: "".join([itos[i] for i in l])
stoi = vocab["stoi"]
encode = lambda s: [stoi[c] for c in s]
# 3. Instantiate model
config = GPTConfig(**cfg_dict)
model = GPT(config)
load_model(model, weights_file)
model.eval()
# 4. Generate text
my_prompt = "ROMEO:\nMy Lord"
context_list = encode(my_prompt)
context = torch.tensor([context_list], dtype=torch.long)
out = model.generate(context, max_new_tokens=300, temperature=0.8, top_k=20)
print(decode(out[0].tolist()))
- Downloads last month
- 57