Fable-Inject / test_recovered.py
CompressedGemma's picture
Upload 3 files
ddf2414 verified
Raw
History Blame Contribute Delete
1.4 kB
#!/usr/bin/env python3
"""Quick test: load recovered model, generate a few tokens, check sanity."""
import torch
import gc
gc.collect(); torch.cuda.empty_cache()
from transformers import AutoTokenizer, AutoModelForCausalLM, BitsAndBytesConfig
MODEL_PATH = "/home/none/Documents/HPC-Quantize/Ornith-1.0-9B-hpc"
bnb = BitsAndBytesConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4",
bnb_4bit_use_double_quant=True, bnb_4bit_compute_dtype=torch.bfloat16)
print("Loading model in 4-bit...")
tok = AutoTokenizer.from_pretrained(MODEL_PATH, trust_remote_code=True)
tok.padding_side = "right"
if tok.pad_token is None: tok.pad_token = tok.eos_token
model = AutoModelForCausalLM.from_pretrained(
MODEL_PATH, trust_remote_code=True,
quantization_config=bnb, device_map="auto", low_cpu_mem_usage=True,
torch_dtype=torch.bfloat16
)
model.config.use_cache = True
prompts = [
"The capital of France is",
"Once upon a time, in a land far away,",
"Machine learning is a field of study that",
]
print("\n--- Generation test ---")
for p in prompts:
inp = tok(p, return_tensors="pt").to(model.device)
with torch.no_grad():
out = model.generate(**inp, max_new_tokens=32, do_sample=True, temperature=0.7, top_p=0.9)
text = tok.decode(out[0], skip_special_tokens=True)
print(f"\nPrompt: {p}")
print(f"Output: {text}")
print("\nDone.")