psikosen's picture
Release verified v7 candidate with native inference and scoped evaluations
73d596d verified
Raw History Blame Contribute Delete
1.14 kB
from pathlib import Path
import argparse,json,torch
from safetensors.torch import load_file
from transformers import AutoTokenizer
from canopy_r3 import CanopyConfig,CanopyForCausalLM
p=argparse.ArgumentParser();p.add_argument("prompt");args=p.parse_args()
root=Path(__file__).resolve().parent
torch.set_num_threads(2)
device="cuda" if torch.cuda.is_available() else "cpu"
model=CanopyForCausalLM(CanopyConfig(**json.loads((root/"native_config.json").read_text())))
model.load_state_dict(load_file(str(root/"model.safetensors")),strict=True)
model=model.to(device=device,dtype=torch.bfloat16).eval()
tokenizer=AutoTokenizer.from_pretrained(root,local_files_only=True)
ids=tokenizer.encode("\nUser: "+args.prompt+"\nAssistant: ",add_special_tokens=False,return_tensors="pt").to(device)
with torch.inference_mode():
out=model.generate(ids,max_new_tokens=32,temperature=0.,repetition_penalty=1.,ptrm_stochastic_scale=0.,eos_token_id=0,enable_prefix_sliding=False)
tokens=out[0,ids.shape[1]:].tolist()
end=next((i for i,t in enumerate(tokens) if t in (0,2)),len(tokens))
print(tokenizer.decode(tokens[:end],skip_special_tokens=True).strip())