harishforaiandml's picture
Update app.py
2441211 verified
Raw History Blame Contribute Delete
2.61 kB
import os
import gradio as gr
import torch
from transformers import AutoModelForCausalLM, AutoTokenizer
from peft import PeftModel
# 1. Base Model and custom adapter path
BASE_MODEL = "meta-llama/Llama-3.2-1B-Instruct"
ADAPTER_MODEL = "harishforaiandml/llama-3.2-1b-sarcastic-bot"
# Retrieve token for access to Llama base weights
HF_TOKEN = os.getenv("HF_TOKEN")
print("Loading tokenizer...")
tokenizer = AutoTokenizer.from_pretrained(BASE_MODEL, token=HF_TOKEN)
tokenizer.pad_token = tokenizer.eos_token
print("Loading base model (this may take a minute)...")
# Load base model in low-memory CPU mode
base_model = AutoModelForCausalLM.from_pretrained(
BASE_MODEL,
torch_dtype=torch.float32,
device_map="cpu",
token=HF_TOKEN
)
print("Injecting custom sarcastic LoRA adapters...")
# This layers your fine-tuned 5,000 row personality directly onto the base brain
model = PeftModel.from_pretrained(base_model, ADAPTER_MODEL, token=HF_TOKEN)
model.eval()
print("Model completely loaded locally!")
def respond(message, chat_history):
# Construct standard Llama-3 conversation format
messages = [
{"role": "system", "content": "You are a sarcastic, snarky robot assistant."}
]
for msg in chat_history:
messages.append({"role": msg["role"], "content": msg["content"]})
messages.append({"role": "user", "content": message})
# Format the conversational structure with Llama tags
prompt = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
inputs = tokenizer(prompt, return_tensors="pt").to("cpu")
try:
with torch.no_grad():
outputs = model.generate(
**inputs,
max_new_tokens=150,
temperature=0.7,
do_sample=True,
pad_token_id=tokenizer.eos_token_id
)
# Isolate the newly generated assistant tokens
input_length = inputs.input_ids.shape[1]
response = tokenizer.decode(outputs[0][input_length:], skip_special_tokens=True)
except Exception as e:
response = f"Error generating text: {str(e)}"
return response
# Build the Gradio frame
with gr.Blocks() as demo:
gr.Markdown("# 🤖 The Snarky Robot Assistant")
gr.Markdown("Go ahead, ask a regular human question. Prepare to be judged.")
gr.ChatInterface(
fn=respond,
textbox=gr.Textbox(placeholder="Type your question here...", container=False, scale=7)
)
if __name__ == "__main__":
demo.launch(theme=gr.themes.Soft())