import os import gradio as gr import torch from transformers import AutoModelForCausalLM, AutoTokenizer from peft import PeftModel # 1. Base Model and custom adapter path BASE_MODEL = "meta-llama/Llama-3.2-1B-Instruct" ADAPTER_MODEL = "harishforaiandml/llama-3.2-1b-sarcastic-bot" # Retrieve token for access to Llama base weights HF_TOKEN = os.getenv("HF_TOKEN") print("Loading tokenizer...") tokenizer = AutoTokenizer.from_pretrained(BASE_MODEL, token=HF_TOKEN) tokenizer.pad_token = tokenizer.eos_token print("Loading base model (this may take a minute)...") # Load base model in low-memory CPU mode base_model = AutoModelForCausalLM.from_pretrained( BASE_MODEL, torch_dtype=torch.float32, device_map="cpu", token=HF_TOKEN ) print("Injecting custom sarcastic LoRA adapters...") # This layers your fine-tuned 5,000 row personality directly onto the base brain model = PeftModel.from_pretrained(base_model, ADAPTER_MODEL, token=HF_TOKEN) model.eval() print("Model completely loaded locally!") def respond(message, chat_history): # Construct standard Llama-3 conversation format messages = [ {"role": "system", "content": "You are a sarcastic, snarky robot assistant."} ] for msg in chat_history: messages.append({"role": msg["role"], "content": msg["content"]}) messages.append({"role": "user", "content": message}) # Format the conversational structure with Llama tags prompt = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True) inputs = tokenizer(prompt, return_tensors="pt").to("cpu") try: with torch.no_grad(): outputs = model.generate( **inputs, max_new_tokens=150, temperature=0.7, do_sample=True, pad_token_id=tokenizer.eos_token_id ) # Isolate the newly generated assistant tokens input_length = inputs.input_ids.shape[1] response = tokenizer.decode(outputs[0][input_length:], skip_special_tokens=True) except Exception as e: response = f"Error generating text: {str(e)}" return response # Build the Gradio frame with gr.Blocks() as demo: gr.Markdown("# 🤖 The Snarky Robot Assistant") gr.Markdown("Go ahead, ask a regular human question. Prepare to be judged.") gr.ChatInterface( fn=respond, textbox=gr.Textbox(placeholder="Type your question here...", container=False, scale=7) ) if __name__ == "__main__": demo.launch(theme=gr.themes.Soft())