Spaces:
Sleeping
Sleeping
Download app.py from harishforaiandml/sarcastic-bot-demo: direct link, hf CLI and curl.
- Browser
- Download file 2.61 kB
-
https://huggingface.co/spaces/harishforaiandml/sarcastic-bot-demo/resolve/main/app.py
- Command line
-
hf download hf://spaces/harishforaiandml/sarcastic-bot-demo/app.py
-
curl -L -o app.py https://huggingface.co/spaces/harishforaiandml/sarcastic-bot-demo/resolve/main/app.py
2.61 kB
| import os | |
| import gradio as gr | |
| import torch | |
| from transformers import AutoModelForCausalLM, AutoTokenizer | |
| from peft import PeftModel | |
| # 1. Base Model and custom adapter path | |
| BASE_MODEL = "meta-llama/Llama-3.2-1B-Instruct" | |
| ADAPTER_MODEL = "harishforaiandml/llama-3.2-1b-sarcastic-bot" | |
| # Retrieve token for access to Llama base weights | |
| HF_TOKEN = os.getenv("HF_TOKEN") | |
| print("Loading tokenizer...") | |
| tokenizer = AutoTokenizer.from_pretrained(BASE_MODEL, token=HF_TOKEN) | |
| tokenizer.pad_token = tokenizer.eos_token | |
| print("Loading base model (this may take a minute)...") | |
| # Load base model in low-memory CPU mode | |
| base_model = AutoModelForCausalLM.from_pretrained( | |
| BASE_MODEL, | |
| torch_dtype=torch.float32, | |
| device_map="cpu", | |
| token=HF_TOKEN | |
| ) | |
| print("Injecting custom sarcastic LoRA adapters...") | |
| # This layers your fine-tuned 5,000 row personality directly onto the base brain | |
| model = PeftModel.from_pretrained(base_model, ADAPTER_MODEL, token=HF_TOKEN) | |
| model.eval() | |
| print("Model completely loaded locally!") | |
| def respond(message, chat_history): | |
| # Construct standard Llama-3 conversation format | |
| messages = [ | |
| {"role": "system", "content": "You are a sarcastic, snarky robot assistant."} | |
| ] | |
| for msg in chat_history: | |
| messages.append({"role": msg["role"], "content": msg["content"]}) | |
| messages.append({"role": "user", "content": message}) | |
| # Format the conversational structure with Llama tags | |
| prompt = tokenizer.apply_chat_template(messages, tokenize=False, add_generation_prompt=True) | |
| inputs = tokenizer(prompt, return_tensors="pt").to("cpu") | |
| try: | |
| with torch.no_grad(): | |
| outputs = model.generate( | |
| **inputs, | |
| max_new_tokens=150, | |
| temperature=0.7, | |
| do_sample=True, | |
| pad_token_id=tokenizer.eos_token_id | |
| ) | |
| # Isolate the newly generated assistant tokens | |
| input_length = inputs.input_ids.shape[1] | |
| response = tokenizer.decode(outputs[0][input_length:], skip_special_tokens=True) | |
| except Exception as e: | |
| response = f"Error generating text: {str(e)}" | |
| return response | |
| # Build the Gradio frame | |
| with gr.Blocks() as demo: | |
| gr.Markdown("# 🤖 The Snarky Robot Assistant") | |
| gr.Markdown("Go ahead, ask a regular human question. Prepare to be judged.") | |
| gr.ChatInterface( | |
| fn=respond, | |
| textbox=gr.Textbox(placeholder="Type your question here...", container=False, scale=7) | |
| ) | |
| if __name__ == "__main__": | |
| demo.launch(theme=gr.themes.Soft()) |