"""Hemmingway-1 (27B) chat — transformers + bitsandbytes NF4 on ZeroGPU. ZeroGPU canonical pattern: weights load at import (download happens at app start, no GPU needed); @spaces.GPU attaches the GPU per call. First call also pays NF4 packing; subsequent calls are warm. """ from threading import Thread import spaces # noqa: F401 (ZeroGPU shim; must import before torch) import torch import gradio as gr from transformers import ( AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig, TextIteratorStreamer, ) MODEL_ID = "Altworld/Hemmingway-1" tok = AutoTokenizer.from_pretrained(MODEL_ID) model = AutoModelForCausalLM.from_pretrained( MODEL_ID, quantization_config=BitsAndBytesConfig( load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_use_double_quant=True, bnb_4bit_compute_dtype=torch.bfloat16, ), device_map="auto", dtype=torch.bfloat16, ) model.eval() def _history_to_msgs(history, max_turns=6): msgs = [] for h in list(history)[-max_turns:]: if isinstance(h, dict): role, content = h.get("role"), h.get("content") else: role, content = "user", h[0] if content in (None, ""): continue msgs.append({"role": role, "content": content}) return msgs @spaces.GPU(duration=35) def respond(message, history, temperature, max_new_tokens): msgs = _history_to_msgs(history) + [{"role": "user", "content": message}] prompt = tok.apply_chat_template( msgs, add_generation_prompt=True, tokenize=False, enable_thinking=False ) inputs = tok(prompt, return_tensors="pt").to(model.device) streamer = TextIteratorStreamer(tok, skip_prompt=True, skip_special_tokens=True) Thread( target=model.generate, kwargs=dict( **inputs, streamer=streamer, max_new_tokens=int(max_new_tokens), temperature=float(temperature), top_p=0.9, do_sample=True, ), ).start() out = [] for piece in streamer: out.append(piece) yield "".join(out) CSS = """ footer {display: none;} #hero {max-width: 760px; margin: 0 auto 10px auto;} #hero h1 {font-size: 1.7rem; margin: 0 0 4px 0; letter-spacing: -0.02em;} #hero p {color: #9ca3af; margin: 0; font-size: 0.95rem;} """ with gr.Blocks(title="Hemmingway-1") as demo: with gr.Column(): gr.HTML( "

Hemmingway-1

" "

The AI that writes like a person · 27B · 4-bit NF4 · ZeroGPU

" ) gr.ChatInterface( fn=respond, chatbot=gr.Chatbot(height=440, label="Chat"), textbox=gr.Textbox( placeholder="What should Hemmingway write for you?", submit_btn=True, label="Your request", ), examples=[ ["Write the text I send my landlord about the broken boiler."], ["Write a short birthday message for a colleague I barely know."], ["Schreib eine kurze Nachricht an meinen Vermieter wegen der kaputten Heizung."], ], additional_inputs=[ gr.Slider(0.1, 1.5, value=0.7, step=0.1, label="Temperature"), gr.Slider(64, 768, value=384, step=32, label="Max new tokens"), ], additional_inputs_accordion=gr.Accordion("Settings", open=False), ) if __name__ == "__main__": demo.launch(theme=gr.themes.Monochrome(), css=CSS)