Spaces:
Build error
Build error
Download app.py from Dhanrajtz5rt/Qwen3.5-0.8B-Chatbot: direct link, hf CLI and curl.
- Browser
- Download file 13.3 kB
-
https://huggingface.co/spaces/Dhanrajtz5rt/Qwen3.5-0.8B-Chatbot/resolve/main/app.py
- Command line
-
hf download hf://spaces/Dhanrajtz5rt/Qwen3.5-0.8B-Chatbot/app.py
-
curl -L -o app.py https://huggingface.co/spaces/Dhanrajtz5rt/Qwen3.5-0.8B-Chatbot/resolve/main/app.py
13.3 kB
| """ | |
| Qwen3.5-0.8B AI Chatbot Backend | |
| ================================ | |
| Premium Claude-like chatbot with: | |
| - Qwen3.5-0.8B multimodal model (FP16) | |
| - Thinking mode display | |
| - Image + video input | |
| - Adjustable context window (1k - 252k tokens) | |
| - Custom system prompt | |
| - Streaming responses | |
| """ | |
| import os | |
| import re | |
| import base64 | |
| import tempfile | |
| import asyncio | |
| import json | |
| from pathlib import Path | |
| from typing import Optional, List, Generator | |
| import torch | |
| from fastapi import FastAPI, File, UploadFile, Form, HTTPException | |
| from fastapi.responses import StreamingResponse, HTMLResponse | |
| from fastapi.staticfiles import StaticFiles | |
| from fastapi.middleware.cors import CORSMiddleware | |
| from pydantic import BaseModel | |
| import uvicorn | |
| # βββ Model Loading ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| print("π Loading Qwen3.5-0.8B model in FP16...") | |
| MODEL_ID = "Qwen/Qwen3.5-0.8B" | |
| DEVICE = "cpu" # HuggingFace Spaces CPU tier | |
| from transformers import AutoModelForCausalLM, AutoTokenizer, AutoProcessor, TextIteratorStreamer | |
| from threading import Thread | |
| # Try to load as VL (multimodal) first, fall back to text-only | |
| try: | |
| from transformers import Qwen3VLForConditionalGeneration | |
| from qwen_vl_utils import process_vision_info | |
| processor = AutoProcessor.from_pretrained(MODEL_ID, trust_remote_code=True) | |
| model = Qwen3VLForConditionalGeneration.from_pretrained( | |
| MODEL_ID, | |
| torch_dtype=torch.float16, | |
| device_map=DEVICE, | |
| trust_remote_code=True, | |
| low_cpu_mem_usage=True, | |
| ) | |
| MULTIMODAL = True | |
| print("β Loaded as multimodal VL model") | |
| except Exception as e: | |
| print(f"β οΈ VL load failed ({e}), loading as text-only...") | |
| from transformers import AutoModelForCausalLM, AutoTokenizer | |
| tokenizer = AutoTokenizer.from_pretrained(MODEL_ID, trust_remote_code=True) | |
| model = AutoModelForCausalLM.from_pretrained( | |
| MODEL_ID, | |
| torch_dtype=torch.float16, | |
| device_map=DEVICE, | |
| trust_remote_code=True, | |
| low_cpu_mem_usage=True, | |
| ) | |
| processor = tokenizer | |
| MULTIMODAL = False | |
| print("β Loaded as text-only model") | |
| model.eval() | |
| print(f"β Model ready! Multimodal: {MULTIMODAL} | Device: {DEVICE} | dtype: torch.float16") | |
| # βββ FastAPI App βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| app = FastAPI(title="Qwen3.5 Chatbot", version="1.0.0") | |
| app.add_middleware( | |
| CORSMiddleware, | |
| allow_origins=["*"], | |
| allow_methods=["*"], | |
| allow_headers=["*"], | |
| ) | |
| # Serve static files | |
| STATIC_DIR = Path(__file__).parent / "static" | |
| app.mount("/static", StaticFiles(directory=str(STATIC_DIR)), name="static") | |
| UPLOAD_DIR = Path(__file__).parent / "uploads" | |
| UPLOAD_DIR.mkdir(exist_ok=True) | |
| # βββ Pydantic Models βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class Message(BaseModel): | |
| role: str # "user" | "assistant" | "system" | |
| content: str | |
| class ChatRequest(BaseModel): | |
| messages: List[Message] | |
| system_prompt: Optional[str] = "You are a helpful, harmless, and honest AI assistant." | |
| max_context_tokens: Optional[int] = 8192 | |
| enable_thinking: Optional[bool] = True | |
| temperature: Optional[float] = 0.6 | |
| top_p: Optional[float] = 0.95 | |
| top_k: Optional[int] = 20 | |
| max_new_tokens: Optional[int] = 2048 | |
| # βββ Helper: Parse Thinking βββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def parse_thinking(text: str): | |
| """Split response into thinking content and final answer.""" | |
| think_match = re.search(r"<think>(.*?)</think>", text, re.DOTALL) | |
| if think_match: | |
| thinking = think_match.group(1).strip() | |
| answer = text[think_match.end():].strip() | |
| return thinking, answer | |
| return None, text.strip() | |
| # βββ Helper: Build Messages βββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def build_messages(req: ChatRequest, image_data=None, video_path=None): | |
| """Build messages list for the model.""" | |
| messages = [] | |
| if req.system_prompt: | |
| messages.append({"role": "system", "content": req.system_prompt}) | |
| for i, msg in enumerate(req.messages): | |
| if msg.role == "system": | |
| continue # Already handled | |
| if msg.role == "user" and i == len(req.messages) - 1: | |
| # Last user message β possibly attach media | |
| if MULTIMODAL and (image_data or video_path): | |
| content = [] | |
| if image_data: | |
| for img in image_data: | |
| content.append({ | |
| "type": "image", | |
| "image": f"data:image/jpeg;base64,{img}" | |
| }) | |
| if video_path: | |
| content.append({ | |
| "type": "video", | |
| "video": video_path, | |
| "fps": 1.0, | |
| "max_pixels": 360 * 420 | |
| }) | |
| content.append({"type": "text", "text": msg.content}) | |
| messages.append({"role": "user", "content": content}) | |
| else: | |
| messages.append({"role": "user", "content": msg.content}) | |
| else: | |
| messages.append({"role": msg.role, "content": msg.content}) | |
| return messages | |
| # βββ Streaming Text Generator βββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def generate_stream(req: ChatRequest, messages: list): | |
| """Generate streaming response from model.""" | |
| # Apply chat template | |
| if MULTIMODAL and hasattr(processor, 'apply_chat_template'): | |
| try: | |
| image_inputs, video_inputs = process_vision_info(messages) | |
| text = processor.apply_chat_template( | |
| messages, | |
| tokenize=False, | |
| add_generation_prompt=True, | |
| enable_thinking=req.enable_thinking | |
| ) | |
| inputs = processor( | |
| text=[text], | |
| images=image_inputs, | |
| videos=video_inputs, | |
| padding=True, | |
| return_tensors="pt" | |
| ) | |
| except Exception: | |
| # Fallback: text-only processing | |
| text = processor.apply_chat_template( | |
| messages, | |
| tokenize=False, | |
| add_generation_prompt=True, | |
| enable_thinking=req.enable_thinking | |
| ) | |
| inputs = processor(text=[text], return_tensors="pt") | |
| else: | |
| # Text-only tokenizer | |
| text = processor.apply_chat_template( | |
| messages, | |
| tokenize=False, | |
| add_generation_prompt=True, | |
| enable_thinking=req.enable_thinking | |
| ) | |
| inputs = processor([text], return_tensors="pt") | |
| # Truncate to context window | |
| max_input = min(req.max_context_tokens - req.max_new_tokens, inputs["input_ids"].shape[1]) | |
| for key in inputs: | |
| if hasattr(inputs[key], '__len__') and inputs[key].dim() > 1: | |
| inputs[key] = inputs[key][:, -max_input:] | |
| inputs = {k: v.to(DEVICE) for k, v in inputs.items()} | |
| streamer = TextIteratorStreamer( | |
| processor if not MULTIMODAL else processor.tokenizer, | |
| skip_prompt=True, | |
| skip_special_tokens=True | |
| ) | |
| gen_kwargs = { | |
| **inputs, | |
| "streamer": streamer, | |
| "max_new_tokens": req.max_new_tokens, | |
| "temperature": req.temperature, | |
| "top_p": req.top_p, | |
| "top_k": req.top_k, | |
| "do_sample": True, | |
| } | |
| thread = Thread(target=model.generate, kwargs=gen_kwargs) | |
| thread.start() | |
| return streamer, thread | |
| # βββ API Routes βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| async def root(): | |
| html_path = STATIC_DIR / "index.html" | |
| return HTMLResponse(content=html_path.read_text(), status_code=200) | |
| async def get_info(): | |
| return { | |
| "model": MODEL_ID, | |
| "multimodal": MULTIMODAL, | |
| "device": DEVICE, | |
| "dtype": "float16", | |
| "max_context": 262144, | |
| } | |
| async def chat_stream( | |
| messages: str = Form(...), | |
| system_prompt: str = Form("You are a helpful, harmless, and honest AI assistant."), | |
| max_context_tokens: int = Form(8192), | |
| enable_thinking: bool = Form(True), | |
| temperature: float = Form(0.6), | |
| top_p: float = Form(0.95), | |
| top_k: int = Form(20), | |
| max_new_tokens: int = Form(2048), | |
| images: List[UploadFile] = File(default=[]), | |
| video: Optional[UploadFile] = File(default=None), | |
| ): | |
| """Streaming chat endpoint with optional image/video attachments.""" | |
| # Parse messages JSON | |
| try: | |
| msgs_data = json.loads(messages) | |
| req = ChatRequest( | |
| messages=[Message(**m) for m in msgs_data], | |
| system_prompt=system_prompt, | |
| max_context_tokens=max_context_tokens, | |
| enable_thinking=enable_thinking, | |
| temperature=temperature, | |
| top_p=top_p, | |
| top_k=top_k, | |
| max_new_tokens=max_new_tokens, | |
| ) | |
| except Exception as e: | |
| raise HTTPException(status_code=400, detail=f"Invalid request: {e}") | |
| # Process images | |
| image_data = [] | |
| for img_file in images: | |
| if img_file.filename: | |
| raw = await img_file.read() | |
| image_data.append(base64.b64encode(raw).decode()) | |
| # Process video | |
| video_path = None | |
| video_temp = None | |
| if video and video.filename: | |
| video_temp = tempfile.NamedTemporaryFile( | |
| delete=False, suffix=Path(video.filename).suffix, dir=str(UPLOAD_DIR) | |
| ) | |
| video_temp.write(await video.read()) | |
| video_temp.close() | |
| video_path = video_temp.name | |
| # Build messages for model | |
| model_messages = build_messages(req, image_data or None, video_path) | |
| async def stream_generator(): | |
| full_text = "" | |
| thinking_sent = False | |
| answer_started = False | |
| thinking_buffer = "" | |
| answer_buffer = "" | |
| in_think = False | |
| try: | |
| streamer, thread = generate_stream(req, model_messages) | |
| for token in streamer: | |
| full_text += token | |
| # Real-time parse of <think>...</think> | |
| if "<think>" in full_text and not in_think: | |
| in_think = True | |
| yield f"data: {json.dumps({'type': 'thinking_start'})}\n\n" | |
| continue | |
| if in_think: | |
| if "</think>" in full_text: | |
| in_think = False | |
| # Extract remaining thinking content | |
| think_end = full_text.index("</think>") | |
| think_content = full_text[full_text.index("<think>")+7:think_end] | |
| yield f"data: {json.dumps({'type': 'thinking_chunk', 'content': token})}\n\n" | |
| yield f"data: {json.dumps({'type': 'thinking_end'})}\n\n" | |
| else: | |
| yield f"data: {json.dumps({'type': 'thinking_chunk', 'content': token})}\n\n" | |
| continue | |
| # Answer content | |
| yield f"data: {json.dumps({'type': 'answer_chunk', 'content': token})}\n\n" | |
| thread.join() | |
| yield f"data: {json.dumps({'type': 'done'})}\n\n" | |
| except Exception as e: | |
| yield f"data: {json.dumps({'type': 'error', 'content': str(e)})}\n\n" | |
| finally: | |
| if video_path and os.path.exists(video_path): | |
| os.unlink(video_path) | |
| return StreamingResponse( | |
| stream_generator(), | |
| media_type="text/event-stream", | |
| headers={ | |
| "Cache-Control": "no-cache", | |
| "X-Accel-Buffering": "no", | |
| } | |
| ) | |
| async def chat_simple(req: ChatRequest): | |
| """Non-streaming chat endpoint.""" | |
| model_messages = build_messages(req) | |
| streamer, thread = generate_stream(req, model_messages) | |
| full_text = "" | |
| for token in streamer: | |
| full_text += token | |
| thread.join() | |
| thinking, answer = parse_thinking(full_text) | |
| return { | |
| "thinking": thinking, | |
| "answer": answer, | |
| "full_text": full_text, | |
| } | |
| # βββ Main ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| if __name__ == "__main__": | |
| uvicorn.run(app, host="0.0.0.0", port=7860, log_level="info") | |