""" Qwen3.5-0.8B AI Chatbot Backend ================================ Premium Claude-like chatbot with: - Qwen3.5-0.8B multimodal model (FP16) - Thinking mode display - Image + video input - Adjustable context window (1k - 252k tokens) - Custom system prompt - Streaming responses """ import os import re import base64 import tempfile import asyncio import json from pathlib import Path from typing import Optional, List, Generator import torch from fastapi import FastAPI, File, UploadFile, Form, HTTPException from fastapi.responses import StreamingResponse, HTMLResponse from fastapi.staticfiles import StaticFiles from fastapi.middleware.cors import CORSMiddleware from pydantic import BaseModel import uvicorn # ─── Model Loading ──────────────────────────────────────────────────────────── print("🚀 Loading Qwen3.5-0.8B model in FP16...") MODEL_ID = "Qwen/Qwen3.5-0.8B" DEVICE = "cpu" # HuggingFace Spaces CPU tier from transformers import AutoModelForCausalLM, AutoTokenizer, AutoProcessor, TextIteratorStreamer from threading import Thread # Try to load as VL (multimodal) first, fall back to text-only try: from transformers import Qwen3VLForConditionalGeneration from qwen_vl_utils import process_vision_info processor = AutoProcessor.from_pretrained(MODEL_ID, trust_remote_code=True) model = Qwen3VLForConditionalGeneration.from_pretrained( MODEL_ID, torch_dtype=torch.float16, device_map=DEVICE, trust_remote_code=True, low_cpu_mem_usage=True, ) MULTIMODAL = True print("✅ Loaded as multimodal VL model") except Exception as e: print(f"⚠️ VL load failed ({e}), loading as text-only...") from transformers import AutoModelForCausalLM, AutoTokenizer tokenizer = AutoTokenizer.from_pretrained(MODEL_ID, trust_remote_code=True) model = AutoModelForCausalLM.from_pretrained( MODEL_ID, torch_dtype=torch.float16, device_map=DEVICE, trust_remote_code=True, low_cpu_mem_usage=True, ) processor = tokenizer MULTIMODAL = False print("✅ Loaded as text-only model") model.eval() print(f"✅ Model ready! Multimodal: {MULTIMODAL} | Device: {DEVICE} | dtype: torch.float16") # ─── FastAPI App ─────────────────────────────────────────────────────────────── app = FastAPI(title="Qwen3.5 Chatbot", version="1.0.0") app.add_middleware( CORSMiddleware, allow_origins=["*"], allow_methods=["*"], allow_headers=["*"], ) # Serve static files STATIC_DIR = Path(__file__).parent / "static" app.mount("/static", StaticFiles(directory=str(STATIC_DIR)), name="static") UPLOAD_DIR = Path(__file__).parent / "uploads" UPLOAD_DIR.mkdir(exist_ok=True) # ─── Pydantic Models ─────────────────────────────────────────────────────────── class Message(BaseModel): role: str # "user" | "assistant" | "system" content: str class ChatRequest(BaseModel): messages: List[Message] system_prompt: Optional[str] = "You are a helpful, harmless, and honest AI assistant." max_context_tokens: Optional[int] = 8192 enable_thinking: Optional[bool] = True temperature: Optional[float] = 0.6 top_p: Optional[float] = 0.95 top_k: Optional[int] = 20 max_new_tokens: Optional[int] = 2048 # ─── Helper: Parse Thinking ─────────────────────────────────────────────────── def parse_thinking(text: str): """Split response into thinking content and final answer.""" think_match = re.search(r"(.*?)", text, re.DOTALL) if think_match: thinking = think_match.group(1).strip() answer = text[think_match.end():].strip() return thinking, answer return None, text.strip() # ─── Helper: Build Messages ─────────────────────────────────────────────────── def build_messages(req: ChatRequest, image_data=None, video_path=None): """Build messages list for the model.""" messages = [] if req.system_prompt: messages.append({"role": "system", "content": req.system_prompt}) for i, msg in enumerate(req.messages): if msg.role == "system": continue # Already handled if msg.role == "user" and i == len(req.messages) - 1: # Last user message — possibly attach media if MULTIMODAL and (image_data or video_path): content = [] if image_data: for img in image_data: content.append({ "type": "image", "image": f"data:image/jpeg;base64,{img}" }) if video_path: content.append({ "type": "video", "video": video_path, "fps": 1.0, "max_pixels": 360 * 420 }) content.append({"type": "text", "text": msg.content}) messages.append({"role": "user", "content": content}) else: messages.append({"role": "user", "content": msg.content}) else: messages.append({"role": msg.role, "content": msg.content}) return messages # ─── Streaming Text Generator ───────────────────────────────────────────────── def generate_stream(req: ChatRequest, messages: list): """Generate streaming response from model.""" # Apply chat template if MULTIMODAL and hasattr(processor, 'apply_chat_template'): try: image_inputs, video_inputs = process_vision_info(messages) text = processor.apply_chat_template( messages, tokenize=False, add_generation_prompt=True, enable_thinking=req.enable_thinking ) inputs = processor( text=[text], images=image_inputs, videos=video_inputs, padding=True, return_tensors="pt" ) except Exception: # Fallback: text-only processing text = processor.apply_chat_template( messages, tokenize=False, add_generation_prompt=True, enable_thinking=req.enable_thinking ) inputs = processor(text=[text], return_tensors="pt") else: # Text-only tokenizer text = processor.apply_chat_template( messages, tokenize=False, add_generation_prompt=True, enable_thinking=req.enable_thinking ) inputs = processor([text], return_tensors="pt") # Truncate to context window max_input = min(req.max_context_tokens - req.max_new_tokens, inputs["input_ids"].shape[1]) for key in inputs: if hasattr(inputs[key], '__len__') and inputs[key].dim() > 1: inputs[key] = inputs[key][:, -max_input:] inputs = {k: v.to(DEVICE) for k, v in inputs.items()} streamer = TextIteratorStreamer( processor if not MULTIMODAL else processor.tokenizer, skip_prompt=True, skip_special_tokens=True ) gen_kwargs = { **inputs, "streamer": streamer, "max_new_tokens": req.max_new_tokens, "temperature": req.temperature, "top_p": req.top_p, "top_k": req.top_k, "do_sample": True, } thread = Thread(target=model.generate, kwargs=gen_kwargs) thread.start() return streamer, thread # ─── API Routes ─────────────────────────────────────────────────────────────── @app.get("/", response_class=HTMLResponse) async def root(): html_path = STATIC_DIR / "index.html" return HTMLResponse(content=html_path.read_text(), status_code=200) @app.get("/api/info") async def get_info(): return { "model": MODEL_ID, "multimodal": MULTIMODAL, "device": DEVICE, "dtype": "float16", "max_context": 262144, } @app.post("/api/chat/stream") async def chat_stream( messages: str = Form(...), system_prompt: str = Form("You are a helpful, harmless, and honest AI assistant."), max_context_tokens: int = Form(8192), enable_thinking: bool = Form(True), temperature: float = Form(0.6), top_p: float = Form(0.95), top_k: int = Form(20), max_new_tokens: int = Form(2048), images: List[UploadFile] = File(default=[]), video: Optional[UploadFile] = File(default=None), ): """Streaming chat endpoint with optional image/video attachments.""" # Parse messages JSON try: msgs_data = json.loads(messages) req = ChatRequest( messages=[Message(**m) for m in msgs_data], system_prompt=system_prompt, max_context_tokens=max_context_tokens, enable_thinking=enable_thinking, temperature=temperature, top_p=top_p, top_k=top_k, max_new_tokens=max_new_tokens, ) except Exception as e: raise HTTPException(status_code=400, detail=f"Invalid request: {e}") # Process images image_data = [] for img_file in images: if img_file.filename: raw = await img_file.read() image_data.append(base64.b64encode(raw).decode()) # Process video video_path = None video_temp = None if video and video.filename: video_temp = tempfile.NamedTemporaryFile( delete=False, suffix=Path(video.filename).suffix, dir=str(UPLOAD_DIR) ) video_temp.write(await video.read()) video_temp.close() video_path = video_temp.name # Build messages for model model_messages = build_messages(req, image_data or None, video_path) async def stream_generator(): full_text = "" thinking_sent = False answer_started = False thinking_buffer = "" answer_buffer = "" in_think = False try: streamer, thread = generate_stream(req, model_messages) for token in streamer: full_text += token # Real-time parse of ... if "" in full_text and not in_think: in_think = True yield f"data: {json.dumps({'type': 'thinking_start'})}\n\n" continue if in_think: if "" in full_text: in_think = False # Extract remaining thinking content think_end = full_text.index("") think_content = full_text[full_text.index("")+7:think_end] yield f"data: {json.dumps({'type': 'thinking_chunk', 'content': token})}\n\n" yield f"data: {json.dumps({'type': 'thinking_end'})}\n\n" else: yield f"data: {json.dumps({'type': 'thinking_chunk', 'content': token})}\n\n" continue # Answer content yield f"data: {json.dumps({'type': 'answer_chunk', 'content': token})}\n\n" thread.join() yield f"data: {json.dumps({'type': 'done'})}\n\n" except Exception as e: yield f"data: {json.dumps({'type': 'error', 'content': str(e)})}\n\n" finally: if video_path and os.path.exists(video_path): os.unlink(video_path) return StreamingResponse( stream_generator(), media_type="text/event-stream", headers={ "Cache-Control": "no-cache", "X-Accel-Buffering": "no", } ) @app.post("/api/chat") async def chat_simple(req: ChatRequest): """Non-streaming chat endpoint.""" model_messages = build_messages(req) streamer, thread = generate_stream(req, model_messages) full_text = "" for token in streamer: full_text += token thread.join() thinking, answer = parse_thinking(full_text) return { "thinking": thinking, "answer": answer, "full_text": full_text, } # ─── Main ────────────────────────────────────────────────────────────────────── if __name__ == "__main__": uvicorn.run(app, host="0.0.0.0", port=7860, log_level="info")