"""
Qwen3.5-0.8B AI Chatbot Backend
================================
Premium Claude-like chatbot with:
- Qwen3.5-0.8B multimodal model (FP16)
- Thinking mode display
- Image + video input
- Adjustable context window (1k - 252k tokens)
- Custom system prompt
- Streaming responses
"""
import os
import re
import base64
import tempfile
import asyncio
import json
from pathlib import Path
from typing import Optional, List, Generator
import torch
from fastapi import FastAPI, File, UploadFile, Form, HTTPException
from fastapi.responses import StreamingResponse, HTMLResponse
from fastapi.staticfiles import StaticFiles
from fastapi.middleware.cors import CORSMiddleware
from pydantic import BaseModel
import uvicorn
# ─── Model Loading ────────────────────────────────────────────────────────────
print("🚀 Loading Qwen3.5-0.8B model in FP16...")
MODEL_ID = "Qwen/Qwen3.5-0.8B"
DEVICE = "cpu" # HuggingFace Spaces CPU tier
from transformers import AutoModelForCausalLM, AutoTokenizer, AutoProcessor, TextIteratorStreamer
from threading import Thread
# Try to load as VL (multimodal) first, fall back to text-only
try:
from transformers import Qwen3VLForConditionalGeneration
from qwen_vl_utils import process_vision_info
processor = AutoProcessor.from_pretrained(MODEL_ID, trust_remote_code=True)
model = Qwen3VLForConditionalGeneration.from_pretrained(
MODEL_ID,
torch_dtype=torch.float16,
device_map=DEVICE,
trust_remote_code=True,
low_cpu_mem_usage=True,
)
MULTIMODAL = True
print("✅ Loaded as multimodal VL model")
except Exception as e:
print(f"⚠️ VL load failed ({e}), loading as text-only...")
from transformers import AutoModelForCausalLM, AutoTokenizer
tokenizer = AutoTokenizer.from_pretrained(MODEL_ID, trust_remote_code=True)
model = AutoModelForCausalLM.from_pretrained(
MODEL_ID,
torch_dtype=torch.float16,
device_map=DEVICE,
trust_remote_code=True,
low_cpu_mem_usage=True,
)
processor = tokenizer
MULTIMODAL = False
print("✅ Loaded as text-only model")
model.eval()
print(f"✅ Model ready! Multimodal: {MULTIMODAL} | Device: {DEVICE} | dtype: torch.float16")
# ─── FastAPI App ───────────────────────────────────────────────────────────────
app = FastAPI(title="Qwen3.5 Chatbot", version="1.0.0")
app.add_middleware(
CORSMiddleware,
allow_origins=["*"],
allow_methods=["*"],
allow_headers=["*"],
)
# Serve static files
STATIC_DIR = Path(__file__).parent / "static"
app.mount("/static", StaticFiles(directory=str(STATIC_DIR)), name="static")
UPLOAD_DIR = Path(__file__).parent / "uploads"
UPLOAD_DIR.mkdir(exist_ok=True)
# ─── Pydantic Models ───────────────────────────────────────────────────────────
class Message(BaseModel):
role: str # "user" | "assistant" | "system"
content: str
class ChatRequest(BaseModel):
messages: List[Message]
system_prompt: Optional[str] = "You are a helpful, harmless, and honest AI assistant."
max_context_tokens: Optional[int] = 8192
enable_thinking: Optional[bool] = True
temperature: Optional[float] = 0.6
top_p: Optional[float] = 0.95
top_k: Optional[int] = 20
max_new_tokens: Optional[int] = 2048
# ─── Helper: Parse Thinking ───────────────────────────────────────────────────
def parse_thinking(text: str):
"""Split response into thinking content and final answer."""
think_match = re.search(r"(.*?)", text, re.DOTALL)
if think_match:
thinking = think_match.group(1).strip()
answer = text[think_match.end():].strip()
return thinking, answer
return None, text.strip()
# ─── Helper: Build Messages ───────────────────────────────────────────────────
def build_messages(req: ChatRequest, image_data=None, video_path=None):
"""Build messages list for the model."""
messages = []
if req.system_prompt:
messages.append({"role": "system", "content": req.system_prompt})
for i, msg in enumerate(req.messages):
if msg.role == "system":
continue # Already handled
if msg.role == "user" and i == len(req.messages) - 1:
# Last user message — possibly attach media
if MULTIMODAL and (image_data or video_path):
content = []
if image_data:
for img in image_data:
content.append({
"type": "image",
"image": f"data:image/jpeg;base64,{img}"
})
if video_path:
content.append({
"type": "video",
"video": video_path,
"fps": 1.0,
"max_pixels": 360 * 420
})
content.append({"type": "text", "text": msg.content})
messages.append({"role": "user", "content": content})
else:
messages.append({"role": "user", "content": msg.content})
else:
messages.append({"role": msg.role, "content": msg.content})
return messages
# ─── Streaming Text Generator ─────────────────────────────────────────────────
def generate_stream(req: ChatRequest, messages: list):
"""Generate streaming response from model."""
# Apply chat template
if MULTIMODAL and hasattr(processor, 'apply_chat_template'):
try:
image_inputs, video_inputs = process_vision_info(messages)
text = processor.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True,
enable_thinking=req.enable_thinking
)
inputs = processor(
text=[text],
images=image_inputs,
videos=video_inputs,
padding=True,
return_tensors="pt"
)
except Exception:
# Fallback: text-only processing
text = processor.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True,
enable_thinking=req.enable_thinking
)
inputs = processor(text=[text], return_tensors="pt")
else:
# Text-only tokenizer
text = processor.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True,
enable_thinking=req.enable_thinking
)
inputs = processor([text], return_tensors="pt")
# Truncate to context window
max_input = min(req.max_context_tokens - req.max_new_tokens, inputs["input_ids"].shape[1])
for key in inputs:
if hasattr(inputs[key], '__len__') and inputs[key].dim() > 1:
inputs[key] = inputs[key][:, -max_input:]
inputs = {k: v.to(DEVICE) for k, v in inputs.items()}
streamer = TextIteratorStreamer(
processor if not MULTIMODAL else processor.tokenizer,
skip_prompt=True,
skip_special_tokens=True
)
gen_kwargs = {
**inputs,
"streamer": streamer,
"max_new_tokens": req.max_new_tokens,
"temperature": req.temperature,
"top_p": req.top_p,
"top_k": req.top_k,
"do_sample": True,
}
thread = Thread(target=model.generate, kwargs=gen_kwargs)
thread.start()
return streamer, thread
# ─── API Routes ───────────────────────────────────────────────────────────────
@app.get("/", response_class=HTMLResponse)
async def root():
html_path = STATIC_DIR / "index.html"
return HTMLResponse(content=html_path.read_text(), status_code=200)
@app.get("/api/info")
async def get_info():
return {
"model": MODEL_ID,
"multimodal": MULTIMODAL,
"device": DEVICE,
"dtype": "float16",
"max_context": 262144,
}
@app.post("/api/chat/stream")
async def chat_stream(
messages: str = Form(...),
system_prompt: str = Form("You are a helpful, harmless, and honest AI assistant."),
max_context_tokens: int = Form(8192),
enable_thinking: bool = Form(True),
temperature: float = Form(0.6),
top_p: float = Form(0.95),
top_k: int = Form(20),
max_new_tokens: int = Form(2048),
images: List[UploadFile] = File(default=[]),
video: Optional[UploadFile] = File(default=None),
):
"""Streaming chat endpoint with optional image/video attachments."""
# Parse messages JSON
try:
msgs_data = json.loads(messages)
req = ChatRequest(
messages=[Message(**m) for m in msgs_data],
system_prompt=system_prompt,
max_context_tokens=max_context_tokens,
enable_thinking=enable_thinking,
temperature=temperature,
top_p=top_p,
top_k=top_k,
max_new_tokens=max_new_tokens,
)
except Exception as e:
raise HTTPException(status_code=400, detail=f"Invalid request: {e}")
# Process images
image_data = []
for img_file in images:
if img_file.filename:
raw = await img_file.read()
image_data.append(base64.b64encode(raw).decode())
# Process video
video_path = None
video_temp = None
if video and video.filename:
video_temp = tempfile.NamedTemporaryFile(
delete=False, suffix=Path(video.filename).suffix, dir=str(UPLOAD_DIR)
)
video_temp.write(await video.read())
video_temp.close()
video_path = video_temp.name
# Build messages for model
model_messages = build_messages(req, image_data or None, video_path)
async def stream_generator():
full_text = ""
thinking_sent = False
answer_started = False
thinking_buffer = ""
answer_buffer = ""
in_think = False
try:
streamer, thread = generate_stream(req, model_messages)
for token in streamer:
full_text += token
# Real-time parse of ...
if "" in full_text and not in_think:
in_think = True
yield f"data: {json.dumps({'type': 'thinking_start'})}\n\n"
continue
if in_think:
if "" in full_text:
in_think = False
# Extract remaining thinking content
think_end = full_text.index("")
think_content = full_text[full_text.index("")+7:think_end]
yield f"data: {json.dumps({'type': 'thinking_chunk', 'content': token})}\n\n"
yield f"data: {json.dumps({'type': 'thinking_end'})}\n\n"
else:
yield f"data: {json.dumps({'type': 'thinking_chunk', 'content': token})}\n\n"
continue
# Answer content
yield f"data: {json.dumps({'type': 'answer_chunk', 'content': token})}\n\n"
thread.join()
yield f"data: {json.dumps({'type': 'done'})}\n\n"
except Exception as e:
yield f"data: {json.dumps({'type': 'error', 'content': str(e)})}\n\n"
finally:
if video_path and os.path.exists(video_path):
os.unlink(video_path)
return StreamingResponse(
stream_generator(),
media_type="text/event-stream",
headers={
"Cache-Control": "no-cache",
"X-Accel-Buffering": "no",
}
)
@app.post("/api/chat")
async def chat_simple(req: ChatRequest):
"""Non-streaming chat endpoint."""
model_messages = build_messages(req)
streamer, thread = generate_stream(req, model_messages)
full_text = ""
for token in streamer:
full_text += token
thread.join()
thinking, answer = parse_thinking(full_text)
return {
"thinking": thinking,
"answer": answer,
"full_text": full_text,
}
# ─── Main ──────────────────────────────────────────────────────────────────────
if __name__ == "__main__":
uvicorn.run(app, host="0.0.0.0", port=7860, log_level="info")