""" Cerebellum-2B High-Performance Decision Server Includes REST API and Built-in Interactive Web UI """ import os import sys import time from typing import List, Dict, Optional import torch import uvicorn from fastapi import FastAPI, HTTPException from fastapi.responses import HTMLResponse from pydantic import BaseModel, Field # Local model import sys.path.append(os.path.dirname(os.path.abspath(__file__))) from modeling_cerebellum import CerebellumModel app = FastAPI( title="Cerebellum-2B Decision Server", description="25ms Non-Autoregressive Agent System 1 Decision Engine", version="1.0.0" ) # Global model instance model: Optional[CerebellumModel] = None class DecideRequest(BaseModel): state: str = Field(..., description="Agent dialogue history, environment observation, or context") candidates: List[str] = Field(..., min_items=1, description="List of candidate API calls, tools, or DOM actions") instruction: Optional[str] = Field("Select the best action to execute next.", description="Decision instruction") escalate_threshold: Optional[float] = Field(0.50, description="Escalate to human/LLM threshold") class DecideResponse(BaseModel): action: str action_index: int confidence: float probabilities: Dict[str, float] needs_escalation: bool escalate_probability: float latency_ms: float class BatchDecideRequest(BaseModel): queries: List[DecideRequest] class BatchDecideResponse(BaseModel): results: List[DecideResponse] total_latency_ms: float @app.on_event("startup") def startup(): global model model_dir = os.path.dirname(os.path.abspath(__file__)) device = "cuda:0" if torch.cuda.is_available() else "cpu" print(f"[Cerebellum] Loading model from {model_dir} on {device}...") model = CerebellumModel.from_pretrained(model_dir, device=device) # Warmup _ = model.decide("Hello", ["Action A", "Action B"]) print("[Cerebellum] Model ready for fast non-autoregressive decisions!") @app.get("/health") def health(): return {"status": "ok", "model": "Cerebellum-2B", "device": "cuda:0" if torch.cuda.is_available() else "cpu"} @app.post("/v1/decide", response_model=DecideResponse) def decide(req: DecideRequest): if model is None: raise HTTPException(status_code=503, detail="Model is still initializing") if len(req.candidates) == 0: raise HTTPException(status_code=400, detail="Candidates list cannot be empty") t0 = time.perf_counter() dec = model.decide( state=req.state, candidates=req.candidates, instruction=req.instruction, escalate_threshold=req.escalate_threshold ) return DecideResponse( action=dec.action, action_index=dec.action_index, confidence=dec.confidence, probabilities=dec.probabilities, needs_escalation=dec.needs_escalation, escalate_probability=dec.escalate_probability, latency_ms=dec.latency_ms ) @app.post("/v1/batch_decide", response_model=BatchDecideResponse) def batch_decide(req: BatchDecideRequest): if model is None: raise HTTPException(status_code=503, detail="Model is still initializing") t0 = time.perf_counter() results = [] for q in req.queries: dec = model.decide( state=q.state, candidates=q.candidates, instruction=q.instruction, escalate_threshold=q.escalate_threshold ) results.append(DecideResponse( action=dec.action, action_index=dec.action_index, confidence=dec.confidence, probabilities=dec.probabilities, needs_escalation=dec.needs_escalation, escalate_probability=dec.escalate_probability, latency_ms=dec.latency_ms )) total_latency = (time.perf_counter() - t0) * 1000 return BatchDecideResponse(results=results, total_latency_ms=total_latency) @app.get("/", response_class=HTMLResponse) def dashboard(): return """ Cerebellum-2B Interactive Decision Console

🧠 Cerebellum-2B (小脑-2B)

25ms Non-Autoregressive AI Agent System 1 Decision Engine

⚡ O(1) Single Forward Pass

Decision Result

Selected Action:
Candidate Probability Distribution:
""" if __name__ == "__main__": uvicorn.run("serve:app", host="0.0.0.0", port=8000, workers=1)