Image-Text-to-Text
Transformers
Safetensors
English
qwen3_5
decision-model
typed-decisions
one-pass
option-probabilities
conversational
Instructions to use thegovind/blink-mimo-9b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use thegovind/blink-mimo-9b with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("image-text-to-text", model="thegovind/blink-mimo-9b") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# Load model directly from transformers import AutoProcessor, AutoModelForMultimodalLM processor = AutoProcessor.from_pretrained("thegovind/blink-mimo-9b") model = AutoModelForMultimodalLM.from_pretrained("thegovind/blink-mimo-9b", device_map="auto") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=40) print(processor.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use thegovind/blink-mimo-9b with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "thegovind/blink-mimo-9b" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "thegovind/blink-mimo-9b", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker
docker model run hf.co/thegovind/blink-mimo-9b
- SGLang
How to use thegovind/blink-mimo-9b with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "thegovind/blink-mimo-9b" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "thegovind/blink-mimo-9b", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "thegovind/blink-mimo-9b" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "thegovind/blink-mimo-9b", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }' - Docker Model Runner
How to use thegovind/blink-mimo-9b with Docker Model Runner:
docker model run hf.co/thegovind/blink-mimo-9b
v1.2: TypeSafe API wire compatibility in serve.py (server-side SDKs, GET /v1/models, optional API key, error detail, request ids; a full batching queue answers 529); score answers carry confidence; weights unchanged
Browse files
README.md
CHANGED
|
@@ -203,8 +203,8 @@ import os, sys
|
|
| 203 |
from huggingface_hub import hf_hub_download
|
| 204 |
|
| 205 |
os.environ["BLINK_MODEL"] = "thegovind/blink-mimo-9b"
|
| 206 |
-
os.environ["BLINK_REVISION"] = "v1.
|
| 207 |
-
sys.path.insert(0, os.path.dirname(hf_hub_download("thegovind/blink-mimo-9b", "blink.py", revision="v1.
|
| 208 |
import blink
|
| 209 |
|
| 210 |
out = blink.decide(
|
|
@@ -224,14 +224,17 @@ print(out["answers"]["intent"]["probabilities"])
|
|
| 224 |
|
| 225 |
## Run it as a server
|
| 226 |
|
| 227 |
-
`serve.py`
|
| 228 |
-
|
| 229 |
-
`
|
|
|
|
|
|
|
| 230 |
|
| 231 |
```sh
|
| 232 |
pip install "torch==2.13.0" "transformers==5.17.0" "flash-linear-attention==0.5.2" "accelerate>=1.1.0" safetensors huggingface_hub
|
| 233 |
-
hf download thegovind/blink-mimo-9b --revision v1.
|
| 234 |
python blink-mimo-9b/serve.py --model ./blink-mimo-9b --port 8000
|
|
|
|
| 235 |
```
|
| 236 |
|
| 237 |
In another terminal: `curl -s http://127.0.0.1:8000/healthz`.
|
|
@@ -250,25 +253,30 @@ docker build -t blink-mimo-9b . && docker run --rm --gpus all -p 127.0.0.1:8000:
|
|
| 250 |
answers), `kernels` (fast path or slower fallback without flash-linear-attention), `versions` and `hub_offline`.
|
| 251 |
- Limits: 255 options per choice, 2–10 score levels, 131,072 input tokens per question and 512 questions
|
| 252 |
per request. Over-limit requests get HTTP 422 with the reason; nothing is truncated.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 253 |
- Requests run one at a time. Questions are batched; each batch takes one forward pass (large requests
|
| 254 |
can take more than one). Serving the downloaded folder or Docker image enables Hugging Face offline
|
| 255 |
mode before model loading (`hub_offline: true`). The server doesn't otherwise restrict network access.
|
| 256 |
- Set `--batch-window-ms 5` to turn on cross-request batching with a 5 ms collection window, up to
|
| 257 |
`--max-batch-requests` requests at a time, which defaults to 16. The window defaults to 0, so requests still
|
| 258 |
run one at a time. `--max-queued-requests` lets up to 64 requests wait for a batch by default. Excess requests
|
| 259 |
-
get HTTP
|
| 260 |
-
throughput rose about 15% with 4 concurrent clients and 16%
|
| 261 |
-
|
| 262 |
-
the public TypeSafe cases covered by the FP32 reference,
|
| 263 |
-
checks against an FP32 reference as one-at-a-time answers,
|
| 264 |
-
differences. TypeSafe documents were sent as JSON objects. The FP32
|
| 265 |
-
TypeSafe documents, so parity covered 170 of 354 TypeSafe questions.
|
| 266 |
-
the same option every time on the other 184. A few near-tied answers
|
| 267 |
-
|
| 268 |
-
the flag:
|
| 269 |
|
| 270 |
```sh
|
| 271 |
-
hf download thegovind/blink-mimo-9b serve.py blink.py --revision v1.
|
| 272 |
python blink-mimo-9b/serve.py --model ./blink-mimo-9b --port 8000 --batch-window-ms 5
|
| 273 |
```
|
| 274 |
- blink-mimo-9b weights are 18.8 GB in bf16. Long prompts need more memory.
|
|
|
|
| 203 |
from huggingface_hub import hf_hub_download
|
| 204 |
|
| 205 |
os.environ["BLINK_MODEL"] = "thegovind/blink-mimo-9b"
|
| 206 |
+
os.environ["BLINK_REVISION"] = "v1.2"
|
| 207 |
+
sys.path.insert(0, os.path.dirname(hf_hub_download("thegovind/blink-mimo-9b", "blink.py", revision="v1.2")))
|
| 208 |
import blink
|
| 209 |
|
| 210 |
out = blink.decide(
|
|
|
|
| 224 |
|
| 225 |
## Run it as a server
|
| 226 |
|
| 227 |
+
`serve.py` handles TypeSafe's request and answer fields at `POST /v1/systemone` and lists its model at
|
| 228 |
+
`GET /v1/models`. From server-side code, point TypeSafe's Python or JavaScript SDK at the server with
|
| 229 |
+
`TYPESAFE_BASE_URL`;
|
| 230 |
+
JevBench's stock `typesafe` adapter and the Decision Index kit's `http` engine still work unchanged.
|
| 231 |
+
`GET /healthz` reports startup checks. `v1.2` changes code only; its weights are identical to `v1.0`.
|
| 232 |
|
| 233 |
```sh
|
| 234 |
pip install "torch==2.13.0" "transformers==5.17.0" "flash-linear-attention==0.5.2" "accelerate>=1.1.0" safetensors huggingface_hub
|
| 235 |
+
hf download thegovind/blink-mimo-9b --revision v1.2 --local-dir blink-mimo-9b
|
| 236 |
python blink-mimo-9b/serve.py --model ./blink-mimo-9b --port 8000
|
| 237 |
+
# TypeSafe SDKs: export TYPESAFE_BASE_URL=http://127.0.0.1:8000 TYPESAFE_API_KEY=any
|
| 238 |
```
|
| 239 |
|
| 240 |
In another terminal: `curl -s http://127.0.0.1:8000/healthz`.
|
|
|
|
| 253 |
answers), `kernels` (fast path or slower fallback without flash-linear-attention), `versions` and `hub_offline`.
|
| 254 |
- Limits: 255 options per choice, 2–10 score levels, 131,072 input tokens per question and 512 questions
|
| 255 |
per request. Over-limit requests get HTTP 422 with the reason; nothing is truncated.
|
| 256 |
+
- `GET /v1/models` lists the one served model with a blank `release_date`. Every request uses that model
|
| 257 |
+
regardless of its `model` field.
|
| 258 |
+
- The server is open by default. Set `--api-key` or `BLINK_API_KEY` to require `Authorization: Bearer <key>` on both API
|
| 259 |
+
routes. Missing or wrong keys get 401; `/healthz` stays open.
|
| 260 |
+
- Error bodies put the reason in `error` and `detail`. Over-limit requests return 422, with nothing cut.
|
| 261 |
- Requests run one at a time. Questions are batched; each batch takes one forward pass (large requests
|
| 262 |
can take more than one). Serving the downloaded folder or Docker image enables Hugging Face offline
|
| 263 |
mode before model loading (`hub_offline: true`). The server doesn't otherwise restrict network access.
|
| 264 |
- Set `--batch-window-ms 5` to turn on cross-request batching with a 5 ms collection window, up to
|
| 265 |
`--max-batch-requests` requests at a time, which defaults to 16. The window defaults to 0, so requests still
|
| 266 |
run one at a time. `--max-queued-requests` lets up to 64 requests wait for a batch by default. Excess requests
|
| 267 |
+
get HTTP 529 with `Retry-After`, so clients should retry. `v1.1` returned HTTP 503 for a full queue. On a
|
| 268 |
+
1,000-request Decision Index sample over HTTP, throughput rose about 15% with 4 concurrent clients and 16%
|
| 269 |
+
with 16. One client saw no gain. Offline runs on long documents showed no meaningful gain. On the Decision
|
| 270 |
+
Index sample, a set of long workflow documents, and the public TypeSafe cases covered by the FP32 reference,
|
| 271 |
+
batched answers passed the same numerical-parity checks against an FP32 reference as one-at-a-time answers,
|
| 272 |
+
covering argmax agreement and probability differences. TypeSafe documents were sent as JSON objects. The FP32
|
| 273 |
+
reference could not run the five longest TypeSafe documents, so parity covered 170 of 354 TypeSafe questions.
|
| 274 |
+
Batched and one-at-a-time answers chose the same option every time on the other 184. A few near-tied answers
|
| 275 |
+
can still flip. Batching arrived in `v1.1`. The current code revision is `v1.2`, with weights identical to
|
| 276 |
+
`v1.0`. Update the two code files in an existing `v1.0` download, then restart with the flag:
|
| 277 |
|
| 278 |
```sh
|
| 279 |
+
hf download thegovind/blink-mimo-9b serve.py blink.py --revision v1.2 --local-dir blink-mimo-9b
|
| 280 |
python blink-mimo-9b/serve.py --model ./blink-mimo-9b --port 8000 --batch-window-ms 5
|
| 281 |
```
|
| 282 |
- blink-mimo-9b weights are 18.8 GB in bf16. Long prompts need more memory.
|
blink.py
CHANGED
|
@@ -223,12 +223,16 @@ def answer_for(q: dict, keys: list[str], probs: list[float]) -> dict:
|
|
| 223 |
return {"type": "noul", "noul": p["yes"], "probabilities": p}
|
| 224 |
levels = [str(i) for i in range(len(q["criteria"]))]
|
| 225 |
ev = sum(int(k) * p[k] for k in levels)
|
|
|
|
|
|
|
| 226 |
return {
|
| 227 |
"type": "score",
|
| 228 |
"score": ev,
|
| 229 |
"probabilities": {k: p[k] for k in levels},
|
| 230 |
"legend": {k: text(q["criteria"][int(k)]) for k in levels},
|
| 231 |
-
"choice":
|
|
|
|
|
|
|
| 232 |
}
|
| 233 |
|
| 234 |
|
|
|
|
| 223 |
return {"type": "noul", "noul": p["yes"], "probabilities": p}
|
| 224 |
levels = [str(i) for i in range(len(q["criteria"]))]
|
| 225 |
ev = sum(int(k) * p[k] for k in levels)
|
| 226 |
+
top = max(levels, key=lambda k: (p[k], -int(k)))
|
| 227 |
+
K = len(levels)
|
| 228 |
return {
|
| 229 |
"type": "score",
|
| 230 |
"score": ev,
|
| 231 |
"probabilities": {k: p[k] for k in levels},
|
| 232 |
"legend": {k: text(q["criteria"][int(k)]) for k in levels},
|
| 233 |
+
"choice": top,
|
| 234 |
+
# added after the fields above, which are unchanged: the choice formula over the levels (TypeSafe-shaped)
|
| 235 |
+
"confidence": min(1.0, max(0.0, (p[top] - 1 / K) / (1 - 1 / K))),
|
| 236 |
}
|
| 237 |
|
| 238 |
|
serve.py
CHANGED
|
@@ -1,32 +1,44 @@
|
|
| 1 |
-
"""blink server: a Jev-compatible decision endpoint.
|
| 2 |
|
| 3 |
pip install "torch==2.13.0" "transformers==5.17.0" "flash-linear-attention==0.5.2" accelerate safetensors huggingface_hub
|
| 4 |
hf download thegovind/blink-4b --revision v1.0 --local-dir blink-4b
|
| 5 |
python blink-4b/serve.py --model ./blink-4b --port 8000
|
| 6 |
|
| 7 |
-
POST /v1/systemone {"state": ..., "questions": {...}} -> {"model", "answers", "usage"}
|
| 8 |
-
GET /
|
| 9 |
-
|
| 10 |
-
|
| 11 |
-
|
| 12 |
-
the
|
| 13 |
-
|
| 14 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 15 |
|
| 16 |
Opt-in cross-request batching (--batch-window-ms, default 0 = off): requests that arrive within the window are
|
| 17 |
decided together, up to --max-batch-requests; at most --max-queued-requests wait, and past that a request gets
|
| 18 |
-
HTTP
|
| 19 |
"""
|
| 20 |
|
| 21 |
from __future__ import annotations
|
| 22 |
|
| 23 |
import argparse
|
| 24 |
import hashlib
|
|
|
|
| 25 |
import json
|
| 26 |
import os
|
| 27 |
import socket
|
| 28 |
import sys
|
| 29 |
import threading
|
|
|
|
| 30 |
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
| 31 |
|
| 32 |
WARMUP = ("Order 4471 arrived with a cracked screen. The customer attached photos and wants a replacement.",
|
|
@@ -95,6 +107,42 @@ def int_range(lo: int, hi: int):
|
|
| 95 |
return parse
|
| 96 |
|
| 97 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 98 |
def main() -> None:
|
| 99 |
ap = argparse.ArgumentParser(description="Serve blink over a Jev-compatible HTTP API.")
|
| 100 |
ap.add_argument("--model", default=os.environ.get("BLINK_MODEL", "thegovind/blink-4b"))
|
|
@@ -108,7 +156,11 @@ def main() -> None:
|
|
| 108 |
ap.add_argument("--max-batch-requests", type=int_range(1, 64), default=16,
|
| 109 |
help="most requests decided together (1-64)")
|
| 110 |
ap.add_argument("--max-queued-requests", type=int_range(1, 1024), default=64,
|
| 111 |
-
help="most requests waiting for a batch; past this a request gets HTTP
|
|
|
|
|
|
|
|
|
|
|
|
|
| 112 |
a = ap.parse_args()
|
| 113 |
|
| 114 |
local = os.path.isdir(a.model)
|
|
@@ -151,6 +203,14 @@ def main() -> None:
|
|
| 151 |
if a.batch_window_ms > 0 else None)
|
| 152 |
health["batching"] = ({"window_ms": a.batch_window_ms, "max_requests": a.max_batch_requests,
|
| 153 |
"max_queued": a.max_queued_requests} if batcher else None)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 154 |
|
| 155 |
def current_health() -> dict:
|
| 156 |
if batcher is None:
|
|
@@ -171,31 +231,61 @@ def main() -> None:
|
|
| 171 |
# on the client's delayed ACK, a flat ~40 ms on every request of a kept-alive connection
|
| 172 |
self.connection.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1)
|
| 173 |
|
| 174 |
-
def _send(self, code: int, obj: dict, headers: dict | None = None) -> None:
|
| 175 |
body = json.dumps(obj, ensure_ascii=False).encode("utf-8")
|
| 176 |
-
self.send_response(code)
|
| 177 |
self.send_header("Content-Type", "application/json")
|
| 178 |
self.send_header("Content-Length", str(len(body)))
|
|
|
|
| 179 |
for name, value in (headers or {}).items():
|
| 180 |
self.send_header(name, value)
|
| 181 |
self.end_headers()
|
| 182 |
self.wfile.write(body)
|
| 183 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 184 |
def do_GET(self):
|
| 185 |
-
|
|
|
|
| 186 |
return self._send(200, current_health())
|
| 187 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 188 |
|
| 189 |
def do_POST(self):
|
| 190 |
if self.path.rstrip("/") != "/v1/systemone":
|
| 191 |
-
return self.
|
|
|
|
|
|
|
| 192 |
try:
|
| 193 |
size = int(self.headers.get("Content-Length") or 0)
|
| 194 |
req = json.loads(self.rfile.read(size) or b"{}")
|
| 195 |
except (ValueError, json.JSONDecodeError) as exc:
|
| 196 |
-
return self.
|
| 197 |
if not isinstance(req, dict):
|
| 198 |
-
return self.
|
| 199 |
try:
|
| 200 |
if batcher is not None:
|
| 201 |
out = batcher.submit(req.get("state"), req.get("questions"))
|
|
@@ -203,11 +293,11 @@ def main() -> None:
|
|
| 203 |
with lock:
|
| 204 |
out = blink.decide(req.get("state"), req.get("questions"))
|
| 205 |
except blink.BlinkError as exc:
|
| 206 |
-
return self.
|
| 207 |
except busy as exc:
|
| 208 |
-
return self.
|
| 209 |
except Exception as exc: # noqa: BLE001 - report, keep serving
|
| 210 |
-
return self.
|
| 211 |
return self._send(200, {
|
| 212 |
"model": a.model,
|
| 213 |
"answers": out["answers"],
|
|
@@ -215,7 +305,8 @@ def main() -> None:
|
|
| 215 |
})
|
| 216 |
|
| 217 |
server = ThreadingHTTPServer((a.host, a.port), Handler)
|
| 218 |
-
print(f"blink serving {a.model} on http://{a.host}:{a.port} ({kernels}; weights_verified={verified}
|
|
|
|
| 219 |
server.serve_forever()
|
| 220 |
|
| 221 |
|
|
|
|
| 1 |
+
"""blink server: a Jev-compatible decision endpoint (the TypeSafe API's wire format).
|
| 2 |
|
| 3 |
pip install "torch==2.13.0" "transformers==5.17.0" "flash-linear-attention==0.5.2" accelerate safetensors huggingface_hub
|
| 4 |
hf download thegovind/blink-4b --revision v1.0 --local-dir blink-4b
|
| 5 |
python blink-4b/serve.py --model ./blink-4b --port 8000
|
| 6 |
|
| 7 |
+
POST /v1/systemone {"state": ..., "model": ..., "questions": {...}} -> {"model", "answers", "usage"}
|
| 8 |
+
GET /v1/models -> {"models": [{"name", "description", "release_date"}]}
|
| 9 |
+
GET /healthz -> {"ok", "model", "revision", "weights_verified", "hub_offline", "warmup", "kernels", "versions",
|
| 10 |
+
"batching", "api_key_required"}
|
| 11 |
+
|
| 12 |
+
A client written for the TypeSafe API works unchanged against this server once its base URL points here (for the
|
| 13 |
+
official SDKs: TYPESAFE_BASE_URL=http://127.0.0.1:8000). The request's "model" is accepted and not used: this
|
| 14 |
+
server answers with the one model it serves and names it in the response. Any Authorization header is accepted and
|
| 15 |
+
ignored unless a key is set with --api-key or BLINK_API_KEY; then /v1/systemone and /v1/models need
|
| 16 |
+
"Authorization: Bearer <key>" and answer HTTP 401 without it (/healthz stays open).
|
| 17 |
+
|
| 18 |
+
Requests are served one at a time. A request blink can't answer (a malformed question, or one over a limit: options
|
| 19 |
+
per choice, context length, questions per request) gets HTTP 422 with the reason; nothing is truncated. A body that
|
| 20 |
+
isn't a JSON object gets HTTP 400. Error bodies are {"error": reason, "detail": ...}, where detail is a list of
|
| 21 |
+
{"loc", "msg", "type"} for a 400 or 422 and the reason again otherwise. Every response carries an
|
| 22 |
+
x-typesafe-request-id header. With weights.sha256 beside the weights, every listed file is hashed before serving
|
| 23 |
+
(weights_verified). Serving a local folder switches the Hugging Face libraries to offline mode before any of them
|
| 24 |
+
loads (hub_offline reports the setting the libraries actually use); the server does not otherwise restrict the network.
|
| 25 |
|
| 26 |
Opt-in cross-request batching (--batch-window-ms, default 0 = off): requests that arrive within the window are
|
| 27 |
decided together, up to --max-batch-requests; at most --max-queued-requests wait, and past that a request gets
|
| 28 |
+
HTTP 529 (overloaded) with Retry-After. Each request still gets its own answers or its own error.
|
| 29 |
"""
|
| 30 |
|
| 31 |
from __future__ import annotations
|
| 32 |
|
| 33 |
import argparse
|
| 34 |
import hashlib
|
| 35 |
+
import hmac
|
| 36 |
import json
|
| 37 |
import os
|
| 38 |
import socket
|
| 39 |
import sys
|
| 40 |
import threading
|
| 41 |
+
import uuid
|
| 42 |
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
| 43 |
|
| 44 |
WARMUP = ("Order 4471 arrived with a cracked screen. The customer attached photos and wants a replacement.",
|
|
|
|
| 107 |
return parse
|
| 108 |
|
| 109 |
|
| 110 |
+
def api_key(value: str):
|
| 111 |
+
"""--api-key / BLINK_API_KEY: empty means open; a key must fit in an Authorization header (printable ASCII, no
|
| 112 |
+
whitespace), as TypeSafe's SDKs require of theirs."""
|
| 113 |
+
key = value.strip()
|
| 114 |
+
if not key:
|
| 115 |
+
return None
|
| 116 |
+
if not (key.isascii() and key.isprintable()) or " " in key:
|
| 117 |
+
raise argparse.ArgumentTypeError("must be printable ASCII without whitespace")
|
| 118 |
+
return key
|
| 119 |
+
|
| 120 |
+
|
| 121 |
+
def bearer(header) -> str | None:
|
| 122 |
+
"""The token of an "Authorization: Bearer <token>" header, else None."""
|
| 123 |
+
scheme, _, token = (header or "").strip().partition(" ")
|
| 124 |
+
if scheme.lower() != "bearer":
|
| 125 |
+
return None
|
| 126 |
+
return token.strip() or None
|
| 127 |
+
|
| 128 |
+
|
| 129 |
+
def error_loc(blink, req: dict, message: str) -> list:
|
| 130 |
+
"""Where a refused request went wrong, as a FastAPI-style location (TypeSafe's 422 body): the question the reason
|
| 131 |
+
opens with (blink's refusals that name one start "question '<key>' "), else the first question blink can't read,
|
| 132 |
+
else the questions map. Only the opening counts: a key quoted later in a reason, say inside a bad type, isn't it."""
|
| 133 |
+
questions = req.get("questions")
|
| 134 |
+
if isinstance(questions, dict) and 0 < len(questions) <= getattr(blink, "MAX_QUESTIONS", 512):
|
| 135 |
+
for key in questions:
|
| 136 |
+
if message.startswith(f"question {key!r} "):
|
| 137 |
+
return ["body", "questions", key]
|
| 138 |
+
for key, q in questions.items():
|
| 139 |
+
try:
|
| 140 |
+
blink.question_options(q)
|
| 141 |
+
except Exception: # noqa: BLE001 - only locating the refusal, which is already decided
|
| 142 |
+
return ["body", "questions", key]
|
| 143 |
+
return ["body", "questions"]
|
| 144 |
+
|
| 145 |
+
|
| 146 |
def main() -> None:
|
| 147 |
ap = argparse.ArgumentParser(description="Serve blink over a Jev-compatible HTTP API.")
|
| 148 |
ap.add_argument("--model", default=os.environ.get("BLINK_MODEL", "thegovind/blink-4b"))
|
|
|
|
| 156 |
ap.add_argument("--max-batch-requests", type=int_range(1, 64), default=16,
|
| 157 |
help="most requests decided together (1-64)")
|
| 158 |
ap.add_argument("--max-queued-requests", type=int_range(1, 1024), default=64,
|
| 159 |
+
help="most requests waiting for a batch; past this a request gets HTTP 529 (1-1024)")
|
| 160 |
+
# a string default goes through api_key() too, so BLINK_API_KEY is checked the same way
|
| 161 |
+
ap.add_argument("--api-key", type=api_key, default=os.environ.get("BLINK_API_KEY", ""),
|
| 162 |
+
help="require 'Authorization: Bearer <key>' on /v1/systemone and /v1/models (default: open, any "
|
| 163 |
+
"Authorization header is accepted and ignored); BLINK_API_KEY keeps it out of the process list")
|
| 164 |
a = ap.parse_args()
|
| 165 |
|
| 166 |
local = os.path.isdir(a.model)
|
|
|
|
| 203 |
if a.batch_window_ms > 0 else None)
|
| 204 |
health["batching"] = ({"window_ms": a.batch_window_ms, "max_requests": a.max_batch_requests,
|
| 205 |
"max_queued": a.max_queued_requests} if batcher else None)
|
| 206 |
+
health["api_key_required"] = a.api_key is not None
|
| 207 |
+
# TypeSafe's model list, for clients that ask which names the model field takes; any name is accepted here
|
| 208 |
+
listing = {"models": [{
|
| 209 |
+
"name": a.model,
|
| 210 |
+
"description": "blink: typed decisions (noul, choice, score) with option probabilities from one forward pass. "
|
| 211 |
+
"This server serves one model; a request's model field is accepted and not used.",
|
| 212 |
+
"release_date": "",
|
| 213 |
+
}]}
|
| 214 |
|
| 215 |
def current_health() -> dict:
|
| 216 |
if batcher is None:
|
|
|
|
| 231 |
# on the client's delayed ACK, a flat ~40 ms on every request of a kept-alive connection
|
| 232 |
self.connection.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1)
|
| 233 |
|
| 234 |
+
def _send(self, code: int, obj: dict, headers: dict | None = None, reason: str | None = None) -> None:
|
| 235 |
body = json.dumps(obj, ensure_ascii=False).encode("utf-8")
|
| 236 |
+
self.send_response(code, reason)
|
| 237 |
self.send_header("Content-Type", "application/json")
|
| 238 |
self.send_header("Content-Length", str(len(body)))
|
| 239 |
+
self.send_header("x-typesafe-request-id", uuid.uuid4().hex) # read by TypeSafe's SDKs
|
| 240 |
for name, value in (headers or {}).items():
|
| 241 |
self.send_header(name, value)
|
| 242 |
self.end_headers()
|
| 243 |
self.wfile.write(body)
|
| 244 |
|
| 245 |
+
def _fail(self, code: int, reason: str, detail=None, headers: dict | None = None, *, unread: bool = False,
|
| 246 |
+
status_text: str | None = None) -> None:
|
| 247 |
+
"""An error body: blink's {"error"} plus TypeSafe's (FastAPI's) {"detail"}. With the request body still
|
| 248 |
+
unread, the connection is closed rather than left holding it."""
|
| 249 |
+
if unread:
|
| 250 |
+
self.close_connection = True
|
| 251 |
+
headers = {**(headers or {}), "Connection": "close"}
|
| 252 |
+
self._send(code, {"error": reason, "detail": reason if detail is None else detail}, headers, status_text)
|
| 253 |
+
|
| 254 |
+
def _refused(self, code: int, reason: str, loc: list, kind: str) -> None:
|
| 255 |
+
self._fail(code, reason, [{"loc": loc, "msg": reason, "type": kind}])
|
| 256 |
+
|
| 257 |
+
def _authorized(self) -> bool:
|
| 258 |
+
if a.api_key is None:
|
| 259 |
+
return True
|
| 260 |
+
token = bearer(self.headers.get("Authorization"))
|
| 261 |
+
return token is not None and hmac.compare_digest(token.encode("utf-8"), a.api_key.encode("utf-8"))
|
| 262 |
+
|
| 263 |
+
def _unauthorized(self, unread: bool) -> None:
|
| 264 |
+
self._fail(401, "missing or invalid API key: send Authorization: Bearer <key>",
|
| 265 |
+
headers={"WWW-Authenticate": "Bearer"}, unread=unread)
|
| 266 |
+
|
| 267 |
def do_GET(self):
|
| 268 |
+
path = self.path.rstrip("/")
|
| 269 |
+
if path in ("/healthz", "/health"):
|
| 270 |
return self._send(200, current_health())
|
| 271 |
+
if path == "/v1/models":
|
| 272 |
+
if not self._authorized():
|
| 273 |
+
return self._unauthorized(unread=False)
|
| 274 |
+
return self._send(200, listing)
|
| 275 |
+
return self._fail(404, "not found")
|
| 276 |
|
| 277 |
def do_POST(self):
|
| 278 |
if self.path.rstrip("/") != "/v1/systemone":
|
| 279 |
+
return self._fail(404, "not found", unread=True)
|
| 280 |
+
if not self._authorized():
|
| 281 |
+
return self._unauthorized(unread=True)
|
| 282 |
try:
|
| 283 |
size = int(self.headers.get("Content-Length") or 0)
|
| 284 |
req = json.loads(self.rfile.read(size) or b"{}")
|
| 285 |
except (ValueError, json.JSONDecodeError) as exc:
|
| 286 |
+
return self._refused(400, f"invalid JSON: {exc}", ["body"], "json_invalid")
|
| 287 |
if not isinstance(req, dict):
|
| 288 |
+
return self._refused(400, "the body must be a JSON object", ["body"], "value_error")
|
| 289 |
try:
|
| 290 |
if batcher is not None:
|
| 291 |
out = batcher.submit(req.get("state"), req.get("questions"))
|
|
|
|
| 293 |
with lock:
|
| 294 |
out = blink.decide(req.get("state"), req.get("questions"))
|
| 295 |
except blink.BlinkError as exc:
|
| 296 |
+
return self._refused(422, str(exc), error_loc(blink, req, str(exc)), "value_error")
|
| 297 |
except busy as exc:
|
| 298 |
+
return self._fail(529, str(exc), headers={"Retry-After": "1"}, status_text="Overloaded")
|
| 299 |
except Exception as exc: # noqa: BLE001 - report, keep serving
|
| 300 |
+
return self._fail(500, f"{type(exc).__name__}: {exc}")
|
| 301 |
return self._send(200, {
|
| 302 |
"model": a.model,
|
| 303 |
"answers": out["answers"],
|
|
|
|
| 305 |
})
|
| 306 |
|
| 307 |
server = ThreadingHTTPServer((a.host, a.port), Handler)
|
| 308 |
+
print(f"blink serving {a.model} on http://{a.host}:{a.port} ({kernels}; weights_verified={verified}; "
|
| 309 |
+
f"api key {'required' if a.api_key else 'not required'})", flush=True)
|
| 310 |
server.serve_forever()
|
| 311 |
|
| 312 |
|