Spaces:
Running
Running
Require the API key on /v1 and default thinking off
Browse files- gateway.py +14 -2
gateway.py
CHANGED
|
@@ -23,6 +23,11 @@ MAX_QUEUE_SIZE = int(os.getenv("MAX_QUEUE_SIZE", "32"))
|
|
| 23 |
REQUEST_TIMEOUT = float(os.getenv("REQUEST_TIMEOUT_SECONDS", "900"))
|
| 24 |
API_KEY = os.getenv("API_KEY", "").strip()
|
| 25 |
LLAMA_API_KEY = os.getenv("LLAMA_API_KEY", "").strip()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 26 |
ENGLISH_GUARD = (
|
| 27 |
"Respond entirely in English unless the user explicitly requests another "
|
| 28 |
"language. Preserve code, identifiers, and quoted source material exactly."
|
|
@@ -94,6 +99,13 @@ def require_api_key(authorization: str | None = Header(default=None)) -> None:
|
|
| 94 |
raise HTTPException(status_code=401, detail="Invalid API key")
|
| 95 |
|
| 96 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 97 |
def ensure_language_guard(payload: dict[str, Any]) -> dict[str, Any]:
|
| 98 |
if DEFAULT_LANGUAGE.lower() != "en":
|
| 99 |
return payload
|
|
@@ -139,7 +151,7 @@ async def execute_completion(payload: dict[str, Any]) -> dict[str, Any]:
|
|
| 139 |
async with httpx.AsyncClient(timeout=REQUEST_TIMEOUT) as client:
|
| 140 |
response = await client.post(
|
| 141 |
f"{LLAMA_URL}/v1/chat/completions",
|
| 142 |
-
json=
|
| 143 |
headers=llama_auth_headers(),
|
| 144 |
)
|
| 145 |
response.raise_for_status()
|
|
@@ -265,7 +277,7 @@ async def chat_completions(request: Request):
|
|
| 265 |
except httpx.HTTPStatusError as exc:
|
| 266 |
raise HTTPException(exc.response.status_code, exc.response.text) from exc
|
| 267 |
|
| 268 |
-
payload =
|
| 269 |
|
| 270 |
async def stream_response():
|
| 271 |
await wait_for_llama()
|
|
|
|
| 23 |
REQUEST_TIMEOUT = float(os.getenv("REQUEST_TIMEOUT_SECONDS", "900"))
|
| 24 |
API_KEY = os.getenv("API_KEY", "").strip()
|
| 25 |
LLAMA_API_KEY = os.getenv("LLAMA_API_KEY", "").strip()
|
| 26 |
+
# Spark X2.5 is a reasoning model: with thinking on it routinely spends the
|
| 27 |
+
# whole max_tokens budget reasoning and returns an empty answer, which breaks
|
| 28 |
+
# agent callers. Default it off; a caller can still pass
|
| 29 |
+
# chat_template_kwargs={"enable_thinking": true} explicitly.
|
| 30 |
+
DEFAULT_ENABLE_THINKING = os.getenv("DEFAULT_ENABLE_THINKING", "false").lower() == "true"
|
| 31 |
ENGLISH_GUARD = (
|
| 32 |
"Respond entirely in English unless the user explicitly requests another "
|
| 33 |
"language. Preserve code, identifiers, and quoted source material exactly."
|
|
|
|
| 99 |
raise HTTPException(status_code=401, detail="Invalid API key")
|
| 100 |
|
| 101 |
|
| 102 |
+
def apply_defaults(payload: dict[str, Any]) -> dict[str, Any]:
|
| 103 |
+
kwargs = dict(payload.get("chat_template_kwargs") or {})
|
| 104 |
+
kwargs.setdefault("enable_thinking", DEFAULT_ENABLE_THINKING)
|
| 105 |
+
payload = {**payload, "chat_template_kwargs": kwargs}
|
| 106 |
+
return ensure_language_guard(payload)
|
| 107 |
+
|
| 108 |
+
|
| 109 |
def ensure_language_guard(payload: dict[str, Any]) -> dict[str, Any]:
|
| 110 |
if DEFAULT_LANGUAGE.lower() != "en":
|
| 111 |
return payload
|
|
|
|
| 151 |
async with httpx.AsyncClient(timeout=REQUEST_TIMEOUT) as client:
|
| 152 |
response = await client.post(
|
| 153 |
f"{LLAMA_URL}/v1/chat/completions",
|
| 154 |
+
json=apply_defaults(payload),
|
| 155 |
headers=llama_auth_headers(),
|
| 156 |
)
|
| 157 |
response.raise_for_status()
|
|
|
|
| 277 |
except httpx.HTTPStatusError as exc:
|
| 278 |
raise HTTPException(exc.response.status_code, exc.response.text) from exc
|
| 279 |
|
| 280 |
+
payload = apply_defaults(payload)
|
| 281 |
|
| 282 |
async def stream_response():
|
| 283 |
await wait_for_llama()
|