betterwithage commited on
Commit
c107f18
·
verified ·
1 Parent(s): 34cadf2

chore(sync): mirror backend .py + Dockerfile to Space (hf-sync-backend)

Browse files

Automated backend sync from szl-holdings/a11oy main via hf-sync-backend.
Updated (differed from the Space): szl_llm_registry.py
Deleted (gone from the repo + Dockerfile COPY set): (none)

Keeps the Space-built backend (serve.py + the Dockerfile-COPY'd .py
modules) identical to GitHub main so the Space never rebuilds from a
stale backend, new endpoints don't 404 there, and orphaned modules
removed from the repo don't linger in the Space tree.

Files changed (1) hide show
  1. szl_llm_registry.py +248 -75
szl_llm_registry.py CHANGED
@@ -53,6 +53,21 @@ DOCTRINE = "v11"
53
  _KERNEL = "c7c0ba17"
54
  _LAMBDA_FLOOR = 0.90
55
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
56
  # ─────────────────────────────────────────────────────────────────────────────
57
  # THE CANONICAL LLM ROSTER — a11oy is the hub; every model lives here.
58
  # Source of truth: OPERATOR_FULL_CAPABILITY_BRIEF_2026-05-31_2135.md §2,
@@ -203,6 +218,47 @@ MODEL_REGISTRY: list[dict[str, Any]] = [
203
  "Honest stub when the env is unset or the node is offline.",
204
  "honest_stub": True,
205
  },
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
206
  # ── ADDITIONAL: Perplexity (online search-augmented) ──
207
  {
208
  "model_id": "perplexity_sonar_pro",
@@ -361,9 +417,34 @@ except (TypeError, ValueError):
361
  _SOVEREIGN_GEN_TIMEOUT_S = 120.0
362
 
363
 
 
 
 
 
364
  def _sovereign_base() -> str:
365
- """Resolve the sovereign-local base URL from env (empty string when unset)."""
366
- return (os.environ.get(_SOVEREIGN_ENV, "") or "").strip().rstrip("/")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
367
 
368
 
369
  def _sovereign_model_slug() -> str:
@@ -403,20 +484,21 @@ def sovereign_probe(base: str = "", timeout: float | None = None) -> dict[str, A
403
  """
404
  base = (base or _sovereign_base())
405
  to = _SOVEREIGN_PROBE_TIMEOUT_S if timeout is None else float(timeout)
 
406
  out: dict[str, Any] = {
407
  "env_var": _SOVEREIGN_ENV,
408
- "env_present": bool(base),
409
  "base_url": base or None,
410
  "live": False,
411
  "models": [],
412
  "probed": [],
413
  "note": "",
414
  }
415
- if not base:
416
- out["note"] = ("SZL_LOCAL_LLM_URL not set — sovereign_local is an HONEST STUB "
417
- "(point it at https://gpu.a-11-oy.com to wire the local fleet).")
418
- return out
419
- b = base.rstrip("/")
420
  # 1) ollama native /api/tags
421
  tags_url = b + "/api/tags"
422
  doc, err = _http_json(tags_url, timeout=to)
@@ -441,9 +523,12 @@ def sovereign_probe(base: str = "", timeout: float | None = None) -> dict[str, A
441
  out["api_style"] = "openai /v1"
442
  out["note"] = "node live (OpenAI-compatible /v1/models); model list is real THIS request."
443
  return out
444
- out["note"] = ("SZL_LOCAL_LLM_URL set but node did not respond live this request "
445
- "(honest stub). Errors: %s" % "; ".join(
446
- str(p.get("error")) for p in out["probed"] if p.get("error")))
 
 
 
447
  return out
448
 
449
 
@@ -462,13 +547,9 @@ def sovereign_generate(prompt: str, base: str = "", model: str = "",
462
  res: dict[str, Any] = {
463
  "wired": False, "live": False, "text": None, "model": model,
464
  "api_style": None, "base_url": base or None, "env_var": _SOVEREIGN_ENV,
465
- "env_present": bool(base), "note": "",
466
  }
467
- if not base:
468
- res["note"] = ("SZL_LOCAL_LLM_URL not set — HONEST STUB (no local fleet base). "
469
- "Tier selection + Λ-receipt still REAL.")
470
- return res
471
- b = base.rstrip("/")
472
  # 1) ollama native /api/generate
473
  gen_url = b + "/api/generate"
474
  body = json.dumps({"model": model, "prompt": prompt, "stream": False}).encode("utf-8")
@@ -499,8 +580,9 @@ def sovereign_generate(prompt: str, base: str = "", model: str = "",
499
  if isinstance(doc2.get("usage"), dict):
500
  res["raw"] = {"usage": doc2["usage"]}
501
  return res
502
- res["note"] = ("SZL_LOCAL_LLM_URL set but node did not generate live this request "
503
- "(honest stub). Errors: %s / %s" % (err or "", err2 or ""))
 
504
  return res
505
 
506
 
@@ -621,24 +703,35 @@ def _enrich_model(m: dict, *, probe_local: bool = False) -> dict:
621
  env_var = m.get("api_env_var", "") or ""
622
  model_id = m.get("model_id", "")
623
 
624
- if model_id == "sovereign_local" or env_var == _SOVEREIGN_ENV:
625
  base = _sovereign_base()
626
- wired = bool(base)
 
 
 
627
  out["api_key_wired"] = wired
628
  out["wired"] = wired
629
- out["provider"] = m.get("provider", "SZL Holdings (local)")
630
  out["env_used"] = _SOVEREIGN_ENV
631
- out["env_present"] = wired
632
  out["base_url"] = base or m.get("api_base")
633
- out["honest_stub"] = not wired
634
  out["is_local"] = True
 
 
 
 
 
635
  if probe_local:
636
  probe = sovereign_probe(base)
637
- out["local_live"] = bool(probe.get("live"))
 
 
638
  out["local_models"] = probe.get("models", [])
639
  out["local_probe_note"] = probe.get("note", "")
640
- # honest_stub only clears when the node is actually live this request
641
- out["honest_stub"] = not bool(probe.get("live"))
 
 
642
  return out
643
 
644
  wired = _api_key_wired(env_var)
@@ -701,8 +794,27 @@ def register(app: FastAPI) -> dict:
701
  "base_url": m.get("base_url"),
702
  "honest_stub": bool(m.get("honest_stub", True)),
703
  "is_local": bool(m.get("is_local")),
 
 
704
  } for m in models]
705
  all_stub = (len(wired) == 0)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
706
  return JSONResponse({
707
  "timestamp": _now(),
708
  "hub": "a11oy",
@@ -723,6 +835,7 @@ def register(app: FastAPI) -> dict:
723
  "route_endpoint": "/api/a11oy/v1/llm/route",
724
  "router_status_endpoint": "/api/a11oy/v1/llm/router/status",
725
  "sovereign_health_endpoint": "/api/a11oy/v1/llm/sovereign/health",
 
726
  "doctrine": DOCTRINE,
727
  "kernel_commit": _KERNEL,
728
  "honest_note": (
@@ -819,55 +932,87 @@ def register(app: FastAPI) -> dict:
819
 
820
  lam = _lambda_gm(axis_scores)
821
 
822
- # ── SOVEREIGN LOCAL selection ────────────────────────────────────────
823
- # Route to the local fleet when EITHER the caller asks for it explicitly
824
- # (model_id='sovereign_local' | task_hint in {sovereign,local,offline} |
825
- # prefer_local=true) OR SZL_LOCAL_LLM_URL is set and NO cloud key is wired
826
- # (air-gap/offline preference). When selected, we make a REAL guarded call
827
- # to the node; wired=true ONLY if the node answered live THIS request.
 
 
 
 
 
 
828
  _req_model = str(body.get("model_id", "")).strip().lower()
829
- _prefer_local = bool(body.get("prefer_local", False))
 
 
 
830
  _sov_hints = {"sovereign", "local", "offline", "air-gap", "airgap"}
831
  _any_cloud_wired = any(_api_key_wired(ev) for ev, _ in _PROVIDER_ENV_VARS)
832
  _sov_base = _sovereign_base()
833
- _want_sovereign = (
834
- _req_model == "sovereign_local"
835
- or task_hint in _sov_hints
836
- or _prefer_local
837
- or (bool(_sov_base) and not _any_cloud_wired
838
- and str(body.get("offline_mode", "")).lower() in ("1", "true", "yes"))
839
  )
 
 
 
 
 
 
 
 
840
  if _want_sovereign:
841
- sov_model = _MODEL_BY_ID.get("sovereign_local", MODEL_REGISTRY[0])
 
 
 
842
  sov_enriched = _enrich_model(sov_model)
843
  gen = sovereign_generate(prompt or _DEFAULT_SOVEREIGN_PROMPT)
844
- sov_enriched["wired"] = bool(gen.get("wired"))
845
- sov_enriched["api_key_wired"] = bool(gen.get("wired"))
846
- sov_enriched["local_live"] = bool(gen.get("live"))
847
- sov_enriched["honest_stub"] = not bool(gen.get("wired"))
848
- sov_reason = ("sovereign_local selected (%s); "
849
- % ("explicit request" if (_req_model == "sovereign_local"
850
- or task_hint in _sov_hints or _prefer_local)
851
- else "offline preference, no cloud key wired"))
852
- if gen.get("wired"):
853
- sov_reason += "node LIVE this request — REAL local generation."
 
 
 
 
 
854
  response_text = gen.get("text") or ""
855
  else:
856
- sov_reason += "node NOT live — HONEST STUB."
857
  response_text = (
858
- "[HONEST STUB] sovereign_local (%s). %s Tier selection + Λ=%.4f "
859
- "+ receipt are REAL." % (sov_model.get("model_slug", ""),
860
- gen.get("note", ""), lam))
 
 
861
  sov_receipt = {
862
  "schema": "szl.llm_route.lambda_receipt/v1",
863
  "ts": _now(), "hub": "a11oy", "lambda": round(lam, 6),
864
  "lambda_floor": _LAMBDA_FLOOR, "axis_scores": axis_scores,
865
  "tier_selected": sov_model.get("tier", 5),
866
- "model_id": "sovereign_local",
 
 
867
  "model_display": sov_model.get("display_name"),
868
  "reason": sov_reason, "task_hint": task_hint,
869
- "api_key_wired": bool(gen.get("wired")),
870
- "local_live": bool(gen.get("live")),
 
 
 
871
  "local_api_style": gen.get("api_style"),
872
  "local_base_url": gen.get("base_url"),
873
  "doctrine": DOCTRINE, "kernel_commit": _KERNEL,
@@ -879,11 +1024,16 @@ def register(app: FastAPI) -> dict:
879
  "response": response_text,
880
  "model_selected": sov_enriched,
881
  "lambda_receipt": sov_receipt,
882
- "routed_via": "sovereign_local (%s)" % (gen.get("api_style") or "honest stub"),
 
 
 
 
883
  "local": {k: gen.get(k) for k in
884
  ("wired", "live", "api_style", "base_url", "model", "note", "raw")
885
  if k in gen},
886
  "doctrine": DOCTRINE,
 
887
  }
888
  if _harness_fallback_note:
889
  _sov_resp["harness_note"] = _harness_fallback_note
@@ -1075,26 +1225,42 @@ def register(app: FastAPI) -> dict:
1075
 
1076
  @app.get("/api/a11oy/v1/llm/sovereign/health")
1077
  async def llm_sovereign_health() -> JSONResponse:
1078
- """Ping the sovereign local node (browser UA) and report live + model list.
 
 
 
1079
 
1080
- `live` is True ONLY on a real 2xx JSON response THIS request — never
1081
- fabricated. When SZL_LOCAL_LLM_URL is unset, reports env_present=false and
1082
- an honest note. gpu.a-11-oy.com serves llama3.1:8b; gpu2 serves
1083
- glm-4.7-flash + qwen2.5:3b (per fleet ground truth).
1084
  """
1085
- probe = sovereign_probe()
1086
- wired = bool(probe.get("env_present"))
 
 
 
1087
  return JSONResponse({
 
 
 
 
 
 
 
 
 
1088
  "timestamp": _now(),
1089
  "hub": "a11oy",
1090
- "model_id": "sovereign_local",
1091
- "model_slug": "llama3-szl-finetuned-q4",
 
 
1092
  "env_var": _SOVEREIGN_ENV,
1093
- "env_present": wired,
1094
- "wired": wired, # env present => operator intends local routing
1095
- "live": bool(probe.get("live")), # stronger THIS-request liveness
1096
- "honest_stub": not bool(probe.get("live")),
1097
- "base_url": probe.get("base_url"),
1098
  "api_style": probe.get("api_style"),
1099
  "served_models": probe.get("models", []),
1100
  "configured_model": _sovereign_model_slug(),
@@ -1103,6 +1269,7 @@ def register(app: FastAPI) -> dict:
1103
  "note": probe.get("note", ""),
1104
  "doctrine": DOCTRINE,
1105
  "kernel_commit": _KERNEL,
 
1106
  })
1107
 
1108
  # ── GET /api/a11oy/v1/llm/router/status ───────────────────────────────────
@@ -1142,15 +1309,21 @@ def register(app: FastAPI) -> dict:
1142
  base = _sovereign_base()
1143
  if do_probe:
1144
  sov = sovereign_probe(base)
 
1145
  local_node = {
 
1146
  "env_var": _SOVEREIGN_ENV, "base_url": base or None,
1147
- "env_present": bool(base), "live": bool(sov.get("live")),
 
 
1148
  "served_models": sov.get("models", []), "note": sov.get("note", ""),
1149
  }
1150
  else:
1151
  local_node = {
 
1152
  "env_var": _SOVEREIGN_ENV, "base_url": base or None,
1153
- "env_present": bool(base), "live": None,
 
1154
  "note": "pass ?probe=1 to ping the node for THIS-request liveness",
1155
  }
1156
 
 
53
  _KERNEL = "c7c0ba17"
54
  _LAMBDA_FLOOR = 0.90
55
 
56
+ # Wave M (DEV 1) — first-class sovereign backend id + honest provenance.
57
+ # The founder stands up a LOCAL sovereign model on the Tower (OMEN, RTX 4060 Ti)
58
+ # via Ollama (model tag `llama3-szl-finetuned-q4`, base llama3.1:8b wrapped in a
59
+ # Doctrine-v11 SYSTEM prompt), served OpenAI-compatible at SZL_LOCAL_LLM_URL
60
+ # (default http://localhost:11434/v1). It is NOT reachable from CI/cloud, so this
61
+ # backend MUST degrade to an honest UNAVAILABLE label (never fabricate a response).
62
+ _SOVEREIGN_BACKEND_ID = "szl-sovereign-local" # canonical Wave-M backend id
63
+ _SOVEREIGN_LEGACY_ID = "sovereign_local" # pre-Wave-M alias (kept, additive)
64
+ _SOVEREIGN_MODEL_TAG = "llama3-szl-finetuned-q4" # ollama model tag (Stage A wrapper / Stage B LoRA)
65
+ _SOVEREIGN_PROVENANCE = "SZL sovereign (Ollama, local, Doctrine-v11 system prompt)"
66
+ # Honest label vocabulary (Doctrine v11): the sovereign backend is LIVE only when
67
+ # the node answers this request; otherwise UNAVAILABLE (never SIMULATED/fabricated).
68
+ _LABEL_LIVE = "LIVE"
69
+ _LABEL_UNAVAILABLE = "UNAVAILABLE"
70
+
71
  # ─────────────────────────────────────────────────────────────────────────────
72
  # THE CANONICAL LLM ROSTER — a11oy is the hub; every model lives here.
73
  # Source of truth: OPERATOR_FULL_CAPABILITY_BRIEF_2026-05-31_2135.md §2,
 
218
  "Honest stub when the env is unset or the node is offline.",
219
  "honest_stub": True,
220
  },
221
+ # ── WAVE M (DEV 1): FIRST-CLASS SOVEREIGN BACKEND (own-metal, routed FIRST) ──
222
+ # `szl-sovereign-local` is the canonical Wave-M backend id for the founder's
223
+ # Tower model (Ollama, model tag llama3-szl-finetuned-q4, Doctrine-v11 system
224
+ # prompt). It targets SZL_LOCAL_LLM_URL (default http://localhost:11434/v1,
225
+ # OpenAI-compatible /chat/completions; native Ollama /api/generate fallback).
226
+ # Per the router's "own metal first" doctrine it slots FIRST in the routing
227
+ # order (before free, before paid) — BUT ONLY when the node answers live this
228
+ # request. When the Tower is offline it degrades to an HONEST UNAVAILABLE
229
+ # label; it NEVER fabricates a response.
230
+ {
231
+ "model_id": _SOVEREIGN_BACKEND_ID, # "szl-sovereign-local"
232
+ "display_name": "SZL Sovereign Local (own-metal, Doctrine-v11)",
233
+ "provider": _SOVEREIGN_PROVENANCE, # "SZL sovereign (Ollama, local, Doctrine-v11 system prompt)"
234
+ "provider_slug": "szl-sovereign",
235
+ "tier": 5,
236
+ "tier_name": "sovereign",
237
+ "context_window": 32_000,
238
+ "use_case": "Own-metal sovereign inference — routed FIRST when reachable",
239
+ "why": "SZL's OWN governed model on the Tower; zero external dependency; "
240
+ "routed before free/paid per 'own metal first' doctrine when live",
241
+ "routing_condition": "own-metal FIRST when node is reachable; else honest UNAVAILABLE",
242
+ "api_env_var": "SZL_LOCAL_LLM_URL",
243
+ "api_base": "http://localhost:11434/v1", # OpenAI-compatible base (default)
244
+ "model_slug": _SOVEREIGN_MODEL_TAG, # "llama3-szl-finetuned-q4"
245
+ "modalities": ["text"],
246
+ "streaming": True,
247
+ "operator_mirrored": False,
248
+ "ecosystem_mirror": ["policy", "reasoning", "killinchu"],
249
+ "lean_gate": "deterministicReplay",
250
+ "own_metal": True,
251
+ "route_first": True,
252
+ "legacy_alias": _SOVEREIGN_LEGACY_ID,
253
+ "notes": "Stage A = Doctrine-v11 system-prompt derivative (now); Stage B = real "
254
+ "LoRA fine-tune (later) under the SAME model tag. Wired when "
255
+ "SZL_LOCAL_LLM_URL is set; makes a REAL guarded ollama /api/generate "
256
+ "or OpenAI-compatible /v1/chat/completions call and is LIVE ONLY when "
257
+ "the node answers this request. Honest UNAVAILABLE otherwise "
258
+ "(see GET /api/a11oy/v1/llm/sovereign/health). The Tower is NOT "
259
+ "reachable from CI/cloud — CI reports UNAVAILABLE, never fabricates.",
260
+ "honest_stub": True,
261
+ },
262
  # ── ADDITIONAL: Perplexity (online search-augmented) ──
263
  {
264
  "model_id": "perplexity_sonar_pro",
 
417
  _SOVEREIGN_GEN_TIMEOUT_S = 120.0
418
 
419
 
420
+ # Doctrine v11 default: OpenAI-compatible base at the Tower's Ollama /v1 endpoint.
421
+ _SOVEREIGN_DEFAULT_URL = "http://localhost:11434/v1"
422
+
423
+
424
  def _sovereign_base() -> str:
425
+ """Resolve the sovereign-local base URL from env.
426
+
427
+ Wave M: default to the OpenAI-compatible Tower endpoint
428
+ (http://localhost:11434/v1) when SZL_LOCAL_LLM_URL is unset, so the backend
429
+ always has a concrete target to reachability-probe. The probe/generate paths
430
+ normalise a trailing `/v1` back to the Ollama root for native /api calls, so
431
+ either form (`.../11434` or `.../11434/v1`) works.
432
+ """
433
+ val = (os.environ.get(_SOVEREIGN_ENV, "") or "").strip().rstrip("/")
434
+ return val or _SOVEREIGN_DEFAULT_URL
435
+
436
+
437
+ def _sovereign_env_present() -> bool:
438
+ """True only when SZL_LOCAL_LLM_URL was explicitly set (operator intent)."""
439
+ return bool((os.environ.get(_SOVEREIGN_ENV, "") or "").strip())
440
+
441
+
442
+ def _ollama_root(base: str) -> str:
443
+ """Return the Ollama server root, stripping a trailing OpenAI-compat `/v1`."""
444
+ b = (base or "").rstrip("/")
445
+ if b.endswith("/v1"):
446
+ b = b[: -len("/v1")]
447
+ return b
448
 
449
 
450
  def _sovereign_model_slug() -> str:
 
484
  """
485
  base = (base or _sovereign_base())
486
  to = _SOVEREIGN_PROBE_TIMEOUT_S if timeout is None else float(timeout)
487
+ env_present = _sovereign_env_present()
488
  out: dict[str, Any] = {
489
  "env_var": _SOVEREIGN_ENV,
490
+ "env_present": env_present,
491
  "base_url": base or None,
492
  "live": False,
493
  "models": [],
494
  "probed": [],
495
  "note": "",
496
  }
497
+ # Wave M: we ALWAYS have a concrete base (env or the localhost /v1 default), so
498
+ # we always reachability-probe. env_present tells the caller whether the
499
+ # operator explicitly targeted a Tower or we fell back to the localhost default
500
+ # (which, from CI/cloud, is honestly UNAVAILABLE).
501
+ b = _ollama_root(base)
502
  # 1) ollama native /api/tags
503
  tags_url = b + "/api/tags"
504
  doc, err = _http_json(tags_url, timeout=to)
 
523
  out["api_style"] = "openai /v1"
524
  out["note"] = "node live (OpenAI-compatible /v1/models); model list is real THIS request."
525
  return out
526
+ _tgt = ("SZL_LOCAL_LLM_URL=%s" % base) if env_present else (
527
+ "SZL_LOCAL_LLM_URL unset — probing localhost default %s" % base)
528
+ out["note"] = ("Sovereign node NOT reachable this request — honest UNAVAILABLE "
529
+ "(never fabricate). %s. The Tower is not reachable from CI/cloud. "
530
+ "Errors: %s" % (_tgt, "; ".join(
531
+ str(p.get("error")) for p in out["probed"] if p.get("error"))))
532
  return out
533
 
534
 
 
547
  res: dict[str, Any] = {
548
  "wired": False, "live": False, "text": None, "model": model,
549
  "api_style": None, "base_url": base or None, "env_var": _SOVEREIGN_ENV,
550
+ "env_present": _sovereign_env_present(), "note": "",
551
  }
552
+ b = _ollama_root(base)
 
 
 
 
553
  # 1) ollama native /api/generate
554
  gen_url = b + "/api/generate"
555
  body = json.dumps({"model": model, "prompt": prompt, "stream": False}).encode("utf-8")
 
580
  if isinstance(doc2.get("usage"), dict):
581
  res["raw"] = {"usage": doc2["usage"]}
582
  return res
583
+ res["note"] = ("Sovereign node NOT reachable this request — honest UNAVAILABLE "
584
+ "(never fabricate a response). Tier selection + Λ-receipt still "
585
+ "REAL. Errors: %s / %s" % (err or "", err2 or ""))
586
  return res
587
 
588
 
 
703
  env_var = m.get("api_env_var", "") or ""
704
  model_id = m.get("model_id", "")
705
 
706
+ if model_id in (_SOVEREIGN_LEGACY_ID, _SOVEREIGN_BACKEND_ID) or env_var == _SOVEREIGN_ENV:
707
  base = _sovereign_base()
708
+ env_present = _sovereign_env_present()
709
+ # `wired` = operator intent (env explicitly set). `label` = honest state:
710
+ # LIVE only when the node answers this request; else UNAVAILABLE.
711
+ wired = env_present
712
  out["api_key_wired"] = wired
713
  out["wired"] = wired
714
+ out["provider"] = m.get("provider", _SOVEREIGN_PROVENANCE)
715
  out["env_used"] = _SOVEREIGN_ENV
716
+ out["env_present"] = env_present
717
  out["base_url"] = base or m.get("api_base")
 
718
  out["is_local"] = True
719
+ out["own_metal"] = True
720
+ # Default (no probe): honest UNAVAILABLE until proven live this request.
721
+ out["honest_stub"] = True
722
+ out["label"] = _LABEL_UNAVAILABLE
723
+ out["reachable"] = False
724
  if probe_local:
725
  probe = sovereign_probe(base)
726
+ live = bool(probe.get("live"))
727
+ out["local_live"] = live
728
+ out["reachable"] = live
729
  out["local_models"] = probe.get("models", [])
730
  out["local_probe_note"] = probe.get("note", "")
731
+ out["api_style"] = probe.get("api_style")
732
+ # honest_stub clears / label flips to LIVE ONLY when actually live.
733
+ out["honest_stub"] = not live
734
+ out["label"] = _LABEL_LIVE if live else _LABEL_UNAVAILABLE
735
  return out
736
 
737
  wired = _api_key_wired(env_var)
 
794
  "base_url": m.get("base_url"),
795
  "honest_stub": bool(m.get("honest_stub", True)),
796
  "is_local": bool(m.get("is_local")),
797
+ **({"label": m.get("label"), "reachable": m.get("reachable"),
798
+ "own_metal": bool(m.get("own_metal"))} if m.get("own_metal") else {}),
799
  } for m in models]
800
  all_stub = (len(wired) == 0)
801
+ # Wave-M sovereign availability snapshot: honest reachability of the
802
+ # own-metal backend. `reachable`/`label` come from a real probe only when
803
+ # ?probe=1; otherwise we honestly report unknown (probe not run).
804
+ _sov_badge = next((m for m in models if m.get("model_id") == _SOVEREIGN_BACKEND_ID), None)
805
+ sovereign_snapshot = {
806
+ "backend_id": _SOVEREIGN_BACKEND_ID,
807
+ "model": _SOVEREIGN_MODEL_TAG,
808
+ "provider": _SOVEREIGN_PROVENANCE,
809
+ "url": _sovereign_base(),
810
+ "env_present": _sovereign_env_present(),
811
+ "probed": do_probe,
812
+ "reachable": (bool(_sov_badge.get("reachable")) if (do_probe and _sov_badge) else None),
813
+ "label": (_sov_badge.get("label") if (do_probe and _sov_badge)
814
+ else "UNPROBED (pass ?probe=1 for THIS-request reachability)"),
815
+ "route_order": "own-metal/sovereign FIRST (when reachable) → free → paid",
816
+ "health_endpoint": "/api/a11oy/v1/llm/sovereign/health",
817
+ }
818
  return JSONResponse({
819
  "timestamp": _now(),
820
  "hub": "a11oy",
 
835
  "route_endpoint": "/api/a11oy/v1/llm/route",
836
  "router_status_endpoint": "/api/a11oy/v1/llm/router/status",
837
  "sovereign_health_endpoint": "/api/a11oy/v1/llm/sovereign/health",
838
+ "sovereign": sovereign_snapshot,
839
  "doctrine": DOCTRINE,
840
  "kernel_commit": _KERNEL,
841
  "honest_note": (
 
932
 
933
  lam = _lambda_gm(axis_scores)
934
 
935
+ # ── SOVEREIGN / OWN-METAL selection (routed FIRST when reachable) ─────
936
+ # Doctrine: "own metal first" — the sovereign backend slots FIRST in the
937
+ # routing order (before free, before paid) BUT ONLY when the Tower node
938
+ # actually answers this request. Selection triggers when EITHER:
939
+ # (a) the caller asks for it explicitly (model_id in the sovereign ids |
940
+ # task_hint in {sovereign,local,offline,…} | prefer_local=true), OR
941
+ # (b) OWN-METAL-FIRST: the node is reachable this request (auto-probe),
942
+ # unless the caller opts out with prefer_local=false + a cloud model,
943
+ # (c) offline preference (env set, no cloud key, offline_mode=true).
944
+ # When the node is NOT reachable we NEVER fabricate — the caller falls
945
+ # through to free/paid tiers, and if sovereign was explicitly requested we
946
+ # return an honest UNAVAILABLE label (no fabricated text).
947
  _req_model = str(body.get("model_id", "")).strip().lower()
948
+ _sov_ids = {_SOVEREIGN_LEGACY_ID, _SOVEREIGN_BACKEND_ID}
949
+ _prefer_local_raw = body.get("prefer_local", None)
950
+ _prefer_local = bool(_prefer_local_raw)
951
+ _opted_out = (_prefer_local_raw is False) # explicit prefer_local=false
952
  _sov_hints = {"sovereign", "local", "offline", "air-gap", "airgap"}
953
  _any_cloud_wired = any(_api_key_wired(ev) for ev, _ in _PROVIDER_ENV_VARS)
954
  _sov_base = _sovereign_base()
955
+ _explicit_sovereign = (
956
+ _req_model in _sov_ids or task_hint in _sov_hints or _prefer_local
957
+ )
958
+ _offline_pref = (
959
+ _sovereign_env_present() and not _any_cloud_wired
960
+ and str(body.get("offline_mode", "")).lower() in ("1", "true", "yes")
961
  )
962
+ # OWN-METAL-FIRST: reachability-probe (short timeout) and, if the node is
963
+ # LIVE this request, route sovereign FIRST — unless the caller explicitly
964
+ # opted out. If the caller explicitly requested sovereign we ALSO select it
965
+ # (even when unreachable) so we can return an honest UNAVAILABLE.
966
+ _sov_probe = sovereign_probe(_sov_base)
967
+ _sov_reachable = bool(_sov_probe.get("live"))
968
+ _own_metal_first = _sov_reachable and not _opted_out
969
+ _want_sovereign = _explicit_sovereign or _offline_pref or _own_metal_first
970
  if _want_sovereign:
971
+ # Prefer the first-class Wave-M backend id; fall back to legacy alias.
972
+ sov_model = (_MODEL_BY_ID.get(_SOVEREIGN_BACKEND_ID)
973
+ or _MODEL_BY_ID.get(_SOVEREIGN_LEGACY_ID)
974
+ or MODEL_REGISTRY[0])
975
  sov_enriched = _enrich_model(sov_model)
976
  gen = sovereign_generate(prompt or _DEFAULT_SOVEREIGN_PROMPT)
977
+ _live = bool(gen.get("live"))
978
+ sov_enriched["wired"] = _sovereign_env_present()
979
+ sov_enriched["reachable"] = _live
980
+ sov_enriched["local_live"] = _live
981
+ sov_enriched["honest_stub"] = not _live
982
+ sov_enriched["label"] = _LABEL_LIVE if _live else _LABEL_UNAVAILABLE
983
+ if _explicit_sovereign:
984
+ _why = "explicit request"
985
+ elif _own_metal_first:
986
+ _why = "own-metal-first (node reachable → routed before free/paid)"
987
+ else:
988
+ _why = "offline preference, no cloud key wired"
989
+ sov_reason = "%s selected (%s); " % (sov_model.get("model_id"), _why)
990
+ if _live:
991
+ sov_reason += "node LIVE this request — REAL local generation [LIVE]."
992
  response_text = gen.get("text") or ""
993
  else:
994
+ sov_reason += "node NOT reachable — honest UNAVAILABLE (never fabricate)."
995
  response_text = (
996
+ "[UNAVAILABLE] %s (%s) — the sovereign Tower endpoint is not "
997
+ "reachable this request, so NO response is fabricated. %s "
998
+ "Tier selection + Λ=%.4f + receipt are REAL."
999
+ % (sov_model.get("model_id"), sov_model.get("model_slug", ""),
1000
+ gen.get("note", ""), lam))
1001
  sov_receipt = {
1002
  "schema": "szl.llm_route.lambda_receipt/v1",
1003
  "ts": _now(), "hub": "a11oy", "lambda": round(lam, 6),
1004
  "lambda_floor": _LAMBDA_FLOOR, "axis_scores": axis_scores,
1005
  "tier_selected": sov_model.get("tier", 5),
1006
+ "model_id": sov_model.get("model_id"),
1007
+ "model_slug": sov_model.get("model_slug"),
1008
+ "provider": sov_model.get("provider"),
1009
  "model_display": sov_model.get("display_name"),
1010
  "reason": sov_reason, "task_hint": task_hint,
1011
+ "own_metal_first": bool(_own_metal_first),
1012
+ "reachable": _live,
1013
+ "label": _LABEL_LIVE if _live else _LABEL_UNAVAILABLE,
1014
+ "api_key_wired": _sovereign_env_present(),
1015
+ "local_live": _live,
1016
  "local_api_style": gen.get("api_style"),
1017
  "local_base_url": gen.get("base_url"),
1018
  "doctrine": DOCTRINE, "kernel_commit": _KERNEL,
 
1024
  "response": response_text,
1025
  "model_selected": sov_enriched,
1026
  "lambda_receipt": sov_receipt,
1027
+ "label": _LABEL_LIVE if _live else _LABEL_UNAVAILABLE,
1028
+ "reachable": _live,
1029
+ "routed_via": "%s (%s)" % (
1030
+ sov_model.get("model_id"),
1031
+ gen.get("api_style") if _live else "honest UNAVAILABLE"),
1032
  "local": {k: gen.get(k) for k in
1033
  ("wired", "live", "api_style", "base_url", "model", "note", "raw")
1034
  if k in gen},
1035
  "doctrine": DOCTRINE,
1036
+ "conjecture_note": "Λ = Conjecture 1 — advisory, never 'green'/theorem.",
1037
  }
1038
  if _harness_fallback_note:
1039
  _sov_resp["harness_note"] = _harness_fallback_note
 
1225
 
1226
  @app.get("/api/a11oy/v1/llm/sovereign/health")
1227
  async def llm_sovereign_health() -> JSONResponse:
1228
+ """Ping the sovereign local node (browser UA) and report honest health.
1229
+
1230
+ Wave-M contract (DEV 1): returns the required compact block
1231
+ {reachable, model, url, provider, label} — plus rich diagnostics.
1232
 
1233
+ `reachable` is True ONLY on a real 2xx JSON response THIS request — never
1234
+ fabricated. When the Tower is offline / unreachable (e.g. from CI/cloud),
1235
+ `reachable=false` and `label=UNAVAILABLE` (honest degradation, no fake).
 
1236
  """
1237
+ base = _sovereign_base()
1238
+ probe = sovereign_probe(base)
1239
+ reachable = bool(probe.get("live"))
1240
+ env_present = bool(probe.get("env_present"))
1241
+ label = _LABEL_LIVE if reachable else _LABEL_UNAVAILABLE
1242
  return JSONResponse({
1243
+ # ── Wave-M required compact contract ──
1244
+ # `model` = canonical sovereign model tag; `configured_model` (below)
1245
+ # is the runtime-overridable ollama tag the node is asked to serve.
1246
+ "reachable": reachable,
1247
+ "model": _SOVEREIGN_MODEL_TAG,
1248
+ "url": base,
1249
+ "provider": _SOVEREIGN_PROVENANCE,
1250
+ "label": label,
1251
+ # ── rich diagnostics (additive, honest) ──
1252
  "timestamp": _now(),
1253
  "hub": "a11oy",
1254
+ "backend_id": _SOVEREIGN_BACKEND_ID,
1255
+ "model_id": _SOVEREIGN_BACKEND_ID,
1256
+ "legacy_alias": _SOVEREIGN_LEGACY_ID,
1257
+ "model_slug": _SOVEREIGN_MODEL_TAG,
1258
  "env_var": _SOVEREIGN_ENV,
1259
+ "env_present": env_present,
1260
+ "wired": env_present, # env present => operator intends local routing
1261
+ "live": reachable, # THIS-request liveness (== reachable)
1262
+ "honest_stub": not reachable,
1263
+ "base_url": base,
1264
  "api_style": probe.get("api_style"),
1265
  "served_models": probe.get("models", []),
1266
  "configured_model": _sovereign_model_slug(),
 
1269
  "note": probe.get("note", ""),
1270
  "doctrine": DOCTRINE,
1271
  "kernel_commit": _KERNEL,
1272
+ "conjecture_note": "Λ = Conjecture 1 — advisory, never 'green'/theorem.",
1273
  })
1274
 
1275
  # ── GET /api/a11oy/v1/llm/router/status ───────────────────────────────────
 
1309
  base = _sovereign_base()
1310
  if do_probe:
1311
  sov = sovereign_probe(base)
1312
+ _live = bool(sov.get("live"))
1313
  local_node = {
1314
+ "backend_id": _SOVEREIGN_BACKEND_ID,
1315
  "env_var": _SOVEREIGN_ENV, "base_url": base or None,
1316
+ "env_present": _sovereign_env_present(), "live": _live,
1317
+ "reachable": _live,
1318
+ "label": _LABEL_LIVE if _live else _LABEL_UNAVAILABLE,
1319
  "served_models": sov.get("models", []), "note": sov.get("note", ""),
1320
  }
1321
  else:
1322
  local_node = {
1323
+ "backend_id": _SOVEREIGN_BACKEND_ID,
1324
  "env_var": _SOVEREIGN_ENV, "base_url": base or None,
1325
+ "env_present": _sovereign_env_present(), "live": None,
1326
+ "reachable": None, "label": "UNPROBED",
1327
  "note": "pass ?probe=1 to ping the node for THIS-request liveness",
1328
  }
1329