betterwithage Claude Opus 4.7 commited on
Commit
470a64f
·
verified ·
1 Parent(s): 111b0a8

deploy(hf): sync szl-holdings/a11oy@54a649b250083ce10352dc11064d220b26589c67 derived COPY set

Browse files

Reusable Dockerfile-COPY-derived deploy from szl-holdings/a11oy 54a649b250083ce10352dc11064d220b26589c67.
Files: 1187 Pruned: 0
Derived from Dockerfile COPY sources (NO hand-maintained allowlist).

Signed-off-by: SZL Holdings <noreply@szlholdings.ai>
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>

Files changed (1) hide show
  1. szl_budget_router.py +1002 -121
szl_budget_router.py CHANGED
@@ -10,63 +10,97 @@
10
  # GraphPlanner — ulab-uiuc/GraphPlanner — MIT —
11
  # https://github.com/ulab-uiuc/GraphPlanner
12
  #
13
- # We adopt THREE patterns and EVOLVE them into one a11oy-native mechanism:
14
- # • BudgetMem's module-level BUDGET TIERS (Low/Mid/High) + a lightweight query-aware
15
- # budget-tier ROUTER. We map the tiers onto Warhacker MISSION CONSTRAINTS —
16
- # Tactical (≤1s), Operational (≤10s), Strategic (≤1min) — so sovereign hardware is
17
- # never choked by unnecessary chain-of-thought on a time-critical decision.
18
- # • GraphPlanner's idea of routing by graph-derived signal: we route by DECISION RISK
19
- # (sensitivity × time-budget × reversibility) rather than by learned RL value.
20
- # • MemSkill's meta-memory ("learn WHAT to preserve"): we evolve it into "DECISION
21
- # SKELETONS" — reusable, distilled patterns of how the substrate successfully
22
- # navigated a class of governed trade-offs, kept across runs.
23
- # Original, dependency-free implementation — no BudgetMem/MemSkill/GraphPlanner
24
- # source, no torch/numpy. Reuses the live governance catalog from
25
- # szl_governance_gateway when present. Λ = Conjecture 1. Doctrine v11 LOCKED 749/14/163.
26
- """szl_budget_router — ADDITIVE cost-aware budget-tier router + Decision Skeletons.
27
-
28
- Endpoints (mounted before the SPA catch-all):
29
- GET /budget-router — operator tab (HTML, 0 CDN)
30
- GET /api/a11oy/v1/budget/tiers — the 3 mission budget tiers
31
- POST /api/a11oy/v1/budget/route — route a decision by risk → tier → model
32
- GET /api/a11oy/v1/budget/skeletons — learned Decision Skeletons (meta-memory)
33
  """
34
- from __future__ import annotations
35
 
 
 
 
36
  import re
37
  import time
38
- from typing import Any
 
 
39
 
40
  from fastapi import FastAPI, Request
41
  from fastapi.responses import HTMLResponse, JSONResponse
42
 
43
  DOCTRINE = {"version": "v11", "counts": "749/14/163", "lambda": "Conjecture 1"}
44
 
45
- # --- Mission budget tiers (BudgetMem Low/Mid/High → Warhacker mission constraints) ---
 
 
 
46
  TIERS: list[dict[str, Any]] = [
47
- {"tier": "TACTICAL", "budget": "LOW", "deadline_s": 1, "max_model_tier": "T1",
48
- "cot": "none", "note": "time-critical: cheapest sufficient model, no chain-of-thought"},
49
- {"tier": "OPERATIONAL", "budget": "MID", "deadline_s": 10, "max_model_tier": "T3",
50
- "cot": "short", "note": "balanced: mid-tier reasoning with a short rationale"},
51
- {"tier": "STRATEGIC", "budget": "HIGH", "deadline_s": 60, "max_model_tier": "T6",
52
- "cot": "full", "note": "deliberate: high-tier model, full chain-of-thought permitted"},
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
53
  ]
54
- _TIER_BY_NAME = {t["tier"]: t for t in TIERS}
55
 
56
- # Reversibility cues raise risk: an irreversible action deserves a bigger budget.
57
  _IRREVERSIBLE_RX = re.compile(
58
  r"\b(launch|fire|strike|delete|destroy|deploy|commit|authoriz|release|publish|"
59
- r"terminate|engage|weapon|kill)\b", re.I)
60
- _REVERSIBLE_RX = re.compile(r"\b(draft|preview|simulate|estimate|summari|triage|sort|list|query)\b", re.I)
 
 
 
 
 
61
 
62
 
63
  def _classify_sensitivity(text: str, declared: str | None) -> dict[str, Any]:
64
- """Reuse the live governance classifier when present; else a safe fallback."""
 
65
  try:
66
- import szl_governance_gateway as _gg # type: ignore
67
- return _gg.classify(text, declared)
 
68
  except Exception:
69
- return {"class": (declared or "PUBLIC").upper(), "rank": 0, "signals": ["fallback"]}
 
 
 
 
70
 
71
 
72
  def _reversibility(text: str) -> tuple[float, str]:
@@ -77,84 +111,113 @@ def _reversibility(text: str) -> tuple[float, str]:
77
  return 0.5, "uncertain"
78
 
79
 
80
- def assess_risk(query: str, *, declared: str | None = None,
81
- deadline_s: float | None = None) -> dict[str, Any]:
82
- """Decision risk = f(sensitivity, reversibility). Higher risk → higher tier.
83
- A tighter deadline CAPS the tier (you cannot afford full CoT under 1s)."""
84
- sens = _classify_sensitivity(query, declared)
85
- sens_norm = sens["rank"] / 3.0 # 0..1 over PUBLIC..SECRET
86
- rev_score, rev_label = _reversibility(query)
87
- # Weighted risk: sensitivity dominates, reversibility modulates.
88
- risk = round(0.65 * sens_norm + 0.35 * rev_score, 4)
 
 
 
 
89
  if risk >= 0.6:
90
- want = "STRATEGIC"
91
  elif risk >= 0.3:
92
- want = "OPERATIONAL"
93
  else:
94
- want = "TACTICAL"
95
- # Deadline cap: if the operator imposes a tighter deadline, downshift the tier.
96
- chosen = want
97
  capped_by_deadline = False
98
  if deadline_s is not None:
99
- affordable = [t["tier"] for t in TIERS if t["deadline_s"] <= deadline_s]
 
 
100
  order = ["TACTICAL", "OPERATIONAL", "STRATEGIC"]
101
  if affordable:
102
- best_affordable = max(affordable, key=lambda n: order.index(n))
103
- if order.index(best_affordable) < order.index(want):
104
  chosen = best_affordable
105
  capped_by_deadline = True
106
  else:
107
- chosen = "TACTICAL"; capped_by_deadline = True
 
 
108
  return {
109
- "sensitivity": sens, "reversibility": {"score": rev_score, "label": rev_label},
110
- "risk": risk, "risk_band": ("HIGH" if risk >= 0.6 else "MODERATE" if risk >= 0.3 else "LOW"),
111
- "risk_implied_tier": want, "chosen_tier": chosen,
 
 
 
 
 
 
 
 
112
  "capped_by_deadline": capped_by_deadline,
113
  }
114
 
115
 
116
  def _pick_model(max_model_tier: str, sensitivity_rank: int) -> dict[str, Any] | None:
117
- """Smallest sufficient model within the tier ceiling, respecting the air-gap floor."""
 
118
  try:
119
- import szl_governance_gateway as _gg # type: ignore
120
- cat = getattr(_gg, "CATALOG", [])
121
- order = getattr(_gg, "_TIER_ORDER", {})
122
- floor = getattr(_gg, "_AIRGAP_FLOOR", 2)
 
123
  except Exception:
124
  return None
125
- ceil = order.get(max_model_tier, 9)
126
- airgap_required = sensitivity_rank >= floor
127
- elig = [m for m in cat
128
- if order.get(m["tier"], 9) <= ceil
129
- and (not airgap_required or m.get("zone") == "AIRGAP")]
130
- if not elig:
 
 
 
 
131
  return None
132
- elig.sort(key=lambda m: (order.get(m["tier"], 9), -m.get("ctx", 0)))
133
- return elig[0]
134
 
135
 
136
- def route(query: str, *, declared: str | None = None,
137
- deadline_s: float | None = None) -> dict[str, Any]:
 
 
 
 
138
  risk = assess_risk(query, declared=declared, deadline_s=deadline_s)
139
  tier = _TIER_BY_NAME[risk["chosen_tier"]]
140
  model = _pick_model(tier["max_model_tier"], risk["sensitivity"]["rank"])
141
  decision = {
142
- "query": query, "tier": tier, "risk": risk, "chosen_model": model,
 
 
 
143
  "airgap_required": risk["sensitivity"]["rank"] >= 2,
144
- "policy": "risk(sensitivity×reversibility) → mission tier (deadline-capped) → smallest sufficient air-gap-safe model",
145
- "honest": None if model else "no model within tier ceiling + air-gap floor — raise the tier or add an AIRGAP model",
 
 
 
 
 
 
 
146
  }
147
  _learn_skeleton(decision)
148
  return decision
149
 
150
 
151
- # ---------------------------------------------------------------------------
152
- # Decision Skeletons (MemSkill meta-memory, evolved). A skeleton is a DISTILLED,
153
- # reusable pattern for a class of decisions: (risk_band, tier, airgap) → how it was
154
- # handled, plus how many times that pattern recurred. We keep WHAT generalizes
155
- # (the band+tier+model family) and forget the per-query noise — the "skill of
156
- # remembering" applied to governance trade-offs. In-memory ring (bounded), no PII.
157
- # ---------------------------------------------------------------------------
158
  _SKELETONS: dict[str, dict[str, Any]] = {}
159
 
160
 
@@ -162,70 +225,889 @@ def _learn_skeleton(decision: dict[str, Any]) -> None:
162
  band = decision["risk"]["risk_band"]
163
  tier = decision["tier"]["tier"]
164
  airgap = decision["airgap_required"]
165
- model_fam = ((decision.get("chosen_model") or {}).get("id") or "—").split("-")[0]
166
- key = f"{band}|{tier}|{'airgap' if airgap else 'cloud'}|{model_fam}"
167
- sk = _SKELETONS.get(key)
168
- if sk is None:
169
- sk = {
170
- "key": key, "risk_band": band, "tier": tier,
171
- "airgap": airgap, "model_family": model_fam,
172
- "uses": 0, "first_seen": time.time(), "last_seen": time.time(),
 
 
 
 
 
173
  "lesson": _lesson(band, tier, airgap),
174
  }
175
- _SKELETONS[key] = sk
176
- sk["uses"] += 1
177
- sk["last_seen"] = time.time()
178
- # Bound the meta-memory: keep the 64 most-used skeletons (forget the rest).
179
  if len(_SKELETONS) > 64:
180
- worst = min(_SKELETONS.values(), key=lambda s: (s["uses"], s["last_seen"]))
181
- _SKELETONS.pop(worst["key"], None)
 
 
 
182
 
183
 
184
  def _lesson(band: str, tier: str, airgap: bool) -> str:
185
- z = "air-gapped (classified)" if airgap else "cloud-eligible"
186
- return (f"{band}-risk decisions routed to the {tier} budget tier on {z} compute "
187
- f"resolved within budget; reuse this skeleton for the same risk class.")
 
 
188
 
189
 
190
  def skeletons() -> dict[str, Any]:
191
- sks = sorted(_SKELETONS.values(), key=lambda s: s["uses"], reverse=True)
192
- return {"skeleton_count": len(sks), "skeletons": sks,
193
- "hint": None if sks else "no skeletons yet — route decisions via POST /api/a11oy/v1/budget/route to learn them"}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
194
 
195
 
196
  def register(app: FastAPI, ns: str = "a11oy") -> str:
197
  @app.get(f"/api/{ns}/v1/budget/tiers", include_in_schema=False)
198
- async def _tiers() -> JSONResponse:
199
- return JSONResponse({"doctrine": DOCTRINE, "tiers": TIERS,
200
- "pattern_source": "ViktorAxelsen/BudgetMem (Apache-2.0) — module budget tiers, evolved to mission constraints"})
 
 
 
 
 
 
 
 
201
 
202
  @app.post(f"/api/{ns}/v1/budget/route", include_in_schema=False)
203
- async def _route(request: Request) -> JSONResponse:
204
  try:
205
  body = await request.json()
206
  except Exception:
207
  body = {}
208
- q = (body or {}).get("query") or (body or {}).get("q") or ""
209
  declared = (body or {}).get("classification")
210
- dl = (body or {}).get("deadline_s")
211
  try:
212
- dl = float(dl) if dl is not None else None
213
  except Exception:
214
- dl = None
215
- if not q:
216
  return JSONResponse({"error": "provide {query: ...}"}, status_code=400)
217
- return JSONResponse({"doctrine": DOCTRINE, **route(q, declared=declared, deadline_s=dl)})
 
 
 
 
 
218
 
219
  @app.get(f"/api/{ns}/v1/budget/skeletons", include_in_schema=False)
220
- async def _sk() -> JSONResponse:
221
- return JSONResponse({"doctrine": DOCTRINE, **skeletons(),
222
- "pattern_source": "ViktorAxelsen/MemSkill (Apache-2.0) — meta-memory, evolved to Decision Skeletons"})
 
 
 
 
 
 
 
 
 
 
223
 
224
  @app.get("/budget-router", include_in_schema=False)
225
- async def _page() -> HTMLResponse:
226
  return HTMLResponse(_PAGE_HTML)
227
 
228
- return f"budget-router mounted: GET /budget-router + /api/{ns}/v1/budget/(tiers|route|skeletons)"
 
 
 
229
 
230
 
231
  _PAGE_HTML = """<!DOCTYPE html>
@@ -302,7 +1184,6 @@ patterns: BudgetMem + MemSkill (Apache-2.0), GraphPlanner (MIT), evolved · sove
302
  </div>
303
  <script>
304
  const $=s=>document.querySelector(s);
305
- function band(b){return b==='HIGH'?'red':b==='MODERATE'?'amber':'green'}
306
  async function tiers(){
307
  const d=await(await fetch('/api/a11oy/v1/budget/tiers')).json();
308
  const box=$('#tiers');box.innerHTML='';
@@ -341,9 +1222,9 @@ async function loadSk(){
341
  }
342
  }
343
  $('#go').addEventListener('click',route);
344
- $('#q').addEventListener('keydown',e=>{if(e.key==='Enter')route()});
345
  $('#ref').addEventListener('click',loadSk);
346
- document.querySelectorAll('.chip').forEach(c=>c.addEventListener('click',()=>{$('#q').value=c.textContent;route();}));
347
  tiers();loadSk();
348
  </script>
349
  </body></html>"""
 
10
  # GraphPlanner — ulab-uiuc/GraphPlanner — MIT —
11
  # https://github.com/ulab-uiuc/GraphPlanner
12
  #
13
+ # We retain those cited budget/meta-memory patterns and extend the same bounded
14
+ # runtime seam with original SZL token-ingress controls. No third-party source is
15
+ # copied into the implementation.
16
+ """Cost-aware budget routing, Decision Skeletons, and semantic token ingress.
17
+
18
+ Existing endpoints (preserved):
19
+ GET /budget-router
20
+ GET /api/a11oy/v1/budget/tiers
21
+ POST /api/a11oy/v1/budget/route
22
+ GET /api/a11oy/v1/budget/skeletons
23
+
24
+ Token-ingress endpoints (bounded computation only):
25
+ GET /api/a11oy/v1/token-ingress/status
26
+ POST /api/a11oy/v1/token-ingress/route
27
+ POST /api/a11oy/v1/token-ingress/qualify
28
+ POST /api/a11oy/v1/token-ingress/verification-budget
29
+
30
+ No token-ingress route reads arbitrary repository files, persists a prefix,
31
+ contacts a provider, mutates a model, signs data, or performs a deployment.
 
32
  """
 
33
 
34
+ import hashlib
35
+ import json
36
+ import math
37
  import re
38
  import time
39
+ from dataclasses import dataclass, field
40
+ from pathlib import Path
41
+ from typing import Any, Iterable, Mapping, Sequence
42
 
43
  from fastapi import FastAPI, Request
44
  from fastapi.responses import HTMLResponse, JSONResponse
45
 
46
  DOCTRINE = {"version": "v11", "counts": "749/14/163", "lambda": "Conjecture 1"}
47
 
48
+ # ---------------------------------------------------------------------------
49
+ # Existing mission budget router
50
+ # ---------------------------------------------------------------------------
51
+
52
  TIERS: list[dict[str, Any]] = [
53
+ {
54
+ "tier": "TACTICAL",
55
+ "budget": "LOW",
56
+ "deadline_s": 1,
57
+ "max_model_tier": "T1",
58
+ "cot": "none",
59
+ "note": "time-critical: cheapest sufficient model, no chain-of-thought",
60
+ },
61
+ {
62
+ "tier": "OPERATIONAL",
63
+ "budget": "MID",
64
+ "deadline_s": 10,
65
+ "max_model_tier": "T3",
66
+ "cot": "short",
67
+ "note": "balanced: mid-tier reasoning with a short rationale",
68
+ },
69
+ {
70
+ "tier": "STRATEGIC",
71
+ "budget": "HIGH",
72
+ "deadline_s": 60,
73
+ "max_model_tier": "T6",
74
+ "cot": "full",
75
+ "note": "deliberate: high-tier model, full chain-of-thought permitted",
76
+ },
77
  ]
78
+ _TIER_BY_NAME = {tier["tier"]: tier for tier in TIERS}
79
 
 
80
  _IRREVERSIBLE_RX = re.compile(
81
  r"\b(launch|fire|strike|delete|destroy|deploy|commit|authoriz|release|publish|"
82
+ r"terminate|engage|weapon|kill)\b",
83
+ re.I,
84
+ )
85
+ _REVERSIBLE_RX = re.compile(
86
+ r"\b(draft|preview|simulate|estimate|summari|triage|sort|list|query)\b",
87
+ re.I,
88
+ )
89
 
90
 
91
  def _classify_sensitivity(text: str, declared: str | None) -> dict[str, Any]:
92
+ """Reuse the live governance classifier when present; otherwise fail safe."""
93
+
94
  try:
95
+ import szl_governance_gateway as governance_gateway # type: ignore
96
+
97
+ return governance_gateway.classify(text, declared)
98
  except Exception:
99
+ return {
100
+ "class": (declared or "PUBLIC").upper(),
101
+ "rank": 0,
102
+ "signals": ["fallback"],
103
+ }
104
 
105
 
106
  def _reversibility(text: str) -> tuple[float, str]:
 
111
  return 0.5, "uncertain"
112
 
113
 
114
+ def assess_risk(
115
+ query: str,
116
+ *,
117
+ declared: str | None = None,
118
+ deadline_s: float | None = None,
119
+ ) -> dict[str, Any]:
120
+ """Decision risk = f(sensitivity, reversibility), deadline-capped."""
121
+
122
+ sensitivity = _classify_sensitivity(query, declared)
123
+ sensitivity_normalized = sensitivity["rank"] / 3.0
124
+ reversibility_score, reversibility_label = _reversibility(query)
125
+ risk = round(0.65 * sensitivity_normalized + 0.35 * reversibility_score, 4)
126
+
127
  if risk >= 0.6:
128
+ wanted = "STRATEGIC"
129
  elif risk >= 0.3:
130
+ wanted = "OPERATIONAL"
131
  else:
132
+ wanted = "TACTICAL"
133
+
134
+ chosen = wanted
135
  capped_by_deadline = False
136
  if deadline_s is not None:
137
+ affordable = [
138
+ tier["tier"] for tier in TIERS if tier["deadline_s"] <= deadline_s
139
+ ]
140
  order = ["TACTICAL", "OPERATIONAL", "STRATEGIC"]
141
  if affordable:
142
+ best_affordable = max(affordable, key=order.index)
143
+ if order.index(best_affordable) < order.index(wanted):
144
  chosen = best_affordable
145
  capped_by_deadline = True
146
  else:
147
+ chosen = "TACTICAL"
148
+ capped_by_deadline = True
149
+
150
  return {
151
+ "sensitivity": sensitivity,
152
+ "reversibility": {
153
+ "score": reversibility_score,
154
+ "label": reversibility_label,
155
+ },
156
+ "risk": risk,
157
+ "risk_band": (
158
+ "HIGH" if risk >= 0.6 else "MODERATE" if risk >= 0.3 else "LOW"
159
+ ),
160
+ "risk_implied_tier": wanted,
161
+ "chosen_tier": chosen,
162
  "capped_by_deadline": capped_by_deadline,
163
  }
164
 
165
 
166
  def _pick_model(max_model_tier: str, sensitivity_rank: int) -> dict[str, Any] | None:
167
+ """Choose the smallest sufficient model inside the declared ceiling."""
168
+
169
  try:
170
+ import szl_governance_gateway as governance_gateway # type: ignore
171
+
172
+ catalog = getattr(governance_gateway, "CATALOG", [])
173
+ order = getattr(governance_gateway, "_TIER_ORDER", {})
174
+ airgap_floor = getattr(governance_gateway, "_AIRGAP_FLOOR", 2)
175
  except Exception:
176
  return None
177
+
178
+ ceiling = order.get(max_model_tier, 9)
179
+ airgap_required = sensitivity_rank >= airgap_floor
180
+ eligible = [
181
+ model
182
+ for model in catalog
183
+ if order.get(model["tier"], 9) <= ceiling
184
+ and (not airgap_required or model.get("zone") == "AIRGAP")
185
+ ]
186
+ if not eligible:
187
  return None
188
+ eligible.sort(key=lambda model: (order.get(model["tier"], 9), -model.get("ctx", 0)))
189
+ return eligible[0]
190
 
191
 
192
+ def route(
193
+ query: str,
194
+ *,
195
+ declared: str | None = None,
196
+ deadline_s: float | None = None,
197
+ ) -> dict[str, Any]:
198
  risk = assess_risk(query, declared=declared, deadline_s=deadline_s)
199
  tier = _TIER_BY_NAME[risk["chosen_tier"]]
200
  model = _pick_model(tier["max_model_tier"], risk["sensitivity"]["rank"])
201
  decision = {
202
+ "query": query,
203
+ "tier": tier,
204
+ "risk": risk,
205
+ "chosen_model": model,
206
  "airgap_required": risk["sensitivity"]["rank"] >= 2,
207
+ "policy": (
208
+ "risk(sensitivity×reversibility) → mission tier (deadline-capped) "
209
+ "→ smallest sufficient air-gap-safe model"
210
+ ),
211
+ "honest": (
212
+ None
213
+ if model
214
+ else "no model within tier ceiling + air-gap floor — raise the tier or add an AIRGAP model"
215
+ ),
216
  }
217
  _learn_skeleton(decision)
218
  return decision
219
 
220
 
 
 
 
 
 
 
 
221
  _SKELETONS: dict[str, dict[str, Any]] = {}
222
 
223
 
 
225
  band = decision["risk"]["risk_band"]
226
  tier = decision["tier"]["tier"]
227
  airgap = decision["airgap_required"]
228
+ model_family = ((decision.get("chosen_model") or {}).get("id") or "—").split("-")[0]
229
+ key = f"{band}|{tier}|{'airgap' if airgap else 'cloud'}|{model_family}"
230
+ skeleton = _SKELETONS.get(key)
231
+ if skeleton is None:
232
+ skeleton = {
233
+ "key": key,
234
+ "risk_band": band,
235
+ "tier": tier,
236
+ "airgap": airgap,
237
+ "model_family": model_family,
238
+ "uses": 0,
239
+ "first_seen": time.time(),
240
+ "last_seen": time.time(),
241
  "lesson": _lesson(band, tier, airgap),
242
  }
243
+ _SKELETONS[key] = skeleton
244
+ skeleton["uses"] += 1
245
+ skeleton["last_seen"] = time.time()
 
246
  if len(_SKELETONS) > 64:
247
+ least_valuable = min(
248
+ _SKELETONS.values(),
249
+ key=lambda item: (item["uses"], item["last_seen"]),
250
+ )
251
+ _SKELETONS.pop(least_valuable["key"], None)
252
 
253
 
254
  def _lesson(band: str, tier: str, airgap: bool) -> str:
255
+ zone = "air-gapped (classified)" if airgap else "cloud-eligible"
256
+ return (
257
+ f"{band}-risk decisions routed to the {tier} budget tier on {zone} compute "
258
+ "resolved within budget; reuse this skeleton for the same risk class."
259
+ )
260
 
261
 
262
  def skeletons() -> dict[str, Any]:
263
+ rows = sorted(_SKELETONS.values(), key=lambda item: item["uses"], reverse=True)
264
+ return {
265
+ "skeleton_count": len(rows),
266
+ "skeletons": rows,
267
+ "hint": (
268
+ None
269
+ if rows
270
+ else "no skeletons yet — route decisions via POST /api/a11oy/v1/budget/route to learn them"
271
+ ),
272
+ }
273
+
274
+
275
+ # ---------------------------------------------------------------------------
276
+ # Semantic token ingress
277
+ # ---------------------------------------------------------------------------
278
+
279
+ MAX_PREFIX_ENTRIES = 256
280
+ MAX_PREFIX_BYTES = 16 * 1024 * 1024
281
+ MAX_INGEST_FILES = 4096
282
+ MAX_INGEST_FILE_BYTES = 8 * 1024 * 1024
283
+ MAX_INGEST_TOTAL_BYTES = 128 * 1024 * 1024
284
+ MAX_TOKEN_BODY = 64 * 1024
285
+ MAX_TOKEN_NODES = 64
286
+ MAX_TOKEN_CASES = 256
287
+ MAX_TOKEN_IDS_PER_CASE = 8192
288
+ MAX_TOKEN_TEXT_CHARS_PER_CASE = 256 * 1024
289
+
290
+ _SEMANTIC_DIGEST_FIELDS = (
291
+ "vocabulary_sha256",
292
+ "normalization_sha256",
293
+ "special_tokens_sha256",
294
+ "added_tokens_sha256",
295
+ "chat_template_sha256",
296
+ "document_separator_sha256",
297
+ )
298
+
299
+
300
+ def _is_sha256(value: str) -> bool:
301
+ return len(value) == 64 and all(character in "0123456789abcdef" for character in value)
302
+
303
+
304
+ def _canonical_sha256(value: object) -> str:
305
+ payload = json.dumps(
306
+ value,
307
+ ensure_ascii=False,
308
+ sort_keys=True,
309
+ separators=(",", ":"),
310
+ allow_nan=False,
311
+ ).encode("utf-8")
312
+ return hashlib.sha256(payload).hexdigest()
313
+
314
+
315
+ @dataclass(frozen=True)
316
+ class TokenizerNodeSignal:
317
+ node_id: str
318
+ tokenizer_tokens_per_sec: float
319
+ tokenizer_cache_warmth: float
320
+ prefix_cache_hit_rate: float
321
+ kv_cache_hit_rate: float
322
+ available: bool = True
323
+ measured: bool = False
324
+
325
+ def validate(self) -> None:
326
+ if not self.node_id.strip():
327
+ raise ValueError("node_id is required")
328
+ if not math.isfinite(self.tokenizer_tokens_per_sec) or self.tokenizer_tokens_per_sec < 0:
329
+ raise ValueError("tokenizer_tokens_per_sec must be finite and non-negative")
330
+ for name, value in (
331
+ ("tokenizer_cache_warmth", self.tokenizer_cache_warmth),
332
+ ("prefix_cache_hit_rate", self.prefix_cache_hit_rate),
333
+ ("kv_cache_hit_rate", self.kv_cache_hit_rate),
334
+ ):
335
+ if not math.isfinite(value) or not 0.0 <= value <= 1.0:
336
+ raise ValueError(f"{name} must be finite and between 0 and 1")
337
+
338
+
339
+ @dataclass(frozen=True)
340
+ class IngressWorkload:
341
+ prefix_heavy: bool = False
342
+ corpus_heavy: bool = False
343
+ prefill_heavy: bool = False
344
+
345
+ @property
346
+ def ingress_weight(self) -> float:
347
+ flags = sum((self.prefix_heavy, self.corpus_heavy, self.prefill_heavy))
348
+ return min(1.0, 0.25 + 0.25 * flags)
349
+
350
+
351
+ def choose_ingress_node(
352
+ nodes: Sequence[TokenizerNodeSignal],
353
+ workload: IngressWorkload,
354
+ ) -> dict[str, object]:
355
+ eligible = [node for node in nodes if node.available]
356
+ if not eligible:
357
+ return {
358
+ "status": "BLOCKED",
359
+ "reason": "no available ingress nodes",
360
+ "node": None,
361
+ }
362
+ for node in eligible:
363
+ node.validate()
364
+
365
+ maximum_throughput = max(node.tokenizer_tokens_per_sec for node in eligible) or 1.0
366
+ ingress_weight = workload.ingress_weight
367
+
368
+ def score(node: TokenizerNodeSignal) -> float:
369
+ throughput = node.tokenizer_tokens_per_sec / maximum_throughput
370
+ cache_locality = (
371
+ 0.45 * node.tokenizer_cache_warmth
372
+ + 0.35 * node.prefix_cache_hit_rate
373
+ + 0.20 * node.kv_cache_hit_rate
374
+ )
375
+ return round(
376
+ (1.0 - ingress_weight) * throughput + ingress_weight * cache_locality,
377
+ 6,
378
+ )
379
+
380
+ ranked = sorted(eligible, key=lambda node: (-score(node), node.node_id))
381
+ winner = ranked[0]
382
+ return {
383
+ "status": "PASS",
384
+ "node": winner.node_id,
385
+ "score": score(winner),
386
+ "evidence": "MEASURED" if winner.measured else "SAMPLE",
387
+ "policy": "tokenizer-throughput + cache-warmth + prefix/KV reuse",
388
+ "ranking": [{"node": node.node_id, "score": score(node)} for node in ranked],
389
+ }
390
+
391
+
392
+ @dataclass(frozen=True)
393
+ class SemanticTokenContract:
394
+ source: str
395
+ tokenizer_family: str
396
+ vocabulary_sha256: str
397
+ normalization_sha256: str
398
+ special_tokens_sha256: str
399
+ added_tokens_sha256: str
400
+ chat_template_sha256: str
401
+ document_separator_sha256: str
402
+
403
+ def validate(self) -> None:
404
+ if not self.source.strip():
405
+ raise ValueError("semantic token contract source is required")
406
+ if not self.tokenizer_family.strip():
407
+ raise ValueError("tokenizer_family is required")
408
+ for field_name in _SEMANTIC_DIGEST_FIELDS:
409
+ if not _is_sha256(getattr(self, field_name)):
410
+ raise ValueError(f"{field_name} must be one lowercase SHA-256 digest")
411
+
412
+ def semantic_fields(self) -> dict[str, str]:
413
+ self.validate()
414
+ return {
415
+ "tokenizer_family": self.tokenizer_family,
416
+ **{name: getattr(self, name) for name in _SEMANTIC_DIGEST_FIELDS},
417
+ }
418
+
419
+ def digest(self) -> str:
420
+ return _canonical_sha256(self.semantic_fields())
421
+
422
+ def mismatches(self, other: "SemanticTokenContract") -> list[str]:
423
+ left = self.semantic_fields()
424
+ right = other.semantic_fields()
425
+ return sorted(name for name in left if left[name] != right[name])
426
+
427
+
428
+ @dataclass(frozen=True)
429
+ class TokenizerParityCase:
430
+ name: str
431
+ oracle_ids: tuple[int, ...]
432
+ candidate_ids: tuple[int, ...]
433
+ oracle_decoded_text: str
434
+ candidate_decoded_text: str
435
+
436
+ def exact_match(self) -> bool:
437
+ return (
438
+ self.oracle_ids == self.candidate_ids
439
+ and self.oracle_decoded_text == self.candidate_decoded_text
440
+ )
441
+
442
+
443
+ def qualify_tokenizer_candidate(
444
+ oracle: SemanticTokenContract,
445
+ candidate: SemanticTokenContract,
446
+ cases: Sequence[TokenizerParityCase],
447
+ ) -> dict[str, object]:
448
+ oracle.validate()
449
+ candidate.validate()
450
+ if not cases:
451
+ return {
452
+ "status": "BLOCKED",
453
+ "eligible": False,
454
+ "reason": "no representative semantic-parity cases supplied",
455
+ "oracle_source": oracle.source,
456
+ "candidate_source": candidate.source,
457
+ }
458
+
459
+ contract_mismatches = oracle.mismatches(candidate)
460
+ case_mismatches = [case.name for case in cases if not case.exact_match()]
461
+ eligible = not contract_mismatches and not case_mismatches
462
+ return {
463
+ "status": "PASS" if eligible else "FAIL",
464
+ "eligible": eligible,
465
+ "oracle_source": oracle.source,
466
+ "candidate_source": candidate.source,
467
+ "oracle_contract_sha256": oracle.digest(),
468
+ "candidate_contract_sha256": candidate.digest(),
469
+ "contract_mismatches": contract_mismatches,
470
+ "case_mismatches": case_mismatches,
471
+ "cases": len(cases),
472
+ "policy": (
473
+ "exact vocabulary/normalization/special-token/added-token/chat-template/"
474
+ "document-separator digests + token IDs + decoded text"
475
+ ),
476
+ }
477
+
478
+
479
+ @dataclass
480
+ class PrefixFoundry:
481
+ max_entries: int = MAX_PREFIX_ENTRIES
482
+ max_bytes: int = MAX_PREFIX_BYTES
483
+ _entries: dict[str, bytes] = field(default_factory=dict)
484
+ _bytes: int = 0
485
+
486
+ @staticmethod
487
+ def digest(namespace: str, semantic_contract_sha256: str, content: bytes) -> str:
488
+ if not namespace.strip():
489
+ raise ValueError("namespace is required")
490
+ if not _is_sha256(semantic_contract_sha256):
491
+ raise ValueError("semantic_contract_sha256 must be one lowercase SHA-256 digest")
492
+ digest = hashlib.sha256()
493
+ digest.update(namespace.encode("utf-8"))
494
+ digest.update(b"\0")
495
+ digest.update(semantic_contract_sha256.encode("ascii"))
496
+ digest.update(b"\0")
497
+ digest.update(content)
498
+ return digest.hexdigest()
499
+
500
+ def put(self, namespace: str, semantic_contract_sha256: str, content: bytes) -> str:
501
+ if self.max_entries < 1 or self.max_bytes < 1:
502
+ raise ValueError("foundry budgets must be positive")
503
+ if not content:
504
+ raise ValueError("prefix content must not be empty")
505
+ if len(content) > self.max_bytes:
506
+ raise ValueError("prefix exceeds foundry byte budget")
507
+ key = self.digest(namespace, semantic_contract_sha256, content)
508
+ if key in self._entries:
509
+ return key
510
+ while self._entries and (
511
+ len(self._entries) >= self.max_entries
512
+ or self._bytes + len(content) > self.max_bytes
513
+ ):
514
+ oldest_key = next(iter(self._entries))
515
+ old = self._entries.pop(oldest_key)
516
+ self._bytes -= len(old)
517
+ self._entries[key] = bytes(content)
518
+ self._bytes += len(content)
519
+ return key
520
+
521
+ def get(self, key: str) -> bytes | None:
522
+ return self._entries.get(key)
523
+
524
+ def snapshot(self) -> dict[str, int]:
525
+ return {"entries": len(self._entries), "bytes": self._bytes}
526
+
527
+
528
+ @dataclass(frozen=True)
529
+ class IngestedFile:
530
+ path: str
531
+ sha256: str
532
+ size_bytes: int
533
+ text: bool
534
+
535
+
536
+ def _is_probably_binary(data: bytes) -> bool:
537
+ return bool(data) and b"\0" in data[:4096]
538
+
539
+
540
+ def _path_contains_symlink(root: Path, relative_path: Path) -> bool:
541
+ cursor = root
542
+ for part in relative_path.parts:
543
+ cursor = cursor / part
544
+ if cursor.is_symlink():
545
+ return True
546
+ return False
547
+
548
+
549
+ def ingest_repository_files(
550
+ root: Path,
551
+ relative_paths: Iterable[str],
552
+ *,
553
+ max_files: int = MAX_INGEST_FILES,
554
+ max_file_bytes: int = MAX_INGEST_FILE_BYTES,
555
+ max_total_bytes: int = MAX_INGEST_TOTAL_BYTES,
556
+ ) -> dict[str, object]:
557
+ root = root.resolve()
558
+ if not root.is_dir():
559
+ raise ValueError("root must be an existing directory")
560
+ if max_files < 1 or max_file_bytes < 1 or max_total_bytes < 1:
561
+ raise ValueError("ingest budgets must be positive")
562
+
563
+ normalized = sorted(set(relative_paths))
564
+ if len(normalized) > max_files:
565
+ return {
566
+ "status": "BLOCKED",
567
+ "reason": "file-count-budget",
568
+ "files": [],
569
+ "text_payloads": {},
570
+ "skipped": [],
571
+ "total_bytes": 0,
572
+ }
573
+
574
+ total = 0
575
+ manifest: list[IngestedFile] = []
576
+ text_payloads: dict[str, str] = {}
577
+ skipped: list[dict[str, str]] = []
578
+
579
+ for raw_path in normalized:
580
+ relative = Path(raw_path)
581
+ if relative.is_absolute() or ".." in relative.parts:
582
+ raise ValueError(f"path escapes repository root: {raw_path}")
583
+ if _path_contains_symlink(root, relative):
584
+ skipped.append({"path": relative.as_posix(), "reason": "symlink"})
585
+ continue
586
+
587
+ target = (root / relative).resolve()
588
+ try:
589
+ target.relative_to(root)
590
+ except ValueError as exc:
591
+ raise ValueError(f"path escapes repository root: {raw_path}") from exc
592
+ if not target.is_file():
593
+ skipped.append({"path": relative.as_posix(), "reason": "not-a-file"})
594
+ continue
595
+
596
+ stat_size = target.stat().st_size
597
+ if stat_size > max_file_bytes:
598
+ skipped.append({"path": relative.as_posix(), "reason": "file-budget"})
599
+ continue
600
+ if total + stat_size > max_total_bytes:
601
+ return {
602
+ "status": "BLOCKED",
603
+ "reason": "total-ingest-byte-budget",
604
+ "files": [item.__dict__ for item in manifest],
605
+ "text_payloads": text_payloads,
606
+ "skipped": skipped,
607
+ "total_bytes": total,
608
+ }
609
+
610
+ data = target.read_bytes()
611
+ if len(data) > max_file_bytes or total + len(data) > max_total_bytes:
612
+ return {
613
+ "status": "BLOCKED",
614
+ "reason": "post-read-byte-budget",
615
+ "files": [item.__dict__ for item in manifest],
616
+ "text_payloads": text_payloads,
617
+ "skipped": skipped,
618
+ "total_bytes": total,
619
+ }
620
+
621
+ total += len(data)
622
+ is_text = not _is_probably_binary(data)
623
+ manifest.append(
624
+ IngestedFile(
625
+ path=relative.as_posix(),
626
+ sha256=hashlib.sha256(data).hexdigest(),
627
+ size_bytes=len(data),
628
+ text=is_text,
629
+ )
630
+ )
631
+ if is_text:
632
+ text_payloads[relative.as_posix()] = data.decode("utf-8", errors="replace")
633
+ else:
634
+ skipped.append({"path": relative.as_posix(), "reason": "binary"})
635
+
636
+ rows = [item.__dict__ for item in manifest]
637
+ return {
638
+ "status": "PASS",
639
+ "files": rows,
640
+ "text_payloads": text_payloads,
641
+ "skipped": skipped,
642
+ "total_bytes": total,
643
+ "batch_sha256": _canonical_sha256(rows),
644
+ }
645
+
646
+
647
+ def verifier_reinvestment(
648
+ saved_milliseconds: float,
649
+ *,
650
+ measured: bool = False,
651
+ weights: Mapping[str, float] | None = None,
652
+ ) -> dict[str, object]:
653
+ if not math.isfinite(saved_milliseconds) or saved_milliseconds < 0:
654
+ raise ValueError("saved_milliseconds must be finite and non-negative")
655
+ allocation = dict(
656
+ weights
657
+ or {
658
+ "branch_scoring": 0.30,
659
+ "static_analysis": 0.25,
660
+ "policy_checks": 0.20,
661
+ "replay": 0.15,
662
+ "counterexamples": 0.10,
663
+ }
664
+ )
665
+ if not allocation or any(
666
+ not math.isfinite(value) or value < 0 for value in allocation.values()
667
+ ):
668
+ raise ValueError("verification weights must be finite and non-negative")
669
+ total = sum(allocation.values())
670
+ if total <= 0:
671
+ raise ValueError("verification weights must have positive total")
672
+ budget = {
673
+ name: round(saved_milliseconds * value / total, 3)
674
+ for name, value in allocation.items()
675
+ }
676
+ return {
677
+ "evidence": "MEASURED" if measured else "MODELED",
678
+ "saved_milliseconds": saved_milliseconds,
679
+ "verification_budget_ms": budget,
680
+ "policy": "reinvest ingress savings into verification before expanding interactive traffic",
681
+ }
682
+
683
+
684
+ _TOKEN_FOUNDRY = PrefixFoundry()
685
+
686
+
687
+ def _token_error(status: int, code: str, message: str) -> JSONResponse:
688
+ return JSONResponse(
689
+ {
690
+ "ready": status < 500,
691
+ "accepted": False,
692
+ "status": "BLOCKED" if status < 500 else "UNAVAILABLE",
693
+ "error": {"code": code, "message": message[:240]},
694
+ "effectors": 0,
695
+ },
696
+ status_code=status,
697
+ )
698
+
699
+
700
+ def _closed_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
701
+ result: dict[str, Any] = {}
702
+ for key, value in pairs:
703
+ if key in result:
704
+ raise ValueError(f"duplicate JSON field: {key}")
705
+ result[key] = value
706
+ return result
707
+
708
+
709
+ def _reject_json_constant(value: str) -> None:
710
+ raise ValueError(f"non-finite JSON number is forbidden: {value}")
711
+
712
+
713
+ async def _token_body(request: Request) -> dict[str, Any]:
714
+ content_type = request.headers.get("content-type", "").split(";", 1)[0].strip().lower()
715
+ if content_type != "application/json":
716
+ raise ValueError("content-type must be application/json")
717
+
718
+ declared = request.headers.get("content-length")
719
+ if declared is not None:
720
+ try:
721
+ size = int(declared)
722
+ except ValueError as exc:
723
+ raise ValueError("content-length must be an integer") from exc
724
+ if size < 0 or size > MAX_TOKEN_BODY:
725
+ raise ValueError("request body exceeds 64 KiB")
726
+
727
+ raw = bytearray()
728
+ async for chunk in request.stream():
729
+ if len(raw) + len(chunk) > MAX_TOKEN_BODY:
730
+ raise ValueError("request body exceeds 64 KiB")
731
+ raw.extend(chunk)
732
+
733
+ try:
734
+ value = json.loads(
735
+ bytes(raw).decode("utf-8"),
736
+ object_pairs_hook=_closed_object,
737
+ parse_constant=_reject_json_constant,
738
+ )
739
+ except (UnicodeDecodeError, json.JSONDecodeError, ValueError) as exc:
740
+ raise ValueError("request body must be strict JSON with unique fields") from exc
741
+ if not isinstance(value, dict):
742
+ raise ValueError("request body must be one JSON object")
743
+ return value
744
+
745
+
746
+ def _closed_fields(value: dict[str, Any], allowed: set[str], field: str) -> None:
747
+ extras = sorted(set(value) - allowed)
748
+ if extras:
749
+ raise ValueError(f"{field} contains unsupported fields: {','.join(extras)}")
750
+
751
+
752
+ def _number(value: Any, field_name: str) -> float:
753
+ if isinstance(value, bool) or not isinstance(value, (int, float)):
754
+ raise ValueError(f"{field_name} must be a number")
755
+ number = float(value)
756
+ if not math.isfinite(number):
757
+ raise ValueError(f"{field_name} must be finite")
758
+ return number
759
+
760
+
761
+ def _strict_bool(value: Any, field_name: str, *, default: bool) -> bool:
762
+ if value is None:
763
+ return default
764
+ if not isinstance(value, bool):
765
+ raise ValueError(f"{field_name} must be a boolean")
766
+ return value
767
+
768
+
769
+ def _token_ids(value: Any, field_name: str) -> tuple[int, ...]:
770
+ if not isinstance(value, list) or len(value) > MAX_TOKEN_IDS_PER_CASE:
771
+ raise ValueError(
772
+ f"{field_name} must be an array with at most {MAX_TOKEN_IDS_PER_CASE} entries"
773
+ )
774
+ if any(
775
+ isinstance(item, bool) or not isinstance(item, int) or item < 0
776
+ for item in value
777
+ ):
778
+ raise ValueError(f"{field_name} must contain non-negative integer token IDs")
779
+ return tuple(value)
780
+
781
+
782
+ def _bounded_text(value: Any, field_name: str) -> str:
783
+ if not isinstance(value, str):
784
+ raise ValueError(f"{field_name} must be a string")
785
+ if len(value) > MAX_TOKEN_TEXT_CHARS_PER_CASE:
786
+ raise ValueError(f"{field_name} exceeds the text boundary")
787
+ return value
788
+
789
+
790
+ def _semantic_contract(value: Any, field_name: str) -> SemanticTokenContract:
791
+ if not isinstance(value, dict):
792
+ raise ValueError(f"{field_name} must be an object")
793
+ allowed = {"source", "tokenizer_family", *_SEMANTIC_DIGEST_FIELDS}
794
+ _closed_fields(value, allowed, field_name)
795
+ source = value.get("source")
796
+ family = value.get("tokenizer_family")
797
+ if not isinstance(source, str) or not isinstance(family, str):
798
+ raise ValueError(
799
+ f"{field_name}.source and {field_name}.tokenizer_family must be strings"
800
+ )
801
+ digests: dict[str, str] = {}
802
+ for digest_field in _SEMANTIC_DIGEST_FIELDS:
803
+ digest = value.get(digest_field)
804
+ if not isinstance(digest, str):
805
+ raise ValueError(f"{field_name}.{digest_field} must be a string")
806
+ digests[digest_field] = digest
807
+ contract = SemanticTokenContract(
808
+ source=source,
809
+ tokenizer_family=family,
810
+ **digests,
811
+ )
812
+ contract.validate()
813
+ return contract
814
+
815
+
816
+ def _register_token_ingress(app: FastAPI, ns: str) -> dict[str, Any]:
817
+ prefix = f"/api/{ns}/v1/token-ingress"
818
+ if any(
819
+ getattr(existing, "path", None) == f"{prefix}/status"
820
+ for existing in app.router.routes
821
+ ):
822
+ return {"ok": True, "state": "ALREADY_REGISTERED", "routes": []}
823
+
824
+ @app.get(f"{prefix}/status", include_in_schema=False)
825
+ async def token_ingress_status() -> JSONResponse:
826
+ return JSONResponse(
827
+ {
828
+ "ready": True,
829
+ "implementation": "REAL",
830
+ "execution": "BOUNDED_COMPUTATION_ONLY",
831
+ "telemetry": "CALLER_SAMPLE_ONLY",
832
+ "tokenizer_promotion": "FAIL_CLOSED_SEMANTIC_CONTRACT_REQUIRED",
833
+ "semantic_contract_fields": [
834
+ "tokenizer_family",
835
+ *_SEMANTIC_DIGEST_FIELDS,
836
+ ],
837
+ "prefix_foundry": _TOKEN_FOUNDRY.snapshot(),
838
+ "repository_ingestion": "INTERNAL_LIBRARY_ONLY",
839
+ "effectors": 0,
840
+ "provider_calls": 0,
841
+ "network_calls": 0,
842
+ }
843
+ )
844
+
845
+ @app.post(f"{prefix}/route", include_in_schema=False)
846
+ async def token_ingress_route(request: Request) -> JSONResponse:
847
+ try:
848
+ payload = await _token_body(request)
849
+ _closed_fields(payload, {"nodes", "workload"}, "request")
850
+ raw_nodes = payload.get("nodes")
851
+ if not isinstance(raw_nodes, list) or not 1 <= len(raw_nodes) <= MAX_TOKEN_NODES:
852
+ raise ValueError(f"nodes must contain 1..{MAX_TOKEN_NODES} entries")
853
+
854
+ nodes: list[TokenizerNodeSignal] = []
855
+ node_fields = {
856
+ "node_id",
857
+ "tokenizer_tokens_per_sec",
858
+ "tokenizer_cache_warmth",
859
+ "prefix_cache_hit_rate",
860
+ "kv_cache_hit_rate",
861
+ "available",
862
+ "measured",
863
+ }
864
+ for index, item in enumerate(raw_nodes):
865
+ if not isinstance(item, dict):
866
+ raise ValueError("every node must be an object")
867
+ _closed_fields(item, node_fields, f"node[{index}]")
868
+ node_id = item.get("node_id")
869
+ if not isinstance(node_id, str):
870
+ raise ValueError("node_id must be a string")
871
+ nodes.append(
872
+ TokenizerNodeSignal(
873
+ node_id=node_id,
874
+ tokenizer_tokens_per_sec=_number(
875
+ item.get("tokenizer_tokens_per_sec", 0),
876
+ "tokenizer_tokens_per_sec",
877
+ ),
878
+ tokenizer_cache_warmth=_number(
879
+ item.get("tokenizer_cache_warmth", 0),
880
+ "tokenizer_cache_warmth",
881
+ ),
882
+ prefix_cache_hit_rate=_number(
883
+ item.get("prefix_cache_hit_rate", 0),
884
+ "prefix_cache_hit_rate",
885
+ ),
886
+ kv_cache_hit_rate=_number(
887
+ item.get("kv_cache_hit_rate", 0),
888
+ "kv_cache_hit_rate",
889
+ ),
890
+ available=_strict_bool(
891
+ item.get("available"),
892
+ "available",
893
+ default=True,
894
+ ),
895
+ measured=False,
896
+ )
897
+ )
898
+
899
+ raw_workload = payload.get("workload") or {}
900
+ if not isinstance(raw_workload, dict):
901
+ raise ValueError("workload must be an object")
902
+ _closed_fields(
903
+ raw_workload,
904
+ {"prefix_heavy", "corpus_heavy", "prefill_heavy"},
905
+ "workload",
906
+ )
907
+ workload = IngressWorkload(
908
+ prefix_heavy=_strict_bool(
909
+ raw_workload.get("prefix_heavy"),
910
+ "prefix_heavy",
911
+ default=False,
912
+ ),
913
+ corpus_heavy=_strict_bool(
914
+ raw_workload.get("corpus_heavy"),
915
+ "corpus_heavy",
916
+ default=False,
917
+ ),
918
+ prefill_heavy=_strict_bool(
919
+ raw_workload.get("prefill_heavy"),
920
+ "prefill_heavy",
921
+ default=False,
922
+ ),
923
+ )
924
+ result = choose_ingress_node(nodes, workload)
925
+ result["evidence"] = "SAMPLE"
926
+ result["telemetry_authority"] = "CALLER_SUPPLIED_NOT_MEASURED"
927
+ accepted = result["status"] == "PASS"
928
+ return JSONResponse(
929
+ {"ready": True, "accepted": accepted, **result},
930
+ status_code=200 if accepted else 409,
931
+ )
932
+ except (TypeError, ValueError) as exc:
933
+ return _token_error(422, "invalid_ingress_route", str(exc))
934
+
935
+ @app.post(f"{prefix}/qualify", include_in_schema=False)
936
+ async def token_ingress_qualify(request: Request) -> JSONResponse:
937
+ try:
938
+ payload = await _token_body(request)
939
+ _closed_fields(
940
+ payload,
941
+ {"oracle_contract", "candidate_contract", "cases"},
942
+ "request",
943
+ )
944
+ oracle = _semantic_contract(
945
+ payload.get("oracle_contract"),
946
+ "oracle_contract",
947
+ )
948
+ candidate = _semantic_contract(
949
+ payload.get("candidate_contract"),
950
+ "candidate_contract",
951
+ )
952
+ raw_cases = payload.get("cases")
953
+ if not isinstance(raw_cases, list) or len(raw_cases) > MAX_TOKEN_CASES:
954
+ raise ValueError(
955
+ f"cases must be a list with at most {MAX_TOKEN_CASES} entries"
956
+ )
957
+
958
+ cases: list[TokenizerParityCase] = []
959
+ case_fields = {
960
+ "name",
961
+ "oracle_ids",
962
+ "candidate_ids",
963
+ "oracle_decoded_text",
964
+ "candidate_decoded_text",
965
+ }
966
+ for index, item in enumerate(raw_cases):
967
+ if not isinstance(item, dict):
968
+ raise ValueError("every parity case must be an object")
969
+ _closed_fields(item, case_fields, f"case[{index}]")
970
+ name = item.get("name")
971
+ if not isinstance(name, str) or not name:
972
+ raise ValueError("case name must be a non-empty string")
973
+ cases.append(
974
+ TokenizerParityCase(
975
+ name=name,
976
+ oracle_ids=_token_ids(item.get("oracle_ids"), "oracle_ids"),
977
+ candidate_ids=_token_ids(
978
+ item.get("candidate_ids"),
979
+ "candidate_ids",
980
+ ),
981
+ oracle_decoded_text=_bounded_text(
982
+ item.get("oracle_decoded_text"),
983
+ "oracle_decoded_text",
984
+ ),
985
+ candidate_decoded_text=_bounded_text(
986
+ item.get("candidate_decoded_text"),
987
+ "candidate_decoded_text",
988
+ ),
989
+ )
990
+ )
991
+
992
+ result = qualify_tokenizer_candidate(oracle, candidate, cases)
993
+ status_code = (
994
+ 200
995
+ if result["status"] == "PASS"
996
+ else 409
997
+ if result["status"] == "FAIL"
998
+ else 422
999
+ )
1000
+ return JSONResponse(
1001
+ {
1002
+ "ready": True,
1003
+ "accepted": result["status"] == "PASS",
1004
+ **result,
1005
+ "effectors": 0,
1006
+ },
1007
+ status_code=status_code,
1008
+ )
1009
+ except (TypeError, ValueError) as exc:
1010
+ return _token_error(422, "invalid_tokenizer_qualification", str(exc))
1011
+
1012
+ @app.post(f"{prefix}/verification-budget", include_in_schema=False)
1013
+ async def token_ingress_verification_budget(request: Request) -> JSONResponse:
1014
+ try:
1015
+ payload = await _token_body(request)
1016
+ _closed_fields(
1017
+ payload,
1018
+ {"saved_milliseconds", "measured"},
1019
+ "request",
1020
+ )
1021
+ saved = _number(
1022
+ payload.get("saved_milliseconds", 0),
1023
+ "saved_milliseconds",
1024
+ )
1025
+ result = verifier_reinvestment(saved, measured=False)
1026
+ result["evidence"] = "MODELED"
1027
+ result["measurement_authority"] = "NOT_ACCEPTED_FROM_PUBLIC_CALLER"
1028
+ return JSONResponse(
1029
+ {"ready": True, "accepted": True, **result, "effectors": 0}
1030
+ )
1031
+ except (TypeError, ValueError) as exc:
1032
+ return _token_error(422, "invalid_verification_budget", str(exc))
1033
+
1034
+ return {
1035
+ "ok": True,
1036
+ "state": "REGISTERED",
1037
+ "routes": [
1038
+ f"{prefix}/status",
1039
+ f"{prefix}/route",
1040
+ f"{prefix}/qualify",
1041
+ f"{prefix}/verification-budget",
1042
+ ],
1043
+ "effectors": 0,
1044
+ }
1045
+
1046
+
1047
+ # ---------------------------------------------------------------------------
1048
+ # Route registration
1049
+ # ---------------------------------------------------------------------------
1050
 
1051
 
1052
  def register(app: FastAPI, ns: str = "a11oy") -> str:
1053
  @app.get(f"/api/{ns}/v1/budget/tiers", include_in_schema=False)
1054
+ async def budget_tiers() -> JSONResponse:
1055
+ return JSONResponse(
1056
+ {
1057
+ "doctrine": DOCTRINE,
1058
+ "tiers": TIERS,
1059
+ "pattern_source": (
1060
+ "ViktorAxelsen/BudgetMem (Apache-2.0) — module budget tiers, "
1061
+ "evolved to mission constraints"
1062
+ ),
1063
+ }
1064
+ )
1065
 
1066
  @app.post(f"/api/{ns}/v1/budget/route", include_in_schema=False)
1067
+ async def budget_route(request: Request) -> JSONResponse:
1068
  try:
1069
  body = await request.json()
1070
  except Exception:
1071
  body = {}
1072
+ query = (body or {}).get("query") or (body or {}).get("q") or ""
1073
  declared = (body or {}).get("classification")
1074
+ deadline = (body or {}).get("deadline_s")
1075
  try:
1076
+ deadline = float(deadline) if deadline is not None else None
1077
  except Exception:
1078
+ deadline = None
1079
+ if not query:
1080
  return JSONResponse({"error": "provide {query: ...}"}, status_code=400)
1081
+ return JSONResponse(
1082
+ {
1083
+ "doctrine": DOCTRINE,
1084
+ **route(query, declared=declared, deadline_s=deadline),
1085
+ }
1086
+ )
1087
 
1088
  @app.get(f"/api/{ns}/v1/budget/skeletons", include_in_schema=False)
1089
+ async def budget_skeletons() -> JSONResponse:
1090
+ return JSONResponse(
1091
+ {
1092
+ "doctrine": DOCTRINE,
1093
+ **skeletons(),
1094
+ "pattern_source": (
1095
+ "ViktorAxelsen/MemSkill (Apache-2.0) — meta-memory, "
1096
+ "evolved to Decision Skeletons"
1097
+ ),
1098
+ }
1099
+ )
1100
+
1101
+ token_status = _register_token_ingress(app, ns)
1102
 
1103
  @app.get("/budget-router", include_in_schema=False)
1104
+ async def budget_router_page() -> HTMLResponse:
1105
  return HTMLResponse(_PAGE_HTML)
1106
 
1107
+ return (
1108
+ f"budget-router mounted: GET /budget-router + /api/{ns}/v1/budget/"
1109
+ f"(tiers|route|skeletons); token-ingress={token_status['state']}"
1110
+ )
1111
 
1112
 
1113
  _PAGE_HTML = """<!DOCTYPE html>
 
1184
  </div>
1185
  <script>
1186
  const $=s=>document.querySelector(s);
 
1187
  async function tiers(){
1188
  const d=await(await fetch('/api/a11oy/v1/budget/tiers')).json();
1189
  const box=$('#tiers');box.innerHTML='';
 
1222
  }
1223
  }
1224
  $('#go').addEventListener('click',route);
1225
+ $('#q').addEventListener('keydown',event=>{if(event.key==='Enter')route()});
1226
  $('#ref').addEventListener('click',loadSk);
1227
+ document.querySelectorAll('.chip').forEach(chip=>chip.addEventListener('click',()=>{$('#q').value=chip.textContent;route();}));
1228
  tiers();loadSk();
1229
  </script>
1230
  </body></html>"""