betterwithage commited on
Commit
9d184e3
·
verified ·
1 Parent(s): 4355a3c

chore(sync): mirror backend .py + Dockerfile to Space (hf-sync-backend)

Browse files

Automated backend sync from szl-holdings/a11oy main via hf-sync-backend.
Updated (differed from the Space): Dockerfile, serve.py, szl3d_holographic.py, szl_brainexplain.py
Deleted (gone from the repo + Dockerfile COPY set): (none)

Keeps the Space-built backend (serve.py + the Dockerfile-COPY'd .py
modules) identical to GitHub main so the Space never rebuilds from a
stale backend, new endpoints don't 404 there, and orphaned modules
removed from the repo don't linger in the Space tree.

Files changed (4) hide show
  1. Dockerfile +11 -0
  2. serve.py +19 -0
  3. szl3d_holographic.py +1 -0
  4. szl_brainexplain.py +664 -0
Dockerfile CHANGED
@@ -1556,6 +1556,17 @@ COPY szl_brainqueryaudit.py ./szl_brainqueryaudit.py
1556
  # Node-origin lineage — NOT per-answer provenance, NOT build/model attestation. Adds
1557
  # NOTHING to the locked-8; Λ = Conjecture 1; trust 0.97.
1558
  COPY szl_brainlineage.py ./szl_brainlineage.py
 
 
 
 
 
 
 
 
 
 
 
1559
 
1560
 
1561
 
 
1556
  # Node-origin lineage — NOT per-answer provenance, NOT build/model attestation. Adds
1557
  # NOTHING to the locked-8; Λ = Conjecture 1; trust 0.97.
1558
  COPY szl_brainlineage.py ./szl_brainlineage.py
1559
+ # BRAIN EXPLAIN (feat/frontier-brainexplain) — per-file COPY (this Dockerfile has NO
1560
+ # `COPY . .`; the copy-completeness guard requires every module reachable from serve.py
1561
+ # to appear in the COPY set). szl_brainexplain.py is imported by serve.py and turns the
1562
+ # brain's REAL retrieval (szl_brain_api.ask) into a deterministic, plain-language
1563
+ # explanation trace — which query terms matched which seed nodes, why each supporting
1564
+ # node ranked where it did (ppr vs salience), which communities were traversed, each
1565
+ # node's OWN label VERBATIM — with an EXPLAINABLE/PARTIALLY-EXPLAINABLE/OPAQUE verdict
1566
+ # (MODELED). Its 3D surface brainexplain.js ships via the whole-tree `COPY static/3d/
1567
+ # ./static/3d/` above. DESCRIBES only — adds NOTHING to the locked-8; Λ = Conjecture 1.
1568
+ COPY szl_brainexplain.py ./szl_brainexplain.py
1569
+
1570
 
1571
 
1572
 
serve.py CHANGED
@@ -1199,6 +1199,25 @@ except Exception as _brainlineage_e: # pragma: no cover
1199
  print(f"[a11oy] Brain lineage NOT registered: {_brainlineage_e!r}; SPA + API unaffected", file=__import__("sys").stderr)
1200
 
1201
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1202
  # -- BRAIN COMMAND view (Wave O / Dev 5) — the founder's "Brain powering the
1203
  # ecosystem" dashboard. Read-only command rollup over the Brain nervous-system hub:
1204
  # GET /api/a11oy/v1/brain/command → {knowledge harvested, energy harnessed, organs/
 
1199
  print(f"[a11oy] Brain lineage NOT registered: {_brainlineage_e!r}; SPA + API unaffected", file=__import__("sys").stderr)
1200
 
1201
 
1202
+ # -- BRAIN EXPLAIN (feat/frontier-brainexplain) — a transparent, human-readable
1203
+ # explanation of WHY the brain retrieved what it did for a query: GET
1204
+ # /api/a11oy/v1/brain/explain/info (static describe), GET /api/a11oy/v1/brain/explain?q=&k=
1205
+ # (a MODELED explanation trace over the REAL retrieval subgraph — which query terms
1206
+ # matched which seed nodes, why each supporting node ranked where it did (ppr vs
1207
+ # salience), which communities were traversed, each node's OWN label VERBATIM; verdict
1208
+ # EXPLAINABLE/PARTIALLY-EXPLAINABLE/OPAQUE; pure read, mints nothing), POST
1209
+ # /api/a11oy/v1/brain/explain/receipt (same trace + an unsigned SHA-256 content digest,
1210
+ # receipt-on-write). Reuses szl_brain_api.get_index().ask (invents no node, re-ranks
1211
+ # nothing); never fabricates a rationale (honest OPAQUE beats a fake one); never upgrades
1212
+ # a label. Registered BEFORE the SPA /{full_path:path} catch-all. Additive, guarded.
1213
+ try:
1214
+ import szl_brainexplain as _szl_brainexplain
1215
+ _brainexplain_status = _szl_brainexplain.register(app, ns="a11oy")
1216
+ print(f"[a11oy] Brain explain registered: {_brainexplain_status}", file=__import__("sys").stderr)
1217
+ except Exception as _brainexplain_e: # pragma: no cover
1218
+ print(f"[a11oy] Brain explain NOT registered: {_brainexplain_e!r}; SPA + API unaffected", file=__import__("sys").stderr)
1219
+
1220
+
1221
  # -- BRAIN COMMAND view (Wave O / Dev 5) — the founder's "Brain powering the
1222
  # ecosystem" dashboard. Read-only command rollup over the Brain nervous-system hub:
1223
  # GET /api/a11oy/v1/brain/command → {knowledge harvested, energy harnessed, organs/
szl3d_holographic.py CHANGED
@@ -147,6 +147,7 @@ SURFACES: List[Dict[str, str]] = [
147
  {"id": "brainconsensus", "cat": "brain", "title": "Brain Consensus · honest corroboration of a brain grounding · measures how MANY distinct nodes support a query and how many distinct communities they span (cross-community agreement is stronger than one clique) → CORROBORATED/WEAK-CORROBORATION/SINGLE-SOURCE, single-source-risk flag when support collapses to one node/community, never CORROBORATED while that flag is set, MODELED corroboration honesty not a truth guarantee, unsigned SHA-256 receipt-on-write", "owner": "WaveU-Dev1"},
148
  {"id": "brainqueryaudit", "cat": "brain", "title": "Brain Query Audit · append-only hash-linked ledger of brain queries + the honest verdict each returned · POST appends {query, verdict, grounding_label} and mints an UNSIGNED SHA-256 receipt chained to the prior entry (tamper-evident); GET recomputes the whole chain → CHAIN-INTACT/CHAIN-BROKEN (never softened), ephemeral in-memory ledger labelled honestly, MODELED, receipt-on-write-not-on-read", "owner": "WaveT-Dev1"},
149
  {"id": "brainlineage", "cat": "brain", "title": "Brain Lineage · node-origin chain · how each knowledge-graph node ENTERED the graph, read VERBATIM from its OWN real origin fields (source/url → structural derivation → none) → TRACED/PARTIAL-LINEAGE/UNKNOWN-ORIGIN, a node with no source is UNKNOWN-ORIGIN never a fabricated source, aggregate never TRACED while any origin UNKNOWN, unsigned SHA-256 receipt-on-write (node-origin lineage, NOT per-answer provenance, NOT build/model attestation)", "owner": "WaveT-Dev1"},
 
150
  ]
151
 
152
  # Content-type by extension (the only extensions we serve from the 3d tree).
 
147
  {"id": "brainconsensus", "cat": "brain", "title": "Brain Consensus · honest corroboration of a brain grounding · measures how MANY distinct nodes support a query and how many distinct communities they span (cross-community agreement is stronger than one clique) → CORROBORATED/WEAK-CORROBORATION/SINGLE-SOURCE, single-source-risk flag when support collapses to one node/community, never CORROBORATED while that flag is set, MODELED corroboration honesty not a truth guarantee, unsigned SHA-256 receipt-on-write", "owner": "WaveU-Dev1"},
148
  {"id": "brainqueryaudit", "cat": "brain", "title": "Brain Query Audit · append-only hash-linked ledger of brain queries + the honest verdict each returned · POST appends {query, verdict, grounding_label} and mints an UNSIGNED SHA-256 receipt chained to the prior entry (tamper-evident); GET recomputes the whole chain → CHAIN-INTACT/CHAIN-BROKEN (never softened), ephemeral in-memory ledger labelled honestly, MODELED, receipt-on-write-not-on-read", "owner": "WaveT-Dev1"},
149
  {"id": "brainlineage", "cat": "brain", "title": "Brain Lineage · node-origin chain · how each knowledge-graph node ENTERED the graph, read VERBATIM from its OWN real origin fields (source/url → structural derivation → none) → TRACED/PARTIAL-LINEAGE/UNKNOWN-ORIGIN, a node with no source is UNKNOWN-ORIGIN never a fabricated source, aggregate never TRACED while any origin UNKNOWN, unsigned SHA-256 receipt-on-write (node-origin lineage, NOT per-answer provenance, NOT build/model attestation)", "owner": "WaveT-Dev1"},
150
+ {"id": "brainexplain", "cat": "brain", "title": "Brain Explain · transparent explanation of WHY the brain retrieved what it did · MODELED descriptive trace over the REAL retrieval subgraph (which query terms matched which seed nodes, per-node ppr-vs-salience rationale, communities traversed, each node's OWN label VERBATIM) → EXPLAINABLE/PARTIALLY-EXPLAINABLE/OPAQUE (never invents a rationale; honest OPAQUE beats a fake one), unsigned SHA-256 receipt-on-write", "owner": "WaveT-Dev1"},
151
  ]
152
 
153
  # Content-type by extension (the only extensions we serve from the 3d tree).
szl_brainexplain.py ADDED
@@ -0,0 +1,664 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ # © 2026 Lutar, Stephen P. — SZL Holdings · ORCID 0009-0001-0110-4173
4
+ # Doctrine v11 LOCKED · Λ = Conjecture 1
5
+ # Signed-off-by: Stephen Lutar <stephenlutar2@gmail.com>
6
+ """szl_brainexplain.py — BRAIN EXPLAIN: a transparent, human-readable explanation of
7
+ WHY the brain retrieved what it did for a query.
8
+
9
+ Brain Explain is an explainability trace over the REAL retrieval subgraph. For a
10
+ query it reuses the SAME honest retrieval the brain already runs
11
+ (szl_brain_api.get_index().ask) and turns it into a deterministic, plain-language
12
+ account of the retrieval — never a rationale it invents. It is PURE honesty /
13
+ observability over the knowledge graph: it advances NO detection / fusion /
14
+ effector / targeting / cueing capability. It only DESCRIBES the retrieval the
15
+ estate's own brain already performed.
16
+
17
+ WHAT IT DESCRIBES, at request time (all read VERBATIM from the live retrieval; this
18
+ module invents no node, harvests nothing, and ranks nothing anew):
19
+ * seed matches — which query terms literally matched which seed nodes
20
+ (exact token overlap / substring / MODELED vector proxy).
21
+ * per-node rationale — for each supporting node: its rank, its personalized
22
+ PageRank (ppr) and how much that lifted it above its
23
+ baseline salience (ppr_gain), and the honest BASIS for its
24
+ inclusion (direct-term-match / substring-match /
25
+ vector-similarity / graph-traversal / unattributed).
26
+ * communities — which knowledge-graph communities the grounding traversed.
27
+ * honest labels — every supporting node's OWN label VERBATIM, never upgraded.
28
+
29
+ The explanation is DESCRIPTIVE of the real retrieval. Where a node has no
30
+ attributable signal it is reported as `unattributed` honestly, not rationalized.
31
+ The trace label is MODELED (a derived account over a real subgraph, never a
32
+ MEASURED fact about the world).
33
+
34
+ VERDICT over the reachable evidence:
35
+ EXPLAINABLE — at least one supporting node traces to a direct query-term
36
+ match and every supporting node has an attributable basis.
37
+ PARTIALLY-EXPLAINABLE — the retrieval is traceable but rests only on a MODELED
38
+ similarity proxy / traversal (no direct term anchor), or
39
+ some supporting node is unattributed.
40
+ OPAQUE — retrieval returned too little to explain (no query-matched
41
+ seed, or no supporting nodes): the grounding would be
42
+ generic global salience, not query-driven, so no
43
+ query-relevance rationale is fabricated.
44
+ An OPAQUE/PARTIAL result is never softened to EXPLAINABLE; a truthful OPAQUE beats a
45
+ fabricated rationale.
46
+
47
+ RECEIPTS — RECEIPT-ON-WRITE, NOT ON-READ. The GET info/explain reads mint NOTHING.
48
+ Only the POST receipt endpoint emits an UNSIGNED SHA-256 content digest over the
49
+ explanation trace (mirrors the govern/honestywall content-digest pattern) — a plain
50
+ content hash, never a fabricated signature, never a receipt on a GET.
51
+
52
+ DOCTRINE v11:
53
+ * Adds NOTHING to the locked-8 {F1,F4,F7,F11,F12,F18,F19,F22}; it only DESCRIBES.
54
+ Touches no locked formula and no kernel.
55
+ * Λ stays Conjecture 1 (advisory); introduces no theorem, no green/1.0, no proof
56
+ of Λ. Khipu BFT remains Conjecture 2. Trust ceiling 0.97, never 100%.
57
+ * No label is ever upgraded; an OPAQUE trace can never be reported as EXPLAINABLE.
58
+ * Pure stdlib (+numpy tolerated, not required). Additive routes, registered before
59
+ the SPA catch-all; canonical domain a-11-oy.com; 0 runtime CDN.
60
+ """
61
+
62
+ import datetime
63
+ import hashlib
64
+ import json
65
+ import re
66
+
67
+ # Honesty-label vocabulary (doctrine v11). Re-stated here (not imported) so a broken
68
+ # import can never silently blank the vocabulary; tests grep these exact strings.
69
+ HONEST_LABELS = (
70
+ "LIVE", "MEASURED", "MODELED", "SAMPLE", "SIMULATED", "CACHED", "PROVEN",
71
+ "CONJECTURE", "ROADMAP", "DEGRADED", "REPLAY", "STRUCTURAL-ONLY", "HONEST-STUB",
72
+ "UNSIGNED-LOCAL", "UNAVAILABLE",
73
+ )
74
+
75
+ # An explanation trace is a derived account over a real subgraph — MODELED, never
76
+ # MEASURED. Absent retrieval degrades honestly to UNAVAILABLE.
77
+ LBL_MODELED = "MODELED"
78
+ LBL_UNAVAILABLE = "UNAVAILABLE"
79
+
80
+ # Explainability verdicts.
81
+ EXPLAINABLE = "EXPLAINABLE"
82
+ PARTIALLY_EXPLAINABLE = "PARTIALLY-EXPLAINABLE"
83
+ OPAQUE = "OPAQUE"
84
+
85
+ # Inclusion bases (the honest reason a node is in the grounding set).
86
+ BASIS_DIRECT = "direct-term-match"
87
+ BASIS_SUBSTRING = "substring-match"
88
+ BASIS_VECTOR = "vector-similarity"
89
+ BASIS_TRAVERSAL = "graph-traversal"
90
+ BASIS_UNATTRIBUTED = "unattributed"
91
+
92
+ TRUST_CEILING = 0.97
93
+ LOCKED_SET = ["F1", "F4", "F7", "F11", "F12", "F18", "F19", "F22"]
94
+ LOCKED_COUNT = 8
95
+ KERNEL_COMMIT = "c7c0ba17"
96
+
97
+ # This surface's own id (must match szl3d_holographic.SURFACES + holographic.html).
98
+ SURFACE_ID = "brainexplain"
99
+
100
+ _TOKEN_RE = re.compile(r"[a-z0-9]+")
101
+
102
+
103
+ def _now_iso() -> str:
104
+ return datetime.datetime.now(datetime.timezone.utc).isoformat()
105
+
106
+
107
+ def _tokens(*parts) -> list:
108
+ """Lowercase alnum tokens (len>=2) from the given strings, in order. Mirrors the
109
+ tokenizer szl_brain_api uses so a matched term is a term that literally appears in
110
+ the node's own text — never one this module invents."""
111
+ out = []
112
+ for p in parts:
113
+ for t in _TOKEN_RE.findall((str(p) or "").lower()):
114
+ if len(t) >= 2:
115
+ out.append(t)
116
+ return out
117
+
118
+
119
+ def _doctrine_block(note: str = "") -> dict:
120
+ d = {
121
+ "version": "v11",
122
+ "label_top": LBL_MODELED,
123
+ "locked_proven": LOCKED_COUNT,
124
+ "locked_set": list(LOCKED_SET),
125
+ "kernel_commit": KERNEL_COMMIT,
126
+ "adds_to_locked_8": 0,
127
+ "lambda": "Conjecture 1",
128
+ "khipu_bft": "Conjecture 2",
129
+ "trust_ceiling": TRUST_CEILING,
130
+ "trust_100_percent": False,
131
+ "runtime_cdn": 0,
132
+ }
133
+ if note:
134
+ d["note"] = note
135
+ return d
136
+
137
+
138
+ def _node_label(v) -> str:
139
+ """A node's OWN honesty label, VERBATIM. A missing label is 'UNLABELLED' — never
140
+ fabricated up to MEASURED/PROVEN."""
141
+ return str(v) if v is not None else "UNLABELLED"
142
+
143
+
144
+ # --------------------------------------------------------------------------- #
145
+ # The explanation trace — a PURE, deterministic account of one real retrieval.
146
+ # --------------------------------------------------------------------------- #
147
+
148
+ def build_trace(*, query: str, seeds: list, grounding_nodes: list,
149
+ community_summaries: list, ns: str = "a11oy",
150
+ content_hash: str = "") -> dict:
151
+ """Build a deterministic, plain-language explanation of a single retrieval.
152
+
153
+ Pure and deterministic: given the same retrieval primitives it returns the same
154
+ trace. It DESCRIBES the retrieval — it never re-ranks, and it never invents a
155
+ rationale for a node that has no attributable signal.
156
+
157
+ Args:
158
+ query: the raw query string.
159
+ seeds: the retrieval's seed hits — dicts with id/title/kind/score
160
+ and a 'match' block {exact_token_overlap, vector_cosine,
161
+ substring}. May be empty (no query match => OPAQUE).
162
+ grounding_nodes: the grounding subgraph node views — dicts with id/title/
163
+ kind/node_label/salience/community and a per-query 'ppr'.
164
+ community_summaries: the community context covering the grounding set.
165
+ content_hash: the graph content hash (for correlating traces).
166
+ """
167
+ q_tokens = set(_tokens(query))
168
+ seeds = list(seeds or [])
169
+ seeds_by_id = {s.get("id"): s for s in seeds}
170
+ seed_ids = set(seeds_by_id)
171
+
172
+ # Deterministic ordering: strongest personalized rank first, id as tiebreak.
173
+ gnodes = sorted(
174
+ list(grounding_nodes or []),
175
+ key=lambda n: (-float(n.get("ppr") or 0.0), str(n.get("id"))))
176
+
177
+ supporting = []
178
+ for rank, n in enumerate(gnodes, 1):
179
+ nid = n.get("id")
180
+ text_tokens = set(_tokens(n.get("title", ""), n.get("kind", ""), nid))
181
+ matched = sorted(q_tokens & text_tokens)
182
+ ppr = round(float(n.get("ppr") or 0.0), 8)
183
+ sal = round(float(n.get("salience") or 0.0), 8)
184
+ ppr_gain = round(ppr - sal, 8)
185
+ is_seed = nid in seed_ids
186
+ match_info = (seeds_by_id.get(nid, {}) or {}).get("match", {}) or {}
187
+ vec = float(match_info.get("vector_cosine") or 0.0)
188
+ substr = bool(match_info.get("substring"))
189
+
190
+ if matched:
191
+ basis = BASIS_DIRECT
192
+ why = ("query term(s) [" + ", ".join(matched) + "] appear in this node's "
193
+ "own text; ranked #%d by personalized PageRank (ppr=%s, %+0.8f vs "
194
+ "baseline salience %s)" % (rank, ppr, ppr_gain, sal))
195
+ elif is_seed and substr:
196
+ basis = BASIS_SUBSTRING
197
+ why = ("the query is a substring of this node's title/id; ranked #%d "
198
+ "(ppr=%s, %+0.8f vs baseline salience %s)" % (rank, ppr, ppr_gain, sal))
199
+ elif is_seed and vec > 0.0:
200
+ basis = BASIS_VECTOR
201
+ why = ("no exact term overlap; included via a MODELED hash-embedding "
202
+ "similarity proxy (cosine=%s); ranked #%d (ppr=%s)"
203
+ % (round(vec, 6), rank, ppr))
204
+ elif ppr > 0.0:
205
+ basis = BASIS_TRAVERSAL
206
+ why = ("not a direct query match; reached by graph traversal from the "
207
+ "matched seed node(s) (personalized PageRank ppr=%s, %+0.8f vs "
208
+ "baseline salience %s)" % (ppr, ppr_gain, sal))
209
+ else:
210
+ basis = BASIS_UNATTRIBUTED
211
+ why = ("present in the grounding set with no query-term match, similarity, "
212
+ "or traversal signal to attribute — reported honestly, not rationalized")
213
+
214
+ supporting.append({
215
+ "rank": rank,
216
+ "id": nid,
217
+ "title": n.get("title", nid),
218
+ "kind": n.get("kind"),
219
+ "node_label": _node_label(n.get("node_label")), # VERBATIM
220
+ "community": n.get("community"),
221
+ "ppr": ppr,
222
+ "salience": sal,
223
+ "ppr_gain": ppr_gain,
224
+ "is_seed": is_seed,
225
+ "matched_terms": matched,
226
+ "basis": basis,
227
+ "why": why,
228
+ })
229
+
230
+ # ---- seed-term matches (which query terms matched which seed) ---------- #
231
+ seed_matches = []
232
+ for s in seeds:
233
+ st = set(_tokens(s.get("title", ""), s.get("kind", ""), s.get("id")))
234
+ m = sorted(q_tokens & st)
235
+ sm = s.get("match", {}) or {}
236
+ seed_matches.append({
237
+ "id": s.get("id"),
238
+ "title": s.get("title", s.get("id")),
239
+ "node_label": _node_label(s.get("node_label")), # VERBATIM
240
+ "matched_terms": m,
241
+ "exact_token_overlap": sm.get("exact_token_overlap"),
242
+ "vector_cosine": sm.get("vector_cosine"),
243
+ "substring": bool(sm.get("substring")),
244
+ "score": s.get("score"),
245
+ })
246
+
247
+ # ---- communities traversed by the grounding set ----------------------- #
248
+ comm_by_id = {c.get("id"): c for c in (community_summaries or []) if isinstance(c, dict)}
249
+ traversed: dict = {}
250
+ for n in gnodes:
251
+ cid = n.get("community")
252
+ if cid is None:
253
+ continue
254
+ traversed[cid] = traversed.get(cid, 0) + 1
255
+ communities = []
256
+ for cid, cnt in sorted(traversed.items(), key=lambda kv: (-kv[1], str(kv[0]))):
257
+ c = comm_by_id.get(cid, {})
258
+ communities.append({
259
+ "id": cid,
260
+ "nodes_in_grounding": cnt,
261
+ "size": c.get("size"),
262
+ "summary": c.get("summary"),
263
+ })
264
+
265
+ # ---- honest verdict over the reachable evidence ----------------------- #
266
+ n_support = len(supporting)
267
+ direct = [s for s in supporting if s["basis"] in (BASIS_DIRECT, BASIS_SUBSTRING)]
268
+ unattributed = [s for s in supporting if s["basis"] == BASIS_UNATTRIBUTED]
269
+
270
+ if not seeds or n_support == 0:
271
+ verdict = OPAQUE
272
+ reason = (("no query-matched seed node" if not seeds
273
+ else "no supporting node in the grounding set")
274
+ + " — retrieval returned too little to explain; the grounding would "
275
+ "be generic global salience, not query-driven, so no query-relevance "
276
+ "rationale is fabricated.")
277
+ elif unattributed:
278
+ verdict = PARTIALLY_EXPLAINABLE
279
+ reason = ("%d of %d supporting node(s) have no attributable retrieval signal; "
280
+ "the retrieval is only partially explainable and is reported honestly."
281
+ % (len(unattributed), n_support))
282
+ elif not direct:
283
+ verdict = PARTIALLY_EXPLAINABLE
284
+ reason = ("no exact query-term match anchors the retrieval; the explanation "
285
+ "rests only on a MODELED similarity proxy and graph traversal, so it "
286
+ "is partially explainable.")
287
+ else:
288
+ verdict = EXPLAINABLE
289
+ reason = ("%d of %d supporting node(s) trace to a direct query-term match; the "
290
+ "rest are reached by transparent graph traversal from those matches."
291
+ % (len(direct), n_support))
292
+
293
+ explainable_share = (round((n_support - len(unattributed)) / n_support, 6)
294
+ if n_support else 0.0)
295
+
296
+ return {
297
+ "label": LBL_MODELED,
298
+ "surface_id": SURFACE_ID,
299
+ "ns": ns,
300
+ "query": query,
301
+ "content_hash": content_hash,
302
+ "verdict": verdict,
303
+ "verdict_reason": reason,
304
+ "retrieval": ("hippoRAG-PPR(local) ⊕ graphRAG-community(global) — the SAME "
305
+ "honest retrieval szl_brain_api runs; this trace only describes it."),
306
+ "seed_matches": seed_matches,
307
+ "supporting_nodes": supporting,
308
+ "communities_traversed": communities,
309
+ "summary": {
310
+ "seed_count": len(seeds),
311
+ "supporting_count": n_support,
312
+ "direct_match_count": len(direct),
313
+ "unattributed_count": len(unattributed),
314
+ "community_count": len(communities),
315
+ "explainable_share": explainable_share,
316
+ },
317
+ "method": ("descriptive explainability trace over the real retrieval subgraph: "
318
+ "seed-term matches, per-node ppr-vs-salience rationale, communities "
319
+ "traversed, and each supporting node's OWN label VERBATIM. MODELED — "
320
+ "never invents a rationale."),
321
+ "honest_labels_vocabulary": list(HONEST_LABELS),
322
+ "doctrine": _doctrine_block(
323
+ "additive DESCRIBE-only surface over the knowledge graph; touches no locked "
324
+ "formula and no kernel; Λ = Conjecture 1, never a theorem."),
325
+ "timestamp_utc": _now_iso(),
326
+ }
327
+
328
+
329
+ # --------------------------------------------------------------------------- #
330
+ # Live trace — reuse the AUDITED brain index / ask() (never re-harvest, never re-rank).
331
+ # --------------------------------------------------------------------------- #
332
+
333
+ def live_explanation(ns: str = "a11oy", q: str = "", k: int = 12) -> dict:
334
+ """Read the live retrieval for q and build its explanation trace.
335
+
336
+ Fully guarded: if the brain index/retrieval is unavailable, returns an honest
337
+ UNAVAILABLE/OPAQUE trace rather than raising or fabricating a rationale."""
338
+ try:
339
+ import szl_brain_api as _brain_api
340
+ idx = _brain_api.get_index(ns)
341
+ a = idx.ask(q, k=max(1, int(k)))
342
+ grounding = a.get("grounding_subgraph", {}) or {}
343
+ return build_trace(
344
+ query=q,
345
+ seeds=a.get("seeds", []) or [],
346
+ grounding_nodes=grounding.get("nodes", []) or [],
347
+ community_summaries=a.get("community_context", []) or [],
348
+ ns=ns,
349
+ content_hash=getattr(idx, "content_hash", ""),
350
+ )
351
+ except Exception as exc: # honest degrade — never a fabricated rationale
352
+ return {
353
+ "label": LBL_UNAVAILABLE,
354
+ "surface_id": SURFACE_ID,
355
+ "ns": ns,
356
+ "query": q,
357
+ "verdict": OPAQUE,
358
+ "verdict_reason": ("brain retrieval unavailable this request; no explanation "
359
+ "fabricated (honest OPAQUE/UNAVAILABLE)."),
360
+ "error": str(exc)[:200],
361
+ "seed_matches": [],
362
+ "supporting_nodes": [],
363
+ "communities_traversed": [],
364
+ "summary": {"seed_count": 0, "supporting_count": 0, "direct_match_count": 0,
365
+ "unattributed_count": 0, "community_count": 0,
366
+ "explainable_share": 0.0},
367
+ "doctrine": _doctrine_block("retrieval unavailable; no rationale fabricated."),
368
+ "timestamp_utc": _now_iso(),
369
+ }
370
+
371
+
372
+ # --------------------------------------------------------------------------- #
373
+ # Receipt — UNSIGNED SHA-256 content digest. RECEIPT-ON-WRITE (POST), never GET.
374
+ # --------------------------------------------------------------------------- #
375
+
376
+ def _canonical_core(trace: dict) -> str:
377
+ """Deterministic canonical serialization of the explanation-bearing content
378
+ (excludes the volatile timestamp), so the digest attests the VERDICT + evidence,
379
+ not the clock."""
380
+ core = {
381
+ "query": trace.get("query"),
382
+ "verdict": trace.get("verdict"),
383
+ "content_hash": trace.get("content_hash"),
384
+ "seed_matches": [
385
+ {"id": s.get("id"), "matched_terms": s.get("matched_terms"),
386
+ "node_label": s.get("node_label")}
387
+ for s in trace.get("seed_matches", [])
388
+ ],
389
+ "supporting_nodes": [
390
+ {"id": s.get("id"), "rank": s.get("rank"), "basis": s.get("basis"),
391
+ "node_label": s.get("node_label"), "matched_terms": s.get("matched_terms"),
392
+ "ppr": s.get("ppr"), "salience": s.get("salience")}
393
+ for s in trace.get("supporting_nodes", [])
394
+ ],
395
+ "communities_traversed": [c.get("id") for c in trace.get("communities_traversed", [])],
396
+ }
397
+ return json.dumps(core, sort_keys=True, separators=(",", ":"), default=str)
398
+
399
+
400
+ def _content_receipt(trace: dict) -> dict:
401
+ """An UNSIGNED SHA-256 content-digest receipt over the explanation trace (no
402
+ signature fabricated). RECEIPT-ON-WRITE — only the POST receipt path calls this."""
403
+ canonical = _canonical_core(trace)
404
+ digest = hashlib.sha256(canonical.encode("utf-8")).hexdigest()
405
+ return {
406
+ "kind": "szl.brainexplain.trace",
407
+ "algorithm": "sha256",
408
+ "content_sha256": digest,
409
+ "signed": False,
410
+ "mode": "UNSIGNED-CONTENT-DIGEST",
411
+ "receipt_on": "write (POST receipt)",
412
+ "note": ("unsigned SHA-256 content digest of the explanation trace; "
413
+ "RECEIPT-ON-WRITE, never on a GET read. No signature fabricated."),
414
+ "computed_at": _now_iso(),
415
+ }
416
+
417
+
418
+ # --------------------------------------------------------------------------- #
419
+ # Handlers.
420
+ # --------------------------------------------------------------------------- #
421
+
422
+ def handle_info(ns: str = "a11oy") -> dict:
423
+ """GET /brain/explain/info — static self-describing manifest (no compute). PURE READ."""
424
+ base = f"/api/{ns}/v1/brain/explain"
425
+ return {
426
+ "ok": True,
427
+ "service": "a11oy.brain.explain",
428
+ "endpoint": "brain/explain/info",
429
+ "surface_id": SURFACE_ID,
430
+ "label": LBL_MODELED,
431
+ "title": "Brain Explain — why the brain retrieved what it did",
432
+ "what": ("produces a deterministic, plain-language explanation trace over the "
433
+ "REAL retrieval subgraph for a query: which query terms matched which "
434
+ "seed nodes, why each supporting node ranked where it did (ppr vs "
435
+ "salience), which communities were traversed, and each supporting "
436
+ "node's OWN honesty label VERBATIM. Pure honesty/observability over "
437
+ "the knowledge graph; advances no detection/fusion/effector/targeting/"
438
+ "cueing capability. DESCRIBES the retrieval — never invents a rationale; "
439
+ "never upgrades a label."),
440
+ "endpoints": {
441
+ "info": f"GET {base}/info",
442
+ "explain": f"GET {base}?q=&k=",
443
+ "receipt": f"POST {base}/receipt",
444
+ },
445
+ "method": ("reuses szl_brain_api.get_index().ask (invents no node, re-ranks "
446
+ "nothing); describes the seeds, the grounding subgraph's per-node "
447
+ "personalized-PageRank rationale, and the communities traversed."),
448
+ "verdicts": [EXPLAINABLE, PARTIALLY_EXPLAINABLE, OPAQUE],
449
+ "verdict_legend": {
450
+ EXPLAINABLE: ("a direct query-term match anchors the retrieval and every "
451
+ "supporting node has an attributable basis"),
452
+ PARTIALLY_EXPLAINABLE: ("traceable but rests only on a MODELED similarity "
453
+ "proxy / traversal, or some node is unattributed"),
454
+ OPAQUE: ("retrieval returned too little to explain (no query-matched seed "
455
+ "or no supporting nodes); no rationale fabricated"),
456
+ },
457
+ "inclusion_bases": [BASIS_DIRECT, BASIS_SUBSTRING, BASIS_VECTOR,
458
+ BASIS_TRAVERSAL, BASIS_UNATTRIBUTED],
459
+ "receipt_policy": ("RECEIPT-ON-WRITE-NOT-ON-READ — only POST /receipt emits an "
460
+ "unsigned SHA-256 content digest; GET mints nothing."),
461
+ "honest_labels_vocabulary": list(HONEST_LABELS),
462
+ "doctrine": _doctrine_block(
463
+ "additive DESCRIBE-only surface over the knowledge graph; touches no locked "
464
+ "formula and no kernel; Λ = Conjecture 1, never a theorem."),
465
+ "timestamp_utc": _now_iso(),
466
+ }
467
+
468
+
469
+ def handle_explain(ns: str = "a11oy", q: str = "", k: int = 12) -> dict:
470
+ """GET /brain/explain?q=&k= — the explanation trace for q. PURE READ (mints nothing)."""
471
+ trace = live_explanation(ns, q, k)
472
+ trace["ok"] = trace.get("label") != LBL_UNAVAILABLE
473
+ trace["endpoint"] = "brain/explain"
474
+ trace["receipt_policy"] = ("RECEIPT-ON-WRITE-NOT-ON-READ — GET mints nothing; "
475
+ "POST /receipt digests.")
476
+ return trace
477
+
478
+
479
+ def handle_receipt(ns: str = "a11oy", q: str = "", k: int = 12) -> dict:
480
+ """POST /brain/explain/receipt — the explanation trace + an UNSIGNED SHA-256
481
+ content-digest receipt (RECEIPT-ON-WRITE). Never 500s: honest degraded response."""
482
+ try:
483
+ trace = live_explanation(ns, q, k)
484
+ out = dict(trace)
485
+ out["ok"] = True
486
+ out["endpoint"] = "brain/explain/receipt"
487
+ out["receipt"] = _content_receipt(trace)
488
+ return out
489
+ except Exception as exc:
490
+ return {
491
+ "ok": False, "endpoint": "brain/explain/receipt", "label": LBL_UNAVAILABLE,
492
+ "verdict": OPAQUE, "error": str(exc)[:200],
493
+ "doctrine": "v11: receipt unavailable; no fabricated verdict/receipt emitted.",
494
+ "timestamp_utc": _now_iso(),
495
+ }
496
+
497
+
498
+ # --------------------------------------------------------------------------- #
499
+ # FastAPI router registration.
500
+ # GET info/explain — normal FastAPI GET handlers (pure reads; mint nothing).
501
+ # POST receipt — raw-Request handler via app.router.add_route (Starlette passes
502
+ # the Request positionally, version-proof under fastapi==0.137.x),
503
+ # with app.add_api_route as the fallback. The handler is annotated
504
+ # request: fastapi.Request. Registered BEFORE the SPA catch-all.
505
+ # --------------------------------------------------------------------------- #
506
+
507
+ def register(app, ns: str = "a11oy") -> str:
508
+ from fastapi.responses import JSONResponse
509
+
510
+ base = f"/api/{ns}/v1/brain/explain"
511
+
512
+ @app.get(f"{base}/info")
513
+ def _brainexplain_info():
514
+ """Self-describing brain-explain manifest (pure read; mints nothing)."""
515
+ return JSONResponse(handle_info(ns))
516
+
517
+ @app.get(base)
518
+ def _brainexplain_explain(q: str = "", k: int = 12):
519
+ """Explanation trace for q; MODELED (pure read; mints nothing)."""
520
+ return JSONResponse(handle_explain(ns, q, k))
521
+
522
+ async def _brainexplain_receipt(request):
523
+ """POST: the explanation trace for the query (q/k from the JSON body or query
524
+ params) + an UNSIGNED SHA-256 content digest (RECEIPT-ON-WRITE)."""
525
+ q, k = "", 12
526
+ try:
527
+ raw = await request.body()
528
+ if raw:
529
+ body = json.loads(raw)
530
+ if isinstance(body, dict):
531
+ q = str(body.get("q", body.get("query", "")) or "")
532
+ k = int(body.get("k", 12) or 12)
533
+ except Exception: # a malformed body degrades to an empty query, never a 500
534
+ q, k = "", 12
535
+ # query params win when present (parity with the GET path).
536
+ try:
537
+ qp = request.query_params
538
+ if qp.get("q") is not None:
539
+ q = str(qp.get("q"))
540
+ if qp.get("k") is not None:
541
+ k = int(qp.get("k"))
542
+ except Exception:
543
+ pass
544
+ return JSONResponse(handle_receipt(ns, q, k))
545
+
546
+ # Annotate the raw-Request handler as fastapi.Request so any FastAPI signature
547
+ # analysis (in the add_api_route fallback path) treats the param as the request
548
+ # object (0.137.x gotcha).
549
+ try:
550
+ import fastapi as _fastapi
551
+ _brainexplain_receipt.__annotations__["request"] = _fastapi.Request
552
+ except Exception: # noqa: BLE001 — annotation is best-effort only
553
+ pass
554
+
555
+ rcpt_path = f"{base}/receipt"
556
+ add_route = getattr(getattr(app, "router", None), "add_route", None)
557
+ add_api_route = getattr(app, "add_api_route", None)
558
+ try:
559
+ if callable(add_route):
560
+ app.router.add_route(rcpt_path, _brainexplain_receipt, methods=["POST"])
561
+ elif callable(add_api_route):
562
+ app.add_api_route(rcpt_path, _brainexplain_receipt, methods=["POST"])
563
+ else: # pragma: no cover — last-resort Starlette Route append
564
+ from starlette.routing import Route
565
+ app.router.routes.append(Route(rcpt_path, _brainexplain_receipt, methods=["POST"]))
566
+ except Exception as exc: # additive register must never break boot
567
+ print(f"[{ns}] brainexplain receipt POST route NOT wired (guarded): {exc!r}",
568
+ file=__import__("sys").stderr)
569
+ return "brainexplain-wired:2(get-only)"
570
+
571
+ return "brainexplain-wired:3"
572
+
573
+
574
+ # --------------------------------------------------------------------------- #
575
+ # Self-test — descriptive trace, honest verdicts, receipt only on write.
576
+ # --------------------------------------------------------------------------- #
577
+
578
+ if __name__ == "__main__":
579
+ import sys as _sys
580
+
581
+ print("=" * 72)
582
+ print("szl_brainexplain — self-test (retrieval explainability trace)")
583
+ print("=" * 72)
584
+
585
+ # A small, fully synthetic retrieval fixture (no network, no heavy index build).
586
+ seeds = [
587
+ {"id": "n1", "title": "brain graph harvest", "kind": "module", "score": 0.8,
588
+ "node_label": "HARVESTED",
589
+ "match": {"exact_token_overlap": 0.5, "vector_cosine": 0.3, "substring": False}},
590
+ ]
591
+ grounding = [
592
+ {"id": "n1", "title": "brain graph harvest", "kind": "module",
593
+ "node_label": "HARVESTED", "community": "c0", "salience": 0.10, "ppr": 0.30},
594
+ {"id": "n2", "title": "estate ledger", "kind": "module",
595
+ "node_label": "MODELED", "community": "c0", "salience": 0.05, "ppr": 0.12},
596
+ ]
597
+ comms = [{"id": "c0", "size": 2, "summary": "community c0: 2 nodes"}]
598
+
599
+ trace = build_trace(query="brain graph", seeds=seeds, grounding_nodes=grounding,
600
+ community_summaries=comms, ns="a11oy", content_hash="deadbeef")
601
+
602
+ # 1) descriptive + MODELED + EXPLAINABLE with a direct match anchoring it.
603
+ assert trace["label"] == LBL_MODELED
604
+ assert trace["verdict"] == EXPLAINABLE, trace["verdict"]
605
+ sup = trace["supporting_nodes"]
606
+ assert sup[0]["id"] == "n1" and sup[0]["basis"] == BASIS_DIRECT
607
+ assert "graph" in sup[0]["matched_terms"] and "brain" in sup[0]["matched_terms"]
608
+ assert sup[1]["basis"] == BASIS_TRAVERSAL # reached via PPR, not a direct match
609
+ # node labels are VERBATIM.
610
+ assert sup[0]["node_label"] == "HARVESTED" and sup[1]["node_label"] == "MODELED"
611
+ print(f"[1] EXPLAINABLE, MODELED; n1 direct-term-match, n2 graph-traversal; "
612
+ f"labels verbatim OK")
613
+
614
+ # 2) determinism: same inputs => identical trace (minus the volatile timestamp).
615
+ t2 = build_trace(query="brain graph", seeds=seeds, grounding_nodes=grounding,
616
+ community_summaries=comms, ns="a11oy", content_hash="deadbeef")
617
+ a = dict(trace); a.pop("timestamp_utc")
618
+ b = dict(t2); b.pop("timestamp_utc")
619
+ assert a == b, "trace must be deterministic"
620
+ print("[2] deterministic trace (same retrieval => same account) OK")
621
+
622
+ # 3) OPAQUE when retrieval returns too little (no query-matched seed).
623
+ op = build_trace(query="brain graph", seeds=[], grounding_nodes=grounding,
624
+ community_summaries=comms, ns="a11oy", content_hash="deadbeef")
625
+ assert op["verdict"] == OPAQUE and op["summary"]["seed_count"] == 0
626
+ print("[3] no query-matched seed => OPAQUE (no rationale fabricated) OK")
627
+
628
+ # 4) PARTIALLY-EXPLAINABLE when only a MODELED similarity proxy anchors it.
629
+ # (seed matched by vector only; its text carries none of the query terms.)
630
+ vseeds = [{"id": "v1", "title": "alpha", "kind": "module", "score": 0.2,
631
+ "node_label": "MODELED",
632
+ "match": {"exact_token_overlap": 0.0, "vector_cosine": 0.4,
633
+ "substring": False}}]
634
+ vground = [{"id": "v1", "title": "alpha", "kind": "module", "node_label": "MODELED",
635
+ "community": "c1", "salience": 0.05, "ppr": 0.20}]
636
+ part = build_trace(query="zulu quebec", seeds=vseeds, grounding_nodes=vground,
637
+ community_summaries=[], ns="a11oy", content_hash="beef")
638
+ assert part["verdict"] == PARTIALLY_EXPLAINABLE, part["verdict"]
639
+ assert part["supporting_nodes"][0]["basis"] == BASIS_VECTOR
640
+ print("[4] vector-only anchor => PARTIALLY-EXPLAINABLE OK")
641
+
642
+ # 5) RECEIPT-ON-WRITE: deterministic unsigned sha256; GET explain mints nothing.
643
+ r1 = _content_receipt(trace)
644
+ r2 = _content_receipt(trace)
645
+ assert r1["algorithm"] == "sha256" and len(r1["content_sha256"]) == 64
646
+ assert r1["signed"] is False and r1["mode"] == "UNSIGNED-CONTENT-DIGEST"
647
+ assert r1["content_sha256"] == r2["content_sha256"], "digest must be deterministic"
648
+ g = handle_explain("a11oy", "") # live read (may be OPAQUE off-box) — mints nothing
649
+ assert "receipt" not in g, "GET explain must NOT mint a receipt (receipt-on-write)"
650
+ print(f"[5] POST digest={r1['content_sha256'][:16]}… unsigned + deterministic; "
651
+ f"GET explain mints nothing OK")
652
+
653
+ # 6) doctrine: locked-8 exact, +0, Λ Conjecture 1, trust 0.97 not 100%.
654
+ d = trace["doctrine"]
655
+ assert d["locked_proven"] == 8 and d["locked_set"] == LOCKED_SET
656
+ assert d["adds_to_locked_8"] == 0
657
+ assert d["lambda"] == "Conjecture 1" and d["khipu_bft"] == "Conjecture 2"
658
+ assert d["trust_ceiling"] == 0.97 and d["trust_100_percent"] is False
659
+ assert d["runtime_cdn"] == 0
660
+ assert LBL_MODELED in HONEST_LABELS and LBL_UNAVAILABLE in HONEST_LABELS
661
+ print("[6] doctrine: locked-8 exact, +0, Λ=Conjecture 1, trust 0.97 (not 100%) OK")
662
+
663
+ print("\nok:true checks:6")
664
+ _sys.exit(0)