Spaces:
Running
Running
chore(sync): mirror backend .py + Dockerfile to Space (hf-sync-backend)
Browse filesAutomated backend sync from szl-holdings/a11oy main via hf-sync-backend.
Updated (differed from the Space): Dockerfile, serve.py, szl3d_holographic.py, szl_brain_api.py
Deleted (gone from the repo + Dockerfile COPY set): (none)
Keeps the Space-built backend (serve.py + the Dockerfile-COPY'd .py
modules) identical to GitHub main so the Space never rebuilds from a
stale backend, new endpoints don't 404 there, and orphaned modules
removed from the repo don't linger in the Space tree.
- Dockerfile +4 -0
- serve.py +17 -0
- szl3d_holographic.py +1 -0
- szl_brain_api.py +844 -0
Dockerfile
CHANGED
|
@@ -679,6 +679,10 @@ COPY szl_eu_energy.py ./
|
|
| 679 |
# GET /api/a11oy/v1/brain/graph falls through to the SPA HTML shell (no JSON). Harvests
|
| 680 |
# the real estate (surfaces+formulas+repos+topics) into a layered node/link brain graph.
|
| 681 |
COPY a11oy_brain_graph.py ./
|
|
|
|
|
|
|
|
|
|
|
|
|
| 682 |
# HARVESTED FIELD LEADERS (2026-07-07) — real research graph JSONL (papers/repos/labs/
|
| 683 |
# people/datasets/benchmarks/standards/axes, each with a verified url). a11oy_brain_graph
|
| 684 |
# reads these at runtime to merge the outer "field" layer into /brain/graph; MUST be
|
|
|
|
| 679 |
# GET /api/a11oy/v1/brain/graph falls through to the SPA HTML shell (no JSON). Harvests
|
| 680 |
# the real estate (surfaces+formulas+repos+topics) into a layered node/link brain graph.
|
| 681 |
COPY a11oy_brain_graph.py ./
|
| 682 |
+
# QUERYABLE BRAIN API (WAVE 1) — imported by serve.py (guarded); MUST be per-file COPY'd or
|
| 683 |
+
# GET /api/a11oy/v1/brain/{search,neighbors,community,subgraph,salience,ask,stats,index}
|
| 684 |
+
# falls through to a runtime stub. Reuses a11oy_brain_graph to make the brain traversable.
|
| 685 |
+
COPY szl_brain_api.py ./szl_brain_api.py
|
| 686 |
# HARVESTED FIELD LEADERS (2026-07-07) — real research graph JSONL (papers/repos/labs/
|
| 687 |
# people/datasets/benchmarks/standards/axes, each with a verified url). a11oy_brain_graph
|
| 688 |
# reads these at runtime to merge the outer "field" layer into /brain/graph; MUST be
|
serve.py
CHANGED
|
@@ -790,6 +790,23 @@ except Exception as _brain_graph_e: # pragma: no cover
|
|
| 790 |
print(f"[a11oy] Brain graph NOT registered: {_brain_graph_e!r}; SPA + API unaffected", file=__import__("sys").stderr)
|
| 791 |
|
| 792 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 793 |
# -- szl3d HOLOGRAPHIC ESTATE (Dev0 foundation) -- the vendored three.js r170 toolkit +
|
| 794 |
# the /holographic shell hosting the 3D surfaces (frontier tier + the 9 estate surfaces),
|
| 795 |
# each lazy-loading its per-surface module and lit by REAL a11oy endpoints with honesty
|
|
|
|
| 790 |
print(f"[a11oy] Brain graph NOT registered: {_brain_graph_e!r}; SPA + API unaffected", file=__import__("sys").stderr)
|
| 791 |
|
| 792 |
|
| 793 |
+
# -- szl_brain_api QUERYABLE BRAIN (WAVE 1) -- turns the SAME honest brain graph into a
|
| 794 |
+
# real server-side retrieval API: GET /api/a11oy/v1/brain/{search,neighbors,community,
|
| 795 |
+
# subgraph,salience,ask,stats,index}. Reuses a11oy_brain_graph.get_brain_graph (invents
|
| 796 |
+
# no nodes, harvests nothing, restates no counts). Embedded, zero external DB: NetworkX
|
| 797 |
+
# PageRank/communities, numpy/sqlite-vec vectors, hash-embedding fallback (MODELED, never
|
| 798 |
+
# MEASURED), HippoRAG-style PPR local + GraphRAG community global merged LightRAG-mix. /ask
|
| 799 |
+
# returns a REAL grounding subgraph always; generated prose is UNAVAILABLE unless a sovereign
|
| 800 |
+
# model is reachable — never fabricated. Pure read (0 sign-on-GET). Registered BEFORE the SPA
|
| 801 |
+
# /{full_path:path} catch-all. Additive, try/except-guarded.
|
| 802 |
+
try:
|
| 803 |
+
import szl_brain_api as _szl_brain_api
|
| 804 |
+
_brain_api_status = _szl_brain_api.register(app, ns="a11oy")
|
| 805 |
+
print(f"[a11oy] Brain API registered: {_brain_api_status}", file=__import__("sys").stderr)
|
| 806 |
+
except Exception as _brain_api_e: # pragma: no cover
|
| 807 |
+
print(f"[a11oy] Brain API NOT registered: {_brain_api_e!r}; SPA + API unaffected", file=__import__("sys").stderr)
|
| 808 |
+
|
| 809 |
+
|
| 810 |
# -- szl3d HOLOGRAPHIC ESTATE (Dev0 foundation) -- the vendored three.js r170 toolkit +
|
| 811 |
# the /holographic shell hosting the 3D surfaces (frontier tier + the 9 estate surfaces),
|
| 812 |
# each lazy-loading its per-surface module and lit by REAL a11oy endpoints with honesty
|
szl3d_holographic.py
CHANGED
|
@@ -111,6 +111,7 @@ SURFACES: List[Dict[str, str]] = [
|
|
| 111 |
{"id": "governedagent", "title": "Governed Agent Loop · plan→act→self-eval→gate→retry", "owner": "WaveJ-Dev5"},
|
| 112 |
{"id": "governedrag", "title": "Governed RAG · Retrieval-with-Receipts", "owner": "WaveJ-Dev4"},
|
| 113 |
{"id": "ecosystem", "title": "Harness · Ecosystem Status", "owner": "Wave30-Dev3"},
|
|
|
|
| 114 |
]
|
| 115 |
|
| 116 |
# Content-type by extension (the only extensions we serve from the 3d tree).
|
|
|
|
| 111 |
{"id": "governedagent", "title": "Governed Agent Loop · plan→act→self-eval→gate→retry", "owner": "WaveJ-Dev5"},
|
| 112 |
{"id": "governedrag", "title": "Governed RAG · Retrieval-with-Receipts", "owner": "WaveJ-Dev4"},
|
| 113 |
{"id": "ecosystem", "title": "Harness · Ecosystem Status", "owner": "Wave30-Dev3"},
|
| 114 |
+
{"id": "brainquery", "title": "Brain Query · ask the graph (retrieval)", "owner": "Wave1-Frontier"},
|
| 115 |
]
|
| 116 |
|
| 117 |
# Content-type by extension (the only extensions we serve from the 3d tree).
|
szl_brain_api.py
ADDED
|
@@ -0,0 +1,844 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# SPDX-License-Identifier: Apache-2.0
|
| 2 |
+
# © 2026 Lutar, Stephen P. — SZL Holdings · ORCID 0009-0001-0110-4173 · Doctrine v11 LOCKED
|
| 3 |
+
# Signed-off-by: Stephen Lutar <stephenlutar2@gmail.com>
|
| 4 |
+
"""szl_brain_api.py — make the estate brain QUERYABLE / TRAVERSABLE.
|
| 5 |
+
|
| 6 |
+
WAVE 1 of the frontier program. a11oy_brain_graph.py already HARVESTS the real
|
| 7 |
+
estate into a node/link graph (GET /api/<ns>/v1/brain/graph) — but until now that
|
| 8 |
+
graph was build-only: you could render it in 3D, not *ask* it anything. This
|
| 9 |
+
module turns the SAME honest graph into a real server-side retrieval API:
|
| 10 |
+
|
| 11 |
+
GET /api/<ns>/v1/brain/search ?q=&k= hybrid exact+vector top-k nodes
|
| 12 |
+
GET /api/<ns>/v1/brain/neighbors ?id=&hops= k-hop neighbourhood subgraph
|
| 13 |
+
GET /api/<ns>/v1/brain/community ?id= community of a node + summary
|
| 14 |
+
GET /api/<ns>/v1/brain/subgraph ?ids= induced subgraph over ids
|
| 15 |
+
GET /api/<ns>/v1/brain/salience ?top= PageRank salience ranking
|
| 16 |
+
GET /api/<ns>/v1/brain/ask ?q= PPR grounding subgraph (+answer IF
|
| 17 |
+
a sovereign model is reachable, ELSE
|
| 18 |
+
honest UNAVAILABLE — never fabricated)
|
| 19 |
+
GET /api/<ns>/v1/brain/stats node/edge/community counts, honest
|
| 20 |
+
distinct-vs-total framing
|
| 21 |
+
GET /api/<ns>/v1/brain/index index build status: stack chosen +
|
| 22 |
+
fallbacks + honest tier labels
|
| 23 |
+
|
| 24 |
+
REUSE, NEVER RE-HARVEST: every node/edge comes from
|
| 25 |
+
a11oy_brain_graph.get_brain_graph(ns). This module invents no nodes, harvests
|
| 26 |
+
nothing, and restates no counts — it slices, ranks and traverses the honest graph.
|
| 27 |
+
|
| 28 |
+
EMBEDDED, ZERO EXTERNAL DB. Best-available stack, each tier honestly labelled and
|
| 29 |
+
degrading in the open (a truthful fallback beats a fake dependency):
|
| 30 |
+
|
| 31 |
+
vectors sqlite-vec IF importable ELSE numpy cosine ELSE pure-python cosine
|
| 32 |
+
embeddings local Ollama nomic-embed-text IF SZL_LOCAL_LLM_URL reachable
|
| 33 |
+
ELSE deterministic hash-embedding (labelled MODELED — a
|
| 34 |
+
hash-embedding similarity is NEVER presented as MEASURED)
|
| 35 |
+
communities python-igraph Leiden IF importable
|
| 36 |
+
ELSE networkx greedy_modularity_communities
|
| 37 |
+
local RAG HippoRAG-style Personalized PageRank (networkx.pagerank seeded by
|
| 38 |
+
query-matched nodes)
|
| 39 |
+
global RAG GraphRAG-style community summaries
|
| 40 |
+
merge LightRAG-'mix' style (local subgraph ⊕ global community context)
|
| 41 |
+
|
| 42 |
+
Deterministic: the index is cached keyed by the CONTENT HASH of the graph, so it
|
| 43 |
+
rebuilds only when the underlying graph changes. Pure read — signs nothing,
|
| 44 |
+
appends to no provenance chain (receipts belong on writes, never on GETs).
|
| 45 |
+
"""
|
| 46 |
+
|
| 47 |
+
import datetime
|
| 48 |
+
import hashlib
|
| 49 |
+
import json
|
| 50 |
+
import math
|
| 51 |
+
import os
|
| 52 |
+
import re
|
| 53 |
+
import urllib.error
|
| 54 |
+
import urllib.request
|
| 55 |
+
|
| 56 |
+
from fastapi import FastAPI
|
| 57 |
+
from fastapi.responses import JSONResponse
|
| 58 |
+
|
| 59 |
+
import a11oy_brain_graph as _brain
|
| 60 |
+
|
| 61 |
+
# --------------------------------------------------------------------------- #
|
| 62 |
+
# Best-available libraries — each import is guarded; the tier is labelled by
|
| 63 |
+
# what actually loaded, never by what we wish had loaded.
|
| 64 |
+
# --------------------------------------------------------------------------- #
|
| 65 |
+
try:
|
| 66 |
+
import networkx as _nx # graph algorithms (pagerank, ego, communities)
|
| 67 |
+
_HAVE_NX = True
|
| 68 |
+
except Exception: # pragma: no cover - networkx is a core dep, but stay honest
|
| 69 |
+
_nx = None
|
| 70 |
+
_HAVE_NX = False
|
| 71 |
+
|
| 72 |
+
try:
|
| 73 |
+
import numpy as _np
|
| 74 |
+
_HAVE_NUMPY = True
|
| 75 |
+
except Exception: # pragma: no cover
|
| 76 |
+
_np = None
|
| 77 |
+
_HAVE_NUMPY = False
|
| 78 |
+
|
| 79 |
+
try: # optional embedded vector index
|
| 80 |
+
import sqlite3
|
| 81 |
+
import sqlite_vec as _sqlite_vec
|
| 82 |
+
_HAVE_SQLITE_VEC = True
|
| 83 |
+
except Exception:
|
| 84 |
+
_sqlite_vec = None
|
| 85 |
+
_HAVE_SQLITE_VEC = False
|
| 86 |
+
|
| 87 |
+
try: # optional Leiden community detection
|
| 88 |
+
import igraph as _igraph
|
| 89 |
+
_HAVE_IGRAPH = True
|
| 90 |
+
except Exception:
|
| 91 |
+
_igraph = None
|
| 92 |
+
_HAVE_IGRAPH = False
|
| 93 |
+
|
| 94 |
+
# Honest Doctrine v11 labels (verbatim — never upgraded).
|
| 95 |
+
LBL_MODELED = "MODELED"
|
| 96 |
+
LBL_UNAVAILABLE = "UNAVAILABLE"
|
| 97 |
+
|
| 98 |
+
EMBED_DIM = 256
|
| 99 |
+
_TOKEN_RE = re.compile(r"[a-z0-9]+")
|
| 100 |
+
_PERSON_KINDS = _brain._PERSON_KINDS
|
| 101 |
+
|
| 102 |
+
|
| 103 |
+
def _tokens(*parts: str) -> list:
|
| 104 |
+
"""Lowercase alnum tokens (len>=2) from the given strings, in order."""
|
| 105 |
+
out = []
|
| 106 |
+
for p in parts:
|
| 107 |
+
for t in _TOKEN_RE.findall((p or "").lower()):
|
| 108 |
+
if len(t) >= 2:
|
| 109 |
+
out.append(t)
|
| 110 |
+
return out
|
| 111 |
+
|
| 112 |
+
|
| 113 |
+
# --------------------------------------------------------------------------- #
|
| 114 |
+
# Embeddings
|
| 115 |
+
# --------------------------------------------------------------------------- #
|
| 116 |
+
def _local_llm_url() -> str:
|
| 117 |
+
return (os.environ.get("SZL_LOCAL_LLM_URL")
|
| 118 |
+
or os.environ.get("OLLAMA_URL")
|
| 119 |
+
or "").rstrip("/")
|
| 120 |
+
|
| 121 |
+
|
| 122 |
+
def _ollama_embed(texts: list, url: str, model: str = "nomic-embed-text",
|
| 123 |
+
timeout: float = 2.0):
|
| 124 |
+
"""Embed via a local Ollama server. Returns list[list[float]] or raises."""
|
| 125 |
+
vecs = []
|
| 126 |
+
for text in texts:
|
| 127 |
+
body = json.dumps({"model": model, "prompt": text}).encode("utf-8")
|
| 128 |
+
req = urllib.request.Request(
|
| 129 |
+
f"{url}/api/embeddings", data=body,
|
| 130 |
+
headers={"Content-Type": "application/json"})
|
| 131 |
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
| 132 |
+
payload = json.loads(resp.read().decode("utf-8"))
|
| 133 |
+
emb = payload.get("embedding")
|
| 134 |
+
if not emb:
|
| 135 |
+
raise ValueError("ollama returned no embedding")
|
| 136 |
+
vecs.append([float(x) for x in emb])
|
| 137 |
+
return vecs
|
| 138 |
+
|
| 139 |
+
|
| 140 |
+
def _hash_embed_one(text: str, dim: int = EMBED_DIM) -> list:
|
| 141 |
+
"""Deterministic feature-hashed bag-of-tokens embedding (MODELED).
|
| 142 |
+
|
| 143 |
+
A hash-embedding similarity is NEVER a MEASURED semantic similarity — it is a
|
| 144 |
+
deterministic token-overlap proxy. Labelled MODELED everywhere it surfaces."""
|
| 145 |
+
vec = [0.0] * dim
|
| 146 |
+
for tok in _tokens(text):
|
| 147 |
+
h = int(hashlib.blake2b(tok.encode("utf-8"), digest_size=8).hexdigest(), 16)
|
| 148 |
+
idx = h % dim
|
| 149 |
+
sign = 1.0 if (h >> 63) & 1 else -1.0
|
| 150 |
+
vec[idx] += sign
|
| 151 |
+
return vec
|
| 152 |
+
|
| 153 |
+
|
| 154 |
+
def _l2_normalize(vec: list) -> list:
|
| 155 |
+
norm = math.sqrt(sum(x * x for x in vec))
|
| 156 |
+
if norm == 0.0:
|
| 157 |
+
return vec
|
| 158 |
+
return [x / norm for x in vec]
|
| 159 |
+
|
| 160 |
+
|
| 161 |
+
class _Embedder:
|
| 162 |
+
"""Chooses the best embedding source at build time and stays there."""
|
| 163 |
+
|
| 164 |
+
def __init__(self):
|
| 165 |
+
self.model = "nomic-embed-text"
|
| 166 |
+
url = _local_llm_url()
|
| 167 |
+
self.source = "hash-fallback"
|
| 168 |
+
self.tier = LBL_MODELED
|
| 169 |
+
self.dim = EMBED_DIM
|
| 170 |
+
self._use_ollama = False
|
| 171 |
+
if url:
|
| 172 |
+
try: # probe once with a trivial text
|
| 173 |
+
probe = _ollama_embed(["a11oy"], url, self.model, timeout=2.0)
|
| 174 |
+
self._url = url
|
| 175 |
+
self.dim = len(probe[0])
|
| 176 |
+
self.source = f"ollama:{self.model}"
|
| 177 |
+
self._use_ollama = True
|
| 178 |
+
except Exception:
|
| 179 |
+
self._use_ollama = False
|
| 180 |
+
|
| 181 |
+
def embed(self, texts: list) -> list:
|
| 182 |
+
if self._use_ollama:
|
| 183 |
+
try:
|
| 184 |
+
raw = _ollama_embed(texts, self._url, self.model, timeout=4.0)
|
| 185 |
+
return [_l2_normalize(v) for v in raw]
|
| 186 |
+
except Exception:
|
| 187 |
+
# A mid-flight failure downgrades honestly rather than erroring.
|
| 188 |
+
self._use_ollama = False
|
| 189 |
+
self.source = "hash-fallback"
|
| 190 |
+
return [_l2_normalize(_hash_embed_one(t, self.dim)) for t in texts]
|
| 191 |
+
|
| 192 |
+
|
| 193 |
+
def _cosine(a: list, b: list) -> float:
|
| 194 |
+
return sum(x * y for x, y in zip(a, b))
|
| 195 |
+
|
| 196 |
+
|
| 197 |
+
# --------------------------------------------------------------------------- #
|
| 198 |
+
# The queryable index over the honest graph.
|
| 199 |
+
# --------------------------------------------------------------------------- #
|
| 200 |
+
class BrainIndex:
|
| 201 |
+
"""A retrieval index built over a11oy_brain_graph.get_brain_graph(ns).
|
| 202 |
+
|
| 203 |
+
Holds: node table, a directed weighted graph (for PageRank), an undirected
|
| 204 |
+
graph (for neighbourhoods/communities), node embeddings + a vector backend,
|
| 205 |
+
and community assignments + summaries. Deterministic; keyed by content hash."""
|
| 206 |
+
|
| 207 |
+
def __init__(self, ns: str, graph: dict):
|
| 208 |
+
self.ns = ns
|
| 209 |
+
self.graph = graph
|
| 210 |
+
self.content_hash = _content_hash(graph)
|
| 211 |
+
self.nodes = graph.get("nodes", [])
|
| 212 |
+
self.links = graph.get("links", [])
|
| 213 |
+
self.by_id = {n["id"]: n for n in self.nodes}
|
| 214 |
+
self.ids = [n["id"] for n in self.nodes]
|
| 215 |
+
self._pos = {nid: i for i, nid in enumerate(self.ids)}
|
| 216 |
+
|
| 217 |
+
self._build_graphs()
|
| 218 |
+
self._build_embeddings()
|
| 219 |
+
self._build_communities()
|
| 220 |
+
self._pagerank_global = self._compute_pagerank(None)
|
| 221 |
+
|
| 222 |
+
# ---- graph structures -------------------------------------------------- #
|
| 223 |
+
def _build_graphs(self):
|
| 224 |
+
if _HAVE_NX:
|
| 225 |
+
dg = _nx.DiGraph()
|
| 226 |
+
dg.add_nodes_from(self.ids)
|
| 227 |
+
for l in self.links:
|
| 228 |
+
s, t = l["source"], l["target"]
|
| 229 |
+
if s in self.by_id and t in self.by_id:
|
| 230 |
+
w = dg.get_edge_data(s, t, {}).get("weight", 0.0) + 1.0
|
| 231 |
+
dg.add_edge(s, t, weight=w)
|
| 232 |
+
self.DG = dg
|
| 233 |
+
self.UG = dg.to_undirected(as_view=False)
|
| 234 |
+
else: # pragma: no cover - networkx is present in this estate
|
| 235 |
+
self.DG = None
|
| 236 |
+
self.UG = None
|
| 237 |
+
# Pure-python adjacency (used for neighbours if networkx is absent).
|
| 238 |
+
adj: dict = {nid: set() for nid in self.ids}
|
| 239 |
+
for l in self.links:
|
| 240 |
+
s, t = l["source"], l["target"]
|
| 241 |
+
if s in adj and t in adj:
|
| 242 |
+
adj[s].add(t)
|
| 243 |
+
adj[t].add(s)
|
| 244 |
+
self.adj = adj
|
| 245 |
+
|
| 246 |
+
# ---- embeddings + vector backend -------------------------------------- #
|
| 247 |
+
def _node_text(self, n: dict) -> str:
|
| 248 |
+
return " ".join(str(x) for x in (
|
| 249 |
+
n.get("kind", ""), n.get("title", ""), n.get("id", ""),
|
| 250 |
+
n.get("axis") or "", n.get("source") or "") if x)
|
| 251 |
+
|
| 252 |
+
def _build_embeddings(self):
|
| 253 |
+
self.embedder = _Embedder()
|
| 254 |
+
texts = [self._node_text(n) for n in self.nodes]
|
| 255 |
+
self.embeddings = self.embedder.embed(texts) if texts else []
|
| 256 |
+
self.embed_source = self.embedder.source
|
| 257 |
+
self.embed_tier = self.embedder.tier # always MODELED
|
| 258 |
+
self.embed_dim = self.embedder.dim
|
| 259 |
+
self._init_vector_backend()
|
| 260 |
+
|
| 261 |
+
def _init_vector_backend(self):
|
| 262 |
+
self.vector_backend = "python-cosine"
|
| 263 |
+
self._np_matrix = None
|
| 264 |
+
self._vec_db = None
|
| 265 |
+
if _HAVE_SQLITE_VEC and self.embeddings:
|
| 266 |
+
try:
|
| 267 |
+
db = sqlite3.connect(":memory:")
|
| 268 |
+
db.enable_load_extension(True)
|
| 269 |
+
_sqlite_vec.load(db)
|
| 270 |
+
db.enable_load_extension(False)
|
| 271 |
+
db.execute(
|
| 272 |
+
f"CREATE VIRTUAL TABLE vec USING vec0("
|
| 273 |
+
f"emb float[{self.embed_dim}])")
|
| 274 |
+
for i, emb in enumerate(self.embeddings):
|
| 275 |
+
db.execute("INSERT INTO vec(rowid, emb) VALUES (?, ?)",
|
| 276 |
+
(i, json.dumps(emb)))
|
| 277 |
+
db.commit()
|
| 278 |
+
self._vec_db = db
|
| 279 |
+
self.vector_backend = "sqlite-vec"
|
| 280 |
+
return
|
| 281 |
+
except Exception:
|
| 282 |
+
self._vec_db = None
|
| 283 |
+
if _HAVE_NUMPY and self.embeddings:
|
| 284 |
+
self._np_matrix = _np.asarray(self.embeddings, dtype="float32")
|
| 285 |
+
self.vector_backend = "numpy-cosine"
|
| 286 |
+
|
| 287 |
+
def _vector_scores(self, qvec: list) -> list:
|
| 288 |
+
"""Cosine of qvec against every node embedding (index-aligned)."""
|
| 289 |
+
if not self.embeddings:
|
| 290 |
+
return []
|
| 291 |
+
if self.vector_backend == "sqlite-vec" and self._vec_db is not None:
|
| 292 |
+
try:
|
| 293 |
+
rows = self._vec_db.execute(
|
| 294 |
+
"SELECT rowid, distance FROM vec "
|
| 295 |
+
"WHERE emb MATCH ? ORDER BY distance LIMIT ?",
|
| 296 |
+
(json.dumps(qvec), len(self.ids))).fetchall()
|
| 297 |
+
scores = [0.0] * len(self.ids)
|
| 298 |
+
for rowid, dist in rows:
|
| 299 |
+
# vec0 default is L2; embeddings are L2-normalised so
|
| 300 |
+
# cosine = 1 - dist^2 / 2.
|
| 301 |
+
scores[rowid] = 1.0 - (float(dist) ** 2) / 2.0
|
| 302 |
+
return scores
|
| 303 |
+
except Exception:
|
| 304 |
+
pass
|
| 305 |
+
if self.vector_backend == "numpy-cosine" and self._np_matrix is not None:
|
| 306 |
+
q = _np.asarray(qvec, dtype="float32")
|
| 307 |
+
return list(map(float, self._np_matrix @ q))
|
| 308 |
+
return [_cosine(qvec, e) for e in self.embeddings]
|
| 309 |
+
|
| 310 |
+
# ---- communities ------------------------------------------------------- #
|
| 311 |
+
def _build_communities(self):
|
| 312 |
+
self.community_of: dict = {}
|
| 313 |
+
self.communities: dict = {}
|
| 314 |
+
self.community_algo = "none"
|
| 315 |
+
if not self.ids:
|
| 316 |
+
return
|
| 317 |
+
parts = None
|
| 318 |
+
if _HAVE_IGRAPH and self.UG is not None:
|
| 319 |
+
try:
|
| 320 |
+
g = _igraph.Graph()
|
| 321 |
+
g.add_vertices(self.ids)
|
| 322 |
+
g.add_edges([(s, t) for s, t in self.UG.edges()])
|
| 323 |
+
clustering = g.community_leiden(
|
| 324 |
+
objective_function="modularity")
|
| 325 |
+
parts = [[g.vs[v]["name"] for v in comm] for comm in clustering]
|
| 326 |
+
self.community_algo = "igraph-leiden"
|
| 327 |
+
except Exception:
|
| 328 |
+
parts = None
|
| 329 |
+
if parts is None and _HAVE_NX and self.UG is not None:
|
| 330 |
+
try:
|
| 331 |
+
from networkx.algorithms.community import (
|
| 332 |
+
greedy_modularity_communities)
|
| 333 |
+
comms = greedy_modularity_communities(self.UG)
|
| 334 |
+
parts = [sorted(c) for c in comms]
|
| 335 |
+
self.community_algo = "networkx-greedy-modularity"
|
| 336 |
+
except Exception:
|
| 337 |
+
parts = None
|
| 338 |
+
if parts is None: # last resort: connected components
|
| 339 |
+
if _HAVE_NX and self.UG is not None:
|
| 340 |
+
parts = [sorted(c) for c in _nx.connected_components(self.UG)]
|
| 341 |
+
else:
|
| 342 |
+
parts = self._components_purepy()
|
| 343 |
+
self.community_algo = "connected-components"
|
| 344 |
+
|
| 345 |
+
parts = sorted(parts, key=lambda c: (-len(c), c[0] if c else ""))
|
| 346 |
+
for cidx, members in enumerate(parts):
|
| 347 |
+
cid = f"c{cidx}"
|
| 348 |
+
self.communities[cid] = members
|
| 349 |
+
for m in members:
|
| 350 |
+
self.community_of[m] = cid
|
| 351 |
+
self.community_summaries = {
|
| 352 |
+
cid: self._summarize_community(cid, members)
|
| 353 |
+
for cid, members in self.communities.items()}
|
| 354 |
+
|
| 355 |
+
def _components_purepy(self) -> list:
|
| 356 |
+
seen = set()
|
| 357 |
+
comps = []
|
| 358 |
+
for start in self.ids:
|
| 359 |
+
if start in seen:
|
| 360 |
+
continue
|
| 361 |
+
stack = [start]
|
| 362 |
+
comp = []
|
| 363 |
+
while stack:
|
| 364 |
+
x = stack.pop()
|
| 365 |
+
if x in seen:
|
| 366 |
+
continue
|
| 367 |
+
seen.add(x)
|
| 368 |
+
comp.append(x)
|
| 369 |
+
stack.extend(self.adj.get(x, ()))
|
| 370 |
+
comps.append(sorted(comp))
|
| 371 |
+
return comps
|
| 372 |
+
|
| 373 |
+
def _summarize_community(self, cid: str, members: list) -> dict:
|
| 374 |
+
kinds: dict = {}
|
| 375 |
+
for m in members:
|
| 376 |
+
n = self.by_id.get(m, {})
|
| 377 |
+
kinds[n.get("kind", "?")] = kinds.get(n.get("kind", "?"), 0) + 1
|
| 378 |
+
top = sorted(
|
| 379 |
+
members,
|
| 380 |
+
key=lambda m: (-(self.by_id.get(m, {}).get("degree", 0)), m))[:8]
|
| 381 |
+
top_nodes = [{"id": m,
|
| 382 |
+
"title": self.by_id.get(m, {}).get("title", m),
|
| 383 |
+
"kind": self.by_id.get(m, {}).get("kind"),
|
| 384 |
+
"degree": self.by_id.get(m, {}).get("degree", 0)}
|
| 385 |
+
for m in top]
|
| 386 |
+
labels = sorted({self.by_id.get(m, {}).get("title", "")
|
| 387 |
+
for m in top if self.by_id.get(m, {}).get("title")})[:5]
|
| 388 |
+
return {
|
| 389 |
+
"id": cid,
|
| 390 |
+
"label": LBL_MODELED,
|
| 391 |
+
"size": len(members),
|
| 392 |
+
"by_kind": dict(sorted(kinds.items(), key=lambda kv: (-kv[1], kv[0]))),
|
| 393 |
+
"top_nodes": top_nodes,
|
| 394 |
+
"summary": (f"community {cid}: {len(members)} nodes, "
|
| 395 |
+
f"dominant kinds {', '.join(list(kinds)[:3])}; "
|
| 396 |
+
f"anchors: {', '.join(labels)}"),
|
| 397 |
+
}
|
| 398 |
+
|
| 399 |
+
# ---- pagerank ---------------------------------------------------------- #
|
| 400 |
+
def _compute_pagerank(self, personalization) -> dict:
|
| 401 |
+
if _HAVE_NX and self.DG is not None and self.DG.number_of_nodes():
|
| 402 |
+
try:
|
| 403 |
+
return _nx.pagerank(self.DG, alpha=0.85,
|
| 404 |
+
personalization=personalization,
|
| 405 |
+
weight="weight")
|
| 406 |
+
except Exception:
|
| 407 |
+
pass
|
| 408 |
+
# Pure-python power iteration fallback on the undirected adjacency.
|
| 409 |
+
return self._pagerank_purepy(personalization)
|
| 410 |
+
|
| 411 |
+
def _pagerank_purepy(self, personalization, iters: int = 50,
|
| 412 |
+
damping: float = 0.85) -> dict:
|
| 413 |
+
n = len(self.ids)
|
| 414 |
+
if n == 0:
|
| 415 |
+
return {}
|
| 416 |
+
if personalization:
|
| 417 |
+
tot = sum(personalization.values()) or 1.0
|
| 418 |
+
teleport = {nid: personalization.get(nid, 0.0) / tot
|
| 419 |
+
for nid in self.ids}
|
| 420 |
+
else:
|
| 421 |
+
teleport = {nid: 1.0 / n for nid in self.ids}
|
| 422 |
+
rank = {nid: 1.0 / n for nid in self.ids}
|
| 423 |
+
for _ in range(iters):
|
| 424 |
+
nxt = {nid: (1.0 - damping) * teleport[nid] for nid in self.ids}
|
| 425 |
+
dangling = 0.0
|
| 426 |
+
for nid in self.ids:
|
| 427 |
+
nbrs = self.adj.get(nid, ())
|
| 428 |
+
if not nbrs:
|
| 429 |
+
dangling += rank[nid]
|
| 430 |
+
continue
|
| 431 |
+
share = damping * rank[nid] / len(nbrs)
|
| 432 |
+
for m in nbrs:
|
| 433 |
+
nxt[m] += share
|
| 434 |
+
if dangling:
|
| 435 |
+
for nid in self.ids:
|
| 436 |
+
nxt[nid] += damping * dangling * teleport[nid]
|
| 437 |
+
rank = nxt
|
| 438 |
+
return rank
|
| 439 |
+
|
| 440 |
+
# ---- retrieval primitives --------------------------------------------- #
|
| 441 |
+
def search(self, q: str, k: int = 10) -> list:
|
| 442 |
+
"""Hybrid exact(token/substring) + vector(cosine) top-k nodes."""
|
| 443 |
+
qtokens = set(_tokens(q))
|
| 444 |
+
qlower = (q or "").lower().strip()
|
| 445 |
+
qvec = self.embedder.embed([q])[0] if q else None
|
| 446 |
+
vscores = self._vector_scores(qvec) if qvec else [0.0] * len(self.ids)
|
| 447 |
+
results = []
|
| 448 |
+
for i, n in enumerate(self.nodes):
|
| 449 |
+
ntokens = set(_tokens(self._node_text(n)))
|
| 450 |
+
overlap = len(qtokens & ntokens)
|
| 451 |
+
exact = 0.0
|
| 452 |
+
if qtokens:
|
| 453 |
+
exact = overlap / max(1, len(qtokens))
|
| 454 |
+
substr = 0.0
|
| 455 |
+
if qlower and (qlower in str(n.get("title", "")).lower()
|
| 456 |
+
or qlower in str(n.get("id", "")).lower()):
|
| 457 |
+
substr = 1.0
|
| 458 |
+
vec = vscores[i] if i < len(vscores) else 0.0
|
| 459 |
+
score = 0.5 * exact + 0.35 * vec + 0.15 * substr
|
| 460 |
+
if score <= 0.0:
|
| 461 |
+
continue
|
| 462 |
+
results.append((score, exact, vec, substr, n))
|
| 463 |
+
results.sort(key=lambda r: (-r[0], r[4]["id"]))
|
| 464 |
+
out = []
|
| 465 |
+
for score, exact, vec, substr, n in results[:max(1, k)]:
|
| 466 |
+
out.append({
|
| 467 |
+
"id": n["id"], "title": n.get("title", n["id"]),
|
| 468 |
+
"kind": n.get("kind"), "layer": n.get("layer"),
|
| 469 |
+
"degree": n.get("degree", 0),
|
| 470 |
+
"node_label": n.get("label"),
|
| 471 |
+
"score": round(score, 6),
|
| 472 |
+
"match": {"exact_token_overlap": round(exact, 4),
|
| 473 |
+
"vector_cosine": round(vec, 6),
|
| 474 |
+
"substring": bool(substr)},
|
| 475 |
+
"community": self.community_of.get(n["id"]),
|
| 476 |
+
})
|
| 477 |
+
return out
|
| 478 |
+
|
| 479 |
+
def neighbors(self, nid: str, hops: int = 1) -> dict:
|
| 480 |
+
hops = max(1, min(hops, 4))
|
| 481 |
+
if nid not in self.by_id:
|
| 482 |
+
return {}
|
| 483 |
+
frontier = {nid}
|
| 484 |
+
seen = {nid}
|
| 485 |
+
for _ in range(hops):
|
| 486 |
+
nxt = set()
|
| 487 |
+
for x in frontier:
|
| 488 |
+
for m in self.adj.get(x, ()):
|
| 489 |
+
if m not in seen:
|
| 490 |
+
nxt.add(m)
|
| 491 |
+
seen |= nxt
|
| 492 |
+
frontier = nxt
|
| 493 |
+
if not frontier:
|
| 494 |
+
break
|
| 495 |
+
return self._induced(seen, center=nid, hops=hops)
|
| 496 |
+
|
| 497 |
+
def subgraph(self, ids: list) -> dict:
|
| 498 |
+
present = [i for i in ids if i in self.by_id]
|
| 499 |
+
return self._induced(set(present), requested=ids)
|
| 500 |
+
|
| 501 |
+
def _induced(self, id_set: set, **meta) -> dict:
|
| 502 |
+
nodes = [self._node_view(self.by_id[i]) for i in self.ids if i in id_set]
|
| 503 |
+
links = [dict(l) for l in self.links
|
| 504 |
+
if l["source"] in id_set and l["target"] in id_set]
|
| 505 |
+
out = {
|
| 506 |
+
"label": LBL_MODELED,
|
| 507 |
+
"node_count": len(nodes),
|
| 508 |
+
"link_count": len(links),
|
| 509 |
+
"nodes": nodes, "links": links,
|
| 510 |
+
}
|
| 511 |
+
out.update(meta)
|
| 512 |
+
return out
|
| 513 |
+
|
| 514 |
+
def _node_view(self, n: dict) -> dict:
|
| 515 |
+
v = {"id": n["id"], "title": n.get("title", n["id"]),
|
| 516 |
+
"kind": n.get("kind"), "layer": n.get("layer"),
|
| 517 |
+
"degree": n.get("degree", 0),
|
| 518 |
+
"salience": round(self._pagerank_global.get(n["id"], 0.0), 8),
|
| 519 |
+
"community": self.community_of.get(n["id"]),
|
| 520 |
+
"node_label": n.get("label")}
|
| 521 |
+
for opt in ("url", "source", "axis", "formula_id", "locked",
|
| 522 |
+
"conjecture", "proof_status"):
|
| 523 |
+
if n.get(opt) is not None:
|
| 524 |
+
v[opt] = n[opt]
|
| 525 |
+
return v
|
| 526 |
+
|
| 527 |
+
def salience(self, top: int = 25) -> list:
|
| 528 |
+
ranked = sorted(self._pagerank_global.items(),
|
| 529 |
+
key=lambda kv: (-kv[1], kv[0]))[:max(1, top)]
|
| 530 |
+
return [{"id": nid,
|
| 531 |
+
"title": self.by_id.get(nid, {}).get("title", nid),
|
| 532 |
+
"kind": self.by_id.get(nid, {}).get("kind"),
|
| 533 |
+
"salience": round(score, 8),
|
| 534 |
+
"degree": self.by_id.get(nid, {}).get("degree", 0),
|
| 535 |
+
"community": self.community_of.get(nid)}
|
| 536 |
+
for nid, score in ranked]
|
| 537 |
+
|
| 538 |
+
def community(self, nid: str) -> dict:
|
| 539 |
+
cid = self.community_of.get(nid)
|
| 540 |
+
if cid is None:
|
| 541 |
+
return {}
|
| 542 |
+
summ = dict(self.community_summaries.get(cid, {}))
|
| 543 |
+
summ["members"] = self.communities.get(cid, [])
|
| 544 |
+
summ["queried_node"] = nid
|
| 545 |
+
return summ
|
| 546 |
+
|
| 547 |
+
def ask(self, q: str, k: int = 12) -> dict:
|
| 548 |
+
"""HippoRAG-style PPR local retrieval ⊕ GraphRAG community context.
|
| 549 |
+
|
| 550 |
+
Returns a REAL grounding subgraph regardless. Generated prose is ONLY
|
| 551 |
+
produced if a sovereign model is reachable; otherwise it is honestly
|
| 552 |
+
UNAVAILABLE — never fabricated."""
|
| 553 |
+
seeds = self.search(q, k=max(5, k))
|
| 554 |
+
seed_ids = [s["id"] for s in seeds]
|
| 555 |
+
personalization = None
|
| 556 |
+
if seed_ids:
|
| 557 |
+
personalization = {nid: 0.0 for nid in self.ids}
|
| 558 |
+
for s in seeds:
|
| 559 |
+
personalization[s["id"]] = max(s["score"], 1e-6)
|
| 560 |
+
ppr = self._compute_pagerank(personalization)
|
| 561 |
+
ranked = sorted(ppr.items(), key=lambda kv: (-kv[1], kv[0]))
|
| 562 |
+
# Grounding node set: seeds + top PPR nodes (local retrieval).
|
| 563 |
+
ground_ids = list(dict.fromkeys(
|
| 564 |
+
seed_ids + [nid for nid, _ in ranked[:k]]))
|
| 565 |
+
grounding = self._induced(set(ground_ids))
|
| 566 |
+
for n in grounding["nodes"]:
|
| 567 |
+
n["ppr"] = round(ppr.get(n["id"], 0.0), 8)
|
| 568 |
+
grounding["nodes"].sort(key=lambda n: (-n.get("ppr", 0.0), n["id"]))
|
| 569 |
+
# Global retrieval: community summaries covering the grounding set.
|
| 570 |
+
cids = []
|
| 571 |
+
for nid in ground_ids:
|
| 572 |
+
cid = self.community_of.get(nid)
|
| 573 |
+
if cid and cid not in cids:
|
| 574 |
+
cids.append(cid)
|
| 575 |
+
global_ctx = [self.community_summaries[c] for c in cids[:5]
|
| 576 |
+
if c in self.community_summaries]
|
| 577 |
+
|
| 578 |
+
answer, answer_label, model = self._maybe_generate(q, grounding, global_ctx)
|
| 579 |
+
return {
|
| 580 |
+
"label": LBL_MODELED,
|
| 581 |
+
"query": q,
|
| 582 |
+
"retrieval": "hippoRAG-PPR(local) ⊕ graphRAG-community(global), "
|
| 583 |
+
"LightRAG-mix merge",
|
| 584 |
+
"seeds": seeds,
|
| 585 |
+
"grounding_subgraph": grounding,
|
| 586 |
+
"cited_node_ids": ground_ids,
|
| 587 |
+
"community_context": global_ctx,
|
| 588 |
+
"answer": answer,
|
| 589 |
+
"answer_label": answer_label,
|
| 590 |
+
"answer_model": model,
|
| 591 |
+
"note": ("grounding_subgraph is REAL (retrieved from the honest brain "
|
| 592 |
+
"graph) whether or not a sovereign model was reachable; "
|
| 593 |
+
"generated prose is UNAVAILABLE unless a local model answered."),
|
| 594 |
+
}
|
| 595 |
+
|
| 596 |
+
def _maybe_generate(self, q, grounding, global_ctx):
|
| 597 |
+
"""Return (answer, label, model). Never fabricates — no model => None."""
|
| 598 |
+
url = _local_llm_url()
|
| 599 |
+
model = os.environ.get("SZL_LOCAL_LLM_MODEL", "").strip()
|
| 600 |
+
if not url or not model:
|
| 601 |
+
return None, LBL_UNAVAILABLE, None
|
| 602 |
+
cited = ", ".join(n["id"] for n in grounding["nodes"][:12])
|
| 603 |
+
ctx_lines = [f"- {n['id']}: {n.get('title')}"
|
| 604 |
+
for n in grounding["nodes"][:12]]
|
| 605 |
+
ctx = "\n".join(ctx_lines)
|
| 606 |
+
prompt = (
|
| 607 |
+
"Answer ONLY from the grounding nodes below. Cite node ids you use. "
|
| 608 |
+
"If they do not contain the answer, say so.\n\n"
|
| 609 |
+
f"Question: {q}\n\nGrounding nodes:\n{ctx}\n\n"
|
| 610 |
+
f"(available node ids: {cited})")
|
| 611 |
+
try:
|
| 612 |
+
body = json.dumps({"model": model, "prompt": prompt,
|
| 613 |
+
"stream": False}).encode("utf-8")
|
| 614 |
+
req = urllib.request.Request(
|
| 615 |
+
f"{url}/api/generate", data=body,
|
| 616 |
+
headers={"Content-Type": "application/json"})
|
| 617 |
+
with urllib.request.urlopen(req, timeout=20.0) as resp:
|
| 618 |
+
payload = json.loads(resp.read().decode("utf-8"))
|
| 619 |
+
text = (payload.get("response") or "").strip()
|
| 620 |
+
if not text:
|
| 621 |
+
return None, LBL_UNAVAILABLE, None
|
| 622 |
+
# A locally-generated, graph-grounded answer is MODELED, never
|
| 623 |
+
# MEASURED — it is a model's prose over a real subgraph.
|
| 624 |
+
return text, LBL_MODELED, f"{url.split('//')[-1]}:{model}"
|
| 625 |
+
except Exception:
|
| 626 |
+
return None, LBL_UNAVAILABLE, None
|
| 627 |
+
|
| 628 |
+
def stats(self) -> dict:
|
| 629 |
+
g = self.graph
|
| 630 |
+
comm_sizes = sorted((len(m) for m in self.communities.values()),
|
| 631 |
+
reverse=True)
|
| 632 |
+
return {
|
| 633 |
+
"label": LBL_MODELED,
|
| 634 |
+
"ns": self.ns,
|
| 635 |
+
"content_hash": self.content_hash,
|
| 636 |
+
"node_count": g.get("node_count", len(self.nodes)),
|
| 637 |
+
"link_count": g.get("link_count", len(self.links)),
|
| 638 |
+
"distinct_artifacts": g.get("distinct_artifacts"),
|
| 639 |
+
"person_node_count": g.get("person_node_count"),
|
| 640 |
+
"artifact_note": g.get("artifact_note"),
|
| 641 |
+
"community_count": len(self.communities),
|
| 642 |
+
"community_algo": self.community_algo,
|
| 643 |
+
"largest_communities": comm_sizes[:10],
|
| 644 |
+
"by_kind": g.get("summary", {}).get("by_kind"),
|
| 645 |
+
"by_layer": g.get("summary", {}).get("by_layer"),
|
| 646 |
+
"index": self.index_status(),
|
| 647 |
+
"note": ("counts are the honest totals from a11oy_brain_graph "
|
| 648 |
+
"(reused, never restated). distinct_artifacts excludes arXiv "
|
| 649 |
+
"co-author person nodes; never present the raw total as all "
|
| 650 |
+
"distinct work."),
|
| 651 |
+
}
|
| 652 |
+
|
| 653 |
+
def index_status(self) -> dict:
|
| 654 |
+
return {
|
| 655 |
+
"label": LBL_MODELED,
|
| 656 |
+
"content_hash": self.content_hash,
|
| 657 |
+
"node_count": len(self.nodes),
|
| 658 |
+
"link_count": len(self.links),
|
| 659 |
+
"embed_source": self.embed_source,
|
| 660 |
+
"embed_tier": self.embed_tier,
|
| 661 |
+
"embed_dim": self.embed_dim,
|
| 662 |
+
"vector_backend": self.vector_backend,
|
| 663 |
+
"community_algo": self.community_algo,
|
| 664 |
+
"community_count": len(self.communities),
|
| 665 |
+
"pagerank": "networkx" if (_HAVE_NX and self.DG is not None)
|
| 666 |
+
else "pure-python",
|
| 667 |
+
"stack": {
|
| 668 |
+
"networkx": _HAVE_NX,
|
| 669 |
+
"numpy": _HAVE_NUMPY,
|
| 670 |
+
"sqlite_vec": _HAVE_SQLITE_VEC,
|
| 671 |
+
"igraph": _HAVE_IGRAPH,
|
| 672 |
+
"ollama_embeddings": self.embed_source.startswith("ollama"),
|
| 673 |
+
},
|
| 674 |
+
"note": ("hash-embedding similarity is MODELED (a deterministic "
|
| 675 |
+
"token-overlap proxy), NEVER MEASURED."),
|
| 676 |
+
}
|
| 677 |
+
|
| 678 |
+
|
| 679 |
+
def _content_hash(graph: dict) -> str:
|
| 680 |
+
h = hashlib.sha256()
|
| 681 |
+
for n in graph.get("nodes", []):
|
| 682 |
+
h.update(n["id"].encode("utf-8"))
|
| 683 |
+
h.update(b"\x00")
|
| 684 |
+
h.update(b"||links||")
|
| 685 |
+
for l in graph.get("links", []):
|
| 686 |
+
h.update(f"{l['source']}>{l['target']}:{l.get('rel','')}".encode("utf-8"))
|
| 687 |
+
h.update(b"\x00")
|
| 688 |
+
return h.hexdigest()[:16]
|
| 689 |
+
|
| 690 |
+
|
| 691 |
+
# --------------------------------------------------------------------------- #
|
| 692 |
+
# Cache — one index per (ns, content_hash); rebuilt only when the graph changes.
|
| 693 |
+
# --------------------------------------------------------------------------- #
|
| 694 |
+
_INDEX_CACHE: dict = {}
|
| 695 |
+
|
| 696 |
+
|
| 697 |
+
def get_index(ns: str = "a11oy", *, refresh: bool = False) -> BrainIndex:
|
| 698 |
+
graph = _brain.get_brain_graph(ns, refresh=refresh)
|
| 699 |
+
chash = _content_hash(graph)
|
| 700 |
+
cached = _INDEX_CACHE.get(ns)
|
| 701 |
+
if cached is None or cached.content_hash != chash or refresh:
|
| 702 |
+
_INDEX_CACHE[ns] = BrainIndex(ns, graph)
|
| 703 |
+
return _INDEX_CACHE[ns]
|
| 704 |
+
|
| 705 |
+
|
| 706 |
+
# --------------------------------------------------------------------------- #
|
| 707 |
+
# FastAPI registration — additive, before the SPA catch-all. Pure reads.
|
| 708 |
+
# --------------------------------------------------------------------------- #
|
| 709 |
+
def register(app: FastAPI, ns: str = "a11oy") -> str:
|
| 710 |
+
base = f"/api/{ns}/v1/brain"
|
| 711 |
+
|
| 712 |
+
@app.get(f"{base}/search")
|
| 713 |
+
async def brain_search(q: str = "", k: int = 10): # noqa: ANN202
|
| 714 |
+
idx = get_index(ns)
|
| 715 |
+
return JSONResponse({"label": LBL_MODELED, "query": q, "k": k,
|
| 716 |
+
"results": idx.search(q, k),
|
| 717 |
+
"index": idx.index_status()})
|
| 718 |
+
|
| 719 |
+
@app.get(f"{base}/neighbors")
|
| 720 |
+
async def brain_neighbors(id: str = "", hops: int = 1): # noqa: ANN202,A002
|
| 721 |
+
idx = get_index(ns)
|
| 722 |
+
sub = idx.neighbors(id, hops)
|
| 723 |
+
if not sub:
|
| 724 |
+
return JSONResponse(
|
| 725 |
+
{"label": LBL_UNAVAILABLE, "error": f"unknown node id: {id!r}",
|
| 726 |
+
"id": id}, status_code=404)
|
| 727 |
+
return JSONResponse(sub)
|
| 728 |
+
|
| 729 |
+
@app.get(f"{base}/community")
|
| 730 |
+
async def brain_community(id: str = ""): # noqa: ANN202,A002
|
| 731 |
+
idx = get_index(ns)
|
| 732 |
+
c = idx.community(id)
|
| 733 |
+
if not c:
|
| 734 |
+
return JSONResponse(
|
| 735 |
+
{"label": LBL_UNAVAILABLE, "error": f"unknown node id: {id!r}",
|
| 736 |
+
"id": id}, status_code=404)
|
| 737 |
+
return JSONResponse(c)
|
| 738 |
+
|
| 739 |
+
@app.get(f"{base}/subgraph")
|
| 740 |
+
async def brain_subgraph(ids: str = ""): # noqa: ANN202
|
| 741 |
+
idx = get_index(ns)
|
| 742 |
+
id_list = [x for x in re.split(r"[,\s]+", ids.strip()) if x]
|
| 743 |
+
return JSONResponse(idx.subgraph(id_list))
|
| 744 |
+
|
| 745 |
+
@app.get(f"{base}/salience")
|
| 746 |
+
async def brain_salience(top: int = 25): # noqa: ANN202
|
| 747 |
+
idx = get_index(ns)
|
| 748 |
+
return JSONResponse({"label": LBL_MODELED, "top": top,
|
| 749 |
+
"ranking": idx.salience(top),
|
| 750 |
+
"method": "PageRank (α=0.85)"})
|
| 751 |
+
|
| 752 |
+
@app.get(f"{base}/ask")
|
| 753 |
+
async def brain_ask(q: str = "", k: int = 12): # noqa: ANN202
|
| 754 |
+
idx = get_index(ns)
|
| 755 |
+
return JSONResponse(idx.ask(q, k))
|
| 756 |
+
|
| 757 |
+
@app.get(f"{base}/stats")
|
| 758 |
+
async def brain_stats(): # noqa: ANN202
|
| 759 |
+
return JSONResponse(get_index(ns).stats())
|
| 760 |
+
|
| 761 |
+
@app.get(f"{base}/index")
|
| 762 |
+
async def brain_index(): # noqa: ANN202
|
| 763 |
+
return JSONResponse(get_index(ns).index_status())
|
| 764 |
+
|
| 765 |
+
idx = get_index(ns)
|
| 766 |
+
st = idx.index_status()
|
| 767 |
+
return (f"brain-api mounted: GET {base}/"
|
| 768 |
+
f"{{search,neighbors,community,subgraph,salience,ask,stats,index}} "
|
| 769 |
+
f"({st['node_count']} nodes, {st['community_count']} communities via "
|
| 770 |
+
f"{st['community_algo']}; vectors={st['vector_backend']}, "
|
| 771 |
+
f"embeddings={st['embed_source']} [{st['embed_tier']}]; "
|
| 772 |
+
f"0 sign-on-read)")
|
| 773 |
+
|
| 774 |
+
|
| 775 |
+
# --------------------------------------------------------------------------- #
|
| 776 |
+
# Self-test — runs against the REAL reused graph.
|
| 777 |
+
# --------------------------------------------------------------------------- #
|
| 778 |
+
def _selftest() -> None:
|
| 779 |
+
idx = get_index("a11oy")
|
| 780 |
+
assert idx.nodes, "index must reuse a non-empty brain graph"
|
| 781 |
+
assert idx.content_hash and len(idx.content_hash) == 16, "content hash"
|
| 782 |
+
|
| 783 |
+
st = idx.index_status()
|
| 784 |
+
assert st["embed_tier"] == LBL_MODELED, "embeddings are MODELED, never MEASURED"
|
| 785 |
+
assert st["vector_backend"] in (
|
| 786 |
+
"sqlite-vec", "numpy-cosine", "python-cosine"), st["vector_backend"]
|
| 787 |
+
assert st["community_count"] >= 1, "at least one community"
|
| 788 |
+
|
| 789 |
+
# search: a token that exists in the estate graph.
|
| 790 |
+
res = idx.search("brain graph", k=5)
|
| 791 |
+
assert res, "search must return results for an estate term"
|
| 792 |
+
assert all("score" in r and "match" in r for r in res), "honest scoring"
|
| 793 |
+
assert res == sorted(res, key=lambda r: (-r["score"], r["id"])), "sorted"
|
| 794 |
+
|
| 795 |
+
# neighbors: pick the highest-degree node and expand 1 hop.
|
| 796 |
+
hub = max(idx.nodes, key=lambda n: n.get("degree", 0))["id"]
|
| 797 |
+
nb = idx.neighbors(hub, hops=1)
|
| 798 |
+
assert nb["node_count"] >= 1, "hub must have a neighbourhood"
|
| 799 |
+
ids = {n["id"] for n in nb["nodes"]}
|
| 800 |
+
assert all(l["source"] in ids and l["target"] in ids
|
| 801 |
+
for l in nb["links"]), "induced links stay within the subgraph"
|
| 802 |
+
|
| 803 |
+
# community lookup + summary.
|
| 804 |
+
c = idx.community(hub)
|
| 805 |
+
assert c and c["size"] >= 1 and c["label"] == LBL_MODELED, "community summary"
|
| 806 |
+
|
| 807 |
+
# subgraph over explicit ids.
|
| 808 |
+
some = [n["id"] for n in idx.nodes[:5]]
|
| 809 |
+
sg = idx.subgraph(some)
|
| 810 |
+
assert sg["node_count"] == len(some), "induced subgraph over requested ids"
|
| 811 |
+
|
| 812 |
+
# salience: PageRank ranking is sorted and sums≈1 over all nodes.
|
| 813 |
+
sal = idx.salience(top=10)
|
| 814 |
+
assert sal and sal == sorted(sal, key=lambda r: (-r["salience"], r["id"]))
|
| 815 |
+
tot = sum(idx._pagerank_global.values())
|
| 816 |
+
assert abs(tot - 1.0) < 1e-3, f"pagerank must sum to ~1, got {tot}"
|
| 817 |
+
|
| 818 |
+
# ask: grounding subgraph is REAL; prose UNAVAILABLE without a model.
|
| 819 |
+
a = idx.ask("what proves the estate thesis", k=8)
|
| 820 |
+
assert a["grounding_subgraph"]["node_count"] >= 1, "real grounding subgraph"
|
| 821 |
+
assert a["cited_node_ids"], "cited node ids present"
|
| 822 |
+
if a["answer_model"] is None:
|
| 823 |
+
assert a["answer"] is None and a["answer_label"] == LBL_UNAVAILABLE, \
|
| 824 |
+
"no model => UNAVAILABLE, never a fabricated answer"
|
| 825 |
+
|
| 826 |
+
# stats: honest distinct-vs-total framing reused from the builder.
|
| 827 |
+
s = idx.stats()
|
| 828 |
+
assert s["node_count"] == idx.graph["node_count"], "reuse builder node_count"
|
| 829 |
+
assert s["distinct_artifacts"] == idx.graph["distinct_artifacts"], \
|
| 830 |
+
"reuse builder distinct_artifacts (never restated)"
|
| 831 |
+
assert s["community_count"] == len(idx.communities)
|
| 832 |
+
|
| 833 |
+
# cache: same content hash => same object (no needless rebuild).
|
| 834 |
+
assert get_index("a11oy") is idx, "index cached by content hash"
|
| 835 |
+
|
| 836 |
+
print(f"szl_brain_api: ALL OK — {st['node_count']} nodes, "
|
| 837 |
+
f"{st['community_count']} communities via {st['community_algo']}; "
|
| 838 |
+
f"vectors={st['vector_backend']}, embeddings={st['embed_source']} "
|
| 839 |
+
f"[{st['embed_tier']}]; pagerank sum={tot:.6f}; "
|
| 840 |
+
f"content_hash={idx.content_hash}")
|
| 841 |
+
|
| 842 |
+
|
| 843 |
+
if __name__ == "__main__":
|
| 844 |
+
_selftest()
|