Spaces:
Running
Running
chore(sync): mirror backend .py + Dockerfile to Space (hf-sync-backend)
Browse filesAutomated backend sync from szl-holdings/a11oy main via hf-sync-backend.
Updated (differed from the Space): szl_energy_live.py, szl_governed_api.py
Deleted (gone from the repo + Dockerfile COPY set): (none)
Keeps the Space-built backend (serve.py + the Dockerfile-COPY'd .py
modules) identical to GitHub main so the Space never rebuilds from a
stale backend, new endpoints don't 404 there, and orphaned modules
removed from the repo don't linger in the Space tree.
- szl_energy_live.py +124 -11
- szl_governed_api.py +4 -2
szl_energy_live.py
CHANGED
|
@@ -266,6 +266,10 @@ def govern_posture(deadline_s: float = MESH_PROBE_DEADLINE_S) -> dict:
|
|
| 266 |
"role": _role_for_engine(eng, i),
|
| 267 |
"model": eng.get("model"),
|
| 268 |
"gpu_model": _gpu_model_token(eng.get("name") or ""),
|
|
|
|
|
|
|
|
|
|
|
|
|
| 269 |
"is_glm": bool(eng.get("is_glm")),
|
| 270 |
"live": live_by_idx.get(i),
|
| 271 |
})
|
|
@@ -339,15 +343,54 @@ def build_live() -> dict:
|
|
| 339 |
# ---------------------------------------------------------------------------
|
| 340 |
# /energy/mesh — per-node energy + governance posture for the 3D view.
|
| 341 |
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 342 |
def build_mesh() -> dict:
|
| 343 |
snap = meter_snapshot()
|
| 344 |
reachable = bool(snap.get("reachable"))
|
| 345 |
posture = govern_posture()
|
| 346 |
gpus = snap.get("gpus") or [] if reachable else []
|
|
|
|
|
|
|
|
|
|
|
|
|
| 347 |
|
| 348 |
-
#
|
| 349 |
# node "…RTX 4060 Ti…" gets the meter GPU whose name label contains "RTX 4060 Ti").
|
| 350 |
-
#
|
|
|
|
| 351 |
def _match_gpu(gpu_model):
|
| 352 |
if not gpu_model:
|
| 353 |
return None
|
|
@@ -360,9 +403,19 @@ def build_mesh() -> dict:
|
|
| 360 |
nodes = []
|
| 361 |
watt_vals = []
|
| 362 |
for n in (posture.get("nodes") or []):
|
| 363 |
-
|
| 364 |
-
|
| 365 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 366 |
if isinstance(watts, (int, float)):
|
| 367 |
watt_vals.append(watts)
|
| 368 |
nodes.append({
|
|
@@ -372,7 +425,7 @@ def build_mesh() -> dict:
|
|
| 372 |
"watts": watts,
|
| 373 |
"joules": joules,
|
| 374 |
"joules_label": LABEL_MEASURED if isinstance(joules, (int, float)) else LABEL_UNAVAILABLE,
|
| 375 |
-
"source":
|
| 376 |
})
|
| 377 |
|
| 378 |
# Normalized 0..1 draw for visualization (relative to the busiest live node this tick).
|
|
@@ -381,21 +434,31 @@ def build_mesh() -> dict:
|
|
| 381 |
w = node["watts"]
|
| 382 |
node["draw"] = (round(w / max_w, 6) if (isinstance(w, (int, float)) and max_w > 0) else None)
|
| 383 |
|
| 384 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 385 |
return {
|
| 386 |
"ts": _now_iso(),
|
| 387 |
"label": label,
|
| 388 |
"nodes": nodes,
|
| 389 |
"node_count": len(nodes),
|
| 390 |
"live_count": posture.get("live_count", 0),
|
| 391 |
-
"total_watts":
|
| 392 |
-
"total_joules":
|
| 393 |
"joules_label": label,
|
| 394 |
"meter_url": METER_URL,
|
| 395 |
"meter_status": snap.get("status"),
|
| 396 |
"draw_basis": "watts normalized 0..1 vs the busiest live node this tick (null when no live watts)",
|
| 397 |
-
"note": ("per-node watts/joules attributed by
|
| 398 |
-
"
|
|
|
|
| 399 |
"doctrine": "v11 — honest empty-states; joules MEASURED only with a real exporter reading.",
|
| 400 |
}
|
| 401 |
|
|
@@ -784,6 +847,56 @@ def _selftest() -> dict:
|
|
| 784 |
assert mesh["label"] in (LABEL_MEASURED, LABEL_UNAVAILABLE)
|
| 785 |
out["mesh_shape"] = True
|
| 786 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 787 |
out["ok"] = all(v is True for v in out.values())
|
| 788 |
return out
|
| 789 |
|
|
|
|
| 266 |
"role": _role_for_engine(eng, i),
|
| 267 |
"model": eng.get("model"),
|
| 268 |
"gpu_model": _gpu_model_token(eng.get("name") or ""),
|
| 269 |
+
# The NVML joule-meter engine label this node reports (e.g. 'omen' for the
|
| 270 |
+
# tower, 'betterwithage' for the laptop) — used to attribute live watts/joules
|
| 271 |
+
# from the MERGED multi-meter scrape by engine name (see build_mesh).
|
| 272 |
+
"exporter": eng.get("exporter"),
|
| 273 |
"is_glm": bool(eng.get("is_glm")),
|
| 274 |
"live": live_by_idx.get(i),
|
| 275 |
})
|
|
|
|
| 343 |
# ---------------------------------------------------------------------------
|
| 344 |
# /energy/mesh — per-node energy + governance posture for the 3D view.
|
| 345 |
# ---------------------------------------------------------------------------
|
| 346 |
+
def _merged_engine_readings() -> dict:
|
| 347 |
+
"""Per-engine live watts + cumulative joules from the MERGED multi-meter scrape.
|
| 348 |
+
|
| 349 |
+
Reuses szl_energy_operator._fetch_joule_meter(), which scrapes EVERY meter in
|
| 350 |
+
A11OY_JOULE_METER_URLS (e.g. the tower's meter.a-11-oy.com AND the laptop's
|
| 351 |
+
meter2.a-11-oy.com) and merges their engines[] into one dict. Returns a map keyed
|
| 352 |
+
by lower-cased engine name ('omen', 'betterwithage') → {watts, joules}, so a mesh
|
| 353 |
+
node can be attributed its OWN engine's live NVML numbers the same way the operator
|
| 354 |
+
does (_exporter_sample_for_node). Honest (Doctrine v11): an unreachable meter simply
|
| 355 |
+
contributes no engine, so that node stays null/UNAVAILABLE — never fabricated; watts
|
| 356 |
+
are read only from a GPU flagged live, else None."""
|
| 357 |
+
try:
|
| 358 |
+
import szl_energy_operator as _op
|
| 359 |
+
meter = _op._fetch_joule_meter()
|
| 360 |
+
except Exception: # noqa: BLE001 — operator/meter absent => empty map, honest fallback
|
| 361 |
+
return {}
|
| 362 |
+
out: dict = {}
|
| 363 |
+
for e in (meter or {}).get("engines", []) or []:
|
| 364 |
+
name = str(e.get("engine") or "").strip().lower()
|
| 365 |
+
if not name:
|
| 366 |
+
continue
|
| 367 |
+
watts = None
|
| 368 |
+
for g in (e.get("gpus") or []):
|
| 369 |
+
if g.get("live") and isinstance(g.get("power_w"), (int, float)):
|
| 370 |
+
watts = float(g["power_w"])
|
| 371 |
+
break
|
| 372 |
+
joules = e.get("joules")
|
| 373 |
+
out[name] = {
|
| 374 |
+
"watts": watts,
|
| 375 |
+
"joules": float(joules) if isinstance(joules, (int, float)) else None,
|
| 376 |
+
}
|
| 377 |
+
return out
|
| 378 |
+
|
| 379 |
+
|
| 380 |
def build_mesh() -> dict:
|
| 381 |
snap = meter_snapshot()
|
| 382 |
reachable = bool(snap.get("reachable"))
|
| 383 |
posture = govern_posture()
|
| 384 |
gpus = snap.get("gpus") or [] if reachable else []
|
| 385 |
+
# Live NVML per-engine readings from the MERGED multi-meter scrape (both the tower's
|
| 386 |
+
# and the laptop's meters). This is what lets the laptop node pull its own engine's
|
| 387 |
+
# watts/joules instead of showing null when its meter is a SEPARATE URL from METER_URL.
|
| 388 |
+
engine_map = _merged_engine_readings()
|
| 389 |
|
| 390 |
+
# Fallback attribution: match a meter GPU to a mesh node by GPU model token (e.g. the
|
| 391 |
# node "…RTX 4060 Ti…" gets the meter GPU whose name label contains "RTX 4060 Ti").
|
| 392 |
+
# Only used when the node has no engine reading. A node with neither keeps null
|
| 393 |
+
# watts/joules (UNAVAILABLE), never a guess.
|
| 394 |
def _match_gpu(gpu_model):
|
| 395 |
if not gpu_model:
|
| 396 |
return None
|
|
|
|
| 403 |
nodes = []
|
| 404 |
watt_vals = []
|
| 405 |
for n in (posture.get("nodes") or []):
|
| 406 |
+
# Primary: attribute by the node's own NVML engine name from the merged meters
|
| 407 |
+
# ('omen' → tower, 'betterwithage' → laptop). Same mapping the operator uses.
|
| 408 |
+
exporter = str(n.get("exporter") or "").strip().lower()
|
| 409 |
+
reading = engine_map.get(exporter) if exporter else None
|
| 410 |
+
if reading is not None:
|
| 411 |
+
watts = reading.get("watts")
|
| 412 |
+
joules = reading.get("joules")
|
| 413 |
+
source = "NVML"
|
| 414 |
+
else:
|
| 415 |
+
g = _match_gpu(n.get("gpu_model"))
|
| 416 |
+
watts = g.get("watts") if g else None
|
| 417 |
+
joules = g.get("joules") if g else None
|
| 418 |
+
source = "NVML" if g else "mesh-posture"
|
| 419 |
if isinstance(watts, (int, float)):
|
| 420 |
watt_vals.append(watts)
|
| 421 |
nodes.append({
|
|
|
|
| 425 |
"watts": watts,
|
| 426 |
"joules": joules,
|
| 427 |
"joules_label": LABEL_MEASURED if isinstance(joules, (int, float)) else LABEL_UNAVAILABLE,
|
| 428 |
+
"source": source,
|
| 429 |
})
|
| 430 |
|
| 431 |
# Normalized 0..1 draw for visualization (relative to the busiest live node this tick).
|
|
|
|
| 434 |
w = node["watts"]
|
| 435 |
node["draw"] = (round(w / max_w, 6) if (isinstance(w, (int, float)) and max_w > 0) else None)
|
| 436 |
|
| 437 |
+
# Totals from the per-node attributed readings (honest sum of what we actually
|
| 438 |
+
# metered this tick); fall back to the single-meter snapshot when nothing attributed.
|
| 439 |
+
node_watts = [nd["watts"] for nd in nodes if isinstance(nd["watts"], (int, float))]
|
| 440 |
+
node_joules = [nd["joules"] for nd in nodes if isinstance(nd["joules"], (int, float))]
|
| 441 |
+
any_reading = bool(node_watts) or bool(node_joules) or reachable
|
| 442 |
+
total_watts = (round(sum(node_watts), 6) if node_watts
|
| 443 |
+
else (snap.get("total_watts") if reachable else None))
|
| 444 |
+
total_joules = (round(sum(node_joules), 6) if node_joules
|
| 445 |
+
else (snap.get("total_joules") if reachable else None))
|
| 446 |
+
label = LABEL_MEASURED if any_reading else LABEL_UNAVAILABLE
|
| 447 |
return {
|
| 448 |
"ts": _now_iso(),
|
| 449 |
"label": label,
|
| 450 |
"nodes": nodes,
|
| 451 |
"node_count": len(nodes),
|
| 452 |
"live_count": posture.get("live_count", 0),
|
| 453 |
+
"total_watts": total_watts,
|
| 454 |
+
"total_joules": total_joules,
|
| 455 |
"joules_label": label,
|
| 456 |
"meter_url": METER_URL,
|
| 457 |
"meter_status": snap.get("status"),
|
| 458 |
"draw_basis": "watts normalized 0..1 vs the busiest live node this tick (null when no live watts)",
|
| 459 |
+
"note": ("per-node watts/joules attributed by NVML engine name from the MERGED "
|
| 460 |
+
"multi-meter scrape ('omen' → tower, 'betterwithage' → laptop), GPU-model "
|
| 461 |
+
"match as fallback; unmatched nodes are UNAVAILABLE, never fabricated"),
|
| 462 |
"doctrine": "v11 — honest empty-states; joules MEASURED only with a real exporter reading.",
|
| 463 |
}
|
| 464 |
|
|
|
|
| 847 |
assert mesh["label"] in (LABEL_MEASURED, LABEL_UNAVAILABLE)
|
| 848 |
out["mesh_shape"] = True
|
| 849 |
|
| 850 |
+
# (f) Mesh per-node attribution by NVML engine name from the MERGED multi-meter map:
|
| 851 |
+
# the laptop node ('betterwithage') pulls its OWN live watts/joules exactly like the
|
| 852 |
+
# tower node ('omen'). When its engine is absent (meter2 down) it stays null —
|
| 853 |
+
# UNAVAILABLE, never fabricated. This is the regression this fix guards.
|
| 854 |
+
_lap = "Sovereign GPU 1 (laptop · RTX 5050 · Blackwell)"
|
| 855 |
+
_tow = "Sovereign GPU 2 (tower · RTX 4060 Ti · anchor)"
|
| 856 |
+
_prev_posture = globals()["govern_posture"]
|
| 857 |
+
_prev_readings = globals()["_merged_engine_readings"]
|
| 858 |
+
try:
|
| 859 |
+
# Deterministic posture (no network) with both nodes' engine labels wired.
|
| 860 |
+
globals()["govern_posture"] = lambda *a, **k: {
|
| 861 |
+
"available": True, "live_count": 2, "nodes": [
|
| 862 |
+
{"name": _tow, "role": "anchor", "gpu_model": "RTX 4060 Ti",
|
| 863 |
+
"exporter": "omen", "is_glm": False, "live": True},
|
| 864 |
+
{"name": _lap, "role": "blackwell", "gpu_model": "RTX 5050",
|
| 865 |
+
"exporter": "betterwithage", "is_glm": False, "live": True},
|
| 866 |
+
]}
|
| 867 |
+
# Reachable-but-empty single meter so the GPU-model fallback yields nothing
|
| 868 |
+
# (proves attribution comes from the engine map, and null is honest when absent).
|
| 869 |
+
with _snap_lock:
|
| 870 |
+
_snap_cache["data"] = {"gpus": [], "total_watts": 0.0, "total_joules": None,
|
| 871 |
+
"reachable": True, "status": "ok"}
|
| 872 |
+
_snap_cache["ts"] = time.time()
|
| 873 |
+
|
| 874 |
+
# Both engines present in the merged meters → both nodes MEASURED from their own engine.
|
| 875 |
+
globals()["_merged_engine_readings"] = lambda: {
|
| 876 |
+
"omen": {"watts": 13.53, "joules": 15896.0},
|
| 877 |
+
"betterwithage": {"watts": 8.0, "joules": 37000.0},
|
| 878 |
+
}
|
| 879 |
+
m2 = build_mesh()
|
| 880 |
+
by = {nd["name"]: nd for nd in m2["nodes"]}
|
| 881 |
+
assert by[_lap]["watts"] == 8.0 and by[_lap]["joules"] == 37000.0, by[_lap]
|
| 882 |
+
assert by[_lap]["joules_label"] == LABEL_MEASURED and by[_lap]["source"] == "NVML", by[_lap]
|
| 883 |
+
assert by[_tow]["watts"] == 13.53 and by[_tow]["joules"] == 15896.0, by[_tow]
|
| 884 |
+
|
| 885 |
+
# Laptop engine absent (meter2 unreachable) → null, NEVER fabricated; tower unaffected.
|
| 886 |
+
globals()["_merged_engine_readings"] = lambda: {"omen": {"watts": 13.53, "joules": 15896.0}}
|
| 887 |
+
m3 = build_mesh()
|
| 888 |
+
by3 = {nd["name"]: nd for nd in m3["nodes"]}
|
| 889 |
+
assert by3[_lap]["watts"] is None and by3[_lap]["joules"] is None, by3[_lap]
|
| 890 |
+
assert by3[_lap]["joules_label"] == LABEL_UNAVAILABLE, by3[_lap]
|
| 891 |
+
assert by3[_tow]["watts"] == 13.53, by3[_tow]
|
| 892 |
+
finally:
|
| 893 |
+
globals()["govern_posture"] = _prev_posture
|
| 894 |
+
globals()["_merged_engine_readings"] = _prev_readings
|
| 895 |
+
with _snap_lock:
|
| 896 |
+
_snap_cache["data"] = None
|
| 897 |
+
_snap_cache["ts"] = 0.0
|
| 898 |
+
out["mesh_laptop_attribution"] = True
|
| 899 |
+
|
| 900 |
out["ok"] = all(v is True for v in out.values())
|
| 901 |
return out
|
| 902 |
|
szl_governed_api.py
CHANGED
|
@@ -58,9 +58,11 @@ except Exception: # pragma: no cover
|
|
| 58 |
# Override the whole mesh with SZL_MESH_JSON='[{"name":..,"ollama":..,"meter":..,"model":..}]'.
|
| 59 |
_DEFAULT_MESH = [
|
| 60 |
{"name": "Sovereign GPU 2 (tower · RTX 4060 Ti · anchor)",
|
| 61 |
-
"ollama": "https://gpu.a-11-oy.com", "meter": "https://meter.a-11-oy.com", "model": "llama3.1:8b"
|
|
|
|
| 62 |
{"name": "Sovereign GPU 1 (laptop · RTX 5050 · Blackwell)",
|
| 63 |
-
"ollama": "https://gpu2.a-11-oy.com", "meter": "https://meter2.a-11-oy.com", "model": "qwen2.5:3b"
|
|
|
|
| 64 |
]
|
| 65 |
try:
|
| 66 |
MESH = json.loads(os.environ["SZL_MESH_JSON"]) if os.environ.get("SZL_MESH_JSON") else _DEFAULT_MESH
|
|
|
|
| 58 |
# Override the whole mesh with SZL_MESH_JSON='[{"name":..,"ollama":..,"meter":..,"model":..}]'.
|
| 59 |
_DEFAULT_MESH = [
|
| 60 |
{"name": "Sovereign GPU 2 (tower · RTX 4060 Ti · anchor)",
|
| 61 |
+
"ollama": "https://gpu.a-11-oy.com", "meter": "https://meter.a-11-oy.com", "model": "llama3.1:8b",
|
| 62 |
+
"exporter": "omen"},
|
| 63 |
{"name": "Sovereign GPU 1 (laptop · RTX 5050 · Blackwell)",
|
| 64 |
+
"ollama": "https://gpu2.a-11-oy.com", "meter": "https://meter2.a-11-oy.com", "model": "qwen2.5:3b",
|
| 65 |
+
"exporter": "betterwithage"},
|
| 66 |
]
|
| 67 |
try:
|
| 68 |
MESH = json.loads(os.environ["SZL_MESH_JSON"]) if os.environ.get("SZL_MESH_JSON") else _DEFAULT_MESH
|