betterwithage commited on
Commit
9db537a
·
verified ·
1 Parent(s): 3ab95d1

chore(sync): mirror backend .py + Dockerfile to Space (hf-sync-backend)

Browse files

Automated backend sync from szl-holdings/a11oy main via hf-sync-backend.
Updated (differed from the Space): szl_energy_live.py, szl_governed_api.py
Deleted (gone from the repo + Dockerfile COPY set): (none)

Keeps the Space-built backend (serve.py + the Dockerfile-COPY'd .py
modules) identical to GitHub main so the Space never rebuilds from a
stale backend, new endpoints don't 404 there, and orphaned modules
removed from the repo don't linger in the Space tree.

Files changed (2) hide show
  1. szl_energy_live.py +124 -11
  2. szl_governed_api.py +4 -2
szl_energy_live.py CHANGED
@@ -266,6 +266,10 @@ def govern_posture(deadline_s: float = MESH_PROBE_DEADLINE_S) -> dict:
266
  "role": _role_for_engine(eng, i),
267
  "model": eng.get("model"),
268
  "gpu_model": _gpu_model_token(eng.get("name") or ""),
 
 
 
 
269
  "is_glm": bool(eng.get("is_glm")),
270
  "live": live_by_idx.get(i),
271
  })
@@ -339,15 +343,54 @@ def build_live() -> dict:
339
  # ---------------------------------------------------------------------------
340
  # /energy/mesh — per-node energy + governance posture for the 3D view.
341
  # ---------------------------------------------------------------------------
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
342
  def build_mesh() -> dict:
343
  snap = meter_snapshot()
344
  reachable = bool(snap.get("reachable"))
345
  posture = govern_posture()
346
  gpus = snap.get("gpus") or [] if reachable else []
 
 
 
 
347
 
348
- # Honest attribution: match a meter GPU to a mesh node by GPU model token (e.g. the
349
  # node "…RTX 4060 Ti…" gets the meter GPU whose name label contains "RTX 4060 Ti").
350
- # A node with no model match keeps null watts/joules (UNAVAILABLE), never a guess.
 
351
  def _match_gpu(gpu_model):
352
  if not gpu_model:
353
  return None
@@ -360,9 +403,19 @@ def build_mesh() -> dict:
360
  nodes = []
361
  watt_vals = []
362
  for n in (posture.get("nodes") or []):
363
- g = _match_gpu(n.get("gpu_model"))
364
- watts = g.get("watts") if g else None
365
- joules = g.get("joules") if g else None
 
 
 
 
 
 
 
 
 
 
366
  if isinstance(watts, (int, float)):
367
  watt_vals.append(watts)
368
  nodes.append({
@@ -372,7 +425,7 @@ def build_mesh() -> dict:
372
  "watts": watts,
373
  "joules": joules,
374
  "joules_label": LABEL_MEASURED if isinstance(joules, (int, float)) else LABEL_UNAVAILABLE,
375
- "source": "NVML" if g else "mesh-posture",
376
  })
377
 
378
  # Normalized 0..1 draw for visualization (relative to the busiest live node this tick).
@@ -381,21 +434,31 @@ def build_mesh() -> dict:
381
  w = node["watts"]
382
  node["draw"] = (round(w / max_w, 6) if (isinstance(w, (int, float)) and max_w > 0) else None)
383
 
384
- label = LABEL_MEASURED if reachable else LABEL_UNAVAILABLE
 
 
 
 
 
 
 
 
 
385
  return {
386
  "ts": _now_iso(),
387
  "label": label,
388
  "nodes": nodes,
389
  "node_count": len(nodes),
390
  "live_count": posture.get("live_count", 0),
391
- "total_watts": snap.get("total_watts") if reachable else None,
392
- "total_joules": snap.get("total_joules") if reachable else None,
393
  "joules_label": label,
394
  "meter_url": METER_URL,
395
  "meter_status": snap.get("status"),
396
  "draw_basis": "watts normalized 0..1 vs the busiest live node this tick (null when no live watts)",
397
- "note": ("per-node watts/joules attributed by GPU-model match to the live NVML exporter; "
398
- "unmatched nodes are UNAVAILABLE, never fabricated"),
 
399
  "doctrine": "v11 — honest empty-states; joules MEASURED only with a real exporter reading.",
400
  }
401
 
@@ -784,6 +847,56 @@ def _selftest() -> dict:
784
  assert mesh["label"] in (LABEL_MEASURED, LABEL_UNAVAILABLE)
785
  out["mesh_shape"] = True
786
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
787
  out["ok"] = all(v is True for v in out.values())
788
  return out
789
 
 
266
  "role": _role_for_engine(eng, i),
267
  "model": eng.get("model"),
268
  "gpu_model": _gpu_model_token(eng.get("name") or ""),
269
+ # The NVML joule-meter engine label this node reports (e.g. 'omen' for the
270
+ # tower, 'betterwithage' for the laptop) — used to attribute live watts/joules
271
+ # from the MERGED multi-meter scrape by engine name (see build_mesh).
272
+ "exporter": eng.get("exporter"),
273
  "is_glm": bool(eng.get("is_glm")),
274
  "live": live_by_idx.get(i),
275
  })
 
343
  # ---------------------------------------------------------------------------
344
  # /energy/mesh — per-node energy + governance posture for the 3D view.
345
  # ---------------------------------------------------------------------------
346
+ def _merged_engine_readings() -> dict:
347
+ """Per-engine live watts + cumulative joules from the MERGED multi-meter scrape.
348
+
349
+ Reuses szl_energy_operator._fetch_joule_meter(), which scrapes EVERY meter in
350
+ A11OY_JOULE_METER_URLS (e.g. the tower's meter.a-11-oy.com AND the laptop's
351
+ meter2.a-11-oy.com) and merges their engines[] into one dict. Returns a map keyed
352
+ by lower-cased engine name ('omen', 'betterwithage') → {watts, joules}, so a mesh
353
+ node can be attributed its OWN engine's live NVML numbers the same way the operator
354
+ does (_exporter_sample_for_node). Honest (Doctrine v11): an unreachable meter simply
355
+ contributes no engine, so that node stays null/UNAVAILABLE — never fabricated; watts
356
+ are read only from a GPU flagged live, else None."""
357
+ try:
358
+ import szl_energy_operator as _op
359
+ meter = _op._fetch_joule_meter()
360
+ except Exception: # noqa: BLE001 — operator/meter absent => empty map, honest fallback
361
+ return {}
362
+ out: dict = {}
363
+ for e in (meter or {}).get("engines", []) or []:
364
+ name = str(e.get("engine") or "").strip().lower()
365
+ if not name:
366
+ continue
367
+ watts = None
368
+ for g in (e.get("gpus") or []):
369
+ if g.get("live") and isinstance(g.get("power_w"), (int, float)):
370
+ watts = float(g["power_w"])
371
+ break
372
+ joules = e.get("joules")
373
+ out[name] = {
374
+ "watts": watts,
375
+ "joules": float(joules) if isinstance(joules, (int, float)) else None,
376
+ }
377
+ return out
378
+
379
+
380
  def build_mesh() -> dict:
381
  snap = meter_snapshot()
382
  reachable = bool(snap.get("reachable"))
383
  posture = govern_posture()
384
  gpus = snap.get("gpus") or [] if reachable else []
385
+ # Live NVML per-engine readings from the MERGED multi-meter scrape (both the tower's
386
+ # and the laptop's meters). This is what lets the laptop node pull its own engine's
387
+ # watts/joules instead of showing null when its meter is a SEPARATE URL from METER_URL.
388
+ engine_map = _merged_engine_readings()
389
 
390
+ # Fallback attribution: match a meter GPU to a mesh node by GPU model token (e.g. the
391
  # node "…RTX 4060 Ti…" gets the meter GPU whose name label contains "RTX 4060 Ti").
392
+ # Only used when the node has no engine reading. A node with neither keeps null
393
+ # watts/joules (UNAVAILABLE), never a guess.
394
  def _match_gpu(gpu_model):
395
  if not gpu_model:
396
  return None
 
403
  nodes = []
404
  watt_vals = []
405
  for n in (posture.get("nodes") or []):
406
+ # Primary: attribute by the node's own NVML engine name from the merged meters
407
+ # ('omen' → tower, 'betterwithage' → laptop). Same mapping the operator uses.
408
+ exporter = str(n.get("exporter") or "").strip().lower()
409
+ reading = engine_map.get(exporter) if exporter else None
410
+ if reading is not None:
411
+ watts = reading.get("watts")
412
+ joules = reading.get("joules")
413
+ source = "NVML"
414
+ else:
415
+ g = _match_gpu(n.get("gpu_model"))
416
+ watts = g.get("watts") if g else None
417
+ joules = g.get("joules") if g else None
418
+ source = "NVML" if g else "mesh-posture"
419
  if isinstance(watts, (int, float)):
420
  watt_vals.append(watts)
421
  nodes.append({
 
425
  "watts": watts,
426
  "joules": joules,
427
  "joules_label": LABEL_MEASURED if isinstance(joules, (int, float)) else LABEL_UNAVAILABLE,
428
+ "source": source,
429
  })
430
 
431
  # Normalized 0..1 draw for visualization (relative to the busiest live node this tick).
 
434
  w = node["watts"]
435
  node["draw"] = (round(w / max_w, 6) if (isinstance(w, (int, float)) and max_w > 0) else None)
436
 
437
+ # Totals from the per-node attributed readings (honest sum of what we actually
438
+ # metered this tick); fall back to the single-meter snapshot when nothing attributed.
439
+ node_watts = [nd["watts"] for nd in nodes if isinstance(nd["watts"], (int, float))]
440
+ node_joules = [nd["joules"] for nd in nodes if isinstance(nd["joules"], (int, float))]
441
+ any_reading = bool(node_watts) or bool(node_joules) or reachable
442
+ total_watts = (round(sum(node_watts), 6) if node_watts
443
+ else (snap.get("total_watts") if reachable else None))
444
+ total_joules = (round(sum(node_joules), 6) if node_joules
445
+ else (snap.get("total_joules") if reachable else None))
446
+ label = LABEL_MEASURED if any_reading else LABEL_UNAVAILABLE
447
  return {
448
  "ts": _now_iso(),
449
  "label": label,
450
  "nodes": nodes,
451
  "node_count": len(nodes),
452
  "live_count": posture.get("live_count", 0),
453
+ "total_watts": total_watts,
454
+ "total_joules": total_joules,
455
  "joules_label": label,
456
  "meter_url": METER_URL,
457
  "meter_status": snap.get("status"),
458
  "draw_basis": "watts normalized 0..1 vs the busiest live node this tick (null when no live watts)",
459
+ "note": ("per-node watts/joules attributed by NVML engine name from the MERGED "
460
+ "multi-meter scrape ('omen' → tower, 'betterwithage' → laptop), GPU-model "
461
+ "match as fallback; unmatched nodes are UNAVAILABLE, never fabricated"),
462
  "doctrine": "v11 — honest empty-states; joules MEASURED only with a real exporter reading.",
463
  }
464
 
 
847
  assert mesh["label"] in (LABEL_MEASURED, LABEL_UNAVAILABLE)
848
  out["mesh_shape"] = True
849
 
850
+ # (f) Mesh per-node attribution by NVML engine name from the MERGED multi-meter map:
851
+ # the laptop node ('betterwithage') pulls its OWN live watts/joules exactly like the
852
+ # tower node ('omen'). When its engine is absent (meter2 down) it stays null —
853
+ # UNAVAILABLE, never fabricated. This is the regression this fix guards.
854
+ _lap = "Sovereign GPU 1 (laptop · RTX 5050 · Blackwell)"
855
+ _tow = "Sovereign GPU 2 (tower · RTX 4060 Ti · anchor)"
856
+ _prev_posture = globals()["govern_posture"]
857
+ _prev_readings = globals()["_merged_engine_readings"]
858
+ try:
859
+ # Deterministic posture (no network) with both nodes' engine labels wired.
860
+ globals()["govern_posture"] = lambda *a, **k: {
861
+ "available": True, "live_count": 2, "nodes": [
862
+ {"name": _tow, "role": "anchor", "gpu_model": "RTX 4060 Ti",
863
+ "exporter": "omen", "is_glm": False, "live": True},
864
+ {"name": _lap, "role": "blackwell", "gpu_model": "RTX 5050",
865
+ "exporter": "betterwithage", "is_glm": False, "live": True},
866
+ ]}
867
+ # Reachable-but-empty single meter so the GPU-model fallback yields nothing
868
+ # (proves attribution comes from the engine map, and null is honest when absent).
869
+ with _snap_lock:
870
+ _snap_cache["data"] = {"gpus": [], "total_watts": 0.0, "total_joules": None,
871
+ "reachable": True, "status": "ok"}
872
+ _snap_cache["ts"] = time.time()
873
+
874
+ # Both engines present in the merged meters → both nodes MEASURED from their own engine.
875
+ globals()["_merged_engine_readings"] = lambda: {
876
+ "omen": {"watts": 13.53, "joules": 15896.0},
877
+ "betterwithage": {"watts": 8.0, "joules": 37000.0},
878
+ }
879
+ m2 = build_mesh()
880
+ by = {nd["name"]: nd for nd in m2["nodes"]}
881
+ assert by[_lap]["watts"] == 8.0 and by[_lap]["joules"] == 37000.0, by[_lap]
882
+ assert by[_lap]["joules_label"] == LABEL_MEASURED and by[_lap]["source"] == "NVML", by[_lap]
883
+ assert by[_tow]["watts"] == 13.53 and by[_tow]["joules"] == 15896.0, by[_tow]
884
+
885
+ # Laptop engine absent (meter2 unreachable) → null, NEVER fabricated; tower unaffected.
886
+ globals()["_merged_engine_readings"] = lambda: {"omen": {"watts": 13.53, "joules": 15896.0}}
887
+ m3 = build_mesh()
888
+ by3 = {nd["name"]: nd for nd in m3["nodes"]}
889
+ assert by3[_lap]["watts"] is None and by3[_lap]["joules"] is None, by3[_lap]
890
+ assert by3[_lap]["joules_label"] == LABEL_UNAVAILABLE, by3[_lap]
891
+ assert by3[_tow]["watts"] == 13.53, by3[_tow]
892
+ finally:
893
+ globals()["govern_posture"] = _prev_posture
894
+ globals()["_merged_engine_readings"] = _prev_readings
895
+ with _snap_lock:
896
+ _snap_cache["data"] = None
897
+ _snap_cache["ts"] = 0.0
898
+ out["mesh_laptop_attribution"] = True
899
+
900
  out["ok"] = all(v is True for v in out.values())
901
  return out
902
 
szl_governed_api.py CHANGED
@@ -58,9 +58,11 @@ except Exception: # pragma: no cover
58
  # Override the whole mesh with SZL_MESH_JSON='[{"name":..,"ollama":..,"meter":..,"model":..}]'.
59
  _DEFAULT_MESH = [
60
  {"name": "Sovereign GPU 2 (tower · RTX 4060 Ti · anchor)",
61
- "ollama": "https://gpu.a-11-oy.com", "meter": "https://meter.a-11-oy.com", "model": "llama3.1:8b"},
 
62
  {"name": "Sovereign GPU 1 (laptop · RTX 5050 · Blackwell)",
63
- "ollama": "https://gpu2.a-11-oy.com", "meter": "https://meter2.a-11-oy.com", "model": "qwen2.5:3b"},
 
64
  ]
65
  try:
66
  MESH = json.loads(os.environ["SZL_MESH_JSON"]) if os.environ.get("SZL_MESH_JSON") else _DEFAULT_MESH
 
58
  # Override the whole mesh with SZL_MESH_JSON='[{"name":..,"ollama":..,"meter":..,"model":..}]'.
59
  _DEFAULT_MESH = [
60
  {"name": "Sovereign GPU 2 (tower · RTX 4060 Ti · anchor)",
61
+ "ollama": "https://gpu.a-11-oy.com", "meter": "https://meter.a-11-oy.com", "model": "llama3.1:8b",
62
+ "exporter": "omen"},
63
  {"name": "Sovereign GPU 1 (laptop · RTX 5050 · Blackwell)",
64
+ "ollama": "https://gpu2.a-11-oy.com", "meter": "https://meter2.a-11-oy.com", "model": "qwen2.5:3b",
65
+ "exporter": "betterwithage"},
66
  ]
67
  try:
68
  MESH = json.loads(os.environ["SZL_MESH_JSON"]) if os.environ.get("SZL_MESH_JSON") else _DEFAULT_MESH