Download sAGI/speak.py from PYTHAI/bankml: direct link, hf CLI and curl.
- Browser
- Download file 32.4 kB
-
https://huggingface.co/spaces/PYTHAI/bankml/resolve/main/sAGI/speak.py
- Command line
-
hf download hf://spaces/PYTHAI/bankml/sAGI/speak.py
-
curl -L -o speak.py https://huggingface.co/spaces/PYTHAI/bankml/resolve/main/sAGI/speak.py
32.4 kB
| #!/usr/bin/env python3 | |
| # SPDX-License-Identifier: MIT OR Apache-2.0 | |
| """Savante's voice in the Savante UI — one voice, hers: pre-rendered statements for the aivatar card. | |
| SAVANTE, AS OF 0.1.3 — A VOICE OF HER OWN, FROM OPEN PARTS (operator, 2026-09-28: "combine from Jaimla as template to | |
| improve from Cori to create Savante as a unique voice that is Savante"; and "add confidence and a slower, more | |
| thoughtful response style"): | |
| body Piper `en_GB-cori-high` — trained on public-domain LibriVox recordings (open; Jenny/Jaimla's dataset is a | |
| custom licence, so it is the TEMPLATE, never the source) — rendered slower and steadier: | |
| length_scale 1.18 (thoughtful), noise_scale 0.50 and noise_w 0.60 (steady, confident), 0.8 s after | |
| each sentence | |
| template Jaimla's measured f0 (182 Hz, docspeech_voices.json): each clip's own f0 is measured and moved onto it | |
| with rubberband, formants preserved — lower and grounded, not a slowed tape | |
| resonance the house SAVANTE recipe: her own voice an octave below, 80–2600 Hz, heard only in echo (4 taps at 29 ms, | |
| decay 0.5), at 0.30 — derived from the same clip, so locked to her delivery exactly | |
| eq SAVANTE's curve (+1.5 at 200, −2.5 at 500) with presence +6 dB at 3 kHz and +4 dB above 5 kHz | |
| measured f0 181 Hz · spectral centroid ≈ 2270 Hz (Cori's own brightness, kept: forcing Jaimla's 2783 needs presence | |
| that turns sibilant) — written into the manifest per clip | |
| The eSpeak stand-in below remains only as the fallback where Piper or the model is absent. | |
| The voice is SAVANTE as the house defines her. Her full voice (mindX data/config/docspeech_voices.json, id | |
| "savante") is layered piper — Jaimla's body, a male resonance an octave below in echo locked to her delivery, | |
| breath at the edges — and renders on the house render host. On a computer without that host, the house's own | |
| stand-in for her is the cast entry "savante" in the DeltaVerse voice index (vendor/espeak-ng/mindx-voices/ | |
| index.json): eSpeak NG `en-gb-x-rp+jaimla` at 168 wpm, "savante is FROM jaimla; her eSpeak stand-in carries it". | |
| That stand-in is what listen.html plays for her in a browser, and it is what this module renders — here, on this | |
| computer, with the same WASM build (vendor/espeak-ng/0.3.5-en) under Node. Every agent in this UI speaks with it. | |
| Her name is said sav-ont: the house pronunciation table (mindX data/config/pronunciation.json, voaice-pronunciation/1) | |
| is applied to what is spoken — case-insensitive, one pass, longest match first — never to what is shown. | |
| Renders are cached under BANKML_UI_STATE/voice/savante/ by a key over (engine, voice, wpm, the variant file's hash, | |
| the table version, the spoken text), encoded to Opus, with a manifest. Nothing is written into the canon. | |
| """ | |
| from __future__ import annotations | |
| import hashlib | |
| import json | |
| import os | |
| import re | |
| import shutil | |
| import subprocess | |
| import tempfile | |
| from pathlib import Path | |
| HOME = Path.home() | |
| ESPEAK = Path(os.environ.get("BANKML_ESPEAK", HOME / "DeltaVerse" / "vendor" / "espeak-ng" / "0.3.5-en")) | |
| VOICES = Path(os.environ.get("BANKML_ESPEAK_VOICES", HOME / "DeltaVerse" / "vendor" / "espeak-ng" / "mindx-voices")) | |
| PRON = Path(os.environ.get("BANKML_PRONUNCIATION", HOME / "mindX" / "data" / "config" / "pronunciation.json")) | |
| PRERENDER = Path(__file__).resolve().parent / "voice" / "prerender.mjs" | |
| # Rendered audio lives OUTSIDE every dot-directory (Gradio refuses to serve a path with one) and outside git. | |
| PIPER = Path(os.environ.get("BANKML_PIPER", HOME / ".local" / "share" / "bankml" / "piper")) | |
| NEURAL = {"id": "savante", "engine": "piper 2023.11.14-2 + ffmpeg (rubberband)", "model": "en_GB-cori-high", | |
| "model_licence": "public domain (LibriVox) — rhasspy/piper-voices MODEL_CARD", "template": "jaimla (docspeech_voices.json): f0 182 Hz", | |
| "length_scale": 1.18, "noise_scale": 0.5, "noise_w": 0.6, "sentence_silence": 0.8, "target_f0": 182.0, | |
| "resonance": {"octave": -1, "band": [80, 2600], "echo_ms": 29, "taps": 4, "decay": 0.5, "gain": 0.30}, | |
| "eq": [[200, 200, 1.5], [500, 240, -2.5], [3050, 2600, 6.0]], "high_shelf": [5000, 4.0], "pad_s": 0.35} | |
| VOICE_DIR = Path(os.environ.get("BANKML_VOICE_DIR", Path(__file__).resolve().parent / "voice" / "cache")).expanduser() | |
| ENGINE = "espeak-ng-wasm 0.3.5 (echogarden, vendor/espeak-ng/0.3.5-en)" | |
| _BUILTIN_TABLE = {"format": "voaice-pronunciation/1", "version": 2, "entries": [ | |
| {"match": "PYTHAIML", "say": "Pith AI M L"}, {"match": "SAVANTE", "say": "Sav ont"}, {"match": "PYTHAI", "say": "Pith AI"}]} | |
| def neural_available() -> bool: | |
| return (PIPER / "piper" / "piper").is_file() and (PIPER / f"{NEURAL['model']}.onnx").is_file() and bool(shutil.which("ffmpeg")) | |
| def savante_voice() -> dict: | |
| """Savante's voice as used here: the neural one (Cori body, Jaimla template) when present, else the eSpeak stand-in.""" | |
| if neural_available(): | |
| return {**NEURAL, "voice": NEURAL["model"] + " → savante", "wpm": round(168 / NEURAL["length_scale"]), "standIn": False, | |
| "why": "a voice of her own: Cori's body (public domain) on Jaimla's pitch, slower and steadier, with SAVANTE's resonance", | |
| "source": "ui/speak.py NEURAL"} | |
| return _espeak_voice() | |
| def _espeak_voice() -> dict: | |
| """The cast entry "savante" from the house voice index; the stand-in for her layered voice.""" | |
| try: | |
| c = json.loads((VOICES / "index.json").read_text(encoding="utf-8"))["cast"]["savante"] | |
| return {"id": "savante", "voice": c["voice"], "wpm": int(c["rate"]), "standIn": bool(c.get("standIn")), "why": c.get("why", ""), | |
| "source": str(VOICES / "index.json") + " cast.savante"} | |
| except (OSError, KeyError, ValueError): | |
| return {"id": "savante", "voice": "en-gb-x-rp+jaimla", "wpm": 168, "standIn": True, | |
| "why": "savante is FROM jaimla; her eSpeak stand-in carries it", "source": "built-in copy of the cast entry"} | |
| def table() -> dict: | |
| try: | |
| t = json.loads(PRON.read_text(encoding="utf-8")) | |
| if t.get("format") == "voaice-pronunciation/1" and isinstance(t.get("entries"), list): | |
| return t | |
| except (OSError, ValueError): | |
| pass | |
| return _BUILTIN_TABLE | |
| # said, not shown: the token's split name, the runtime's name, and a web address read as "dot" | |
| _NUM = {"0": "zero", "1": "one", "2": "two", "3": "three", "4": "four", "5": "five", "6": "six", "7": "seven", "8": "eight", "9": "nine", | |
| "16": "sixteen", "32": "thirty-two", "64": "sixty-four", "128": "one twenty-eight"} | |
| _ROMAN = {"I": 1, "II": 2, "III": 3, "IV": 4, "V": 5, "VI": 6} | |
| def _quant(m) -> str: | |
| """Q1_0 → "Q one zero", Q2_0_g64 → "Q two zero, g sixty-four", Q4_K_M → "Q four K M" (a ggml type, spelled).""" | |
| out = [] | |
| for part in m.group(0).split("_"): | |
| mm = re.fullmatch(r"([A-Za-z]*)(\d*)", part) | |
| if mm and mm.group(2) and mm.group(1).lower() == "g": | |
| out.append(", g " + _NUM.get(mm.group(2), mm.group(2))) | |
| elif mm: | |
| out.append(" ".join(list(mm.group(1).upper()) + ([_NUM.get(mm.group(2), mm.group(2))] if mm.group(2) else []))) | |
| else: | |
| out.append(part) | |
| return " ".join(out).replace(" ,", ",") | |
| def _section(m) -> str: | |
| return f"section {_NUM.get(str(_ROMAN.get(m.group(1), 0)), m.group(1))} point {_NUM.get(m.group(2), m.group(2))}" | |
| # said, not shown; applied in order, before the underscore rule and the house table. Maths symbols are read only in a | |
| # maths context (between numbers or single-letter variables), so a name such as "Professor / OVERLORD" or "DAIO · | |
| # savante_sagi" keeps its separator. | |
| LOCAL_SAY = ( | |
| (r"SCIEN·TIFIC", "Sci-en, Tiffic"), (r"(?i)\bbankml\.rs\b", "bank M L dot R S"), (r"(?i)\bbankml\b", "bank M L"), | |
| (r"\b([a-z0-9-]+)\.(pythai)\.(net)\b", r"\1 dot \2 dot \3"), | |
| # TECHNICAL.md's references: a parenthesis that only points somewhere is dropped; a section named in a sentence is read | |
| (r"\s*\((?:see\s+)?(?:§[^()]*|PERFORMANCE\.md[^()]*|`?[a-z0-9_]+\.rs`?(?:,[^()]*)?)\)", ""), (r"\bof\s+§\s*I\.2\b", "described earlier"), | |
| (r"§\s*([IVX]+)\.(\d+)", _section), (r"\b(?:[a-z0-9_]+)\.rs\b", lambda m: m.group(0)[:-3].replace("_", " ") + " dot R S"), | |
| (r"\b(?:PT|P|T|I)?Q\d(?:_[A-Za-z0-9]+)+\b", _quant), (r"(?i)\bq8_0\b", "Q eight zero"), (r"\*", ""), | |
| (r"\+ or −", "plus or minus"), (r"(?<=[\dA-Za-z)])\s\+\s(?=[\d(A-Za-z])", " plus "), (r"(?<=[\w)])\s=\s", " equals "), | |
| (r"(?<=\d)\s/\s(?=\d)", " over "), (r"log₂\s*3", "log base two of three"), (r"3⁵", "three to the fifth"), | |
| (r"Σ_\{w=\+1\}", "the sum over plus-one weights of"), (r"Σ_\{w=−1\}", "the sum over minus-one weights of"), | |
| (r"Σ_\{sᵢ=\+1\}", "the sum over plus signs of"), (r"Σ_\{sᵢ=−1\}", "the sum over minus signs of"), | |
| (r"Σ", "the sum of "), (r"≈", " about "), (r"≤", " at most "), (r"∈", " in "), (r"(?<=[\d\w)])\s?×\s?(?=[\d\w(])", " times "), | |
| (r"\{−1, 0, \+1, \+2\}", "minus one, zero, plus one, or plus two"), | |
| (r"\{\s*−1,\s*0,\s*\+1\s*\}", "minus one, zero, or plus one"), (r"\{\s*−1,\s*\+1\s*\}", "minus one or plus one"), | |
| (r"−(?=\s?\d)", " minus "), (r"\s−\s", " minus "), (r"(?<![\w])\+(?=\d)", "plus "), (r"ᵢ", " i"), | |
| (r"(?<=\b[a-z\d])\s·\s(?=[a-z\d]\b)", " times "), (r"\s{2,}", " ")) | |
| _RX_CACHE: dict = {} | |
| def say(text: str, t: dict | None = None) -> str: | |
| """The respelling the synthesiser hears: one pass, case-insensitive, longest match first.""" | |
| t = t or table() | |
| for m_, s_ in LOCAL_SAY: # bankml's own words first, while Q1_0 and q1_0.rs still have their underscores | |
| text = re.sub(m_, s_, text) | |
| text = re.sub(r"(?<=\w)_(?=\w)", " ", text).strip() # savante_sagi, APPROVE_WITH_CONDITIONS: said as words, shown as written | |
| ents = t["entries"] | |
| if not ents: | |
| return text | |
| key = (id(t), t.get("version"), len(ents)) | |
| if key not in _RX_CACHE: # the table compiled once, not per sentence | |
| es = sorted(ents, key=lambda e: -len(e["match"])) | |
| _RX_CACHE[key] = (re.compile("|".join(re.escape(e["match"]) for e in es), re.I), {e["match"].lower(): e["say"] for e in es}) | |
| rx, by = _RX_CACHE[key] | |
| return rx.sub(lambda m: by[m.group(0).lower()], text) | |
| def _salt(v: dict, t: dict) -> str: | |
| if v.get("model"): | |
| return json.dumps([v["engine"], v["model"], {k: v[k] for k in sorted(NEURAL) if k not in ("model_licence",)}, t.get("version")], sort_keys=True) | |
| return json.dumps([ENGINE, v["voice"], v["wpm"], _variant_sha(v["voice"]), t.get("version")]) | |
| def _f0(path: str) -> float | None: | |
| """Median f0 (autocorrelation over voiced frames), as the house measures a voice; numpy if present.""" | |
| try: | |
| import wave | |
| import numpy as np | |
| w = wave.open(path) | |
| sr = w.getframerate() | |
| x = np.frombuffer(w.readframes(w.getnframes()), dtype=np.int16).astype(np.float32) | |
| f0 = [] | |
| for i in range(0, len(x) - 2048, 512): | |
| fr = x[i:i + 2048] * np.hanning(2048) | |
| if np.sqrt((fr ** 2).mean()) < 300: | |
| continue | |
| ac = np.correlate(fr, fr, "full")[2047:] | |
| lo, hi = int(sr / 400), int(sr / 70) | |
| k = lo + int(np.argmax(ac[lo:hi])) | |
| if ac[k] > 0.35 * ac[0]: | |
| f0.append(sr / k) | |
| return float(np.median(f0)) if f0 else None | |
| except Exception: # noqa: BLE001 | |
| return None | |
| def _render_neural(todo: list): | |
| """Piper (Cori, slower and steadier) → per-clip pitch onto Jaimla's f0 → SAVANTE's resonance and EQ → Opus.""" | |
| import fcntl | |
| n = NEURAL | |
| VOICE_DIR.mkdir(parents=True, exist_ok=True) | |
| lock = open(PIPER / ".render.lock", "w") # beside Piper: machine-wide, whatever voice directory asked | |
| fcntl.flock(lock, fcntl.LOCK_EX) # one Piper at a time on this machine, whoever asked (memory, not speed, is the limit) | |
| todo = [i for i in todo if not i["file"].is_file()] # another renderer may have made them while we waited | |
| try: | |
| _render_neural_locked(todo, n) | |
| finally: | |
| fcntl.flock(lock, fcntl.LOCK_UN) | |
| lock.close() | |
| def _render_neural_locked(todo: list, n: dict): | |
| if not todo: | |
| return | |
| with tempfile.TemporaryDirectory(prefix="bankml-savante-") as tmp: | |
| lines = "".join(json.dumps({"text": i["say"], "output_file": f"{tmp}/{i['key']}.wav"}) + "\n" for i in todo) | |
| p = subprocess.run([str(PIPER / "piper" / "piper"), "--model", str(PIPER / f"{n['model']}.onnx"), "--json-input", "--quiet", | |
| "--length_scale", str(n["length_scale"]), "--noise_scale", str(n["noise_scale"]), "--noise_w", str(n["noise_w"]), | |
| "--sentence_silence", str(n["sentence_silence"])], input=lines, capture_output=True, text=True, timeout=7200) | |
| if p.returncode: | |
| raise RuntimeError("piper failed: " + p.stderr.strip()[-400:]) | |
| r = n["resonance"] | |
| eq = ",".join(f"equalizer=f={f}:t=h:w={w}:g={g}" for f, w, g in n["eq"]) | |
| taps = "|".join(str(r["echo_ms"] * (k + 1)) for k in range(r["taps"])) | |
| decays = "|".join(f"{r['decay'] ** (k + 1):.4f}" for k in range(r["taps"])) | |
| for i in todo: | |
| src = f"{tmp}/{i['key']}.wav" | |
| f0 = _f0(src) | |
| ratio = max(0.70, min(1.0, n["target_f0"] / f0)) if f0 else 0.80 | |
| fc = (f"[0:a]rubberband=pitch={ratio:.4f}:formant=preserved:transients=smooth,asplit=2[b0][r];" | |
| f"[b0]highpass=f=70,{eq},highshelf=f={n['high_shelf'][0]}:g={n['high_shelf'][1]}[b];" | |
| f"[r]rubberband=pitch=0.5:formant=preserved,highpass=f={r['band'][0]},lowpass=f={r['band'][1]}," | |
| f"aecho=0.0:1.0:{taps}:{decays},volume={r['gain']}[rv];" | |
| f"[b][rv]amix=inputs=2:weights=1 1:normalize=0,alimiter=limit=0.89:level=false,apad=pad_dur={n['pad_s']}[o]") | |
| subprocess.run(["ffmpeg", "-v", "error", "-y", "-i", src, "-filter_complex", fc, "-map", "[o]", "-ac", "1", | |
| "-c:a", "libopus", "-b:a", "40k", str(i["file"])], check=True, timeout=300) | |
| dur = subprocess.run(["ffprobe", "-v", "error", "-show_entries", "format=duration", "-of", "csv=p=0", str(i["file"])], | |
| capture_output=True, text=True).stdout.strip() | |
| i["seconds"] = float(dur or 0) | |
| i["measured"] = {"body_f0": round(f0, 1) if f0 else None, "pitch_ratio": round(ratio, 4), "target_f0": n["target_f0"]} | |
| def _variant_sha(voice: str) -> str: | |
| name = voice.split("+", 1)[1] if "+" in voice else "" | |
| p = VOICES / "voices" / "!v" / name | |
| return hashlib.sha256(p.read_bytes()).hexdigest()[:16] if name and p.is_file() else "none" | |
| def available() -> tuple: | |
| if neural_available(): | |
| return True, "ok" | |
| for need, what in ((shutil.which("node"), "node"), (shutil.which("ffmpeg"), "ffmpeg"), ((ESPEAK / "espeak-ng.js").is_file(), "the espeak-ng WASM build"), | |
| ((VOICES / "voices").is_dir(), "the house voice files"), (PRERENDER.is_file(), "sAGI/voice/prerender.mjs")): | |
| if not need: | |
| return False, f"needs {what}" | |
| return True, "ok" | |
| def render(texts: list, state: Path | None = None) -> list: | |
| """[{text, say, key, file, seconds}] for each statement, rendered once and cached; [] with no engine.""" | |
| ok, why = available() | |
| if not ok or not texts: | |
| return [] | |
| v, t = savante_voice(), table() | |
| d = VOICE_DIR / "savante" | |
| d.mkdir(parents=True, exist_ok=True) | |
| salt = _salt(v, t) | |
| items = [] | |
| for text in texts: | |
| s = say(text, t) | |
| key = hashlib.sha256((salt + "\x1f" + s).encode()).hexdigest()[:24] | |
| items.append({"text": text, "say": s, "key": key, "file": d / f"{key}.ogg"}) | |
| todo = [i for i in items if not i["file"].is_file()] | |
| if todo and v.get("model"): | |
| _render_neural(todo) | |
| elif todo: | |
| with tempfile.TemporaryDirectory(prefix="bankml-voice-") as tmp: | |
| job = {"espeak": str(ESPEAK), "voices": str(VOICES), "voice": v["voice"], "wpm": v["wpm"], "out": tmp, | |
| "items": [{"key": i["key"], "say": i["say"]} for i in todo]} | |
| p = subprocess.run(["node", str(PRERENDER)], input=json.dumps(job), capture_output=True, text=True, timeout=300) | |
| if p.returncode: | |
| raise RuntimeError("prerender failed: " + p.stderr.strip()[-400:]) | |
| res = {r["key"]: r for r in json.loads(p.stdout)["items"]} | |
| for i in todo: | |
| subprocess.run(["ffmpeg", "-v", "error", "-y", "-i", res[i["key"]]["file"], "-c:a", "libopus", "-b:a", "32k", | |
| "-ac", "1", str(i["file"])], check=True, timeout=60) | |
| i["seconds"] = res[i["key"]]["seconds"] | |
| import fcntl | |
| man_p = d / "manifest.json" | |
| lock = open(d / ".manifest.lock", "w") | |
| fcntl.flock(lock, fcntl.LOCK_EX) # renders may run in parallel (python3 sAGI/speak.py --shard i/n) | |
| man = json.loads(man_p.read_text(encoding="utf-8")) if man_p.is_file() else {} | |
| for i in items: | |
| if "seconds" in i: | |
| man[i["key"]] = {"text": i["text"], "said": i["say"], "seconds": round(i["seconds"], 3), | |
| "opus_sha256": hashlib.sha256(i["file"].read_bytes()).hexdigest(), **({"measured": i["measured"]} if i.get("measured") else {})} | |
| i["seconds"] = man.get(i["key"], {}).get("seconds") | |
| man["_voice"] = {**v, "engine": v.get("engine") or ENGINE, "variant_sha": _variant_sha(v["voice"]), "pronunciation": {"path": str(PRON), "version": t.get("version")}, | |
| "note": "Savante's own voice (Cori body, Jaimla template, SAVANTE resonance), rendered on this computer" if v.get("model") | |
| else "the house stand-in for SAVANTE's layered voice (docspeech_voices.json id savante), rendered on this computer"} | |
| _atomic(man_p, json.dumps(man, indent=1, ensure_ascii=False) + "\n") # readers never see half a manifest | |
| fcntl.flock(lock, fcntl.LOCK_UN) | |
| lock.close() | |
| return items | |
| # ── the introduction a new participant listens to: chapters read from the canon, verbatim ───────────────────── | |
| def speech(md: str) -> list: | |
| """Markdown → the sentences a listener hears: code blocks, tables, images and HTML dropped; links read as their | |
| text; emphasis and list markers removed; headings kept as short sentences of their own.""" | |
| out = [] | |
| md = re.sub(r"```.*?```", " ", md, flags=re.S) | |
| for para in re.split(r"\n\s*\n", md): | |
| lines = [l for l in para.splitlines() if l.strip() and not l.lstrip().startswith(("|", "<", "![", ">|"))] | |
| if not lines: | |
| continue | |
| t = " ".join(re.sub(r"^([-*+]|\d+\.)\s+", "", l.strip()) for l in lines) # list markers at line starts only | |
| t = re.sub(r"!\[[^\]]*\]\([^)]*\)", " ", t) | |
| t = re.sub(r"\[([^\]]+)\]\([^)]*\)", r"\1", t) | |
| t = re.sub(r"`([^`]*)`", r"\1", t) | |
| t = re.sub(r"^#+\s*|\s#+\s", " ", t) | |
| t = re.sub(r"(?<![\w*])[*_]{1,3}(?=\S)([^*_]+?)(?<=\S)[*_]{1,3}(?![\w*])", r"\1", t) # emphasis only; savante_sagi keeps its _ | |
| t = re.sub(r"https?://\S+", "", t) | |
| t = re.sub(r"\s+", " ", t).strip(" >") | |
| if len(t) < 3: | |
| continue | |
| for sent in re.split(r"(?<=[.!?:;])\s+(?=[A-Z0-9\"'(])", t): | |
| if sent.strip(): | |
| out.append(sent.strip()) | |
| return out | |
| def intro_chapters(canon: Path, persona: dict, card: dict) -> list: | |
| """[(title, [sentences])] — what Savante reads to a new participant, all of it from her canon.""" | |
| ch = [] | |
| who = [card.get("description") or "", persona.get("mantra") or ""] | |
| ch.append(("Who I am", [x for d in who for x in speech(d)])) | |
| ch.append(("My oath", speech(persona.get("oath") or ""))) | |
| bel = [b.get("belief", "") for b in ((persona.get("bdi") or {}).get("beliefs") or []) if isinstance(b, dict)] | |
| ch.append(("What I believe", [x for b in bel for x in speech(b)])) | |
| ch.append(("How I work — my system prompt", speech(persona.get("system_prompt") or ""))) | |
| for title, name in (("Why I exist", "explanation.md"), ("The manifesto", "MANIFESTO.md"), ("Savante, in full", "Savante.md")): | |
| try: | |
| ch.append((title, speech((canon / name).read_text(encoding="utf-8")))) | |
| except OSError: | |
| pass | |
| if name == "explanation.md": # the one chapter that is bankml's, not her canon: SCIEN·TIFIC as the measure | |
| ch.append(("SCIEN·TIFIC: two to the 256, minus one", speech((READINGS / "scientific.md").read_text(encoding="utf-8")))) | |
| return [(t, s) for t, s in ch if s] | |
| READINGS = Path(__file__).resolve().parent / "voice" / "readings" | |
| REPO = Path(__file__).resolve().parents[1] | |
| # the reading: bankml's thesis and the binary / ternary argument, from TECHNICAL.md by heading, verbatim | |
| READING = (("The thesis", "## Thesis"), ("Binary and ternary weights", "### II.1"), ("The binary choice: Q1_0", "### III.2"), | |
| ("The ternary kernel: Q2_0_g64", "### III.5"), ("Can a binary computer perform a ternary operation?", "### III.8")) | |
| def _section(md: str, head: str) -> str: | |
| """The text under the first heading that starts with `head`, up to the next heading of the same or higher level.""" | |
| lv = len(head.split(" ", 1)[0]) | |
| out, on = [], False | |
| for line in md.splitlines(): | |
| h = re.match(r"(#+) ", line) | |
| if on and h and len(h.group(1)) <= lv: | |
| break | |
| if on: | |
| out.append(line) | |
| elif line.startswith(head): | |
| on = True | |
| return "\n".join(out) | |
| def reading_chapters(repo: Path = REPO) -> list: | |
| """[(title, [sentences])]: Savante reads bankml's thesis and TECHNICAL.md's binary/ternary sections aloud.""" | |
| try: | |
| md = (repo / "docs" / "TECHNICAL.md").read_text(encoding="utf-8") | |
| except OSError: | |
| return [] | |
| return [(t, s) for t, h in READING if (s := speech(_section(md, h)))] | |
| # ── export: a chapter set as one Ogg Opus file, complete or not at all ──────────────────────────────────────── | |
| EXPORT_DIR = Path(os.environ.get("BANKML_EXPORT_DIR", Path(__file__).resolve().parent / "voice" / "export")).expanduser() | |
| GAP_S = 0.7 # silence between sentences; 2.0 between chapters | |
| def _duration(f) -> float: | |
| p = subprocess.run(["ffprobe", "-v", "error", "-show_entries", "format=duration", "-of", "csv=p=0", str(f)], capture_output=True, text=True, timeout=60) | |
| try: | |
| return float(p.stdout.strip()) | |
| except ValueError: | |
| return 0.0 | |
| def _atomic(path: Path, text: str) -> None: | |
| tmp = path.with_name(path.name + ".tmp") | |
| tmp.write_text(text, encoding="utf-8") | |
| os.replace(tmp, path) | |
| def export_state(name: str, chapters: list) -> dict: | |
| """{"file", "ready", "total", "current"}: current means the file on disk was built from exactly these clips.""" | |
| items = [i for _, s in chapters for i in cached(s)] | |
| keys = [i["key"] if i else None for i in items] | |
| sig = hashlib.sha256("|".join(k or "-" for k in keys).encode()).hexdigest()[:16] | |
| f = EXPORT_DIR / f"{name}.opus" | |
| meta = EXPORT_DIR / f"{name}.json" | |
| try: | |
| cur = f.is_file() and json.loads(meta.read_text(encoding="utf-8")).get("sig") == sig | |
| except (OSError, ValueError): # absent or damaged: not current (it is rebuilt), never an error for the caller | |
| cur = False | |
| return {"file": f, "ready": sum(1 for k in keys if k), "total": len(keys), "current": cur, "sig": sig, "items": items} | |
| def export(name: str, chapters: list, lead: list = ()) -> dict: | |
| """Concatenate the rendered clips of `lead` + every chapter into EXPORT_DIR/<name>.opus (Ogg Opus, 40 kbit/s, | |
| one chapter mark per chapter). Refuses unless every sentence is rendered: an export is complete.""" | |
| chs = ([("Voice examples", list(lead))] if lead else []) + list(chapters) | |
| st = export_state(name, chs) | |
| if st["current"]: | |
| return st | |
| if st["ready"] < st["total"]: | |
| raise RuntimeError(f"{name}: {st['ready']}/{st['total']} sentences rendered; an export is complete or it is not made") | |
| EXPORT_DIR.mkdir(parents=True, exist_ok=True) | |
| q = lambda p: "file '" + str(p).replace("'", "'\\''") + "'" # the concat demuxer's quoting | |
| with tempfile.TemporaryDirectory(prefix=".bankml-export-", dir=EXPORT_DIR) as tmp: # same filesystem: the rename is atomic | |
| tmp = Path(tmp) | |
| for n, secs in (("gap", GAP_S), ("chap", 2.0)): | |
| subprocess.run(["ffmpeg", "-v", "error", "-y", "-f", "lavfi", "-i", f"anullsrc=r=48000:cl=mono", "-t", str(secs), | |
| "-c:a", "libopus", "-b:a", "40k", str(tmp / f"{n}.ogg")], check=True, timeout=60) | |
| lines, meta, t, k = [], [";FFMETADATA1"], 0.0, 0 | |
| for ci, (title, sents) in enumerate(chs): | |
| if ci: | |
| lines.append(q(tmp / "chap.ogg")) | |
| t += 2.0 | |
| start = t | |
| for j, _ in enumerate(sents): | |
| i = st["items"][k] | |
| k += 1 | |
| if j: | |
| lines.append(q(tmp / "gap.ogg")) | |
| t += GAP_S | |
| lines.append(q(i["file"])) | |
| t += i["seconds"] if i["seconds"] is not None else _duration(i["file"]) # a clip missing from the manifest is measured | |
| meta += ["[CHAPTER]", "TIMEBASE=1/1000", f"START={int(start * 1000)}", f"END={int(t * 1000)}", "title=" + title.replace("=", "\\=")] | |
| (tmp / "list.txt").write_text("\n".join(lines) + "\n", encoding="utf-8") | |
| (tmp / "meta.txt").write_text("\n".join(meta) + "\n", encoding="utf-8") | |
| out = tmp / "out.opus" | |
| subprocess.run(["ffmpeg", "-v", "error", "-y", "-f", "concat", "-safe", "0", "-i", str(tmp / "list.txt"), "-i", str(tmp / "meta.txt"), | |
| "-map", "0:a", "-map_metadata", "1", "-map_chapters", "1", "-c:a", "libopus", "-b:a", "40k", "-ac", "1", | |
| "-metadata", f"title=Savante: {name}", "-metadata", "artist=Savante (bankml; Cori body, Jaimla template)", | |
| "-f", "ogg", str(out)], check=True, timeout=1800) | |
| os.replace(out, st["file"]) | |
| _atomic(EXPORT_DIR / f"{name}.json", json.dumps({"sig": st["sig"], "sentences": st["total"], "chapters": [c for c, _ in chs], | |
| "seconds": round(t, 1), "sha256": hashlib.sha256(st["file"].read_bytes()).hexdigest()}, | |
| indent=1, ensure_ascii=False) + "\n") | |
| return export_state(name, chs) | |
| EXPORTS = {"Savante": "the introduction: her voice examples, then every chapter", "Savante-reading": "the thesis, and binary and ternary, from TECHNICAL.md"} | |
| def export_sets(canon: Path, persona: dict, card: dict) -> dict: | |
| """name -> (lead, chapters) for the two exports.""" | |
| lead = [x for x in persona.get("voice_examples") or [] if isinstance(x, str) and x.strip()] | |
| return {"Savante": (lead, intro_chapters(canon, persona, card)), "Savante-reading": ([], reading_chapters())} | |
| INTRO = {"state": "idle", "done": 0, "total": 0, "error": None} | |
| def render_intro_async(canon: Path, persona: dict, card: dict): | |
| """Render the introduction in the background (once; cached thereafter).""" | |
| import threading | |
| if INTRO["state"] == "running" or os.environ.get("BANKML_VOICE_ASYNC", "1") == "0": | |
| return | |
| chapters = intro_chapters(canon, persona, card) | |
| INTRO.update(state="running", done=0, total=sum(len(s) for _, s in chapters + reading_chapters()), error=None) | |
| def run(): | |
| try: | |
| for _, sents in chapters + reading_chapters(): | |
| for k in range(0, len(sents), 25): | |
| render(sents[k:k + 25]) | |
| INTRO["done"] += len(sents[k:k + 25]) | |
| for name, (lead, chs) in export_sets(canon, persona, card).items(): | |
| export(name, chs, lead) | |
| INTRO["state"] = "done" | |
| except Exception as e: # noqa: BLE001 | |
| INTRO.update(state="error", error=str(e)) | |
| threading.Thread(target=run, daemon=True).start() | |
| _ASYNC = set() | |
| def render_async(texts: list): | |
| """render() in a background thread, once per set of texts at a time (the page never waits on a voice).""" | |
| import threading | |
| if os.environ.get("BANKML_VOICE_ASYNC", "1") == "0": | |
| return | |
| key = hashlib.sha256("\x1e".join(texts).encode()).hexdigest() | |
| if key in _ASYNC: | |
| return | |
| _ASYNC.add(key) | |
| def run(): | |
| try: | |
| for k in range(0, len(texts), 10): | |
| render(texts[k:k + 10]) | |
| finally: | |
| _ASYNC.discard(key) | |
| threading.Thread(target=run, daemon=True).start() | |
| _MAN = {"mtime": None, "data": {}} | |
| def _manifest() -> dict: | |
| """The voice manifest, re-read only when it changes on disk (view mode polls every few seconds); {} if damaged.""" | |
| p = VOICE_DIR / "savante" / "manifest.json" | |
| try: | |
| mt = p.stat().st_mtime_ns | |
| if mt != _MAN["mtime"]: | |
| _MAN.update(mtime=mt, data=json.loads(p.read_text(encoding="utf-8"))) | |
| except (OSError, ValueError): | |
| return {} | |
| return _MAN["data"] | |
| def cached(texts: list) -> list: | |
| """render() without rendering: the items whose audio already exists (None where it does not).""" | |
| v, t = savante_voice(), table() | |
| salt = _salt(v, t) | |
| man = _manifest() | |
| out = [] | |
| for text in texts: | |
| key = hashlib.sha256((salt + "\x1f" + say(text, t)).encode()).hexdigest()[:24] | |
| f = VOICE_DIR / "savante" / f"{key}.ogg" | |
| out.append({"text": text, "key": key, "file": f, "seconds": man.get(key, {}).get("seconds")} if f.is_file() else None) | |
| return out | |
| if __name__ == "__main__": # pre-render Savante's introduction and voice examples from the command line | |
| import sys | |
| c = Path(os.environ.get("SAVANTE_CANON", HOME / "cryptoAGI" / "savante" if (HOME / "cryptoAGI" / "savante").exists() or not (HOME / "savante").exists() else HOME / "savante")) | |
| per = json.loads((c / "savante.persona").read_text(encoding="utf-8")) | |
| crd = json.loads((c / "savante.agentcard.json").read_text(encoding="utf-8")) | |
| chs = intro_chapters(c, per, crd) | |
| # listening order: her voice examples first (short), then the introduction chapter by chapter, then the reading | |
| allt = [x for x in per.get("voice_examples") or [] if isinstance(x, str)] + [x for _, s in chs + reading_chapters() for x in s] | |
| if "--shard" in sys.argv: # python3 sAGI/speak.py --shard 1/2 : every n-th statement, for parallel renders | |
| k, n = map(int, sys.argv[sys.argv.index("--shard") + 1].split("/")) | |
| allt = allt[k - 1::n] | |
| print(f"{len(chs)} chapters, {len(allt)} statements, {sum(len(x.split()) for x in allt)} words", flush=True) | |
| for k in range(0, len(allt), 10): # small batches: each finished batch is playable at once | |
| render(allt[k:k + 10]) | |
| print(f" {min(k + 10, len(allt))}/{len(allt)}", flush=True) | |
| items = cached(allt) | |
| print(f"cached {sum(1 for i in items if i)} of {len(allt)} · {sum(i['seconds'] or 0 for i in items if i) / 60:.1f} min · {VOICE_DIR}") | |
| if "--shard" not in sys.argv: # complete → one file per set: sAGI/voice/export/Savante.opus, Savante-reading.opus | |
| for name, (lead, cs) in export_sets(c, per, crd).items(): | |
| try: | |
| e = export(name, cs, lead) | |
| print(f"export {e['file']} · {e['total']} sentences · {e['file'].stat().st_size / 1e6:.1f} MB") | |
| except RuntimeError as err: | |
| print("export skipped:", err) | |
| if "--prune" in sys.argv: # drop clips no current text uses (older voices, edited canon): the committed cache stays lean | |
| keep = {i["key"] for i in cached(allt) if i} | |
| d = VOICE_DIR / "savante" | |
| gone = [f for f in d.glob("*.ogg") if f.stem not in keep] | |
| for f in gone: | |
| f.unlink() | |
| man_p = d / "manifest.json" | |
| man = json.loads(man_p.read_text(encoding="utf-8")) | |
| man_p.write_text(json.dumps({k: v for k, v in man.items() if k.startswith("_") or k in keep}, indent=1, ensure_ascii=False) + "\n", encoding="utf-8") | |
| print(f"pruned {len(gone)} clips no current text uses") | |