bankml / sAGI /models.py
Gregory-L's picture
Your own bankML in one command: ./install.sh --space (github main a27c4b6)
0ab4e0b verified
Raw History Blame Contribute Delete
54.8 kB
# SPDX-License-Identifier: MIT OR Apache-2.0
"""bankml's model importer: the models Savante can speak through, brought in without friction and never unverified.
Three sources, one discipline:
- a curated catalogue of open-source models (Bonsai 1-bit/ternary, Qwen3, SmolLM, Granite), each pinned here to
the sha256 its publisher's repository lists at a fixed revision;
- any Hugging Face GGUF by URL (https://huggingface.co/OWNER/REPO[/blob|resolve/REV/FILE]): the licence is read from
the repository and must be open source, and the pin is the repository's own LFS sha256 at the resolved revision;
- Ollama: search ollama.com, import from the registry (the layer digest is the GGUF's sha256), or adopt a model
the local Ollama already holds, with no download at all.
Every import is streamed to disk, hashed on the way, and kept only if it equals the published sha256. Then bankml's
header guard (`bankml guard --json`) must say play, and a FORK.json record is written so `bankml serve` will pin it.
A model whose licence is not open source is refused, not hidden behind a warning ("open source or go away"); Gemma and
Llama are refused this way. Nothing is ever uploaded. The only network calls are GETs to huggingface.co, ollama.com
and registry.ollama.ai.
The carrier is `bankml serve MODEL --fork FORK.json --spawn llama-server`. `switch()` replaces it, and rolls back to
the previous model if the new one does not come up."""
import hashlib, json, os, re, shlex, shutil, signal, subprocess, threading, time, urllib.parse, urllib.request
from pathlib import Path
from typing import Mapping, Optional
HOME = Path.home()
REPO = Path(os.environ.get("BANKML_REPO", Path(__file__).resolve().parents[1])).expanduser()
MODELS = Path(os.environ.get("BANKML_MODELS", REPO / ".models")).expanduser()
FORKS = Path(os.environ.get("BANKML_FORKS", HOME / ".local" / "share" / "bankml" / "forks")).expanduser()
DATA = Path(os.environ.get("BANKML_DATA", HOME / ".local" / "share" / "bankml")).expanduser()
LLAMA_TAG = "b11192" # the engine release bankML is checked against (install.sh LLAMA_TAG)
def _install_env(data: Path) -> dict:
"""KEY=value pairs install.sh recorded in <data>/install.env (written with printf %q). Read as data, never executed;
an unreadable or malformed file reads as empty."""
out = {}
try:
lines = (data / "install.env").read_text(encoding="utf-8").splitlines()
except OSError:
return out
for line in lines:
key, sep, raw = line.partition("=")
if not sep or not key.isidentifier():
continue
try:
words = shlex.split(raw)
except ValueError:
continue
if len(words) == 1:
out[key] = words[0]
return out
def llama_server(env: Optional[Mapping[str, str]] = None, data: Optional[Path] = None, home: Optional[Path] = None) -> Path:
"""The llama-server binary, found exactly as install.sh finds it: BANKML_LLAMA_SERVER; else the INSTALL_LLAMA_SERVER
the installer recorded; else the first executable of the installer's download and the development checkout. When
none is executable, the installer's download path, so a refusal names where `./install.sh engine` puts it."""
env = os.environ if env is None else env
data = DATA if data is None else data
home = HOME if home is None else home
explicit = env.get("BANKML_LLAMA_SERVER") or _install_env(data).get("INSTALL_LLAMA_SERVER")
if explicit:
return Path(explicit).expanduser()
download = data / f"llama-{LLAMA_TAG}" / "llama-server"
for c in (download, home / "sAGI" / "bonsai" / f"llama-{LLAMA_TAG}" / "llama-server"):
if c.is_file() and os.access(c, os.X_OK):
return c
return download
LLAMA = llama_server()
BANKML = Path(os.environ.get("BANKML_BIN", REPO / "target" / "release" / "bankml")).expanduser()
OLLAMA_STORE = Path(os.environ.get("BANKML_OLLAMA_MODELS", "/usr/share/ollama/.ollama/models"))
LISTEN = os.environ.get("BANKML_SERVE_LISTEN", "127.0.0.1:18093")
UPSTREAM = os.environ.get("BANKML_UPSTREAM", "127.0.0.1:18092")
LOG = Path(os.environ.get("BANKML_UI_STATE", HOME / ".local" / "share" / "bankml" / "savante")).expanduser() / "carrier.log"
DISK_MARGIN = 1_500_000_000 # never fill the disk: keep 1.5 GB free after an import
EMBEDDING_ARCHS = {"bert", "nomic-bert", "jina-bert-v2", "xlm-roberta", "modern-bert", "neo-bert", "t5encoder"} # encoders: they embed, they do not chat
UA = {"User-Agent": "bankml-importer (+https://github.com/cryptoAGI/bankml)"}
# OSI-approved licences (SPDX ids as Hugging Face tags them), plus public-domain dedications
OPEN = {"apache-2.0", "mit", "bsd-2-clause", "bsd-3-clause", "bsd", "isc", "mpl-2.0", "gpl-2.0", "gpl-3.0", "lgpl-2.1",
"lgpl-3.0", "agpl-3.0", "unlicense", "cc0-1.0", "zlib", "artistic-2.0", "epl-2.0", "ecl-2.0", "upl-1.0", "0bsd"}
# the curated catalogue: sha256 and bytes as each repository lists them at `revision` (read 2026-09-28); an import
# re-reads the repository and refuses unless all three still agree
CATALOG = [
{"id": "bonsai-8b", "title": "Bonsai-8B · 1-bit (Q1_0)", "repo": "PYTHAI/Bonsai-8B-gguf-fork", "file": "Bonsai-8B-Q1_0.gguf",
"revision": "87c4ff9b04e32b7bc71d66d62a803f3e3aa002b4", "bytes": 1158654496, "sha256": "284a335aa3fb2ced3b1b01fcb40b08aa783e3b70832767f0dd2e3fdfa134bd54", "licence": "apache-2.0", "default": True,
"note": "Savante's default carrier: an 8B Qwen3 in 1.16 GB; bankml's Q1_0 kernel is bit-exact against it"},
{"id": "ternary-bonsai-8b", "title": "Ternary-Bonsai-8B · ternary (Q2_0_g64)", "repo": "PYTHAI/Ternary-Bonsai-8B-gguf-fork",
"file": "Ternary-Bonsai-8B-Q2_0_g64.gguf", "revision": "2445960052d465ace262a43b5f5fef52a0c1e1ef", "bytes": 2310125920, "sha256": "e17b298d84ee78797916ae5c2ecc8211469cc65cccfe3080cd9a9bb503fbc55e", "licence": "apache-2.0",
"note": "the better answers of the two Bonsai-8B, slower in the reference runtime (TECHNICAL.md §I.2)"},
{"id": "bonsai-4b", "title": "Bonsai-4B · 1-bit (Q1_0)", "repo": "prism-ml/Bonsai-4B-gguf", "file": "Bonsai-4B-Q1_0.gguf",
"revision": "78f2c2bacd0904ffaba24b4873ed975e5818354a", "bytes": 572270624, "sha256": "4524b3f997f0f06444e568d1f26e2efd69effa3218c7ad3047432fb171e42168", "licence": "apache-2.0", "note": "small and quick"},
{"id": "bonsai-1.7b", "title": "Bonsai-1.7B · 1-bit (Q1_0)", "repo": "prism-ml/Bonsai-1.7B-gguf", "file": "Bonsai-1.7B-Q1_0.gguf",
"revision": "210a9e99f79cb184909d49595906526eb2b3dd9a", "bytes": 248302272, "sha256": "3d7c6c90dd98717a203adb22d5eacd2581850e40aa5327e144b97766cae5f7e3", "licence": "apache-2.0", "note": "the oracle's test model; a quarter gigabyte"},
{"id": "qwen3-0.6b", "title": "Qwen3-0.6B · Q8_0", "repo": "Qwen/Qwen3-0.6B-GGUF", "file": "Qwen3-0.6B-Q8_0.gguf",
"revision": "23749fefcc72300e3a2ad315e1317431b06b590a", "bytes": 639446688, "sha256": "9465e63a22add5354d9bb4b99e90117043c7124007664907259bd16d043bb031", "licence": "apache-2.0", "note": "Qwen's own GGUF; the smallest Qwen3"},
{"id": "qwen3-1.7b", "title": "Qwen3-1.7B · Q4_K_M", "repo": "ggml-org/Qwen3-1.7B-GGUF", "file": "Qwen3-1.7B-Q4_K_M.gguf",
"revision": "daeb8e2d528a760970442092f6bf1e55c3b659eb", "bytes": 1282439264, "sha256": "d2387ca2dbfee2ffabce7120d3770dadca0b293052bc2f0e138fdc940d9bc7b5", "licence": "apache-2.0", "note": "a standard 4-bit K-quant"},
{"id": "qwen3-4b", "title": "Qwen3-4B · Q4_K_M", "repo": "Qwen/Qwen3-4B-GGUF", "file": "Qwen3-4B-Q4_K_M.gguf",
"revision": "bc640142c66e1fdd12af0bd68f40445458f3869b", "bytes": 2497280256, "sha256": "7485fe6f11af29433bc51cab58009521f205840f5b4ae3a32fa7f92e8534fdf5", "licence": "apache-2.0", "note": "Qwen's own GGUF"},
{"id": "qwen3-8b", "title": "Qwen3-8B · Q4_K_M", "repo": "Qwen/Qwen3-8B-GGUF", "file": "Qwen3-8B-Q4_K_M.gguf",
"revision": "7c41481f57cb95916b40956ab2f0b139b296d974", "bytes": 5027783488, "sha256": "d98cdcbd03e17ce47681435b5150e34c1417f50b5c0019dd560e4882c5745785", "licence": "apache-2.0",
"note": "the full-precision-quality 8B; needs about 5 GB of disk and memory"},
{"id": "smollm2-1.7b", "title": "SmolLM2-1.7B-Instruct · Q4_K_M", "repo": "HuggingFaceTB/SmolLM2-1.7B-Instruct-GGUF",
"file": "smollm2-1.7b-instruct-q4_k_m.gguf", "revision": "2d4a76a30b4af41ecd395c35725ac11688d4cfe4", "bytes": 1055609536, "sha256": "decd2598bc2c8ed08c19adc3c8fdd461ee19ed5708679d1c54ef54a5a30d4f33",
"licence": "apache-2.0", "note": "Hugging Face's own small model, open data"},
{"id": "smollm3-3b", "title": "SmolLM3-3B · Q4_K_M", "repo": "ggml-org/SmolLM3-3B-GGUF", "file": "SmolLM3-Q4_K_M.gguf",
"revision": "4965cb60b150737b68a0408c36aeefb65078f894", "bytes": 1915305312, "sha256": "8334b850b7bd46238c16b0c550df2138f0889bf433809008cc17a8b05761863e", "licence": "apache-2.0", "note": "reasoning on or off (/no_think)"},
{"id": "granite-3.3-2b", "title": "Granite-3.3-2B-Instruct · Q4_K_M", "repo": "ibm-granite/granite-3.3-2b-instruct-GGUF",
"file": "granite-3.3-2b-instruct-Q4_K_M.gguf", "revision": "7cdf86ccd1f1bb3491c9b7017b033f2e51367397", "bytes": 1545303328, "sha256": "ac71e9e32c0bea919b409c5918f69ca74339854b0319c5065e4e9fb6d95c4852",
"licence": "apache-2.0", "note": "IBM's open model, tuned for instructions and tools"},
# coders (the codephreak and simplecoder agents): open-source only, so StarCoder and WizardCoder are not here
# (OpenRAIL-M / Llama 2 licences). The newest coder, Qwen3-Coder-Next (80B, Apache-2.0), ships as four files of
# 48 GB in all; it is named in those agents' .model as the latest, not offered for import here.
{"id": "qwen2.5-coder-1.5b", "title": "Qwen2.5-Coder-1.5B-Instruct · Q4_K_M · coder", "repo": "Qwen/Qwen2.5-Coder-1.5B-Instruct-GGUF",
"file": "qwen2.5-coder-1.5b-instruct-q4_k_m.gguf", "revision": "f86cb2c1fa58255f8052cc32aeede1b7482d4361", "bytes": 1117320768,
"sha256": "cc324af070c2ecbfd324a30884d2f951a7ff756aba85cb811a6ec436933bb046", "licence": "apache-2.0", "coder": True,
"note": "the coder agents' default on a small machine: Qwen's own GGUF, 1.1 GB"},
{"id": "qwen2.5-coder-7b", "title": "Qwen2.5-Coder-7B-Instruct · Q4_K_M · coder", "repo": "Qwen/Qwen2.5-Coder-7B-Instruct-GGUF",
"file": "qwen2.5-coder-7b-instruct-q4_k_m.gguf", "revision": "13fb94bfda8c8cf22497dc57b78f391a9acb426a", "bytes": 4683073536,
"sha256": "509287f78cb4d4cf6b3843734733b914b2c158e43e22a7f4bf5e963800894d3c", "licence": "apache-2.0", "coder": True,
"note": "the stronger small coder; needs about 5 GB of disk and memory"},
{"id": "qwen3.8-27b", "title": "Qwen3.8-27B · UD-Q4_K_M · latest Qwen (code and general)", "repo": "unsloth/Qwen3.8-27B-GGUF",
"file": "Qwen3.8-27B-UD-Q4_K_M.gguf", "revision": "4ca720788d1e01f1bff70c033e0d0028fd02e502", "bytes": 16464440224,
"sha256": "322e194ff79741c7baa497c240f677f54b201b0efab44ca8e50f122b39123482", "licence": "apache-2.0", "coder": True,
"note": "the newest Qwen (Aug 2026), strong at code; 16.5 GB, for a machine with 24 GB or more"},
]
# 0.3.4 (O4): models with no published GGUF, converted here from their pinned safetensors by llama.cpp b11192's own
# `convert_hf_to_gguf.py --outtype f16` (commit 171e8846, unmodified). The conversion is reproducible (run twice, the
# same sha256) and was checked tensor by tensor: every GGUF tensor equals the safetensors tensor rounded to f16 (the
# norms kept f32), Q/K rows in llama's NORM-RoPE order. `pin_converted` re-reads the source repository's LFS sha256
# at the revision before it writes the FORK.json; the conversion itself needs torch and is not run by bankML.
CONVERTED = [
{"id": "smollm2-135m-instruct", "title": "SmolLM2-135M-Instruct · F16", "file": "SmolLM2-135M-Instruct-F16.gguf",
"bytes": 270885888, "sha256": "e9aba089704487f72efa3c6cbb6d4c748d4c515428247f97679063137e45a222", "licence": "apache-2.0",
"kind": "model", "repo": "HuggingFaceTB/SmolLM2-135M-Instruct", "revision": "12fd25f77366fa6b3b4b768ec3050bf629380bac",
"source_file": "model.safetensors", "source_bytes": 269060552, "source_sha256": "5af571cbf074e6d21a03528d2330792e532ca608f24ac70a143f6b369968ab8c",
"tools": "llama.cpp b11192 convert_hf_to_gguf.py --outtype f16; torch 2.11.0+cpu, transformers 4.57.6, numpy 2.2.6 (b11192's requirements)",
"note": "the base of mindX's lineage family (SmolLM2-135M, Llama architecture); HuggingFaceTB publishes no F16 GGUF of it"},
{"id": "mindx-gen39", "title": "mindx-gen39 · F16 (mindXtrain39, the last accepted generation)", "file": "mindx-gen39-F16.gguf",
"bytes": 270885600, "sha256": "6b64c748d96ad26fd72402299bd27b2ae82f489bd0469498dd18eb6054058266", "licence": "apache-2.0",
"kind": "dataset", "repo": "PYTHAI/mindXascension", "revision": "4bd31b9db75e3c4c0159af631edc27b1e142ff11",
"source_file": "weights/gen39/ollama_push/merged/model.safetensors", "source_bytes": 269060552,
"source_sha256": "19b62829de298cc06925947976b33d34f9ffb24f5d0e05f349feabae4d83357c",
"tools": "llama.cpp b11192 convert_hf_to_gguf.py --outtype f16; torch 2.11.0+cpu, transformers 5.8.0 (its tokenizer_config.json was "
"written by transformers 5.8.0, which 4.57.6 cannot read), numpy 2.2.6",
"note": "mindX's own generation 39: SmolLM2-135M + its LoRA, merged by mindXtrain; Ollama serves it as mindx-gen39"},
]
# ── small helpers ──────────────────────────────────────────────────────────────────────────────────────────
def _get(url: str, timeout=30, raw=False):
with urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=timeout) as r:
b = r.read()
return b if raw else json.loads(b)
def licence_open(tag) -> bool:
tags = tag if isinstance(tag, list) else [tag]
return any(isinstance(t, str) and t.lower().removeprefix("license:") in OPEN for t in tags)
def free_bytes(p: Path = MODELS) -> int:
p = p if p.exists() else p.parent
return shutil.disk_usage(p.resolve()).free
def mem_total() -> int:
try:
return int(re.search(r"MemTotal:\s+(\d+)", Path("/proc/meminfo").read_text()).group(1)) * 1024
except (OSError, AttributeError):
return 0
def fits(nbytes: int, have: bool = False) -> tuple:
"""(ok, why): disk for the file plus the margin (skipped when the file is already here), and a model the memory
can hold (weights ≲ 70 % of RAM; always checked)."""
if not have and nbytes + DISK_MARGIN > free_bytes():
return False, f"needs {nbytes / 1e9:.1f} GB + {DISK_MARGIN / 1e9:.1f} GB margin; {free_bytes() / 1e9:.1f} GB free on disk"
return fits_memory(nbytes)
def fits_memory(nbytes: int) -> tuple:
mt = mem_total()
if mt and nbytes > 0.7 * mt:
return False, f"{nbytes / 1e9:.1f} GB of weights on a {mt / 1e9:.1f} GB machine would page to swap"
return True, "fits"
def _safe_name(s: str) -> str:
s = re.sub(r"[^A-Za-z0-9._-]+", "-", s).strip("-.")
if not s.endswith(".gguf"):
s += ".gguf"
return s
# ── installed models and their pins ───────────────────────────────────────────────────────────────────────
def forks() -> dict:
"""basename -> (FORK.json path, record) for every pinned file in FORKS."""
out = {}
for f in sorted(FORKS.glob("*.json")):
try:
j = json.loads(f.read_text(encoding="utf-8"))
except (OSError, ValueError):
continue
for rec in j.get("files") or []:
if isinstance(rec, dict) and str(rec.get("path", "")).endswith(".gguf") and re.fullmatch(r"[0-9a-f]{64}", str(rec.get("sha256", ""))):
out[rec["path"]] = (f, {**rec, "source": j.get("source_repo") or j.get("source"), "licence": j.get("licence_tag")})
return out
def installed() -> list:
"""[{file, path, bytes, pinned, fork, source, licence, catalog}] for every GGUF in MODELS."""
pins, cat = forks(), {c["file"]: c for c in CATALOG}
out = []
for p in sorted(MODELS.glob("*.gguf")):
try:
size = p.stat().st_size
except OSError:
continue # a dangling link
fk = pins.get(p.name)
out.append({"file": p.name, "path": str(p), "bytes": size, "pinned": bool(fk), "fork": str(fk[0]) if fk else None,
"source": (fk[1]["source"] if fk else None) or (cat.get(p.name) or {}).get("repo"),
"licence": (fk[1]["licence"] if fk else None) or (cat.get(p.name) or {}).get("licence"), "catalog": (cat.get(p.name) or {}).get("id")})
return out
def write_fork(file: str, nbytes: int, sha: str, source: str, url: str, revision: str, licence: str, pinned_from: str) -> Path:
FORKS.mkdir(parents=True, exist_ok=True)
f = FORKS / f"{file}.FORK.json"
f.write_text(json.dumps({"kind": "bankml import (weights downloaded, verified against the published sha256)", "source_repo": source,
"source_url": url, "source_revision": revision, "licence_tag": licence,
"imported_at_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), "pinned_from": pinned_from,
"files": [{"path": file, "bytes": nbytes, "sha256": sha}]}, indent=1) + "\n", encoding="utf-8")
return f
def guard(path: Path) -> dict:
"""bankml's header guard on a file: {verdict, arch, types, reasons}."""
p = subprocess.run([str(BANKML), "guard", str(path), "--json"], capture_output=True, text=True, timeout=120)
try:
return json.loads(p.stdout)
except ValueError:
return {"verdict": "refuse", "reasons": [(p.stderr or p.stdout).strip()[-300:] or "bankml guard gave no report"]}
# ── Hugging Face ──────────────────────────────────────────────────────────────────────────────────────────
def hf_parse(url: str) -> tuple:
"""(repo, revision|None, file|None) from a Hugging Face URL or an OWNER/REPO id."""
u = url.strip()
u = re.sub(r"^hf://", "", u)
u = re.sub(r"^https?://(www\.)?(huggingface\.co|hf\.co)/", "", u)
u = u.split("?")[0].split("#")[0].strip("/")
parts = u.split("/")
seg = lambda x: bool(re.fullmatch(r"[A-Za-z0-9._-]+", x)) and x not in (".", "..")
if len(parts) < 2 or not all(seg(x) for x in parts[:2]):
raise ValueError("not a Hugging Face model: give https://huggingface.co/OWNER/REPO (optionally …/blob/REV/FILE.gguf)")
repo = "/".join(parts[:2])
if len(parts) >= 4 and parts[2] in ("blob", "resolve", "tree"):
rev, f = urllib.parse.unquote(parts[3]), urllib.parse.unquote("/".join(parts[4:])) if parts[4:] else None
if not re.fullmatch(r"[A-Za-z0-9._-]+", rev) or rev in (".", "..") or (f and any(x in ("", ".", "..") for x in f.split("/"))):
raise ValueError("a revision is a branch, tag or commit; a file path has no '.' or '..' segments")
return repo, rev, f
return repo, None, None
def hf_resolve(url: str) -> dict:
"""{repo, revision, licence, open, files:[{file, bytes, sha256}], chosen} for a Hugging Face URL."""
repo, rev, file = hf_parse(url)
m = _get(f"https://huggingface.co/api/models/{repo}" + (f"/revision/{urllib.parse.quote(rev)}" if rev else ""))
sha = m.get("sha") or rev
lic = (m.get("cardData") or {}).get("license") or [t for t in m.get("tags", []) if t.startswith("license:")]
tree = _get(f"https://huggingface.co/api/models/{repo}/tree/{sha}?recursive=true")
files = [{"file": e["path"], "bytes": e.get("size", 0), "sha256": (e.get("lfs") or {}).get("oid")}
for e in tree if e.get("type") == "file" and e["path"].endswith(".gguf") and (e.get("lfs") or {}).get("oid")]
if file and not any(x["file"] == file for x in files):
raise ValueError(f"{file} is not a GGUF in {repo}@{sha[:10]}")
return {"repo": repo, "revision": sha, "licence": lic, "open": licence_open(lic), "files": files, "chosen": file,
"gated": bool(m.get("gated"))} # gated repositories need an account and an accepted agreement: not imported
# ── Ollama ────────────────────────────────────────────────────────────────────────────────────────────────
def _ollama_name(name: str) -> tuple:
n = name.strip().removeprefix("https://ollama.com/").removeprefix("library/")
n, _, tag = n.partition(":")
if not re.fullmatch(r"[a-z0-9][a-z0-9._-]*(/[a-z0-9][a-z0-9._-]*)?", n) or ".." in n:
raise ValueError(f"not an Ollama model name: {name}")
if tag and (not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._-]*", tag) or ".." in tag):
raise ValueError(f"not an Ollama tag: {tag}")
return n, tag or "latest"
def ollama_search(q: str) -> list:
"""ollama.com's search, as [{name, description, sizes, pulls}] (its HTML; there is no JSON search API)."""
import html
page = _get("https://ollama.com/search?q=" + urllib.parse.quote(q.strip()), raw=True).decode("utf-8", "replace")
out = []
for li in re.findall(r"<li[^>]*>(.*?)</li>", page, flags=re.S):
m = re.search(r'href="/library/([^"/:]+)"', li)
if not m:
continue
d = re.search(r"<p[^>]*>([^<]{8,})</p>", li)
sizes = re.findall(r"text-blue-600[^>]*>\s*([0-9.]+[mbk]|[0-9x.]+b)\s*<", li)
pulls = re.search(r"<span\s*>([0-9.]+[KMB]?)</span>\s*<span[^>]*>&nbsp;Pulls", li)
cloud = "cloud" in li and ">cloud<" in li
out.append({"name": m.group(1), "description": html.unescape(d.group(1).strip()) if d else "", "sizes": sizes,
"pulls": pulls.group(1) if pulls else "", "cloud": cloud})
return out
def ollama_tags(name: str) -> list:
n, _ = _ollama_name(name)
page = _get(f"https://ollama.com/library/{n}/tags", raw=True).decode("utf-8", "replace")
return sorted(set(re.findall(rf'href="/library/{re.escape(n)}:([^"]+)"', page)))
def _licence_of_text(t: str) -> str | None:
s = t[:4000].lower()
if "gemma terms of use" in s:
return "gemma (not open source)"
if "llama" in s and "community license" in s:
return "llama community licence (not open source)"
if "apache license" in s and "version 2.0" in s:
return "apache-2.0"
if "mit license" in s or "permission is hereby granted, free of charge" in s:
return "mit"
if "gnu general public license" in s:
return "gpl-3.0" if "version 3" in s else "gpl-2.0"
if "redistribution and use in source and binary forms" in s:
return "bsd-3-clause"
return None
def ollama_resolve(name: str) -> dict:
"""{repo, tag, file, bytes, sha256, licence, open, url} from the Ollama registry (manifest + licence layer)."""
n, tag = _ollama_name(name)
path = n if "/" in n else f"library/{n}"
req = urllib.request.Request(f"https://registry.ollama.ai/v2/{path}/manifests/{tag}",
headers={**UA, "Accept": "application/vnd.docker.distribution.manifest.v2+json"})
with urllib.request.urlopen(req, timeout=30) as r:
man = json.loads(r.read())
layers = man.get("layers") or []
model = [l for l in layers if l.get("mediaType") == "application/vnd.ollama.image.model"]
if not model:
raise ValueError(f"{n}:{tag} has no model weights in the registry (a cloud-only tag?)")
lic_layer = [l for l in layers if l.get("mediaType") == "application/vnd.ollama.image.license"]
lic = None
if lic_layer:
lic = _licence_of_text(_get(f"https://registry.ollama.ai/v2/{path}/blobs/{lic_layer[0]['digest']}", raw=True).decode("utf-8", "replace"))
d = model[0]["digest"].removeprefix("sha256:")
return {"repo": f"ollama:{path}", "tag": tag, "file": _safe_name(f"{n.replace('/', '-')}-{tag}"), "bytes": model[0]["size"], "sha256": d,
"licence": lic or "unrecognised", "open": bool(lic and licence_open(lic)), "url": f"https://registry.ollama.ai/v2/{path}/blobs/sha256:{d}",
"local": str(OLLAMA_STORE / "blobs" / f"sha256-{d}") if (OLLAMA_STORE / "blobs" / f"sha256-{d}").is_file() else None}
def ollama_local_spec(name: str) -> dict:
"""The same record as ollama_resolve(), built from the local Ollama's own manifest and licence layer: offline, and
pinned to what was actually pulled (the registry's tag may have moved since)."""
n, tag = _ollama_name(name)
ns, base = (n.split("/", 1) if "/" in n else ("library", n))
f = OLLAMA_STORE / "manifests" / "registry.ollama.ai" / ns / base / tag
man = json.loads(f.read_text())
layers = man.get("layers") or []
model = [l for l in layers if l.get("mediaType") == "application/vnd.ollama.image.model"]
if not model:
raise ValueError(f"{n}:{tag}: the local manifest has no model layer")
blob = OLLAMA_STORE / "blobs" / model[0]["digest"].replace(":", "-")
lic_layer = [l for l in layers if l.get("mediaType") == "application/vnd.ollama.image.license"]
lic = _licence_of_text((OLLAMA_STORE / "blobs" / lic_layer[0]["digest"].replace(":", "-")).read_text(errors="replace")) if lic_layer else None
d = model[0]["digest"].removeprefix("sha256:")
return {"repo": f"ollama:{ns}/{base}", "tag": tag, "file": _safe_name(f"{n.replace('/', '-')}-{tag}"), "bytes": model[0]["size"], "sha256": d,
"licence": lic or "unrecognised", "open": bool(lic and licence_open(lic)), "url": "", "local": str(blob) if blob.is_file() else None}
def ollama_local() -> list:
"""Models the local Ollama already holds: [{name, tag, sha256, bytes, blob}]."""
out = []
root = OLLAMA_STORE / "manifests" / "registry.ollama.ai"
for f in sorted(root.glob("*/*/*")) if root.is_dir() else []:
try:
man = json.loads(f.read_text())
except (OSError, ValueError):
continue
for l in man.get("layers") or []:
if l.get("mediaType") == "application/vnd.ollama.image.model":
blob = OLLAMA_STORE / "blobs" / l["digest"].replace(":", "-")
if blob.is_file():
ns, name = f.parent.parent.name, f.parent.name
out.append({"name": name if ns == "library" else f"{ns}/{name}", "tag": f.name, "sha256": l["digest"].removeprefix("sha256:"),
"bytes": l["size"], "blob": str(blob)})
return out
# ── the import job (one at a time; the UI polls JOB) ──────────────────────────────────────────────────────
JOB = {"state": "idle", "what": "", "done": 0, "total": 0, "error": None, "result": None, "started": None, "seq": 0, "cancel": False}
_LOCK = threading.Lock()
def _hash_file(p: Path, on=None) -> str:
h = hashlib.sha256()
with open(p, "rb") as f:
while b := f.read(1 << 22):
h.update(b)
if on:
on(len(b))
return h.hexdigest()
def _download(url: str, dest: Path, nbytes: int, want: str) -> None:
"""Stream url → dest.part, hashing on the way; rename only if the sha256 is the published one. Resumes a .part;
never writes past the published size."""
part = dest.with_name(dest.name + ".part")
h = hashlib.sha256()
have = part.stat().st_size if part.exists() else 0
if have > nbytes:
part.unlink()
have = 0
if have: # resume: hash what is there first
with open(part, "rb") as f:
while b := f.read(1 << 22):
h.update(b)
JOB["done"] = have
if have < nbytes:
req = urllib.request.Request(url, headers={**UA, **({"Range": f"bytes={have}-"} if have else {})})
with urllib.request.urlopen(req, timeout=60) as r:
cr = r.headers.get("Content-Range") or ""
if have and (r.status != 206 or not cr.startswith(f"bytes {have}-")): # the range was ignored or misplaced: start over
have = 0
h = hashlib.sha256()
JOB["done"] = 0
with open(part, "ab" if have else "wb") as f:
while b := r.read(1 << 20):
if JOB.get("cancel"):
raise RuntimeError("cancelled (the partial file is kept; importing again resumes it)")
if JOB["done"] + len(b) > nbytes:
f.close()
part.unlink(missing_ok=True)
raise RuntimeError(f"the source sent more than the published {nbytes} bytes: discarded")
f.write(b)
h.update(b)
JOB["done"] += len(b)
got = h.hexdigest()
if got != want:
part.unlink(missing_ok=True)
raise RuntimeError(f"sha256 {got[:16]}… is not the published {want[:16]}…: discarded")
os.replace(part, dest)
def import_spec(spec: dict) -> dict:
"""Bring one model in. spec: {file, bytes, sha256, url, repo, revision, licence, pinned_from, local?}."""
if not licence_open(spec.get("licence")):
raise PermissionError(f"{spec.get('repo')}: licence {spec.get('licence')!r} is not open source; bankml imports open-source models only")
if not re.fullmatch(r"[0-9a-f]{64}", spec.get("sha256") or ""):
raise ValueError("no published sha256 for this file: nothing to pin it to, refused")
dest = MODELS / _safe_name(spec["file"])
ok, why = fits(spec["bytes"], have=dest.exists() or bool(spec.get("local")))
if not ok:
raise RuntimeError(why)
MODELS.mkdir(parents=True, exist_ok=True)
JOB.update(total=spec["bytes"], done=0)
if dest.exists():
JOB["what"] = f"checking {dest.name} already here"
if _hash_file(dest, lambda n: JOB.__setitem__("done", JOB["done"] + n)) != spec["sha256"]:
raise RuntimeError(f"{dest.name} exists but is not the published file (sha256 differs); move it aside first")
elif spec.get("local"): # the local Ollama already has it: link, then hash
JOB["what"] = f"adopting {spec['file']} from the local Ollama (no download)"
if _hash_file(Path(spec["local"]), lambda n: JOB.__setitem__("done", JOB["done"] + n)) != spec["sha256"]:
raise RuntimeError("the local Ollama blob does not hash to its digest: refused")
os.symlink(spec["local"], dest)
else:
JOB["what"] = f"downloading {spec['file']} from {spec['repo']}"
_download(spec["url"], dest, spec["bytes"], spec["sha256"])
g = guard(dest)
if g.get("verdict") != "play":
dest.unlink(missing_ok=True) # a symlink goes; the Ollama blob it points at stays
raise RuntimeError(f"bankml's guard refused {dest.name}: {'; '.join(g.get('reasons') or [])}")
fk = write_fork(dest.name, spec["bytes"], spec["sha256"], spec["repo"], spec.get("page") or spec["url"], spec.get("revision", ""),
spec["licence"] if isinstance(spec["licence"], str) else ",".join(spec["licence"]), spec["pinned_from"])
return {"file": dest.name, "path": str(dest), "fork": str(fk), "arch": g.get("arch"), "types": g.get("types")}
def spec_catalog(cid: str) -> dict:
c = next(x for x in CATALOG if x["id"] == cid)
r = hf_resolve(f"https://huggingface.co/{c['repo']}/blob/{c['revision']}/{c['file']}")
f = next(x for x in r["files"] if x["file"] == c["file"])
if not (f["sha256"] == c["sha256"] and r["revision"] == c["revision"] and f["bytes"] == c["bytes"]):
raise RuntimeError(f"{c['repo']} no longer lists the catalogued file (sha256/size/revision changed): refused")
return spec_hf(r, c["file"])
def spec_hf(r: dict, file: str) -> dict:
f = next(x for x in r["files"] if x["file"] == file)
lic = r["licence"] if isinstance(r["licence"], str) else ",".join(t.removeprefix("license:") for t in r["licence"])
return {"file": _safe_name(file.replace("/", "-")), "bytes": f["bytes"], "sha256": f["sha256"], "repo": r["repo"], "revision": r["revision"], "licence": lic,
"url": f"https://huggingface.co/{r['repo']}/resolve/{r['revision']}/{urllib.parse.quote(file)}",
"page": f"https://huggingface.co/{r['repo']}/blob/{r['revision']}/{file}", "pinned_from": f"Hugging Face LFS sha256 of {file} at {r['revision']}"}
def spec_ollama(o: dict) -> dict:
return {"file": o["file"], "bytes": o["bytes"], "sha256": o["sha256"], "repo": o["repo"], "revision": o["tag"], "licence": o["licence"],
"url": o["url"], "page": f"https://ollama.com/{o['repo'].removeprefix('ollama:').removeprefix('library/')}:{o['tag']}", "local": o.get("local"),
"pinned_from": "Ollama registry layer digest (sha256 of the GGUF)"}
def start_job(what: str, fn, *a) -> bool:
"""Run fn(*a) in the background as the one import/switch job; False if one is already running."""
with _LOCK:
if JOB["state"] == "running":
return False
JOB.update(state="running", what=what, done=0, total=0, error=None, result=None, started=time.time(), cancel=False)
def run():
try:
JOB["result"] = fn(*a)
JOB["state"] = "done"
except Exception as e: # noqa: BLE001
JOB.update(state="error", error=f"{type(e).__name__}: {e}")
finally:
JOB["seq"] += 1 # the UI refreshes its lists when this moves
threading.Thread(target=run, daemon=True).start()
return True
def cancel_job() -> bool:
"""Ask a running download to stop at its next chunk (the .part is kept for resume)."""
if JOB["state"] == "running":
JOB["cancel"] = True
return True
return False
# ── the carrier: bankml serve + llama-server ──────────────────────────────────────────────────────────────
def serve_status(timeout=3) -> dict:
try:
with urllib.request.urlopen(f"http://{LISTEN}/bankml", timeout=timeout) as r:
return json.loads(r.read())
except Exception as e: # noqa: BLE001
return {"error": str(e)}
def _listeners() -> dict:
"""port -> pid for the carrier's two ports (from ss), only if the process is this user's llama-server or bankml.
Matches the configured host (127.0.0.1, 0.0.0.0, [::1] …) exactly."""
out = {}
try:
s = subprocess.run(["ss", "-ltnpH"], capture_output=True, text=True, timeout=10).stdout
except (OSError, subprocess.SubprocessError):
return out
for addr in (LISTEN, UPSTREAM):
host, port = addr.rsplit(":", 1)
hosts = {host, f"[{host.strip('[]')}]", "*"} if host in ("0.0.0.0", "::", "[::]") else {host, f"[{host.strip('[]')}]"}
for line in s.splitlines():
cols = line.split()
h, _, pt = cols[3].rpartition(":") if len(cols) >= 4 else ("", "", "")
if pt != port or h not in hosts:
continue
m = re.search(r"pid=(\d+)", line)
if not m:
continue
pid = int(m.group(1))
try:
exe = Path(f"/proc/{pid}/cmdline").read_bytes().split(b"\0")[0].decode()
mine = os.stat(f"/proc/{pid}").st_uid == os.getuid()
except OSError:
continue
if mine and Path(exe).name in ("llama-server", "bankml"):
out[port] = pid
return out
def _kill(pid: int, sig) -> None:
try:
os.kill(pid, sig)
except ProcessLookupError:
pass
def _stop_carrier():
for pid in set(_listeners().values()):
_kill(pid, signal.SIGTERM)
for _ in range(100):
if not _listeners():
return
time.sleep(0.2)
for pid in set(_listeners().values()):
_kill(pid, signal.SIGKILL)
time.sleep(0.5)
if _listeners():
raise RuntimeError(f"the carrier's ports are still held by {sorted(set(_listeners().values()))}; stop them by hand")
CTX = int(os.environ.get("BANKML_CTX", "2048")) # the engine's context: 2048 keeps an 8B model's KV cache near 0.3 GB
THREADS = int(os.environ.get("BANKML_THREADS_SERVE", "3"))
RESOURCES = LOG.parent / "resources.json"
SLOTS = LOG.parent / "slots" # the engine's saved KV slots (--slot-save-path): a restart restores the system prompt # the operator's CPU and RAM choice (the Resources sliders), used by every start
OVERHEAD = 250_000_000 # llama-server's compute buffers and runtime beside weights and KV (measured order of magnitude)
CTX_MIN, CTX_MAX = 512, 32768
def resources() -> dict:
"""{"threads", "ctx", "ram_gb", "spec_ngram", "engine", "gpu_limit"}: the saved choice, else the defaults
(BANKML_THREADS_SERVE, BANKML_CTX). engine: "auto" (bankML's own forward pass for the ternary Qwen3 files, where it
is about 8x llama-server; llama-server otherwise), "native" or "llama.cpp". gpu_limit (0.3.7): the share of the
GPU's memory and time the native engine may use (BANKML_GPU_LIMIT), 0 for none (BANKML_GPU=off)."""
r = {"threads": THREADS, "ctx": CTX, "ram_gb": None, "spec_ngram": False, "engine": "auto", "gpu_limit": 0.8}
try:
r.update({k: v for k, v in json.loads(RESOURCES.read_text()).items() if k in r})
except (OSError, ValueError):
pass
return r
def plan(model: Path, ram_gb: float) -> dict:
"""What a RAM budget buys for a model: weights stay resident, the rest is KV cache (f16, the guard's
bytes-per-token for this architecture) after the engine's overhead. ctx is rounded down to 256."""
g = guard(model)
kv = int(g.get("kv_f16_bytes_per_token") or 0) or 147_456 # Qwen3-8B's, if the guard cannot say
w = model.stat().st_size
room = int(ram_gb * 1e9) - w - OVERHEAD
ctx = max(0, min(CTX_MAX, room // kv // 256 * 256))
return {"ctx": ctx, "fits": ctx >= CTX_MIN, "kv_gb": round(ctx * kv / 1e9, 2), "weights_gb": round(w / 1e9, 2),
"kv_bytes_per_token": kv, "overhead_gb": OVERHEAD / 1e9, "arch": g.get("arch"),
"min_ram_gb": round((w + OVERHEAD + CTX_MIN * kv) / 1e9, 2)}
def usage() -> dict:
"""What the carrier uses now, from bankml itself (`GET /bankml/usage`, sys.rs reading /proc — bankml's psutil,
no crates); the Python reading below is the fallback for a serve older than 0.1.8."""
try:
with urllib.request.urlopen(f"http://{LISTEN}/bankml/usage", timeout=5) as r:
u = json.loads(r.read())
return {"pids": [p["pid"] for p in u["processes"]], "rss_gb": round(u["rss_bytes"] / 1e9, 2), "cpu_pct": round(u["cpu_percent"]),
"cores": u["cores"], "mem_total_gb": round(u["mem_total_bytes"] / 1e9, 1), "mem_available_gb": round(u["mem_available_bytes"] / 1e9, 2),
"source": u["source"], "processes": u["processes"]}
except (OSError, ValueError, KeyError):
pass
pids = sorted(set(_listeners().values()))
def ticks(pid):
try:
f = Path(f"/proc/{pid}/stat").read_text().rsplit(")", 1)[1].split()
return int(f[11]) + int(f[12]) # utime + stime
except (OSError, IndexError, ValueError):
return 0
def rss(pid):
try:
return int(re.search(r"VmRSS:\s+(\d+)", Path(f"/proc/{pid}/status").read_text()).group(1)) * 1024
except (OSError, AttributeError):
return 0
t0 = {p: ticks(p) for p in pids}
time.sleep(0.5)
hz = os.sysconf("SC_CLK_TCK")
cpu = sum(ticks(p) - t0[p] for p in pids) / hz / 0.5 * 100
try:
avail = int(re.search(r"MemAvailable:\s+(\d+)", Path("/proc/meminfo").read_text()).group(1)) * 1024
except (OSError, AttributeError):
avail = 0
return {"pids": pids, "rss_gb": round(sum(rss(p) for p in pids) / 1e9, 2), "cpu_pct": round(cpu), "cores": os.cpu_count(),
"mem_total_gb": round(mem_total() / 1e9, 1), "mem_available_gb": round(avail / 1e9, 2), "source": "sAGI/models.py (/proc)"}
def allow_origin() -> str | None:
"""The one web page bankml serve lets in (`--allow-origin`): BANKML_ALLOW_ORIGIN, else what `./install.sh --space`
recorded; None (this computer only) when neither is set."""
o = os.environ.get("BANKML_ALLOW_ORIGIN")
if o is None:
o = _install_env(DATA).get("INSTALL_ALLOW_ORIGIN", "")
return o.strip() or None
def native_for(model: Path, engine: str | None = None) -> bool:
"""Whether the carrier answers from bankML's own forward pass (`bankml serve --native`, 0.3.0) for this model:
"native" always (bankml refuses a model its forward pass does not run), "llama.cpp" never, "auto" for the
ternary (Q2_0_g64) files — there bankML is about 8x llama-server with the same tokens; on the 1-bit files
llama-server is still faster."""
e = engine or resources().get("engine", "auto")
return e == "native" or (e == "auto" and "Q2_0" in model.name)
def apply_resources(threads: int, ram_gb: float, busy=lambda: False, spec_ngram: bool = False, engine: str = "auto",
gpu_limit: float | None = None) -> dict:
"""Save the choice and restart the carrier on the same model with it (a verified switch, with rollback)."""
threads = max(1, min(int(threads), os.cpu_count() or 1))
st = serve_status()
sha = (st.get("verified") or {}).get("model_sha256")
pins = forks()
path, fork = _pinned_path(sha, st.get("model"), pins) if sha else (None, None)
if path is None:
raise RuntimeError("no verified carrier is running; choose a model first (the settings are saved and used when it starts)")
pl = plan(path, ram_gb)
if not pl["fits"]:
raise RuntimeError(f"{ram_gb:.1f} GB cannot hold {path.name}: it needs at least {pl['min_ram_gb']} GB")
prev = resources() # the last settings that ran, restored if these do not
RESOURCES.parent.mkdir(parents=True, exist_ok=True)
gl = prev.get("gpu_limit", 0.8) if gpu_limit is None else max(0.0, min(1.0, float(gpu_limit)))
RESOURCES.write_text(json.dumps({"threads": threads, "ctx": pl["ctx"], "ram_gb": ram_gb, "spec_ngram": bool(spec_ngram),
"engine": engine if engine in ("auto", "native", "llama.cpp") else "auto", "gpu_limit": gl}) + "\n")
if busy():
raise RuntimeError("saved; an answer is being written, so the engine restarts with these settings on the next switch")
JOB["what"] = f"restarting {path.name} with {threads} threads and a {pl['ctx']}-token context"
_stop_carrier()
try:
return _start_carrier(path, fork, sha)
except Exception as first:
RESOURCES.write_text(json.dumps(prev) + "\n")
try:
_stop_carrier()
_start_carrier(path, fork, sha)
except Exception as again: # noqa: BLE001
raise RuntimeError(f"{first}; restoring the previous settings also failed: {again}") from first
raise
def _start_carrier(model: Path, fork: Path, want_sha: str | None = None, threads=None, ctx=None, wait=1800) -> dict:
"""Start `bankml serve --spawn` and wait until it answers verified, with `want_sha` when given. bankml hashes the
whole file before it binds, so the wait is on the process, not on the ports: it fails only when the process exits
or the wait runs out, and then the process group is killed so no second carrier is left behind."""
LOG.parent.mkdir(parents=True, exist_ok=True)
with open(LOG, "ab") as log:
at = log.seek(0, 2)
log.write(f"\n# {time.strftime('%Y-%m-%d %H:%M:%S')} bankml serve {model.name}\n".encode())
log.flush()
SLOTS.mkdir(parents=True, exist_ok=True)
r = resources()
if not ctx and r.get("ram_gb"):
pl = plan(model, float(r["ram_gb"]))
if not pl["fits"]:
raise RuntimeError(f"the saved RAM budget ({r['ram_gb']} GB) cannot hold {model.name}: it needs at least {pl['min_ram_gb']} GB")
ctx = pl["ctx"]
n_threads, n_ctx = str(threads or resources()["threads"]), str(ctx or resources()["ctx"])
if native_for(model):
# bankML's own forward pass answers, on the gateway and on the engine address (0.3.0)
# 0.3.8: the native engine saves and restores slots too (Savante's warm start)
cmd = [str(BANKML), "serve", str(model), "--fork", str(fork), "--native", "--upstream", UPSTREAM, "--listen", LISTEN, "--ctx", n_ctx,
"--slot-dir", str(SLOTS)]
gl = float(resources().get("gpu_limit", 0.8))
env = {**os.environ, "BANKML_THREADS": n_threads, **({"BANKML_GPU": "off"} if gl <= 0 else {"BANKML_GPU_LIMIT": f"{gl:.2f}"})}
else:
if not (LLAMA.is_file() and os.access(LLAMA, os.X_OK)):
raise RuntimeError(f"no llama-server at {LLAMA}: run ./install.sh engine, or set BANKML_LLAMA_SERVER to a llama.cpp {LLAMA_TAG} build")
cmd = ([str(BANKML), "serve", str(model), "--fork", str(fork), "--spawn", str(LLAMA), "--upstream", UPSTREAM, "--listen", LISTEN,
"--threads", n_threads, "--ctx", n_ctx] + (["--spec-ngram"] if resources().get("spec_ngram") else []) + ["--slot-dir", str(SLOTS)])
env = None
if allow_origin():
# ./install.sh --space: bankML's Hugging Face page may talk to this serve from the browser (CORS for it alone)
cmd += ["--allow-origin", allow_origin()]
proc = subprocess.Popen(cmd, stdout=log, stderr=log, stdin=subprocess.DEVNULL, start_new_session=True, cwd=str(REPO), env=env)
t0 = time.time()
why = "timed out" # ctx: per model — the saved RAM budget is re-planned for this model's weights and KV size
while time.time() - t0 < wait:
st = serve_status(2)
v = st.get("verified") or {}
if v and (want_sha is None or v.get("model_sha256") == want_sha):
return st
if v:
why = f"another carrier answers (sha256 {str(v.get('model_sha256'))[:12]}…), not {model.name}"
break
if proc.poll() is not None:
why = f"exited with {proc.returncode}"
break
time.sleep(1)
try:
os.killpg(proc.pid, signal.SIGKILL) # serve and the llama-server it spawned share the group
except ProcessLookupError:
pass
proc.wait(timeout=10)
with open(LOG, "rb") as f:
f.seek(at)
said = [l for l in f.read().decode(errors="replace").splitlines()[1:] if l.strip()]
raise RuntimeError(f"bankml serve did not come up with {model.name} ({why}): " + (" / ".join(said[-3:])[-400:] or "no output"))
def _pinned_path(sha: str, model: str | None, pins: dict):
"""(path, fork) for a verified sha256: a file in MODELS first, else the carrier's own path if a pin names it."""
for f, (fk, rec) in pins.items():
if rec["sha256"] == sha and (MODELS / f).exists():
return MODELS / f, fk
if model:
m = Path(model)
rec = pins.get(m.name)
if rec and rec[1]["sha256"] == sha and m.exists():
return m, rec[0]
return None, None
def switch(file: str, busy=lambda: False) -> dict:
"""Make `file` (in MODELS, pinned) the carrier: success only when bankml serve answers verified with this file's
sha256. If the new one fails, the previous model is restored (found by its verified sha256)."""
if busy():
raise RuntimeError("an answer is being written; switch when it is done")
p = MODELS / file
if not p.exists():
raise FileNotFoundError(f"{file} is not in {MODELS}")
ok, why = fits_memory(p.stat().st_size)
if not ok:
raise RuntimeError(why)
arch = guard(p).get("arch")
if arch in EMBEDDING_ARCHS:
raise RuntimeError(f"{file} is an embedding model ({arch}); it cannot be the chat carrier (see docs/embedding.md)")
pins = forks()
if file not in pins:
adopt(file)
pins = forks()
want = pins[file][1]["sha256"]
prev = serve_status()
prev_sha = (prev.get("verified") or {}).get("model_sha256")
if prev_sha == want:
return prev
prev_path, prev_fork = _pinned_path(prev_sha, prev.get("model"), pins) if prev_sha else (None, None)
JOB["what"] = f"stopping the carrier, starting {file} (bankml hashes the whole file, then llama-server loads it)"
try:
_stop_carrier()
return _start_carrier(p, pins[file][0], want)
except Exception:
if prev_path:
JOB["what"] = f"{file} failed; restoring {prev_path.name}"
_stop_carrier()
_start_carrier(prev_path, prev_fork, prev_sha)
raise
def pin_converted(file: str) -> Path:
"""Pin a converted model (CONVERTED): the file must hash to the recorded conversion, and the source repository must
still list the source safetensors with the recorded sha256 at the revision (read now). The FORK.json says how the
file was made."""
c = next((x for x in CONVERTED if x["file"] == file), None)
if not c:
raise RuntimeError(f"{file} is not a recorded conversion")
if not licence_open(c["licence"]):
raise PermissionError(f"{c['repo']}: licence {c['licence']!r} is not open source")
kind = "datasets/" if c["kind"] == "dataset" else ""
folder = c["source_file"].rsplit("/", 1)[0] if "/" in c["source_file"] else ""
listing = _get(f"https://huggingface.co/api/{'datasets' if kind else 'models'}/{c['repo']}/tree/{c['revision']}/{folder}")
src = next((x for x in listing if x.get("path") == c["source_file"]), None)
if not src or (src.get("lfs") or {}).get("oid") != c["source_sha256"] or src.get("size") != c["source_bytes"]:
raise RuntimeError(f"{c['repo']}@{c['revision'][:12]} no longer lists {c['source_file']} with the recorded sha256: refused")
p = MODELS / file
JOB.update(what=f"hashing {file} to pin it", total=c["bytes"], done=0)
if _hash_file(p, lambda n: JOB.__setitem__("done", JOB["done"] + n)) != c["sha256"]:
raise RuntimeError(f"{file} is not the recorded conversion (sha256 differs): refused")
g = guard(p)
if g.get("verdict") != "play":
raise RuntimeError(f"bankml's guard refused {file}: {'; '.join(g.get('reasons') or [])}")
page = f"https://huggingface.co/{'datasets/' if kind else ''}{c['repo']}/blob/{c['revision']}/{c['source_file']}"
f = write_fork(file, c["bytes"], c["sha256"], c["repo"], page, c["revision"], c["licence"],
f"sha256 of the GGUF that {c['tools'].split(';')[0]} makes from {c['source_file']} (LFS sha256 {c['source_sha256']}) at {c['revision']}")
j = json.loads(f.read_text(encoding="utf-8"))
j["kind"] = "bankml conversion (safetensors pinned by LFS sha256, converted by llama.cpp b11192, checked tensor by tensor)"
j["converted_from"] = {"repo": c["repo"], "revision": c["revision"], "file": c["source_file"], "bytes": c["source_bytes"],
"sha256": c["source_sha256"], "tools": c["tools"]}
f.write_text(json.dumps(j, indent=1) + "\n", encoding="utf-8")
return f
def adopt(file: str) -> Path:
"""Pin a file that is already here but has no FORK.json, if the catalogue knows it: hash it and compare to the
repository's published sha256 (fetched now, not trusted from the catalogue's prefix alone). A recorded conversion
(CONVERTED) is pinned by `pin_converted`."""
if any(x["file"] == file for x in CONVERTED):
return pin_converted(file)
c = next((x for x in CATALOG if x["file"] == file), None)
if not c:
raise RuntimeError(f"{file} has no FORK.json pin and is not in the catalogue: import it from its source to pin it")
s = spec_catalog(c["id"])
JOB.update(what=f"hashing {file} to pin it", total=s["bytes"], done=0)
if _hash_file(MODELS / file, lambda n: JOB.__setitem__("done", JOB["done"] + n)) != s["sha256"]:
raise RuntimeError(f"{file} is not the published file (sha256 differs): refused")
return write_fork(file, s["bytes"], s["sha256"], s["repo"], s["page"], s["revision"], s["licence"], s["pinned_from"])
def first_run(busy=lambda: False) -> dict:
"""Seamless start: if no carrier answers, bring in Bonsai-8B (download only if absent) and start it."""
st = serve_status()
if "verified" in st:
return st
c = next(x for x in CATALOG if x.get("default"))
p = MODELS / c["file"]
if p.is_symlink() and not p.exists(): # a dangling link (the target moved): replace it with the real file
p.unlink()
if not p.exists():
import_spec(spec_catalog(c["id"]))
return switch(c["file"], busy)
if __name__ == "__main__": # python3 sAGI/models.py [list | catalog | import ID|URL|ollama:NAME:TAG | use FILE | first-run | search Q]
import sys
a = sys.argv[1:] or ["list"]
if a[0] == "list":
for m in installed():
print(f"{'pinned ' if m['pinned'] else 'UNPINNED'} {m['bytes'] / 1e9:5.2f} GB {m['file']} ({m['source'] or '?'}, {m['licence'] or '?'})")
print("carrier:", Path(serve_status().get("model", "none")).name)
elif a[0] == "catalog":
for c in CATALOG:
print(f"{c['id']:18} {c['bytes'] / 1e9:5.2f} GB {c['title']} [{fits(c['bytes'], (MODELS / c['file']).exists())[1]}]")
elif a[0] == "search":
for r in ollama_search(" ".join(a[1:])):
print(f"{r['name']:24} {','.join(r['sizes']):20} {r['description'][:80]}")
elif a[0] == "import":
t = a[1]
loc = {f"{m['name']}:{m['tag']}" for m in ollama_local()}
name = t.removeprefix("ollama:")
spec = (spec_ollama(ollama_local_spec(name) if name in loc or f"{name}:latest" in loc else ollama_resolve(name)) if t.startswith("ollama:") else
spec_catalog(t) if any(c["id"] == t for c in CATALOG) else None)
if spec is None:
r = hf_resolve(t)
spec = spec_hf(r, r["chosen"] or (sys.exit("pick a file: " + ", ".join(f["file"] for f in r["files"])) if len(r["files"]) != 1 else r["files"][0]["file"]))
print(json.dumps(import_spec(spec), indent=1))
elif a[0] == "use":
print(json.dumps(switch(a[1]), indent=1))
elif a[0] == "first-run":
print(json.dumps(first_run(), indent=1))