Download sAGI/models.py from PYTHAI/bankml: direct link, hf CLI and curl.
- Browser
- Download file 54.8 kB
-
https://huggingface.co/spaces/PYTHAI/bankml/resolve/main/sAGI/models.py
- Command line
-
hf download hf://spaces/PYTHAI/bankml/sAGI/models.py
-
curl -L -o models.py https://huggingface.co/spaces/PYTHAI/bankml/resolve/main/sAGI/models.py
54.8 kB
| # SPDX-License-Identifier: MIT OR Apache-2.0 | |
| """bankml's model importer: the models Savante can speak through, brought in without friction and never unverified. | |
| Three sources, one discipline: | |
| - a curated catalogue of open-source models (Bonsai 1-bit/ternary, Qwen3, SmolLM, Granite), each pinned here to | |
| the sha256 its publisher's repository lists at a fixed revision; | |
| - any Hugging Face GGUF by URL (https://huggingface.co/OWNER/REPO[/blob|resolve/REV/FILE]): the licence is read from | |
| the repository and must be open source, and the pin is the repository's own LFS sha256 at the resolved revision; | |
| - Ollama: search ollama.com, import from the registry (the layer digest is the GGUF's sha256), or adopt a model | |
| the local Ollama already holds, with no download at all. | |
| Every import is streamed to disk, hashed on the way, and kept only if it equals the published sha256. Then bankml's | |
| header guard (`bankml guard --json`) must say play, and a FORK.json record is written so `bankml serve` will pin it. | |
| A model whose licence is not open source is refused, not hidden behind a warning ("open source or go away"); Gemma and | |
| Llama are refused this way. Nothing is ever uploaded. The only network calls are GETs to huggingface.co, ollama.com | |
| and registry.ollama.ai. | |
| The carrier is `bankml serve MODEL --fork FORK.json --spawn llama-server`. `switch()` replaces it, and rolls back to | |
| the previous model if the new one does not come up.""" | |
| import hashlib, json, os, re, shlex, shutil, signal, subprocess, threading, time, urllib.parse, urllib.request | |
| from pathlib import Path | |
| from typing import Mapping, Optional | |
| HOME = Path.home() | |
| REPO = Path(os.environ.get("BANKML_REPO", Path(__file__).resolve().parents[1])).expanduser() | |
| MODELS = Path(os.environ.get("BANKML_MODELS", REPO / ".models")).expanduser() | |
| FORKS = Path(os.environ.get("BANKML_FORKS", HOME / ".local" / "share" / "bankml" / "forks")).expanduser() | |
| DATA = Path(os.environ.get("BANKML_DATA", HOME / ".local" / "share" / "bankml")).expanduser() | |
| LLAMA_TAG = "b11192" # the engine release bankML is checked against (install.sh LLAMA_TAG) | |
| def _install_env(data: Path) -> dict: | |
| """KEY=value pairs install.sh recorded in <data>/install.env (written with printf %q). Read as data, never executed; | |
| an unreadable or malformed file reads as empty.""" | |
| out = {} | |
| try: | |
| lines = (data / "install.env").read_text(encoding="utf-8").splitlines() | |
| except OSError: | |
| return out | |
| for line in lines: | |
| key, sep, raw = line.partition("=") | |
| if not sep or not key.isidentifier(): | |
| continue | |
| try: | |
| words = shlex.split(raw) | |
| except ValueError: | |
| continue | |
| if len(words) == 1: | |
| out[key] = words[0] | |
| return out | |
| def llama_server(env: Optional[Mapping[str, str]] = None, data: Optional[Path] = None, home: Optional[Path] = None) -> Path: | |
| """The llama-server binary, found exactly as install.sh finds it: BANKML_LLAMA_SERVER; else the INSTALL_LLAMA_SERVER | |
| the installer recorded; else the first executable of the installer's download and the development checkout. When | |
| none is executable, the installer's download path, so a refusal names where `./install.sh engine` puts it.""" | |
| env = os.environ if env is None else env | |
| data = DATA if data is None else data | |
| home = HOME if home is None else home | |
| explicit = env.get("BANKML_LLAMA_SERVER") or _install_env(data).get("INSTALL_LLAMA_SERVER") | |
| if explicit: | |
| return Path(explicit).expanduser() | |
| download = data / f"llama-{LLAMA_TAG}" / "llama-server" | |
| for c in (download, home / "sAGI" / "bonsai" / f"llama-{LLAMA_TAG}" / "llama-server"): | |
| if c.is_file() and os.access(c, os.X_OK): | |
| return c | |
| return download | |
| LLAMA = llama_server() | |
| BANKML = Path(os.environ.get("BANKML_BIN", REPO / "target" / "release" / "bankml")).expanduser() | |
| OLLAMA_STORE = Path(os.environ.get("BANKML_OLLAMA_MODELS", "/usr/share/ollama/.ollama/models")) | |
| LISTEN = os.environ.get("BANKML_SERVE_LISTEN", "127.0.0.1:18093") | |
| UPSTREAM = os.environ.get("BANKML_UPSTREAM", "127.0.0.1:18092") | |
| LOG = Path(os.environ.get("BANKML_UI_STATE", HOME / ".local" / "share" / "bankml" / "savante")).expanduser() / "carrier.log" | |
| DISK_MARGIN = 1_500_000_000 # never fill the disk: keep 1.5 GB free after an import | |
| EMBEDDING_ARCHS = {"bert", "nomic-bert", "jina-bert-v2", "xlm-roberta", "modern-bert", "neo-bert", "t5encoder"} # encoders: they embed, they do not chat | |
| UA = {"User-Agent": "bankml-importer (+https://github.com/cryptoAGI/bankml)"} | |
| # OSI-approved licences (SPDX ids as Hugging Face tags them), plus public-domain dedications | |
| OPEN = {"apache-2.0", "mit", "bsd-2-clause", "bsd-3-clause", "bsd", "isc", "mpl-2.0", "gpl-2.0", "gpl-3.0", "lgpl-2.1", | |
| "lgpl-3.0", "agpl-3.0", "unlicense", "cc0-1.0", "zlib", "artistic-2.0", "epl-2.0", "ecl-2.0", "upl-1.0", "0bsd"} | |
| # the curated catalogue: sha256 and bytes as each repository lists them at `revision` (read 2026-09-28); an import | |
| # re-reads the repository and refuses unless all three still agree | |
| CATALOG = [ | |
| {"id": "bonsai-8b", "title": "Bonsai-8B · 1-bit (Q1_0)", "repo": "PYTHAI/Bonsai-8B-gguf-fork", "file": "Bonsai-8B-Q1_0.gguf", | |
| "revision": "87c4ff9b04e32b7bc71d66d62a803f3e3aa002b4", "bytes": 1158654496, "sha256": "284a335aa3fb2ced3b1b01fcb40b08aa783e3b70832767f0dd2e3fdfa134bd54", "licence": "apache-2.0", "default": True, | |
| "note": "Savante's default carrier: an 8B Qwen3 in 1.16 GB; bankml's Q1_0 kernel is bit-exact against it"}, | |
| {"id": "ternary-bonsai-8b", "title": "Ternary-Bonsai-8B · ternary (Q2_0_g64)", "repo": "PYTHAI/Ternary-Bonsai-8B-gguf-fork", | |
| "file": "Ternary-Bonsai-8B-Q2_0_g64.gguf", "revision": "2445960052d465ace262a43b5f5fef52a0c1e1ef", "bytes": 2310125920, "sha256": "e17b298d84ee78797916ae5c2ecc8211469cc65cccfe3080cd9a9bb503fbc55e", "licence": "apache-2.0", | |
| "note": "the better answers of the two Bonsai-8B, slower in the reference runtime (TECHNICAL.md §I.2)"}, | |
| {"id": "bonsai-4b", "title": "Bonsai-4B · 1-bit (Q1_0)", "repo": "prism-ml/Bonsai-4B-gguf", "file": "Bonsai-4B-Q1_0.gguf", | |
| "revision": "78f2c2bacd0904ffaba24b4873ed975e5818354a", "bytes": 572270624, "sha256": "4524b3f997f0f06444e568d1f26e2efd69effa3218c7ad3047432fb171e42168", "licence": "apache-2.0", "note": "small and quick"}, | |
| {"id": "bonsai-1.7b", "title": "Bonsai-1.7B · 1-bit (Q1_0)", "repo": "prism-ml/Bonsai-1.7B-gguf", "file": "Bonsai-1.7B-Q1_0.gguf", | |
| "revision": "210a9e99f79cb184909d49595906526eb2b3dd9a", "bytes": 248302272, "sha256": "3d7c6c90dd98717a203adb22d5eacd2581850e40aa5327e144b97766cae5f7e3", "licence": "apache-2.0", "note": "the oracle's test model; a quarter gigabyte"}, | |
| {"id": "qwen3-0.6b", "title": "Qwen3-0.6B · Q8_0", "repo": "Qwen/Qwen3-0.6B-GGUF", "file": "Qwen3-0.6B-Q8_0.gguf", | |
| "revision": "23749fefcc72300e3a2ad315e1317431b06b590a", "bytes": 639446688, "sha256": "9465e63a22add5354d9bb4b99e90117043c7124007664907259bd16d043bb031", "licence": "apache-2.0", "note": "Qwen's own GGUF; the smallest Qwen3"}, | |
| {"id": "qwen3-1.7b", "title": "Qwen3-1.7B · Q4_K_M", "repo": "ggml-org/Qwen3-1.7B-GGUF", "file": "Qwen3-1.7B-Q4_K_M.gguf", | |
| "revision": "daeb8e2d528a760970442092f6bf1e55c3b659eb", "bytes": 1282439264, "sha256": "d2387ca2dbfee2ffabce7120d3770dadca0b293052bc2f0e138fdc940d9bc7b5", "licence": "apache-2.0", "note": "a standard 4-bit K-quant"}, | |
| {"id": "qwen3-4b", "title": "Qwen3-4B · Q4_K_M", "repo": "Qwen/Qwen3-4B-GGUF", "file": "Qwen3-4B-Q4_K_M.gguf", | |
| "revision": "bc640142c66e1fdd12af0bd68f40445458f3869b", "bytes": 2497280256, "sha256": "7485fe6f11af29433bc51cab58009521f205840f5b4ae3a32fa7f92e8534fdf5", "licence": "apache-2.0", "note": "Qwen's own GGUF"}, | |
| {"id": "qwen3-8b", "title": "Qwen3-8B · Q4_K_M", "repo": "Qwen/Qwen3-8B-GGUF", "file": "Qwen3-8B-Q4_K_M.gguf", | |
| "revision": "7c41481f57cb95916b40956ab2f0b139b296d974", "bytes": 5027783488, "sha256": "d98cdcbd03e17ce47681435b5150e34c1417f50b5c0019dd560e4882c5745785", "licence": "apache-2.0", | |
| "note": "the full-precision-quality 8B; needs about 5 GB of disk and memory"}, | |
| {"id": "smollm2-1.7b", "title": "SmolLM2-1.7B-Instruct · Q4_K_M", "repo": "HuggingFaceTB/SmolLM2-1.7B-Instruct-GGUF", | |
| "file": "smollm2-1.7b-instruct-q4_k_m.gguf", "revision": "2d4a76a30b4af41ecd395c35725ac11688d4cfe4", "bytes": 1055609536, "sha256": "decd2598bc2c8ed08c19adc3c8fdd461ee19ed5708679d1c54ef54a5a30d4f33", | |
| "licence": "apache-2.0", "note": "Hugging Face's own small model, open data"}, | |
| {"id": "smollm3-3b", "title": "SmolLM3-3B · Q4_K_M", "repo": "ggml-org/SmolLM3-3B-GGUF", "file": "SmolLM3-Q4_K_M.gguf", | |
| "revision": "4965cb60b150737b68a0408c36aeefb65078f894", "bytes": 1915305312, "sha256": "8334b850b7bd46238c16b0c550df2138f0889bf433809008cc17a8b05761863e", "licence": "apache-2.0", "note": "reasoning on or off (/no_think)"}, | |
| {"id": "granite-3.3-2b", "title": "Granite-3.3-2B-Instruct · Q4_K_M", "repo": "ibm-granite/granite-3.3-2b-instruct-GGUF", | |
| "file": "granite-3.3-2b-instruct-Q4_K_M.gguf", "revision": "7cdf86ccd1f1bb3491c9b7017b033f2e51367397", "bytes": 1545303328, "sha256": "ac71e9e32c0bea919b409c5918f69ca74339854b0319c5065e4e9fb6d95c4852", | |
| "licence": "apache-2.0", "note": "IBM's open model, tuned for instructions and tools"}, | |
| # coders (the codephreak and simplecoder agents): open-source only, so StarCoder and WizardCoder are not here | |
| # (OpenRAIL-M / Llama 2 licences). The newest coder, Qwen3-Coder-Next (80B, Apache-2.0), ships as four files of | |
| # 48 GB in all; it is named in those agents' .model as the latest, not offered for import here. | |
| {"id": "qwen2.5-coder-1.5b", "title": "Qwen2.5-Coder-1.5B-Instruct · Q4_K_M · coder", "repo": "Qwen/Qwen2.5-Coder-1.5B-Instruct-GGUF", | |
| "file": "qwen2.5-coder-1.5b-instruct-q4_k_m.gguf", "revision": "f86cb2c1fa58255f8052cc32aeede1b7482d4361", "bytes": 1117320768, | |
| "sha256": "cc324af070c2ecbfd324a30884d2f951a7ff756aba85cb811a6ec436933bb046", "licence": "apache-2.0", "coder": True, | |
| "note": "the coder agents' default on a small machine: Qwen's own GGUF, 1.1 GB"}, | |
| {"id": "qwen2.5-coder-7b", "title": "Qwen2.5-Coder-7B-Instruct · Q4_K_M · coder", "repo": "Qwen/Qwen2.5-Coder-7B-Instruct-GGUF", | |
| "file": "qwen2.5-coder-7b-instruct-q4_k_m.gguf", "revision": "13fb94bfda8c8cf22497dc57b78f391a9acb426a", "bytes": 4683073536, | |
| "sha256": "509287f78cb4d4cf6b3843734733b914b2c158e43e22a7f4bf5e963800894d3c", "licence": "apache-2.0", "coder": True, | |
| "note": "the stronger small coder; needs about 5 GB of disk and memory"}, | |
| {"id": "qwen3.8-27b", "title": "Qwen3.8-27B · UD-Q4_K_M · latest Qwen (code and general)", "repo": "unsloth/Qwen3.8-27B-GGUF", | |
| "file": "Qwen3.8-27B-UD-Q4_K_M.gguf", "revision": "4ca720788d1e01f1bff70c033e0d0028fd02e502", "bytes": 16464440224, | |
| "sha256": "322e194ff79741c7baa497c240f677f54b201b0efab44ca8e50f122b39123482", "licence": "apache-2.0", "coder": True, | |
| "note": "the newest Qwen (Aug 2026), strong at code; 16.5 GB, for a machine with 24 GB or more"}, | |
| ] | |
| # 0.3.4 (O4): models with no published GGUF, converted here from their pinned safetensors by llama.cpp b11192's own | |
| # `convert_hf_to_gguf.py --outtype f16` (commit 171e8846, unmodified). The conversion is reproducible (run twice, the | |
| # same sha256) and was checked tensor by tensor: every GGUF tensor equals the safetensors tensor rounded to f16 (the | |
| # norms kept f32), Q/K rows in llama's NORM-RoPE order. `pin_converted` re-reads the source repository's LFS sha256 | |
| # at the revision before it writes the FORK.json; the conversion itself needs torch and is not run by bankML. | |
| CONVERTED = [ | |
| {"id": "smollm2-135m-instruct", "title": "SmolLM2-135M-Instruct · F16", "file": "SmolLM2-135M-Instruct-F16.gguf", | |
| "bytes": 270885888, "sha256": "e9aba089704487f72efa3c6cbb6d4c748d4c515428247f97679063137e45a222", "licence": "apache-2.0", | |
| "kind": "model", "repo": "HuggingFaceTB/SmolLM2-135M-Instruct", "revision": "12fd25f77366fa6b3b4b768ec3050bf629380bac", | |
| "source_file": "model.safetensors", "source_bytes": 269060552, "source_sha256": "5af571cbf074e6d21a03528d2330792e532ca608f24ac70a143f6b369968ab8c", | |
| "tools": "llama.cpp b11192 convert_hf_to_gguf.py --outtype f16; torch 2.11.0+cpu, transformers 4.57.6, numpy 2.2.6 (b11192's requirements)", | |
| "note": "the base of mindX's lineage family (SmolLM2-135M, Llama architecture); HuggingFaceTB publishes no F16 GGUF of it"}, | |
| {"id": "mindx-gen39", "title": "mindx-gen39 · F16 (mindXtrain39, the last accepted generation)", "file": "mindx-gen39-F16.gguf", | |
| "bytes": 270885600, "sha256": "6b64c748d96ad26fd72402299bd27b2ae82f489bd0469498dd18eb6054058266", "licence": "apache-2.0", | |
| "kind": "dataset", "repo": "PYTHAI/mindXascension", "revision": "4bd31b9db75e3c4c0159af631edc27b1e142ff11", | |
| "source_file": "weights/gen39/ollama_push/merged/model.safetensors", "source_bytes": 269060552, | |
| "source_sha256": "19b62829de298cc06925947976b33d34f9ffb24f5d0e05f349feabae4d83357c", | |
| "tools": "llama.cpp b11192 convert_hf_to_gguf.py --outtype f16; torch 2.11.0+cpu, transformers 5.8.0 (its tokenizer_config.json was " | |
| "written by transformers 5.8.0, which 4.57.6 cannot read), numpy 2.2.6", | |
| "note": "mindX's own generation 39: SmolLM2-135M + its LoRA, merged by mindXtrain; Ollama serves it as mindx-gen39"}, | |
| ] | |
| # ── small helpers ────────────────────────────────────────────────────────────────────────────────────────── | |
| def _get(url: str, timeout=30, raw=False): | |
| with urllib.request.urlopen(urllib.request.Request(url, headers=UA), timeout=timeout) as r: | |
| b = r.read() | |
| return b if raw else json.loads(b) | |
| def licence_open(tag) -> bool: | |
| tags = tag if isinstance(tag, list) else [tag] | |
| return any(isinstance(t, str) and t.lower().removeprefix("license:") in OPEN for t in tags) | |
| def free_bytes(p: Path = MODELS) -> int: | |
| p = p if p.exists() else p.parent | |
| return shutil.disk_usage(p.resolve()).free | |
| def mem_total() -> int: | |
| try: | |
| return int(re.search(r"MemTotal:\s+(\d+)", Path("/proc/meminfo").read_text()).group(1)) * 1024 | |
| except (OSError, AttributeError): | |
| return 0 | |
| def fits(nbytes: int, have: bool = False) -> tuple: | |
| """(ok, why): disk for the file plus the margin (skipped when the file is already here), and a model the memory | |
| can hold (weights ≲ 70 % of RAM; always checked).""" | |
| if not have and nbytes + DISK_MARGIN > free_bytes(): | |
| return False, f"needs {nbytes / 1e9:.1f} GB + {DISK_MARGIN / 1e9:.1f} GB margin; {free_bytes() / 1e9:.1f} GB free on disk" | |
| return fits_memory(nbytes) | |
| def fits_memory(nbytes: int) -> tuple: | |
| mt = mem_total() | |
| if mt and nbytes > 0.7 * mt: | |
| return False, f"{nbytes / 1e9:.1f} GB of weights on a {mt / 1e9:.1f} GB machine would page to swap" | |
| return True, "fits" | |
| def _safe_name(s: str) -> str: | |
| s = re.sub(r"[^A-Za-z0-9._-]+", "-", s).strip("-.") | |
| if not s.endswith(".gguf"): | |
| s += ".gguf" | |
| return s | |
| # ── installed models and their pins ─────────────────────────────────────────────────────────────────────── | |
| def forks() -> dict: | |
| """basename -> (FORK.json path, record) for every pinned file in FORKS.""" | |
| out = {} | |
| for f in sorted(FORKS.glob("*.json")): | |
| try: | |
| j = json.loads(f.read_text(encoding="utf-8")) | |
| except (OSError, ValueError): | |
| continue | |
| for rec in j.get("files") or []: | |
| if isinstance(rec, dict) and str(rec.get("path", "")).endswith(".gguf") and re.fullmatch(r"[0-9a-f]{64}", str(rec.get("sha256", ""))): | |
| out[rec["path"]] = (f, {**rec, "source": j.get("source_repo") or j.get("source"), "licence": j.get("licence_tag")}) | |
| return out | |
| def installed() -> list: | |
| """[{file, path, bytes, pinned, fork, source, licence, catalog}] for every GGUF in MODELS.""" | |
| pins, cat = forks(), {c["file"]: c for c in CATALOG} | |
| out = [] | |
| for p in sorted(MODELS.glob("*.gguf")): | |
| try: | |
| size = p.stat().st_size | |
| except OSError: | |
| continue # a dangling link | |
| fk = pins.get(p.name) | |
| out.append({"file": p.name, "path": str(p), "bytes": size, "pinned": bool(fk), "fork": str(fk[0]) if fk else None, | |
| "source": (fk[1]["source"] if fk else None) or (cat.get(p.name) or {}).get("repo"), | |
| "licence": (fk[1]["licence"] if fk else None) or (cat.get(p.name) or {}).get("licence"), "catalog": (cat.get(p.name) or {}).get("id")}) | |
| return out | |
| def write_fork(file: str, nbytes: int, sha: str, source: str, url: str, revision: str, licence: str, pinned_from: str) -> Path: | |
| FORKS.mkdir(parents=True, exist_ok=True) | |
| f = FORKS / f"{file}.FORK.json" | |
| f.write_text(json.dumps({"kind": "bankml import (weights downloaded, verified against the published sha256)", "source_repo": source, | |
| "source_url": url, "source_revision": revision, "licence_tag": licence, | |
| "imported_at_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), "pinned_from": pinned_from, | |
| "files": [{"path": file, "bytes": nbytes, "sha256": sha}]}, indent=1) + "\n", encoding="utf-8") | |
| return f | |
| def guard(path: Path) -> dict: | |
| """bankml's header guard on a file: {verdict, arch, types, reasons}.""" | |
| p = subprocess.run([str(BANKML), "guard", str(path), "--json"], capture_output=True, text=True, timeout=120) | |
| try: | |
| return json.loads(p.stdout) | |
| except ValueError: | |
| return {"verdict": "refuse", "reasons": [(p.stderr or p.stdout).strip()[-300:] or "bankml guard gave no report"]} | |
| # ── Hugging Face ────────────────────────────────────────────────────────────────────────────────────────── | |
| def hf_parse(url: str) -> tuple: | |
| """(repo, revision|None, file|None) from a Hugging Face URL or an OWNER/REPO id.""" | |
| u = url.strip() | |
| u = re.sub(r"^hf://", "", u) | |
| u = re.sub(r"^https?://(www\.)?(huggingface\.co|hf\.co)/", "", u) | |
| u = u.split("?")[0].split("#")[0].strip("/") | |
| parts = u.split("/") | |
| seg = lambda x: bool(re.fullmatch(r"[A-Za-z0-9._-]+", x)) and x not in (".", "..") | |
| if len(parts) < 2 or not all(seg(x) for x in parts[:2]): | |
| raise ValueError("not a Hugging Face model: give https://huggingface.co/OWNER/REPO (optionally …/blob/REV/FILE.gguf)") | |
| repo = "/".join(parts[:2]) | |
| if len(parts) >= 4 and parts[2] in ("blob", "resolve", "tree"): | |
| rev, f = urllib.parse.unquote(parts[3]), urllib.parse.unquote("/".join(parts[4:])) if parts[4:] else None | |
| if not re.fullmatch(r"[A-Za-z0-9._-]+", rev) or rev in (".", "..") or (f and any(x in ("", ".", "..") for x in f.split("/"))): | |
| raise ValueError("a revision is a branch, tag or commit; a file path has no '.' or '..' segments") | |
| return repo, rev, f | |
| return repo, None, None | |
| def hf_resolve(url: str) -> dict: | |
| """{repo, revision, licence, open, files:[{file, bytes, sha256}], chosen} for a Hugging Face URL.""" | |
| repo, rev, file = hf_parse(url) | |
| m = _get(f"https://huggingface.co/api/models/{repo}" + (f"/revision/{urllib.parse.quote(rev)}" if rev else "")) | |
| sha = m.get("sha") or rev | |
| lic = (m.get("cardData") or {}).get("license") or [t for t in m.get("tags", []) if t.startswith("license:")] | |
| tree = _get(f"https://huggingface.co/api/models/{repo}/tree/{sha}?recursive=true") | |
| files = [{"file": e["path"], "bytes": e.get("size", 0), "sha256": (e.get("lfs") or {}).get("oid")} | |
| for e in tree if e.get("type") == "file" and e["path"].endswith(".gguf") and (e.get("lfs") or {}).get("oid")] | |
| if file and not any(x["file"] == file for x in files): | |
| raise ValueError(f"{file} is not a GGUF in {repo}@{sha[:10]}") | |
| return {"repo": repo, "revision": sha, "licence": lic, "open": licence_open(lic), "files": files, "chosen": file, | |
| "gated": bool(m.get("gated"))} # gated repositories need an account and an accepted agreement: not imported | |
| # ── Ollama ──────────────────────────────────────────────────────────────────────────────────────────────── | |
| def _ollama_name(name: str) -> tuple: | |
| n = name.strip().removeprefix("https://ollama.com/").removeprefix("library/") | |
| n, _, tag = n.partition(":") | |
| if not re.fullmatch(r"[a-z0-9][a-z0-9._-]*(/[a-z0-9][a-z0-9._-]*)?", n) or ".." in n: | |
| raise ValueError(f"not an Ollama model name: {name}") | |
| if tag and (not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._-]*", tag) or ".." in tag): | |
| raise ValueError(f"not an Ollama tag: {tag}") | |
| return n, tag or "latest" | |
| def ollama_search(q: str) -> list: | |
| """ollama.com's search, as [{name, description, sizes, pulls}] (its HTML; there is no JSON search API).""" | |
| import html | |
| page = _get("https://ollama.com/search?q=" + urllib.parse.quote(q.strip()), raw=True).decode("utf-8", "replace") | |
| out = [] | |
| for li in re.findall(r"<li[^>]*>(.*?)</li>", page, flags=re.S): | |
| m = re.search(r'href="/library/([^"/:]+)"', li) | |
| if not m: | |
| continue | |
| d = re.search(r"<p[^>]*>([^<]{8,})</p>", li) | |
| sizes = re.findall(r"text-blue-600[^>]*>\s*([0-9.]+[mbk]|[0-9x.]+b)\s*<", li) | |
| pulls = re.search(r"<span\s*>([0-9.]+[KMB]?)</span>\s*<span[^>]*> Pulls", li) | |
| cloud = "cloud" in li and ">cloud<" in li | |
| out.append({"name": m.group(1), "description": html.unescape(d.group(1).strip()) if d else "", "sizes": sizes, | |
| "pulls": pulls.group(1) if pulls else "", "cloud": cloud}) | |
| return out | |
| def ollama_tags(name: str) -> list: | |
| n, _ = _ollama_name(name) | |
| page = _get(f"https://ollama.com/library/{n}/tags", raw=True).decode("utf-8", "replace") | |
| return sorted(set(re.findall(rf'href="/library/{re.escape(n)}:([^"]+)"', page))) | |
| def _licence_of_text(t: str) -> str | None: | |
| s = t[:4000].lower() | |
| if "gemma terms of use" in s: | |
| return "gemma (not open source)" | |
| if "llama" in s and "community license" in s: | |
| return "llama community licence (not open source)" | |
| if "apache license" in s and "version 2.0" in s: | |
| return "apache-2.0" | |
| if "mit license" in s or "permission is hereby granted, free of charge" in s: | |
| return "mit" | |
| if "gnu general public license" in s: | |
| return "gpl-3.0" if "version 3" in s else "gpl-2.0" | |
| if "redistribution and use in source and binary forms" in s: | |
| return "bsd-3-clause" | |
| return None | |
| def ollama_resolve(name: str) -> dict: | |
| """{repo, tag, file, bytes, sha256, licence, open, url} from the Ollama registry (manifest + licence layer).""" | |
| n, tag = _ollama_name(name) | |
| path = n if "/" in n else f"library/{n}" | |
| req = urllib.request.Request(f"https://registry.ollama.ai/v2/{path}/manifests/{tag}", | |
| headers={**UA, "Accept": "application/vnd.docker.distribution.manifest.v2+json"}) | |
| with urllib.request.urlopen(req, timeout=30) as r: | |
| man = json.loads(r.read()) | |
| layers = man.get("layers") or [] | |
| model = [l for l in layers if l.get("mediaType") == "application/vnd.ollama.image.model"] | |
| if not model: | |
| raise ValueError(f"{n}:{tag} has no model weights in the registry (a cloud-only tag?)") | |
| lic_layer = [l for l in layers if l.get("mediaType") == "application/vnd.ollama.image.license"] | |
| lic = None | |
| if lic_layer: | |
| lic = _licence_of_text(_get(f"https://registry.ollama.ai/v2/{path}/blobs/{lic_layer[0]['digest']}", raw=True).decode("utf-8", "replace")) | |
| d = model[0]["digest"].removeprefix("sha256:") | |
| return {"repo": f"ollama:{path}", "tag": tag, "file": _safe_name(f"{n.replace('/', '-')}-{tag}"), "bytes": model[0]["size"], "sha256": d, | |
| "licence": lic or "unrecognised", "open": bool(lic and licence_open(lic)), "url": f"https://registry.ollama.ai/v2/{path}/blobs/sha256:{d}", | |
| "local": str(OLLAMA_STORE / "blobs" / f"sha256-{d}") if (OLLAMA_STORE / "blobs" / f"sha256-{d}").is_file() else None} | |
| def ollama_local_spec(name: str) -> dict: | |
| """The same record as ollama_resolve(), built from the local Ollama's own manifest and licence layer: offline, and | |
| pinned to what was actually pulled (the registry's tag may have moved since).""" | |
| n, tag = _ollama_name(name) | |
| ns, base = (n.split("/", 1) if "/" in n else ("library", n)) | |
| f = OLLAMA_STORE / "manifests" / "registry.ollama.ai" / ns / base / tag | |
| man = json.loads(f.read_text()) | |
| layers = man.get("layers") or [] | |
| model = [l for l in layers if l.get("mediaType") == "application/vnd.ollama.image.model"] | |
| if not model: | |
| raise ValueError(f"{n}:{tag}: the local manifest has no model layer") | |
| blob = OLLAMA_STORE / "blobs" / model[0]["digest"].replace(":", "-") | |
| lic_layer = [l for l in layers if l.get("mediaType") == "application/vnd.ollama.image.license"] | |
| lic = _licence_of_text((OLLAMA_STORE / "blobs" / lic_layer[0]["digest"].replace(":", "-")).read_text(errors="replace")) if lic_layer else None | |
| d = model[0]["digest"].removeprefix("sha256:") | |
| return {"repo": f"ollama:{ns}/{base}", "tag": tag, "file": _safe_name(f"{n.replace('/', '-')}-{tag}"), "bytes": model[0]["size"], "sha256": d, | |
| "licence": lic or "unrecognised", "open": bool(lic and licence_open(lic)), "url": "", "local": str(blob) if blob.is_file() else None} | |
| def ollama_local() -> list: | |
| """Models the local Ollama already holds: [{name, tag, sha256, bytes, blob}].""" | |
| out = [] | |
| root = OLLAMA_STORE / "manifests" / "registry.ollama.ai" | |
| for f in sorted(root.glob("*/*/*")) if root.is_dir() else []: | |
| try: | |
| man = json.loads(f.read_text()) | |
| except (OSError, ValueError): | |
| continue | |
| for l in man.get("layers") or []: | |
| if l.get("mediaType") == "application/vnd.ollama.image.model": | |
| blob = OLLAMA_STORE / "blobs" / l["digest"].replace(":", "-") | |
| if blob.is_file(): | |
| ns, name = f.parent.parent.name, f.parent.name | |
| out.append({"name": name if ns == "library" else f"{ns}/{name}", "tag": f.name, "sha256": l["digest"].removeprefix("sha256:"), | |
| "bytes": l["size"], "blob": str(blob)}) | |
| return out | |
| # ── the import job (one at a time; the UI polls JOB) ────────────────────────────────────────────────────── | |
| JOB = {"state": "idle", "what": "", "done": 0, "total": 0, "error": None, "result": None, "started": None, "seq": 0, "cancel": False} | |
| _LOCK = threading.Lock() | |
| def _hash_file(p: Path, on=None) -> str: | |
| h = hashlib.sha256() | |
| with open(p, "rb") as f: | |
| while b := f.read(1 << 22): | |
| h.update(b) | |
| if on: | |
| on(len(b)) | |
| return h.hexdigest() | |
| def _download(url: str, dest: Path, nbytes: int, want: str) -> None: | |
| """Stream url → dest.part, hashing on the way; rename only if the sha256 is the published one. Resumes a .part; | |
| never writes past the published size.""" | |
| part = dest.with_name(dest.name + ".part") | |
| h = hashlib.sha256() | |
| have = part.stat().st_size if part.exists() else 0 | |
| if have > nbytes: | |
| part.unlink() | |
| have = 0 | |
| if have: # resume: hash what is there first | |
| with open(part, "rb") as f: | |
| while b := f.read(1 << 22): | |
| h.update(b) | |
| JOB["done"] = have | |
| if have < nbytes: | |
| req = urllib.request.Request(url, headers={**UA, **({"Range": f"bytes={have}-"} if have else {})}) | |
| with urllib.request.urlopen(req, timeout=60) as r: | |
| cr = r.headers.get("Content-Range") or "" | |
| if have and (r.status != 206 or not cr.startswith(f"bytes {have}-")): # the range was ignored or misplaced: start over | |
| have = 0 | |
| h = hashlib.sha256() | |
| JOB["done"] = 0 | |
| with open(part, "ab" if have else "wb") as f: | |
| while b := r.read(1 << 20): | |
| if JOB.get("cancel"): | |
| raise RuntimeError("cancelled (the partial file is kept; importing again resumes it)") | |
| if JOB["done"] + len(b) > nbytes: | |
| f.close() | |
| part.unlink(missing_ok=True) | |
| raise RuntimeError(f"the source sent more than the published {nbytes} bytes: discarded") | |
| f.write(b) | |
| h.update(b) | |
| JOB["done"] += len(b) | |
| got = h.hexdigest() | |
| if got != want: | |
| part.unlink(missing_ok=True) | |
| raise RuntimeError(f"sha256 {got[:16]}… is not the published {want[:16]}…: discarded") | |
| os.replace(part, dest) | |
| def import_spec(spec: dict) -> dict: | |
| """Bring one model in. spec: {file, bytes, sha256, url, repo, revision, licence, pinned_from, local?}.""" | |
| if not licence_open(spec.get("licence")): | |
| raise PermissionError(f"{spec.get('repo')}: licence {spec.get('licence')!r} is not open source; bankml imports open-source models only") | |
| if not re.fullmatch(r"[0-9a-f]{64}", spec.get("sha256") or ""): | |
| raise ValueError("no published sha256 for this file: nothing to pin it to, refused") | |
| dest = MODELS / _safe_name(spec["file"]) | |
| ok, why = fits(spec["bytes"], have=dest.exists() or bool(spec.get("local"))) | |
| if not ok: | |
| raise RuntimeError(why) | |
| MODELS.mkdir(parents=True, exist_ok=True) | |
| JOB.update(total=spec["bytes"], done=0) | |
| if dest.exists(): | |
| JOB["what"] = f"checking {dest.name} already here" | |
| if _hash_file(dest, lambda n: JOB.__setitem__("done", JOB["done"] + n)) != spec["sha256"]: | |
| raise RuntimeError(f"{dest.name} exists but is not the published file (sha256 differs); move it aside first") | |
| elif spec.get("local"): # the local Ollama already has it: link, then hash | |
| JOB["what"] = f"adopting {spec['file']} from the local Ollama (no download)" | |
| if _hash_file(Path(spec["local"]), lambda n: JOB.__setitem__("done", JOB["done"] + n)) != spec["sha256"]: | |
| raise RuntimeError("the local Ollama blob does not hash to its digest: refused") | |
| os.symlink(spec["local"], dest) | |
| else: | |
| JOB["what"] = f"downloading {spec['file']} from {spec['repo']}" | |
| _download(spec["url"], dest, spec["bytes"], spec["sha256"]) | |
| g = guard(dest) | |
| if g.get("verdict") != "play": | |
| dest.unlink(missing_ok=True) # a symlink goes; the Ollama blob it points at stays | |
| raise RuntimeError(f"bankml's guard refused {dest.name}: {'; '.join(g.get('reasons') or [])}") | |
| fk = write_fork(dest.name, spec["bytes"], spec["sha256"], spec["repo"], spec.get("page") or spec["url"], spec.get("revision", ""), | |
| spec["licence"] if isinstance(spec["licence"], str) else ",".join(spec["licence"]), spec["pinned_from"]) | |
| return {"file": dest.name, "path": str(dest), "fork": str(fk), "arch": g.get("arch"), "types": g.get("types")} | |
| def spec_catalog(cid: str) -> dict: | |
| c = next(x for x in CATALOG if x["id"] == cid) | |
| r = hf_resolve(f"https://huggingface.co/{c['repo']}/blob/{c['revision']}/{c['file']}") | |
| f = next(x for x in r["files"] if x["file"] == c["file"]) | |
| if not (f["sha256"] == c["sha256"] and r["revision"] == c["revision"] and f["bytes"] == c["bytes"]): | |
| raise RuntimeError(f"{c['repo']} no longer lists the catalogued file (sha256/size/revision changed): refused") | |
| return spec_hf(r, c["file"]) | |
| def spec_hf(r: dict, file: str) -> dict: | |
| f = next(x for x in r["files"] if x["file"] == file) | |
| lic = r["licence"] if isinstance(r["licence"], str) else ",".join(t.removeprefix("license:") for t in r["licence"]) | |
| return {"file": _safe_name(file.replace("/", "-")), "bytes": f["bytes"], "sha256": f["sha256"], "repo": r["repo"], "revision": r["revision"], "licence": lic, | |
| "url": f"https://huggingface.co/{r['repo']}/resolve/{r['revision']}/{urllib.parse.quote(file)}", | |
| "page": f"https://huggingface.co/{r['repo']}/blob/{r['revision']}/{file}", "pinned_from": f"Hugging Face LFS sha256 of {file} at {r['revision']}"} | |
| def spec_ollama(o: dict) -> dict: | |
| return {"file": o["file"], "bytes": o["bytes"], "sha256": o["sha256"], "repo": o["repo"], "revision": o["tag"], "licence": o["licence"], | |
| "url": o["url"], "page": f"https://ollama.com/{o['repo'].removeprefix('ollama:').removeprefix('library/')}:{o['tag']}", "local": o.get("local"), | |
| "pinned_from": "Ollama registry layer digest (sha256 of the GGUF)"} | |
| def start_job(what: str, fn, *a) -> bool: | |
| """Run fn(*a) in the background as the one import/switch job; False if one is already running.""" | |
| with _LOCK: | |
| if JOB["state"] == "running": | |
| return False | |
| JOB.update(state="running", what=what, done=0, total=0, error=None, result=None, started=time.time(), cancel=False) | |
| def run(): | |
| try: | |
| JOB["result"] = fn(*a) | |
| JOB["state"] = "done" | |
| except Exception as e: # noqa: BLE001 | |
| JOB.update(state="error", error=f"{type(e).__name__}: {e}") | |
| finally: | |
| JOB["seq"] += 1 # the UI refreshes its lists when this moves | |
| threading.Thread(target=run, daemon=True).start() | |
| return True | |
| def cancel_job() -> bool: | |
| """Ask a running download to stop at its next chunk (the .part is kept for resume).""" | |
| if JOB["state"] == "running": | |
| JOB["cancel"] = True | |
| return True | |
| return False | |
| # ── the carrier: bankml serve + llama-server ────────────────────────────────────────────────────────────── | |
| def serve_status(timeout=3) -> dict: | |
| try: | |
| with urllib.request.urlopen(f"http://{LISTEN}/bankml", timeout=timeout) as r: | |
| return json.loads(r.read()) | |
| except Exception as e: # noqa: BLE001 | |
| return {"error": str(e)} | |
| def _listeners() -> dict: | |
| """port -> pid for the carrier's two ports (from ss), only if the process is this user's llama-server or bankml. | |
| Matches the configured host (127.0.0.1, 0.0.0.0, [::1] …) exactly.""" | |
| out = {} | |
| try: | |
| s = subprocess.run(["ss", "-ltnpH"], capture_output=True, text=True, timeout=10).stdout | |
| except (OSError, subprocess.SubprocessError): | |
| return out | |
| for addr in (LISTEN, UPSTREAM): | |
| host, port = addr.rsplit(":", 1) | |
| hosts = {host, f"[{host.strip('[]')}]", "*"} if host in ("0.0.0.0", "::", "[::]") else {host, f"[{host.strip('[]')}]"} | |
| for line in s.splitlines(): | |
| cols = line.split() | |
| h, _, pt = cols[3].rpartition(":") if len(cols) >= 4 else ("", "", "") | |
| if pt != port or h not in hosts: | |
| continue | |
| m = re.search(r"pid=(\d+)", line) | |
| if not m: | |
| continue | |
| pid = int(m.group(1)) | |
| try: | |
| exe = Path(f"/proc/{pid}/cmdline").read_bytes().split(b"\0")[0].decode() | |
| mine = os.stat(f"/proc/{pid}").st_uid == os.getuid() | |
| except OSError: | |
| continue | |
| if mine and Path(exe).name in ("llama-server", "bankml"): | |
| out[port] = pid | |
| return out | |
| def _kill(pid: int, sig) -> None: | |
| try: | |
| os.kill(pid, sig) | |
| except ProcessLookupError: | |
| pass | |
| def _stop_carrier(): | |
| for pid in set(_listeners().values()): | |
| _kill(pid, signal.SIGTERM) | |
| for _ in range(100): | |
| if not _listeners(): | |
| return | |
| time.sleep(0.2) | |
| for pid in set(_listeners().values()): | |
| _kill(pid, signal.SIGKILL) | |
| time.sleep(0.5) | |
| if _listeners(): | |
| raise RuntimeError(f"the carrier's ports are still held by {sorted(set(_listeners().values()))}; stop them by hand") | |
| CTX = int(os.environ.get("BANKML_CTX", "2048")) # the engine's context: 2048 keeps an 8B model's KV cache near 0.3 GB | |
| THREADS = int(os.environ.get("BANKML_THREADS_SERVE", "3")) | |
| RESOURCES = LOG.parent / "resources.json" | |
| SLOTS = LOG.parent / "slots" # the engine's saved KV slots (--slot-save-path): a restart restores the system prompt # the operator's CPU and RAM choice (the Resources sliders), used by every start | |
| OVERHEAD = 250_000_000 # llama-server's compute buffers and runtime beside weights and KV (measured order of magnitude) | |
| CTX_MIN, CTX_MAX = 512, 32768 | |
| def resources() -> dict: | |
| """{"threads", "ctx", "ram_gb", "spec_ngram", "engine", "gpu_limit"}: the saved choice, else the defaults | |
| (BANKML_THREADS_SERVE, BANKML_CTX). engine: "auto" (bankML's own forward pass for the ternary Qwen3 files, where it | |
| is about 8x llama-server; llama-server otherwise), "native" or "llama.cpp". gpu_limit (0.3.7): the share of the | |
| GPU's memory and time the native engine may use (BANKML_GPU_LIMIT), 0 for none (BANKML_GPU=off).""" | |
| r = {"threads": THREADS, "ctx": CTX, "ram_gb": None, "spec_ngram": False, "engine": "auto", "gpu_limit": 0.8} | |
| try: | |
| r.update({k: v for k, v in json.loads(RESOURCES.read_text()).items() if k in r}) | |
| except (OSError, ValueError): | |
| pass | |
| return r | |
| def plan(model: Path, ram_gb: float) -> dict: | |
| """What a RAM budget buys for a model: weights stay resident, the rest is KV cache (f16, the guard's | |
| bytes-per-token for this architecture) after the engine's overhead. ctx is rounded down to 256.""" | |
| g = guard(model) | |
| kv = int(g.get("kv_f16_bytes_per_token") or 0) or 147_456 # Qwen3-8B's, if the guard cannot say | |
| w = model.stat().st_size | |
| room = int(ram_gb * 1e9) - w - OVERHEAD | |
| ctx = max(0, min(CTX_MAX, room // kv // 256 * 256)) | |
| return {"ctx": ctx, "fits": ctx >= CTX_MIN, "kv_gb": round(ctx * kv / 1e9, 2), "weights_gb": round(w / 1e9, 2), | |
| "kv_bytes_per_token": kv, "overhead_gb": OVERHEAD / 1e9, "arch": g.get("arch"), | |
| "min_ram_gb": round((w + OVERHEAD + CTX_MIN * kv) / 1e9, 2)} | |
| def usage() -> dict: | |
| """What the carrier uses now, from bankml itself (`GET /bankml/usage`, sys.rs reading /proc — bankml's psutil, | |
| no crates); the Python reading below is the fallback for a serve older than 0.1.8.""" | |
| try: | |
| with urllib.request.urlopen(f"http://{LISTEN}/bankml/usage", timeout=5) as r: | |
| u = json.loads(r.read()) | |
| return {"pids": [p["pid"] for p in u["processes"]], "rss_gb": round(u["rss_bytes"] / 1e9, 2), "cpu_pct": round(u["cpu_percent"]), | |
| "cores": u["cores"], "mem_total_gb": round(u["mem_total_bytes"] / 1e9, 1), "mem_available_gb": round(u["mem_available_bytes"] / 1e9, 2), | |
| "source": u["source"], "processes": u["processes"]} | |
| except (OSError, ValueError, KeyError): | |
| pass | |
| pids = sorted(set(_listeners().values())) | |
| def ticks(pid): | |
| try: | |
| f = Path(f"/proc/{pid}/stat").read_text().rsplit(")", 1)[1].split() | |
| return int(f[11]) + int(f[12]) # utime + stime | |
| except (OSError, IndexError, ValueError): | |
| return 0 | |
| def rss(pid): | |
| try: | |
| return int(re.search(r"VmRSS:\s+(\d+)", Path(f"/proc/{pid}/status").read_text()).group(1)) * 1024 | |
| except (OSError, AttributeError): | |
| return 0 | |
| t0 = {p: ticks(p) for p in pids} | |
| time.sleep(0.5) | |
| hz = os.sysconf("SC_CLK_TCK") | |
| cpu = sum(ticks(p) - t0[p] for p in pids) / hz / 0.5 * 100 | |
| try: | |
| avail = int(re.search(r"MemAvailable:\s+(\d+)", Path("/proc/meminfo").read_text()).group(1)) * 1024 | |
| except (OSError, AttributeError): | |
| avail = 0 | |
| return {"pids": pids, "rss_gb": round(sum(rss(p) for p in pids) / 1e9, 2), "cpu_pct": round(cpu), "cores": os.cpu_count(), | |
| "mem_total_gb": round(mem_total() / 1e9, 1), "mem_available_gb": round(avail / 1e9, 2), "source": "sAGI/models.py (/proc)"} | |
| def allow_origin() -> str | None: | |
| """The one web page bankml serve lets in (`--allow-origin`): BANKML_ALLOW_ORIGIN, else what `./install.sh --space` | |
| recorded; None (this computer only) when neither is set.""" | |
| o = os.environ.get("BANKML_ALLOW_ORIGIN") | |
| if o is None: | |
| o = _install_env(DATA).get("INSTALL_ALLOW_ORIGIN", "") | |
| return o.strip() or None | |
| def native_for(model: Path, engine: str | None = None) -> bool: | |
| """Whether the carrier answers from bankML's own forward pass (`bankml serve --native`, 0.3.0) for this model: | |
| "native" always (bankml refuses a model its forward pass does not run), "llama.cpp" never, "auto" for the | |
| ternary (Q2_0_g64) files — there bankML is about 8x llama-server with the same tokens; on the 1-bit files | |
| llama-server is still faster.""" | |
| e = engine or resources().get("engine", "auto") | |
| return e == "native" or (e == "auto" and "Q2_0" in model.name) | |
| def apply_resources(threads: int, ram_gb: float, busy=lambda: False, spec_ngram: bool = False, engine: str = "auto", | |
| gpu_limit: float | None = None) -> dict: | |
| """Save the choice and restart the carrier on the same model with it (a verified switch, with rollback).""" | |
| threads = max(1, min(int(threads), os.cpu_count() or 1)) | |
| st = serve_status() | |
| sha = (st.get("verified") or {}).get("model_sha256") | |
| pins = forks() | |
| path, fork = _pinned_path(sha, st.get("model"), pins) if sha else (None, None) | |
| if path is None: | |
| raise RuntimeError("no verified carrier is running; choose a model first (the settings are saved and used when it starts)") | |
| pl = plan(path, ram_gb) | |
| if not pl["fits"]: | |
| raise RuntimeError(f"{ram_gb:.1f} GB cannot hold {path.name}: it needs at least {pl['min_ram_gb']} GB") | |
| prev = resources() # the last settings that ran, restored if these do not | |
| RESOURCES.parent.mkdir(parents=True, exist_ok=True) | |
| gl = prev.get("gpu_limit", 0.8) if gpu_limit is None else max(0.0, min(1.0, float(gpu_limit))) | |
| RESOURCES.write_text(json.dumps({"threads": threads, "ctx": pl["ctx"], "ram_gb": ram_gb, "spec_ngram": bool(spec_ngram), | |
| "engine": engine if engine in ("auto", "native", "llama.cpp") else "auto", "gpu_limit": gl}) + "\n") | |
| if busy(): | |
| raise RuntimeError("saved; an answer is being written, so the engine restarts with these settings on the next switch") | |
| JOB["what"] = f"restarting {path.name} with {threads} threads and a {pl['ctx']}-token context" | |
| _stop_carrier() | |
| try: | |
| return _start_carrier(path, fork, sha) | |
| except Exception as first: | |
| RESOURCES.write_text(json.dumps(prev) + "\n") | |
| try: | |
| _stop_carrier() | |
| _start_carrier(path, fork, sha) | |
| except Exception as again: # noqa: BLE001 | |
| raise RuntimeError(f"{first}; restoring the previous settings also failed: {again}") from first | |
| raise | |
| def _start_carrier(model: Path, fork: Path, want_sha: str | None = None, threads=None, ctx=None, wait=1800) -> dict: | |
| """Start `bankml serve --spawn` and wait until it answers verified, with `want_sha` when given. bankml hashes the | |
| whole file before it binds, so the wait is on the process, not on the ports: it fails only when the process exits | |
| or the wait runs out, and then the process group is killed so no second carrier is left behind.""" | |
| LOG.parent.mkdir(parents=True, exist_ok=True) | |
| with open(LOG, "ab") as log: | |
| at = log.seek(0, 2) | |
| log.write(f"\n# {time.strftime('%Y-%m-%d %H:%M:%S')} bankml serve {model.name}\n".encode()) | |
| log.flush() | |
| SLOTS.mkdir(parents=True, exist_ok=True) | |
| r = resources() | |
| if not ctx and r.get("ram_gb"): | |
| pl = plan(model, float(r["ram_gb"])) | |
| if not pl["fits"]: | |
| raise RuntimeError(f"the saved RAM budget ({r['ram_gb']} GB) cannot hold {model.name}: it needs at least {pl['min_ram_gb']} GB") | |
| ctx = pl["ctx"] | |
| n_threads, n_ctx = str(threads or resources()["threads"]), str(ctx or resources()["ctx"]) | |
| if native_for(model): | |
| # bankML's own forward pass answers, on the gateway and on the engine address (0.3.0) | |
| # 0.3.8: the native engine saves and restores slots too (Savante's warm start) | |
| cmd = [str(BANKML), "serve", str(model), "--fork", str(fork), "--native", "--upstream", UPSTREAM, "--listen", LISTEN, "--ctx", n_ctx, | |
| "--slot-dir", str(SLOTS)] | |
| gl = float(resources().get("gpu_limit", 0.8)) | |
| env = {**os.environ, "BANKML_THREADS": n_threads, **({"BANKML_GPU": "off"} if gl <= 0 else {"BANKML_GPU_LIMIT": f"{gl:.2f}"})} | |
| else: | |
| if not (LLAMA.is_file() and os.access(LLAMA, os.X_OK)): | |
| raise RuntimeError(f"no llama-server at {LLAMA}: run ./install.sh engine, or set BANKML_LLAMA_SERVER to a llama.cpp {LLAMA_TAG} build") | |
| cmd = ([str(BANKML), "serve", str(model), "--fork", str(fork), "--spawn", str(LLAMA), "--upstream", UPSTREAM, "--listen", LISTEN, | |
| "--threads", n_threads, "--ctx", n_ctx] + (["--spec-ngram"] if resources().get("spec_ngram") else []) + ["--slot-dir", str(SLOTS)]) | |
| env = None | |
| if allow_origin(): | |
| # ./install.sh --space: bankML's Hugging Face page may talk to this serve from the browser (CORS for it alone) | |
| cmd += ["--allow-origin", allow_origin()] | |
| proc = subprocess.Popen(cmd, stdout=log, stderr=log, stdin=subprocess.DEVNULL, start_new_session=True, cwd=str(REPO), env=env) | |
| t0 = time.time() | |
| why = "timed out" # ctx: per model — the saved RAM budget is re-planned for this model's weights and KV size | |
| while time.time() - t0 < wait: | |
| st = serve_status(2) | |
| v = st.get("verified") or {} | |
| if v and (want_sha is None or v.get("model_sha256") == want_sha): | |
| return st | |
| if v: | |
| why = f"another carrier answers (sha256 {str(v.get('model_sha256'))[:12]}…), not {model.name}" | |
| break | |
| if proc.poll() is not None: | |
| why = f"exited with {proc.returncode}" | |
| break | |
| time.sleep(1) | |
| try: | |
| os.killpg(proc.pid, signal.SIGKILL) # serve and the llama-server it spawned share the group | |
| except ProcessLookupError: | |
| pass | |
| proc.wait(timeout=10) | |
| with open(LOG, "rb") as f: | |
| f.seek(at) | |
| said = [l for l in f.read().decode(errors="replace").splitlines()[1:] if l.strip()] | |
| raise RuntimeError(f"bankml serve did not come up with {model.name} ({why}): " + (" / ".join(said[-3:])[-400:] or "no output")) | |
| def _pinned_path(sha: str, model: str | None, pins: dict): | |
| """(path, fork) for a verified sha256: a file in MODELS first, else the carrier's own path if a pin names it.""" | |
| for f, (fk, rec) in pins.items(): | |
| if rec["sha256"] == sha and (MODELS / f).exists(): | |
| return MODELS / f, fk | |
| if model: | |
| m = Path(model) | |
| rec = pins.get(m.name) | |
| if rec and rec[1]["sha256"] == sha and m.exists(): | |
| return m, rec[0] | |
| return None, None | |
| def switch(file: str, busy=lambda: False) -> dict: | |
| """Make `file` (in MODELS, pinned) the carrier: success only when bankml serve answers verified with this file's | |
| sha256. If the new one fails, the previous model is restored (found by its verified sha256).""" | |
| if busy(): | |
| raise RuntimeError("an answer is being written; switch when it is done") | |
| p = MODELS / file | |
| if not p.exists(): | |
| raise FileNotFoundError(f"{file} is not in {MODELS}") | |
| ok, why = fits_memory(p.stat().st_size) | |
| if not ok: | |
| raise RuntimeError(why) | |
| arch = guard(p).get("arch") | |
| if arch in EMBEDDING_ARCHS: | |
| raise RuntimeError(f"{file} is an embedding model ({arch}); it cannot be the chat carrier (see docs/embedding.md)") | |
| pins = forks() | |
| if file not in pins: | |
| adopt(file) | |
| pins = forks() | |
| want = pins[file][1]["sha256"] | |
| prev = serve_status() | |
| prev_sha = (prev.get("verified") or {}).get("model_sha256") | |
| if prev_sha == want: | |
| return prev | |
| prev_path, prev_fork = _pinned_path(prev_sha, prev.get("model"), pins) if prev_sha else (None, None) | |
| JOB["what"] = f"stopping the carrier, starting {file} (bankml hashes the whole file, then llama-server loads it)" | |
| try: | |
| _stop_carrier() | |
| return _start_carrier(p, pins[file][0], want) | |
| except Exception: | |
| if prev_path: | |
| JOB["what"] = f"{file} failed; restoring {prev_path.name}" | |
| _stop_carrier() | |
| _start_carrier(prev_path, prev_fork, prev_sha) | |
| raise | |
| def pin_converted(file: str) -> Path: | |
| """Pin a converted model (CONVERTED): the file must hash to the recorded conversion, and the source repository must | |
| still list the source safetensors with the recorded sha256 at the revision (read now). The FORK.json says how the | |
| file was made.""" | |
| c = next((x for x in CONVERTED if x["file"] == file), None) | |
| if not c: | |
| raise RuntimeError(f"{file} is not a recorded conversion") | |
| if not licence_open(c["licence"]): | |
| raise PermissionError(f"{c['repo']}: licence {c['licence']!r} is not open source") | |
| kind = "datasets/" if c["kind"] == "dataset" else "" | |
| folder = c["source_file"].rsplit("/", 1)[0] if "/" in c["source_file"] else "" | |
| listing = _get(f"https://huggingface.co/api/{'datasets' if kind else 'models'}/{c['repo']}/tree/{c['revision']}/{folder}") | |
| src = next((x for x in listing if x.get("path") == c["source_file"]), None) | |
| if not src or (src.get("lfs") or {}).get("oid") != c["source_sha256"] or src.get("size") != c["source_bytes"]: | |
| raise RuntimeError(f"{c['repo']}@{c['revision'][:12]} no longer lists {c['source_file']} with the recorded sha256: refused") | |
| p = MODELS / file | |
| JOB.update(what=f"hashing {file} to pin it", total=c["bytes"], done=0) | |
| if _hash_file(p, lambda n: JOB.__setitem__("done", JOB["done"] + n)) != c["sha256"]: | |
| raise RuntimeError(f"{file} is not the recorded conversion (sha256 differs): refused") | |
| g = guard(p) | |
| if g.get("verdict") != "play": | |
| raise RuntimeError(f"bankml's guard refused {file}: {'; '.join(g.get('reasons') or [])}") | |
| page = f"https://huggingface.co/{'datasets/' if kind else ''}{c['repo']}/blob/{c['revision']}/{c['source_file']}" | |
| f = write_fork(file, c["bytes"], c["sha256"], c["repo"], page, c["revision"], c["licence"], | |
| f"sha256 of the GGUF that {c['tools'].split(';')[0]} makes from {c['source_file']} (LFS sha256 {c['source_sha256']}) at {c['revision']}") | |
| j = json.loads(f.read_text(encoding="utf-8")) | |
| j["kind"] = "bankml conversion (safetensors pinned by LFS sha256, converted by llama.cpp b11192, checked tensor by tensor)" | |
| j["converted_from"] = {"repo": c["repo"], "revision": c["revision"], "file": c["source_file"], "bytes": c["source_bytes"], | |
| "sha256": c["source_sha256"], "tools": c["tools"]} | |
| f.write_text(json.dumps(j, indent=1) + "\n", encoding="utf-8") | |
| return f | |
| def adopt(file: str) -> Path: | |
| """Pin a file that is already here but has no FORK.json, if the catalogue knows it: hash it and compare to the | |
| repository's published sha256 (fetched now, not trusted from the catalogue's prefix alone). A recorded conversion | |
| (CONVERTED) is pinned by `pin_converted`.""" | |
| if any(x["file"] == file for x in CONVERTED): | |
| return pin_converted(file) | |
| c = next((x for x in CATALOG if x["file"] == file), None) | |
| if not c: | |
| raise RuntimeError(f"{file} has no FORK.json pin and is not in the catalogue: import it from its source to pin it") | |
| s = spec_catalog(c["id"]) | |
| JOB.update(what=f"hashing {file} to pin it", total=s["bytes"], done=0) | |
| if _hash_file(MODELS / file, lambda n: JOB.__setitem__("done", JOB["done"] + n)) != s["sha256"]: | |
| raise RuntimeError(f"{file} is not the published file (sha256 differs): refused") | |
| return write_fork(file, s["bytes"], s["sha256"], s["repo"], s["page"], s["revision"], s["licence"], s["pinned_from"]) | |
| def first_run(busy=lambda: False) -> dict: | |
| """Seamless start: if no carrier answers, bring in Bonsai-8B (download only if absent) and start it.""" | |
| st = serve_status() | |
| if "verified" in st: | |
| return st | |
| c = next(x for x in CATALOG if x.get("default")) | |
| p = MODELS / c["file"] | |
| if p.is_symlink() and not p.exists(): # a dangling link (the target moved): replace it with the real file | |
| p.unlink() | |
| if not p.exists(): | |
| import_spec(spec_catalog(c["id"])) | |
| return switch(c["file"], busy) | |
| if __name__ == "__main__": # python3 sAGI/models.py [list | catalog | import ID|URL|ollama:NAME:TAG | use FILE | first-run | search Q] | |
| import sys | |
| a = sys.argv[1:] or ["list"] | |
| if a[0] == "list": | |
| for m in installed(): | |
| print(f"{'pinned ' if m['pinned'] else 'UNPINNED'} {m['bytes'] / 1e9:5.2f} GB {m['file']} ({m['source'] or '?'}, {m['licence'] or '?'})") | |
| print("carrier:", Path(serve_status().get("model", "none")).name) | |
| elif a[0] == "catalog": | |
| for c in CATALOG: | |
| print(f"{c['id']:18} {c['bytes'] / 1e9:5.2f} GB {c['title']} [{fits(c['bytes'], (MODELS / c['file']).exists())[1]}]") | |
| elif a[0] == "search": | |
| for r in ollama_search(" ".join(a[1:])): | |
| print(f"{r['name']:24} {','.join(r['sizes']):20} {r['description'][:80]}") | |
| elif a[0] == "import": | |
| t = a[1] | |
| loc = {f"{m['name']}:{m['tag']}" for m in ollama_local()} | |
| name = t.removeprefix("ollama:") | |
| spec = (spec_ollama(ollama_local_spec(name) if name in loc or f"{name}:latest" in loc else ollama_resolve(name)) if t.startswith("ollama:") else | |
| spec_catalog(t) if any(c["id"] == t for c in CATALOG) else None) | |
| if spec is None: | |
| r = hf_resolve(t) | |
| spec = spec_hf(r, r["chosen"] or (sys.exit("pick a file: " + ", ".join(f["file"] for f in r["files"])) if len(r["files"]) != 1 else r["files"][0]["file"])) | |
| print(json.dumps(import_spec(spec), indent=1)) | |
| elif a[0] == "use": | |
| print(json.dumps(switch(a[1]), indent=1)) | |
| elif a[0] == "first-run": | |
| print(json.dumps(first_run(), indent=1)) | |