// SPDX-License-Identifier: MIT OR Apache-2.0 // Ask bankML, from this static page: two ways, never confused. // · Your own bankML — this browser talks to `bankml serve` on the visitor's machine (started with // --allow-origin for this page): bankML's verified arithmetic, the visitor's CPU and RAM, a receipt on every // answer whose sha256 this page checks. Free. // · A Hugging Face provider — sign in with Hugging Face; a provider-hosted model answers under bankML's persona, // billed to the visitor's inference quota. NOT bankML's arithmetic, no receipt; said on every answer. // Nothing an engine or a provider returns becomes markup: text only. import { oauthLoginUrl, oauthHandleRedirectIfPresent } from "https://cdn.jsdelivr.net/npm/@huggingface/hub@2.17.1/+esm"; import { InferenceClient } from "https://cdn.jsdelivr.net/npm/@huggingface/inference@4.13.28/+esm"; const $ = (id) => document.getElementById(id); const SESSION = "bankml.oauth"; const DEFAULT_ENDPOINT = "http://127.0.0.1:18093"; const DEFAULT_MODEL = "Qwen/Qwen3-8B"; const history = []; let persona = null; let oauth = null; const mode = () => document.querySelector('input[name="mode"]:checked').value; // ── the persona: the same file the bankML console speaks from ───────────────────────────────────────────────── async function loadPersona() { try { const r = await fetch("sAGI/personas/bankml.persona", { cache: "no-store" }); persona = await r.json(); } catch (e) { persona = null; $("askstatus").textContent = "the persona could not be read: " + e; } } // ── the SELF block, as the console builds it: one sentence per measurement, "not measured" when it was not ────── const get = async (ep, path) => { try { const r = await fetch(ep + path, { cache: "no-store" }); return r.ok ? await r.json() : null; } catch { return null; } }; async function selfText(ep) { const [b, u, m] = [await get(ep, "/bankml") || {}, await get(ep, "/bankml/usage") || {}, await get(ep, "/bankml/metrics") || {}]; const last = (m.records || [{}]).slice(-1)[0] || {}; const v = (x, unit = "", scale = 1, d = 1) => (x === null || x === undefined ? "not measured" : (x * scale).toFixed(d) + unit); const ver = b.verified || {}; return [ `- the model I am running: ${ver.name || b.resident || "none"} (sha256 ${String(ver.model_sha256 || "").slice(0, 16)}…), bankML ${ver.bankml}`, `- tokens I have read in total (prompts): ${v(m.prompt_tokens, "", 1, 0)}`, `- tokens I have written in total (answers): ${v(m.completion_tokens, "", 1, 0)}`, `- time to first token of my last answer: ${v(last.ttft_ms, " milliseconds", 1, 0)}`, `- prompt reading speed of my last answer: ${v(last.prompt_tps, " tokens per second")}`, `- generation speed of my last answer: ${v(last.eval_tps, " tokens per second")}`, `- CPU use right now: ${v(u.cpu_percent, " percent of one core")}`, `- memory I hold (RSS): ${v(u.rss_bytes, " GB", 1e-9, 2)}; memory still available on this machine: ${v(u.mem_available_bytes, " GB", 1e-9, 2)}`, `- power the CPU package draws: ${v(u.package_watts, " watts")}; energy per token I write: ${v(m.joules_per_token, " joules", 1, 3)}`, ].join("\n"); } // ── the conversation ────────────────────────────────────────────────────────────────────────────────────────── function bubble(role, text) { const d = document.createElement("div"); d.className = "msg " + role; d.textContent = text; $("chat").append(d); d.scrollIntoView({ block: "nearest" }); return d; } function note(el, text, cls) { const p = document.createElement("div"); p.className = "meta " + (cls || ""); p.textContent = text; el.after(p); return p; } // "…" alone looks like no reply: say what is happening and how long it has taken, until the first piece arrives function waiting(out, what) { const t0 = Date.now(); const tick = () => { if (out.dataset.started !== "1") out.textContent = `… ${what} · ${Math.round((Date.now() - t0) / 1000)} s`; }; tick(); const id = setInterval(tick, 1000); return () => { out.dataset.started = "1"; clearInterval(id); }; } // a thinking model may still put its reasoning in the content as …: show the answer only const answerOnly = (t) => t.replace(/[\s\S]*?(<\/think>|$)/g, "").replace(/^\s+/, ""); async function sha256hex(text) { const h = await crypto.subtle.digest("SHA-256", new TextEncoder().encode(text)); return [...new Uint8Array(h)].map((b) => b.toString(16).padStart(2, "0")).join(""); } // ── your own bankML ─────────────────────────────────────────────────────────────────────────────────────────── async function connect() { const ep = $("endpoint").value.trim().replace(/\/+$/, "") || DEFAULT_ENDPOINT; $("localstatus").textContent = "connecting…"; const b = await get(ep, "/bankml"); if (!b) { $("localstatus").textContent = "not reachable — is bankml serve running with --allow-origin " + location.origin + " ? (see below)"; return false; } const v = b.verified || {}; $("localstatus").textContent = v.guard === "play" ? `✓ connected: ${v.name || b.resident || "model"} · bankML ${v.bankml} · sha256 ${String(v.model_sha256 || "").slice(0, 12)}…` : "connected, but no verified model is loaded yet"; return v.guard === "play"; } async function askLocal(message) { const ep = $("endpoint").value.trim().replace(/\/+$/, "") || DEFAULT_ENDPOINT; const system = persona.system_prompt + "\n\nSELF (measured by bankML just now):\n" + await selfText(ep); const out = bubble("assistant", "…"); const started = waiting(out, "your bankML is reading the prompt (the first answer of a session reads the whole persona; it can take a few minutes on a busy CPU)"); let text = "", receipt = null; const r = await fetch(ep + "/v1/chat/completions", { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ messages: [{ role: "system", content: system }, ...history.slice(-12), { role: "user", content: message }], stream: true, max_tokens: 384 }), }); if (!r.ok) throw new Error(`HTTP ${r.status}: ${(await r.text()).slice(0, 300)}`); const reader = r.body.getReader(), dec = new TextDecoder(); let buf = ""; for (;;) { const { done, value } = await reader.read(); if (done) break; buf += dec.decode(value, { stream: true }); let i; while ((i = buf.indexOf("\n")) >= 0) { const line = buf.slice(0, i).trim(); buf = buf.slice(i + 1); if (!line.startsWith("data:") || line === "data: [DONE]") continue; const d = JSON.parse(line.slice(5)); if (d.bankml_receipt) { receipt = d.bankml_receipt; continue; } const piece = d.choices?.[0]?.delta?.content; if (piece) { started(); text += piece; out.textContent = text; } } } started(); const ok = receipt && (await sha256hex(text)) === receipt.response_sha256; note(out, receipt ? `${ok ? "✓" : "✗"} receipt — bankML ${receipt.bankml} · model sha256 ${String(receipt.model_sha256 || "").slice(0, 12)}… · answer sha256 ${ok ? "matches the text received" : "does NOT match the text received"}` : "no receipt came with this answer", ok ? "ok" : "bad"); history.push({ role: "user", content: message }, { role: "assistant", content: text }); } // ── a Hugging Face provider (not bankML) ────────────────────────────────────────────────────────────────────── function readSession() { try { return JSON.parse(sessionStorage.getItem(SESSION) || "null"); } catch { return null; } } function writeSession(v) { try { v ? sessionStorage.setItem(SESSION, JSON.stringify(v)) : sessionStorage.removeItem(SESSION); } catch {} } const signedIn = () => !!(oauth && oauth.accessToken && new Date(oauth.accessTokenExpiresAt) > new Date()); function renderAuth() { $("signin").hidden = signedIn(); $("signout").hidden = !signedIn(); $("who").textContent = signedIn() ? `signed in as ${oauth.userInfo?.preferred_username || oauth.userInfo?.name || "you"} — answers spend your inference quota` : "not signed in"; } async function askProvider(message) { const model = $("model").value.trim() || DEFAULT_MODEL; if (!/^[\w.-]+\/[\w.-]+$/.test(model)) throw new Error("the model must be a Hugging Face repository id, owner/name"); // the persona, told the truth about where it is running const system = persona.system_prompt + `\n\nWHERE THIS REPLY COMES FROM: you speak for bankML — its design, its rule, its voice — but this particular reply is generated by ${model} through a Hugging Face inference provider, not by bankML's engine: there is no bankML arithmetic, no receipt and no SELF block behind it. Answer as bankML would about what bankML is and does; when asked about your speed, use, receipt, verification or what is running right now, say plainly that this reply comes from ${model}, not from bankML, and that the visitor's own bankML gives verified answers.`; const out = bubble("assistant", "…"); const started = waiting(out, `asking ${model} through a Hugging Face provider`); let text = "", reasoning = ""; const client = new InferenceClient(oauth.accessToken); // Qwen3 and other thinking models: /no_think in the prompt (honoured by the model itself) and enable_thinking off // (honoured by some providers) — otherwise the whole budget can go to reasoning and no answer arrives const stream = client.chatCompletionStream({ provider: "auto", model, max_tokens: 512, chat_template_kwargs: { enable_thinking: false }, messages: [{ role: "system", content: system + "\n/no_think" }, ...history.slice(-12), { role: "user", content: message }] }); for await (const chunk of stream) { const d = chunk?.choices?.[0]?.delta || {}; if (d.reasoning_content) reasoning += d.reasoning_content; if (d.content) { started(); text += d.content; out.textContent = answerOnly(text) || "…"; } } started(); text = answerOnly(text); if (!text) { out.textContent = reasoning ? "The provider returned only the model's reasoning and no answer — ask again, or choose a model that does not think aloud." : "The provider returned an empty answer."; } note(out, `not bankML — ${model} via a Hugging Face provider · no receipt · your quota`, "warn"); history.push({ role: "user", content: message }, { role: "assistant", content: text }); } // ── wiring ───────────────────────────────────────────────────────────────────────────────────────────────────── function renderMode() { const m = mode(); $("localrow").hidden = m !== "local"; $("hfrow").hidden = m !== "hf"; $("modenote").textContent = m === "local" ? "bankML's own arithmetic on your machine: verified model, receipt checked here, nothing sent anywhere else." : "Not bankML: a provider-hosted model speaks with bankML's persona, without its arithmetic or a receipt."; } document.querySelectorAll('input[name="mode"]').forEach((r) => r.addEventListener("change", renderMode)); $("connect").addEventListener("click", connect); $("signin").addEventListener("click", async () => { if (!window.huggingface?.variables?.OAUTH_CLIENT_ID) { $("who").textContent = "sign-in works only on the Hugging Face Space itself"; return; } window.location.href = await oauthLoginUrl({ scopes: window.huggingface.variables.OAUTH_SCOPES }); }); $("signout").addEventListener("click", () => { writeSession(null); oauth = null; renderAuth(); }); $("send").addEventListener("click", async () => { const message = $("message").value.trim(); if (!message || !persona) return; if (mode() === "hf" && !signedIn()) { $("askstatus").textContent = "sign in with Hugging Face first"; return; } $("message").value = ""; $("send").disabled = true; $("askstatus").textContent = ""; bubble("user", message); try { await (mode() === "local" ? askLocal(message) : askProvider(message)); } catch (e) { const local = mode() === "local"; $("askstatus").textContent = (local ? "your bankML did not answer: " : "the provider did not answer: ") + String(e.message || e).slice(0, 300) + (local ? " — is bankml serve running with --allow-origin " + location.origin + " ?" : ""); } finally { $("send").disabled = false; } }); $("message").addEventListener("keydown", (e) => { if (e.key === "Enter" && (e.ctrlKey || e.metaKey)) $("send").click(); }); $("origin").textContent = location.origin; (async () => { $("endpoint").value = DEFAULT_ENDPOINT; $("model").value = DEFAULT_MODEL; renderMode(); await loadPersona(); if (persona) $("mantra").textContent = "“" + persona.mantra + "”"; try { const res = await oauthHandleRedirectIfPresent(); if (res) writeSession(res); } catch (e) { $("who").textContent = "sign-in failed: " + String(e).slice(0, 120); } oauth = readSession(); renderAuth(); })();