"""Run a Jebadiah GGUF inside Ollama and print typed answers. Same interface and output as decide_gguf.py, but it talks to Ollama's /api/generate. The prompt is rendered in Python exactly as AINode's /v1/systemone does (jebadiah_prompt.py, the chat template with thinking off) and sent with "raw": true, so Ollama's own template never touches it. One token is requested with logprobs on and top_logprobs 20 (Ollama's cap). The answer is read off the log probabilities of the option labels at that position by exact token text, renormalised over the labels, with the per-type temperature from temperatures.json applied. Ollama reports full-vocabulary log probabilities from the raw logits (before temperature), which is what the temperatures were fitted on. Every call checks that Ollama counted the same number of prompt tokens as the local tokenizer. ollama pull hf.co/frontier-infra/jebadiah-9b-v2-GGUF:Q8_0 python scripts/decide_ollama.py --model hf.co/frontier-infra/jebadiah-9b-v2-GGUF:Q8_0 --request scripts/example-request.json At most 20 options per question for exact probabilities: a label outside Ollama's top 20 gets the smallest returned value (an upper bound), and the script reports how many labels that hit. Needs `transformers` (the tokenizer only, no torch) and nothing else outside the standard library. """ from __future__ import annotations import argparse import json import os import sys import urllib.request HERE = os.path.dirname(os.path.abspath(__file__)) sys.path.insert(0, HERE) from decide_gguf import load_renderer, option_probs, read_temperatures # noqa: E402 from jebadiah_prompt import answer_from_probs # noqa: E402 TOP_MAX = 20 # Ollama rejects top_logprobs above 20 (server/routes.go) def post(server: str, path: str, body: dict, timeout: float = 900) -> dict: req = urllib.request.Request(server.rstrip("/") + path, data=json.dumps(body).encode(), headers={"Content-Type": "application/json"}) with urllib.request.urlopen(req, timeout=timeout) as r: return json.loads(r.read()) def label_logprobs(server: str, model: str, rd) -> tuple[list[float], int, int]: r = post(server, "/api/generate", { "model": model, "prompt": rd.prompt, "raw": True, "stream": False, "think": False, "logprobs": True, "top_logprobs": TOP_MAX, "options": {"num_predict": 1, "temperature": 0}}) top = r["logprobs"][0]["top_logprobs"] lp = {} for t in top: lp.setdefault(t["token"], t["logprob"]) floor = min(lp.values()) return [lp.get(L, floor) for L in rd.letters], sum(1 for L in rd.letters if L not in lp), r["prompt_eval_count"] def main(): ap = argparse.ArgumentParser() ap.add_argument("--server", default="http://127.0.0.1:11434", help="Ollama's server") ap.add_argument("--model", required=True, help="the Ollama model name for the Jebadiah GGUF") ap.add_argument("--request", required=True) ap.add_argument("--tokenizer", default=os.path.dirname(HERE)) ap.add_argument("--temperatures", default=os.path.join(os.path.dirname(HERE), "temperatures.json")) ap.add_argument("--no-temperatures", action="store_true") a = ap.parse_args() req = json.load(open(a.request)) renderer = load_renderer(a.tokenizer) temps = {} if a.no_temperatures else read_temperatures(a.temperatures) out = {"temperatures_applied": temps, "answers": {}} for qid, q in req["questions"].items(): rd = renderer.render(req["state"], q) n_local = len(renderer.tok.encode(rd.prompt, add_special_tokens=False)) lps, missing, n_prompt = label_logprobs(a.server, a.model, rd) if n_prompt != n_local: sys.exit(f"{qid}: Ollama counted {n_prompt} prompt tokens, the local tokenizer {n_local}.") if missing: print(f"{qid}: {missing} of {len(rd.letters)} labels were outside Ollama's top {TOP_MAX}", file=sys.stderr) probs = option_probs(lps, float(temps.get(q["type"], 1.0))) out["answers"][qid] = answer_from_probs(q, rd.keys, probs) print(json.dumps(out, indent=1)) if __name__ == "__main__": main()