"""Run a Jebadiah GGUF inside LM Studio and print typed answers. Same interface and output as decide_gguf.py, but it talks to LM Studio's local server (Developer tab, or `lms server start`) instead of llama-server. LM Studio applies the chat template itself, so this script sends the chat MESSAGES that AINode's /v1/systemone builds (jebadiah_prompt.py) with reasoning off ("reasoning_effort": "none"). With thinking off, LM Studio renders the same bytes that decide_gguf.py sends raw: the Qwen template with an empty think block and the generation prompt. The script checks this on every call, by comparing the prompt token count LM Studio reports with the count from the local tokenizer. The answer is read off the log probabilities of the single-token option labels ("A", "B", ...) at the first generated position, renormalised over those labels, with the per-type temperature from temperatures.json applied. LM Studio returns at most the top 20 log probabilities. That is enough for any question AINode's own route accepts (20 options at most). On a wider question, a label outside the top 20 gets the smallest returned value, which is an upper bound; the script reports how many labels that hit. lms load jebadiah-9b-v2 --identifier jebadiah python scripts/decide_lmstudio.py --model jebadiah --request scripts/example-request.json If "Require Authentication" is on in LM Studio's server settings, pass a token with --api-key or set LM_API_TOKEN. Needs `transformers` (the tokenizer only, no torch) and nothing else outside the standard library. """ from __future__ import annotations import argparse import json import math import os import sys import urllib.request HERE = os.path.dirname(os.path.abspath(__file__)) sys.path.insert(0, HERE) from ainode_prompt_verbatim import build_messages, option_label # noqa: E402 from jebadiah_prompt import CHAT_TEMPLATE_KWARGS, Renderer, answer_from_probs # noqa: E402 TOP_MAX = 20 # LM Studio's cap on top_logprobs class MessageRenderer(Renderer): """jebadiah_prompt.Renderer that also keeps the chat messages it rendered, so they can be sent to a server that applies the template itself.""" def render_messages(self, state_text, question, options, letters=None): messages = build_messages(state_text, None, question, options) if letters is not None: user = messages[1]["content"] head, _, rest = user.partition("\nOPTIONS:\n") lines = rest.split("\n") for i in range(len(options)): assert lines[i].startswith(f"{option_label(i)}. ") lines[i] = f"{letters[i]}. " + lines[i][len(option_label(i)) + 2:] messages[1]["content"] = head + "\nOPTIONS:\n" + "\n".join(lines) self.last_messages = messages return self.tok.apply_chat_template(messages, tokenize=False, **CHAT_TEMPLATE_KWARGS) def render(self, state, q, order=None): rd = super().render(state, q, order) rd.messages = self.last_messages rd.n_tokens = len(self.tok.encode(rd.prompt, add_special_tokens=False)) return rd def load_renderer(tokenizer: str, max_tokens: int = 2048) -> MessageRenderer: from transformers import AutoTokenizer tok = AutoTokenizer.from_pretrained(tokenizer) if tok.pad_token_id is None: tok.pad_token = tok.eos_token return MessageRenderer(tok, max_tokens) def read_temperatures(path: str | None) -> dict: if not path or not os.path.exists(path): return {} return {k: float(v) for k, v in json.load(open(path))["temperatures"].items()} def post(server: str, path: str, body: dict, api_key: str | None = None, timeout: float = 900) -> dict: headers = {"Content-Type": "application/json"} if api_key: headers["Authorization"] = "Bearer " + api_key req = urllib.request.Request(server.rstrip("/") + path, data=json.dumps(body).encode(), headers=headers) with urllib.request.urlopen(req, timeout=timeout) as r: return json.loads(r.read()) def label_logprobs(server: str, model: str, rd, api_key: str | None = None) -> tuple[list[float], int, int]: """Log probabilities of each option label at the first answer position, from LM Studio's OpenAI-compatible chat endpoint with thinking off. A label outside the returned top 20 gets the smallest returned value (an upper bound). Returns (logprobs, labels not returned, prompt tokens LM Studio counted).""" r = post(server, "/v1/chat/completions", { "model": model, "messages": rd.messages, "max_tokens": 1, "temperature": 0, "logprobs": True, "top_logprobs": TOP_MAX, "reasoning_effort": "none", "stream": False}, api_key) top = r["choices"][0]["logprobs"]["content"][0]["top_logprobs"] lp = {} for t in top: lp.setdefault(t["token"], t["logprob"]) # exact token text: "A", not " A" floor = min(lp.values()) return [lp.get(L, floor) for L in rd.letters], sum(1 for L in rd.letters if L not in lp), r["usage"]["prompt_tokens"] def option_probs(logprobs: list[float], temperature: float = 1.0) -> list[float]: """Softmax over the labels only, after dividing by the temperature (see decide_gguf.py).""" z = [x / temperature for x in logprobs] m = max(z) e = [math.exp(x - m) for x in z] s = sum(e) return [x / s for x in e] def main(): ap = argparse.ArgumentParser() ap.add_argument("--server", default="http://127.0.0.1:1234", help="LM Studio's local server") ap.add_argument("--model", required=True, help="the model identifier LM Studio shows for the loaded Jebadiah") ap.add_argument("--api-key", default=os.environ.get("LM_API_TOKEN"), help="LM Studio API token, if auth is on") ap.add_argument("--request", required=True, help="JSON file: {state, questions: {id: {type, instructions, criteria}}}") ap.add_argument("--tokenizer", default=os.path.dirname(HERE), help="folder or Hub repo id with the tokenizer and chat template") ap.add_argument("--temperatures", default=os.path.join(os.path.dirname(HERE), "temperatures.json")) ap.add_argument("--no-temperatures", action="store_true", help="raw probabilities, as the served route returns today") a = ap.parse_args() req = json.load(open(a.request)) renderer = load_renderer(a.tokenizer) temps = {} if a.no_temperatures else read_temperatures(a.temperatures) out = {"temperatures_applied": temps, "answers": {}} for qid, q in req["questions"].items(): rd = renderer.render(req["state"], q) lps, missing, n_prompt = label_logprobs(a.server, a.model, rd, a.api_key) if n_prompt != rd.n_tokens: sys.exit(f"{qid}: LM Studio rendered {n_prompt} prompt tokens, the local template {rd.n_tokens}. " "Is thinking off, and is the loaded model's prompt template the GGUF's own?") if missing: print(f"{qid}: {missing} of {len(rd.letters)} labels were outside LM Studio's top {TOP_MAX}", file=sys.stderr) probs = option_probs(lps, float(temps.get(q["type"], 1.0))) out["answers"][qid] = answer_from_probs(q, rd.keys, probs) print(json.dumps(out, indent=1)) if __name__ == "__main__": main()