jebadiah-4b-v2-GGUF / scripts /decide_lmstudio.py
jbrashear's picture
Add decide_lmstudio.py and a "Use it in LM Studio" section
7cd0c19 verified
Raw History Blame Contribute Delete
7.23 kB
"""Run a Jebadiah GGUF inside LM Studio and print typed answers.
Same interface and output as decide_gguf.py, but it talks to LM Studio's local server
(Developer tab, or `lms server start`) instead of llama-server.
LM Studio applies the chat template itself, so this script sends the chat MESSAGES that AINode's
/v1/systemone builds (jebadiah_prompt.py) with reasoning off ("reasoning_effort": "none"). With
thinking off, LM Studio renders the same bytes that decide_gguf.py sends raw: the Qwen template
with an empty think block and the generation prompt. The script checks this on every call, by
comparing the prompt token count LM Studio reports with the count from the local tokenizer. The
answer is read off the log probabilities of the single-token option labels ("A", "B", ...) at the
first generated position, renormalised over those labels, with the per-type temperature from
temperatures.json applied.
LM Studio returns at most the top 20 log probabilities. That is enough for any question AINode's
own route accepts (20 options at most). On a wider question, a label outside the top 20 gets the
smallest returned value, which is an upper bound; the script reports how many labels that hit.
lms load jebadiah-9b-v2 --identifier jebadiah
python scripts/decide_lmstudio.py --model jebadiah --request scripts/example-request.json
If "Require Authentication" is on in LM Studio's server settings, pass a token with --api-key or
set LM_API_TOKEN. Needs `transformers` (the tokenizer only, no torch) and nothing else outside
the standard library.
"""
from __future__ import annotations
import argparse
import json
import math
import os
import sys
import urllib.request
HERE = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, HERE)
from ainode_prompt_verbatim import build_messages, option_label # noqa: E402
from jebadiah_prompt import CHAT_TEMPLATE_KWARGS, Renderer, answer_from_probs # noqa: E402
TOP_MAX = 20 # LM Studio's cap on top_logprobs
class MessageRenderer(Renderer):
"""jebadiah_prompt.Renderer that also keeps the chat messages it rendered, so they can be
sent to a server that applies the template itself."""
def render_messages(self, state_text, question, options, letters=None):
messages = build_messages(state_text, None, question, options)
if letters is not None:
user = messages[1]["content"]
head, _, rest = user.partition("\nOPTIONS:\n")
lines = rest.split("\n")
for i in range(len(options)):
assert lines[i].startswith(f"{option_label(i)}. ")
lines[i] = f"{letters[i]}. " + lines[i][len(option_label(i)) + 2:]
messages[1]["content"] = head + "\nOPTIONS:\n" + "\n".join(lines)
self.last_messages = messages
return self.tok.apply_chat_template(messages, tokenize=False, **CHAT_TEMPLATE_KWARGS)
def render(self, state, q, order=None):
rd = super().render(state, q, order)
rd.messages = self.last_messages
rd.n_tokens = len(self.tok.encode(rd.prompt, add_special_tokens=False))
return rd
def load_renderer(tokenizer: str, max_tokens: int = 2048) -> MessageRenderer:
from transformers import AutoTokenizer
tok = AutoTokenizer.from_pretrained(tokenizer)
if tok.pad_token_id is None:
tok.pad_token = tok.eos_token
return MessageRenderer(tok, max_tokens)
def read_temperatures(path: str | None) -> dict:
if not path or not os.path.exists(path):
return {}
return {k: float(v) for k, v in json.load(open(path))["temperatures"].items()}
def post(server: str, path: str, body: dict, api_key: str | None = None, timeout: float = 900) -> dict:
headers = {"Content-Type": "application/json"}
if api_key:
headers["Authorization"] = "Bearer " + api_key
req = urllib.request.Request(server.rstrip("/") + path, data=json.dumps(body).encode(), headers=headers)
with urllib.request.urlopen(req, timeout=timeout) as r:
return json.loads(r.read())
def label_logprobs(server: str, model: str, rd, api_key: str | None = None) -> tuple[list[float], int, int]:
"""Log probabilities of each option label at the first answer position, from LM Studio's
OpenAI-compatible chat endpoint with thinking off. A label outside the returned top 20 gets the
smallest returned value (an upper bound). Returns (logprobs, labels not returned, prompt tokens
LM Studio counted)."""
r = post(server, "/v1/chat/completions", {
"model": model, "messages": rd.messages, "max_tokens": 1, "temperature": 0,
"logprobs": True, "top_logprobs": TOP_MAX, "reasoning_effort": "none", "stream": False}, api_key)
top = r["choices"][0]["logprobs"]["content"][0]["top_logprobs"]
lp = {}
for t in top:
lp.setdefault(t["token"], t["logprob"]) # exact token text: "A", not " A"
floor = min(lp.values())
return [lp.get(L, floor) for L in rd.letters], sum(1 for L in rd.letters if L not in lp), r["usage"]["prompt_tokens"]
def option_probs(logprobs: list[float], temperature: float = 1.0) -> list[float]:
"""Softmax over the labels only, after dividing by the temperature (see decide_gguf.py)."""
z = [x / temperature for x in logprobs]
m = max(z)
e = [math.exp(x - m) for x in z]
s = sum(e)
return [x / s for x in e]
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--server", default="http://127.0.0.1:1234", help="LM Studio's local server")
ap.add_argument("--model", required=True, help="the model identifier LM Studio shows for the loaded Jebadiah")
ap.add_argument("--api-key", default=os.environ.get("LM_API_TOKEN"), help="LM Studio API token, if auth is on")
ap.add_argument("--request", required=True, help="JSON file: {state, questions: {id: {type, instructions, criteria}}}")
ap.add_argument("--tokenizer", default=os.path.dirname(HERE), help="folder or Hub repo id with the tokenizer and chat template")
ap.add_argument("--temperatures", default=os.path.join(os.path.dirname(HERE), "temperatures.json"))
ap.add_argument("--no-temperatures", action="store_true", help="raw probabilities, as the served route returns today")
a = ap.parse_args()
req = json.load(open(a.request))
renderer = load_renderer(a.tokenizer)
temps = {} if a.no_temperatures else read_temperatures(a.temperatures)
out = {"temperatures_applied": temps, "answers": {}}
for qid, q in req["questions"].items():
rd = renderer.render(req["state"], q)
lps, missing, n_prompt = label_logprobs(a.server, a.model, rd, a.api_key)
if n_prompt != rd.n_tokens:
sys.exit(f"{qid}: LM Studio rendered {n_prompt} prompt tokens, the local template {rd.n_tokens}. "
"Is thinking off, and is the loaded model's prompt template the GGUF's own?")
if missing:
print(f"{qid}: {missing} of {len(rd.letters)} labels were outside LM Studio's top {TOP_MAX}", file=sys.stderr)
probs = option_probs(lps, float(temps.get(q["type"], 1.0)))
out["answers"][qid] = answer_from_probs(q, rd.keys, probs)
print(json.dumps(out, indent=1))
if __name__ == "__main__":
main()