File size: 7,226 Bytes
7cd0c19
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
"""Run a Jebadiah GGUF inside LM Studio and print typed answers.

Same interface and output as decide_gguf.py, but it talks to LM Studio's local server
(Developer tab, or `lms server start`) instead of llama-server.

LM Studio applies the chat template itself, so this script sends the chat MESSAGES that AINode's
/v1/systemone builds (jebadiah_prompt.py) with reasoning off ("reasoning_effort": "none"). With
thinking off, LM Studio renders the same bytes that decide_gguf.py sends raw: the Qwen template
with an empty think block and the generation prompt. The script checks this on every call, by
comparing the prompt token count LM Studio reports with the count from the local tokenizer. The
answer is read off the log probabilities of the single-token option labels ("A", "B", ...) at the
first generated position, renormalised over those labels, with the per-type temperature from
temperatures.json applied.

LM Studio returns at most the top 20 log probabilities. That is enough for any question AINode's
own route accepts (20 options at most). On a wider question, a label outside the top 20 gets the
smallest returned value, which is an upper bound; the script reports how many labels that hit.

  lms load jebadiah-9b-v2 --identifier jebadiah
  python scripts/decide_lmstudio.py --model jebadiah --request scripts/example-request.json

If "Require Authentication" is on in LM Studio's server settings, pass a token with --api-key or
set LM_API_TOKEN. Needs `transformers` (the tokenizer only, no torch) and nothing else outside
the standard library.
"""
from __future__ import annotations

import argparse
import json
import math
import os
import sys
import urllib.request

HERE = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, HERE)
from ainode_prompt_verbatim import build_messages, option_label  # noqa: E402
from jebadiah_prompt import CHAT_TEMPLATE_KWARGS, Renderer, answer_from_probs  # noqa: E402

TOP_MAX = 20   # LM Studio's cap on top_logprobs


class MessageRenderer(Renderer):
    """jebadiah_prompt.Renderer that also keeps the chat messages it rendered, so they can be
    sent to a server that applies the template itself."""

    def render_messages(self, state_text, question, options, letters=None):
        messages = build_messages(state_text, None, question, options)
        if letters is not None:
            user = messages[1]["content"]
            head, _, rest = user.partition("\nOPTIONS:\n")
            lines = rest.split("\n")
            for i in range(len(options)):
                assert lines[i].startswith(f"{option_label(i)}. ")
                lines[i] = f"{letters[i]}. " + lines[i][len(option_label(i)) + 2:]
            messages[1]["content"] = head + "\nOPTIONS:\n" + "\n".join(lines)
        self.last_messages = messages
        return self.tok.apply_chat_template(messages, tokenize=False, **CHAT_TEMPLATE_KWARGS)

    def render(self, state, q, order=None):
        rd = super().render(state, q, order)
        rd.messages = self.last_messages
        rd.n_tokens = len(self.tok.encode(rd.prompt, add_special_tokens=False))
        return rd


def load_renderer(tokenizer: str, max_tokens: int = 2048) -> MessageRenderer:
    from transformers import AutoTokenizer
    tok = AutoTokenizer.from_pretrained(tokenizer)
    if tok.pad_token_id is None:
        tok.pad_token = tok.eos_token
    return MessageRenderer(tok, max_tokens)


def read_temperatures(path: str | None) -> dict:
    if not path or not os.path.exists(path):
        return {}
    return {k: float(v) for k, v in json.load(open(path))["temperatures"].items()}


def post(server: str, path: str, body: dict, api_key: str | None = None, timeout: float = 900) -> dict:
    headers = {"Content-Type": "application/json"}
    if api_key:
        headers["Authorization"] = "Bearer " + api_key
    req = urllib.request.Request(server.rstrip("/") + path, data=json.dumps(body).encode(), headers=headers)
    with urllib.request.urlopen(req, timeout=timeout) as r:
        return json.loads(r.read())


def label_logprobs(server: str, model: str, rd, api_key: str | None = None) -> tuple[list[float], int, int]:
    """Log probabilities of each option label at the first answer position, from LM Studio's
    OpenAI-compatible chat endpoint with thinking off. A label outside the returned top 20 gets the
    smallest returned value (an upper bound). Returns (logprobs, labels not returned, prompt tokens
    LM Studio counted)."""
    r = post(server, "/v1/chat/completions", {
        "model": model, "messages": rd.messages, "max_tokens": 1, "temperature": 0,
        "logprobs": True, "top_logprobs": TOP_MAX, "reasoning_effort": "none", "stream": False}, api_key)
    top = r["choices"][0]["logprobs"]["content"][0]["top_logprobs"]
    lp = {}
    for t in top:
        lp.setdefault(t["token"], t["logprob"])   # exact token text: "A", not " A"
    floor = min(lp.values())
    return [lp.get(L, floor) for L in rd.letters], sum(1 for L in rd.letters if L not in lp), r["usage"]["prompt_tokens"]


def option_probs(logprobs: list[float], temperature: float = 1.0) -> list[float]:
    """Softmax over the labels only, after dividing by the temperature (see decide_gguf.py)."""
    z = [x / temperature for x in logprobs]
    m = max(z)
    e = [math.exp(x - m) for x in z]
    s = sum(e)
    return [x / s for x in e]


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--server", default="http://127.0.0.1:1234", help="LM Studio's local server")
    ap.add_argument("--model", required=True, help="the model identifier LM Studio shows for the loaded Jebadiah")
    ap.add_argument("--api-key", default=os.environ.get("LM_API_TOKEN"), help="LM Studio API token, if auth is on")
    ap.add_argument("--request", required=True, help="JSON file: {state, questions: {id: {type, instructions, criteria}}}")
    ap.add_argument("--tokenizer", default=os.path.dirname(HERE), help="folder or Hub repo id with the tokenizer and chat template")
    ap.add_argument("--temperatures", default=os.path.join(os.path.dirname(HERE), "temperatures.json"))
    ap.add_argument("--no-temperatures", action="store_true", help="raw probabilities, as the served route returns today")
    a = ap.parse_args()
    req = json.load(open(a.request))
    renderer = load_renderer(a.tokenizer)
    temps = {} if a.no_temperatures else read_temperatures(a.temperatures)
    out = {"temperatures_applied": temps, "answers": {}}
    for qid, q in req["questions"].items():
        rd = renderer.render(req["state"], q)
        lps, missing, n_prompt = label_logprobs(a.server, a.model, rd, a.api_key)
        if n_prompt != rd.n_tokens:
            sys.exit(f"{qid}: LM Studio rendered {n_prompt} prompt tokens, the local template {rd.n_tokens}. "
                     "Is thinking off, and is the loaded model's prompt template the GGUF's own?")
        if missing:
            print(f"{qid}: {missing} of {len(rd.letters)} labels were outside LM Studio's top {TOP_MAX}", file=sys.stderr)
        probs = option_probs(lps, float(temps.get(q["type"], 1.0)))
        out["answers"][qid] = answer_from_probs(q, rd.keys, probs)
    print(json.dumps(out, indent=1))


if __name__ == "__main__":
    main()