Harry19081 commited on
Commit
9662694
·
verified ·
1 Parent(s): 3a67163

Card: GGUF / llama.cpp / Ollama section (EN + ZH), tags; wald-serve 0.1.1 (llama.cpp backend, vLLM path unchanged); MANIFEST with the GGUF files

Browse files
MANIFEST.json CHANGED
@@ -20,8 +20,8 @@
20
  "sha256": "69225fb813578b0d4f9e453a9f0c140c848ec354cf6cca877bcae828c082d3d7"
21
  },
22
  "README.md": {
23
- "bytes": 28587,
24
- "sha256": "7dde057672062641b000143ffaeab5848b19d17fa395c846b79b378cb3746e76"
25
  },
26
  "RUNBOOK.md": {
27
  "bytes": 4980,
@@ -44,8 +44,8 @@
44
  "sha256": "e385b92e333eaaa9395a09daf26c09420645bee242456552a55500db45919bc7"
45
  },
46
  "docs/readmes/README.zh.md": {
47
- "bytes": 24299,
48
- "sha256": "3754868c44ea219cc16220d84c70dbfb170e6a8e699b20ab57584d39f366b421"
49
  },
50
  "evaluation/benchmark-summary.json": {
51
  "bytes": 36014,
@@ -140,48 +140,48 @@
140
  "sha256": "d5b7b3e62d43c5f0dae4779f723b1eb4408aa9c21fcbdb6ed65e1da049dd050a"
141
  },
142
  "server/README.md": {
143
- "bytes": 849,
144
- "sha256": "e32014b15ad7e0aa37623d7a5c4561d68f5cd25c9f5514339a8e36d886c1b36f"
145
  },
146
  "server/pyproject.toml": {
147
- "bytes": 565,
148
- "sha256": "f5d1763d77797fab431a8d05b1a457278d5468f1487636545ae290c4b6e7ff53"
149
  },
150
  "server/src/wald_serve/__init__.py": {
151
  "bytes": 139,
152
- "sha256": "6794303c203fd5ae4dc9246eb4ce75be8006a6e058c9c7f572896e1b880ab961"
153
  },
154
  "server/src/wald_serve/__main__.py": {
155
  "bytes": 51,
156
  "sha256": "540fd0a5992ca535482a9d44b2e268ce3a39bb47a0576f301214afa5ef4494b9"
157
  },
158
  "server/src/wald_serve/engine.py": {
159
- "bytes": 13300,
160
- "sha256": "09672d5a560ab5f5f2b9f432d176f6a151b7fcce5ea7729763e8f50af1d6231a"
161
  },
162
  "server/src/wald_serve/prompt.py": {
163
  "bytes": 4192,
164
  "sha256": "bb45a10734f2546eb1ea1bd7e8b5717ecd8305a4aded0b816f80cd7e85188f7a"
165
  },
166
  "server/src/wald_serve/server.py": {
167
- "bytes": 7195,
168
- "sha256": "6eaf25b7b9eecf3d375b7f9b87ed4b16de211af7e5689966287c76e1e7733cfd"
169
  },
170
  "server/src/wald_serve/wire.py": {
171
  "bytes": 3670,
172
  "sha256": "836ce54ce93016a98f94a6beaa508cc03982e24c3d3b4d62b8ef667389cd14af"
173
  },
174
  "server/tests/fakevllm.py": {
175
- "bytes": 2895,
176
- "sha256": "e1a6c2cdf113d491bad0cb8efc292d1323c16ee70609ee0dfe81a8fab222a08e"
177
  },
178
  "server/tests/test_parity.py": {
179
  "bytes": 4392,
180
  "sha256": "cd1e4142090076114f2b07efce49fdc5466e302a5c79996b993f3833a3457566"
181
  },
182
  "server/tests/test_server.py": {
183
- "bytes": 11924,
184
- "sha256": "8f8fecbcff18abb6b44da743f6e041472124ba4188881f9e75845348e46ca405"
185
  },
186
  "serving.json": {
187
  "bytes": 128,
@@ -242,5 +242,21 @@
242
  "evaluation/v1.2/release-check.json": {
243
  "bytes": 3036,
244
  "sha256": "0de642d7c1ddb2ececafce16c06ae2d2eb0e663ec94b4afcfab605a61aadfb70"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
245
  }
246
  }
 
20
  "sha256": "69225fb813578b0d4f9e453a9f0c140c848ec354cf6cca877bcae828c082d3d7"
21
  },
22
  "README.md": {
23
+ "bytes": 29939,
24
+ "sha256": "5f02a79558567e33186c0d1e4d462b70a0fa2d960080c2af27b2e08567eb8b98"
25
  },
26
  "RUNBOOK.md": {
27
  "bytes": 4980,
 
44
  "sha256": "e385b92e333eaaa9395a09daf26c09420645bee242456552a55500db45919bc7"
45
  },
46
  "docs/readmes/README.zh.md": {
47
+ "bytes": 25676,
48
+ "sha256": "667a7db8482decc260fdf2538de52365f3e130984a173554fb7483e1f5f02d26"
49
  },
50
  "evaluation/benchmark-summary.json": {
51
  "bytes": 36014,
 
140
  "sha256": "d5b7b3e62d43c5f0dae4779f723b1eb4408aa9c21fcbdb6ed65e1da049dd050a"
141
  },
142
  "server/README.md": {
143
+ "bytes": 1678,
144
+ "sha256": "2ca738dcacd770e7c490ceeb59f2f11df7515f5bc56784eb9d476af4c7145b4d"
145
  },
146
  "server/pyproject.toml": {
147
+ "bytes": 578,
148
+ "sha256": "0fd8155bfa466e0f7a95fddce986ab881cce9260187c1620884fa27e5008dfbe"
149
  },
150
  "server/src/wald_serve/__init__.py": {
151
  "bytes": 139,
152
+ "sha256": "7db760d45fae0f1d9cea6dbd3e1421a2627a92020981ae3cd56ab57181423aac"
153
  },
154
  "server/src/wald_serve/__main__.py": {
155
  "bytes": 51,
156
  "sha256": "540fd0a5992ca535482a9d44b2e268ce3a39bb47a0576f301214afa5ef4494b9"
157
  },
158
  "server/src/wald_serve/engine.py": {
159
+ "bytes": 14814,
160
+ "sha256": "40fbea84c4c5b93ef888da44508d4fec2677c58732a09d731f0bcf17c2b48dd6"
161
  },
162
  "server/src/wald_serve/prompt.py": {
163
  "bytes": 4192,
164
  "sha256": "bb45a10734f2546eb1ea1bd7e8b5717ecd8305a4aded0b816f80cd7e85188f7a"
165
  },
166
  "server/src/wald_serve/server.py": {
167
+ "bytes": 8878,
168
+ "sha256": "49fb06ee52eb2ebe7059404432386d02e049556003acb7d2f6494ba92d4b9703"
169
  },
170
  "server/src/wald_serve/wire.py": {
171
  "bytes": 3670,
172
  "sha256": "836ce54ce93016a98f94a6beaa508cc03982e24c3d3b4d62b8ef667389cd14af"
173
  },
174
  "server/tests/fakevllm.py": {
175
+ "bytes": 4167,
176
+ "sha256": "9fb222a21bc2b17221e5a1a3605973b4f71730db9bbfede439bd3388b91e2570"
177
  },
178
  "server/tests/test_parity.py": {
179
  "bytes": 4392,
180
  "sha256": "cd1e4142090076114f2b07efce49fdc5466e302a5c79996b993f3833a3457566"
181
  },
182
  "server/tests/test_server.py": {
183
+ "bytes": 13844,
184
+ "sha256": "46f2007e9743bfeb0f7f79fca093cf444a70df1bfc348229f34cd251f88a19b3"
185
  },
186
  "serving.json": {
187
  "bytes": 128,
 
242
  "evaluation/v1.2/release-check.json": {
243
  "bytes": 3036,
244
  "sha256": "0de642d7c1ddb2ececafce16c06ae2d2eb0e663ec94b4afcfab605a61aadfb70"
245
+ },
246
+ "Wald-4B-v1.2-Q4_K_M.gguf": {
247
+ "bytes": 2708804000,
248
+ "sha256": "e843f7658793b9f3effb3a4941cd67f80707a07b468aaaf94d440613751d76fa"
249
+ },
250
+ "Wald-4B-v1.2-Q5_K_M.gguf": {
251
+ "bytes": 3074986400,
252
+ "sha256": "6fecc4e7655adb71e8c642ce5018f967564ba259852d562ecd3109386542b0cb"
253
+ },
254
+ "Wald-4B-v1.2-Q6_K.gguf": {
255
+ "bytes": 3464055200,
256
+ "sha256": "5870f15f3b2eef773f672e56bb36e343a95e32fd09baaf7c9ce8b0da1e5dd897"
257
+ },
258
+ "Wald-4B-v1.2-Q8_0.gguf": {
259
+ "bytes": 4482402720,
260
+ "sha256": "09c02494ebdfb7d088a2c76726eb2b290a438dbc5658c3be16da8333fddd6038"
261
  }
262
  }
README.md CHANGED
@@ -31,6 +31,9 @@ tags:
31
  - qwen3.5
32
  - 4b
33
  - vllm
 
 
 
34
  - reasoning
35
  model-index:
36
  - name: Wald-Q4B v1.1 (revision v1.1)
@@ -202,6 +205,27 @@ curl http://localhost:8000/v1/systemone \
202
 
203
  The answer for `route` contains the chosen key and a probability for each of `returns`, `delivery` and `other`. Request and response fields, yes/no and score questions, and a clarification example: [API reference](https://huggingface.co/org2ai/Wald-4B/blob/main/docs/api.md). `GET /health` reports the effective policy. The server uses vLLM 0.30.0 and the included `wald-serve` package; Docker and exact evaluation settings are in [RUNBOOK.md](https://huggingface.co/org2ai/Wald-4B/blob/main/RUNBOOK.md). Generic text-generation calls do not reproduce the decision API's readout.
204
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
205
  ## Thinking effort
206
 
207
  | Effort | When it thinks | Thought budget |
 
31
  - qwen3.5
32
  - 4b
33
  - vllm
34
+ - gguf
35
+ - llama.cpp
36
+ - ollama
37
  - reasoning
38
  model-index:
39
  - name: Wald-Q4B v1.1 (revision v1.1)
 
205
 
206
  The answer for `route` contains the chosen key and a probability for each of `returns`, `delivery` and `other`. Request and response fields, yes/no and score questions, and a clarification example: [API reference](https://huggingface.co/org2ai/Wald-4B/blob/main/docs/api.md). `GET /health` reports the effective policy. The server uses vLLM 0.30.0 and the included `wald-serve` package; Docker and exact evaluation settings are in [RUNBOOK.md](https://huggingface.co/org2ai/Wald-4B/blob/main/RUNBOOK.md). Generic text-generation calls do not reproduce the decision API's readout.
207
 
208
+ ## GGUF: llama.cpp, Ollama, LM Studio
209
+
210
+ `main` also carries v1.2 as GGUF files for CPUs, Apple Silicon and consumer GPUs (llama.cpp `b11312`, the same weights). Parity on the JevBench public set (231 items, `none`), against the BF16 weights on vLLM (204/231):
211
+
212
+ | File | Size | JevBench public | Same option as BF16 |
213
+ |---|---:|---:|---:|
214
+ | `Wald-4B-v1.2-Q8_0.gguf` | 4.5 GB | 206/231 | 229/231 |
215
+ | `Wald-4B-v1.2-Q6_K.gguf` | 3.5 GB | 205/231 | 227/231 |
216
+ | `Wald-4B-v1.2-Q5_K_M.gguf` | 3.1 GB | 205/231 | 227/231 |
217
+ | `Wald-4B-v1.2-Q4_K_M.gguf` | 2.7 GB | 202/231 | 223/231 |
218
+
219
+ For calibrated probabilities, run the included server on llama.cpp (`llama-server` on your `PATH`):
220
+
221
+ ```sh
222
+ hf download org2ai/Wald-4B --include "Wald-4B-v1.2-Q8_0.gguf" "serving.json" "temperature.json" "server/*" --local-dir ./wald-gguf
223
+ pip install ./wald-gguf/server
224
+ wald-serve --gguf ./wald-gguf/Wald-4B-v1.2-Q8_0.gguf --max-model-len 32768 --port 8000
225
+ ```
226
+
227
+ Ollama: `ollama run hf.co/org2ai/Wald-4B:Q4_K_M`. LM Studio: search for `Wald-4B`. Neither app was tested with these files, and a chat session returns text, not the option probabilities. Download a single file with `--include`; a plain `hf download org2ai/Wald-4B` of `main` also fetches all four GGUFs (14 GB). Details: [org2ai/Wald-4B-GGUF](https://huggingface.co/org2ai/Wald-4B-GGUF).
228
+
229
  ## Thinking effort
230
 
231
  | Effort | When it thinks | Thought budget |
docs/readmes/README.zh.md CHANGED
@@ -83,6 +83,27 @@ curl http://localhost:8000/v1/systemone \
83
 
84
  `route` 的答案包含选中的键,以及 `returns`、`delivery`、`other` 各自的概率。请求与响应字段、是非题和评分题、澄清判断示例见 [API 说明](https://huggingface.co/org2ai/Wald-4B/blob/main/docs/api.md)。`GET /health` 返回当前生效策略。服务使用 vLLM 0.30.0 和仓库内的 `wald-serve`;Docker 与精确评测配置见 [RUNBOOK.md](https://huggingface.co/org2ai/Wald-4B/blob/main/RUNBOOK.md)。普通文本生成接口不会复现决策 API 的读出流程。
85
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
86
  ## 思考程度
87
 
88
  | Effort | 何时思考 | 思考预算 |
 
83
 
84
  `route` 的答案包含选中的键,以及 `returns`、`delivery`、`other` 各自的概率。请求与响应字段、是非题和评分题、澄清判断示例见 [API 说明](https://huggingface.co/org2ai/Wald-4B/blob/main/docs/api.md)。`GET /health` 返回当前生效策略。服务使用 vLLM 0.30.0 和仓库内的 `wald-serve`;Docker 与精确评测配置见 [RUNBOOK.md](https://huggingface.co/org2ai/Wald-4B/blob/main/RUNBOOK.md)。普通文本生成接口不会复现决策 API 的读出流程。
85
 
86
+ ## GGUF:llama.cpp、Ollama、LM Studio
87
+
88
+ `main` 上另有 v1.2 的 GGUF 文件,适合 CPU、Apple Silicon 和消费级显卡(llama.cpp `b11312`,同一份权重)。在 JevBench 公开集(231 题,`none`)上与 vLLM 上的 BF16 权重(204/231)对照:
89
+
90
+ | 文件 | 大小 | JevBench 公开集 | 与 BF16 选同一选项 |
91
+ |---|---:|---:|---:|
92
+ | `Wald-4B-v1.2-Q8_0.gguf` | 4.5 GB | 206/231 | 229/231 |
93
+ | `Wald-4B-v1.2-Q6_K.gguf` | 3.5 GB | 205/231 | 227/231 |
94
+ | `Wald-4B-v1.2-Q5_K_M.gguf` | 3.1 GB | 205/231 | 227/231 |
95
+ | `Wald-4B-v1.2-Q4_K_M.gguf` | 2.7 GB | 202/231 | 223/231 |
96
+
97
+ 要拿到校准后的概率,用自带的服务跑在 llama.cpp 上(`llama-server` 需在 `PATH` 里):
98
+
99
+ ```sh
100
+ hf download org2ai/Wald-4B --include "Wald-4B-v1.2-Q8_0.gguf" "serving.json" "temperature.json" "server/*" --local-dir ./wald-gguf
101
+ pip install ./wald-gguf/server
102
+ wald-serve --gguf ./wald-gguf/Wald-4B-v1.2-Q8_0.gguf --max-model-len 32768 --port 8000
103
+ ```
104
+
105
+ Ollama:`ollama run hf.co/org2ai/Wald-4B:Q4_K_M`。LM Studio:搜索 `Wald-4B`。这两个应用都没有用这些文件实测过;聊天界面返回的是文字,不是各选项的概率。请用 `--include` 只下载一个文件;直接 `hf download org2ai/Wald-4B` 下载 `main` 会把四个 GGUF(14 GB)一起拉下来。详见 [org2ai/Wald-4B-GGUF](https://huggingface.co/org2ai/Wald-4B-GGUF)。
106
+
107
  ## 思考程度
108
 
109
  | Effort | 何时思考 | 思考预算 |
server/README.md CHANGED
@@ -1,14 +1,28 @@
1
  # wald-serve
2
 
3
- The TypeSafe `POST /v1/systemone` server for Wald-4B: the option-letter readout over a vLLM OpenAI-compatible server,
4
- the effort gate (none / low / medium / high / high-k), the knockout for more than 26 options, and the temperature table.
5
- It needs only public packages: `pydantic`, plus `vllm==0.30.0` on the GPU machine.
 
6
 
7
  ```sh
8
  pip install ".[vllm]" # on a Linux CUDA machine
9
  wald-serve --model /path/to/Wald-4B --port 8000 # starts vLLM on the weights and serves /v1/systemone
10
  ```
11
 
12
- Tests (no model, no GPU; a fake vLLM stands in): `pip install ".[test]" && cd tests && pytest -q`.
 
 
 
 
 
 
 
 
 
 
 
 
 
13
  `tests/test_parity.py` compares prompts and answers byte for byte with the reference implementation the published
14
  numbers were measured with; it is skipped unless that implementation is importable (WALD_PARITY_SRC set).
 
1
  # wald-serve
2
 
3
+ The TypeSafe `POST /v1/systemone` server for Wald-4B: the option-letter readout over a vLLM OpenAI-compatible server or
4
+ llama.cpp's `llama-server`, the effort gate (none / low / medium / high / high-k), the knockout for more than 26
5
+ options, and the temperature table. It needs only public packages: `pydantic`, plus `vllm==0.30.0` or llama.cpp on the
6
+ machine that runs the weights.
7
 
8
  ```sh
9
  pip install ".[vllm]" # on a Linux CUDA machine
10
  wald-serve --model /path/to/Wald-4B --port 8000 # starts vLLM on the weights and serves /v1/systemone
11
  ```
12
 
13
+ GGUF weights (CPU, Apple Silicon or any GPU llama.cpp supports):
14
+
15
+ ```sh
16
+ pip install .
17
+ wald-serve --gguf ./Wald-4B-Q8_0.gguf --port 8000 # starts llama-server (on PATH) on the file
18
+ wald-serve --llamacpp http://127.0.0.1:8080 --temperature temperature.json \
19
+ --prompt-format repeat_state_plain --effort none --port 8000 # or attach to a running llama-server
20
+ ```
21
+
22
+ With `--gguf`, `serving.json` and `temperature.json` are read from the GGUF's folder. The llama.cpp backend sends token
23
+ ids to `/completion` and reads the letters from its top 100 next-token logprobs (`n_probs`); a letter outside them gets
24
+ the same floor as in vLLM. llama.cpp's context is `--max-model-len` × `--llama-parallel` (default 4 slots).
25
+
26
+ Tests (no model, no GPU; a fake server stands in for vLLM and llama-server): `pip install ".[test]" && cd tests && pytest -q`.
27
  `tests/test_parity.py` compares prompts and answers byte for byte with the reference implementation the published
28
  numbers were measured with; it is skipped unless that implementation is importable (WALD_PARITY_SRC set).
server/pyproject.toml CHANGED
@@ -1,7 +1,7 @@
1
  [project]
2
  name = "wald-serve"
3
- version = "0.1.0"
4
- description = "TypeSafe POST /v1/systemone server for Wald-4B: option-letter readout over vLLM, effort gate, knockout, calibration"
5
  readme = "README.md"
6
  license = { text = "Apache-2.0" }
7
  requires-python = ">=3.10"
 
1
  [project]
2
  name = "wald-serve"
3
+ version = "0.1.1"
4
+ description = "TypeSafe POST /v1/systemone server for Wald-4B: option-letter readout over vLLM or llama.cpp, effort gate, knockout, calibration"
5
  readme = "README.md"
6
  license = { text = "Apache-2.0" }
7
  requires-python = ">=3.10"
server/src/wald_serve/__init__.py CHANGED
@@ -1,2 +1,2 @@
1
  """Wald-4B serving: TypeSafe `POST /v1/systemone` over vLLM (letter readout, effort gate, knockout, calibration)."""
2
- __version__ = "0.1.0"
 
1
  """Wald-4B serving: TypeSafe `POST /v1/systemone` over vLLM (letter readout, effort gate, knockout, calibration)."""
2
+ __version__ = "0.1.1"
server/src/wald_serve/engine.py CHANGED
@@ -136,16 +136,18 @@ class Client:
136
  def ids(self, text: str) -> list[int]:
137
  return self.post("/tokenize", {"model": self.served, "prompt": text, "add_special_tokens": False})["tokens"]
138
 
139
- def readout(self, ids: list[int], n: int):
140
- """-> (softmax over the first n letters, total letter mass)."""
141
- if len(ids) + 1 > self.max_len:
142
- raise Capacity(f"prompt of {len(ids)} tokens is longer than the maximum context length {self.max_len}")
143
- want = [t for L in self.letter_ids[:n] for t in L]
144
  body = {"model": self.served, "prompt": ids, "max_tokens": 1, "temperature": 0.0, "logprobs": 20,
145
  "logprob_token_ids": want, "return_tokens_as_token_ids": True}
146
  r = self.post("/v1/completions", body)
147
  top = r["choices"][0]["logprobs"]["top_logprobs"][0]
148
- lp = {int(k.split(":", 1)[1]): v for k, v in top.items() if k.startswith("token_id:") and v is not None and v > -9999}
 
 
 
 
 
 
149
  floor = min(lp.values()) - 2.0 if lp else -30.0
150
  z = [math.log(sum(math.exp(lp.get(t, floor)) for t in L)) for L in self.letter_ids[:n]]
151
  m = max(z)
@@ -167,6 +169,33 @@ class Client:
167
  return c.get("text") or "", ntok
168
 
169
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
170
  # --- one question -----------------------------------------------------------------------------------------------------
171
  def chunks(n: int, size: int = 26) -> list[list[int]]:
172
  k = math.ceil(n / size)
 
136
  def ids(self, text: str) -> list[int]:
137
  return self.post("/tokenize", {"model": self.served, "prompt": text, "add_special_tokens": False})["tokens"]
138
 
139
+ def letter_logprobs(self, ids: list[int], want: list[int]) -> dict:
 
 
 
 
140
  body = {"model": self.served, "prompt": ids, "max_tokens": 1, "temperature": 0.0, "logprobs": 20,
141
  "logprob_token_ids": want, "return_tokens_as_token_ids": True}
142
  r = self.post("/v1/completions", body)
143
  top = r["choices"][0]["logprobs"]["top_logprobs"][0]
144
+ return {int(k.split(":", 1)[1]): v for k, v in top.items() if k.startswith("token_id:") and v is not None and v > -9999}
145
+
146
+ def readout(self, ids: list[int], n: int):
147
+ """-> (softmax over the first n letters, total letter mass)."""
148
+ if len(ids) + 1 > self.max_len:
149
+ raise Capacity(f"prompt of {len(ids)} tokens is longer than the maximum context length {self.max_len}")
150
+ lp = self.letter_logprobs(ids, [t for L in self.letter_ids[:n] for t in L])
151
  floor = min(lp.values()) - 2.0 if lp else -30.0
152
  z = [math.log(sum(math.exp(lp.get(t, floor)) for t in L)) for L in self.letter_ids[:n]]
153
  m = max(z)
 
169
  return c.get("text") or "", ntok
170
 
171
 
172
+ class LlamaCppClient(Client):
173
+ """The same reads over llama.cpp's `llama-server` (GGUF weights): token ids in, raw next-token logprobs out.
174
+ llama-server reports the top `n_probs` tokens only, so a letter outside them gets the same floor as in vLLM."""
175
+
176
+ N_PROBS = 100
177
+
178
+ def ids(self, text: str) -> list[int]:
179
+ return self.post("/tokenize", {"content": text, "add_special": False, "parse_special": False})["tokens"]
180
+
181
+ def letter_logprobs(self, ids: list[int], want: list[int]) -> dict:
182
+ body = {"prompt": ids, "n_predict": 1, "temperature": 0.0, "n_probs": self.N_PROBS, "post_sampling_probs": False,
183
+ "cache_prompt": True, "samplers": ["top_k"], "top_k": 1}
184
+ r = self.post("/completion", body)
185
+ top = r["completion_probabilities"][0]["top_logprobs"]
186
+ return {int(t["id"]): float(t["logprob"]) for t in top}
187
+
188
+ def generate(self, ids: list[int], max_tokens: int, seed: int, n: int = 1):
189
+ texts, ntok = [], 0
190
+ for i in range(n):
191
+ body = {"prompt": ids, "n_predict": max_tokens, "temperature": 0.6, "top_p": 0.95, "top_k": 20, "min_p": 0.0,
192
+ "seed": seed + i, "stop": [STOP], "cache_prompt": True}
193
+ r = self.post("/completion", body)
194
+ texts.append(r.get("content") or "")
195
+ ntok += int(r.get("tokens_predicted") or 0)
196
+ return (texts[0] if n == 1 else texts), ntok
197
+
198
+
199
  # --- one question -----------------------------------------------------------------------------------------------------
200
  def chunks(n: int, size: int = 26) -> list[list[int]]:
201
  k = math.ceil(n / size)
server/src/wald_serve/server.py CHANGED
@@ -8,6 +8,11 @@ or attaches to a vLLM server that is already running:
8
 
9
  wald-serve --vllm http://127.0.0.1:8011 --served wald --temperature /path/to/temperature.json --port 8000
10
 
 
 
 
 
 
11
  Defaults come from `<model>/serving.json` when present (`effort`, `prompt_format`, `max_model_len`), else the built-in
12
  ones below; command-line flags override both. The response carries TypeSafe's answer keys plus `mode` (A = one pass,
13
  B = after a thought, K = knockout) and `usage`.
@@ -30,7 +35,7 @@ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
30
  from pathlib import Path
31
 
32
  from . import __version__
33
- from .engine import Capacity, Client, answer, load_tables, policy
34
  from .prompt import PROMPT_FORMATS
35
 
36
  DEFAULTS = {"effort": "medium", "prompt_format": "plain", "max_model_len": 131072}
@@ -47,6 +52,16 @@ def launch_vllm(model: str, served: str, port: int, max_len: int, mem: float, ex
47
  cmd = [sys.executable, "-m", "vllm.entrypoints.openai.api_server", "--model", model, "--served-model-name", served,
48
  "--host", "127.0.0.1", "--port", str(port), "--max-model-len", str(max_len), "--gpu-memory-utilization",
49
  str(mem), "--max-num-seqs", "256", "--seed", "0", *shlex.split(extra)]
 
 
 
 
 
 
 
 
 
 
50
  print(json.dumps({"launching": cmd}), flush=True)
51
  proc = subprocess.Popen(cmd, start_new_session=True)
52
 
@@ -58,13 +73,13 @@ def launch_vllm(model: str, served: str, port: int, max_len: int, mem: float, ex
58
  url = f"http://127.0.0.1:{port}/health"
59
  for _ in range(1200):
60
  if proc.poll() is not None:
61
- raise SystemExit(f"vLLM exited with code {proc.returncode}")
62
  try:
63
  with urllib.request.urlopen(url, timeout=5):
64
  return proc
65
  except OSError:
66
  time.sleep(2)
67
- raise SystemExit("vLLM did not become healthy within 40 minutes")
68
 
69
 
70
  def make_handler(cl: Client, pol: dict, tables: dict, info: dict, workers: int):
@@ -109,6 +124,9 @@ def main(argv=None):
109
  src = ap.add_mutually_exclusive_group(required=True)
110
  src.add_argument("--model", help="weights directory: start vLLM on it (serving.json / temperature.json read from it)")
111
  src.add_argument("--vllm", help="URL of a running vLLM OpenAI-compatible server")
 
 
 
112
  ap.add_argument("--served", default="wald", help="vLLM served model name")
113
  ap.add_argument("--model-name", default="wald-4b", help="name reported in responses")
114
  ap.add_argument("--effort", default=None, help="none | low | medium | high | high-k<k> (default: serving.json, else medium)")
@@ -119,16 +137,20 @@ def main(argv=None):
119
  ap.add_argument("--vllm-port", type=int, default=8011)
120
  ap.add_argument("--gpu-memory-utilization", type=float, default=0.90)
121
  ap.add_argument("--vllm-args", default="", help="extra `vllm serve` arguments, e.g. \"--quantization fp8\"")
 
 
 
122
  ap.add_argument("--host", default="0.0.0.0")
123
  ap.add_argument("--port", type=int, default=8000)
124
  a = ap.parse_args(argv)
125
 
126
- cfg = {**DEFAULTS, **read_serving(a.model)}
 
127
  effort = a.effort or cfg["effort"]
128
  fmt = a.prompt_format or cfg["prompt_format"]
129
  max_len = a.max_model_len or int(cfg["max_model_len"])
130
  pol = policy(effort)
131
- temps = a.temperature or (str(Path(a.model) / cfg.get("temperature", "temperature.json")) if a.model else None)
132
  if temps and not Path(temps).is_file():
133
  raise SystemExit(f"temperature table not found: {temps}")
134
  tables = load_tables(temps, pol["budget"])
@@ -136,10 +158,14 @@ def main(argv=None):
136
  if a.model:
137
  launch_vllm(a.model, a.served, a.vllm_port, max_len, a.gpu_memory_utilization, a.vllm_args)
138
  endpoint = f"http://127.0.0.1:{a.vllm_port}"
 
 
 
139
  else:
140
- endpoint = a.vllm
141
- cl = Client(endpoint, a.served, max_len, prompt_format=fmt)
142
- info = {"model": a.model_name, "version": __version__, "effort": pol["name"], "gate": pol["gate"],
 
143
  "budget": pol["budget"], "think_k": pol["k"], "prompt_format": fmt, "max_model_len": max_len,
144
  "wide": "knockout", "max_options": 676, "temperature": bool(tables)}
145
  ThreadingHTTPServer.daemon_threads = True
 
8
 
9
  wald-serve --vllm http://127.0.0.1:8011 --served wald --temperature /path/to/temperature.json --port 8000
10
 
11
+ GGUF weights run on llama.cpp instead (`llama-server` on PATH, or --llama-server):
12
+
13
+ wald-serve --gguf Wald-4B-Q8_0.gguf --temperature temperature.json --port 8000
14
+ wald-serve --llamacpp http://127.0.0.1:8080 --temperature temperature.json --port 8000
15
+
16
  Defaults come from `<model>/serving.json` when present (`effort`, `prompt_format`, `max_model_len`), else the built-in
17
  ones below; command-line flags override both. The response carries TypeSafe's answer keys plus `mode` (A = one pass,
18
  B = after a thought, K = knockout) and `usage`.
 
35
  from pathlib import Path
36
 
37
  from . import __version__
38
+ from .engine import Capacity, Client, LlamaCppClient, answer, load_tables, policy
39
  from .prompt import PROMPT_FORMATS
40
 
41
  DEFAULTS = {"effort": "medium", "prompt_format": "plain", "max_model_len": 131072}
 
52
  cmd = [sys.executable, "-m", "vllm.entrypoints.openai.api_server", "--model", model, "--served-model-name", served,
53
  "--host", "127.0.0.1", "--port", str(port), "--max-model-len", str(max_len), "--gpu-memory-utilization",
54
  str(mem), "--max-num-seqs", "256", "--seed", "0", *shlex.split(extra)]
55
+ return launch(cmd, port, "vLLM")
56
+
57
+
58
+ def launch_llama(binary: str, gguf: str, port: int, max_len: int, parallel: int, extra: str) -> subprocess.Popen:
59
+ cmd = [binary, "-m", gguf, "--host", "127.0.0.1", "--port", str(port), "-c", str(max_len * parallel),
60
+ "-np", str(parallel), "-ngl", "999", "--seed", "0", *shlex.split(extra)]
61
+ return launch(cmd, port, "llama-server")
62
+
63
+
64
+ def launch(cmd: list, port: int, name: str) -> subprocess.Popen:
65
  print(json.dumps({"launching": cmd}), flush=True)
66
  proc = subprocess.Popen(cmd, start_new_session=True)
67
 
 
73
  url = f"http://127.0.0.1:{port}/health"
74
  for _ in range(1200):
75
  if proc.poll() is not None:
76
+ raise SystemExit(f"{name} exited with code {proc.returncode}")
77
  try:
78
  with urllib.request.urlopen(url, timeout=5):
79
  return proc
80
  except OSError:
81
  time.sleep(2)
82
+ raise SystemExit(f"{name} did not become healthy within 40 minutes")
83
 
84
 
85
  def make_handler(cl: Client, pol: dict, tables: dict, info: dict, workers: int):
 
124
  src = ap.add_mutually_exclusive_group(required=True)
125
  src.add_argument("--model", help="weights directory: start vLLM on it (serving.json / temperature.json read from it)")
126
  src.add_argument("--vllm", help="URL of a running vLLM OpenAI-compatible server")
127
+ src.add_argument("--gguf", help="GGUF weights file: start llama.cpp's llama-server on it (serving.json / temperature.json "
128
+ "read from its folder)")
129
+ src.add_argument("--llamacpp", help="URL of a running llama.cpp llama-server")
130
  ap.add_argument("--served", default="wald", help="vLLM served model name")
131
  ap.add_argument("--model-name", default="wald-4b", help="name reported in responses")
132
  ap.add_argument("--effort", default=None, help="none | low | medium | high | high-k<k> (default: serving.json, else medium)")
 
137
  ap.add_argument("--vllm-port", type=int, default=8011)
138
  ap.add_argument("--gpu-memory-utilization", type=float, default=0.90)
139
  ap.add_argument("--vllm-args", default="", help="extra `vllm serve` arguments, e.g. \"--quantization fp8\"")
140
+ ap.add_argument("--llama-server", default="llama-server", help="llama-server binary for --gguf")
141
+ ap.add_argument("--llama-parallel", type=int, default=4, help="llama-server slots for --gguf (context = max-model-len x slots)")
142
+ ap.add_argument("--llama-args", default="", help="extra llama-server arguments for --gguf")
143
  ap.add_argument("--host", default="0.0.0.0")
144
  ap.add_argument("--port", type=int, default=8000)
145
  a = ap.parse_args(argv)
146
 
147
+ home = a.model or (str(Path(a.gguf).parent) if a.gguf else None)
148
+ cfg = {**DEFAULTS, **read_serving(home)}
149
  effort = a.effort or cfg["effort"]
150
  fmt = a.prompt_format or cfg["prompt_format"]
151
  max_len = a.max_model_len or int(cfg["max_model_len"])
152
  pol = policy(effort)
153
+ temps = a.temperature or (str(Path(home) / cfg.get("temperature", "temperature.json")) if home else None)
154
  if temps and not Path(temps).is_file():
155
  raise SystemExit(f"temperature table not found: {temps}")
156
  tables = load_tables(temps, pol["budget"])
 
158
  if a.model:
159
  launch_vllm(a.model, a.served, a.vllm_port, max_len, a.gpu_memory_utilization, a.vllm_args)
160
  endpoint = f"http://127.0.0.1:{a.vllm_port}"
161
+ elif a.gguf:
162
+ launch_llama(a.llama_server, a.gguf, a.vllm_port, max_len, a.llama_parallel, a.llama_args)
163
+ endpoint = f"http://127.0.0.1:{a.vllm_port}"
164
  else:
165
+ endpoint = a.vllm or a.llamacpp
166
+ cl = (LlamaCppClient if a.gguf or a.llamacpp else Client)(endpoint, a.served, max_len, prompt_format=fmt)
167
+ info = {"model": a.model_name, "version": __version__, "backend": "llama.cpp" if a.gguf or a.llamacpp else "vllm",
168
+ "effort": pol["name"], "gate": pol["gate"],
169
  "budget": pol["budget"], "think_k": pol["k"], "prompt_format": fmt, "max_model_len": max_len,
170
  "wide": "knockout", "max_options": 676, "temperature": bool(tables)}
171
  ThreadingHTTPServer.daemon_threads = True
server/tests/fakevllm.py CHANGED
@@ -3,6 +3,9 @@
3
  Tokenizer: one token per character (so "A" is one token and " A" is two). Completions: the logprob of every requested
4
  token id is a deterministic function of the prompt ids, so a read is reproducible; generation returns a short fixed
5
  thought per seed. A prompt longer than `max_len` answers 400 "maximum context length", as vLLM does.
 
 
 
6
  """
7
  from __future__ import annotations
8
 
@@ -44,7 +47,9 @@ class FakeVLLM:
44
  body = json.loads(self.rfile.read(int(self.headers["content-length"])))
45
  if self.path == "/tokenize":
46
  fake.calls["tokenize"] += 1
47
- return self.send(200, {"tokens": [ord(c) for c in body["prompt"]]})
 
 
48
  prompt = body["prompt"]
49
  if len(prompt) + body["max_tokens"] > fake.max_len:
50
  return self.send(400, {"message": f"This model's maximum context length is {fake.max_len} tokens."})
@@ -57,6 +62,20 @@ class FakeVLLM:
57
  texts = [f" thought {body['seed'] % 997} #{i}: the evidence points one way." for i in range(n)]
58
  return self.send(200, {"choices": [{"text": t} for t in texts], "usage": {"completion_tokens": 9 * n}})
59
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
60
  self.httpd = ThreadingHTTPServer(("127.0.0.1", 0), H)
61
  self.httpd.daemon_threads = True
62
  self.url = f"http://127.0.0.1:{self.httpd.server_address[1]}"
 
3
  Tokenizer: one token per character (so "A" is one token and " A" is two). Completions: the logprob of every requested
4
  token id is a deterministic function of the prompt ids, so a read is reproducible; generation returns a short fixed
5
  thought per seed. A prompt longer than `max_len` answers 400 "maximum context length", as vLLM does.
6
+
7
+ The same server also answers llama.cpp's llama-server shapes (`/tokenize` with `content`, `/completion` with `n_probs`),
8
+ reporting the same logprobs for the printable-ASCII token ids, so both clients see one model.
9
  """
10
  from __future__ import annotations
11
 
 
47
  body = json.loads(self.rfile.read(int(self.headers["content-length"])))
48
  if self.path == "/tokenize":
49
  fake.calls["tokenize"] += 1
50
+ return self.send(200, {"tokens": [ord(c) for c in body.get("prompt", body.get("content"))]})
51
+ if self.path == "/completion":
52
+ return self.llama(body)
53
  prompt = body["prompt"]
54
  if len(prompt) + body["max_tokens"] > fake.max_len:
55
  return self.send(400, {"message": f"This model's maximum context length is {fake.max_len} tokens."})
 
62
  texts = [f" thought {body['seed'] % 997} #{i}: the evidence points one way." for i in range(n)]
63
  return self.send(200, {"choices": [{"text": t} for t in texts], "usage": {"completion_tokens": 9 * n}})
64
 
65
+ def llama(self, body):
66
+ prompt = body["prompt"]
67
+ if len(prompt) + body["n_predict"] > fake.max_len:
68
+ return self.send(400, {"error": {"message": "the request exceeds the available context size"}})
69
+ if body["n_predict"] == 1:
70
+ fake.calls["read"] += 1
71
+ top = sorted(({"id": t, "token": chr(t), "logprob": _lp(prompt, t)} for t in range(32, 127)),
72
+ key=lambda x: -x["logprob"])[: body["n_probs"]]
73
+ return self.send(200, {"content": top[0]["token"],
74
+ "completion_probabilities": [{**top[0], "top_logprobs": top}]})
75
+ fake.calls["generate"] += 1
76
+ return self.send(200, {"content": f" thought {body['seed'] % 997}: the evidence points one way.",
77
+ "tokens_predicted": 9})
78
+
79
  self.httpd = ThreadingHTTPServer(("127.0.0.1", 0), H)
80
  self.httpd.daemon_threads = True
81
  self.url = f"http://127.0.0.1:{self.httpd.server_address[1]}"
server/tests/test_server.py CHANGED
@@ -12,7 +12,7 @@ from pathlib import Path
12
  import pytest
13
 
14
  from fakevllm import FakeVLLM
15
- from wald_serve.engine import (POLICIES, Client, answer, bucket_key, chunks, load_tables, policy, temper, temperature,
16
  wide_read)
17
  from wald_serve.prompt import STATE_REPEAT, format_prompt, question_prompt
18
  from wald_serve.server import make_handler
@@ -249,3 +249,47 @@ def test_thought_that_does_not_fit_falls_back(fake):
249
  cl = Client(fake.url, "wald", 400) # the prompt fits, prompt + 512-token budget does not
250
  a, u = answer(cl, q, policy("high"), {})
251
  assert a["q"]["mode"] == "A" and u["output_tokens"] == 0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
12
  import pytest
13
 
14
  from fakevllm import FakeVLLM
15
+ from wald_serve.engine import (POLICIES, Client, LlamaCppClient, answer, bucket_key, chunks, load_tables, policy, temper, temperature,
16
  wide_read)
17
  from wald_serve.prompt import STATE_REPEAT, format_prompt, question_prompt
18
  from wald_serve.server import make_handler
 
249
  cl = Client(fake.url, "wald", 400) # the prompt fits, prompt + 512-token budget does not
250
  a, u = answer(cl, q, policy("high"), {})
251
  assert a["q"]["mode"] == "A" and u["output_tokens"] == 0
252
+
253
+
254
+ def test_llamacpp_client_matches_vllm_client(fake):
255
+ vl, lc = Client(fake.url, "wald", 100_000), LlamaCppClient(fake.url, "wald", 100_000)
256
+ assert lc.letter_ids == vl.letter_ids
257
+ for req in (REQ, wide_request(77)):
258
+ a, ua = answer(vl, req, policy("none"), {})
259
+ b, ub = answer(lc, req, policy("none"), {})
260
+ assert a == b and ua == ub
261
+ high, uh = answer(lc, REQ, policy("high-k3"), {})
262
+ assert all(x["mode"] == "B" for x in high.values()) and uh["output_tokens"] > 0
263
+ check_wire(REQ, {"answers": high})
264
+
265
+
266
+ def test_llamacpp_capacity_is_422(fake):
267
+ cl = LlamaCppClient(fake.url, "wald", 300)
268
+ httpd, url = serve(cl, "none")
269
+ try:
270
+ code, resp = post(url, {"state": "x" * 400, "questions": {"q": {"type": "noul", "instructions": "Is it?"}}})
271
+ assert code == 422 and "maximum context length" in resp["error"]
272
+ finally:
273
+ httpd.shutdown()
274
+
275
+
276
+ def test_gguf_reads_serving_json_beside_the_file(tmp_path, monkeypatch):
277
+ import wald_serve.server as srv
278
+ (tmp_path / "serving.json").write_text(json.dumps({"effort": "none", "prompt_format": "repeat_state_plain",
279
+ "max_model_len": 4096, "temperature": "temperature.json"}))
280
+ (tmp_path / "temperature.json").write_text(json.dumps(TABLE))
281
+ seen = {}
282
+ monkeypatch.setattr(srv, "launch_llama", lambda *a: seen.update(launch=a))
283
+ monkeypatch.setattr(srv, "LlamaCppClient", lambda *a, **k: seen.update(client=(a, k)))
284
+
285
+ class Stop(Exception):
286
+ pass
287
+
288
+ def fake_http(addr, handler):
289
+ seen["handler"] = handler
290
+ raise Stop
291
+ monkeypatch.setattr(srv, "ThreadingHTTPServer", fake_http)
292
+ with pytest.raises(Stop):
293
+ srv.main(["--gguf", str(tmp_path / "Wald-4B-Q8_0.gguf"), "--port", "0"])
294
+ assert seen["launch"][3] == 4096
295
+ assert seen["client"][1] == {"prompt_format": "repeat_state_plain"}