Lie24 commited on
Commit
a54b311
·
0 Parent(s):

StartLux-Decision-27B-Q8_0-GGUF

Browse files
.gitattributes ADDED
@@ -0,0 +1,37 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
37
+ StartLux-Decision-27B-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,408 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Attribution-NonCommercial 4.0 International
2
+
3
+ =======================================================================
4
+
5
+ Creative Commons Corporation ("Creative Commons") is not a law firm and
6
+ does not provide legal services or legal advice. Distribution of
7
+ Creative Commons public licenses does not create a lawyer-client or
8
+ other relationship. Creative Commons makes its licenses and related
9
+ information available on an "as-is" basis. Creative Commons gives no
10
+ warranties regarding its licenses, any material licensed under their
11
+ terms and conditions, or any related information. Creative Commons
12
+ disclaims all liability for damages resulting from their use to the
13
+ fullest extent possible.
14
+
15
+ Using Creative Commons Public Licenses
16
+
17
+ Creative Commons public licenses provide a standard set of terms and
18
+ conditions that creators and other rights holders may use to share
19
+ original works of authorship and other material subject to copyright
20
+ and certain other rights specified in the public license below. The
21
+ following considerations are for informational purposes only, are not
22
+ exhaustive, and do not form part of our licenses.
23
+
24
+ Considerations for licensors: Our public licenses are
25
+ intended for use by those authorized to give the public
26
+ permission to use material in ways otherwise restricted by
27
+ copyright and certain other rights. Our licenses are
28
+ irrevocable. Licensors should read and understand the terms
29
+ and conditions of the license they choose before applying it.
30
+ Licensors should also secure all rights necessary before
31
+ applying our licenses so that the public can reuse the
32
+ material as expected. Licensors should clearly mark any
33
+ material not subject to the license. This includes other CC-
34
+ licensed material, or material used under an exception or
35
+ limitation to copyright. More considerations for licensors:
36
+ wiki.creativecommons.org/Considerations_for_licensors
37
+
38
+ Considerations for the public: By using one of our public
39
+ licenses, a licensor grants the public permission to use the
40
+ licensed material under specified terms and conditions. If
41
+ the licensor's permission is not necessary for any reason--for
42
+ example, because of any applicable exception or limitation to
43
+ copyright--then that use is not regulated by the license. Our
44
+ licenses grant only permissions under copyright and certain
45
+ other rights that a licensor has authority to grant. Use of
46
+ the licensed material may still be restricted for other
47
+ reasons, including because others have copyright or other
48
+ rights in the material. A licensor may make special requests,
49
+ such as asking that all changes be marked or described.
50
+ Although not required by our licenses, you are encouraged to
51
+ respect those requests where reasonable. More considerations
52
+ for the public:
53
+ wiki.creativecommons.org/Considerations_for_licensees
54
+
55
+ =======================================================================
56
+
57
+ Creative Commons Attribution-NonCommercial 4.0 International Public
58
+ License
59
+
60
+ By exercising the Licensed Rights (defined below), You accept and agree
61
+ to be bound by the terms and conditions of this Creative Commons
62
+ Attribution-NonCommercial 4.0 International Public License ("Public
63
+ License"). To the extent this Public License may be interpreted as a
64
+ contract, You are granted the Licensed Rights in consideration of Your
65
+ acceptance of these terms and conditions, and the Licensor grants You
66
+ such rights in consideration of benefits the Licensor receives from
67
+ making the Licensed Material available under these terms and
68
+ conditions.
69
+
70
+
71
+ Section 1 -- Definitions.
72
+
73
+ a. Adapted Material means material subject to Copyright and Similar
74
+ Rights that is derived from or based upon the Licensed Material
75
+ and in which the Licensed Material is translated, altered,
76
+ arranged, transformed, or otherwise modified in a manner requiring
77
+ permission under the Copyright and Similar Rights held by the
78
+ Licensor. For purposes of this Public License, where the Licensed
79
+ Material is a musical work, performance, or sound recording,
80
+ Adapted Material is always produced where the Licensed Material is
81
+ synched in timed relation with a moving image.
82
+
83
+ b. Adapter's License means the license You apply to Your Copyright
84
+ and Similar Rights in Your contributions to Adapted Material in
85
+ accordance with the terms and conditions of this Public License.
86
+
87
+ c. Copyright and Similar Rights means copyright and/or similar rights
88
+ closely related to copyright including, without limitation,
89
+ performance, broadcast, sound recording, and Sui Generis Database
90
+ Rights, without regard to how the rights are labeled or
91
+ categorized. For purposes of this Public License, the rights
92
+ specified in Section 2(b)(1)-(2) are not Copyright and Similar
93
+ Rights.
94
+ d. Effective Technological Measures means those measures that, in the
95
+ absence of proper authority, may not be circumvented under laws
96
+ fulfilling obligations under Article 11 of the WIPO Copyright
97
+ Treaty adopted on December 20, 1996, and/or similar international
98
+ agreements.
99
+
100
+ e. Exceptions and Limitations means fair use, fair dealing, and/or
101
+ any other exception or limitation to Copyright and Similar Rights
102
+ that applies to Your use of the Licensed Material.
103
+
104
+ f. Licensed Material means the artistic or literary work, database,
105
+ or other material to which the Licensor applied this Public
106
+ License.
107
+
108
+ g. Licensed Rights means the rights granted to You subject to the
109
+ terms and conditions of this Public License, which are limited to
110
+ all Copyright and Similar Rights that apply to Your use of the
111
+ Licensed Material and that the Licensor has authority to license.
112
+
113
+ h. Licensor means the individual(s) or entity(ies) granting rights
114
+ under this Public License.
115
+
116
+ i. NonCommercial means not primarily intended for or directed towards
117
+ commercial advantage or monetary compensation. For purposes of
118
+ this Public License, the exchange of the Licensed Material for
119
+ other material subject to Copyright and Similar Rights by digital
120
+ file-sharing or similar means is NonCommercial provided there is
121
+ no payment of monetary compensation in connection with the
122
+ exchange.
123
+
124
+ j. Share means to provide material to the public by any means or
125
+ process that requires permission under the Licensed Rights, such
126
+ as reproduction, public display, public performance, distribution,
127
+ dissemination, communication, or importation, and to make material
128
+ available to the public including in ways that members of the
129
+ public may access the material from a place and at a time
130
+ individually chosen by them.
131
+
132
+ k. Sui Generis Database Rights means rights other than copyright
133
+ resulting from Directive 96/9/EC of the European Parliament and of
134
+ the Council of 11 March 1996 on the legal protection of databases,
135
+ as amended and/or succeeded, as well as other essentially
136
+ equivalent rights anywhere in the world.
137
+
138
+ l. You means the individual or entity exercising the Licensed Rights
139
+ under this Public License. Your has a corresponding meaning.
140
+
141
+
142
+ Section 2 -- Scope.
143
+
144
+ a. License grant.
145
+
146
+ 1. Subject to the terms and conditions of this Public License,
147
+ the Licensor hereby grants You a worldwide, royalty-free,
148
+ non-sublicensable, non-exclusive, irrevocable license to
149
+ exercise the Licensed Rights in the Licensed Material to:
150
+
151
+ a. reproduce and Share the Licensed Material, in whole or
152
+ in part, for NonCommercial purposes only; and
153
+
154
+ b. produce, reproduce, and Share Adapted Material for
155
+ NonCommercial purposes only.
156
+
157
+ 2. Exceptions and Limitations. For the avoidance of doubt, where
158
+ Exceptions and Limitations apply to Your use, this Public
159
+ License does not apply, and You do not need to comply with
160
+ its terms and conditions.
161
+
162
+ 3. Term. The term of this Public License is specified in Section
163
+ 6(a).
164
+
165
+ 4. Media and formats; technical modifications allowed. The
166
+ Licensor authorizes You to exercise the Licensed Rights in
167
+ all media and formats whether now known or hereafter created,
168
+ and to make technical modifications necessary to do so. The
169
+ Licensor waives and/or agrees not to assert any right or
170
+ authority to forbid You from making technical modifications
171
+ necessary to exercise the Licensed Rights, including
172
+ technical modifications necessary to circumvent Effective
173
+ Technological Measures. For purposes of this Public License,
174
+ simply making modifications authorized by this Section 2(a)
175
+ (4) never produces Adapted Material.
176
+
177
+ 5. Downstream recipients.
178
+
179
+ a. Offer from the Licensor -- Licensed Material. Every
180
+ recipient of the Licensed Material automatically
181
+ receives an offer from the Licensor to exercise the
182
+ Licensed Rights under the terms and conditions of this
183
+ Public License.
184
+
185
+ b. No downstream restrictions. You may not offer or impose
186
+ any additional or different terms or conditions on, or
187
+ apply any Effective Technological Measures to, the
188
+ Licensed Material if doing so restricts exercise of the
189
+ Licensed Rights by any recipient of the Licensed
190
+ Material.
191
+
192
+ 6. No endorsement. Nothing in this Public License constitutes or
193
+ may be construed as permission to assert or imply that You
194
+ are, or that Your use of the Licensed Material is, connected
195
+ with, or sponsored, endorsed, or granted official status by,
196
+ the Licensor or others designated to receive attribution as
197
+ provided in Section 3(a)(1)(A)(i).
198
+
199
+ b. Other rights.
200
+
201
+ 1. Moral rights, such as the right of integrity, are not
202
+ licensed under this Public License, nor are publicity,
203
+ privacy, and/or other similar personality rights; however, to
204
+ the extent possible, the Licensor waives and/or agrees not to
205
+ assert any such rights held by the Licensor to the limited
206
+ extent necessary to allow You to exercise the Licensed
207
+ Rights, but not otherwise.
208
+
209
+ 2. Patent and trademark rights are not licensed under this
210
+ Public License.
211
+
212
+ 3. To the extent possible, the Licensor waives any right to
213
+ collect royalties from You for the exercise of the Licensed
214
+ Rights, whether directly or through a collecting society
215
+ under any voluntary or waivable statutory or compulsory
216
+ licensing scheme. In all other cases the Licensor expressly
217
+ reserves any right to collect such royalties, including when
218
+ the Licensed Material is used other than for NonCommercial
219
+ purposes.
220
+
221
+
222
+ Section 3 -- License Conditions.
223
+
224
+ Your exercise of the Licensed Rights is expressly made subject to the
225
+ following conditions.
226
+
227
+ a. Attribution.
228
+
229
+ 1. If You Share the Licensed Material (including in modified
230
+ form), You must:
231
+
232
+ a. retain the following if it is supplied by the Licensor
233
+ with the Licensed Material:
234
+
235
+ i. identification of the creator(s) of the Licensed
236
+ Material and any others designated to receive
237
+ attribution, in any reasonable manner requested by
238
+ the Licensor (including by pseudonym if
239
+ designated);
240
+
241
+ ii. a copyright notice;
242
+
243
+ iii. a notice that refers to this Public License;
244
+
245
+ iv. a notice that refers to the disclaimer of
246
+ warranties;
247
+
248
+ v. a URI or hyperlink to the Licensed Material to the
249
+ extent reasonably practicable;
250
+
251
+ b. indicate if You modified the Licensed Material and
252
+ retain an indication of any previous modifications; and
253
+
254
+ c. indicate the Licensed Material is licensed under this
255
+ Public License, and include the text of, or the URI or
256
+ hyperlink to, this Public License.
257
+
258
+ 2. You may satisfy the conditions in Section 3(a)(1) in any
259
+ reasonable manner based on the medium, means, and context in
260
+ which You Share the Licensed Material. For example, it may be
261
+ reasonable to satisfy the conditions by providing a URI or
262
+ hyperlink to a resource that includes the required
263
+ information.
264
+
265
+ 3. If requested by the Licensor, You must remove any of the
266
+ information required by Section 3(a)(1)(A) to the extent
267
+ reasonably practicable.
268
+
269
+ 4. If You Share Adapted Material You produce, the Adapter's
270
+ License You apply must not prevent recipients of the Adapted
271
+ Material from complying with this Public License.
272
+
273
+
274
+ Section 4 -- Sui Generis Database Rights.
275
+
276
+ Where the Licensed Rights include Sui Generis Database Rights that
277
+ apply to Your use of the Licensed Material:
278
+
279
+ a. for the avoidance of doubt, Section 2(a)(1) grants You the right
280
+ to extract, reuse, reproduce, and Share all or a substantial
281
+ portion of the contents of the database for NonCommercial purposes
282
+ only;
283
+
284
+ b. if You include all or a substantial portion of the database
285
+ contents in a database in which You have Sui Generis Database
286
+ Rights, then the database in which You have Sui Generis Database
287
+ Rights (but not its individual contents) is Adapted Material; and
288
+
289
+ c. You must comply with the conditions in Section 3(a) if You Share
290
+ all or a substantial portion of the contents of the database.
291
+
292
+ For the avoidance of doubt, this Section 4 supplements and does not
293
+ replace Your obligations under this Public License where the Licensed
294
+ Rights include other Copyright and Similar Rights.
295
+
296
+
297
+ Section 5 -- Disclaimer of Warranties and Limitation of Liability.
298
+
299
+ a. UNLESS OTHERWISE SEPARATELY UNDERTAKEN BY THE LICENSOR, TO THE
300
+ EXTENT POSSIBLE, THE LICENSOR OFFERS THE LICENSED MATERIAL AS-IS
301
+ AND AS-AVAILABLE, AND MAKES NO REPRESENTATIONS OR WARRANTIES OF
302
+ ANY KIND CONCERNING THE LICENSED MATERIAL, WHETHER EXPRESS,
303
+ IMPLIED, STATUTORY, OR OTHER. THIS INCLUDES, WITHOUT LIMITATION,
304
+ WARRANTIES OF TITLE, MERCHANTABILITY, FITNESS FOR A PARTICULAR
305
+ PURPOSE, NON-INFRINGEMENT, ABSENCE OF LATENT OR OTHER DEFECTS,
306
+ ACCURACY, OR THE PRESENCE OR ABSENCE OF ERRORS, WHETHER OR NOT
307
+ KNOWN OR DISCOVERABLE. WHERE DISCLAIMERS OF WARRANTIES ARE NOT
308
+ ALLOWED IN FULL OR IN PART, THIS DISCLAIMER MAY NOT APPLY TO YOU.
309
+
310
+ b. TO THE EXTENT POSSIBLE, IN NO EVENT WILL THE LICENSOR BE LIABLE
311
+ TO YOU ON ANY LEGAL THEORY (INCLUDING, WITHOUT LIMITATION,
312
+ NEGLIGENCE) OR OTHERWISE FOR ANY DIRECT, SPECIAL, INDIRECT,
313
+ INCIDENTAL, CONSEQUENTIAL, PUNITIVE, EXEMPLARY, OR OTHER LOSSES,
314
+ COSTS, EXPENSES, OR DAMAGES ARISING OUT OF THIS PUBLIC LICENSE OR
315
+ USE OF THE LICENSED MATERIAL, EVEN IF THE LICENSOR HAS BEEN
316
+ ADVISED OF THE POSSIBILITY OF SUCH LOSSES, COSTS, EXPENSES, OR
317
+ DAMAGES. WHERE A LIMITATION OF LIABILITY IS NOT ALLOWED IN FULL OR
318
+ IN PART, THIS LIMITATION MAY NOT APPLY TO YOU.
319
+
320
+ c. The disclaimer of warranties and limitation of liability provided
321
+ above shall be interpreted in a manner that, to the extent
322
+ possible, most closely approximates an absolute disclaimer and
323
+ waiver of all liability.
324
+
325
+
326
+ Section 6 -- Term and Termination.
327
+
328
+ a. This Public License applies for the term of the Copyright and
329
+ Similar Rights licensed here. However, if You fail to comply with
330
+ this Public License, then Your rights under this Public License
331
+ terminate automatically.
332
+
333
+ b. Where Your right to use the Licensed Material has terminated under
334
+ Section 6(a), it reinstates:
335
+
336
+ 1. automatically as of the date the violation is cured, provided
337
+ it is cured within 30 days of Your discovery of the
338
+ violation; or
339
+
340
+ 2. upon express reinstatement by the Licensor.
341
+
342
+ For the avoidance of doubt, this Section 6(b) does not affect any
343
+ right the Licensor may have to seek remedies for Your violations
344
+ of this Public License.
345
+
346
+ c. For the avoidance of doubt, the Licensor may also offer the
347
+ Licensed Material under separate terms or conditions or stop
348
+ distributing the Licensed Material at any time; however, doing so
349
+ will not terminate this Public License.
350
+
351
+ d. Sections 1, 5, 6, 7, and 8 survive termination of this Public
352
+ License.
353
+
354
+
355
+ Section 7 -- Other Terms and Conditions.
356
+
357
+ a. The Licensor shall not be bound by any additional or different
358
+ terms or conditions communicated by You unless expressly agreed.
359
+
360
+ b. Any arrangements, understandings, or agreements regarding the
361
+ Licensed Material not stated herein are separate from and
362
+ independent of the terms and conditions of this Public License.
363
+
364
+
365
+ Section 8 -- Interpretation.
366
+
367
+ a. For the avoidance of doubt, this Public License does not, and
368
+ shall not be interpreted to, reduce, limit, restrict, or impose
369
+ conditions on any use of the Licensed Material that could lawfully
370
+ be made without permission under this Public License.
371
+
372
+ b. To the extent possible, if any provision of this Public License is
373
+ deemed unenforceable, it shall be automatically reformed to the
374
+ minimum extent necessary to make it enforceable. If the provision
375
+ cannot be reformed, it shall be severed from this Public License
376
+ without affecting the enforceability of the remaining terms and
377
+ conditions.
378
+
379
+ c. No term or condition of this Public License will be waived and no
380
+ failure to comply consented to unless expressly agreed to by the
381
+ Licensor.
382
+
383
+ d. Nothing in this Public License constitutes or may be interpreted
384
+ as a limitation upon, or waiver of, any privileges and immunities
385
+ that apply to the Licensor or You, including from the legal
386
+ processes of any jurisdiction or authority.
387
+
388
+ =======================================================================
389
+
390
+ Creative Commons is not a party to its public
391
+ licenses. Notwithstanding, Creative Commons may elect to apply one of
392
+ its public licenses to material it publishes and in those instances
393
+ will be considered the “Licensor.” The text of the Creative Commons
394
+ public licenses is dedicated to the public domain under the CC0 Public
395
+ Domain Dedication. Except for the limited purpose of indicating that
396
+ material is shared under a Creative Commons public license or as
397
+ otherwise permitted by the Creative Commons policies published at
398
+ creativecommons.org/policies, Creative Commons does not authorize the
399
+ use of the trademark "Creative Commons" or any other trademark or logo
400
+ of Creative Commons without its prior written consent including,
401
+ without limitation, in connection with any unauthorized modifications
402
+ to any of its public licenses or any other arrangements,
403
+ understandings, or agreements concerning use of licensed material. For
404
+ the avoidance of doubt, this paragraph does not form part of the
405
+ public licenses.
406
+
407
+ Creative Commons may be contacted at creativecommons.org.
408
+
NOTICE ADDED
@@ -0,0 +1,227 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ StartLux-Decision
2
+ Copyright 2026 StartLux Labs
3
+
4
+ Model weights
5
+ -------------
6
+ The StartLux-Decision model weights are licensed under the Creative Commons Attribution-NonCommercial 4.0
7
+ International License (CC BY-NC 4.0); the full text is in LICENSE. You may use, share and adapt them for
8
+ non-commercial purposes, with attribution. Commercial use requires a separate license from StartLux Labs:
9
+ contact contact@startlux.com.
10
+
11
+ The weights are a modified version of a model released under the Apache License, Version 2.0. That model's
12
+ copyright notice is retained here:
13
+
14
+ Copyright 2026 Alibaba Cloud
15
+
16
+ Files taken unchanged from that model, such as the tokenizer files, remain under the Apache License, Version 2.0.
17
+
18
+ Inference code
19
+ --------------
20
+ The inference code in startlux_decision/ is licensed under the Apache License, Version 2.0.
21
+
22
+ The full text of the Apache License, Version 2.0 follows.
23
+
24
+ ========================================================================
25
+
26
+ Apache License
27
+ Version 2.0, January 2004
28
+ http://www.apache.org/licenses/
29
+
30
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
31
+
32
+ 1. Definitions.
33
+
34
+ "License" shall mean the terms and conditions for use, reproduction,
35
+ and distribution as defined by Sections 1 through 9 of this document.
36
+
37
+ "Licensor" shall mean the copyright owner or entity authorized by
38
+ the copyright owner that is granting the License.
39
+
40
+ "Legal Entity" shall mean the union of the acting entity and all
41
+ other entities that control, are controlled by, or are under common
42
+ control with that entity. For the purposes of this definition,
43
+ "control" means (i) the power, direct or indirect, to cause the
44
+ direction or management of such entity, whether by contract or
45
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
46
+ outstanding shares, or (iii) beneficial ownership of such entity.
47
+
48
+ "You" (or "Your") shall mean an individual or Legal Entity
49
+ exercising permissions granted by this License.
50
+
51
+ "Source" form shall mean the preferred form for making modifications,
52
+ including but not limited to software source code, documentation
53
+ source, and configuration files.
54
+
55
+ "Object" form shall mean any form resulting from mechanical
56
+ transformation or translation of a Source form, including but
57
+ not limited to compiled object code, generated documentation,
58
+ and conversions to other media types.
59
+
60
+ "Work" shall mean the work of authorship, whether in Source or
61
+ Object form, made available under the License, as indicated by a
62
+ copyright notice that is included in or attached to the work
63
+ (an example is provided in the Appendix below).
64
+
65
+ "Derivative Works" shall mean any work, whether in Source or Object
66
+ form, that is based on (or derived from) the Work and for which the
67
+ editorial revisions, annotations, elaborations, or other modifications
68
+ represent, as a whole, an original work of authorship. For the purposes
69
+ of this License, Derivative Works shall not include works that remain
70
+ separable from, or merely link (or bind by name) to the interfaces of,
71
+ the Work and Derivative Works thereof.
72
+
73
+ "Contribution" shall mean any work of authorship, including
74
+ the original version of the Work and any modifications or additions
75
+ to that Work or Derivative Works thereof, that is intentionally
76
+ submitted to Licensor for inclusion in the Work by the copyright owner
77
+ or by an individual or Legal Entity authorized to submit on behalf of
78
+ the copyright owner. For the purposes of this definition, "submitted"
79
+ means any form of electronic, verbal, or written communication sent
80
+ to the Licensor or its representatives, including but not limited to
81
+ communication on electronic mailing lists, source code control systems,
82
+ and issue tracking systems that are managed by, or on behalf of, the
83
+ Licensor for the purpose of discussing and improving the Work, but
84
+ excluding communication that is conspicuously marked or otherwise
85
+ designated in writing by the copyright owner as "Not a Contribution."
86
+
87
+ "Contributor" shall mean Licensor and any individual or Legal Entity
88
+ on behalf of whom a Contribution has been received by Licensor and
89
+ subsequently incorporated within the Work.
90
+
91
+ 2. Grant of Copyright License. Subject to the terms and conditions of
92
+ this License, each Contributor hereby grants to You a perpetual,
93
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
94
+ copyright license to reproduce, prepare Derivative Works of,
95
+ publicly display, publicly perform, sublicense, and distribute the
96
+ Work and such Derivative Works in Source or Object form.
97
+
98
+ 3. Grant of Patent License. Subject to the terms and conditions of
99
+ this License, each Contributor hereby grants to You a perpetual,
100
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
101
+ (except as stated in this section) patent license to make, have made,
102
+ use, offer to sell, sell, import, and otherwise transfer the Work,
103
+ where such license applies only to those patent claims licensable
104
+ by such Contributor that are necessarily infringed by their
105
+ Contribution(s) alone or by combination of their Contribution(s)
106
+ with the Work to which such Contribution(s) was submitted. If You
107
+ institute patent litigation against any entity (including a
108
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
109
+ or a Contribution incorporated within the Work constitutes direct
110
+ or contributory patent infringement, then any patent licenses
111
+ granted to You under this License for that Work shall terminate
112
+ as of the date such litigation is filed.
113
+
114
+ 4. Redistribution. You may reproduce and distribute copies of the
115
+ Work or Derivative Works thereof in any medium, with or without
116
+ modifications, and in Source or Object form, provided that You
117
+ meet the following conditions:
118
+
119
+ (a) You must give any other recipients of the Work or
120
+ Derivative Works a copy of this License; and
121
+
122
+ (b) You must cause any modified files to carry prominent notices
123
+ stating that You changed the files; and
124
+
125
+ (c) You must retain, in the Source form of any Derivative Works
126
+ that You distribute, all copyright, patent, trademark, and
127
+ attribution notices from the Source form of the Work,
128
+ excluding those notices that do not pertain to any part of
129
+ the Derivative Works; and
130
+
131
+ (d) If the Work includes a "NOTICE" text file as part of its
132
+ distribution, then any Derivative Works that You distribute must
133
+ include a readable copy of the attribution notices contained
134
+ within such NOTICE file, excluding those notices that do not
135
+ pertain to any part of the Derivative Works, in at least one
136
+ of the following places: within a NOTICE text file distributed
137
+ as part of the Derivative Works; within the Source form or
138
+ documentation, if provided along with the Derivative Works; or,
139
+ within a display generated by the Derivative Works, if and
140
+ wherever such third-party notices normally appear. The contents
141
+ of the NOTICE file are for informational purposes only and
142
+ do not modify the License. You may add Your own attribution
143
+ notices within Derivative Works that You distribute, alongside
144
+ or as an addendum to the NOTICE text from the Work, provided
145
+ that such additional attribution notices cannot be construed
146
+ as modifying the License.
147
+
148
+ You may add Your own copyright statement to Your modifications and
149
+ may provide additional or different license terms and conditions
150
+ for use, reproduction, or distribution of Your modifications, or
151
+ for any such Derivative Works as a whole, provided Your use,
152
+ reproduction, and distribution of the Work otherwise complies with
153
+ the conditions stated in this License.
154
+
155
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
156
+ any Contribution intentionally submitted for inclusion in the Work
157
+ by You to the Licensor shall be under the terms and conditions of
158
+ this License, without any additional terms or conditions.
159
+ Notwithstanding the above, nothing herein shall supersede or modify
160
+ the terms of any separate license agreement you may have executed
161
+ with Licensor regarding such Contributions.
162
+
163
+ 6. Trademarks. This License does not grant permission to use the trade
164
+ names, trademarks, service marks, or product names of the Licensor,
165
+ except as required for reasonable and customary use in describing the
166
+ origin of the Work and reproducing the content of the NOTICE file.
167
+
168
+ 7. Disclaimer of Warranty. Unless required by applicable law or
169
+ agreed to in writing, Licensor provides the Work (and each
170
+ Contributor provides its Contributions) on an "AS IS" BASIS,
171
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
172
+ implied, including, without limitation, any warranties or conditions
173
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
174
+ PARTICULAR PURPOSE. You are solely responsible for determining the
175
+ appropriateness of using or redistributing the Work and assume any
176
+ risks associated with Your exercise of permissions under this License.
177
+
178
+ 8. Limitation of Liability. In no event and under no legal theory,
179
+ whether in tort (including negligence), contract, or otherwise,
180
+ unless required by applicable law (such as deliberate and grossly
181
+ negligent acts) or agreed to in writing, shall any Contributor be
182
+ liable to You for damages, including any direct, indirect, special,
183
+ incidental, or consequential damages of any character arising as a
184
+ result of this License or out of the use or inability to use the
185
+ Work (including but not limited to damages for loss of goodwill,
186
+ work stoppage, computer failure or malfunction, or any and all
187
+ other commercial damages or losses), even if such Contributor
188
+ has been advised of the possibility of such damages.
189
+
190
+ 9. Accepting Warranty or Additional Liability. While redistributing
191
+ the Work or Derivative Works thereof, You may choose to offer,
192
+ and charge a fee for, acceptance of support, warranty, indemnity,
193
+ or other liability obligations and/or rights consistent with this
194
+ License. However, in accepting such obligations, You may act only
195
+ on Your own behalf and on Your sole responsibility, not on behalf
196
+ of any other Contributor, and only if You agree to indemnify,
197
+ defend, and hold each Contributor harmless for any liability
198
+ incurred by, or claims asserted against, such Contributor by reason
199
+ of your accepting any such warranty or additional liability.
200
+
201
+ END OF TERMS AND CONDITIONS
202
+
203
+ APPENDIX: How to apply the Apache License to your work.
204
+
205
+ To apply the Apache License to your work, attach the following
206
+ boilerplate notice, with the fields enclosed by brackets "[]"
207
+ replaced with your own identifying information. (Don't include
208
+ the brackets!) The text should be enclosed in the appropriate
209
+ comment syntax for the file format. We also recommend that a
210
+ file or class name and description of purpose be included on the
211
+ same "printed page" as the copyright notice for easier
212
+ identification within third-party archives.
213
+
214
+ Copyright [yyyy] [name of copyright owner]
215
+
216
+ Licensed under the Apache License, Version 2.0 (the "License");
217
+ you may not use this file except in compliance with the License.
218
+ You may obtain a copy of the License at
219
+
220
+ http://www.apache.org/licenses/LICENSE-2.0
221
+
222
+ Unless required by applicable law or agreed to in writing, software
223
+ distributed under the License is distributed on an "AS IS" BASIS,
224
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
225
+ See the License for the specific language governing permissions and
226
+ limitations under the License.
227
+
README.md ADDED
@@ -0,0 +1,73 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: cc-by-nc-4.0
3
+ base_model: startlux-models/StartLux-Decision-27B
4
+ base_model_relation: quantized
5
+ tags:
6
+ - gguf
7
+ - llama.cpp
8
+ - decision-model
9
+ - typed-decisions
10
+ ---
11
+
12
+ # StartLux-Decision-27B-Q8_0-GGUF
13
+
14
+ [StartLux-Decision-27B](https://huggingface.co/startlux-models/StartLux-Decision-27B) in Q8_0 (8-bit) as a GGUF file for
15
+ llama.cpp. The decision server in this repository turns it into the same typed decisions, with a probability for every
16
+ option, as the original model.
17
+
18
+ ## Precisions
19
+
20
+ <div align="center">
21
+
22
+ | Repository | Bits | Size | Same decision as the original | JevBench public, of 231 | |
23
+ |:---:|:---:|:---:|:---:|:---:|:---:|
24
+ | **StartLux-Decision-27B-Q8_0-GGUF** (this repository) | 8-bit | 28.60 GB | 100.0% | 209 (90.5%) | recommended |
25
+ | [StartLux-Decision-27B-Q4_K_M-GGUF](https://huggingface.co/startlux-models/StartLux-Decision-27B-Q4_K_M-GGUF) | 4-bit | 16.55 GB | 96.5% | 208 (90.0%) | smallest |
26
+ | [StartLux-Decision-27B-BF16-GGUF](https://huggingface.co/startlux-models/StartLux-Decision-27B-BF16-GGUF) | 16-bit | 53.81 GB | 100.0% | 209 (90.5%) | the original weights, unchanged |
27
+ | [StartLux-Decision-27B](https://huggingface.co/startlux-models/StartLux-Decision-27B) (original weights) | | | | 209 (90.5%) | for comparison |
28
+
29
+ </div>
30
+
31
+ "Same decision" is the share of the 231 public JevBench items on which the file picks the same answer as the original
32
+ weights run through the `startlux_decision` package, with the same prompts, readout and temperatures. Q8_0 and Q4_K_M
33
+ are llama.cpp's standard quantizations of BF16, without an importance matrix.
34
+
35
+ ## Download and run
36
+
37
+ ```bash
38
+ hf download startlux-models/StartLux-Decision-27B-Q8_0-GGUF --local-dir StartLux-Decision-27B-Q8_0-GGUF
39
+ cd StartLux-Decision-27B-Q8_0-GGUF
40
+ pip install -r requirements.txt # transformers and torch; a CPU build of torch is enough
41
+
42
+ # llama.cpp serves the weights; the decision server puts the prompt format, readout and calibration on top
43
+ llama-server -m StartLux-Decision-27B-Q8_0.gguf -ngl 99 -c 16384 --parallel 4 --port 8081 &
44
+ python -m startlux_decision.gguf_server --model-dir . --llama http://127.0.0.1:8081 --port 8090
45
+ ```
46
+
47
+ `-ngl 99` puts every layer on the GPU (CUDA or Metal); leave it out on a CPU-only machine. llama.cpp has to be recent
48
+ enough for this model (build b10454 or newer).
49
+
50
+ Requests and responses use the TypeSafe `/v1/systemone` format:
51
+
52
+ ```bash
53
+ curl -s localhost:8090/v1/systemone -H 'Content-Type: application/json' -d '{
54
+ "state": {"ticket": "I was charged twice for order #4411 and the app still shows it as unpaid."},
55
+ "questions": {
56
+ "team": {"type": "choice", "instructions": "Which team should handle this ticket?",
57
+ "criteria": {"billing": "Payments, refunds and invoices",
58
+ "shipping": "Delivery and tracking",
59
+ "technical": "App, login and account problems"}},
60
+ "urgent": {"type": "noul", "instructions": "Should this ticket be answered today?"}
61
+ }
62
+ }'
63
+ ```
64
+
65
+ Plain chat with a GGUF file does not give these decisions: the option-letter readout and the per-type temperatures live
66
+ in `gguf_server.py`, not in the weights. The file holds the text decoder only.
67
+
68
+ ## License
69
+
70
+ The model weights are released under [CC BY-NC 4.0](https://creativecommons.org/licenses/by-nc/4.0/): free for
71
+ research and other non-commercial use, with attribution. Commercial use requires a separate license from
72
+ StartLux Labs; contact [contact@startlux.com](mailto:contact@startlux.com). The inference code in
73
+ `startlux_decision/` is Apache-2.0. See [LICENSE](https://huggingface.co/startlux-models/StartLux-Decision-27B-Q8_0-GGUF/blob/main/LICENSE) and [NOTICE](https://huggingface.co/startlux-models/StartLux-Decision-27B-Q8_0-GGUF/blob/main/NOTICE).
StartLux-Decision-27B-Q8_0.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5868a8a2df103c63353adfbda6f57fec5b6393c9b8a1d97392855a0335001088
3
+ size 28595763648
chat_template.jinja ADDED
@@ -0,0 +1,170 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- set reasoning_instructions = '' %}
46
+ {%- if enable_thinking is undefined or enable_thinking is true %}
47
+ {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}
48
+ {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}
49
+ {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}
50
+ {%- endif %}
51
+ {%- if resolved_reasoning_effort == 'xhigh' %}
52
+ {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}
53
+ {%- elif resolved_reasoning_effort == 'low' %}
54
+ {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}
55
+ {%- endif %}
56
+ {%- endif %}
57
+ {%- if tools and tools is iterable and tools is not mapping %}
58
+ {{- '<|im_start|>system\n' }}
59
+ {%- if reasoning_instructions %}
60
+ {{- reasoning_instructions + '\n\n' }}
61
+ {%- endif %}
62
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
63
+ {%- for tool in tools %}
64
+ {{- "\n" }}
65
+ {{- tool | tojson }}
66
+ {%- endfor %}
67
+ {{- "\n</tools>" }}
68
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
69
+ {%- if messages[0].role == 'system' %}
70
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
71
+ {%- if content %}
72
+ {{- '\n\n' + content }}
73
+ {%- endif %}
74
+ {%- endif %}
75
+ {{- '<|im_end|>\n' }}
76
+ {%- else %}
77
+ {%- if messages[0].role == 'system' %}
78
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
79
+ {%- if content %}
80
+ {{- '<|im_start|>system\n' + (reasoning_instructions + '\n\n' if reasoning_instructions else '') + content + '<|im_end|>\n' }}
81
+ {%- elif reasoning_instructions %}
82
+ {{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
83
+ {%- endif %}
84
+ {%- elif reasoning_instructions %}
85
+ {{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
86
+ {%- endif %}
87
+ {%- endif %}
88
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
89
+ {%- for message in messages[::-1] %}
90
+ {%- set index = (messages|length - 1) - loop.index0 %}
91
+ {%- if ns.multi_step_tool and message.role == "user" %}
92
+ {%- set content = render_content(message.content, false)|trim %}
93
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
94
+ {%- set ns.multi_step_tool = false %}
95
+ {%- set ns.last_query_index = index %}
96
+ {%- endif %}
97
+ {%- endif %}
98
+ {%- endfor %}
99
+ {%- if ns.multi_step_tool %}
100
+ {{- raise_exception('No user query found in messages.') }}
101
+ {%- endif %}
102
+ {%- for message in messages %}
103
+ {%- set content = render_content(message.content, true)|trim %}
104
+ {%- if message.role == "system" %}
105
+ {%- if not loop.first %}
106
+ {{- raise_exception('System message must be at the beginning.') }}
107
+ {%- endif %}
108
+ {%- elif message.role == "user" %}
109
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
110
+ {%- elif message.role == "assistant" %}
111
+ {%- set reasoning_content = '' %}
112
+ {%- if message.reasoning_content is string %}
113
+ {%- set reasoning_content = message.reasoning_content %}
114
+ {%- endif %}
115
+ {%- set reasoning_content = reasoning_content|trim %}
116
+ {%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}
117
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
118
+ {%- else %}
119
+ {{- '<|im_start|>' + message.role + '\n' + content }}
120
+ {%- endif %}
121
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
122
+ {%- for tool_call in message.tool_calls %}
123
+ {%- if tool_call.function is defined %}
124
+ {%- set tool_call = tool_call.function %}
125
+ {%- endif %}
126
+ {%- if loop.first %}
127
+ {%- if content|trim %}
128
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
129
+ {%- else %}
130
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
131
+ {%- endif %}
132
+ {%- else %}
133
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
134
+ {%- endif %}
135
+ {%- if tool_call.arguments is defined and tool_call.arguments != '' %}
136
+ {%- for args_name, args_value in tool_call.arguments|items %}
137
+ {{- '<parameter=' + args_name + '>\n' }}
138
+ {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}
139
+ {{- args_value }}
140
+ {{- '\n</parameter>\n' }}
141
+ {%- endfor %}
142
+ {%- endif %}
143
+ {{- '</function>\n</tool_call>' }}
144
+ {%- endfor %}
145
+ {%- endif %}
146
+ {{- '<|im_end|>\n' }}
147
+ {%- elif message.role == "tool" %}
148
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
149
+ {{- '<|im_start|>user' }}
150
+ {%- endif %}
151
+ {{- '\n<tool_response>\n' }}
152
+ {{- content }}
153
+ {{- '\n</tool_response>' }}
154
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
155
+ {{- '<|im_end|>\n' }}
156
+ {%- elif loop.last %}
157
+ {{- '<|im_end|>\n' }}
158
+ {%- endif %}
159
+ {%- else %}
160
+ {{- raise_exception('Unexpected message role.') }}
161
+ {%- endif %}
162
+ {%- endfor %}
163
+ {%- if add_generation_prompt %}
164
+ {{- '<|im_start|>assistant\n' }}
165
+ {%- if enable_thinking is defined and enable_thinking is false %}
166
+ {{- '<think>\n\n</think>\n\n' }}
167
+ {%- else %}
168
+ {{- '<think>\n' }}
169
+ {%- endif %}
170
+ {%- endif %}
config.json ADDED
@@ -0,0 +1,140 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5ForConditionalGeneration"
4
+ ],
5
+ "image_token_id": 248056,
6
+ "language_model_only": false,
7
+ "model_type": "qwen3_5",
8
+ "text_config": {
9
+ "attention_bias": false,
10
+ "attention_dropout": 0.0,
11
+ "attn_output_gate": true,
12
+ "bos_token_id": 248044,
13
+ "dtype": "bfloat16",
14
+ "eos_token_id": 248044,
15
+ "full_attention_interval": 4,
16
+ "head_dim": 256,
17
+ "hidden_act": "silu",
18
+ "hidden_size": 5120,
19
+ "initializer_range": 0.02,
20
+ "intermediate_size": 17408,
21
+ "layer_types": [
22
+ "linear_attention",
23
+ "linear_attention",
24
+ "linear_attention",
25
+ "full_attention",
26
+ "linear_attention",
27
+ "linear_attention",
28
+ "linear_attention",
29
+ "full_attention",
30
+ "linear_attention",
31
+ "linear_attention",
32
+ "linear_attention",
33
+ "full_attention",
34
+ "linear_attention",
35
+ "linear_attention",
36
+ "linear_attention",
37
+ "full_attention",
38
+ "linear_attention",
39
+ "linear_attention",
40
+ "linear_attention",
41
+ "full_attention",
42
+ "linear_attention",
43
+ "linear_attention",
44
+ "linear_attention",
45
+ "full_attention",
46
+ "linear_attention",
47
+ "linear_attention",
48
+ "linear_attention",
49
+ "full_attention",
50
+ "linear_attention",
51
+ "linear_attention",
52
+ "linear_attention",
53
+ "full_attention",
54
+ "linear_attention",
55
+ "linear_attention",
56
+ "linear_attention",
57
+ "full_attention",
58
+ "linear_attention",
59
+ "linear_attention",
60
+ "linear_attention",
61
+ "full_attention",
62
+ "linear_attention",
63
+ "linear_attention",
64
+ "linear_attention",
65
+ "full_attention",
66
+ "linear_attention",
67
+ "linear_attention",
68
+ "linear_attention",
69
+ "full_attention",
70
+ "linear_attention",
71
+ "linear_attention",
72
+ "linear_attention",
73
+ "full_attention",
74
+ "linear_attention",
75
+ "linear_attention",
76
+ "linear_attention",
77
+ "full_attention",
78
+ "linear_attention",
79
+ "linear_attention",
80
+ "linear_attention",
81
+ "full_attention",
82
+ "linear_attention",
83
+ "linear_attention",
84
+ "linear_attention",
85
+ "full_attention"
86
+ ],
87
+ "linear_conv_kernel_dim": 4,
88
+ "linear_key_head_dim": 128,
89
+ "linear_num_key_heads": 16,
90
+ "linear_num_value_heads": 48,
91
+ "linear_value_head_dim": 128,
92
+ "mamba_ssm_dtype": "float32",
93
+ "max_position_embeddings": 262144,
94
+ "model_type": "qwen3_5_text",
95
+ "mtp_num_hidden_layers": 1,
96
+ "mtp_use_dedicated_embeddings": false,
97
+ "num_attention_heads": 24,
98
+ "num_hidden_layers": 64,
99
+ "num_key_value_heads": 4,
100
+ "output_gate_type": "swish",
101
+ "pad_token_id": null,
102
+ "partial_rotary_factor": 0.25,
103
+ "rms_norm_eps": 1e-06,
104
+ "rope_parameters": {
105
+ "mrope_interleaved": true,
106
+ "mrope_section": [
107
+ 11,
108
+ 11,
109
+ 10
110
+ ],
111
+ "partial_rotary_factor": 0.25,
112
+ "rope_theta": 10000000,
113
+ "rope_type": "default"
114
+ },
115
+ "tie_word_embeddings": false,
116
+ "use_cache": true,
117
+ "vocab_size": 248320
118
+ },
119
+ "tie_word_embeddings": false,
120
+ "transformers_version": "5.8.0.dev0",
121
+ "video_token_id": 248057,
122
+ "vision_config": {
123
+ "deepstack_visual_indexes": [],
124
+ "depth": 27,
125
+ "hidden_act": "gelu_pytorch_tanh",
126
+ "hidden_size": 1152,
127
+ "in_channels": 3,
128
+ "initializer_range": 0.02,
129
+ "intermediate_size": 4304,
130
+ "model_type": "qwen3_5",
131
+ "num_heads": 16,
132
+ "num_position_embeddings": 2304,
133
+ "out_hidden_size": 5120,
134
+ "patch_size": 16,
135
+ "spatial_merge_size": 2,
136
+ "temporal_patch_size": 2
137
+ },
138
+ "vision_end_token_id": 248054,
139
+ "vision_start_token_id": 248053
140
+ }
decision_config.json ADDED
@@ -0,0 +1,42 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "StartLux-Decision-v1",
3
+ "letter_token_ids": [
4
+ 32,
5
+ 33,
6
+ 34,
7
+ 35,
8
+ 36,
9
+ 37,
10
+ 38,
11
+ 39,
12
+ 40,
13
+ 41,
14
+ 42,
15
+ 43,
16
+ 44,
17
+ 45,
18
+ 46,
19
+ 47,
20
+ 48,
21
+ 49,
22
+ 50,
23
+ 51,
24
+ 52,
25
+ 53,
26
+ 54,
27
+ 55,
28
+ 56,
29
+ 57
30
+ ],
31
+ "temperature_by_type": {
32
+ "noul": 1.421,
33
+ "choice": 1.2098,
34
+ "score": 1.2294
35
+ },
36
+ "max_options_per_pass": 26,
37
+ "wide_choice": {
38
+ "group": 25,
39
+ "keep": 3,
40
+ "residual": 0.001
41
+ }
42
+ }
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
requirements.txt ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ transformers
2
+ torch
startlux_decision/__init__.py ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ """StartLux-Decision: typed decisions (choice, noul, score) with calibrated probabilities from one forward pass."""
2
+ from .model import StartLuxDecision
3
+
4
+ __all__ = ["StartLuxDecision"]
startlux_decision/check.py ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Check that the fast kernels will be used for a model, before starting a server on it.
2
+
3
+ python -m startlux_decision.check /path/to/StartLux-Decision-4B
4
+
5
+ Exits with status 1 when flash-linear-attention or causal-conv1d is missing or not importable; transformers would then
6
+ fall back to a plain torch path that is more than ten times slower.
7
+ """
8
+ import sys
9
+
10
+ from .model import fast_kernels_active
11
+
12
+
13
+ def main():
14
+ if len(sys.argv) != 2:
15
+ sys.exit(__doc__.strip())
16
+ ok = fast_kernels_active(sys.argv[1])
17
+ print("fast kernels: " + ("active" if ok else "NOT active, pip install flash-linear-attention causal-conv1d"))
18
+ sys.exit(0 if ok else 1)
19
+
20
+
21
+ if __name__ == "__main__":
22
+ main()
startlux_decision/gguf_server.py ADDED
@@ -0,0 +1,108 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """A /v1/systemone server for a StartLux-Decision GGUF file run by llama.cpp.
2
+
3
+ llama-server -m StartLux-Decision-4B-Q8_0.gguf -ngl 99 -c 16384 --parallel 4 --port 8081
4
+ python -m startlux_decision.gguf_server --model-dir StartLux-Decision-4B-GGUF --llama http://127.0.0.1:8081 --port 8090
5
+
6
+ It is StartLuxDecision with only the forward pass replaced by a llama-server call: prompt rendering, the option-letter
7
+ readout, the per-type temperatures and wide choices are the package's own code, so the answers match the original
8
+ model question by question up to the numerics of the GGUF file. MODEL_DIR holds the tokenizer files, config.json and
9
+ decision_config.json (the GGUF repositories ship them next to the .gguf files). llama-server returns the
10
+ log-probabilities of the whole vocabulary at the answer position; dividing them by the temperature and normalising over
11
+ the listed options is the same as doing it on the logits."""
12
+ import argparse
13
+ import json
14
+ import os
15
+ import urllib.request
16
+ from concurrent.futures import ThreadPoolExecutor
17
+ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
18
+
19
+ import torch
20
+
21
+ from . import jevfmt as J
22
+ from .model import StartLuxDecision
23
+
24
+
25
+ class GGUFDecision(StartLuxDecision):
26
+ def __init__(self, path, llama, workers): # no torch model: llama-server runs the weights
27
+ from transformers import AutoTokenizer
28
+ cfg = json.load(open(os.path.join(path, "decision_config.json")))
29
+ self.tok = AutoTokenizer.from_pretrained(path)
30
+ self.letters = J.check_tokenizer(self.tok)
31
+ if self.letters != cfg["letter_token_ids"]:
32
+ raise ValueError("tokenizer letter ids differ from decision_config.json")
33
+ self.temperature = {k: float(v) for k, v in cfg["temperature_by_type"].items()}
34
+ wide = cfg.get("wide_choice", {})
35
+ self.group, self.keep, self.residual = wide.get("group", 25), wide.get("keep", 3), wide.get("residual", 1e-3)
36
+ self.max_length, self.graphs = 65536, {}
37
+ mc = json.load(open(os.path.join(path, "config.json")))
38
+ self.vocab = int(mc.get("text_config", mc).get("vocab_size", 262144))
39
+ self.url = llama.rstrip("/") + "/completion"
40
+ self.pool = ThreadPoolExecutor(workers)
41
+
42
+ def _letter_logprobs(self, ids, count):
43
+ """Next-token log-probabilities of the option letters at the last prompt position (full-vocabulary softmax;
44
+ dividing by the temperature and normalising over the options afterwards equals doing it on the logits)."""
45
+ targets = self.letters[:count]
46
+ for n in (256, self.vocab):
47
+ body = json.dumps({"prompt": ids, "n_predict": 1, "temperature": -1, "n_probs": n, "cache_prompt": False}).encode()
48
+ with urllib.request.urlopen(urllib.request.Request(self.url, body, {"Content-Type": "application/json"}), timeout=900) as r:
49
+ out = json.loads(r.read())
50
+ rows = out.get("completion_probabilities", out.get("probs"))
51
+ got = {t["id"]: t["logprob"] for t in rows[0]["top_logprobs"]}
52
+ if all(t in got for t in targets):
53
+ return [got[t] for t in targets]
54
+ raise RuntimeError("option letters missing from llama-server's log-probabilities")
55
+
56
+ def _logits(self, rows):
57
+ jobs = []
58
+ for r in rows:
59
+ order = [o["id"] for o in r["options"]]
60
+ ids, _ = J.render_ids(r, self.tok, order, max_length=self.max_length)
61
+ jobs.append((ids, len(order)))
62
+ outs = list(self.pool.map(lambda j: self._letter_logprobs(*j), jobs))
63
+ return [torch.tensor(o, dtype=torch.float32) for o in outs], sum(len(i) for i, _ in jobs)
64
+
65
+
66
+ def main():
67
+ ap = argparse.ArgumentParser()
68
+ ap.add_argument("--model-dir", required=True)
69
+ ap.add_argument("--llama", required=True)
70
+ ap.add_argument("--port", type=int, required=True)
71
+ ap.add_argument("--workers", type=int, default=16)
72
+ ap.add_argument("--name", help="model name reported in responses (default: the GGUF directory name)")
73
+ a = ap.parse_args()
74
+ engine = GGUFDecision(a.model_dir, a.llama, a.workers)
75
+ a.name = a.name or os.path.basename(os.path.abspath(a.model_dir))
76
+
77
+ class Handler(BaseHTTPRequestHandler):
78
+ def _send(self, code, obj):
79
+ data = json.dumps(obj).encode()
80
+ self.send_response(code)
81
+ self.send_header("Content-Type", "application/json")
82
+ self.send_header("Content-Length", str(len(data)))
83
+ self.end_headers()
84
+ self.wfile.write(data)
85
+
86
+ def do_GET(self):
87
+ self._send(200, {"status": "ok", "model": a.name})
88
+
89
+ def do_POST(self):
90
+ try:
91
+ req = json.loads(self.rfile.read(int(self.headers.get("Content-Length", 0))))
92
+ answers, usage = engine.decide(req["state"], req["questions"])
93
+ self._send(200, {"answers": answers, "usage": usage, "model": a.name})
94
+ except Exception as e: # surfaced to the client; suites.py stops on it
95
+ self._send(500, {"error": repr(e)})
96
+
97
+ def log_message(self, *args):
98
+ pass
99
+
100
+ class Server(ThreadingHTTPServer):
101
+ request_queue_size = 1024 # the default backlog of 5 resets connections under load
102
+ daemon_threads = True
103
+
104
+ Server(("127.0.0.1", a.port), Handler).serve_forever()
105
+
106
+
107
+ if __name__ == "__main__":
108
+ main()
startlux_decision/jevfmt.py ADDED
@@ -0,0 +1,188 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Prompt rendering for StartLux-Decision.
2
+
3
+ A question about a state is turned into one chat prompt: a fixed system line, then an Evidence / Question / Options
4
+ block with lettered options, and the thinking-off assistant prefix. The answer is read from the next-token logits of
5
+ the option letters at the last prompt position, so nothing is generated.
6
+
7
+ Rules worth knowing when you build requests by hand: an option without a description is shown by its id alone, yes/no
8
+ questions are shown as yes / no, choice ids that are bare letters or numbers are hidden (a line such as "A) C: Paris"
9
+ would make the answer letter ambiguous). StartLux-Decision shows options in the order the request lists them; score levels
10
+ are always lowest first.
11
+ """
12
+ import hashlib
13
+ import json
14
+ import re
15
+ import string
16
+
17
+ VERSION = "jev_render_v3"
18
+ SYSTEM = "Apply the criterion to the evidence. Choose exactly one listed option. Answer with its letter only."
19
+ LETTERS = string.ascii_uppercase
20
+ TYPES = ("choice", "score", "noul")
21
+ MAX_OPTIONS = 26
22
+ MAX_LEVELS = 10
23
+ _BARE = re.compile(r"^(?:\(?[A-Za-z][\).]?|\(?\d{1,3}[\).]?|option[_ ]?\d{1,3}|opt[_ ]?\d{1,3})$", re.I)
24
+ ANNOTATE_MIN = 8
25
+
26
+
27
+ def short_hash(text, n=16):
28
+ return hashlib.sha1(text.encode("utf-8", "surrogatepass")).hexdigest()[:n]
29
+
30
+
31
+ def annotate_indices(x, min_len=ANNOTATE_MIN):
32
+ """Long arrays carry their element index so that items can be referred to by position."""
33
+ if isinstance(x, list):
34
+ if len(x) >= min_len:
35
+ return [({"_index": i, **annotate_indices(v, min_len)} if isinstance(v, dict)
36
+ else {"_index": i, "value": annotate_indices(v, min_len)}) for i, v in enumerate(x)]
37
+ return [annotate_indices(v, min_len) for v in x]
38
+ if isinstance(x, dict):
39
+ return {k: annotate_indices(v, min_len) for k, v in x.items()}
40
+ return x
41
+
42
+
43
+ def state_text(state):
44
+ if isinstance(state, str):
45
+ return state
46
+ if state is None or state == {} or state == []:
47
+ return ""
48
+ return json.dumps(annotate_indices(state), ensure_ascii=False)
49
+
50
+
51
+ def value_text(v):
52
+ if v is None:
53
+ return None
54
+ s = v if isinstance(v, str) else json.dumps(v, ensure_ascii=False)
55
+ s = s.strip()
56
+ return s or None
57
+
58
+
59
+ def _norm(s):
60
+ return re.sub(r"[\s_\-]+", " ", str(s)).strip().lower()
61
+
62
+
63
+ def validate(row):
64
+ t = row.get("type")
65
+ if t not in TYPES:
66
+ raise ValueError(f"unknown type {t!r}")
67
+ if not isinstance(row.get("state"), str):
68
+ raise ValueError("state must be a string")
69
+ if not isinstance(row.get("instructions"), str) or not row["instructions"].strip():
70
+ raise ValueError("instructions required")
71
+ opts = row.get("options")
72
+ if not isinstance(opts, list) or not 2 <= len(opts) <= MAX_OPTIONS:
73
+ raise ValueError(f"2..{MAX_OPTIONS} options required, got {len(opts) if isinstance(opts, list) else opts!r}")
74
+ ids = []
75
+ for o in opts:
76
+ i = o.get("id")
77
+ if not isinstance(i, str) or not i.strip() or "\n" in i or "\r" in i or len(i) > 200:
78
+ raise ValueError(f"bad option id {i!r}")
79
+ c = o.get("criterion")
80
+ if c is not None and not isinstance(c, str):
81
+ raise ValueError("criterion must be a string or None")
82
+ ids.append(i)
83
+ if len(set(ids)) != len(ids):
84
+ raise ValueError("duplicate option id")
85
+ if t == "noul" and set(ids) != {"true", "false"}:
86
+ raise ValueError("noul ids must be true/false")
87
+ if t == "score" and not 2 <= len(ids) <= MAX_LEVELS:
88
+ raise ValueError(f"score needs 2..{MAX_LEVELS} levels")
89
+ return ids
90
+
91
+
92
+ def option_lines(row, order):
93
+ crit = {o["id"]: o.get("criterion") for o in row["options"]}
94
+ t = row["type"]
95
+ hide = t == "choice" and all(_BARE.match(i) for i in order) and all(crit[i] for i in order)
96
+ lines = []
97
+ for k, i in enumerate(order):
98
+ name = ("yes" if i == "true" else "no") if t == "noul" else i
99
+ c = crit[i]
100
+ if hide:
101
+ body = c
102
+ elif c is None or not c.strip() or _norm(c) == _norm(name):
103
+ body = name
104
+ else:
105
+ body = f"{name}: {c}"
106
+ lines.append(f"{LETTERS[k]}) {body}")
107
+ return lines
108
+
109
+
110
+ def messages(row, order=None):
111
+ validate(row)
112
+ order = [o["id"] for o in row["options"]] if order is None else list(order)
113
+ if sorted(order) != sorted(o["id"] for o in row["options"]):
114
+ raise ValueError("order must be a permutation of option ids")
115
+ state = row["state"] if row["state"].strip() else "(none)"
116
+ content = ("Evidence:\n" + state + "\n\nQuestion: " + row["instructions"].strip() + "\nOptions:\n"
117
+ + "\n".join(option_lines(row, order)))
118
+ return [{"role": "system", "content": SYSTEM}, {"role": "user", "content": content}], order
119
+
120
+
121
+ THINK_OFF_SUFFIX = "<think>\n\n</think>\n\n"
122
+
123
+
124
+ def check_tokenizer(tokenizer):
125
+ """Letter ids and the native thinking-off prefix; returns the 26 letter token ids."""
126
+ probe, _ = messages({"type": "noul", "state": "s", "instructions": "q",
127
+ "options": [{"id": "true", "criterion": None}, {"id": "false", "criterion": None}]})
128
+ text = tokenizer.apply_chat_template(probe, tokenize=False, add_generation_prompt=True, enable_thinking=False)
129
+ if not text.endswith(THINK_OFF_SUFFIX):
130
+ raise ValueError("tokenizer chat template lacks the thinking-off assistant prefix")
131
+ ids = tokenizer.encode(text, add_special_tokens=False)
132
+ letters = []
133
+ for L in LETTERS:
134
+ t = tokenizer.encode(L, add_special_tokens=False)
135
+ if len(t) != 1 or tokenizer.encode(text + L, add_special_tokens=False) != ids + t:
136
+ raise ValueError("letter is not a separate single token after the prefix: " + L)
137
+ letters.append(t[0])
138
+ if len(set(letters)) != 26:
139
+ raise ValueError("letter ids not distinct")
140
+ return letters
141
+
142
+
143
+ def render_ids(row, tokenizer, order=None, max_length=12288):
144
+ """-> (input_ids, order). The answer letter is read at position len(input_ids) - 1."""
145
+ msgs, order = messages(row, order)
146
+ text = tokenizer.apply_chat_template(msgs, tokenize=False, add_generation_prompt=True, enable_thinking=False)
147
+ ids = tokenizer.encode(text, add_special_tokens=False)
148
+ if len(ids) > max_length:
149
+ raise ValueError(f"length {len(ids)} > {max_length}")
150
+ return ids, order
151
+
152
+
153
+ # ---------------------------------------------------------------- /v1/systemone requests
154
+
155
+ def score_keys(crit):
156
+ """Level order of a legend-form Score: numeric keys ascending, otherwise as given."""
157
+ try:
158
+ return sorted(crit, key=lambda k: float(k))
159
+ except (TypeError, ValueError):
160
+ return list(crit)
161
+
162
+
163
+ def from_systemone(state, spec, qid="q"):
164
+ """One /v1/systemone question spec -> the record that gets rendered. Score level ids are "0".."n-1"."""
165
+ t = spec.get("type", "choice")
166
+ t = "noul" if t == "bool" else t
167
+ crit = spec.get("criteria", spec.get("options"))
168
+ ins = value_text(spec.get("instructions", spec.get("question"))) or ""
169
+ if t == "choice":
170
+ if isinstance(crit, (list, tuple)):
171
+ crit = {str(c): None for c in crit}
172
+ opts = [{"id": str(k), "criterion": value_text(v)} for k, v in crit.items()]
173
+ elif t == "score":
174
+ if isinstance(crit, dict):
175
+ crit = [crit[k] for k in score_keys(crit)]
176
+ opts = [{"id": str(i), "criterion": value_text(c)} for i, c in enumerate(crit)]
177
+ elif t == "noul":
178
+ c = crit if isinstance(crit, dict) else {}
179
+ tr, fa = c.get("true", c.get(True)), c.get("false", c.get(False))
180
+ opts = [{"id": "true", "criterion": value_text(tr)}, {"id": "false", "criterion": value_text(fa)}]
181
+ if not ins:
182
+ ins = "Which answer fits the evidence?"
183
+ else:
184
+ raise ValueError(f"unknown question type {t!r}")
185
+ row = {"id": qid, "type": t, "state": state_text(state), "instructions": ins, "options": opts}
186
+ if t == "score":
187
+ row["ordered"] = True
188
+ return row
startlux_decision/model.py ADDED
@@ -0,0 +1,293 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """StartLux-Decision inference.
2
+
3
+ Each question is rendered as one prompt (startlux_decision/jevfmt.py), the model runs one forward pass, and the
4
+ answer is read from the next-token logits of the option letters at the last prompt position, divided by the
5
+ temperature of the question type and normalised over the listed options. Nothing is generated.
6
+
7
+ All questions of one request run in one forward pass, one row per question. On CUDA the pass is a recorded graph:
8
+ one per padded input length (128 ... 4096 tokens) for a single question, and one per (question count, length) for two
9
+ to four questions of up to 1024 tokens each. Replaying a graph removes the per-layer launch overhead (the approach of
10
+ the JevK5 runtime, Apache-2.0). Inputs are right-padded; every layer is causal and rows never mix, so padding after the
11
+ last prompt token never reaches the position that is read. Other requests run eagerly as one padded batch (inputs
12
+ padded to one of PAD_LENGTHS); decide_batch() batches questions across many requests.
13
+ """
14
+ import importlib
15
+ import json
16
+ import os
17
+ import warnings
18
+
19
+ import torch
20
+
21
+ from . import jevfmt as J
22
+
23
+ __all__ = ["StartLuxDecision", "load_model", "fast_kernels_active"]
24
+
25
+ GRAPH_LENGTHS = (128, 192, 256, 320, 384, 512, 640, 768, 1024, 1536, 2048, 3072, 4096)
26
+ GRAPH_ROWS = (1, 2, 3, 4) # questions per request replayed as one graph
27
+ MULTI_ROW_MAX_LENGTH = 1024 # longest prompt for the multi-question graphs (rows x length <= 4096 tokens)
28
+ # Batched inputs are padded to one of these lengths. The linear-attention kernels are compiled once per sequence length
29
+ # (about a second each), so padding to exact lengths would recompile for almost every batch. Above 256 tokens a step
30
+ # adds at most 25% padding.
31
+ PAD_LENGTHS = (128, 192, 256, 320, 384, 448, 512, 640, 768, 896, 1024, 1280, 1536, 1792, 2048, 2560, 3072, 3584, 4096,
32
+ 5120, 6144, 7168, 8192, 10240, 12288, 14336, 16384, 20480, 24576, 28672, 32768, 40960, 49152, 57344, 65536)
33
+
34
+
35
+ def padded_length(n):
36
+ """The length a batch whose longest input has n tokens is padded to."""
37
+ for length in PAD_LENGTHS:
38
+ if n <= length:
39
+ return length
40
+ return -(-n // 8192) * 8192
41
+
42
+
43
+ def fast_kernels_active(path):
44
+ """True when transformers will use the fla / causal-conv1d kernels for the linear-attention layers of the model
45
+ in `path`."""
46
+ from transformers import AutoConfig
47
+
48
+ kind = AutoConfig.from_pretrained(path).model_type
49
+ try:
50
+ modeling = importlib.import_module(f"transformers.models.{kind}.modeling_{kind}")
51
+ except ImportError:
52
+ return False
53
+ return bool(getattr(modeling, "is_fast_path_available", False))
54
+
55
+
56
+ def load_model(path, device):
57
+ """The checkpoint's own transformers class in bf16 on `device`, and its text decoder (the stack without the
58
+ output head; checkpoints that also carry other towers keep the text decoder under .language_model)."""
59
+ import transformers
60
+
61
+ arch = transformers.AutoConfig.from_pretrained(path).architectures[0]
62
+ model = getattr(transformers, arch).from_pretrained(path, dtype=torch.bfloat16, device_map={"": device})
63
+ return model, getattr(model.model, "language_model", model.model)
64
+
65
+
66
+ class StartLuxDecision:
67
+ """decide(state, questions) -> (answers, usage), answers in the TypeSafe /v1/systemone format."""
68
+
69
+ def __init__(self, path, device=None, max_length=65536, max_batch_tokens=65536, graphs=True):
70
+ from transformers import AutoTokenizer
71
+
72
+ if not os.path.isdir(path):
73
+ raise FileNotFoundError(f"{path}: expected a local StartLux-Decision directory (weights are shared separately)")
74
+ cfg = json.load(open(os.path.join(path, "decision_config.json")))
75
+ self.tok = AutoTokenizer.from_pretrained(path)
76
+ self.letters = J.check_tokenizer(self.tok)
77
+ if self.letters != cfg["letter_token_ids"]:
78
+ raise ValueError("tokenizer letter ids differ from decision_config.json")
79
+ self.temperature = {k: float(v) for k, v in cfg["temperature_by_type"].items()}
80
+ wide = cfg.get("wide_choice", {})
81
+ self.group, self.keep, self.residual = wide.get("group", 25), wide.get("keep", 3), wide.get("residual", 1e-3)
82
+ self.device = torch.device(device or ("cuda" if torch.cuda.is_available() else "cpu"))
83
+ self.fast_kernels = fast_kernels_active(path)
84
+ if self.device.type == "cuda" and not self.fast_kernels:
85
+ msg = ("flash-linear-attention and causal-conv1d are not active, so transformers would run the plain torch "
86
+ "path of the linear-attention layers (>10x slower). pip install flash-linear-attention causal-conv1d, "
87
+ "or set STARTLUX_ALLOW_SLOW=1 to run anyway.")
88
+ if os.environ.get("STARTLUX_ALLOW_SLOW") != "1":
89
+ raise RuntimeError(msg)
90
+ warnings.warn(msg)
91
+ model, body = load_model(path, self.device)
92
+ self.body = body.eval()
93
+ head = model.get_output_embeddings().weight
94
+ self.letter_rows = head.index_select(0, torch.tensor(self.letters, device=head.device)).float()
95
+ del model # only the decoder and the letter rows of the output head are used
96
+ self.pad = self.tok.pad_token_id
97
+ self.max_length, self.max_batch_tokens = int(max_length), int(max_batch_tokens)
98
+ self.graphs = {}
99
+ if graphs and self.device.type == "cuda" and os.environ.get("STARTLUX_GRAPHS", "1") != "0":
100
+ self._capture()
101
+
102
+ # ---- CUDA-graph path (the questions of one request as rows, right-padded, no attention mask)
103
+ def _slot_logits(self, ids, last):
104
+ with torch.autocast(self.device.type, dtype=torch.bfloat16):
105
+ hidden = self.body(input_ids=ids, use_cache=False, return_dict=True).last_hidden_state
106
+ h = hidden[torch.arange(ids.shape[0], device=ids.device), last].float()
107
+ return h @ self.letter_rows.T
108
+
109
+ @torch.inference_mode()
110
+ def _capture(self):
111
+ shapes = [(b, n) for b in GRAPH_ROWS for n in GRAPH_LENGTHS if b == 1 or n <= MULTI_ROW_MAX_LENGTH]
112
+ pool = None # one memory pool for every graph, sized by the largest
113
+ for b, n in sorted(shapes, key=lambda s: (-s[0] * s[1], -s[1])):
114
+ ids = torch.full((b, n), self.pad, dtype=torch.long, device=self.device)
115
+ last = torch.full((b,), n - 1, dtype=torch.long, device=self.device)
116
+ stream = torch.cuda.Stream(device=self.device)
117
+ stream.wait_stream(torch.cuda.current_stream(self.device))
118
+ with torch.cuda.stream(stream):
119
+ for _ in range(3): # autotune and warm every kernel outside the capture
120
+ self._slot_logits(ids, last)
121
+ torch.cuda.current_stream(self.device).wait_stream(stream)
122
+ graph = torch.cuda.CUDAGraph()
123
+ with torch.cuda.graph(graph, pool=pool):
124
+ out = self._slot_logits(ids, last)
125
+ pool = graph.pool()
126
+ self.graphs[(b, n)] = (graph, ids, last, out)
127
+
128
+ def _graph_length(self, enc):
129
+ """The recorded length for these rows, or None when no graph fits."""
130
+ longest = max(len(t) for _, t, _ in enc)
131
+ fits = [n for b, n in self.graphs if b == len(enc) and n >= longest]
132
+ return min(fits) if fits else None
133
+
134
+ @torch.inference_mode()
135
+ def _graph_logits(self, enc, n):
136
+ graph, static_ids, last, out = self.graphs[(len(enc), n)]
137
+ ids = torch.full((len(enc), n), self.pad, dtype=torch.long)
138
+ for j, (_, t, _) in enumerate(enc):
139
+ ids[j, :len(t)] = torch.tensor(t)
140
+ static_ids.copy_(ids)
141
+ last.copy_(torch.tensor([len(t) - 1 for _, t, _ in enc]))
142
+ graph.replay()
143
+ res = out.cpu()
144
+ return [res[j, :c].clone() for j, (_, _, c) in enumerate(enc)]
145
+
146
+ # ---- readout
147
+ @torch.no_grad()
148
+ def _logits(self, rows):
149
+ """rows: rendered records (<= 26 options) -> (letter logits per row in option order, prompt tokens)."""
150
+ enc = []
151
+ for i, r in enumerate(rows):
152
+ order = [o["id"] for o in r["options"]]
153
+ ids, _ = J.render_ids(r, self.tok, order, max_length=self.max_length)
154
+ enc.append((i, ids, len(order)))
155
+ out, tokens = [None] * len(rows), 0
156
+ n = self._graph_length(enc) if self.graphs and enc else None
157
+ if n is not None: # every question of the request in one replay
158
+ for (i, t, _), z in zip(enc, self._graph_logits(enc, n)):
159
+ out[i] = z
160
+ tokens += len(t)
161
+ return out, tokens
162
+ enc.sort(key=lambda x: -len(x[1]))
163
+ start = 0
164
+ while start < len(enc):
165
+ longest = padded_length(len(enc[start][1]))
166
+ n = max(1, min(len(enc) - start, self.max_batch_tokens // longest))
167
+ batch = enc[start:start + n]
168
+ start += n
169
+ ids = torch.full((len(batch), longest), self.pad, dtype=torch.long)
170
+ attn = torch.zeros_like(ids)
171
+ for j, (_, t, _) in enumerate(batch):
172
+ ids[j, :len(t)] = torch.tensor(t)
173
+ attn[j, :len(t)] = 1
174
+ tokens += len(t)
175
+ ids, attn = ids.to(self.device), attn.to(self.device)
176
+ with torch.autocast(self.device.type, dtype=torch.bfloat16):
177
+ hidden = self.body(input_ids=ids, attention_mask=attn, use_cache=False, return_dict=True).last_hidden_state
178
+ pos = attn.sum(1) - 1
179
+ h = hidden[torch.arange(len(batch), device=self.device), pos].float()
180
+ logits = h @ self.letter_rows.T
181
+ for j, (i, _, c) in enumerate(batch):
182
+ out[i] = logits[j, :c].cpu()
183
+ return out, tokens
184
+
185
+ def _probs(self, rows):
186
+ logits, tokens = self._logits(rows)
187
+ return [torch.softmax(z / self.temperature.get(r["type"], 1.0), -1).tolist() for r, z in zip(rows, logits)], tokens
188
+
189
+ def self_test(self, state, questions):
190
+ """Largest |p_graph - p_eager| over the questions of one request (0.0 when no graphs were recorded)."""
191
+ if not self.graphs:
192
+ return 0.0
193
+ rows = [J.from_systemone(state, q) for q in questions.values()]
194
+ graphed, _ = self._probs(rows)
195
+ graphs, self.graphs = self.graphs, {}
196
+ try:
197
+ eager, _ = self._probs(rows)
198
+ finally:
199
+ self.graphs = graphs
200
+ return max(abs(a - b) for pg, pe in zip(graphed, eager) for a, b in zip(pg, pe))
201
+
202
+ def _wide(self, state, spec):
203
+ """Choice lists over 26 options: near-equal groups, the top `keep` of each group go to a final round."""
204
+ keys = list(spec["criteria"])
205
+ n_groups = -(-len(keys) // self.group)
206
+ size, extra = divmod(len(keys), n_groups)
207
+ groups, start = [], 0
208
+ for g in range(n_groups):
209
+ end = start + size + (1 if g < extra else 0)
210
+ groups.append(keys[start:end])
211
+ start = end
212
+ rows = [J.from_systemone(state, dict(spec, criteria={k: spec["criteria"][k] for k in g})) for g in groups]
213
+ first, tokens = self._probs(rows)
214
+ first_p = {k: p for g, ps in zip(groups, first) for k, p in zip(g, ps)}
215
+ finalists = [k for g, ps in zip(groups, first) for k, _ in sorted(zip(g, ps), key=lambda x: -x[1])[:self.keep]]
216
+ fin_spec = dict(spec, criteria={k: spec["criteria"][k] for k in finalists})
217
+ if len(finalists) > J.MAX_OPTIONS:
218
+ final_p, more = self._wide(state, fin_spec)
219
+ else:
220
+ ps, more = self._probs([J.from_systemone(state, fin_spec)])
221
+ final_p = dict(zip(finalists, ps[0]))
222
+ rest = [k for k in keys if k not in set(finalists)]
223
+ mass = sum(first_p[k] for k in rest) or 1.0
224
+ probs = {k: final_p[k] * (1 - self.residual) for k in finalists}
225
+ probs.update({k: self.residual * first_p[k] / mass for k in rest})
226
+ z = sum(probs.values())
227
+ return {k: v / z for k, v in probs.items()}, tokens + more
228
+
229
+ # ---- public API
230
+ @staticmethod
231
+ def _answer(row, p, question):
232
+ ids = [o["id"] for o in row["options"]]
233
+ if row["type"] == "noul":
234
+ return {"type": "noul", "noul": p[ids.index("true")]}
235
+ if row["type"] == "score":
236
+ levels = question.get("criteria") or []
237
+ levels = list(levels.values()) if isinstance(levels, dict) else list(levels)
238
+ return {"type": "score", "score": sum(i * v for i, v in enumerate(p)), "confidence": max(p),
239
+ "legend": {str(i): (levels[i] if i < len(levels) else str(i)) for i in range(len(p))},
240
+ "probabilities": {str(i): v for i, v in enumerate(p)}}
241
+ dist = dict(zip(ids, p))
242
+ best = max(dist, key=dist.get)
243
+ return {"type": "choice", "choice": best, "confidence": dist[best], "probabilities": dist}
244
+
245
+ def _split(self, state, questions):
246
+ """-> (answers decided without the model, [(key, rendered row)], tokens spent on wide lists)"""
247
+ answers, rows, tokens = {}, [], 0
248
+ for k, q in questions.items():
249
+ t = q.get("type", "choice")
250
+ crit = q.get("criteria") or {}
251
+ if t == "choice" and isinstance(crit, dict) and len(crit) == 1:
252
+ only = next(iter(crit))
253
+ answers[k] = {"type": "choice", "choice": only, "confidence": 1.0, "probabilities": {only: 1.0}}
254
+ elif t == "choice" and isinstance(crit, dict) and len(crit) > J.MAX_OPTIONS:
255
+ p, n = self._wide(state, q)
256
+ tokens += n
257
+ best = max(p, key=p.get)
258
+ answers[k] = {"type": "choice", "choice": best, "confidence": p[best], "probabilities": p}
259
+ else:
260
+ rows.append((k, J.from_systemone(state, q)))
261
+ return answers, rows, tokens
262
+
263
+ def decide(self, state, questions):
264
+ """One request -> (answers, usage), answers in the TypeSafe /v1/systemone format."""
265
+ answers, rows, tokens = self._split(state, questions)
266
+ if rows:
267
+ probs, n = self._probs([r for _, r in rows])
268
+ tokens += n
269
+ for (k, row), p in zip(rows, probs):
270
+ answers[k] = self._answer(row, p, questions[k])
271
+ return answers, {"input_tokens": tokens, "output_tokens": 0}
272
+
273
+ def decide_batch(self, requests):
274
+ """[(state, questions), ...] -> [answers, ...]. Every question of every request goes through one length-sorted
275
+ set of padded forward passes (up to max_batch_tokens each), which is much faster than calling decide() in a loop
276
+ when the requests are short. Answers are the same as decide() up to bf16 rounding."""
277
+ parts, flat = [], []
278
+ for state, questions in requests:
279
+ answers, rows, _ = self._split(state, questions)
280
+ parts.append((answers, rows, questions))
281
+ flat.extend(r for _, r in rows)
282
+ graphs, self.graphs = self.graphs, {} # batched eager path; graphs are sized for one request
283
+ try:
284
+ probs, _ = self._probs(flat) if flat else ([], 0)
285
+ finally:
286
+ self.graphs = graphs
287
+ out, i = [], 0
288
+ for answers, rows, questions in parts:
289
+ for k, row in rows:
290
+ answers[k] = self._answer(row, probs[i], questions[k])
291
+ i += 1
292
+ out.append(answers)
293
+ return out
startlux_decision/server.py ADDED
@@ -0,0 +1,96 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """TypeSafe-compatible decision server.
2
+
3
+ python -m startlux_decision.server --model /path/to/StartLux-Decision-4B --port 8090
4
+
5
+ POST /v1/systemone {"state": ..., "questions": {key: {"type", "instructions", "criteria"}}}
6
+ -> {"answers": {key: answer}, "usage": {"input_tokens", "output_tokens"}, "model": ...}
7
+ GET /health {"status", "model", "fast_kernels", "cuda_graphs"}
8
+ GET /v1/models
9
+ Requests are served one at a time on one GPU. At start-up one request runs through both the CUDA-graph path and the
10
+ eager path; if their probabilities differ by more than 0.02 the graphs are dropped.
11
+ """
12
+ import argparse
13
+ import json
14
+ import os
15
+ import threading
16
+ import time
17
+ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
18
+
19
+
20
+ def main():
21
+ ap = argparse.ArgumentParser()
22
+ ap.add_argument("--model", required=True, help="local StartLux-Decision directory")
23
+ ap.add_argument("--host", default="127.0.0.1")
24
+ ap.add_argument("--port", type=int, default=8090)
25
+ ap.add_argument("--device", help="cuda or cpu (default: cuda when available)")
26
+ ap.add_argument("--name", help="model name reported in responses (default: the directory name)")
27
+ a = ap.parse_args()
28
+ from .model import StartLuxDecision
29
+
30
+ engine = StartLuxDecision(a.model, device=a.device)
31
+ demo_state = {"ticket": "I was charged twice for order #4411 and the app still shows it as unpaid."}
32
+ demo_questions = {
33
+ "team": {"type": "choice", "instructions": "Which team should handle this ticket?",
34
+ "criteria": {"billing": "Payments, refunds and invoices", "shipping": "Delivery and tracking",
35
+ "technical": "App, login and account problems"}},
36
+ "urgent": {"type": "noul", "instructions": "Should this ticket be answered today?"},
37
+ "severity": {"type": "score", "instructions": "How severe is the impact?",
38
+ "criteria": ["cosmetic", "annoying", "blocks the customer"]},
39
+ }
40
+ engine.decide(demo_state, demo_questions) # warm-up
41
+ if engine.graphs:
42
+ diff = engine.self_test(demo_state, demo_questions)
43
+ print(f"graph self-test: max |p_graph - p_eager| = {diff:.2e}", flush=True)
44
+ if diff > 0.02:
45
+ print("graph readout disagrees with the eager path, graphs disabled", flush=True)
46
+ engine.graphs = {}
47
+ lock = threading.Lock()
48
+ name = a.name or os.path.basename(os.path.normpath(a.model))
49
+
50
+ class Handler(BaseHTTPRequestHandler):
51
+ def _send(self, code, obj):
52
+ data = json.dumps(obj).encode()
53
+ self.send_response(code)
54
+ self.send_header("Content-Type", "application/json")
55
+ self.send_header("Content-Length", str(len(data)))
56
+ self.end_headers()
57
+ self.wfile.write(data)
58
+
59
+ def do_GET(self):
60
+ path = self.path.rstrip("/")
61
+ if path in ("/health", "/v1/health"):
62
+ return self._send(200, {"status": "ok", "model": name, "fast_kernels": engine.fast_kernels,
63
+ "cuda_graphs": len(engine.graphs)})
64
+ if path == "/v1/models":
65
+ return self._send(200, {"models": [{"name": name, "description": "StartLux-Decision typed decision model"}]})
66
+ self._send(404, {"error": "not found"})
67
+
68
+ def do_POST(self):
69
+ if self.path.rstrip("/") != "/v1/systemone":
70
+ return self._send(404, {"error": "not found"})
71
+ try:
72
+ body = json.loads(self.rfile.read(int(self.headers.get("Content-Length") or 0)) or b"{}")
73
+ questions = body.get("questions")
74
+ if not isinstance(questions, dict) or not questions:
75
+ raise ValueError("questions must be a non-empty object")
76
+ except ValueError as e:
77
+ return self._send(400, {"error": str(e)})
78
+ t = time.perf_counter()
79
+ try:
80
+ with lock:
81
+ answers, usage = engine.decide(body.get("state"), questions)
82
+ except ValueError as e: # unknown question type, prompt over the context limit
83
+ return self._send(422, {"error": str(e)})
84
+ self._send(200, {"answers": answers, "usage": usage, "model": name,
85
+ "latency_ms": round(1000 * (time.perf_counter() - t), 2)})
86
+
87
+ def log_message(self, *args):
88
+ pass
89
+
90
+ print(f"{name} serving on http://{a.host}:{a.port}/v1/systemone (fast kernels: {engine.fast_kernels}, "
91
+ f"cuda graphs: {len(engine.graphs)})", flush=True)
92
+ ThreadingHTTPServer((a.host, a.port), Handler).serve_forever()
93
+
94
+
95
+ if __name__ == "__main__":
96
+ main()
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0997f410c57a1f4e53b09e4be8f4a172d90edd9564368fb0847030937229b9f3
3
+ size 12809320
tokenizer_config.json ADDED
@@ -0,0 +1,305 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "added_tokens_decoder": {
4
+ "248044": {
5
+ "content": "<|endoftext|>",
6
+ "lstrip": false,
7
+ "normalized": false,
8
+ "rstrip": false,
9
+ "single_word": false,
10
+ "special": true
11
+ },
12
+ "248045": {
13
+ "content": "<|im_start|>",
14
+ "lstrip": false,
15
+ "normalized": false,
16
+ "rstrip": false,
17
+ "single_word": false,
18
+ "special": true
19
+ },
20
+ "248046": {
21
+ "content": "<|im_end|>",
22
+ "lstrip": false,
23
+ "normalized": false,
24
+ "rstrip": false,
25
+ "single_word": false,
26
+ "special": true
27
+ },
28
+ "248047": {
29
+ "content": "<|object_ref_start|>",
30
+ "lstrip": false,
31
+ "normalized": false,
32
+ "rstrip": false,
33
+ "single_word": false,
34
+ "special": true
35
+ },
36
+ "248048": {
37
+ "content": "<|object_ref_end|>",
38
+ "lstrip": false,
39
+ "normalized": false,
40
+ "rstrip": false,
41
+ "single_word": false,
42
+ "special": true
43
+ },
44
+ "248049": {
45
+ "content": "<|box_start|>",
46
+ "lstrip": false,
47
+ "normalized": false,
48
+ "rstrip": false,
49
+ "single_word": false,
50
+ "special": true
51
+ },
52
+ "248050": {
53
+ "content": "<|box_end|>",
54
+ "lstrip": false,
55
+ "normalized": false,
56
+ "rstrip": false,
57
+ "single_word": false,
58
+ "special": true
59
+ },
60
+ "248051": {
61
+ "content": "<|quad_start|>",
62
+ "lstrip": false,
63
+ "normalized": false,
64
+ "rstrip": false,
65
+ "single_word": false,
66
+ "special": true
67
+ },
68
+ "248052": {
69
+ "content": "<|quad_end|>",
70
+ "lstrip": false,
71
+ "normalized": false,
72
+ "rstrip": false,
73
+ "single_word": false,
74
+ "special": true
75
+ },
76
+ "248053": {
77
+ "content": "<|vision_start|>",
78
+ "lstrip": false,
79
+ "normalized": false,
80
+ "rstrip": false,
81
+ "single_word": false,
82
+ "special": true
83
+ },
84
+ "248054": {
85
+ "content": "<|vision_end|>",
86
+ "lstrip": false,
87
+ "normalized": false,
88
+ "rstrip": false,
89
+ "single_word": false,
90
+ "special": true
91
+ },
92
+ "248055": {
93
+ "content": "<|vision_pad|>",
94
+ "lstrip": false,
95
+ "normalized": false,
96
+ "rstrip": false,
97
+ "single_word": false,
98
+ "special": true
99
+ },
100
+ "248056": {
101
+ "content": "<|image_pad|>",
102
+ "lstrip": false,
103
+ "normalized": false,
104
+ "rstrip": false,
105
+ "single_word": false,
106
+ "special": true
107
+ },
108
+ "248057": {
109
+ "content": "<|video_pad|>",
110
+ "lstrip": false,
111
+ "normalized": false,
112
+ "rstrip": false,
113
+ "single_word": false,
114
+ "special": true
115
+ },
116
+ "248058": {
117
+ "content": "<tool_call>",
118
+ "lstrip": false,
119
+ "normalized": false,
120
+ "rstrip": false,
121
+ "single_word": false,
122
+ "special": false
123
+ },
124
+ "248059": {
125
+ "content": "</tool_call>",
126
+ "lstrip": false,
127
+ "normalized": false,
128
+ "rstrip": false,
129
+ "single_word": false,
130
+ "special": false
131
+ },
132
+ "248060": {
133
+ "content": "<|fim_prefix|>",
134
+ "lstrip": false,
135
+ "normalized": false,
136
+ "rstrip": false,
137
+ "single_word": false,
138
+ "special": false
139
+ },
140
+ "248061": {
141
+ "content": "<|fim_middle|>",
142
+ "lstrip": false,
143
+ "normalized": false,
144
+ "rstrip": false,
145
+ "single_word": false,
146
+ "special": false
147
+ },
148
+ "248062": {
149
+ "content": "<|fim_suffix|>",
150
+ "lstrip": false,
151
+ "normalized": false,
152
+ "rstrip": false,
153
+ "single_word": false,
154
+ "special": false
155
+ },
156
+ "248063": {
157
+ "content": "<|fim_pad|>",
158
+ "lstrip": false,
159
+ "normalized": false,
160
+ "rstrip": false,
161
+ "single_word": false,
162
+ "special": false
163
+ },
164
+ "248064": {
165
+ "content": "<|repo_name|>",
166
+ "lstrip": false,
167
+ "normalized": false,
168
+ "rstrip": false,
169
+ "single_word": false,
170
+ "special": false
171
+ },
172
+ "248065": {
173
+ "content": "<|file_sep|>",
174
+ "lstrip": false,
175
+ "normalized": false,
176
+ "rstrip": false,
177
+ "single_word": false,
178
+ "special": false
179
+ },
180
+ "248066": {
181
+ "content": "<tool_response>",
182
+ "lstrip": false,
183
+ "normalized": false,
184
+ "rstrip": false,
185
+ "single_word": false,
186
+ "special": false
187
+ },
188
+ "248067": {
189
+ "content": "</tool_response>",
190
+ "lstrip": false,
191
+ "normalized": false,
192
+ "rstrip": false,
193
+ "single_word": false,
194
+ "special": false
195
+ },
196
+ "248068": {
197
+ "content": "<think>",
198
+ "lstrip": false,
199
+ "normalized": false,
200
+ "rstrip": false,
201
+ "single_word": false,
202
+ "special": false
203
+ },
204
+ "248069": {
205
+ "content": "</think>",
206
+ "lstrip": false,
207
+ "normalized": false,
208
+ "rstrip": false,
209
+ "single_word": false,
210
+ "special": false
211
+ },
212
+ "248070": {
213
+ "content": "<|audio_start|>",
214
+ "lstrip": false,
215
+ "normalized": false,
216
+ "rstrip": false,
217
+ "single_word": false,
218
+ "special": true
219
+ },
220
+ "248071": {
221
+ "content": "<|audio_end|>",
222
+ "lstrip": false,
223
+ "normalized": false,
224
+ "rstrip": false,
225
+ "single_word": false,
226
+ "special": true
227
+ },
228
+ "248072": {
229
+ "content": "<tts_pad>",
230
+ "lstrip": false,
231
+ "normalized": false,
232
+ "rstrip": false,
233
+ "single_word": false,
234
+ "special": true
235
+ },
236
+ "248073": {
237
+ "content": "<tts_text_bos>",
238
+ "lstrip": false,
239
+ "normalized": false,
240
+ "rstrip": false,
241
+ "single_word": false,
242
+ "special": true
243
+ },
244
+ "248074": {
245
+ "content": "<tts_text_eod>",
246
+ "lstrip": false,
247
+ "normalized": false,
248
+ "rstrip": false,
249
+ "single_word": false,
250
+ "special": true
251
+ },
252
+ "248075": {
253
+ "content": "<tts_text_bos_single>",
254
+ "lstrip": false,
255
+ "normalized": false,
256
+ "rstrip": false,
257
+ "single_word": false,
258
+ "special": true
259
+ },
260
+ "248076": {
261
+ "content": "<|audio_pad|>",
262
+ "lstrip": false,
263
+ "normalized": false,
264
+ "rstrip": false,
265
+ "single_word": false,
266
+ "special": true
267
+ }
268
+ },
269
+ "additional_special_tokens": [
270
+ "<|im_start|>",
271
+ "<|im_end|>",
272
+ "<|object_ref_start|>",
273
+ "<|object_ref_end|>",
274
+ "<|box_start|>",
275
+ "<|box_end|>",
276
+ "<|quad_start|>",
277
+ "<|quad_end|>",
278
+ "<|vision_start|>",
279
+ "<|vision_end|>",
280
+ "<|vision_pad|>",
281
+ "<|image_pad|>",
282
+ "<|video_pad|>"
283
+ ],
284
+ "bos_token": null,
285
+ "chat_template": "{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- macro render_content(content, do_vision_count, is_system_content=false) %}\n {%- if content is string %}\n {{- content }}\n {%- elif content is iterable and content is not mapping %}\n {%- for item in content %}\n {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain images.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Picture ' ~ image_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|image_pad|><|vision_end|>' }}\n {%- elif 'video' in item or item.type == 'video' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain videos.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Video ' ~ video_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|video_pad|><|vision_end|>' }}\n {%- elif 'text' in item %}\n {{- item.text }}\n {%- else %}\n {{- raise_exception('Unexpected item type in content.') }}\n {%- endif %}\n {%- endfor %}\n {%- elif content is none or content is undefined %}\n {{- '' }}\n {%- else %}\n {{- raise_exception('Unexpected content type.') }}\n {%- endif %}\n{%- endmacro %}\n{%- if not messages %}\n {{- raise_exception('No messages provided.') }}\n{%- endif %}\n{%- set reasoning_instructions = '' %}\n{%- if enable_thinking is undefined or enable_thinking is true %}\n {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}\n {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}\n {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}\n {%- endif %}\n {%- if resolved_reasoning_effort == 'xhigh' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}\n {%- elif resolved_reasoning_effort == 'low' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}\n {%- endif %}\n{%- endif %}\n{%- if tools and tools is iterable and tools is not mapping %}\n {{- '<|im_start|>system\\n' }}\n {%- if reasoning_instructions %}\n {{- reasoning_instructions + '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou have access to the following functions:\\n\\n<tools>\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n</tools>\" }}\n {{- '\\n\\nIf you choose to call a function ONLY reply in the following format with NO suffix:\\n\\n<tool_call>\\n<function=example_function_name>\\n<parameter=example_parameter_1>\\nvalue_1\\n</parameter>\\n<parameter=example_parameter_2>\\nThis is the value for the second parameter\\nthat can span\\nmultiple lines\\n</parameter>\\n</function>\\n</tool_call>\\n\\n<IMPORTANT>\\nReminder:\\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\\n- Required parameters MUST be specified\\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\\n</IMPORTANT>' }}\n {%- if messages[0].role == 'system' %}\n {%- set content = render_content(messages[0].content, false, true)|trim %}\n {%- if content %}\n {{- '\\n\\n' + content }}\n {%- endif %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n{%- else %}\n {%- if messages[0].role == 'system' %}\n {%- set content = render_content(messages[0].content, false, true)|trim %}\n {%- if content %}\n {{- '<|im_start|>system\\n' + (reasoning_instructions + '\\n\\n' if reasoning_instructions else '') + content + '<|im_end|>\\n' }}\n {%- elif reasoning_instructions %}\n {{- '<|im_start|>system\\n' + reasoning_instructions + '<|im_end|>\\n' }}\n {%- endif %}\n {%- elif reasoning_instructions %}\n {{- '<|im_start|>system\\n' + reasoning_instructions + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" %}\n {%- set content = render_content(message.content, false)|trim %}\n {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if ns.multi_step_tool %}\n {{- raise_exception('No user query found in messages.') }}\n{%- endif %}\n{%- for message in messages %}\n {%- set content = render_content(message.content, true)|trim %}\n {%- if message.role == \"system\" %}\n {%- if not loop.first %}\n {{- raise_exception('System message must be at the beginning.') }}\n {%- endif %}\n {%- elif message.role == \"user\" %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- endif %}\n {%- set reasoning_content = reasoning_content|trim %}\n {%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}\n {{- '<|im_start|>' + message.role + '\\n<think>\\n' + reasoning_content + '\\n</think>\\n\\n' + content }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {%- if loop.first %}\n {%- if content|trim %}\n {{- '\\n\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- else %}\n {{- '<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- else %}\n {{- '\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- if tool_call.arguments is defined and tool_call.arguments != '' %}\n {%- for args_name, args_value in tool_call.arguments|items %}\n {{- '<parameter=' + args_name + '>\\n' }}\n {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}\n {{- args_value }}\n {{- '\\n</parameter>\\n' }}\n {%- endfor %}\n {%- endif %}\n {{- '</function>\\n</tool_call>' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.previtem and loop.previtem.role != \"tool\" %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n<tool_response>\\n' }}\n {{- content }}\n {{- '\\n</tool_response>' }}\n {%- if not loop.last and loop.nextitem.role != \"tool\" %}\n {{- '<|im_end|>\\n' }}\n {%- elif loop.last %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- else %}\n {{- raise_exception('Unexpected message role.') }}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is false %}\n {{- '<think>\\n\\n</think>\\n\\n' }}\n {%- else %}\n {{- '<think>\\n' }}\n {%- endif %}\n{%- endif %}",
286
+ "clean_up_tokenization_spaces": false,
287
+ "eos_token": "<|im_end|>",
288
+ "errors": "replace",
289
+ "model_max_length": 262144,
290
+ "pad_token": "<|endoftext|>",
291
+ "split_special_tokens": false,
292
+ "tokenizer_class": "Qwen2Tokenizer",
293
+ "unk_token": null,
294
+ "add_bos_token": false,
295
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
296
+ "extra_special_tokens": {
297
+ "audio_bos_token": "<|audio_start|>",
298
+ "audio_eos_token": "<|audio_end|>",
299
+ "audio_token": "<|audio_pad|>",
300
+ "image_token": "<|image_pad|>",
301
+ "video_token": "<|video_pad|>",
302
+ "vision_bos_token": "<|vision_start|>",
303
+ "vision_eos_token": "<|vision_end|>"
304
+ }
305
+ }
vocab.json ADDED
The diff for this file is too large to render. See raw diff