oktayd commited on
Commit
a668ea0
·
verified ·
1 Parent(s): 61cb76f

Publish Q36 v2 naming, complete edition guide and honest benchmark scope

Browse files
BENCHMARK-DEVICE-SUMMARY.json ADDED
@@ -0,0 +1,545 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "status": "local_diagnostic_partial_semantic_audit",
3
+ "not_official_leaderboard": true,
4
+ "score_formula": "C*(70+15F+5D+10T); pending values are null; no overall all-task accuracy claimed",
5
+ "runs": [
6
+ {
7
+ "run": "q36-final-bf16-h200",
8
+ "model": "Q36",
9
+ "device": "H200",
10
+ "attempted": 31,
11
+ "completed": 28,
12
+ "not_attempted": 173,
13
+ "old_scored": 19,
14
+ "old_correct": 10,
15
+ "new_scored": 15,
16
+ "new_content_correct": 10,
17
+ "new_partial": 0,
18
+ "review_pending": 16,
19
+ "mean_score_scored_only": 63.666666666666664,
20
+ "sum_request_seconds": 554.38988994807,
21
+ "outcomes": {
22
+ "incomplete_other": 3,
23
+ "complete": 28
24
+ },
25
+ "repetition_flags": 3,
26
+ "latency": {
27
+ "n": 31,
28
+ "median": 6.871402349323034,
29
+ "p95": 120.04690708965063,
30
+ "min": 0.30404097586870193,
31
+ "max": 120.19533631578088
32
+ },
33
+ "e2e_tps_ge64": {
34
+ "n": 23,
35
+ "median": 21.78950706263202,
36
+ "p95": 22.13126769002376,
37
+ "min": 20.6417058337875,
38
+ "max": 22.282956428042905
39
+ },
40
+ "ttfo": {
41
+ "n": 0,
42
+ "median": null,
43
+ "p95": null,
44
+ "min": null,
45
+ "max": null
46
+ },
47
+ "resources_and_native_timings": {}
48
+ },
49
+ {
50
+ "run": "q36-final-q4_k_m-llamacpp-h200",
51
+ "model": "Q36",
52
+ "device": "H200",
53
+ "attempted": 204,
54
+ "completed": 186,
55
+ "not_attempted": 0,
56
+ "old_scored": 108,
57
+ "old_correct": 61,
58
+ "new_scored": 113,
59
+ "new_content_correct": 87,
60
+ "new_partial": 2,
61
+ "review_pending": 91,
62
+ "mean_score_scored_only": 76.55884955752212,
63
+ "sum_request_seconds": 467.68541045859456,
64
+ "outcomes": {
65
+ "complete": 186,
66
+ "token_limit": 18
67
+ },
68
+ "repetition_flags": 18,
69
+ "latency": {
70
+ "n": 204,
71
+ "median": 0.5681875385344028,
72
+ "p95": 19.078906919807196,
73
+ "min": 0.23451422899961472,
74
+ "max": 19.142078693956137
75
+ },
76
+ "e2e_tps_ge64": {
77
+ "n": 104,
78
+ "median": 156.35807976215648,
79
+ "p95": 214.79182473741946,
80
+ "min": 95.49705544705205,
81
+ "max": 215.13906599984156
82
+ },
83
+ "ttfo": {
84
+ "n": 0,
85
+ "median": null,
86
+ "p95": null,
87
+ "min": null,
88
+ "max": null
89
+ },
90
+ "resources_and_native_timings": {}
91
+ },
92
+ {
93
+ "run": "q36-final-q4_k_m-llamacpp-rtx5090",
94
+ "model": "Q36",
95
+ "device": "RTX 5090",
96
+ "attempted": 204,
97
+ "completed": 190,
98
+ "not_attempted": 0,
99
+ "old_scored": 109,
100
+ "old_correct": 62,
101
+ "new_scored": 116,
102
+ "new_content_correct": 91,
103
+ "new_partial": 1,
104
+ "review_pending": 88,
105
+ "mean_score_scored_only": 77.2869827586207,
106
+ "sum_request_seconds": 347.72151251439936,
107
+ "outcomes": {
108
+ "complete": 190,
109
+ "token_limit": 14
110
+ },
111
+ "repetition_flags": 14,
112
+ "latency": {
113
+ "n": 204,
114
+ "median": 0.508773922920227,
115
+ "p95": 16.40450456715189,
116
+ "min": 0.21930593415163457,
117
+ "max": 16.689129662001505
118
+ },
119
+ "e2e_tps_ge64": {
120
+ "n": 100,
121
+ "median": 174.5699836304055,
122
+ "p95": 249.32785102452453,
123
+ "min": 106.72219620117671,
124
+ "max": 249.90085545053165
125
+ },
126
+ "ttfo": {
127
+ "n": 204,
128
+ "median": 0.2632923311321065,
129
+ "p95": 0.38364072307012975,
130
+ "min": 0.21216707909479737,
131
+ "max": 0.6255119100678712
132
+ },
133
+ "resources_and_native_timings": {
134
+ "attempted": 204,
135
+ "completed": 190,
136
+ "request_latency_seconds": {
137
+ "count": 204,
138
+ "median": 0.508773922920227,
139
+ "p95_nearest_rank": 16.40450456715189,
140
+ "minimum": 0.21930593415163457,
141
+ "maximum": 16.689129662001505
142
+ },
143
+ "time_to_first_output_seconds": {
144
+ "count": 204,
145
+ "median": 0.2632923311321065,
146
+ "p95_nearest_rank": 0.38364072307012975,
147
+ "minimum": 0.21216707909479737,
148
+ "maximum": 0.6255119100678712
149
+ },
150
+ "tokens_per_second_including_prefill_outputs_ge64": {
151
+ "count": 100,
152
+ "median": 174.5699836304055,
153
+ "p95_nearest_rank": 249.32785102452453,
154
+ "minimum": 106.72219620117671,
155
+ "maximum": 249.90085545053165
156
+ },
157
+ "server_native_decode_tokens_per_second": {
158
+ "count": 204,
159
+ "median": 244.94170595571296,
160
+ "p95_nearest_rank": 253.86931857973244,
161
+ "minimum": 92.5497454881999,
162
+ "maximum": 256.53321350605455
163
+ },
164
+ "server_native_prompt_tokens_per_second": {
165
+ "count": 204,
166
+ "median": 759.2535719351886,
167
+ "p95_nearest_rank": 2764.939261319481,
168
+ "minimum": 173.17016855229738,
169
+ "maximum": 5286.562207759986
170
+ },
171
+ "reported_prompt_tokens": 36728,
172
+ "reported_completion_tokens": 73608,
173
+ "requests_without_final_token_usage": 0,
174
+ "limitations": [
175
+ "Incomplete requests may lack final usage; reported token totals are not necessarily all generated tokens.",
176
+ "TTFO includes client/network overhead and stream buffering; not claimed to be server-only TTFT.",
177
+ "Native prompt/decode rates are null when server timings are absent.",
178
+ "For H200 comparison use identical prompt IDs and shared end-to-end metric; H200 did not capture TTFO or sampled peak VRAM.",
179
+ "Different generated lengths and timeout rates affect aggregate throughput; do not treat an aggregate ratio as pure GPU speedup.",
180
+ "Microbenchmark is text-only and separate from multimodal diagnostic; repetitions estimate timing variability, not answer accuracy."
181
+ ]
182
+ }
183
+ },
184
+ {
185
+ "run": "huihui-original-q4_k_m-llamacpp-rtx5090",
186
+ "model": "Huihui",
187
+ "device": "RTX 5090",
188
+ "attempted": 204,
189
+ "completed": 196,
190
+ "not_attempted": 0,
191
+ "old_scored": 114,
192
+ "old_correct": 87,
193
+ "new_scored": 117,
194
+ "new_content_correct": 107,
195
+ "new_partial": 0,
196
+ "review_pending": 87,
197
+ "mean_score_scored_only": 89.91452991452991,
198
+ "sum_request_seconds": 399.81621656077914,
199
+ "outcomes": {
200
+ "complete": 196,
201
+ "token_limit": 8
202
+ },
203
+ "repetition_flags": 3,
204
+ "latency": {
205
+ "n": 204,
206
+ "median": 1.0916633284650743,
207
+ "p95": 6.350672701839358,
208
+ "min": 0.2248853868804872,
209
+ "max": 16.718071810901165
210
+ },
211
+ "e2e_tps_ge64": {
212
+ "n": 164,
213
+ "median": 203.78866356038696,
214
+ "p95": 246.14011225922462,
215
+ "min": 101.47720797551588,
216
+ "max": 250.50201997533694
217
+ },
218
+ "ttfo": {
219
+ "n": 204,
220
+ "median": 0.24578010209370404,
221
+ "p95": 0.38527246192097664,
222
+ "min": 0.21113943797536194,
223
+ "max": 0.6080712059047073
224
+ },
225
+ "resources_and_native_timings": {
226
+ "attempted": 204,
227
+ "completed": 196,
228
+ "request_latency_seconds": {
229
+ "count": 204,
230
+ "median": 1.0916633284650743,
231
+ "p95_nearest_rank": 6.350672701839358,
232
+ "minimum": 0.2248853868804872,
233
+ "maximum": 16.718071810901165
234
+ },
235
+ "time_to_first_output_seconds": {
236
+ "count": 204,
237
+ "median": 0.24578010209370404,
238
+ "p95_nearest_rank": 0.38527246192097664,
239
+ "minimum": 0.21113943797536194,
240
+ "maximum": 0.6080712059047073
241
+ },
242
+ "tokens_per_second_including_prefill_outputs_ge64": {
243
+ "count": 164,
244
+ "median": 203.78866356038696,
245
+ "p95_nearest_rank": 246.14011225922462,
246
+ "minimum": 101.47720797551588,
247
+ "maximum": 250.50201997533694
248
+ },
249
+ "server_native_decode_tokens_per_second": {
250
+ "count": 204,
251
+ "median": 252.68031289342514,
252
+ "p95_nearest_rank": 255.56013413126567,
253
+ "minimum": 98.33325138895718,
254
+ "maximum": 256.3838921478427
255
+ },
256
+ "server_native_prompt_tokens_per_second": {
257
+ "count": 204,
258
+ "median": 774.6375227145426,
259
+ "p95_nearest_rank": 2809.1288376190555,
260
+ "minimum": 198.83482790845645,
261
+ "maximum": 5265.592426671397
262
+ },
263
+ "reported_prompt_tokens": 36728,
264
+ "reported_completion_tokens": 87541,
265
+ "requests_without_final_token_usage": 0,
266
+ "sampled_gpu_power_watts": {
267
+ "count": 396,
268
+ "median": 366.19000000000005,
269
+ "p95_nearest_rank": 379.18,
270
+ "minimum": 25.03,
271
+ "maximum": 408.61
272
+ },
273
+ "sampled_gpu_utilization_percent": {
274
+ "count": 396,
275
+ "median": 84.0,
276
+ "p95_nearest_rank": 84.0,
277
+ "minimum": 0.0,
278
+ "maximum": 97.0
279
+ },
280
+ "limitations": [
281
+ "Incomplete requests may lack final usage; reported token totals are not necessarily all generated tokens.",
282
+ "TTFO includes client/network overhead and stream buffering; not claimed to be server-only TTFT.",
283
+ "Native prompt/decode rates are null when server timings are absent.",
284
+ "For H200 comparison use identical prompt IDs and shared end-to-end metric; H200 did not capture TTFO or sampled peak VRAM.",
285
+ "Different generated lengths and timeout rates affect aggregate throughput; do not treat an aggregate ratio as pure GPU speedup.",
286
+ "Microbenchmark is text-only and separate from multimodal diagnostic; repetitions estimate timing variability, not answer accuracy."
287
+ ]
288
+ }
289
+ },
290
+ {
291
+ "run": "q36-final-q4_k_m-llamacpp-rtx2000ada-mixed24",
292
+ "model": "Q36",
293
+ "device": "RTX 2000 Ada (CPU+GPU)",
294
+ "attempted": 204,
295
+ "completed": 194,
296
+ "not_attempted": 0,
297
+ "old_scored": 109,
298
+ "old_correct": 61,
299
+ "new_scored": 116,
300
+ "new_content_correct": 87,
301
+ "new_partial": 1,
302
+ "review_pending": 88,
303
+ "mean_score_scored_only": 73.70939655172414,
304
+ "sum_request_seconds": 1502.7530024759471,
305
+ "outcomes": {
306
+ "complete": 194,
307
+ "time_limit": 10
308
+ },
309
+ "repetition_flags": 10,
310
+ "latency": {
311
+ "n": 204,
312
+ "median": 3.9858929477632046,
313
+ "p95": 29.909678515046835,
314
+ "min": 0.6382952108979225,
315
+ "max": 45.05058938637376
316
+ },
317
+ "e2e_tps_ge64": {
318
+ "n": 88,
319
+ "median": 18.23039974058431,
320
+ "p95": 20.79240100570237,
321
+ "min": 12.120789798716995,
322
+ "max": 21.60126768643173
323
+ },
324
+ "ttfo": {
325
+ "n": 204,
326
+ "median": 1.0794818717986345,
327
+ "p95": 2.476460698992014,
328
+ "min": 0.5688546188175678,
329
+ "max": 6.211364433169365
330
+ },
331
+ "resources_and_native_timings": {
332
+ "attempted": 204,
333
+ "completed": 194,
334
+ "request_latency_seconds": {
335
+ "count": 204,
336
+ "median": 3.9858929477632046,
337
+ "p95_nearest_rank": 29.909678515046835,
338
+ "minimum": 0.6382952108979225,
339
+ "maximum": 45.05058938637376
340
+ },
341
+ "time_to_first_output_seconds": {
342
+ "count": 204,
343
+ "median": 1.0794818717986345,
344
+ "p95_nearest_rank": 2.476460698992014,
345
+ "minimum": 0.5688546188175678,
346
+ "maximum": 6.211364433169365
347
+ },
348
+ "tokens_per_second_including_prefill_outputs_ge64": {
349
+ "count": 88,
350
+ "median": 18.23039974058431,
351
+ "p95_nearest_rank": 20.79240100570237,
352
+ "minimum": 12.120789798716995,
353
+ "maximum": 21.60126768643173
354
+ },
355
+ "server_native_decode_tokens_per_second": {
356
+ "count": 194,
357
+ "median": 21.914455555451973,
358
+ "p95_nearest_rank": 22.6313741164166,
359
+ "minimum": 13.167722882638389,
360
+ "maximum": 22.996138086060174
361
+ },
362
+ "server_native_prompt_tokens_per_second": {
363
+ "count": 194,
364
+ "median": 88.9779467340917,
365
+ "p95_nearest_rank": 287.9468373175563,
366
+ "minimum": 18.242119004543568,
367
+ "maximum": 391.30143041626985
368
+ },
369
+ "reported_prompt_tokens": 35551,
370
+ "reported_completion_tokens": 17227,
371
+ "requests_without_final_token_usage": 10,
372
+ "sampled_gpu_power_watts": {
373
+ "count": 1479,
374
+ "median": 25.97,
375
+ "p95_nearest_rank": 35.38,
376
+ "minimum": 6.86,
377
+ "maximum": 55.78
378
+ },
379
+ "sampled_gpu_utilization_percent": {
380
+ "count": 1479,
381
+ "median": 17.0,
382
+ "p95_nearest_rank": 79.0,
383
+ "minimum": 0.0,
384
+ "maximum": 100.0
385
+ },
386
+ "sampled_llama_server_cpu_percent_one_core_equals100": {
387
+ "count": 1479,
388
+ "median": 754.3,
389
+ "p95_nearest_rank": 761.8,
390
+ "minimum": 0.0,
391
+ "maximum": 772.7
392
+ },
393
+ "sampled_llama_server_rss_mib": {
394
+ "count": 1479,
395
+ "median": 18108.24609375,
396
+ "p95_nearest_rank": 18288.66796875,
397
+ "minimum": 55.21484375,
398
+ "maximum": 21086.76953125
399
+ },
400
+ "sampled_container_memory_mib_including_file_cache": {
401
+ "count": 1479,
402
+ "median": 21982.6484375,
403
+ "p95_nearest_rank": 29548.671875,
404
+ "minimum": 15274.6640625,
405
+ "maximum": 29552.79296875
406
+ },
407
+ "limitations": [
408
+ "Incomplete requests may lack final usage; reported token totals are not necessarily all generated tokens.",
409
+ "TTFO includes client/network overhead and stream buffering; not claimed to be server-only TTFT.",
410
+ "Native prompt/decode rates are null when server timings are absent.",
411
+ "For H200 comparison use identical prompt IDs and shared end-to-end metric; H200 did not capture TTFO or sampled peak VRAM.",
412
+ "Different generated lengths and timeout rates affect aggregate throughput; do not treat an aggregate ratio as pure GPU speedup.",
413
+ "Microbenchmark is text-only and separate from multimodal diagnostic; repetitions estimate timing variability, not answer accuracy."
414
+ ]
415
+ }
416
+ },
417
+ {
418
+ "run": "huihui-original-q4_k_m-llamacpp-rtx2000ada-mixed24",
419
+ "model": "Huihui",
420
+ "device": "RTX 2000 Ada (CPU+GPU)",
421
+ "attempted": 182,
422
+ "completed": 169,
423
+ "not_attempted": 22,
424
+ "old_scored": 99,
425
+ "old_correct": 76,
426
+ "new_scored": 100,
427
+ "new_content_correct": 92,
428
+ "new_partial": 0,
429
+ "review_pending": 82,
430
+ "mean_score_scored_only": 90.05,
431
+ "sum_request_seconds": 2396.760685328394,
432
+ "outcomes": {
433
+ "time_limit": 13,
434
+ "complete": 169
435
+ },
436
+ "repetition_flags": 0,
437
+ "latency": {
438
+ "n": 182,
439
+ "median": 10.075448336079717,
440
+ "p95": 45.013748209923506,
441
+ "min": 0.711407907307148,
442
+ "max": 45.04958474263549
443
+ },
444
+ "e2e_tps_ge64": {
445
+ "n": 129,
446
+ "median": 19.52059495468179,
447
+ "p95": 21.135017829481033,
448
+ "min": 12.736947373166998,
449
+ "max": 21.456602392611842
450
+ },
451
+ "ttfo": {
452
+ "n": 182,
453
+ "median": 1.100716471672058,
454
+ "p95": 2.4438186809420586,
455
+ "min": 0.5777811706066132,
456
+ "max": 7.080324701964855
457
+ },
458
+ "resources_and_native_timings": {
459
+ "attempted": 182,
460
+ "completed": 169,
461
+ "request_latency_seconds": {
462
+ "count": 182,
463
+ "median": 10.075448336079717,
464
+ "p95_nearest_rank": 45.013748209923506,
465
+ "minimum": 0.711407907307148,
466
+ "maximum": 45.04958474263549
467
+ },
468
+ "time_to_first_output_seconds": {
469
+ "count": 182,
470
+ "median": 1.100716471672058,
471
+ "p95_nearest_rank": 2.4438186809420586,
472
+ "minimum": 0.5777811706066132,
473
+ "maximum": 7.080324701964855
474
+ },
475
+ "tokens_per_second_including_prefill_outputs_ge64": {
476
+ "count": 129,
477
+ "median": 19.52059495468179,
478
+ "p95_nearest_rank": 21.135017829481033,
479
+ "minimum": 12.736947373166998,
480
+ "maximum": 21.456602392611842
481
+ },
482
+ "server_native_decode_tokens_per_second": {
483
+ "count": 169,
484
+ "median": 21.803532864499694,
485
+ "p95_nearest_rank": 22.333440919859836,
486
+ "minimum": 13.217722322089456,
487
+ "maximum": 22.8402607202002
488
+ },
489
+ "server_native_prompt_tokens_per_second": {
490
+ "count": 169,
491
+ "median": 96.93778059824136,
492
+ "p95_nearest_rank": 281.1025909507955,
493
+ "minimum": 34.69781702981594,
494
+ "maximum": 385.468739092297
495
+ },
496
+ "reported_prompt_tokens": 29227,
497
+ "reported_completion_tokens": 35178,
498
+ "requests_without_final_token_usage": 13,
499
+ "sampled_gpu_power_watts": {
500
+ "count": 2351,
501
+ "median": 25.96,
502
+ "p95_nearest_rank": 33.74,
503
+ "minimum": 17.89,
504
+ "maximum": 60.31
505
+ },
506
+ "sampled_gpu_utilization_percent": {
507
+ "count": 2351,
508
+ "median": 17.0,
509
+ "p95_nearest_rank": 35.0,
510
+ "minimum": 0.0,
511
+ "maximum": 100.0
512
+ },
513
+ "sampled_llama_server_cpu_percent_one_core_equals100": {
514
+ "count": 2351,
515
+ "median": 755.1,
516
+ "p95_nearest_rank": 760.5,
517
+ "minimum": 0.0,
518
+ "maximum": 773.7
519
+ },
520
+ "sampled_llama_server_rss_mib": {
521
+ "count": 2351,
522
+ "median": 18181.86328125,
523
+ "p95_nearest_rank": 18359.9609375,
524
+ "minimum": 53.3671875,
525
+ "maximum": 18437.01171875
526
+ },
527
+ "sampled_container_memory_mib_including_file_cache": {
528
+ "count": 2351,
529
+ "median": 29415.12109375,
530
+ "p95_nearest_rank": 29547.53125,
531
+ "minimum": 23668.16796875,
532
+ "maximum": 29553.60546875
533
+ },
534
+ "limitations": [
535
+ "Incomplete requests may lack final usage; reported token totals are not necessarily all generated tokens.",
536
+ "TTFO includes client/network overhead and stream buffering; not claimed to be server-only TTFT.",
537
+ "Native prompt/decode rates are null when server timings are absent.",
538
+ "For H200 comparison use identical prompt IDs and shared end-to-end metric; H200 did not capture TTFO or sampled peak VRAM.",
539
+ "Different generated lengths and timeout rates affect aggregate throughput; do not treat an aggregate ratio as pure GPU speedup.",
540
+ "Microbenchmark is text-only and separate from multimodal diagnostic; repetitions estimate timing variability, not answer accuracy."
541
+ ]
542
+ }
543
+ }
544
+ ]
545
+ }
BENCHMARK-EXTENSION-QWEN38.json ADDED
@@ -0,0 +1,315 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": "q36.benchmark-extension.v1",
3
+ "source_url": "https://huggingface.co/Qwen/Qwen3.8-27B",
4
+ "source_revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0",
5
+ "status": "registered_not_run",
6
+ "publication_scores_not_local_results": true,
7
+ "new_model_baseline": {
8
+ "id": "Qwen/Qwen3.8-27B",
9
+ "revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0",
10
+ "downloaded_weights": false,
11
+ "measured_locally": false
12
+ },
13
+ "benchmarks": [
14
+ {
15
+ "name": "Terminal Bench 2.1 (Terminus)",
16
+ "category": "coding",
17
+ "publisher_reported": "73.0",
18
+ "proposed_local_cases": 5,
19
+ "environment": "terminal_container",
20
+ "status": "registered_adapter_and_data_not_prepared",
21
+ "requirements": "version_and_harness_required",
22
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
23
+ "repetitions": 1,
24
+ "official_comparable": false
25
+ },
26
+ {
27
+ "name": "SWE-bench Pro",
28
+ "category": "coding",
29
+ "publisher_reported": "61.7",
30
+ "proposed_local_cases": 5,
31
+ "environment": "repository_container",
32
+ "status": "registered_adapter_and_data_not_prepared",
33
+ "requirements": "refined_tasks_Claude_Code_256K_temp1_top_p095",
34
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
35
+ "repetitions": 1,
36
+ "official_comparable": false
37
+ },
38
+ {
39
+ "name": "NL2Repo-Bench",
40
+ "category": "coding",
41
+ "publisher_reported": "42.3",
42
+ "proposed_local_cases": 5,
43
+ "environment": "repository_container",
44
+ "status": "registered_adapter_and_data_not_prepared",
45
+ "requirements": "Claude_Code_no_repository_download_reward_hacking",
46
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
47
+ "repetitions": 1,
48
+ "official_comparable": false
49
+ },
50
+ {
51
+ "name": "DeepSWE 1.1",
52
+ "category": "coding",
53
+ "publisher_reported": "42.2",
54
+ "proposed_local_cases": 5,
55
+ "environment": "repository_container",
56
+ "status": "registered_adapter_and_data_not_prepared",
57
+ "requirements": "Claude_Code_256K_temp1_top_p095",
58
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
59
+ "repetitions": 1,
60
+ "official_comparable": false
61
+ },
62
+ {
63
+ "name": "QwenSWEBench",
64
+ "category": "coding",
65
+ "publisher_reported": "79.0",
66
+ "proposed_local_cases": null,
67
+ "environment": "in_house",
68
+ "status": "blocked_in_house_release_needed",
69
+ "requirements": "in_house_avg3_8h_timeout_32768_output_256K_context",
70
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
71
+ "repetitions": 1,
72
+ "official_comparable": false
73
+ },
74
+ {
75
+ "name": "CoWorkBench",
76
+ "category": "agent",
77
+ "publisher_reported": "70.7",
78
+ "proposed_local_cases": null,
79
+ "environment": "in_house",
80
+ "status": "blocked_in_house_release_needed",
81
+ "requirements": "in_house_productivity_tasks",
82
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
83
+ "repetitions": 1,
84
+ "official_comparable": false
85
+ },
86
+ {
87
+ "name": "JobBench",
88
+ "category": "agent",
89
+ "publisher_reported": "33.4",
90
+ "proposed_local_cases": 5,
91
+ "environment": "agent_environment",
92
+ "status": "registered_adapter_and_data_not_prepared",
93
+ "requirements": "dataset_and_harness_availability_unverified",
94
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
95
+ "repetitions": 1,
96
+ "official_comparable": false
97
+ },
98
+ {
99
+ "name": "Agents' Last Exam",
100
+ "category": "agent",
101
+ "publisher_reported": "Pass@1 20.4; Score 42.9",
102
+ "proposed_local_cases": 5,
103
+ "environment": "agent_environment",
104
+ "status": "registered_adapter_and_data_not_prepared",
105
+ "requirements": "two_distinct_metrics",
106
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
107
+ "repetitions": 1,
108
+ "official_comparable": false
109
+ },
110
+ {
111
+ "name": "IFBench",
112
+ "category": "instruction",
113
+ "publisher_reported": "79.5",
114
+ "proposed_local_cases": 20,
115
+ "environment": "text",
116
+ "status": "registered_adapter_and_data_not_prepared",
117
+ "requirements": "not_IFEval",
118
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
119
+ "repetitions": 1,
120
+ "official_comparable": false
121
+ },
122
+ {
123
+ "name": "GPQA Diamond",
124
+ "category": "science",
125
+ "publisher_reported": "89.2",
126
+ "proposed_local_cases": 20,
127
+ "environment": "text",
128
+ "status": "registered_adapter_and_data_not_prepared",
129
+ "requirements": "existing_5_prompt_diagnostic_not_full_benchmark",
130
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
131
+ "repetitions": 1,
132
+ "official_comparable": false
133
+ },
134
+ {
135
+ "name": "HLE",
136
+ "category": "reasoning",
137
+ "publisher_reported": "30.8",
138
+ "proposed_local_cases": 10,
139
+ "environment": "text_multimodal",
140
+ "status": "registered_adapter_and_data_not_prepared",
141
+ "requirements": "publisher_GPT4o_judge_separate_judge_budget",
142
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
143
+ "repetitions": 1,
144
+ "official_comparable": false
145
+ },
146
+ {
147
+ "name": "LiveCodeBench v6",
148
+ "category": "coding",
149
+ "publisher_reported": "90.3",
150
+ "proposed_local_cases": 10,
151
+ "environment": "code_sandbox",
152
+ "status": "registered_adapter_and_data_not_prepared",
153
+ "requirements": "existing_LiveCodeBench_revision_must_be_reconciled",
154
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
155
+ "repetitions": 1,
156
+ "official_comparable": false
157
+ },
158
+ {
159
+ "name": "OSWorld-Verified",
160
+ "category": "desktop",
161
+ "publisher_reported": "84.3",
162
+ "proposed_local_cases": 5,
163
+ "environment": "desktop_vm",
164
+ "status": "registered_adapter_and_data_not_prepared",
165
+ "requirements": "real_actions_not_static_screenshot_qa",
166
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
167
+ "repetitions": 1,
168
+ "official_comparable": false
169
+ },
170
+ {
171
+ "name": "WebArena-Verified",
172
+ "category": "browser",
173
+ "publisher_reported": "64.8",
174
+ "proposed_local_cases": 5,
175
+ "environment": "browser_environment",
176
+ "status": "registered_adapter_and_data_not_prepared",
177
+ "requirements": "official_grader_OSWorld_scaffold",
178
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
179
+ "repetitions": 1,
180
+ "official_comparable": false
181
+ },
182
+ {
183
+ "name": "AndroidWorld",
184
+ "category": "mobile",
185
+ "publisher_reported": "81.9",
186
+ "proposed_local_cases": 5,
187
+ "environment": "android_emulator",
188
+ "status": "registered_adapter_and_data_not_prepared",
189
+ "requirements": "emulator_and_state_reset_required",
190
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
191
+ "repetitions": 1,
192
+ "official_comparable": false
193
+ },
194
+ {
195
+ "name": "RecreationBench",
196
+ "category": "application",
197
+ "publisher_reported": "47.1",
198
+ "proposed_local_cases": null,
199
+ "environment": "in_house",
200
+ "status": "blocked_in_house_release_needed",
201
+ "requirements": "in_house_Ubuntu_macOS_Windows_Android_web",
202
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
203
+ "repetitions": 1,
204
+ "official_comparable": false
205
+ },
206
+ {
207
+ "name": "ClawEval-MM",
208
+ "category": "tools",
209
+ "publisher_reported": "Pass@3 57.4; Average 56.9",
210
+ "proposed_local_cases": 5,
211
+ "environment": "multimodal_agent",
212
+ "status": "registered_adapter_and_data_not_prepared",
213
+ "requirements": "official_pass3_requires_three_trials; local_single_trial_not_pass3",
214
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
215
+ "repetitions": 1,
216
+ "official_comparable": false
217
+ },
218
+ {
219
+ "name": "SWE-MM",
220
+ "category": "coding_vision",
221
+ "publisher_reported": "38.6",
222
+ "proposed_local_cases": 5,
223
+ "environment": "repository_container",
224
+ "status": "registered_adapter_and_data_not_prepared",
225
+ "requirements": "public_dev_split_with_Opus47_appendix8_3_modifications",
226
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
227
+ "repetitions": 1,
228
+ "official_comparable": false
229
+ },
230
+ {
231
+ "name": "Vision2Web",
232
+ "category": "web_development",
233
+ "publisher_reported": "62.9",
234
+ "proposed_local_cases": 5,
235
+ "environment": "browser_and_code",
236
+ "status": "registered_adapter_and_data_not_prepared",
237
+ "requirements": "Claude_Code_judged_gpt5.4_2026_03_05",
238
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
239
+ "repetitions": 1,
240
+ "official_comparable": false
241
+ },
242
+ {
243
+ "name": "MathVision",
244
+ "category": "visual_reasoning",
245
+ "publisher_reported": "Without CI 90.0; With CI 94.6",
246
+ "proposed_local_cases": 10,
247
+ "environment": "image_qa_plus_optional_code",
248
+ "status": "registered_adapter_and_data_not_prepared",
249
+ "requirements": "corrected_annotations_boxed_final_CI_strata_separate",
250
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
251
+ "repetitions": 1,
252
+ "official_comparable": false
253
+ },
254
+ {
255
+ "name": "BabyVision",
256
+ "category": "visual_reasoning",
257
+ "publisher_reported": "Without CI 65.7; With CI 85.6",
258
+ "proposed_local_cases": 10,
259
+ "environment": "image_qa_plus_optional_code",
260
+ "status": "registered_adapter_and_data_not_prepared",
261
+ "requirements": "CI_strata_separate",
262
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
263
+ "repetitions": 1,
264
+ "official_comparable": false
265
+ },
266
+ {
267
+ "name": "CharXiv (RQ)",
268
+ "category": "chart_reasoning",
269
+ "publisher_reported": "Without CI 83.7; With CI 90.2",
270
+ "proposed_local_cases": 10,
271
+ "environment": "image_qa_plus_optional_code",
272
+ "status": "registered_adapter_and_data_not_prepared",
273
+ "requirements": "corrected_annotations_CI_strata_separate",
274
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
275
+ "repetitions": 1,
276
+ "official_comparable": false
277
+ },
278
+ {
279
+ "name": "OmniDocBench 1.5",
280
+ "category": "document",
281
+ "publisher_reported": "91.1",
282
+ "proposed_local_cases": 10,
283
+ "environment": "document_parser",
284
+ "status": "registered_adapter_and_data_not_prepared",
285
+ "requirements": "official_metric_not_generic_accuracy",
286
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
287
+ "repetitions": 1,
288
+ "official_comparable": false
289
+ },
290
+ {
291
+ "name": "RealWorldQA",
292
+ "category": "perception",
293
+ "publisher_reported": "85.9",
294
+ "proposed_local_cases": 20,
295
+ "environment": "image_qa",
296
+ "status": "registered_adapter_and_data_not_prepared",
297
+ "requirements": "dataset_revision_and_image_preprocessing_lock_required",
298
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
299
+ "repetitions": 1,
300
+ "official_comparable": false
301
+ },
302
+ {
303
+ "name": "ERQA",
304
+ "category": "embodied_reasoning",
305
+ "publisher_reported": "65.5",
306
+ "proposed_local_cases": 10,
307
+ "environment": "image_qa",
308
+ "status": "registered_adapter_and_data_not_prepared",
309
+ "requirements": "dataset_revision_and_official_scoring_required",
310
+ "source": "https://huggingface.co/Qwen/Qwen3.8-27B",
311
+ "repetitions": 1,
312
+ "official_comparable": false
313
+ }
314
+ ]
315
+ }
BENCHMARK-PLAN.md ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Zusätzliche Benchmarks und Laptop-Vergleich
2
+
3
+ Alle Einträge sind **registriert, nicht ausgeführt**. Quelle: [Qwen3.8-27B-Modellkarte](https://huggingface.co/Qwen/Qwen3.8-27B), Revision `1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0`. Veröffentlichte Zahlen sind kein direkter Vergleich zu unseren Kurztests.
4
+
5
+ | Benchmark | Veröffentlichtes Qwen3.8-Ergebnis | Lokale Kurztest-Fälle (Vorschlag) | Voraussetzung |
6
+ |---|---|---|---|
7
+ | Terminal Bench 2.1 (Terminus) | 73.0 | 5 | version_and_harness_required |
8
+ | SWE-bench Pro | 61.7 | 5 | refined_tasks_Claude_Code_256K_temp1_top_p095 |
9
+ | NL2Repo-Bench | 42.3 | 5 | Claude_Code_no_repository_download_reward_hacking |
10
+ | DeepSWE 1.1 | 42.2 | 5 | Claude_Code_256K_temp1_top_p095 |
11
+ | QwenSWEBench | 79.0 | offen – intern | in_house_avg3_8h_timeout_32768_output_256K_context |
12
+ | CoWorkBench | 70.7 | offen – intern | in_house_productivity_tasks |
13
+ | JobBench | 33.4 | 5 | dataset_and_harness_availability_unverified |
14
+ | Agents' Last Exam | Pass@1 20.4; Score 42.9 | 5 | two_distinct_metrics |
15
+ | IFBench | 79.5 | 20 | not_IFEval |
16
+ | GPQA Diamond | 89.2 | 20 | existing_5_prompt_diagnostic_not_full_benchmark |
17
+ | HLE | 30.8 | 10 | publisher_GPT4o_judge_separate_judge_budget |
18
+ | LiveCodeBench v6 | 90.3 | 10 | existing_LiveCodeBench_revision_must_be_reconciled |
19
+ | OSWorld-Verified | 84.3 | 5 | real_actions_not_static_screenshot_qa |
20
+ | WebArena-Verified | 64.8 | 5 | official_grader_OSWorld_scaffold |
21
+ | AndroidWorld | 81.9 | 5 | emulator_and_state_reset_required |
22
+ | RecreationBench | 47.1 | offen – intern | in_house_Ubuntu_macOS_Windows_Android_web |
23
+ | ClawEval-MM | Pass@3 57.4; Average 56.9 | 5 | official_pass3_requires_three_trials; local_single_trial_not_pass3 |
24
+ | SWE-MM | 38.6 | 5 | public_dev_split_with_Opus47_appendix8_3_modifications |
25
+ | Vision2Web | 62.9 | 5 | Claude_Code_judged_gpt5.4_2026_03_05 |
26
+ | MathVision | Without CI 90.0; With CI 94.6 | 10 | corrected_annotations_boxed_final_CI_strata_separate |
27
+ | BabyVision | Without CI 65.7; With CI 85.6 | 10 | CI_strata_separate |
28
+ | CharXiv (RQ) | Without CI 83.7; With CI 90.2 | 10 | corrected_annotations_CI_strata_separate |
29
+ | OmniDocBench 1.5 | 91.1 | 10 | official_metric_not_generic_accuracy |
30
+ | RealWorldQA | 85.9 | 20 | dataset_revision_and_image_preprocessing_lock_required |
31
+ | ERQA | 65.5 | 10 | dataset_revision_and_official_scoring_required |
32
+
33
+ ## Vergleichsregeln
34
+
35
+ IFBench ist nicht IFEval, MathVision nicht MathVista und CharXiv nicht ChartQA. Screenshot-Fragen ersetzen keinen OSWorld-/WebArena-Lauf. Die Qwen-Karte nutzt teils korrigierte Referenzen, andere Harnesses, 256K-Kontext, drei Versuche oder externe Richter. Solange diese Voraussetzungen nicht identisch sind, stehen Publisherwerte in einer getrennten Referenzspalte. Bei unbekannter Verfügbarkeit wird kein offizieller Adapter behauptet.
36
+
37
+ ## Laptop
38
+
39
+ 4 Quants × 2 Backends × 2 Runtime-Profile = **16 Q36-Konfigurationen**. Mit beiden Baselines wären es bis zu 48, aber nur wenn dieselben Quants verfügbar und mit dem Laptop kompatibel sind. Modellidentität und Runtime-Optimierung sind getrennte Achsen. Zunächst kleine Smoke-/Speichertests, dann eine feste Qualitätsstichprobe; Leistung dreimal warm messen. Die große Agentenmatrix ist nicht automatisch ein kostenloser Kurzlauf.
LAPTOP-BENCHMARK-MATRIX.json ADDED
@@ -0,0 +1,57 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": "q36.laptop-matrix.v1",
3
+ "status": "planned_not_executed",
4
+ "quants": [
5
+ "IQ4_XS",
6
+ "Q3_K_M",
7
+ "IQ2_M",
8
+ "IQ1_M"
9
+ ],
10
+ "backends": [
11
+ "llama.cpp",
12
+ "Ollama"
13
+ ],
14
+ "runtime_profiles": [
15
+ "stock",
16
+ "optimized"
17
+ ],
18
+ "primary_model": "Q36-v2 (same weights formerly released as v1.3)",
19
+ "baseline_models": [
20
+ "Huihui original",
21
+ "Qwen/Qwen3.8-27B"
22
+ ],
23
+ "primary_cells": 16,
24
+ "potential_all_model_cells": 48,
25
+ "baseline_quant_availability": "not_verified_do_not_substitute_different_quants_silently",
26
+ "gpu_jobs_authorized": false,
27
+ "downloads_started": false,
28
+ "profiles": {
29
+ "stock": "Backend defaults explicitly recorded, with same context, prompt, output and time budgets enforced. Not an undocumented historical server setup.",
30
+ "optimized": "Tune CPU threads, batch sizes, GPU offload and supported KV cache/flash-attention on a disjoint tuning set; freeze per device/quant/backend before test. Never optimize on scored prompts."
31
+ },
32
+ "common_controls": [
33
+ "Exact weights SHA256 and projector SHA256",
34
+ "One immutable prompt/image manifest and renderer-specific chat-template capture",
35
+ "Separate non-thinking controlled track from publisher-recommended thinking track",
36
+ "Same sampling, output cap, context, per-request deadline across comparable cells",
37
+ "One warmup excluded; three text throughput repetitions; quality one trial except separately named pass@3 tests",
38
+ "AC power, power mode, driver, runtime versions, thermals and throttling logged",
39
+ "Randomized balanced cell order; cooldown after sustained thermal throttling",
40
+ "No concurrent workloads; restart server between cells; clear/reuse caches explicitly",
41
+ "Keep runtime profile independent from model identity (stock model vs stock runtime are different axes)"
42
+ ],
43
+ "metrics": [
44
+ "prompt eval tokens/s",
45
+ "decode tokens/s",
46
+ "end-to-end tokens/s for matched IDs and length bins",
47
+ "first visible output seconds (client) and TTFT if native provided",
48
+ "request p50/p95 and total wall time",
49
+ "prompt/completion token counts and missing usage",
50
+ "process CPU percent 100%=one logical core",
51
+ "process RSS, private/anonymous bytes if available, system commit, page faults, swap",
52
+ "VRAM, GPU utilization, power, temperature, thermal limits",
53
+ "content score, protocol score, refusal review, timeout/truncation/loop flags"
54
+ ],
55
+ "memory_policy": "Fail fast on insufficient memory; classify not runnable/OOM, not wrong answer. No model file size presented as total RAM need.",
56
+ "cost_policy": "Preparation/API metadata/local analysis only; no inference provider, remote jobs or paid judge calls without separate authorization."
57
+ }
README.md CHANGED
@@ -1,109 +1,130 @@
1
- ---
2
- license: apache-2.0
3
- library_name: transformers
4
- pipeline_tag: image-text-to-text
5
- base_model: oktayd/Q36-35B-A3B-Opus4.7-Ablit-Heretic-OBLITERATUS-Hermes-MTP-Vision-FT
6
- tags:
7
- - qwen3_5_moe
8
- - vision
9
- - image-text-to-text
10
- - moe
11
- - bf16
12
- ---
13
-
14
- # Q36 v1.3 35B-A3B MTP Vision, BF16
15
-
16
- Standalone BF16 checkpoint after the completed Q36 three-run Soup/PEFT training
17
- sequence. The final LoRA is merged into the source-locked Q36 base. This repository
18
- contains the weights, tokenizer, processor, chat template, configuration and
19
- release validation data. It does not require a separate adapter.
20
-
21
- Architecture: `Qwen3_5MoeForConditionalGeneration`, approximately 35B total
22
- parameters, 256 experts with 8 selected per token. Vision tensors and all 19 source
23
- MTP tensors are preserved. MTP decoding support depends on the inference backend;
24
- preserving these tensors is not a claim that Transformers enables speculative decoding.
25
-
26
- The three runs contain 34,000 selected record uses (6,000 + 12,000 + 16,000).
27
- The final personality stage completed 313/313 optimizer steps. Training loss is
28
- not a benchmark score. Broader quality and hardware benchmarks are pending.
29
-
30
- ## Load with Transformers
31
-
32
- Validated with Transformers 5.16.1 and PyTorch BF16 on NVIDIA H200.
33
- Install a compatible PyTorch/CUDA build, then `pip install transformers==5.16.1 accelerate pillow`.
34
-
35
- ```python
36
- import torch
37
- from transformers import AutoProcessor, Qwen3_5MoeForConditionalGeneration
38
-
39
- repo = "oktayd/Q36-v1.3-35B-A3B-MTP-Vision-BF16"
40
- processor = AutoProcessor.from_pretrained(repo)
41
- model = Qwen3_5MoeForConditionalGeneration.from_pretrained(
42
- repo, dtype=torch.bfloat16, device_map="auto"
43
- )
44
- messages = [{"role": "user", "content": [{"type": "text", "text": "Explain gravity briefly."}]}]
45
- inputs = processor.apply_chat_template(
46
- messages, tokenize=True, add_generation_prompt=True,
47
- return_dict=True, return_tensors="pt", enable_thinking=False
48
- ).to(model.device)
49
- with torch.inference_mode():
50
- output = model.generate(**inputs, max_new_tokens=256, do_sample=False)
51
- print(processor.batch_decode(output[:, inputs.input_ids.shape[1]:], skip_special_tokens=True)[0])
52
- ```
53
-
54
- For image input, include an image content item supported by `AutoProcessor`.
55
- The model is roughly 70 GB of BF16 weights; runtime memory also includes cache and activations.
56
-
57
- ## Validation and provenance
58
-
59
- `RELEASE-VALIDATION.json` records the pinned base revision, final adapter checksum,
60
- tensor coverage, smoke outputs, and exact pre-save/post-reload logit comparison.
61
- BF16 merging introduces rounding relative to unmerged LoRA inference; this
62
- difference is recorded rather than claimed to be bitwise identical.
63
- `SHA256SUMS` covers the files in this release. The model inherits the source's
64
- Apache-2.0 license. This is a research fine-tune; no claim of universal benchmark
65
- improvement or removal of memorization is made.
66
-
67
- ## Bounded inference (runtime safeguard)
68
-
69
- The release enables KV caching for inference. This corrects a training-time
70
- `use_cache: false` export setting; it does not repair learned reasoning behavior.
71
- Short diagnostic tests observed repetitive generations, particularly with thinking
72
- enabled. `enable_thinking=False` is the recommended starting point for interactive
73
- use, not a guarantee of correctness. Thinking mode remains available explicitly.
74
-
75
- The optional, inspectable `q36_runtime.py` helper supports batch size one:
76
-
77
- ```python
78
- from q36_runtime import generate_bounded # download this helper alongside your script
79
- result = generate_bounded(model, processor.tokenizer, inputs,
80
- max_new_tokens=1024, max_seconds=45)
81
- print(result['text'])
82
- print(result['completed'], result['finish_reason'])
83
- ```
84
-
85
- Build `inputs` with `enable_thinking=False` as above. The helper stops on EOS,
86
- token/time limits, or a long repeated-token-block heuristic. A limit/loop stop is
87
- reported as **incomplete**, not as a successful answer. It never executes tools
88
- or retries automatically. Legitimate repeated code/text may trigger the heuristic.
89
- Time limits are cooperative between generation steps, not a hard process deadline.
90
- The helper is opt-in: plain Transformers, GGUF and Ollama imports do not automatically
91
- execute it. These safeguards do not change weights and do not fix factual errors.
92
- Selected regression tests are documented in `RUNTIME-PATCH-VALIDATION.json`; they
93
- are not independent benchmark scores or evidence of universal improvement.
94
-
95
- ## Compact Q4_K_M diagnostic (H200)
96
-
97
- The llama.cpp **Q4_K_M** diagnostic attempted 204 available prompts:
98
- 186 completed, 18 were incomplete or errored.
99
- Of 108 completed outputs with implemented strict automatic checks,
100
- 61 passed. Another 78 completed outputs require
101
- manual review or an official evaluator. These heterogeneous counts are **not a composite
102
- model accuracy score** and are not official leaderboard benchmarks.
103
-
104
- The run used greedy decoding with thinking disabled, a 16K context, a 4,096-token
105
- output cap and per-request time limits. Repetition failures still occurred; inference
106
- limits do not fix learned reasoning errors. Only Q4_K_M on H200 was tested in this
107
- run, with no completed source-model comparison. See `BENCHMARK-DIAGNOSTIC.json`
108
- for per-task counts, configuration, timings and limitations. Raw test prompts,
109
- answers and internal logs are kept outside the public model repository.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ library_name: transformers
4
+ pipeline_tag: image-text-to-text
5
+ base_model: oktayd/Q36-35B-A3B-Opus4.7-Ablit-Heretic-OBLITERATUS-Hermes-MTP-Vision-FT
6
+ tags:
7
+ - qwen3_5_moe
8
+ - vision
9
+ - moe
10
+ - conversational
11
+ - bf16
12
+ ---
13
+
14
+ # Qwen3.6-35B v2 · MoE · Heretic / Uncensor / Hermes
15
+
16
+ **Standalone BF16 / Transformers master.**
17
+
18
+ [BF16 / FT](https://huggingface.co/oktayd/Qwen3.6-35B-v2-MoE-Ablit-Heretic-Uncensor-Hermes-MTP-Vision-FT) · [GGUF / Llama](https://huggingface.co/oktayd/Qwen3.6-35B-v2-MoE-Ablit-Heretic-Uncensor-Hermes-MTP-Vision-Llama) · [Ollama](https://huggingface.co/oktayd/Qwen3.6-35B-v2-MoE-Ablit-Heretic-Uncensor-Hermes-MTP-Vision-Ollama) · [v2 collection](https://huggingface.co/collections/oktayd/qwen36-35b-v2-moe-uncensor-hermes-editions-6a9bcf212c664f52217f7ec5)
19
+
20
+ ## What this release is
21
+
22
+ Q36 v2 is the completed three-run Soup/PEFT research fine-tune, formerly published as **Q36 v1.3**. The v2 update changes packaging, names and documentation — **not model tensors and not another training run**. Quantizations and Ollama packages derive from the same merged BF16 checkpoint.
23
+
24
+ The name follows the project's earlier naming style. `Ablit`, `Heretic`, `Uncensor` and `Hermes` describe inherited project lineage/training intent, not affiliation with those projects or a promise of unrestricted behavior. Full provenance and validation are retained, including the historical Opus4.7-labelled source. No universal superiority, guaranteed compliance or removal of memorization is claimed.
25
+
26
+ ## Architecture, capability scope and validation
27
+
28
+ - `Qwen3_5MoeForConditionalGeneration`; approximately 35B total parameters, 256 experts, 8 selected per token (the previous A3B label).
29
+ - Native vision-language architecture; separate projector required for GGUF image input. Text and elementary red/blue-image smoke tests were recorded. Video, 3D consistency and full desktop/browser agents are not certified by those tests.
30
+ - Training covered selected knowledge, instruction/agent, coding, personality and visual data: **34,000 record uses** across 6,000 + 12,000 + 16,000. Record uses are not unique records. Training loss is not benchmark accuracy.
31
+ - All 19 source MTP tensors were preserved. **MTP/speculative acceleration is not enabled or validated by these benchmark results.** Backend support must be tested separately.
32
+ - Tool-call formatting and end-of-answer behavior are diagnostic targets, not guaranteed features. Known schema mistakes, factual errors and repetition remain.
33
+
34
+ ## Load with Transformers
35
+
36
+ Validated with Transformers 5.16.1 and PyTorch BF16 on NVIDIA H200.
37
+ Install a compatible PyTorch/CUDA build, then `pip install transformers==5.16.1 accelerate pillow`.
38
+
39
+ ```python
40
+ import torch
41
+ from transformers import AutoProcessor, Qwen3_5MoeForConditionalGeneration
42
+
43
+ repo = "oktayd/Qwen3.6-35B-v2-MoE-Ablit-Heretic-Uncensor-Hermes-MTP-Vision-FT"
44
+ processor = AutoProcessor.from_pretrained(repo)
45
+ model = Qwen3_5MoeForConditionalGeneration.from_pretrained(
46
+ repo, dtype=torch.bfloat16, device_map="auto"
47
+ )
48
+ messages = [{"role": "user", "content": [{"type": "text", "text": "Explain gravity briefly."}]}]
49
+ inputs = processor.apply_chat_template(
50
+ messages, tokenize=True, add_generation_prompt=True,
51
+ return_dict=True, return_tensors="pt", enable_thinking=False
52
+ ).to(model.device)
53
+ with torch.inference_mode():
54
+ output = model.generate(**inputs, max_new_tokens=256, do_sample=False)
55
+ print(processor.batch_decode(output[:, inputs.input_ids.shape[1]:], skip_special_tokens=True)[0])
56
+ ```
57
+
58
+ For image input, include an image content item supported by `AutoProcessor`.
59
+ The model is roughly 70 GB of BF16 weights; runtime memory also includes cache and activations.
60
+
61
+ ## Validation and provenance
62
+
63
+ `RELEASE-VALIDATION.json` records the pinned base revision, final adapter checksum,
64
+ tensor coverage, smoke outputs, and exact pre-save/post-reload logit comparison.
65
+ BF16 merging introduces rounding relative to unmerged LoRA inference; this
66
+ difference is recorded rather than claimed to be bitwise identical.
67
+ `SHA256SUMS` covers the files in this release. The model inherits the source's
68
+ Apache-2.0 license. This is a research fine-tune; no claim of universal benchmark
69
+ improvement or removal of memorization is made.
70
+
71
+ ## Bounded inference (runtime safeguard)
72
+
73
+ The release enables KV caching for inference. This corrects a training-time
74
+ `use_cache: false` export setting; it does not repair learned reasoning behavior.
75
+ Short diagnostic tests observed repetitive generations, particularly with thinking
76
+ enabled. `enable_thinking=False` is the recommended starting point for interactive
77
+ use, not a guarantee of correctness. Thinking mode remains available explicitly.
78
+
79
+ The optional, inspectable `q36_runtime.py` helper supports batch size one:
80
+
81
+ ```python
82
+ from q36_runtime import generate_bounded # download this helper alongside your script
83
+ result = generate_bounded(model, processor.tokenizer, inputs,
84
+ max_new_tokens=1024, max_seconds=45)
85
+ print(result['text'])
86
+ print(result['completed'], result['finish_reason'])
87
+ ```
88
+
89
+ Build `inputs` with `enable_thinking=False` as above. The helper stops on EOS,
90
+ token/time limits, or a long repeated-token-block heuristic. A limit/loop stop is
91
+ reported as **incomplete**, not as a successful answer. It never executes tools
92
+ or retries automatically. Legitimate repeated code/text may trigger the heuristic.
93
+ Time limits are cooperative between generation steps, not a hard process deadline.
94
+ The helper is opt-in: plain Transformers, GGUF and Ollama imports do not automatically
95
+ execute it. These safeguards do not change weights and do not fix factual errors.
96
+ Selected regression tests are documented in `RUNTIME-PATCH-VALIDATION.json`; they
97
+ are not independent benchmark scores or evidence of universal improvement.
98
+
99
+
100
+ ## Files to use
101
+
102
+ Standard `model-*.safetensors`, `model-mtp.safetensors` and the index load together with tokenizer, processor, chat template and configs. Keep standard filenames unchanged. About 70 GB of weights plus runtime overhead; use a quantized edition for constrained memory. `q36_runtime.py` is optional and batch-size-one only, not an automatic GGUF/Ollama patch.
103
+
104
+ ## Support files and known limitations
105
+
106
+ [SHA256SUMS](SHA256SUMS) verifies release files. [RELEASE-NAMING.json](RELEASE-NAMING.json) maps old repository/file names to v2 and records unchanged weight hashes. Historical validation reports intentionally retain their original runtime names; the map resolves those names. [RELEASE-VALIDATION.json](RELEASE-VALIDATION.json) covers tensor/merge/provenance checks; [RUNTIME-PATCH-VALIDATION.json](RUNTIME-PATCH-VALIDATION.json) covers selected guard tests.
107
+
108
+ Thinking mode can loop; start with thinking off and bounded output. Guard stops are incomplete answers, not successful corrections. The private training inputs, installer ZIP, raw benchmark prompts/answers and internal debug logs are not part of these end-user repositories. Keep use within applicable rights and deployment requirements.
109
+
110
+ ## Measured diagnostics — scope matters
111
+
112
+ These are local **Q4_K_M / llama.cpp diagnostics**, not full official benchmark scores and not Ollama performance claims. Temperature 0, seed 42, thinking off, 16,384 context, 4,096 output cap, 45-second per-request budget. The RTX 2000 Ada used requested 24 GPU layers plus CPU offload; the actual offload-layer log was unavailable.
113
+
114
+ | Model / device | Attempted / 204 | Completed | Old strict pass / scored | End-to-end output tok/s (median, outputs ≥64 tokens) |
115
+ |---|---:|---:|---:|---:|
116
+ | Q36 / H200 | 204 | 186 | 61 / 108 | 156.4 |
117
+ | Q36 / RTX 5090 | 204 | 190 | 62 / 109 | 174.6 |
118
+ | Huihui / RTX 5090 | 204 | 196 | 87 / 114 | 203.8 |
119
+ | Q36 / RTX 2000 Ada (CPU+GPU) | 204 | 194 | 61 / 109 | 18.2 |
120
+ | Huihui / RTX 2000 Ada (CPU+GPU) | 182 | 169 | 76 / 99 | 19.5 |
121
+
122
+ These strict counts omit pending official/manual evaluators, use changing denominators, and sometimes reject semantically correct formatting variants. **Do not divide passes by all prompts or call these counts overall accuracy.** A separate local regrade keeps content, protocol compliance and delivery apart. It does not certify unreviewed reasoning or execute generated code. The Huihui RTX 2000 Ada run left 22 tasks untested at the global deadline. Earlier BF16 diagnostics used thinking and are not directly comparable.
123
+
124
+ Huihui performed better on the shared automatically assessable RTX 5090 subset; no claim is made that this fine-tune universally surpasses its source or Qwen3.8. Rates mix generated lengths and are not pure hardware speedups. Native decode, prefill, client first-output and resource metrics are separated in [BENCHMARK-DEVICE-SUMMARY.json](BENCHMARK-DEVICE-SUMMARY.json). Raw prompts/answers and private training data are not published here.
125
+
126
+ [Qwen3.8 comparison plan](BENCHMARK-PLAN.md) registers every benchmark family from the publisher card, including internal/unavailable tasks. **Qwen3.8 has not been tested locally.** Publisher scores use different harnesses, settings, annotations and trial counts and are shown only as references. No GPU job is launched by these support files.
127
+
128
+ ## Related collection
129
+
130
+ The previous release remains separate: [Qwen3.6 Opus4.7 Heretic Hermes Agent — Editions](https://huggingface.co/collections/oktayd/qwen36-opus47-heretic-hermes-agent-editions-6a8d09f6c2eb42ed1b112184).
RELEASE-NAMING.json ADDED
@@ -0,0 +1,97 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "old_repo": "oktayd/Q36-v1.3-35B-A3B-MTP-Vision-BF16",
3
+ "new_repo": "oktayd/Qwen3.6-35B-v2-MoE-Ablit-Heretic-Uncensor-Hermes-MTP-Vision-FT",
4
+ "old_revision": "61cb76f35c41e8f800e8b4e9fe9e36f4571580ae",
5
+ "release_display_version": "v2",
6
+ "tensor_changes": false,
7
+ "weights_retrained": false,
8
+ "file_renames": {},
9
+ "historical_reports": "Immutable original diagnostic names retained; resolve via this mapping.",
10
+ "weight_hashes": {
11
+ "model-00001-of-00016.safetensors": {
12
+ "size": 4323955448,
13
+ "sha256": "41d2b50506fcdd2a7dc0ff7a055021f23abe2bb57965da3e281c9e0795aa5587",
14
+ "pointer_size": 135
15
+ },
16
+ "model-00002-of-00016.safetensors": {
17
+ "size": 4506431768,
18
+ "sha256": "4d48740f1c9cdc95cc02eeb051c8795b56a7c04fbb55e1bea4a14662ba07ab1c",
19
+ "pointer_size": 135
20
+ },
21
+ "model-00003-of-00016.safetensors": {
22
+ "size": 4988775056,
23
+ "sha256": "64b74183b13f634dfeaba4c131f07b6fe548a85ffe4324ff4ed8a596f31b57d9",
24
+ "pointer_size": 135
25
+ },
26
+ "model-00004-of-00016.safetensors": {
27
+ "size": 3962207584,
28
+ "sha256": "a014b881f5b7e883243ecb7d8eb8321334117816a725ce2389f6b0bedc56225d",
29
+ "pointer_size": 135
30
+ },
31
+ "model-00005-of-00016.safetensors": {
32
+ "size": 4506431736,
33
+ "sha256": "824a2b762a92a88543a6188d789b616ca78cd7bcfeccfbfa91d3e79a1cf42678",
34
+ "pointer_size": 135
35
+ },
36
+ "model-00006-of-00016.safetensors": {
37
+ "size": 4988775104,
38
+ "sha256": "9e592d029623ddafb2529e3c1c822b475bb5eb007d630e71956e5c09378727d1",
39
+ "pointer_size": 135
40
+ },
41
+ "model-00007-of-00016.safetensors": {
42
+ "size": 3962207632,
43
+ "sha256": "81d3392b22d5f0eb3c13ef62610636285a228e324ddaa26c9936907c4f7a0bae",
44
+ "pointer_size": 135
45
+ },
46
+ "model-00008-of-00016.safetensors": {
47
+ "size": 4506431816,
48
+ "sha256": "810c547161ac480b1cfb5b31b87e4566bf302884ba76af63cb55461b02af217d",
49
+ "pointer_size": 135
50
+ },
51
+ "model-00009-of-00016.safetensors": {
52
+ "size": 4988775104,
53
+ "sha256": "4f52ae41acdec8ecb85adf77cb6271b6bcdc5e2231fa61c32930e97d5a277fd2",
54
+ "pointer_size": 135
55
+ },
56
+ "model-00010-of-00016.safetensors": {
57
+ "size": 3962207632,
58
+ "sha256": "d7387d92fed4e1cc870f2d074ca161ad53d3b54d60f355f55c2767ce43ea3ef8",
59
+ "pointer_size": 135
60
+ },
61
+ "model-00011-of-00016.safetensors": {
62
+ "size": 4506431816,
63
+ "sha256": "9f9a4f448cfe365ae1b410a008b1a8508debf0261f72f78c02eb923bc82651d4",
64
+ "pointer_size": 135
65
+ },
66
+ "model-00012-of-00016.safetensors": {
67
+ "size": 4988775104,
68
+ "sha256": "1cdc8b5f614150ab62d74fa3d7a240c66349ac7a26fb20af7d6f3bb2057a4d51",
69
+ "pointer_size": 135
70
+ },
71
+ "model-00013-of-00016.safetensors": {
72
+ "size": 3962207632,
73
+ "sha256": "bdedbb50762ae1e9f040d7d8fee07faec1fa56e1005123ec3e320301ada17cec",
74
+ "pointer_size": 135
75
+ },
76
+ "model-00014-of-00016.safetensors": {
77
+ "size": 4506431816,
78
+ "sha256": "23225d0ea44a7894d5bb8654d6e8944a65851fb656cad407c051b86e6fdbee90",
79
+ "pointer_size": 135
80
+ },
81
+ "model-00015-of-00016.safetensors": {
82
+ "size": 4988775104,
83
+ "sha256": "301bf7359901f07482f76541259478553dc9dcc237c06863e74c8429c567fbd0",
84
+ "pointer_size": 135
85
+ },
86
+ "model-00016-of-00016.safetensors": {
87
+ "size": 2565674304,
88
+ "sha256": "0414d33074013180c4e41b2adc2aa6a9a920bce33b3869c94cc26a4b865798d8",
89
+ "pointer_size": 135
90
+ },
91
+ "model-mtp.safetensors": {
92
+ "size": 1689283688,
93
+ "sha256": "62e6f4b910fd5ff7594e85863fe28cfd0177890c8251655ee9bcbc35a208d78b",
94
+ "pointer_size": 135
95
+ }
96
+ }
97
+ }
SHA256SUMS CHANGED
@@ -1,6 +1,11 @@
 
1
  807d54b88902798393fc104c3ca80d3e424d64270e8d1e1a34a44d4574f52c7e BENCHMARK-DIAGNOSTIC.json
 
 
 
2
  20a2a90fa761fe5081d31d25989d94656ad7b9766103b4244b71891e714dcc22 LICENSE
3
- 57f8ca092a5579c5b41fb94dc540a7eafdefb746128fa30752f82272e10716e9 README.md
 
4
  b36b05ac4d31eb371f841af813f3b3fdfad5c04375c58214013a94dd929c2ad0 RELEASE-VALIDATION.json
5
  3836329a27f1f7c4320c8aef72fd617f8ab4440e88e7419ed5cf785408c48cb2 RUNTIME-PATCH-VALIDATION.json
6
  55d4931433fe502b794226ee7f4d206a6bdd436ac9f80eb7d8ebb4c639f9ea0c chat_template.jinja
 
1
+ 1857a51f70eb7b787dd3ec5f050b2b013c5170c66c55a8f7278ddf83d94a9330 BENCHMARK-DEVICE-SUMMARY.json
2
  807d54b88902798393fc104c3ca80d3e424d64270e8d1e1a34a44d4574f52c7e BENCHMARK-DIAGNOSTIC.json
3
+ 0e712dd8e4bc65088e170826e98bdc440fb19a2cfdb7613c643e9dabb624ecd8 BENCHMARK-EXTENSION-QWEN38.json
4
+ 67d070e3b12bcd185626054c81cdf4d528a9a1af26774050e1d50be370b5ee14 BENCHMARK-PLAN.md
5
+ 5fa5bdc4687e5ed723af58eb436f545cc62589e5395d5041722639f03bb5ec9b LAPTOP-BENCHMARK-MATRIX.json
6
  20a2a90fa761fe5081d31d25989d94656ad7b9766103b4244b71891e714dcc22 LICENSE
7
+ 2b3481c79e0f3c1d2e650d5628af88bed6c51e9b09c2183caa6d9a8eecf87b66 README.md
8
+ cd2b979e9e94047d3ec45a8fe6c6610765a0045be6a09e4e730e8c501405290c RELEASE-NAMING.json
9
  b36b05ac4d31eb371f841af813f3b3fdfad5c04375c58214013a94dd929c2ad0 RELEASE-VALIDATION.json
10
  3836329a27f1f7c4320c8aef72fd617f8ab4440e88e7419ed5cf785408c48cb2 RUNTIME-PATCH-VALIDATION.json
11
  55d4931433fe502b794226ee7f4d206a6bdd436ac9f80eb7d8ebb4c639f9ea0c chat_template.jinja