AutomatosX commited on
Commit
3555477
·
verified ·
1 Parent(s): f93bb79

Record architecture-specific runtime format audit and n-gram scope

Browse files
Files changed (2) hide show
  1. README.md +11 -0
  2. runtime_audit.json +43 -0
README.md CHANGED
@@ -15,6 +15,17 @@ tags:
15
  - development-preview
16
  ---
17
 
 
 
 
 
 
 
 
 
 
 
 
18
  # AX-Qwen3-Embedding-8B-CUDA-AXQ-NVFP4-W4A4
19
 
20
  **AXQuant CUDA NVFP4 W4A4 mixed precision development preview.** Converted
 
15
  - development-preview
16
  ---
17
 
18
+ ## Runtime format audit (2026-10-06)
19
+
20
+ No quantization-container correction was needed.
21
+ This family has no n-gram tensors; no n-gram file or declaration was added.
22
+ This CUDA pack is outside MLX/oMLX/MTPLX export scope.
23
+
24
+ See [runtime_audit.json](runtime_audit.json) for pinned config/index/header
25
+ bindings, architecture, physical-format findings, and applied corrections.
26
+ This is development evidence; no quality, MTP exactness, speed, or certification
27
+ claim is added. Historical evidence stays bound to its original revision.
28
+
29
  # AX-Qwen3-Embedding-8B-CUDA-AXQ-NVFP4-W4A4
30
 
31
  **AXQuant CUDA NVFP4 W4A4 mixed precision development preview.** Converted
runtime_audit.json ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "applied_config_corrections": [],
3
+ "current_config_sha256": "c0c2c43724dbcc6dc6d5ffee0ecdd8b2caf787dd9dcbaee1c99077cf0b6cfe27",
4
+ "date": "2026-10-06",
5
+ "header_sha256": {
6
+ "model-00001-of-00004.safetensors": "32411907ffceb53ab5721f8c6f5ad77cc3a7a2c56cb3053e211f3d8a7216adac",
7
+ "model-00002-of-00004.safetensors": "4ccc539c50f8bfa4a554dd80dcd649292113b0e11bc03d827a5b58a408fdbb9b",
8
+ "model-00003-of-00004.safetensors": "6fc5f484ac650ae94f03aacfee96a21b96bcd19722fc2f96f0aefbfe2f51d83b",
9
+ "model-00004-of-00004.safetensors": "d9fd30277814f33e80e08bab931af73f15424f6e71ab242f27e5e1783ac19e80"
10
+ },
11
+ "input_bindings": {
12
+ "config.json": "c0c2c43724dbcc6dc6d5ffee0ecdd8b2caf787dd9dcbaee1c99077cf0b6cfe27",
13
+ "model.safetensors.index.json": "28b535b0551125ad1567b2e95b8084829497f7094329e20332dc48d2c1bdf206"
14
+ },
15
+ "issues": [],
16
+ "model_type": "qwen3",
17
+ "mtp_files": [],
18
+ "ngram_action": "none; do not invent n-gram data",
19
+ "ngram_files": [],
20
+ "ngram_quantization": [],
21
+ "ngram_table_metadata": null,
22
+ "ngram_tensor_count": 0,
23
+ "quality_certified": false,
24
+ "quantization": {
25
+ "quantization_config": {
26
+ "quant_method": "compressed-tensors"
27
+ }
28
+ },
29
+ "removed_source_evidence": [],
30
+ "repo_id": "AutomatosX/AX-Qwen3-Embedding-8B-CUDA-AXQ-NVFP4-W4A4",
31
+ "runtime_arch_id": null,
32
+ "runtime_verified": false,
33
+ "schema_version": "axquant.hub-runtime-audit.v1",
34
+ "scope": "Pinned remote config/index/Safetensors header audit; no runtime load or generation claim.",
35
+ "source_revision": "f93bb79d3b6f308750e97e129176a56622af519f",
36
+ "status": "outside-mlx-peer-scope",
37
+ "unchanged_weight_sha256": {
38
+ "model-00001-of-00004.safetensors": "95e7215c3735a750b6b7dba2176eab2c6b09083d9db7d87b3cc6602499d40471",
39
+ "model-00002-of-00004.safetensors": "5bb0706d5abe7aa61e5020dc3ccd1bca9064fd911eb1e38d1a400701f54cdadb",
40
+ "model-00003-of-00004.safetensors": "223ddc84244e0f70af57389fa25d41bf8fb9a4aded4a2512aeb27807862384a1",
41
+ "model-00004-of-00004.safetensors": "36cbc9c60375693629f25743c1e77ebb1724af58e671b2376463193c7fd21ef6"
42
+ }
43
+ }