{
"format": "jang-v2",
"model": "dots3-note-prev",
"source": "/Users/eric/models/dots-studio/dots3-note-prev-fp8",
"plan": {
"name": "dots3-note-JANG2D-95GiB-v2",
"note": "REBUILD v2 after the gather_mm corruption fix. Measured on CLEAN 205k-token capture. Grid: per-unit best of imatrix vs min-max, GPTQ codes on the winner.",
"defaults": {
"routed": {
"bits": 2,
"group_size": 64
},
"attention": {
"bits": 8,
"group_size": 64
},
"shared_expert": {
"bits": 8,
"group_size": 64
},
"dense_mlp": {
"bits": 8,
"group_size": 64
},
"bookend": {
"bits": 8,
"group_size": 64
},
"mtp_linear": {
"bits": 8,
"group_size": 64
},
"vision_linear": {
"bits": 6,
"group_size": 64
},
"vision_expert": {
"bits": 4,
"group_size": 64
},
"audio_linear": {
"bits": 6,
"group_size": 64
}
},
"routed_overrides": {
"26:down_proj": {
"bits": 3,
"group_size": 64
},
"31:down_proj": {
"bits": 3,
"group_size": 64
},
"37:down_proj": {
"bits": 3,
"group_size": 64
},
"38:down_proj": {
"bits": 3,
"group_size": 64
},
"39:down_proj": {
"bits": 3,
"group_size": 64
},
"40:down_proj": {
"bits": 3,
"group_size": 64
},
"42:down_proj": {
"bits": 3,
"group_size": 64
},
"43:down_proj": {
"bits": 3,
"group_size": 64
},
"44:down_proj": {
"bits": 3,
"group_size": 64
}
},
"provenance": {
"unit_scores": "plans/unit_scores.json",
"solver": "plans/solved-87.3-g64only.json",
"vision_ab": "plans/vision_ab.json",
"folds": "~/models/dots3-captures/folds.npz (awq alpha .25 clip [.5,2] + per-expert diag imatrix)",
"capture": "~/models/dots3-captures (104935 calib tokens, 2026-04-23 mix + dots-XML agentic)"
}
},
"mtp_embed_shared": false,
"converted_bytes": 101538100224,
"qat": {
"method": "gptq_error_compensated_codes",
"grid": "per_unit_best_of{imatrix_activation_weighted, minmax}_f16",
"sequencing": "brecq_w1w3_then_w2",
"units_replaced": 135,
"codes_dir": "/Users/eric/models/dots3-gptq-codes",
"applied": "2026-08-16 05:00:21"
},
"chat": {
"sampling_defaults": {
"temperature": 1.0,
"top_p": 0.95,
"top_k": 0,
"min_p": 0.0,
"presence_penalty": 0.0,
"repetition_penalty": 1.0,
"source": "vendor_readme_2026-08-14",
"mode": "thinking_general"
},
"sampling_modes": {
"thinking_general": {
"temperature": 1.0,
"top_p": 0.95,
"top_k": 0,
"min_p": 0.0,
"presence_penalty": 0.0,
"repetition_penalty": 1.0,
"source": "vendor_readme_2026-08-14"
},
"agentic_coding": {
"temperature": 0.6,
"top_p": 0.95,
"top_k": 0,
"min_p": 0.0,
"presence_penalty": 0.0,
"repetition_penalty": 1.0,
"source": "jang_default_coding_preset (DSV4-0731/Qwen36-27B precedent; vendor publishes none; vLLM example 0.7)"
},
"instruct_nothinking": {
"temperature": 0.7,
"top_p": 0.95,
"top_k": 0,
"min_p": 0.0,
"presence_penalty": 0.0,
"repetition_penalty": 1.0,
"source": "vllm_recipe_example_2026-08-14"
}
},
"stop_token_ids": [
151643,
151668
]
},
"reasoning": {
"supported": true,
"parser": "dots3",
"default": "on",
"think_in_template": true,
"enable_kwarg": "enable_thinking",
"off_is_prefilled_closed_block": true,
"off_prefill": "\n\n\n\n",
"off_user_marker": "",
"tiers": null,
"note": "Thinking ON by default: chat_template sets enable_thinking=true when undefined. Disabling appends to the user turn AND prefills a CLOSED think block \u2014 it is not an omission."
},
"tools": {
"supported": true,
"parser": "dots",
"dialect": "dots_xml_function_call",
"call_open": "",
"response_wrap": "",
"mlx_lm_autodetected": false,
"detection_hint": "template literal '' \u2014 no mlx_lm parser exists for this dialect yet"
},
"vision": {
"supported": true,
"video_supported": true,
"tower": "dots3 MoE-ViT (42 blocks, 608 pyramid experts, 1.2B act)",
"processor": "preprocessor_config.json",
"video_processor": "video preprocessing via dots3_note processor (config-embedded; no separate file upstream)"
},
"audio": {
"supported": true,
"tower": "dots3 speech encoder (32 layers, swiglu whisper-shape, conv2d stem, 800M)",
"config": "config.json audio_config (no separate processor file)",
"note": "video inputs include their audio track when available"
},
"drop_mtp": false,
"runtime": {
"bundle_has_mtp": true,
"mtp_layers": 1,
"mtp_mode": "preserved_enabled",
"mtp_num_speculative_tokens": 1,
"mtp_status": "MTP layer 46 preserved for native speculative decode; recommended 1 draft/step on Apple silicon (unmeasured on this artifact \u2014 run a depth sweep to validate)."
},
"mtp": {
"num_layers": 1,
"artifact_available": true,
"tensor_count": 41,
"runtime_available": false,
"dedicated_embeddings": true,
"embed_shared_with_backbone": false,
"layout": "dsv3_fusion_at_layer_46 (eh_proj/enorm/hnorm/shared_head.norm + MLA + dense FFN)",
"upstream_method": "nextn/mtp (SGLang NEXTN, vLLM mtp)",
"upstream_num_speculative_tokens": 3,
"recommended_num_drafts": 1,
"notation": "Draft count is the unambiguous unit. vmlx native-MTP 'depth' and vLLM num_speculative_tokens both count DRAFTS; kit docs' D counts tokens/cycle (kit D2 == 1 draft == vmlx depth 1)."
},
"capabilities": {
"has_vision": true,
"has_video": true,
"has_audio": true,
"modality": "omni",
"modalities": {
"text": true,
"vision": true,
"video": true,
"audio": true
},
"supports_tools": true,
"tool_parser": "dots",
"supports_thinking": true,
"default_reasoning": "on",
"think_in_template": true,
"reasoning_parser": "dots3",
"family": "dots3_note",
"cache_type": "hybrid_full_swa",
"context_length": 524288,
"dsa_index_topk": 2048
}
}