{ "format": "jang-v2", "model": "dots3-note-prev", "source": "/Users/eric/models/dots-studio/dots3-note-prev-fp8", "plan": { "name": "dots3-note-JANG2D-95GiB-v2", "note": "REBUILD v2 after the gather_mm corruption fix. Measured on CLEAN 205k-token capture. Grid: per-unit best of imatrix vs min-max, GPTQ codes on the winner.", "defaults": { "routed": { "bits": 2, "group_size": 64 }, "attention": { "bits": 8, "group_size": 64 }, "shared_expert": { "bits": 8, "group_size": 64 }, "dense_mlp": { "bits": 8, "group_size": 64 }, "bookend": { "bits": 8, "group_size": 64 }, "mtp_linear": { "bits": 8, "group_size": 64 }, "vision_linear": { "bits": 6, "group_size": 64 }, "vision_expert": { "bits": 4, "group_size": 64 }, "audio_linear": { "bits": 6, "group_size": 64 } }, "routed_overrides": { "26:down_proj": { "bits": 3, "group_size": 64 }, "31:down_proj": { "bits": 3, "group_size": 64 }, "37:down_proj": { "bits": 3, "group_size": 64 }, "38:down_proj": { "bits": 3, "group_size": 64 }, "39:down_proj": { "bits": 3, "group_size": 64 }, "40:down_proj": { "bits": 3, "group_size": 64 }, "42:down_proj": { "bits": 3, "group_size": 64 }, "43:down_proj": { "bits": 3, "group_size": 64 }, "44:down_proj": { "bits": 3, "group_size": 64 } }, "provenance": { "unit_scores": "plans/unit_scores.json", "solver": "plans/solved-87.3-g64only.json", "vision_ab": "plans/vision_ab.json", "folds": "~/models/dots3-captures/folds.npz (awq alpha .25 clip [.5,2] + per-expert diag imatrix)", "capture": "~/models/dots3-captures (104935 calib tokens, 2026-04-23 mix + dots-XML agentic)" } }, "mtp_embed_shared": false, "converted_bytes": 101538100224, "qat": { "method": "gptq_error_compensated_codes", "grid": "per_unit_best_of{imatrix_activation_weighted, minmax}_f16", "sequencing": "brecq_w1w3_then_w2", "units_replaced": 135, "codes_dir": "/Users/eric/models/dots3-gptq-codes", "applied": "2026-08-16 05:00:21" }, "chat": { "sampling_defaults": { "temperature": 1.0, "top_p": 0.95, "top_k": 0, "min_p": 0.0, "presence_penalty": 0.0, "repetition_penalty": 1.0, "source": "vendor_readme_2026-08-14", "mode": "thinking_general" }, "sampling_modes": { "thinking_general": { "temperature": 1.0, "top_p": 0.95, "top_k": 0, "min_p": 0.0, "presence_penalty": 0.0, "repetition_penalty": 1.0, "source": "vendor_readme_2026-08-14" }, "agentic_coding": { "temperature": 0.6, "top_p": 0.95, "top_k": 0, "min_p": 0.0, "presence_penalty": 0.0, "repetition_penalty": 1.0, "source": "jang_default_coding_preset (DSV4-0731/Qwen36-27B precedent; vendor publishes none; vLLM example 0.7)" }, "instruct_nothinking": { "temperature": 0.7, "top_p": 0.95, "top_k": 0, "min_p": 0.0, "presence_penalty": 0.0, "repetition_penalty": 1.0, "source": "vllm_recipe_example_2026-08-14" } }, "stop_token_ids": [ 151643, 151668 ] }, "reasoning": { "supported": true, "parser": "dots3", "default": "on", "think_in_template": true, "enable_kwarg": "enable_thinking", "off_is_prefilled_closed_block": true, "off_prefill": "\n\n\n\n", "off_user_marker": "", "tiers": null, "note": "Thinking ON by default: chat_template sets enable_thinking=true when undefined. Disabling appends to the user turn AND prefills a CLOSED think block \u2014 it is not an omission." }, "tools": { "supported": true, "parser": "dots", "dialect": "dots_xml_function_call", "call_open": "", "response_wrap": "", "mlx_lm_autodetected": false, "detection_hint": "template literal '' \u2014 no mlx_lm parser exists for this dialect yet" }, "vision": { "supported": true, "video_supported": true, "tower": "dots3 MoE-ViT (42 blocks, 608 pyramid experts, 1.2B act)", "processor": "preprocessor_config.json", "video_processor": "video preprocessing via dots3_note processor (config-embedded; no separate file upstream)" }, "audio": { "supported": true, "tower": "dots3 speech encoder (32 layers, swiglu whisper-shape, conv2d stem, 800M)", "config": "config.json audio_config (no separate processor file)", "note": "video inputs include their audio track when available" }, "drop_mtp": false, "runtime": { "bundle_has_mtp": true, "mtp_layers": 1, "mtp_mode": "preserved_enabled", "mtp_num_speculative_tokens": 1, "mtp_status": "MTP layer 46 preserved for native speculative decode; recommended 1 draft/step on Apple silicon (unmeasured on this artifact \u2014 run a depth sweep to validate)." }, "mtp": { "num_layers": 1, "artifact_available": true, "tensor_count": 41, "runtime_available": false, "dedicated_embeddings": true, "embed_shared_with_backbone": false, "layout": "dsv3_fusion_at_layer_46 (eh_proj/enorm/hnorm/shared_head.norm + MLA + dense FFN)", "upstream_method": "nextn/mtp (SGLang NEXTN, vLLM mtp)", "upstream_num_speculative_tokens": 3, "recommended_num_drafts": 1, "notation": "Draft count is the unambiguous unit. vmlx native-MTP 'depth' and vLLM num_speculative_tokens both count DRAFTS; kit docs' D counts tokens/cycle (kit D2 == 1 draft == vmlx depth 1)." }, "capabilities": { "has_vision": true, "has_video": true, "has_audio": true, "modality": "omni", "modalities": { "text": true, "vision": true, "video": true, "audio": true }, "supports_tools": true, "tool_parser": "dots", "supports_thinking": true, "default_reasoning": "on", "think_in_template": true, "reasoning_parser": "dots3", "family": "dots3_note", "cache_type": "hybrid_full_swa", "context_length": 524288, "dsa_index_topk": 2048 } }