muse-glimmer-30b / tt_kernel_manifest.json
tt-hous's picture
Add files using upload-large-folder tool
9f1a442 verified
Raw History Blame Contribute Delete
12.9 kB
{
"schema_version": "5.1",
"name": "muse-glimmer-30b",
"tt_metal_version": "0.65.2.dev8976",
"arch": "blackhole",
"device_count": 4,
"producer": {
"tt_kernel_version": "0.1.0",
"created_at": "2026-10-05T16:43:44.785659+00:00",
"hostname": "qb2-120-p11t01"
},
"weights": {
"repo_id": "meta-models/Muse-Glimmer-30B",
"revision": "f84ecc3a0ea984a4c04542a84269e3d065350a6e",
"allow_patterns": null,
"ignore_patterns": null,
"repo_type": "model"
},
"mesh": null,
"entrypoint": null,
"resources": null,
"capabilities": null,
"env": {},
"bundled": null,
"deps": null,
"container": {
"image": {
"registry": "hf",
"repository": "muse-glimmer-30b",
"tag": "tt-model/muse-glimmer-30b:f4f9b83a1f3e",
"digest": "sha256:f4f9b83a1f3e93c9adf7715b3b33375e53f0c7a51e720a470b554e081b82b011"
},
"kind": "vllm-plugin",
"runtime": {
"vllm": {
"version": "0.24.0"
},
"plugin": {
"path": "/home/hous/dev/muse/vllm-tt-plugin",
"sha": "0911fa6d4ac5fcd146d7d3dd3cc05f721fbb3201",
"dirty": false
},
"lock": "requirements.lock"
},
"serve": {
"hardware": "p300x2",
"mesh_device": "P300x2",
"port": 8000,
"max_model_len": 131072,
"max_num_seqs": 32,
"block_size": 64,
"server_timeout": null,
"capabilities": {
"tool_parser": "muse_glimmer",
"reasoning_parser": "muse_glimmer"
},
"additional_config": {
"tt": {
"sample_on_device_mode": "all",
"trace_region_size": 400000000,
"fabric_config": "FABRIC_1D_RING",
"fabric_packet_payload_bytes": 8192,
"l1_small_size": 6144,
"trace_mode": "decode_only"
}
},
"args": [
[
"--tool-parser-plugin",
"/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/tt/muse_glimmer_tool_parser.py"
],
[
"--reasoning-parser-plugin",
"/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/tt/reasoning_parser.py"
]
],
"env": {}
},
"serve_profiles": [
{
"hardware": null,
"mesh_device": null,
"port": null,
"max_model_len": null,
"max_num_seqs": null,
"block_size": null,
"server_timeout": null,
"capabilities": null,
"additional_config": {},
"args": [],
"env": {},
"name": "default",
"description": null
}
],
"default_profile": null,
"code_dir": "code",
"verify": [
"import pathlib; p = pathlib.Path('/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/doc/datatype_sweep/selected_precision_config.json'); assert p.exists(), f'{p} missing: model would silently serve with default precision'",
"import models.autoports.meta_models_muse_glimmer_30b.tt.generator_vllm",
"import vllm; import vllm_tt_plugin.platform as p; import inspect; assert 'MuseGlimmerForConditionalGeneration' in inspect.getsource(p), 'plugin lacks the Muse-Glimmer registration'",
"import vllm; import importlib.util; s = importlib.util.spec_from_file_location('mgtp', '/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/tt/muse_glimmer_tool_parser.py'); m = importlib.util.module_from_spec(s); s.loader.exec_module(m); from vllm.tool_parsers import ToolParserManager; assert 'muse_glimmer' in ToolParserManager.tool_parsers, 'ATEM tool parser did not register'",
"import vllm; from vllm.reasoning import ReasoningParserManager as RM; RM.import_reasoning_parser('/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/tt/reasoning_parser.py'); assert 'muse_glimmer' in RM.list_registered(), 'reasoning parser did not register'; RM.get_reasoning_parser('muse_glimmer')"
],
"built": {
"image": "tt-model/muse-glimmer-30b:f4f9b83a1f3e",
"repo": "tt-hous/muse-glimmer-30b",
"tt_model_version": "0.1.0",
"created_at": "2026-10-05T16:40:10+00:00",
"tt_metal": {
"sha": "bb45e7a41804d6a9db51769448b31a6f9225c2b0",
"describe": "v0.74.0-dev20260622-270-gbb45e7a418",
"dirty": false,
"scm_version": "0.65.2.dev8976",
"mode": "local",
"remote": "https://github.com/tenstorrent/tt-metal.git",
"branch": "fix/muse-glimmer-30b-parsers",
"pushed": false
},
"code_sha256": "7931f069dd3c0235c4487774a56858ae39fec017e288a2fa4b91d31078a03eda",
"plugin": {
"sha": "0911fa6d4ac5fcd146d7d3dd3cc05f721fbb3201",
"path": "/home/hous/dev/muse/vllm-tt-plugin",
"dirty": false
},
"image_digest": "sha256:f4f9b83a1f3e93c9adf7715b3b33375e53f0c7a51e720a470b554e081b82b011"
},
"card": {
"description": null,
"quickstart": "Muse-Glimmer-30B (~29.6 B dense, text-only) is served as an\nOpenAI-compatible endpoint for **agentic coding**: long-context (131k)\ntool-calling work driven by a coding agent.\n\nOn this model the first start takes about 4 minutes (weight loading + kernel\ncompilation). Verify it is running correctly with tool calling:\n```bash\ncurl -s localhost:8000/v1/chat/completions -H 'Content-Type: application/json' -d '{\n \"model\": \"meta-models/Muse-Glimmer-30B\",\n \"messages\": [{\"role\": \"user\", \"content\": \"What is the weather in Paris right now, in Celsius?\"}],\n \"tools\": [{\n \"type\": \"function\",\n \"function\": {\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\"type\": \"string\", \"description\": \"City name\"},\n \"metric\": {\"type\": \"boolean\", \"description\": \"true for Celsius\"}\n },\n \"required\": [\"city\"]\n }\n }\n }],\n \"tool_choice\": \"auto\",\n \"max_tokens\": 256,\n \"temperature\": 0\n}' | python3 -c 'import sys, json; c = json.load(sys.stdin)[\"choices\"][0]; print(c[\"finish_reason\"], json.dumps(c[\"message\"][\"tool_calls\"], indent=2))'\n```\nA correct serve prints `tool_calls` followed by a structured `get_weather` call\nwith JSON arguments (e.g. `{\"city\": \"Paris\", \"metric\": true}`). If the call comes\nback as prose in `message.content` with `finish_reason` `stop`, the tool-call\nparser is not active in the launch.\n\nThen verify plain chat, streamed. The model always thinks first, so this is\nthe path that shows whether the reasoning parser is splitting the channels:\n```bash\ncurl -sN localhost:8000/v1/chat/completions -H 'Content-Type: application/json' -d '{\n \"model\": \"meta-models/Muse-Glimmer-30B\",\n \"messages\": [{\"role\": \"user\", \"content\": \"What is 17 * 23?\"}],\n \"max_tokens\": 512, \"temperature\": 0, \"stream\": true\n}' | grep -c '\"reasoning\"'\n```\nA correct serve prints a positive count: the analysis arrives as `reasoning`\ndeltas and only the answer arrives as `content`. Zero means the stream is raw\nchannel text (` to=self<|message|>...`), which is what a launch without the\ntwo parser plugins produces. Give plain chat a real token budget: at the\ntemplate's default `Reasoning strength: high` a short factual question needs\nroughly 500 to 700 completion tokens, and a turn cut off by `max_tokens`\ninside the analysis returns that analysis as `reasoning` with empty `content`.\n",
"architecture": null,
"status": null,
"intended_use": null,
"out_of_scope_use": null,
"usage": null,
"performance": "Release latency sweep on P300x2: one request at a time (batch 1), 512\noutput tokens, input length swept to the full context. Decode rate is per\nuser. Retried points show the median of three independent runs.\n\n| input tokens | output tokens | TTFT | TPOT | end-to-end | tokens/s/user |\n|---:|---:|---:|---:|---:|---:|\n| 128 | 512 | 69.5 ms | 23.60 ms | 12.1 s | 42.38 |\n| 1,024 | 512 | 144.6 ms | 24.99 ms | 12.9 s | 40.02 |\n| 4,096 | 512 | 454.5 ms | 26.64 ms | 14.1 s | 37.54 |\n| 8,192 | 512 | 912.8 ms | 27.86 ms | 15.1 s | 35.90 |\n| 16,384 | 512 | 2.08 s | 30.26 ms | 17.5 s | 33.05 |\n| 32,768 | 512 | 4.48 s | 35.29 ms | 22.5 s | 28.34 |\n| 65,536 | 512 | 10.17 s | 45.10 ms | 33.2 s | 22.17 |\n| 130,560 | 512 | 25.12 s | 64.76 ms | 58.2 s | 15.44 |\n\nThe last row saturates the advertised context (130,560 + 512 = 131,072).\nServe one request at a time: concurrency at long context is admission-limited\nby the KV cache. The release passed the bounded latency gate: 2% per metric,\nplus a 5 ms absolute TTFT allowance for short-input measurement variance.\n\n### Prefix caching\n\nMeasured with vLLM's own `prefix_repetition` benchmark, the standard dataset\nfor this feature. Eight distinct 4,096-token prefixes, each reused across\neight requests, 64 requests at concurrency 1 -- the same package served twice,\none flag apart.\n\n```\nvllm bench serve --model meta-models/Muse-Glimmer-30B \\\n --dataset-name prefix_repetition \\\n --prefix-repetition-prefix-len 4096 --prefix-repetition-suffix-len 128 \\\n --prefix-repetition-num-prefixes 8 --prefix-repetition-output-len 32 \\\n --num-prompts 64 --max-concurrency 1 --ignore-eos --seed 1234\n```\n\n| metric | caching off | caching on |\n|---|---:|---:|\n| mean TTFT | 497.00 ms | 168.04 ms |\n| median TTFT | 495.82 ms | 98.18 ms |\n| p99 TTFT | 525.84 ms | 952.04 ms |\n| benchmark duration | 80.14 s | 59.14 s |\n| output throughput | 25.56 tok/s | 34.63 tok/s |\n| prefix cache hit rate | 0 | 54.7% |\n\nThe distribution is the evidence, not the mean. With caching off every request\npays the full 4,096-token prefill and TTFT is flat at 496/497/526. With it on\nthe distribution splits: 98 ms median for the 56 requests that hit, 952 ms at\np99 for the 8 cold prefixes.\n\nNote the p99 moves the wrong way, 526 ms to 952 ms. A cold prefix now compiles\nits own SDPA program for its resume offset, so the first request at any\npreviously unseen offset is slower than it was. Median improves 5x; the tail\nregresses 1.8x. Workloads that reuse a small set of prefixes gain; workloads\nwhose offsets keep changing may not.\n\n### Evaluations\n\nRun through tt-inference-server's eval workflow -- its lm-eval command, venv\nand scoring -- against this package at 131,072 context. Sampling is this\ncard's recipe: temperature 1.0, top_p 0.95, top_k 64.\n\n| task | samples | metric | score | reference |\n|---|---:|---|---:|---:|\n| `gpqa_diamond_cot_zeroshot` | 198 | exact_match, flexible-extract | 77.27 | 72.8 |\n| `ifeval` | 541 | prompt_level_strict_acc | 88.72 | 77.0 |\n| `aime25` | 30 | exact_match | not valid, see below | 94.7 |\n\nThe GPQA reference is the GPU reference score for `openai/gpt-oss-20b`, the\nclosest configured analogue. This model had no GPQA row before this run, while\nevery other reasoning model of its size in the catalogue has one. The `ifeval`\nreference is an IFBench floor, not an equivalence target.\n\n`aime25` is not reported as a score. 18 of its 30 responses contained no\nextractable answer and one ran to 294,912 characters, while the same problems\nput to the server directly return correct boxed answers -- so the figure\nmeasures the eval path, not the model. 30 of the 198 GPQA responses show the\nsame pathology, which makes 77.27 a lower bound rather than a point estimate.\n",
"limitations": "- Text-only. The checkpoint carries a perception encoder; this port serves text\n and does not accept images.\n- Serve one request at a time at long context: admission is limited by the KV\n cache (see the latency sweep under Expected performance).\n- The chat template's default `Reasoning strength: high` makes the model think\n before every reply. A short factual question needs roughly 500 to 700\n completion tokens; a `max_tokens` budget that ends inside the analysis returns\n the analysis as `reasoning` and an empty string as `content`. Put\n `Reasoning strength: low` in the system prompt, or send\n `chat_template_kwargs: {\"reasoning_strength\": \"low\"}`, when a short budget is\n required.\n- A request that sends `tools` with `tool_choice: \"none\"` suppresses tool calls\n as the API requires, but if the model still writes a call, a non-streaming\n response carries that call's markup in `content` as text.\n- Prefix caching improves median TTFT 5x but regresses p99 TTFT 1.8x on cold\n prefixes, as measured under Expected performance.\n- `aime25` through the eval harness is not a valid score: 18 of 30 responses had\n no extractable answer and one ran to 294,912 characters, while the same\n problems put to the server directly return correct boxed answers.\n",
"risks": null,
"licensing": null,
"related": null,
"license": null,
"pipeline_tag": null,
"base_model": null
}
}
}