Download tt_kernel_manifest.json from tt-hous/muse-glimmer-30b: direct link, hf CLI and curl.
- Browser
- Download file 12.9 kB
-
https://huggingface.co/tt-hous/muse-glimmer-30b/resolve/main/tt_kernel_manifest.json
- Command line
-
hf download hf://tt-hous/muse-glimmer-30b/tt_kernel_manifest.json
-
curl -L -o tt_kernel_manifest.json https://huggingface.co/tt-hous/muse-glimmer-30b/resolve/main/tt_kernel_manifest.json
12.9 kB
| { | |
| "schema_version": "5.1", | |
| "name": "muse-glimmer-30b", | |
| "tt_metal_version": "0.65.2.dev8976", | |
| "arch": "blackhole", | |
| "device_count": 4, | |
| "producer": { | |
| "tt_kernel_version": "0.1.0", | |
| "created_at": "2026-10-05T16:43:44.785659+00:00", | |
| "hostname": "qb2-120-p11t01" | |
| }, | |
| "weights": { | |
| "repo_id": "meta-models/Muse-Glimmer-30B", | |
| "revision": "f84ecc3a0ea984a4c04542a84269e3d065350a6e", | |
| "allow_patterns": null, | |
| "ignore_patterns": null, | |
| "repo_type": "model" | |
| }, | |
| "mesh": null, | |
| "entrypoint": null, | |
| "resources": null, | |
| "capabilities": null, | |
| "env": {}, | |
| "bundled": null, | |
| "deps": null, | |
| "container": { | |
| "image": { | |
| "registry": "hf", | |
| "repository": "muse-glimmer-30b", | |
| "tag": "tt-model/muse-glimmer-30b:f4f9b83a1f3e", | |
| "digest": "sha256:f4f9b83a1f3e93c9adf7715b3b33375e53f0c7a51e720a470b554e081b82b011" | |
| }, | |
| "kind": "vllm-plugin", | |
| "runtime": { | |
| "vllm": { | |
| "version": "0.24.0" | |
| }, | |
| "plugin": { | |
| "path": "/home/hous/dev/muse/vllm-tt-plugin", | |
| "sha": "0911fa6d4ac5fcd146d7d3dd3cc05f721fbb3201", | |
| "dirty": false | |
| }, | |
| "lock": "requirements.lock" | |
| }, | |
| "serve": { | |
| "hardware": "p300x2", | |
| "mesh_device": "P300x2", | |
| "port": 8000, | |
| "max_model_len": 131072, | |
| "max_num_seqs": 32, | |
| "block_size": 64, | |
| "server_timeout": null, | |
| "capabilities": { | |
| "tool_parser": "muse_glimmer", | |
| "reasoning_parser": "muse_glimmer" | |
| }, | |
| "additional_config": { | |
| "tt": { | |
| "sample_on_device_mode": "all", | |
| "trace_region_size": 400000000, | |
| "fabric_config": "FABRIC_1D_RING", | |
| "fabric_packet_payload_bytes": 8192, | |
| "l1_small_size": 6144, | |
| "trace_mode": "decode_only" | |
| } | |
| }, | |
| "args": [ | |
| [ | |
| "--tool-parser-plugin", | |
| "/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/tt/muse_glimmer_tool_parser.py" | |
| ], | |
| [ | |
| "--reasoning-parser-plugin", | |
| "/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/tt/reasoning_parser.py" | |
| ] | |
| ], | |
| "env": {} | |
| }, | |
| "serve_profiles": [ | |
| { | |
| "hardware": null, | |
| "mesh_device": null, | |
| "port": null, | |
| "max_model_len": null, | |
| "max_num_seqs": null, | |
| "block_size": null, | |
| "server_timeout": null, | |
| "capabilities": null, | |
| "additional_config": {}, | |
| "args": [], | |
| "env": {}, | |
| "name": "default", | |
| "description": null | |
| } | |
| ], | |
| "default_profile": null, | |
| "code_dir": "code", | |
| "verify": [ | |
| "import pathlib; p = pathlib.Path('/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/doc/datatype_sweep/selected_precision_config.json'); assert p.exists(), f'{p} missing: model would silently serve with default precision'", | |
| "import models.autoports.meta_models_muse_glimmer_30b.tt.generator_vllm", | |
| "import vllm; import vllm_tt_plugin.platform as p; import inspect; assert 'MuseGlimmerForConditionalGeneration' in inspect.getsource(p), 'plugin lacks the Muse-Glimmer registration'", | |
| "import vllm; import importlib.util; s = importlib.util.spec_from_file_location('mgtp', '/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/tt/muse_glimmer_tool_parser.py'); m = importlib.util.module_from_spec(s); s.loader.exec_module(m); from vllm.tool_parsers import ToolParserManager; assert 'muse_glimmer' in ToolParserManager.tool_parsers, 'ATEM tool parser did not register'", | |
| "import vllm; from vllm.reasoning import ReasoningParserManager as RM; RM.import_reasoning_parser('/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/tt/reasoning_parser.py'); assert 'muse_glimmer' in RM.list_registered(), 'reasoning parser did not register'; RM.get_reasoning_parser('muse_glimmer')" | |
| ], | |
| "built": { | |
| "image": "tt-model/muse-glimmer-30b:f4f9b83a1f3e", | |
| "repo": "tt-hous/muse-glimmer-30b", | |
| "tt_model_version": "0.1.0", | |
| "created_at": "2026-10-05T16:40:10+00:00", | |
| "tt_metal": { | |
| "sha": "bb45e7a41804d6a9db51769448b31a6f9225c2b0", | |
| "describe": "v0.74.0-dev20260622-270-gbb45e7a418", | |
| "dirty": false, | |
| "scm_version": "0.65.2.dev8976", | |
| "mode": "local", | |
| "remote": "https://github.com/tenstorrent/tt-metal.git", | |
| "branch": "fix/muse-glimmer-30b-parsers", | |
| "pushed": false | |
| }, | |
| "code_sha256": "7931f069dd3c0235c4487774a56858ae39fec017e288a2fa4b91d31078a03eda", | |
| "plugin": { | |
| "sha": "0911fa6d4ac5fcd146d7d3dd3cc05f721fbb3201", | |
| "path": "/home/hous/dev/muse/vllm-tt-plugin", | |
| "dirty": false | |
| }, | |
| "image_digest": "sha256:f4f9b83a1f3e93c9adf7715b3b33375e53f0c7a51e720a470b554e081b82b011" | |
| }, | |
| "card": { | |
| "description": null, | |
| "quickstart": "Muse-Glimmer-30B (~29.6 B dense, text-only) is served as an\nOpenAI-compatible endpoint for **agentic coding**: long-context (131k)\ntool-calling work driven by a coding agent.\n\nOn this model the first start takes about 4 minutes (weight loading + kernel\ncompilation). Verify it is running correctly with tool calling:\n```bash\ncurl -s localhost:8000/v1/chat/completions -H 'Content-Type: application/json' -d '{\n \"model\": \"meta-models/Muse-Glimmer-30B\",\n \"messages\": [{\"role\": \"user\", \"content\": \"What is the weather in Paris right now, in Celsius?\"}],\n \"tools\": [{\n \"type\": \"function\",\n \"function\": {\n \"name\": \"get_weather\",\n \"description\": \"Get the current weather for a city.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"city\": {\"type\": \"string\", \"description\": \"City name\"},\n \"metric\": {\"type\": \"boolean\", \"description\": \"true for Celsius\"}\n },\n \"required\": [\"city\"]\n }\n }\n }],\n \"tool_choice\": \"auto\",\n \"max_tokens\": 256,\n \"temperature\": 0\n}' | python3 -c 'import sys, json; c = json.load(sys.stdin)[\"choices\"][0]; print(c[\"finish_reason\"], json.dumps(c[\"message\"][\"tool_calls\"], indent=2))'\n```\nA correct serve prints `tool_calls` followed by a structured `get_weather` call\nwith JSON arguments (e.g. `{\"city\": \"Paris\", \"metric\": true}`). If the call comes\nback as prose in `message.content` with `finish_reason` `stop`, the tool-call\nparser is not active in the launch.\n\nThen verify plain chat, streamed. The model always thinks first, so this is\nthe path that shows whether the reasoning parser is splitting the channels:\n```bash\ncurl -sN localhost:8000/v1/chat/completions -H 'Content-Type: application/json' -d '{\n \"model\": \"meta-models/Muse-Glimmer-30B\",\n \"messages\": [{\"role\": \"user\", \"content\": \"What is 17 * 23?\"}],\n \"max_tokens\": 512, \"temperature\": 0, \"stream\": true\n}' | grep -c '\"reasoning\"'\n```\nA correct serve prints a positive count: the analysis arrives as `reasoning`\ndeltas and only the answer arrives as `content`. Zero means the stream is raw\nchannel text (` to=self<|message|>...`), which is what a launch without the\ntwo parser plugins produces. Give plain chat a real token budget: at the\ntemplate's default `Reasoning strength: high` a short factual question needs\nroughly 500 to 700 completion tokens, and a turn cut off by `max_tokens`\ninside the analysis returns that analysis as `reasoning` with empty `content`.\n", | |
| "architecture": null, | |
| "status": null, | |
| "intended_use": null, | |
| "out_of_scope_use": null, | |
| "usage": null, | |
| "performance": "Release latency sweep on P300x2: one request at a time (batch 1), 512\noutput tokens, input length swept to the full context. Decode rate is per\nuser. Retried points show the median of three independent runs.\n\n| input tokens | output tokens | TTFT | TPOT | end-to-end | tokens/s/user |\n|---:|---:|---:|---:|---:|---:|\n| 128 | 512 | 69.5 ms | 23.60 ms | 12.1 s | 42.38 |\n| 1,024 | 512 | 144.6 ms | 24.99 ms | 12.9 s | 40.02 |\n| 4,096 | 512 | 454.5 ms | 26.64 ms | 14.1 s | 37.54 |\n| 8,192 | 512 | 912.8 ms | 27.86 ms | 15.1 s | 35.90 |\n| 16,384 | 512 | 2.08 s | 30.26 ms | 17.5 s | 33.05 |\n| 32,768 | 512 | 4.48 s | 35.29 ms | 22.5 s | 28.34 |\n| 65,536 | 512 | 10.17 s | 45.10 ms | 33.2 s | 22.17 |\n| 130,560 | 512 | 25.12 s | 64.76 ms | 58.2 s | 15.44 |\n\nThe last row saturates the advertised context (130,560 + 512 = 131,072).\nServe one request at a time: concurrency at long context is admission-limited\nby the KV cache. The release passed the bounded latency gate: 2% per metric,\nplus a 5 ms absolute TTFT allowance for short-input measurement variance.\n\n### Prefix caching\n\nMeasured with vLLM's own `prefix_repetition` benchmark, the standard dataset\nfor this feature. Eight distinct 4,096-token prefixes, each reused across\neight requests, 64 requests at concurrency 1 -- the same package served twice,\none flag apart.\n\n```\nvllm bench serve --model meta-models/Muse-Glimmer-30B \\\n --dataset-name prefix_repetition \\\n --prefix-repetition-prefix-len 4096 --prefix-repetition-suffix-len 128 \\\n --prefix-repetition-num-prefixes 8 --prefix-repetition-output-len 32 \\\n --num-prompts 64 --max-concurrency 1 --ignore-eos --seed 1234\n```\n\n| metric | caching off | caching on |\n|---|---:|---:|\n| mean TTFT | 497.00 ms | 168.04 ms |\n| median TTFT | 495.82 ms | 98.18 ms |\n| p99 TTFT | 525.84 ms | 952.04 ms |\n| benchmark duration | 80.14 s | 59.14 s |\n| output throughput | 25.56 tok/s | 34.63 tok/s |\n| prefix cache hit rate | 0 | 54.7% |\n\nThe distribution is the evidence, not the mean. With caching off every request\npays the full 4,096-token prefill and TTFT is flat at 496/497/526. With it on\nthe distribution splits: 98 ms median for the 56 requests that hit, 952 ms at\np99 for the 8 cold prefixes.\n\nNote the p99 moves the wrong way, 526 ms to 952 ms. A cold prefix now compiles\nits own SDPA program for its resume offset, so the first request at any\npreviously unseen offset is slower than it was. Median improves 5x; the tail\nregresses 1.8x. Workloads that reuse a small set of prefixes gain; workloads\nwhose offsets keep changing may not.\n\n### Evaluations\n\nRun through tt-inference-server's eval workflow -- its lm-eval command, venv\nand scoring -- against this package at 131,072 context. Sampling is this\ncard's recipe: temperature 1.0, top_p 0.95, top_k 64.\n\n| task | samples | metric | score | reference |\n|---|---:|---|---:|---:|\n| `gpqa_diamond_cot_zeroshot` | 198 | exact_match, flexible-extract | 77.27 | 72.8 |\n| `ifeval` | 541 | prompt_level_strict_acc | 88.72 | 77.0 |\n| `aime25` | 30 | exact_match | not valid, see below | 94.7 |\n\nThe GPQA reference is the GPU reference score for `openai/gpt-oss-20b`, the\nclosest configured analogue. This model had no GPQA row before this run, while\nevery other reasoning model of its size in the catalogue has one. The `ifeval`\nreference is an IFBench floor, not an equivalence target.\n\n`aime25` is not reported as a score. 18 of its 30 responses contained no\nextractable answer and one ran to 294,912 characters, while the same problems\nput to the server directly return correct boxed answers -- so the figure\nmeasures the eval path, not the model. 30 of the 198 GPQA responses show the\nsame pathology, which makes 77.27 a lower bound rather than a point estimate.\n", | |
| "limitations": "- Text-only. The checkpoint carries a perception encoder; this port serves text\n and does not accept images.\n- Serve one request at a time at long context: admission is limited by the KV\n cache (see the latency sweep under Expected performance).\n- The chat template's default `Reasoning strength: high` makes the model think\n before every reply. A short factual question needs roughly 500 to 700\n completion tokens; a `max_tokens` budget that ends inside the analysis returns\n the analysis as `reasoning` and an empty string as `content`. Put\n `Reasoning strength: low` in the system prompt, or send\n `chat_template_kwargs: {\"reasoning_strength\": \"low\"}`, when a short budget is\n required.\n- A request that sends `tools` with `tool_choice: \"none\"` suppresses tool calls\n as the API requires, but if the model still writes a call, a non-streaming\n response carries that call's markup in `content` as text.\n- Prefix caching improves median TTFT 5x but regresses p99 TTFT 1.8x on cold\n prefixes, as measured under Expected performance.\n- `aime25` through the eval harness is not a valid score: 18 of 30 responses had\n no extractable answer and one ran to 294,912 characters, while the same\n problems put to the server directly return correct boxed answers.\n", | |
| "risks": null, | |
| "licensing": null, | |
| "related": null, | |
| "license": null, | |
| "pipeline_tag": null, | |
| "base_model": null | |
| } | |
| } | |
| } |