File size: 12,893 Bytes
5e01d16
 
 
9f1a442
5e01d16
 
 
 
9f1a442
 
5e01d16
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
9f1a442
 
5e01d16
 
 
 
 
 
 
9f1a442
9748538
 
5e01d16
 
 
 
9748538
 
5e01d16
9748538
 
5e01d16
 
 
 
 
 
 
 
 
 
9748538
 
 
5e01d16
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
9748538
 
5e01d16
9748538
 
5e01d16
 
 
 
 
9748538
 
 
5e01d16
 
9748538
5e01d16
 
 
 
 
 
9748538
5e01d16
 
9f1a442
5e01d16
 
9f1a442
5e01d16
9f1a442
 
5e01d16
9f1a442
5e01d16
 
9f1a442
9748538
5e01d16
9f1a442
5e01d16
9748538
9f1a442
9748538
5e01d16
9f1a442
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5e01d16
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
{
  "schema_version": "5.1",
  "name": "muse-glimmer-30b",
  "tt_metal_version": "0.65.2.dev8976",
  "arch": "blackhole",
  "device_count": 4,
  "producer": {
    "tt_kernel_version": "0.1.0",
    "created_at": "2026-10-05T16:43:44.785659+00:00",
    "hostname": "qb2-120-p11t01"
  },
  "weights": {
    "repo_id": "meta-models/Muse-Glimmer-30B",
    "revision": "f84ecc3a0ea984a4c04542a84269e3d065350a6e",
    "allow_patterns": null,
    "ignore_patterns": null,
    "repo_type": "model"
  },
  "mesh": null,
  "entrypoint": null,
  "resources": null,
  "capabilities": null,
  "env": {},
  "bundled": null,
  "deps": null,
  "container": {
    "image": {
      "registry": "hf",
      "repository": "muse-glimmer-30b",
      "tag": "tt-model/muse-glimmer-30b:f4f9b83a1f3e",
      "digest": "sha256:f4f9b83a1f3e93c9adf7715b3b33375e53f0c7a51e720a470b554e081b82b011"
    },
    "kind": "vllm-plugin",
    "runtime": {
      "vllm": {
        "version": "0.24.0"
      },
      "plugin": {
        "path": "/home/hous/dev/muse/vllm-tt-plugin",
        "sha": "0911fa6d4ac5fcd146d7d3dd3cc05f721fbb3201",
        "dirty": false
      },
      "lock": "requirements.lock"
    },
    "serve": {
      "hardware": "p300x2",
      "mesh_device": "P300x2",
      "port": 8000,
      "max_model_len": 131072,
      "max_num_seqs": 32,
      "block_size": 64,
      "server_timeout": null,
      "capabilities": {
        "tool_parser": "muse_glimmer",
        "reasoning_parser": "muse_glimmer"
      },
      "additional_config": {
        "tt": {
          "sample_on_device_mode": "all",
          "trace_region_size": 400000000,
          "fabric_config": "FABRIC_1D_RING",
          "fabric_packet_payload_bytes": 8192,
          "l1_small_size": 6144,
          "trace_mode": "decode_only"
        }
      },
      "args": [
        [
          "--tool-parser-plugin",
          "/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/tt/muse_glimmer_tool_parser.py"
        ],
        [
          "--reasoning-parser-plugin",
          "/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/tt/reasoning_parser.py"
        ]
      ],
      "env": {}
    },
    "serve_profiles": [
      {
        "hardware": null,
        "mesh_device": null,
        "port": null,
        "max_model_len": null,
        "max_num_seqs": null,
        "block_size": null,
        "server_timeout": null,
        "capabilities": null,
        "additional_config": {},
        "args": [],
        "env": {},
        "name": "default",
        "description": null
      }
    ],
    "default_profile": null,
    "code_dir": "code",
    "verify": [
      "import pathlib; p = pathlib.Path('/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/doc/datatype_sweep/selected_precision_config.json'); assert p.exists(), f'{p} missing: model would silently serve with default precision'",
      "import models.autoports.meta_models_muse_glimmer_30b.tt.generator_vllm",
      "import vllm; import vllm_tt_plugin.platform as p; import inspect; assert 'MuseGlimmerForConditionalGeneration' in inspect.getsource(p), 'plugin lacks the Muse-Glimmer registration'",
      "import vllm; import importlib.util; s = importlib.util.spec_from_file_location('mgtp', '/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/tt/muse_glimmer_tool_parser.py'); m = importlib.util.module_from_spec(s); s.loader.exec_module(m); from vllm.tool_parsers import ToolParserManager; assert 'muse_glimmer' in ToolParserManager.tool_parsers, 'ATEM tool parser did not register'",
      "import vllm; from vllm.reasoning import ReasoningParserManager as RM; RM.import_reasoning_parser('/opt/tt-metal/models/autoports/meta_models_muse_glimmer_30b/tt/reasoning_parser.py'); assert 'muse_glimmer' in RM.list_registered(), 'reasoning parser did not register'; RM.get_reasoning_parser('muse_glimmer')"
    ],
    "built": {
      "image": "tt-model/muse-glimmer-30b:f4f9b83a1f3e",
      "repo": "tt-hous/muse-glimmer-30b",
      "tt_model_version": "0.1.0",
      "created_at": "2026-10-05T16:40:10+00:00",
      "tt_metal": {
        "sha": "bb45e7a41804d6a9db51769448b31a6f9225c2b0",
        "describe": "v0.74.0-dev20260622-270-gbb45e7a418",
        "dirty": false,
        "scm_version": "0.65.2.dev8976",
        "mode": "local",
        "remote": "https://github.com/tenstorrent/tt-metal.git",
        "branch": "fix/muse-glimmer-30b-parsers",
        "pushed": false
      },
      "code_sha256": "7931f069dd3c0235c4487774a56858ae39fec017e288a2fa4b91d31078a03eda",
      "plugin": {
        "sha": "0911fa6d4ac5fcd146d7d3dd3cc05f721fbb3201",
        "path": "/home/hous/dev/muse/vllm-tt-plugin",
        "dirty": false
      },
      "image_digest": "sha256:f4f9b83a1f3e93c9adf7715b3b33375e53f0c7a51e720a470b554e081b82b011"
    },
    "card": {
      "description": null,
      "quickstart": "Muse-Glimmer-30B (~29.6 B dense, text-only) is served as an\nOpenAI-compatible endpoint for **agentic coding**: long-context (131k)\ntool-calling work driven by a coding agent.\n\nOn this model the first start takes about 4 minutes (weight loading + kernel\ncompilation). Verify it is running correctly with tool calling:\n```bash\ncurl -s localhost:8000/v1/chat/completions -H 'Content-Type: application/json' -d '{\n  \"model\": \"meta-models/Muse-Glimmer-30B\",\n  \"messages\": [{\"role\": \"user\", \"content\": \"What is the weather in Paris right now, in Celsius?\"}],\n  \"tools\": [{\n    \"type\": \"function\",\n    \"function\": {\n      \"name\": \"get_weather\",\n      \"description\": \"Get the current weather for a city.\",\n      \"parameters\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"city\":   {\"type\": \"string\", \"description\": \"City name\"},\n          \"metric\": {\"type\": \"boolean\", \"description\": \"true for Celsius\"}\n        },\n        \"required\": [\"city\"]\n      }\n    }\n  }],\n  \"tool_choice\": \"auto\",\n  \"max_tokens\": 256,\n  \"temperature\": 0\n}' | python3 -c 'import sys, json; c = json.load(sys.stdin)[\"choices\"][0]; print(c[\"finish_reason\"], json.dumps(c[\"message\"][\"tool_calls\"], indent=2))'\n```\nA correct serve prints `tool_calls` followed by a structured `get_weather` call\nwith JSON arguments (e.g. `{\"city\": \"Paris\", \"metric\": true}`). If the call comes\nback as prose in `message.content` with `finish_reason` `stop`, the tool-call\nparser is not active in the launch.\n\nThen verify plain chat, streamed. The model always thinks first, so this is\nthe path that shows whether the reasoning parser is splitting the channels:\n```bash\ncurl -sN localhost:8000/v1/chat/completions -H 'Content-Type: application/json' -d '{\n  \"model\": \"meta-models/Muse-Glimmer-30B\",\n  \"messages\": [{\"role\": \"user\", \"content\": \"What is 17 * 23?\"}],\n  \"max_tokens\": 512, \"temperature\": 0, \"stream\": true\n}' | grep -c '\"reasoning\"'\n```\nA correct serve prints a positive count: the analysis arrives as `reasoning`\ndeltas and only the answer arrives as `content`. Zero means the stream is raw\nchannel text (` to=self<|message|>...`), which is what a launch without the\ntwo parser plugins produces. Give plain chat a real token budget: at the\ntemplate's default `Reasoning strength: high` a short factual question needs\nroughly 500 to 700 completion tokens, and a turn cut off by `max_tokens`\ninside the analysis returns that analysis as `reasoning` with empty `content`.\n",
      "architecture": null,
      "status": null,
      "intended_use": null,
      "out_of_scope_use": null,
      "usage": null,
      "performance": "Release latency sweep on P300x2: one request at a time (batch 1), 512\noutput tokens, input length swept to the full context. Decode rate is per\nuser. Retried points show the median of three independent runs.\n\n| input tokens | output tokens | TTFT | TPOT | end-to-end | tokens/s/user |\n|---:|---:|---:|---:|---:|---:|\n| 128 | 512 | 69.5 ms | 23.60 ms | 12.1 s | 42.38 |\n| 1,024 | 512 | 144.6 ms | 24.99 ms | 12.9 s | 40.02 |\n| 4,096 | 512 | 454.5 ms | 26.64 ms | 14.1 s | 37.54 |\n| 8,192 | 512 | 912.8 ms | 27.86 ms | 15.1 s | 35.90 |\n| 16,384 | 512 | 2.08 s | 30.26 ms | 17.5 s | 33.05 |\n| 32,768 | 512 | 4.48 s | 35.29 ms | 22.5 s | 28.34 |\n| 65,536 | 512 | 10.17 s | 45.10 ms | 33.2 s | 22.17 |\n| 130,560 | 512 | 25.12 s | 64.76 ms | 58.2 s | 15.44 |\n\nThe last row saturates the advertised context (130,560 + 512 = 131,072).\nServe one request at a time: concurrency at long context is admission-limited\nby the KV cache. The release passed the bounded latency gate: 2% per metric,\nplus a 5 ms absolute TTFT allowance for short-input measurement variance.\n\n### Prefix caching\n\nMeasured with vLLM's own `prefix_repetition` benchmark, the standard dataset\nfor this feature. Eight distinct 4,096-token prefixes, each reused across\neight requests, 64 requests at concurrency 1 -- the same package served twice,\none flag apart.\n\n```\nvllm bench serve --model meta-models/Muse-Glimmer-30B \\\n  --dataset-name prefix_repetition \\\n  --prefix-repetition-prefix-len 4096 --prefix-repetition-suffix-len 128 \\\n  --prefix-repetition-num-prefixes 8 --prefix-repetition-output-len 32 \\\n  --num-prompts 64 --max-concurrency 1 --ignore-eos --seed 1234\n```\n\n| metric | caching off | caching on |\n|---|---:|---:|\n| mean TTFT | 497.00 ms | 168.04 ms |\n| median TTFT | 495.82 ms | 98.18 ms |\n| p99 TTFT | 525.84 ms | 952.04 ms |\n| benchmark duration | 80.14 s | 59.14 s |\n| output throughput | 25.56 tok/s | 34.63 tok/s |\n| prefix cache hit rate | 0 | 54.7% |\n\nThe distribution is the evidence, not the mean. With caching off every request\npays the full 4,096-token prefill and TTFT is flat at 496/497/526. With it on\nthe distribution splits: 98 ms median for the 56 requests that hit, 952 ms at\np99 for the 8 cold prefixes.\n\nNote the p99 moves the wrong way, 526 ms to 952 ms. A cold prefix now compiles\nits own SDPA program for its resume offset, so the first request at any\npreviously unseen offset is slower than it was. Median improves 5x; the tail\nregresses 1.8x. Workloads that reuse a small set of prefixes gain; workloads\nwhose offsets keep changing may not.\n\n### Evaluations\n\nRun through tt-inference-server's eval workflow -- its lm-eval command, venv\nand scoring -- against this package at 131,072 context. Sampling is this\ncard's recipe: temperature 1.0, top_p 0.95, top_k 64.\n\n| task | samples | metric | score | reference |\n|---|---:|---|---:|---:|\n| `gpqa_diamond_cot_zeroshot` | 198 | exact_match, flexible-extract | 77.27 | 72.8 |\n| `ifeval` | 541 | prompt_level_strict_acc | 88.72 | 77.0 |\n| `aime25` | 30 | exact_match | not valid, see below | 94.7 |\n\nThe GPQA reference is the GPU reference score for `openai/gpt-oss-20b`, the\nclosest configured analogue. This model had no GPQA row before this run, while\nevery other reasoning model of its size in the catalogue has one. The `ifeval`\nreference is an IFBench floor, not an equivalence target.\n\n`aime25` is not reported as a score. 18 of its 30 responses contained no\nextractable answer and one ran to 294,912 characters, while the same problems\nput to the server directly return correct boxed answers -- so the figure\nmeasures the eval path, not the model. 30 of the 198 GPQA responses show the\nsame pathology, which makes 77.27 a lower bound rather than a point estimate.\n",
      "limitations": "- Text-only. The checkpoint carries a perception encoder; this port serves text\n  and does not accept images.\n- Serve one request at a time at long context: admission is limited by the KV\n  cache (see the latency sweep under Expected performance).\n- The chat template's default `Reasoning strength: high` makes the model think\n  before every reply. A short factual question needs roughly 500 to 700\n  completion tokens; a `max_tokens` budget that ends inside the analysis returns\n  the analysis as `reasoning` and an empty string as `content`. Put\n  `Reasoning strength: low` in the system prompt, or send\n  `chat_template_kwargs: {\"reasoning_strength\": \"low\"}`, when a short budget is\n  required.\n- A request that sends `tools` with `tool_choice: \"none\"` suppresses tool calls\n  as the API requires, but if the model still writes a call, a non-streaming\n  response carries that call's markup in `content` as text.\n- Prefix caching improves median TTFT 5x but regresses p99 TTFT 1.8x on cold\n  prefixes, as measured under Expected performance.\n- `aime25` through the eval harness is not a valid score: 18 of 30 responses had\n  no extractable answer and one ran to 294,912 characters, while the same\n  problems put to the server directly return correct boxed answers.\n",
      "risks": null,
      "licensing": null,
      "related": null,
      "license": null,
      "pipeline_tag": null,
      "base_model": null
    }
  }
}