Text Generation
Transformers
Safetensors
English
Japanese
qwen3_5_gdn24
qwen
qwen3.5
recurrent
linear-attention
gdn
cuda
custom_code
conversational
Instructions to use summerMC/Qwen3.5-9B-SpeedX9-GDN32 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use summerMC/Qwen3.5-9B-SpeedX9-GDN32 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="summerMC/Qwen3.5-9B-SpeedX9-GDN32", trust_remote_code=True) messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# pip install -U transformers accelerate # Load model directly from transformers import AutoModelForCausalLM model = AutoModelForCausalLM.from_pretrained("summerMC/Qwen3.5-9B-SpeedX9-GDN32", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use summerMC/Qwen3.5-9B-SpeedX9-GDN32 with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "summerMC/Qwen3.5-9B-SpeedX9-GDN32" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "summerMC/Qwen3.5-9B-SpeedX9-GDN32", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/summerMC/Qwen3.5-9B-SpeedX9-GDN32
- SGLang
How to use summerMC/Qwen3.5-9B-SpeedX9-GDN32 with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "summerMC/Qwen3.5-9B-SpeedX9-GDN32" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "summerMC/Qwen3.5-9B-SpeedX9-GDN32", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "summerMC/Qwen3.5-9B-SpeedX9-GDN32" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "summerMC/Qwen3.5-9B-SpeedX9-GDN32", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use summerMC/Qwen3.5-9B-SpeedX9-GDN32 with Docker Model Runner:
docker model run hf.co/summerMC/Qwen3.5-9B-SpeedX9-GDN32
Download uni_engine.py from summerMC/Qwen3.5-9B-SpeedX9-GDN32: direct link, hf CLI and curl.
- Browser
- Download file 11.6 kB
-
https://huggingface.co/summerMC/Qwen3.5-9B-SpeedX9-GDN32/resolve/main/uni_engine.py
- Command line
-
hf download hf://summerMC/Qwen3.5-9B-SpeedX9-GDN32/uni_engine.py
-
curl -L -o uni_engine.py https://huggingface.co/summerMC/Qwen3.5-9B-SpeedX9-GDN32/resolve/main/uni_engine.py
11.6 kB
| from __future__ import annotations | |
| """UNI MAX: unified exact-prefill / quantized-decode recurrent inference. | |
| The GDN24 cache topology is identical between the exact BF16 model and a model | |
| whose **state-safe tail** (final MLP + LM head) is quantized. UNI MAX v1.1 exploits that causal invariant: | |
| 1. prefill the prompt with the exact BF16 model; | |
| 2. copy only the fixed recurrent cache state into a state-compatible persistent decode engine; | |
| 3. decode with the fastest CUDA-Graph candidate (typically FP8 on NVIDIA L4); | |
| 4. optionally verify FP8 draft blocks with exact BF16 chunk forwards to recover | |
| exact greedy-token semantics. | |
| No KV sequence is copied: the bridge is O(1) in context length. Full-MLP quantization is deliberately excluded from this bridge because it changes the hidden trajectory that writes future recurrent states. | |
| """ | |
| from dataclasses import dataclass | |
| import time | |
| from typing import Any | |
| import torch | |
| from .engine import MaxTurboGraphDecoder, snapshot_cache, restore_cache_ | |
| class BridgeReport: | |
| seconds: float | |
| mib: float | |
| seen_tokens: int | |
| class VerifyReport: | |
| generated_tokens: int | |
| drafted_tokens: int | |
| accepted_draft_tokens: int | |
| rejected_blocks: int | |
| verifier_blocks: int | |
| def acceptance_rate(self) -> float: | |
| if self.drafted_tokens <= 0: | |
| return 1.0 | |
| return self.accepted_draft_tokens / self.drafted_tokens | |
| def cache_payload_bytes(cache) -> int: | |
| total = 0 | |
| for layer in cache.layers: | |
| for table_name in ("conv_states", "recurrent_states"): | |
| table = getattr(layer, table_name, {}) | |
| for tensor in table.values(): | |
| if tensor is not None: | |
| total += tensor.numel() * tensor.element_size() | |
| return int(total) | |
| def copy_cache_(dst, src) -> None: | |
| """Copy recurrent state without reallocating destination tensors. | |
| Destination storage must already be materialized (CUDA-Graph capture does | |
| this). Tensor addresses are preserved, so captured graphs remain valid. | |
| """ | |
| if len(dst.layers) != len(src.layers): | |
| raise ValueError("cache topology mismatch") | |
| for d_layer, s_layer in zip(dst.layers, src.layers): | |
| for table_name in ("conv_states", "recurrent_states"): | |
| d_table = getattr(d_layer, table_name) | |
| s_table = getattr(s_layer, table_name) | |
| for idx, s_tensor in s_table.items(): | |
| if s_tensor is None: | |
| continue | |
| d_tensor = d_table.get(idx) | |
| if d_tensor is None: | |
| raise RuntimeError( | |
| f"destination cache storage is not materialized: {table_name}[{idx}]" | |
| ) | |
| if d_tensor.shape != s_tensor.shape or d_tensor.dtype != s_tensor.dtype: | |
| raise RuntimeError( | |
| f"cache state mismatch for {table_name}[{idx}]: " | |
| f"dst={tuple(d_tensor.shape)}/{d_tensor.dtype}, " | |
| f"src={tuple(s_tensor.shape)}/{s_tensor.dtype}" | |
| ) | |
| d_tensor.copy_(s_tensor) | |
| d_layer.has_previous_state.clear() | |
| d_layer.has_previous_state.update(dict(s_layer.has_previous_state)) | |
| d_layer.is_conv_states_initialized.clear() | |
| d_layer.is_conv_states_initialized.update(dict(s_layer.is_conv_states_initialized)) | |
| d_layer.is_recurrent_states_initialized.clear() | |
| d_layer.is_recurrent_states_initialized.update(dict(s_layer.is_recurrent_states_initialized)) | |
| dst.seen_tokens = int(src.seen_tokens) | |
| class UniMaxEngine: | |
| """Phase-specialized recurrent inference engine. | |
| ``exact_model`` always handles prefill. ``decode_model`` can be the same model (UNI-SAFE) or a state-safe quantized clone (UNI-SPEED / UNI-EXACT). | |
| """ | |
| def __init__( | |
| self, | |
| exact_model, | |
| decode_model, | |
| decode_graph: MaxTurboGraphDecoder, | |
| ): | |
| self.exact_model = exact_model.eval() | |
| self.decode_model = decode_model.eval() | |
| self.decode_graph = decode_graph | |
| self.device = self.exact_model.get_input_embeddings().weight.device | |
| if self.device != self.decode_graph.device: | |
| raise ValueError("exact and decode engines must live on the same CUDA device") | |
| self.exact_cache = self.exact_model.make_recurrent_cache() | |
| self.last_bridge: BridgeReport | None = None | |
| def reset(self) -> None: | |
| self.exact_cache.reset() | |
| self.decode_graph.reset() | |
| self.last_bridge = None | |
| def _bridge(self) -> BridgeReport: | |
| if self.device.type == "cuda": | |
| torch.cuda.synchronize(self.device) | |
| t0 = time.perf_counter() | |
| copy_cache_(self.decode_graph.cache, self.exact_cache) | |
| if self.device.type == "cuda": | |
| torch.cuda.synchronize(self.device) | |
| dt = time.perf_counter() - t0 | |
| report = BridgeReport( | |
| seconds=float(dt), | |
| mib=cache_payload_bytes(self.exact_cache) / 2**20, | |
| seen_tokens=int(self.exact_cache.seen_tokens), | |
| ) | |
| self.last_bridge = report | |
| return report | |
| def prefill_exact(self, input_ids: torch.Tensor) -> tuple[torch.Tensor, Any, BridgeReport]: | |
| """Exact BF16 prefill, then O(1)-context state-compatible recurrent bridge.""" | |
| self.exact_cache.reset() | |
| out = self.exact_model( | |
| input_ids=input_ids, | |
| past_key_values=self.exact_cache, | |
| use_cache=True, | |
| logits_to_keep=1, | |
| ) | |
| first = torch.argmax(out.logits[:, -1, :], dim=-1, keepdim=True) | |
| bridge = self._bridge() | |
| self.decode_graph.static_token.copy_(first) | |
| return first, out, bridge | |
| def decode_fast(self, first_token: torch.Tensor, recurrent_forwards: int) -> torch.Tensor: | |
| return self.decode_graph.decode_forwards(first_token, int(recurrent_forwards)) | |
| def generate_fast(self, input_ids: torch.Tensor, max_new_tokens: int) -> torch.Tensor: | |
| """UNI-SPEED generation: exact first token, quantized graph thereafter.""" | |
| n = int(max_new_tokens) | |
| if n <= 0: | |
| return torch.empty((input_ids.shape[0], 0), dtype=torch.long, device=input_ids.device) | |
| first, _, _ = self.prefill_exact(input_ids) | |
| if n == 1: | |
| return first | |
| tail = self.decode_graph.decode_tokens(first, n - 1) | |
| return torch.cat([first, tail], dim=1) | |
| def decode_verified( | |
| self, | |
| first_token: torch.Tensor, | |
| recurrent_forwards: int, | |
| *, | |
| draft_block: int | None = None, | |
| ) -> tuple[torch.Tensor, VerifyReport]: | |
| """Verify ``recurrent_forwards`` tokens after ``first_token``. | |
| ``prefill_exact`` must have been called immediately before this method so | |
| both exact and decode caches represent the same prompt state. | |
| """ | |
| remaining = int(recurrent_forwards) | |
| if remaining <= 0: | |
| empty = torch.empty((first_token.shape[0], 0), dtype=torch.long, device=first_token.device) | |
| return empty, VerifyReport(0, 0, 0, 0, 0) | |
| if first_token.shape[0] != 1: | |
| raise ValueError("UNI-EXACT currently supports batch size 1") | |
| current = first_token | |
| generated: list[torch.Tensor] = [] | |
| k_default = max(1, int(draft_block or self.decode_graph.block_size)) | |
| drafted = accepted_drafts = rejected_blocks = verifier_blocks = 0 | |
| while remaining > 0: | |
| k = min(k_default, remaining) | |
| exact_before = snapshot_cache(self.exact_cache) | |
| draft = self.decode_graph.decode_tokens(current, k) | |
| drafted += k | |
| verify_input = current if k == 1 else torch.cat([current, draft[:, :-1]], dim=1) | |
| verify_out = self.exact_model( | |
| input_ids=verify_input, | |
| past_key_values=self.exact_cache, | |
| use_cache=True, | |
| logits_to_keep=k, | |
| ) | |
| exact_pred = torch.argmax(verify_out.logits[:, -k:, :], dim=-1) | |
| verifier_blocks += 1 | |
| mismatch_positions = (~exact_pred.eq(draft)[0]).nonzero(as_tuple=False) | |
| if mismatch_positions.numel() == 0: | |
| generated.append(draft.detach().clone()) | |
| accepted_drafts += k | |
| current = draft[:, -1:] | |
| remaining -= k | |
| continue | |
| m = int(mismatch_positions[0, 0].item()) | |
| accepted_drafts += m | |
| rejected_blocks += 1 | |
| restore_cache_(self.exact_cache, exact_before) | |
| replay_input = current if m == 0 else torch.cat([current, draft[:, :m]], dim=1) | |
| replay = self.exact_model( | |
| input_ids=replay_input, | |
| past_key_values=self.exact_cache, | |
| use_cache=True, | |
| logits_to_keep=1, | |
| ) | |
| corrected = torch.argmax(replay.logits[:, -1, :], dim=-1, keepdim=True) | |
| if m: | |
| generated.append(draft[:, :m].detach().clone()) | |
| generated.append(corrected.detach().clone()) | |
| current = corrected | |
| emitted = m + 1 | |
| remaining -= emitted | |
| copy_cache_(self.decode_graph.cache, self.exact_cache) | |
| self.decode_graph.static_token.copy_(current) | |
| out = torch.cat(generated, dim=1) | |
| return out, VerifyReport( | |
| generated_tokens=int(out.shape[1]), | |
| drafted_tokens=int(drafted), | |
| accepted_draft_tokens=int(accepted_drafts), | |
| rejected_blocks=int(rejected_blocks), | |
| verifier_blocks=int(verifier_blocks), | |
| ) | |
| def generate_verified( | |
| self, | |
| input_ids: torch.Tensor, | |
| max_new_tokens: int, | |
| *, | |
| draft_block: int | None = None, | |
| ) -> tuple[torch.Tensor, VerifyReport]: | |
| """UNI-EXACT generation with exact greedy-token semantics.""" | |
| n = int(max_new_tokens) | |
| if n <= 0: | |
| empty = torch.empty((input_ids.shape[0], 0), dtype=torch.long, device=input_ids.device) | |
| return empty, VerifyReport(0, 0, 0, 0, 0) | |
| first, _, _ = self.prefill_exact(input_ids) | |
| if n == 1: | |
| return first, VerifyReport(1, 0, 0, 0, 0) | |
| tail, report = self.decode_verified(first, n - 1, draft_block=draft_block) | |
| full = torch.cat([first, tail], dim=1) | |
| return full, VerifyReport( | |
| generated_tokens=int(full.shape[1]), | |
| drafted_tokens=report.drafted_tokens, | |
| accepted_draft_tokens=report.accepted_draft_tokens, | |
| rejected_blocks=report.rejected_blocks, | |
| verifier_blocks=report.verifier_blocks, | |
| ) | |
| def exact_greedy_tokens(model, input_ids: torch.Tensor, max_new_tokens: int) -> torch.Tensor: | |
| n = int(max_new_tokens) | |
| if n <= 0: | |
| return torch.empty((input_ids.shape[0], 0), dtype=torch.long, device=input_ids.device) | |
| cache = model.make_recurrent_cache() | |
| out = model(input_ids=input_ids, past_key_values=cache, use_cache=True, logits_to_keep=1) | |
| token = torch.argmax(out.logits[:, -1, :], dim=-1, keepdim=True) | |
| pieces = [token] | |
| for _ in range(n - 1): | |
| token = model.greedy_step(token, cache) | |
| pieces.append(token) | |
| return torch.cat(pieces, dim=1) | |