Download modeling_decision1.py from vllm-sr/Decision-1.0-Route-0.6B: direct link, hf CLI and curl.
- Browser
- Download file 13.9 kB
-
https://huggingface.co/vllm-sr/Decision-1.0-Route-0.6B/resolve/a5b21dffbf09f7926b6f20e9ec352574c3aa5860/modeling_decision1.py
- Command line
-
hf download hf://vllm-sr/Decision-1.0-Route-0.6B@a5b21dffbf09f7926b6f20e9ec352574c3aa5860/modeling_decision1.py
-
curl -L -o modeling_decision1.py https://huggingface.co/vllm-sr/Decision-1.0-Route-0.6B/resolve/a5b21dffbf09f7926b6f20e9ec352574c3aa5860/modeling_decision1.py
13.9 kB
| # Copyright 2026 The vLLM Semantic Router Authors. | |
| # SPDX-License-Identifier: Apache-2.0 | |
| """Decision 1.0 model for 🤗 Transformers (``trust_remote_code=True``). | |
| ``AutoModel.from_pretrained(repo, trust_remote_code=True)`` downloads the files | |
| named by the repository's ``config.json`` and builds the Decision 1.0 native | |
| inference path from the modules next to this one. ``system_one(state=..., | |
| questions={...})`` answers typed Choice, Noul and Score questions with the System | |
| One response body; there is no text generation and no chat API. | |
| """ | |
| from __future__ import annotations | |
| import os | |
| from pathlib import Path | |
| from typing import Any | |
| import torch | |
| from transformers import PreTrainedModel | |
| from .configuration_decision1 import Decision1Config | |
| from .decision1_system_one import ( | |
| DecisionInputError, | |
| DecisionInputTooLongError, | |
| answer, | |
| build_row, | |
| error_answer, | |
| validate_question, | |
| validate_request, | |
| ) | |
| HUB_OPTIONS = ( | |
| "cache_dir", | |
| "force_download", | |
| "local_files_only", | |
| "proxies", | |
| "revision", | |
| "token", | |
| ) | |
| RUNTIME_OPTIONS = ("device", "threads") | |
| # Options of Transformers' own weight loader, which this model does not use. | |
| LOADER_FLAGS = ( | |
| "trust_remote_code", | |
| "_from_auto", | |
| "_from_pipeline", | |
| "adapter_kwargs", | |
| "code_revision", | |
| "_commit_hash", | |
| "low_cpu_mem_usage", | |
| "use_safetensors", | |
| "resume_download", | |
| "user_agent", | |
| ) | |
| # Input limits and prompt policies of the published models, as the Decision runtime serves them. | |
| FAMILY_PROFILES = { | |
| "vela-encoder": {"max_input_tokens": 1024, "choice_null_description": "render_key"}, | |
| "qwen3.5-decision": { | |
| "max_input_tokens": 16384, | |
| "choice_null_description": "preserve_json_null", | |
| }, | |
| } | |
| MODEL_PROFILES = { | |
| "Decision-1.0-Nox-4B": {"choice_null_description": "render_key"}, | |
| "Decision-1.0-Eos-0.8B": {"rocm_conv": True}, | |
| } | |
| __all__ = ["Decision1Model", "DecisionInputError", "DecisionInputTooLongError"] | |
| def _device_name(value: Any) -> str: | |
| if isinstance(value, bool): | |
| raise ValueError(f"Not a device: {value!r}") | |
| if isinstance(value, int): | |
| return "cpu" if value < 0 else f"cuda:{value}" | |
| if isinstance(value, (str, torch.device)): | |
| return str(torch.device(value)) | |
| raise ValueError(f"Not a device: {value!r}") | |
| def _device(device: Any, device_map: Any) -> torch.device: | |
| if isinstance(device_map, dict): | |
| if set(device_map) != {""}: | |
| raise ValueError( | |
| "Decision 1.0 models run on one device: pass a device name or {'': device}" | |
| ) | |
| device_map = device_map[""] | |
| if device_map == "auto": | |
| device_map = None | |
| names = {_device_name(v) for v in (device, device_map) if v is not None} | |
| if len(names) > 1: | |
| raise ValueError("device and device_map name different devices") | |
| if names: | |
| return torch.device(names.pop()) | |
| return torch.device("cuda:0" if torch.cuda.is_available() else "cpu") | |
| def _offline(options: dict[str, Any]) -> dict[str, Any]: | |
| """With HF_HUB_OFFLINE, read the cache only (huggingface_hub would still list a commit's files).""" | |
| try: | |
| from huggingface_hub import is_offline_mode | |
| offline = is_offline_mode() | |
| except ImportError: | |
| from huggingface_hub import constants | |
| offline = constants.HF_HUB_OFFLINE | |
| return {**options, "local_files_only": True} if offline else options | |
| def _package_dir( | |
| name_or_path: Any, config: Decision1Config, hub: dict[str, Any] | |
| ) -> Path: | |
| """The repository revision as a directory: a local download, or a snapshot in the Hugging Face cache.""" | |
| local = Path(os.fspath(name_or_path)).expanduser() | |
| if local.is_dir(): | |
| return local.resolve() | |
| from huggingface_hub import snapshot_download | |
| revision = hub.get("revision") | |
| # Weights come from the commit the config came from: Transformers 5.18 passes a | |
| # ResolvedRevision, earlier versions record the commit on the config. | |
| commit = getattr(revision, "resolved", None) | |
| if commit is None and getattr(config, "name_or_path", None) == str(name_or_path): | |
| commit = getattr(config, "_commit_hash", None) | |
| options = _offline( | |
| { | |
| k: v | |
| for k, v in hub.items() | |
| if k != "revision" and v is not None and v is not False | |
| } | |
| ) | |
| return Path( | |
| snapshot_download( | |
| str(name_or_path), | |
| revision=commit or revision, | |
| allow_patterns=config.files(), | |
| **options, | |
| ) | |
| ) | |
| class Decision1Model(PreTrainedModel): | |
| """A Decision 1.0 model behind System One: ``system_one(state=..., questions={...})``.""" | |
| config_class = Decision1Config | |
| base_model_prefix = "decision" | |
| main_input_name = "state" | |
| supports_gradient_checkpointing = False | |
| _no_split_modules: list[str] = [] | |
| def __init__(self, config: Decision1Config): | |
| super().__init__(config) | |
| self.runtime = None | |
| self._source: Path | None = None | |
| self._threads: int | None = None | |
| self.post_init() | |
| def _init_weights(self, module: Any) -> None: | |
| """Every weight comes from the repository files; nothing is initialized here.""" | |
| def from_pretrained( | |
| cls, | |
| pretrained_model_name_or_path: str | os.PathLike, | |
| *model_args: Any, | |
| config: Decision1Config | None = None, | |
| **kwargs: Any, | |
| ) -> Decision1Model: | |
| """Load a Hub repository or a local download on one device. | |
| Hub options: ``revision``, ``cache_dir``, ``token``, ``local_files_only``, | |
| ``force_download``. ``device`` or ``device_map`` names one device (default: | |
| cuda:0 if a GPU is visible, else CPU); ``threads`` sets CPU threads. | |
| Numerics are the model's own, so ``dtype`` is only None or "auto". | |
| """ | |
| if model_args: | |
| raise TypeError("Decision 1.0 models take no positional model arguments") | |
| hub = {k: kwargs.pop(k) for k in HUB_OPTIONS if k in kwargs} | |
| if kwargs.pop("subfolder", "") not in ("", None): | |
| raise ValueError("A Decision 1.0 repository loads from its root") | |
| options = {k: kwargs.pop(k) for k in RUNTIME_OPTIONS if k in kwargs} | |
| device_map = kwargs.pop("device_map", None) | |
| for key in ("dtype", "torch_dtype"): | |
| if kwargs.pop(key, None) not in (None, "auto"): | |
| raise ValueError( | |
| f"{key}: Decision 1.0 numerics are fixed (FP32 encoders; decoders BF16 on a " | |
| "GPU with an FP32 head, FP32 on CPU); pass None or 'auto'" | |
| ) | |
| if kwargs.pop("attn_implementation", None) not in (None, "sdpa"): | |
| raise ValueError("Decision 1.0 models use SDPA attention") | |
| loading_info = kwargs.pop("output_loading_info", False) | |
| for key in LOADER_FLAGS: | |
| kwargs.pop(key, None) | |
| if kwargs: | |
| raise TypeError( | |
| f"Unsupported keyword arguments for a Decision 1.0 model: {sorted(kwargs)}" | |
| ) | |
| if config is None: | |
| config = Decision1Config.from_pretrained( | |
| pretrained_model_name_or_path, | |
| **{k: v for k, v in hub.items() if v is not None}, | |
| ) | |
| model = cls(config) | |
| model._source = _package_dir(pretrained_model_name_or_path, config, hub) | |
| model._threads = options.get("threads") | |
| model._load(_device(options.get("device"), device_map)) | |
| model.name_or_path = str(pretrained_model_name_or_path) | |
| if loading_info: | |
| return model, { | |
| "missing_keys": [], | |
| "unexpected_keys": [], | |
| "mismatched_keys": [], | |
| "error_msgs": [], | |
| } | |
| return model | |
| def _load(self, device: torch.device) -> None: | |
| if self._threads: | |
| torch.set_num_threads(self._threads) | |
| descriptor = self.config.descriptor() | |
| profile = dict(FAMILY_PROFILES[descriptor["runtime_family"]]) | |
| profile.update(MODEL_PROFILES.get(descriptor["model_name"], {})) | |
| if descriptor["runtime_family"] == "vela-encoder": | |
| from .decision1_vela import VelaRuntime | |
| runtime = VelaRuntime.load( | |
| self._source, | |
| descriptor, | |
| max_input_tokens=profile["max_input_tokens"], | |
| device=device, | |
| ) | |
| else: | |
| from .decision1_qwen import QwenRuntime | |
| runtime = QwenRuntime.load( | |
| self._source, | |
| descriptor, | |
| max_input_tokens=profile["max_input_tokens"], | |
| choice_null_description=profile["choice_null_description"], | |
| rocm_conv=profile.get("rocm_conv", False), | |
| device=device, | |
| ) | |
| self._modules.pop("decision", None) | |
| self.decision = runtime.model | |
| self.runtime = runtime | |
| super().train(False) | |
| def _require(self) -> Any: | |
| if self.runtime is None: | |
| raise RuntimeError("Load the model with from_pretrained") | |
| return self.runtime | |
| def model_name(self) -> str: | |
| return self.config.model_name | |
| def max_input_tokens(self) -> int: | |
| return self._require().max_input_tokens | |
| def system_one(self, *, state: Any, questions: dict[str, Any]) -> dict[str, Any]: | |
| """Typed Choice / Noul / Score answers about one state: ``{"model", "answers", "usage"}``. | |
| ``questions`` maps question IDs to ``{"type": "choice" | "noul" | "score", | |
| "instructions": ..., "criteria": ...}``. A malformed question is answered with | |
| ``invalid_question``. As in the native runtime, a request is admitted only when | |
| every question fits the input limit; otherwise each question is answered with | |
| ``max_length_exceeded`` and nothing is truncated. | |
| """ | |
| runtime = self._require() | |
| state = validate_request(state, questions) | |
| answers: dict[str, Any] = {} | |
| rows = [] | |
| for question_id, question in questions.items(): | |
| try: | |
| checked = validate_question(question_id, question) | |
| except DecisionInputError: | |
| answers[question_id] = error_answer(question, "invalid_question") | |
| continue | |
| rows.append( | |
| build_row( | |
| question_id, | |
| state, | |
| checked, | |
| noul_default_false=runtime.noul_default_false, | |
| noul_default_true=runtime.noul_default_true, | |
| noul_explicit_null=runtime.noul_explicit_null, | |
| ) | |
| ) | |
| tokens = 0 | |
| if rows: | |
| try: | |
| probabilities, counts = runtime.predict(rows) | |
| except DecisionInputTooLongError: | |
| for row in rows: | |
| answers[row.question_id] = { | |
| "type": row.type, | |
| "error": "max_length_exceeded", | |
| } | |
| else: | |
| tokens = sum(counts) | |
| for row, values in zip(rows, probabilities): | |
| answers[row.question_id] = answer(row, values) | |
| return { | |
| "model": self.config.model_name, | |
| "answers": {question_id: answers[question_id] for question_id in questions}, | |
| "usage": {"input_tokens": tokens, "output_tokens": 0}, | |
| } | |
| def forward( | |
| self, state: Any = None, questions: dict[str, Any] | None = None | |
| ) -> dict[str, Any]: | |
| return self.system_one(state=state, questions=questions) | |
| def to(self, *args: Any, **kwargs: Any) -> Decision1Model: | |
| """Move to another device by loading the repository there through the same path.""" | |
| device, dtype, _, memory_format = torch._C._nn._parse_to(*args, **kwargs) | |
| if dtype is not None or memory_format is not None: | |
| raise TypeError( | |
| "Decision 1.0 numerics are fixed; only the device can change" | |
| ) | |
| if device is None: | |
| return self | |
| current = next(self.decision.parameters()).device | |
| if device.type == "cuda" and device.index is None: | |
| device = torch.device("cuda", torch.cuda.current_device()) | |
| if device != current: | |
| self._load(device) | |
| return self | |
| def cuda(self, device: Any = None) -> Decision1Model: | |
| if isinstance(device, int): | |
| device = torch.device("cuda", device) | |
| return self.to(device if device is not None else "cuda") | |
| def cpu(self) -> Decision1Model: | |
| return self.to("cpu") | |
| def _cast(self, *args: Any, **kwargs: Any) -> Decision1Model: | |
| raise TypeError( | |
| "Decision 1.0 numerics are fixed; dtype casts are not supported" | |
| ) | |
| def half(self, *args: Any, **kwargs: Any) -> Decision1Model: | |
| return self._cast() | |
| def float(self, *args: Any, **kwargs: Any) -> Decision1Model: | |
| return self._cast() | |
| def bfloat16(self, *args: Any, **kwargs: Any) -> Decision1Model: | |
| return self._cast() | |
| def double(self, *args: Any, **kwargs: Any) -> Decision1Model: | |
| return self._cast() | |
| def train(self, mode: bool = True) -> Decision1Model: | |
| if mode: | |
| raise RuntimeError("Decision 1.0 models are inference-only") | |
| return super().train(False) | |
| def save_pretrained(self, *args: Any, **kwargs: Any) -> None: | |
| raise NotImplementedError( | |
| "The repository itself is the model; copy it with " | |
| "huggingface_hub.snapshot_download(repo_id, local_dir=...)" | |
| ) | |
| def push_to_hub(self, *args: Any, **kwargs: Any) -> None: | |
| raise NotImplementedError( | |
| "Decision 1.0 repositories are published by their release" | |
| ) | |