Decision-1.0-Route-0.6B / modeling_decision1.py
Xunzhuo's picture
Load with 🤗 Transformers (trust_remote_code) (#3)
a5b21df
Raw History Blame
13.9 kB
# Copyright 2026 The vLLM Semantic Router Authors.
# SPDX-License-Identifier: Apache-2.0
"""Decision 1.0 model for 🤗 Transformers (``trust_remote_code=True``).
``AutoModel.from_pretrained(repo, trust_remote_code=True)`` downloads the files
named by the repository's ``config.json`` and builds the Decision 1.0 native
inference path from the modules next to this one. ``system_one(state=...,
questions={...})`` answers typed Choice, Noul and Score questions with the System
One response body; there is no text generation and no chat API.
"""
from __future__ import annotations
import os
from pathlib import Path
from typing import Any
import torch
from transformers import PreTrainedModel
from .configuration_decision1 import Decision1Config
from .decision1_system_one import (
DecisionInputError,
DecisionInputTooLongError,
answer,
build_row,
error_answer,
validate_question,
validate_request,
)
HUB_OPTIONS = (
"cache_dir",
"force_download",
"local_files_only",
"proxies",
"revision",
"token",
)
RUNTIME_OPTIONS = ("device", "threads")
# Options of Transformers' own weight loader, which this model does not use.
LOADER_FLAGS = (
"trust_remote_code",
"_from_auto",
"_from_pipeline",
"adapter_kwargs",
"code_revision",
"_commit_hash",
"low_cpu_mem_usage",
"use_safetensors",
"resume_download",
"user_agent",
)
# Input limits and prompt policies of the published models, as the Decision runtime serves them.
FAMILY_PROFILES = {
"vela-encoder": {"max_input_tokens": 1024, "choice_null_description": "render_key"},
"qwen3.5-decision": {
"max_input_tokens": 16384,
"choice_null_description": "preserve_json_null",
},
}
MODEL_PROFILES = {
"Decision-1.0-Nox-4B": {"choice_null_description": "render_key"},
"Decision-1.0-Eos-0.8B": {"rocm_conv": True},
}
__all__ = ["Decision1Model", "DecisionInputError", "DecisionInputTooLongError"]
def _device_name(value: Any) -> str:
if isinstance(value, bool):
raise ValueError(f"Not a device: {value!r}")
if isinstance(value, int):
return "cpu" if value < 0 else f"cuda:{value}"
if isinstance(value, (str, torch.device)):
return str(torch.device(value))
raise ValueError(f"Not a device: {value!r}")
def _device(device: Any, device_map: Any) -> torch.device:
if isinstance(device_map, dict):
if set(device_map) != {""}:
raise ValueError(
"Decision 1.0 models run on one device: pass a device name or {'': device}"
)
device_map = device_map[""]
if device_map == "auto":
device_map = None
names = {_device_name(v) for v in (device, device_map) if v is not None}
if len(names) > 1:
raise ValueError("device and device_map name different devices")
if names:
return torch.device(names.pop())
return torch.device("cuda:0" if torch.cuda.is_available() else "cpu")
def _offline(options: dict[str, Any]) -> dict[str, Any]:
"""With HF_HUB_OFFLINE, read the cache only (huggingface_hub would still list a commit's files)."""
try:
from huggingface_hub import is_offline_mode
offline = is_offline_mode()
except ImportError:
from huggingface_hub import constants
offline = constants.HF_HUB_OFFLINE
return {**options, "local_files_only": True} if offline else options
def _package_dir(
name_or_path: Any, config: Decision1Config, hub: dict[str, Any]
) -> Path:
"""The repository revision as a directory: a local download, or a snapshot in the Hugging Face cache."""
local = Path(os.fspath(name_or_path)).expanduser()
if local.is_dir():
return local.resolve()
from huggingface_hub import snapshot_download
revision = hub.get("revision")
# Weights come from the commit the config came from: Transformers 5.18 passes a
# ResolvedRevision, earlier versions record the commit on the config.
commit = getattr(revision, "resolved", None)
if commit is None and getattr(config, "name_or_path", None) == str(name_or_path):
commit = getattr(config, "_commit_hash", None)
options = _offline(
{
k: v
for k, v in hub.items()
if k != "revision" and v is not None and v is not False
}
)
return Path(
snapshot_download(
str(name_or_path),
revision=commit or revision,
allow_patterns=config.files(),
**options,
)
)
class Decision1Model(PreTrainedModel):
"""A Decision 1.0 model behind System One: ``system_one(state=..., questions={...})``."""
config_class = Decision1Config
base_model_prefix = "decision"
main_input_name = "state"
supports_gradient_checkpointing = False
_no_split_modules: list[str] = []
def __init__(self, config: Decision1Config):
super().__init__(config)
self.runtime = None
self._source: Path | None = None
self._threads: int | None = None
self.post_init()
def _init_weights(self, module: Any) -> None:
"""Every weight comes from the repository files; nothing is initialized here."""
@classmethod
def from_pretrained(
cls,
pretrained_model_name_or_path: str | os.PathLike,
*model_args: Any,
config: Decision1Config | None = None,
**kwargs: Any,
) -> Decision1Model:
"""Load a Hub repository or a local download on one device.
Hub options: ``revision``, ``cache_dir``, ``token``, ``local_files_only``,
``force_download``. ``device`` or ``device_map`` names one device (default:
cuda:0 if a GPU is visible, else CPU); ``threads`` sets CPU threads.
Numerics are the model's own, so ``dtype`` is only None or "auto".
"""
if model_args:
raise TypeError("Decision 1.0 models take no positional model arguments")
hub = {k: kwargs.pop(k) for k in HUB_OPTIONS if k in kwargs}
if kwargs.pop("subfolder", "") not in ("", None):
raise ValueError("A Decision 1.0 repository loads from its root")
options = {k: kwargs.pop(k) for k in RUNTIME_OPTIONS if k in kwargs}
device_map = kwargs.pop("device_map", None)
for key in ("dtype", "torch_dtype"):
if kwargs.pop(key, None) not in (None, "auto"):
raise ValueError(
f"{key}: Decision 1.0 numerics are fixed (FP32 encoders; decoders BF16 on a "
"GPU with an FP32 head, FP32 on CPU); pass None or 'auto'"
)
if kwargs.pop("attn_implementation", None) not in (None, "sdpa"):
raise ValueError("Decision 1.0 models use SDPA attention")
loading_info = kwargs.pop("output_loading_info", False)
for key in LOADER_FLAGS:
kwargs.pop(key, None)
if kwargs:
raise TypeError(
f"Unsupported keyword arguments for a Decision 1.0 model: {sorted(kwargs)}"
)
if config is None:
config = Decision1Config.from_pretrained(
pretrained_model_name_or_path,
**{k: v for k, v in hub.items() if v is not None},
)
model = cls(config)
model._source = _package_dir(pretrained_model_name_or_path, config, hub)
model._threads = options.get("threads")
model._load(_device(options.get("device"), device_map))
model.name_or_path = str(pretrained_model_name_or_path)
if loading_info:
return model, {
"missing_keys": [],
"unexpected_keys": [],
"mismatched_keys": [],
"error_msgs": [],
}
return model
def _load(self, device: torch.device) -> None:
if self._threads:
torch.set_num_threads(self._threads)
descriptor = self.config.descriptor()
profile = dict(FAMILY_PROFILES[descriptor["runtime_family"]])
profile.update(MODEL_PROFILES.get(descriptor["model_name"], {}))
if descriptor["runtime_family"] == "vela-encoder":
from .decision1_vela import VelaRuntime
runtime = VelaRuntime.load(
self._source,
descriptor,
max_input_tokens=profile["max_input_tokens"],
device=device,
)
else:
from .decision1_qwen import QwenRuntime
runtime = QwenRuntime.load(
self._source,
descriptor,
max_input_tokens=profile["max_input_tokens"],
choice_null_description=profile["choice_null_description"],
rocm_conv=profile.get("rocm_conv", False),
device=device,
)
self._modules.pop("decision", None)
self.decision = runtime.model
self.runtime = runtime
super().train(False)
def _require(self) -> Any:
if self.runtime is None:
raise RuntimeError("Load the model with from_pretrained")
return self.runtime
@property
def model_name(self) -> str:
return self.config.model_name
@property
def max_input_tokens(self) -> int:
return self._require().max_input_tokens
def system_one(self, *, state: Any, questions: dict[str, Any]) -> dict[str, Any]:
"""Typed Choice / Noul / Score answers about one state: ``{"model", "answers", "usage"}``.
``questions`` maps question IDs to ``{"type": "choice" | "noul" | "score",
"instructions": ..., "criteria": ...}``. A malformed question is answered with
``invalid_question``. As in the native runtime, a request is admitted only when
every question fits the input limit; otherwise each question is answered with
``max_length_exceeded`` and nothing is truncated.
"""
runtime = self._require()
state = validate_request(state, questions)
answers: dict[str, Any] = {}
rows = []
for question_id, question in questions.items():
try:
checked = validate_question(question_id, question)
except DecisionInputError:
answers[question_id] = error_answer(question, "invalid_question")
continue
rows.append(
build_row(
question_id,
state,
checked,
noul_default_false=runtime.noul_default_false,
noul_default_true=runtime.noul_default_true,
noul_explicit_null=runtime.noul_explicit_null,
)
)
tokens = 0
if rows:
try:
probabilities, counts = runtime.predict(rows)
except DecisionInputTooLongError:
for row in rows:
answers[row.question_id] = {
"type": row.type,
"error": "max_length_exceeded",
}
else:
tokens = sum(counts)
for row, values in zip(rows, probabilities):
answers[row.question_id] = answer(row, values)
return {
"model": self.config.model_name,
"answers": {question_id: answers[question_id] for question_id in questions},
"usage": {"input_tokens": tokens, "output_tokens": 0},
}
def forward(
self, state: Any = None, questions: dict[str, Any] | None = None
) -> dict[str, Any]:
return self.system_one(state=state, questions=questions)
def to(self, *args: Any, **kwargs: Any) -> Decision1Model:
"""Move to another device by loading the repository there through the same path."""
device, dtype, _, memory_format = torch._C._nn._parse_to(*args, **kwargs)
if dtype is not None or memory_format is not None:
raise TypeError(
"Decision 1.0 numerics are fixed; only the device can change"
)
if device is None:
return self
current = next(self.decision.parameters()).device
if device.type == "cuda" and device.index is None:
device = torch.device("cuda", torch.cuda.current_device())
if device != current:
self._load(device)
return self
def cuda(self, device: Any = None) -> Decision1Model:
if isinstance(device, int):
device = torch.device("cuda", device)
return self.to(device if device is not None else "cuda")
def cpu(self) -> Decision1Model:
return self.to("cpu")
def _cast(self, *args: Any, **kwargs: Any) -> Decision1Model:
raise TypeError(
"Decision 1.0 numerics are fixed; dtype casts are not supported"
)
def half(self, *args: Any, **kwargs: Any) -> Decision1Model:
return self._cast()
def float(self, *args: Any, **kwargs: Any) -> Decision1Model:
return self._cast()
def bfloat16(self, *args: Any, **kwargs: Any) -> Decision1Model:
return self._cast()
def double(self, *args: Any, **kwargs: Any) -> Decision1Model:
return self._cast()
def train(self, mode: bool = True) -> Decision1Model:
if mode:
raise RuntimeError("Decision 1.0 models are inference-only")
return super().train(False)
def save_pretrained(self, *args: Any, **kwargs: Any) -> None:
raise NotImplementedError(
"The repository itself is the model; copy it with "
"huggingface_hub.snapshot_download(repo_id, local_dir=...)"
)
def push_to_hub(self, *args: Any, **kwargs: Any) -> None:
raise NotImplementedError(
"Decision 1.0 repositories are published by their release"
)