Instructions to use Azrail/smallm_70_rope with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use Azrail/smallm_70_rope with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="Azrail/smallm_70_rope", trust_remote_code=True)# Load model directly from transformers import AutoModelForCausalLM model = AutoModelForCausalLM.from_pretrained("Azrail/smallm_70_rope", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use Azrail/smallm_70_rope with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "Azrail/smallm_70_rope" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Azrail/smallm_70_rope", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/Azrail/smallm_70_rope
- SGLang
How to use Azrail/smallm_70_rope with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "Azrail/smallm_70_rope" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Azrail/smallm_70_rope", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "Azrail/smallm_70_rope" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Azrail/smallm_70_rope", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use Azrail/smallm_70_rope with Docker Model Runner:
docker model run hf.co/Azrail/smallm_70_rope
Upload SmalLmForCausalLM
Browse files
README.md
CHANGED
|
@@ -2,6 +2,7 @@
|
|
| 2 |
library_name: transformers
|
| 3 |
tags:
|
| 4 |
- generated_from_trainer
|
|
|
|
| 5 |
model-index:
|
| 6 |
- name: smallm_70_rope
|
| 7 |
results: []
|
|
|
|
| 2 |
library_name: transformers
|
| 3 |
tags:
|
| 4 |
- generated_from_trainer
|
| 5 |
+
- smallm
|
| 6 |
model-index:
|
| 7 |
- name: smallm_70_rope
|
| 8 |
results: []
|
config.json
CHANGED
|
@@ -4,6 +4,10 @@
|
|
| 4 |
],
|
| 5 |
"attention_bias": false,
|
| 6 |
"attention_dropout": 0.1,
|
|
|
|
|
|
|
|
|
|
|
|
|
| 7 |
"balancing_coef": 0.0001,
|
| 8 |
"bos_token_id": 1,
|
| 9 |
"embedding_dropout": 0.0,
|
|
@@ -39,7 +43,7 @@
|
|
| 39 |
"sliding_window_attention": true,
|
| 40 |
"sliding_window_context": 1024,
|
| 41 |
"sliding_window_period": 4,
|
| 42 |
-
"static_residual":
|
| 43 |
"token_experts": 3,
|
| 44 |
"torch_dtype": "float32",
|
| 45 |
"transformers_version": "4.50.3",
|
|
|
|
| 4 |
],
|
| 5 |
"attention_bias": false,
|
| 6 |
"attention_dropout": 0.1,
|
| 7 |
+
"auto_map": {
|
| 8 |
+
"AutoConfig": "config.SmalLmConfig",
|
| 9 |
+
"AutoModelForCausalLM": "model.SmalLmForCausalLM"
|
| 10 |
+
},
|
| 11 |
"balancing_coef": 0.0001,
|
| 12 |
"bos_token_id": 1,
|
| 13 |
"embedding_dropout": 0.0,
|
|
|
|
| 43 |
"sliding_window_attention": true,
|
| 44 |
"sliding_window_context": 1024,
|
| 45 |
"sliding_window_period": 4,
|
| 46 |
+
"static_residual": true,
|
| 47 |
"token_experts": 3,
|
| 48 |
"torch_dtype": "float32",
|
| 49 |
"transformers_version": "4.50.3",
|
config.py
CHANGED
|
@@ -56,14 +56,12 @@ class SmalLmConfig(PretrainedConfig):
|
|
| 56 |
eos_token_id: int = 0,
|
| 57 |
pad_token_id: int = 0,
|
| 58 |
static_residual: bool = False,
|
| 59 |
-
moe_type: str = "default",
|
| 60 |
**kwargs,
|
| 61 |
):
|
| 62 |
if positional_bias_type not in ["alibi", "rope"]:
|
| 63 |
raise ValueError(
|
| 64 |
f"positional_bias_type must be 'alibi' or 'rope', got {positional_bias_type}"
|
| 65 |
)
|
| 66 |
-
self.moe_type = moe_type
|
| 67 |
self.static_residual = not static_residual
|
| 68 |
self.no_moe_layers = no_moe_layers
|
| 69 |
self.moe_bias = moe_bias
|
|
|
|
| 56 |
eos_token_id: int = 0,
|
| 57 |
pad_token_id: int = 0,
|
| 58 |
static_residual: bool = False,
|
|
|
|
| 59 |
**kwargs,
|
| 60 |
):
|
| 61 |
if positional_bias_type not in ["alibi", "rope"]:
|
| 62 |
raise ValueError(
|
| 63 |
f"positional_bias_type must be 'alibi' or 'rope', got {positional_bias_type}"
|
| 64 |
)
|
|
|
|
| 65 |
self.static_residual = not static_residual
|
| 66 |
self.no_moe_layers = no_moe_layers
|
| 67 |
self.moe_bias = moe_bias
|
model.py
CHANGED
|
@@ -10,7 +10,7 @@ from transformers.modeling_outputs import (
|
|
| 10 |
from .config import SmalLmConfig
|
| 11 |
from typing import Optional
|
| 12 |
import logging
|
| 13 |
-
from einops import rearrange
|
| 14 |
from transformers.modeling_attn_mask_utils import AttentionMaskConverter
|
| 15 |
from einops._torch_specific import allow_ops_in_compiled_graph
|
| 16 |
|
|
@@ -127,68 +127,6 @@ class MoE(nn.Module):
|
|
| 127 |
return (out + shared_out).view(shape)
|
| 128 |
|
| 129 |
|
| 130 |
-
class ComboMoe(nn.Module):
|
| 131 |
-
def __init__(self, config: SmalLmConfig, *args, **kwargs):
|
| 132 |
-
super().__init__(*args, **kwargs)
|
| 133 |
-
self.config = config
|
| 134 |
-
self.shared_experts = SwiGLU(
|
| 135 |
-
config.hidden_size,
|
| 136 |
-
config.shared_experts * config.expert_size,
|
| 137 |
-
config.moe_bias,
|
| 138 |
-
)
|
| 139 |
-
self.input_router = Router(config)
|
| 140 |
-
self.middle_router = Router(config)
|
| 141 |
-
self.out_router = Router(config)
|
| 142 |
-
self.routed_experts = nn.ModuleList(
|
| 143 |
-
[
|
| 144 |
-
nn.Linear(config.hidden_size, config.expert_size, bias=config.moe_bias)
|
| 145 |
-
for _ in range(config.routed_experts)
|
| 146 |
-
]
|
| 147 |
-
)
|
| 148 |
-
self.middle_routed_experts = nn.ModuleList(
|
| 149 |
-
[
|
| 150 |
-
nn.Linear(config.expert_size, config.hidden_size, bias=config.moe_bias)
|
| 151 |
-
for _ in range(config.routed_experts)
|
| 152 |
-
]
|
| 153 |
-
)
|
| 154 |
-
self.out_routed_experts = nn.ModuleList(
|
| 155 |
-
[
|
| 156 |
-
nn.Linear(config.expert_size, config.hidden_size, bias=config.moe_bias)
|
| 157 |
-
for _ in range(config.routed_experts)
|
| 158 |
-
]
|
| 159 |
-
)
|
| 160 |
-
self.offset = config.routed_experts
|
| 161 |
-
|
| 162 |
-
def forward(self, x: torch.Tensor) -> torch.Tensor:
|
| 163 |
-
shape = x.size()
|
| 164 |
-
x = x.view(-1, self.config.hidden_size)
|
| 165 |
-
iexpert_idx, iexpert_weights, icounts = self.input_router(x)
|
| 166 |
-
iout = torch.zeros((*x.shape[:-1], self.config.expert_size), device=x.device)
|
| 167 |
-
for i, expert in enumerate(self.routed_experts[: self.offset]):
|
| 168 |
-
if icounts[i] == 0:
|
| 169 |
-
continue
|
| 170 |
-
idx, pos = torch.where(iexpert_idx == i)
|
| 171 |
-
iout[idx] += expert(x[idx]) * iexpert_weights[idx, pos, None]
|
| 172 |
-
|
| 173 |
-
mexpert_idx, mexpert_weights, mcounts = self.middle_router(x)
|
| 174 |
-
for i, expert in enumerate(self.middle_routed_experts):
|
| 175 |
-
if mcounts[i] == 0:
|
| 176 |
-
continue
|
| 177 |
-
idx, pos = torch.where(mexpert_idx == i)
|
| 178 |
-
iout[idx] *= F.silu(expert(x[idx]) * mexpert_weights[idx, pos, None])
|
| 179 |
-
|
| 180 |
-
out = torch.zeros_like(x)
|
| 181 |
-
oexpert_idx, oexpert_weights, ocounts = self.out_router(iout)
|
| 182 |
-
for i, expert in enumerate(self.out_routed_experts):
|
| 183 |
-
if ocounts[i] == 0:
|
| 184 |
-
continue
|
| 185 |
-
idx, pos = torch.where(oexpert_idx == i)
|
| 186 |
-
out[idx] += expert(iout[idx]) * oexpert_weights[idx, pos, None]
|
| 187 |
-
|
| 188 |
-
shared_out = self.shared_experts(x)
|
| 189 |
-
return (out + shared_out).view(shape)
|
| 190 |
-
|
| 191 |
-
|
| 192 |
def build_alibi_bias(config: SmalLmConfig) -> torch.Tensor:
|
| 193 |
"""Build ALiBi for specified number of heads:
|
| 194 |
|
|
@@ -471,9 +409,8 @@ class Block(nn.Module):
|
|
| 471 |
self.dropout1 = nn.Dropout(config.layer_dropout)
|
| 472 |
self.dropout2 = nn.Dropout(config.layer_dropout)
|
| 473 |
self.attention = CausalSelfAttention(config, layer_idx)
|
| 474 |
-
moe_class = MoE if config.moe_type == "default" else ComboMoe
|
| 475 |
self.mlp = (
|
| 476 |
-
|
| 477 |
if (
|
| 478 |
config.use_moe
|
| 479 |
and layer_idx % config.moe_period == 0
|
|
|
|
| 10 |
from .config import SmalLmConfig
|
| 11 |
from typing import Optional
|
| 12 |
import logging
|
| 13 |
+
from einops import rearrange
|
| 14 |
from transformers.modeling_attn_mask_utils import AttentionMaskConverter
|
| 15 |
from einops._torch_specific import allow_ops_in_compiled_graph
|
| 16 |
|
|
|
|
| 127 |
return (out + shared_out).view(shape)
|
| 128 |
|
| 129 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 130 |
def build_alibi_bias(config: SmalLmConfig) -> torch.Tensor:
|
| 131 |
"""Build ALiBi for specified number of heads:
|
| 132 |
|
|
|
|
| 409 |
self.dropout1 = nn.Dropout(config.layer_dropout)
|
| 410 |
self.dropout2 = nn.Dropout(config.layer_dropout)
|
| 411 |
self.attention = CausalSelfAttention(config, layer_idx)
|
|
|
|
| 412 |
self.mlp = (
|
| 413 |
+
MoE(config)
|
| 414 |
if (
|
| 415 |
config.use_moe
|
| 416 |
and layer_idx % config.moe_period == 0
|