Azrail commited on
Commit
a914078
·
verified ·
1 Parent(s): 9d0bd70

Upload SmalLmForCausalLM

Browse files
Files changed (4) hide show
  1. README.md +1 -0
  2. config.json +5 -1
  3. config.py +0 -2
  4. model.py +2 -65
README.md CHANGED
@@ -2,6 +2,7 @@
2
  library_name: transformers
3
  tags:
4
  - generated_from_trainer
 
5
  model-index:
6
  - name: smallm_70_rope
7
  results: []
 
2
  library_name: transformers
3
  tags:
4
  - generated_from_trainer
5
+ - smallm
6
  model-index:
7
  - name: smallm_70_rope
8
  results: []
config.json CHANGED
@@ -4,6 +4,10 @@
4
  ],
5
  "attention_bias": false,
6
  "attention_dropout": 0.1,
 
 
 
 
7
  "balancing_coef": 0.0001,
8
  "bos_token_id": 1,
9
  "embedding_dropout": 0.0,
@@ -39,7 +43,7 @@
39
  "sliding_window_attention": true,
40
  "sliding_window_context": 1024,
41
  "sliding_window_period": 4,
42
- "static_residual": false,
43
  "token_experts": 3,
44
  "torch_dtype": "float32",
45
  "transformers_version": "4.50.3",
 
4
  ],
5
  "attention_bias": false,
6
  "attention_dropout": 0.1,
7
+ "auto_map": {
8
+ "AutoConfig": "config.SmalLmConfig",
9
+ "AutoModelForCausalLM": "model.SmalLmForCausalLM"
10
+ },
11
  "balancing_coef": 0.0001,
12
  "bos_token_id": 1,
13
  "embedding_dropout": 0.0,
 
43
  "sliding_window_attention": true,
44
  "sliding_window_context": 1024,
45
  "sliding_window_period": 4,
46
+ "static_residual": true,
47
  "token_experts": 3,
48
  "torch_dtype": "float32",
49
  "transformers_version": "4.50.3",
config.py CHANGED
@@ -56,14 +56,12 @@ class SmalLmConfig(PretrainedConfig):
56
  eos_token_id: int = 0,
57
  pad_token_id: int = 0,
58
  static_residual: bool = False,
59
- moe_type: str = "default",
60
  **kwargs,
61
  ):
62
  if positional_bias_type not in ["alibi", "rope"]:
63
  raise ValueError(
64
  f"positional_bias_type must be 'alibi' or 'rope', got {positional_bias_type}"
65
  )
66
- self.moe_type = moe_type
67
  self.static_residual = not static_residual
68
  self.no_moe_layers = no_moe_layers
69
  self.moe_bias = moe_bias
 
56
  eos_token_id: int = 0,
57
  pad_token_id: int = 0,
58
  static_residual: bool = False,
 
59
  **kwargs,
60
  ):
61
  if positional_bias_type not in ["alibi", "rope"]:
62
  raise ValueError(
63
  f"positional_bias_type must be 'alibi' or 'rope', got {positional_bias_type}"
64
  )
 
65
  self.static_residual = not static_residual
66
  self.no_moe_layers = no_moe_layers
67
  self.moe_bias = moe_bias
model.py CHANGED
@@ -10,7 +10,7 @@ from transformers.modeling_outputs import (
10
  from .config import SmalLmConfig
11
  from typing import Optional
12
  import logging
13
- from einops import rearrange, repeat
14
  from transformers.modeling_attn_mask_utils import AttentionMaskConverter
15
  from einops._torch_specific import allow_ops_in_compiled_graph
16
 
@@ -127,68 +127,6 @@ class MoE(nn.Module):
127
  return (out + shared_out).view(shape)
128
 
129
 
130
- class ComboMoe(nn.Module):
131
- def __init__(self, config: SmalLmConfig, *args, **kwargs):
132
- super().__init__(*args, **kwargs)
133
- self.config = config
134
- self.shared_experts = SwiGLU(
135
- config.hidden_size,
136
- config.shared_experts * config.expert_size,
137
- config.moe_bias,
138
- )
139
- self.input_router = Router(config)
140
- self.middle_router = Router(config)
141
- self.out_router = Router(config)
142
- self.routed_experts = nn.ModuleList(
143
- [
144
- nn.Linear(config.hidden_size, config.expert_size, bias=config.moe_bias)
145
- for _ in range(config.routed_experts)
146
- ]
147
- )
148
- self.middle_routed_experts = nn.ModuleList(
149
- [
150
- nn.Linear(config.expert_size, config.hidden_size, bias=config.moe_bias)
151
- for _ in range(config.routed_experts)
152
- ]
153
- )
154
- self.out_routed_experts = nn.ModuleList(
155
- [
156
- nn.Linear(config.expert_size, config.hidden_size, bias=config.moe_bias)
157
- for _ in range(config.routed_experts)
158
- ]
159
- )
160
- self.offset = config.routed_experts
161
-
162
- def forward(self, x: torch.Tensor) -> torch.Tensor:
163
- shape = x.size()
164
- x = x.view(-1, self.config.hidden_size)
165
- iexpert_idx, iexpert_weights, icounts = self.input_router(x)
166
- iout = torch.zeros((*x.shape[:-1], self.config.expert_size), device=x.device)
167
- for i, expert in enumerate(self.routed_experts[: self.offset]):
168
- if icounts[i] == 0:
169
- continue
170
- idx, pos = torch.where(iexpert_idx == i)
171
- iout[idx] += expert(x[idx]) * iexpert_weights[idx, pos, None]
172
-
173
- mexpert_idx, mexpert_weights, mcounts = self.middle_router(x)
174
- for i, expert in enumerate(self.middle_routed_experts):
175
- if mcounts[i] == 0:
176
- continue
177
- idx, pos = torch.where(mexpert_idx == i)
178
- iout[idx] *= F.silu(expert(x[idx]) * mexpert_weights[idx, pos, None])
179
-
180
- out = torch.zeros_like(x)
181
- oexpert_idx, oexpert_weights, ocounts = self.out_router(iout)
182
- for i, expert in enumerate(self.out_routed_experts):
183
- if ocounts[i] == 0:
184
- continue
185
- idx, pos = torch.where(oexpert_idx == i)
186
- out[idx] += expert(iout[idx]) * oexpert_weights[idx, pos, None]
187
-
188
- shared_out = self.shared_experts(x)
189
- return (out + shared_out).view(shape)
190
-
191
-
192
  def build_alibi_bias(config: SmalLmConfig) -> torch.Tensor:
193
  """Build ALiBi for specified number of heads:
194
 
@@ -471,9 +409,8 @@ class Block(nn.Module):
471
  self.dropout1 = nn.Dropout(config.layer_dropout)
472
  self.dropout2 = nn.Dropout(config.layer_dropout)
473
  self.attention = CausalSelfAttention(config, layer_idx)
474
- moe_class = MoE if config.moe_type == "default" else ComboMoe
475
  self.mlp = (
476
- moe_class(config)
477
  if (
478
  config.use_moe
479
  and layer_idx % config.moe_period == 0
 
10
  from .config import SmalLmConfig
11
  from typing import Optional
12
  import logging
13
+ from einops import rearrange
14
  from transformers.modeling_attn_mask_utils import AttentionMaskConverter
15
  from einops._torch_specific import allow_ops_in_compiled_graph
16
 
 
127
  return (out + shared_out).view(shape)
128
 
129
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
130
  def build_alibi_bias(config: SmalLmConfig) -> torch.Tensor:
131
  """Build ALiBi for specified number of heads:
132
 
 
409
  self.dropout1 = nn.Dropout(config.layer_dropout)
410
  self.dropout2 = nn.Dropout(config.layer_dropout)
411
  self.attention = CausalSelfAttention(config, layer_idx)
 
412
  self.mlp = (
413
+ MoE(config)
414
  if (
415
  config.use_moe
416
  and layer_idx % config.moe_period == 0