Spaces:
Sleeping
Sleeping
Arush kumar commited on
Commit ·
f335923
1
Parent(s): 5b68e97
Update veylon_model.py
Browse files- veylon_model.py +12 -2
veylon_model.py
CHANGED
|
@@ -384,11 +384,21 @@ class TransformerBlock(layers.Layer):
|
|
| 384 |
return x + a + f
|
| 385 |
|
| 386 |
def generate_step(self, x, cache_k=None, cache_v=None, cache_pos=0):
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 387 |
attn_out, nck, ncv = self.attn.generate_step(
|
| 388 |
self.norm1(x), cache_k=cache_k, cache_v=cache_v, cache_pos=cache_pos,
|
| 389 |
)
|
| 390 |
-
|
| 391 |
-
x = x +
|
| 392 |
return x, nck, ncv
|
| 393 |
|
| 394 |
def get_config(self):
|
|
|
|
| 384 |
return x + a + f
|
| 385 |
|
| 386 |
def generate_step(self, x, cache_k=None, cache_v=None, cache_pos=0):
|
| 387 |
+
# Must mirror call()'s PARALLEL residual structure exactly:
|
| 388 |
+
# attn and ffn both read from norm(x) computed on the SAME
|
| 389 |
+
# pre-block x, then both get added to that same original x.
|
| 390 |
+
# Previously this was written as a sequential block (x=x+attn;
|
| 391 |
+
# x=x+ffn(norm2(x))) — ffn ended up conditioned on norm2(x+attn_out)
|
| 392 |
+
# instead of norm2(x), an input distribution the FFN weights were
|
| 393 |
+
# never trained on. That silently produced a different model at
|
| 394 |
+
# generation time than at training time: teacher-forced loss (via
|
| 395 |
+
# call()) never exercises this path and looked fine, while actual
|
| 396 |
+
# autoregressive generation compounded the mismatch every step.
|
| 397 |
attn_out, nck, ncv = self.attn.generate_step(
|
| 398 |
self.norm1(x), cache_k=cache_k, cache_v=cache_v, cache_pos=cache_pos,
|
| 399 |
)
|
| 400 |
+
ffn_out = self.ffn(self.norm2(x), training=False)
|
| 401 |
+
x = x + attn_out + ffn_out
|
| 402 |
return x, nck, ncv
|
| 403 |
|
| 404 |
def get_config(self):
|