Arush kumar commited on
Commit
f335923
·
1 Parent(s): 5b68e97

Update veylon_model.py

Browse files
Files changed (1) hide show
  1. veylon_model.py +12 -2
veylon_model.py CHANGED
@@ -384,11 +384,21 @@ class TransformerBlock(layers.Layer):
384
  return x + a + f
385
 
386
  def generate_step(self, x, cache_k=None, cache_v=None, cache_pos=0):
 
 
 
 
 
 
 
 
 
 
387
  attn_out, nck, ncv = self.attn.generate_step(
388
  self.norm1(x), cache_k=cache_k, cache_v=cache_v, cache_pos=cache_pos,
389
  )
390
- x = x + attn_out
391
- x = x + self.ffn(self.norm2(x), training=False)
392
  return x, nck, ncv
393
 
394
  def get_config(self):
 
384
  return x + a + f
385
 
386
  def generate_step(self, x, cache_k=None, cache_v=None, cache_pos=0):
387
+ # Must mirror call()'s PARALLEL residual structure exactly:
388
+ # attn and ffn both read from norm(x) computed on the SAME
389
+ # pre-block x, then both get added to that same original x.
390
+ # Previously this was written as a sequential block (x=x+attn;
391
+ # x=x+ffn(norm2(x))) — ffn ended up conditioned on norm2(x+attn_out)
392
+ # instead of norm2(x), an input distribution the FFN weights were
393
+ # never trained on. That silently produced a different model at
394
+ # generation time than at training time: teacher-forced loss (via
395
+ # call()) never exercises this path and looked fine, while actual
396
+ # autoregressive generation compounded the mismatch every step.
397
  attn_out, nck, ncv = self.attn.generate_step(
398
  self.norm1(x), cache_k=cache_k, cache_v=cache_v, cache_pos=cache_pos,
399
  )
400
+ ffn_out = self.ffn(self.norm2(x), training=False)
401
+ x = x + attn_out + ffn_out
402
  return x, nck, ncv
403
 
404
  def get_config(self):