mkzero commited on
Commit
cb6c24e
·
verified ·
1 Parent(s): def067f

fix: safe attention mask and auto-dequantization

Browse files
Files changed (1) hide show
  1. modeling_cerebellum.py +2 -1
modeling_cerebellum.py CHANGED
@@ -117,7 +117,8 @@ def bidirectional_state_branch_mask_batch(segs, device, dtype=torch.bfloat16):
117
  allow = allow | torch.eye(L, dtype=torch.bool, device=device)[None]
118
 
119
  mask = torch.zeros((len(segs), 1, L, L), dtype=dtype, device=device)
120
- mask.masked_fill_(~allow[:, None, :, :], torch.finfo(dtype).min)
 
121
  return mask
122
 
123
  class CerebellumModel(nn.Module):
 
117
  allow = allow | torch.eye(L, dtype=torch.bool, device=device)[None]
118
 
119
  mask = torch.zeros((len(segs), 1, L, L), dtype=dtype, device=device)
120
+ mask_min = -1e4 if dtype == torch.float16 else -1e9
121
+ mask.masked_fill_(~allow[:, None, :, :], mask_min)
122
  return mask
123
 
124
  class CerebellumModel(nn.Module):