fix: safe attention mask and auto-dequantization
Browse files- modeling_cerebellum.py +2 -1
modeling_cerebellum.py
CHANGED
|
@@ -117,7 +117,8 @@ def bidirectional_state_branch_mask_batch(segs, device, dtype=torch.bfloat16):
|
|
| 117 |
allow = allow | torch.eye(L, dtype=torch.bool, device=device)[None]
|
| 118 |
|
| 119 |
mask = torch.zeros((len(segs), 1, L, L), dtype=dtype, device=device)
|
| 120 |
-
|
|
|
|
| 121 |
return mask
|
| 122 |
|
| 123 |
class CerebellumModel(nn.Module):
|
|
|
|
| 117 |
allow = allow | torch.eye(L, dtype=torch.bool, device=device)[None]
|
| 118 |
|
| 119 |
mask = torch.zeros((len(segs), 1, L, L), dtype=dtype, device=device)
|
| 120 |
+
mask_min = -1e4 if dtype == torch.float16 else -1e9
|
| 121 |
+
mask.masked_fill_(~allow[:, None, :, :], mask_min)
|
| 122 |
return mask
|
| 123 |
|
| 124 |
class CerebellumModel(nn.Module):
|