tuanh23's picture
Initial upload: sigmoid QE head for TowerInstruct-7B-v0.2
e6d6bac verified
Raw History Blame Contribute Delete
1.4 kB
"""Sigmoid head for token-level Quality Estimation.
A new unembedding head that sits on top of the last hidden states of a frozen
base LM and produces a per-token confidence score via sigmoid (not softmax).
Multiple equally-valid tokens can simultaneously have high scores under
language ambiguity.
Paper: "Sigmoid Head for Quality Estimation under Language Ambiguity"
"""
import torch
from transformers import PreTrainedModel, PretrainedConfig
class SigmoidHeadConfig(PretrainedConfig):
model_type = "sigmoid_head"
def __init__(self, vocab_size: int = 32007, hidden_size: int = 4096, **kwargs):
super().__init__(**kwargs)
self.vocab_size = vocab_size
self.hidden_size = hidden_size
class SigmoidHead(PreTrainedModel):
config_class = SigmoidHeadConfig
def __init__(self, config: SigmoidHeadConfig):
super().__init__(config)
self.weight = torch.nn.Parameter(
torch.empty(config.vocab_size, config.hidden_size)
)
self.post_init()
@torch.no_grad()
def score(self, last_hidden_states: torch.Tensor) -> torch.Tensor:
"""Per-token confidence in (0, 1).
Args:
last_hidden_states: [batch, seq_len, hidden_size]
Returns:
confidence_scores: [batch, seq_len, vocab_size]
"""
return torch.sigmoid(torch.matmul(last_hidden_states, self.weight.T))