"""Sigmoid head for token-level Quality Estimation. A new unembedding head that sits on top of the last hidden states of a frozen base LM and produces a per-token confidence score via sigmoid (not softmax). Multiple equally-valid tokens can simultaneously have high scores under language ambiguity. Paper: "Sigmoid Head for Quality Estimation under Language Ambiguity" """ import torch from transformers import PreTrainedModel, PretrainedConfig class SigmoidHeadConfig(PretrainedConfig): model_type = "sigmoid_head" def __init__(self, vocab_size: int = 32007, hidden_size: int = 4096, **kwargs): super().__init__(**kwargs) self.vocab_size = vocab_size self.hidden_size = hidden_size class SigmoidHead(PreTrainedModel): config_class = SigmoidHeadConfig def __init__(self, config: SigmoidHeadConfig): super().__init__(config) self.weight = torch.nn.Parameter( torch.empty(config.vocab_size, config.hidden_size) ) self.post_init() @torch.no_grad() def score(self, last_hidden_states: torch.Tensor) -> torch.Tensor: """Per-token confidence in (0, 1). Args: last_hidden_states: [batch, seq_len, hidden_size] Returns: confidence_scores: [batch, seq_len, vocab_size] """ return torch.sigmoid(torch.matmul(last_hidden_states, self.weight.T))