Text Classification
Scikit-learn
sentence-transformers
English
information-retrieval
claim-verification
scifact
evidence-relevance
Eval Results (legacy)
Instructions to use andreiaalexa/scifact-relevance-classifier with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Scikit-learn
How to use andreiaalexa/scifact-relevance-classifier with Scikit-learn:
from huggingface_hub import hf_hub_download import joblib model = joblib.load( hf_hub_download("andreiaalexa/scifact-relevance-classifier", "sklearn_model.joblib") ) # only load pickle files from sources you trust # read more about it here https://skops.readthedocs.io/en/stable/persistence.html - sentence-transformers
How to use andreiaalexa/scifact-relevance-classifier with sentence-transformers:
from sentence_transformers import SentenceTransformer model = SentenceTransformer("andreiaalexa/scifact-relevance-classifier") sentences = [ "The weather is lovely today.", "It's so sunny outside!", "He drove to the stadium." ] embeddings = model.encode(sentences) similarities = model.similarity(embeddings, embeddings) print(similarities.shape) # [3, 3] - Notebooks
- Google Colab
- Kaggle
upload scifact_features.py
Browse files- scifact_features.py +34 -0
scifact_features.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Embedding feature builders for claim-document relevance classification."""
|
| 2 |
+
|
| 3 |
+
from __future__ import annotations
|
| 4 |
+
|
| 5 |
+
import numpy as np
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
def e5_queries(texts: list[str]) -> list[str]:
|
| 9 |
+
return [f"query: {text}" for text in texts]
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
def e5_passages(texts: list[str]) -> list[str]:
|
| 13 |
+
return [f"passage: {text}" for text in texts]
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
def pair_features(model, claims: list[str], documents: list[str], show_progress_bar=False):
|
| 17 |
+
"""Build standard sentence-pair features from two embedding vectors.
|
| 18 |
+
|
| 19 |
+
q and d alone give the classifier raw semantic position. abs(q-d) exposes
|
| 20 |
+
distance dimensions. q*d exposes alignment dimensions. cosine gives a
|
| 21 |
+
single retrieval-style similarity signal.
|
| 22 |
+
"""
|
| 23 |
+
q = model.encode(
|
| 24 |
+
e5_queries(claims),
|
| 25 |
+
normalize_embeddings=True,
|
| 26 |
+
show_progress_bar=show_progress_bar,
|
| 27 |
+
)
|
| 28 |
+
d = model.encode(
|
| 29 |
+
e5_passages(documents),
|
| 30 |
+
normalize_embeddings=True,
|
| 31 |
+
show_progress_bar=show_progress_bar,
|
| 32 |
+
)
|
| 33 |
+
cosine = np.sum(q * d, axis=1, keepdims=True)
|
| 34 |
+
return np.hstack([q, d, np.abs(q - d), q * d, cosine])
|