Feature Extraction
sentence-transformers
Safetensors
Model2Vec
static-embeddings
moderation
safety
abuse-detection
lf2
2bit-quantization
cpu-optimized
Instructions to use VTXAI/VTX-MOD-1 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- sentence-transformers
How to use VTXAI/VTX-MOD-1 with sentence-transformers:
from sentence_transformers import SentenceTransformer model = SentenceTransformer("VTXAI/VTX-MOD-1") sentences = [ "The weather is lovely today.", "It's so sunny outside!", "He drove to the stadium." ] embeddings = model.encode(sentences) similarities = model.similarity(embeddings, embeddings) print(similarities.shape) # [3, 3] - Model2Vec
How to use VTXAI/VTX-MOD-1 with Model2Vec:
from model2vec import StaticModel model = StaticModel.from_pretrained("VTXAI/VTX-MOD-1") embeddings = model.encode(["It's dangerous to go alone!", "It's a secret to everybody."]) print(embeddings.shape) - Notebooks
- Google Colab
- Kaggle
Download inference.py from VTXAI/VTX-MOD-1: direct link, hf CLI and curl.
- Browser
- Download file 10.2 kB
-
https://huggingface.co/VTXAI/VTX-MOD-1/resolve/main/inference.py
- Command line
-
hf download hf://VTXAI/VTX-MOD-1/inference.py
-
curl -L -o inference.py https://huggingface.co/VTXAI/VTX-MOD-1/resolve/main/inference.py
10.2 kB
| """JEV-2 Inference Engine with Typed Decision Primitives (Noul, Choice, Score) | |
| Compatible with VTXAI/VTX-JEV-2 and VTXAI/VTX-MOD-1. | |
| Usage: | |
| from inference import JevClient, Choice, Noul, Score | |
| client = JevClient.from_pretrained("VTXAI/VTX-JEV-2") | |
| response = client.system_one( | |
| state="I was charged twice and production is unavailable.", | |
| questions={ | |
| "refund": Noul("Does the customer request a refund?"), | |
| "team": Choice( | |
| "Which team should handle this?", | |
| {"billing": "Payments", "technical": "Production outage"}, | |
| ), | |
| "severity": Score( | |
| "How severe is the impact?", | |
| ["Minor", "Major", "Critical"], | |
| ), | |
| }, | |
| ) | |
| print(response.nouls["refund"].noul) # bool (e.g. True) | |
| print(response.choices["team"].choice) # key (e.g. 'billing') | |
| print(response.scores["severity"].score) # top label or expected rank | |
| print(response.to_dict()) | |
| """ | |
| from __future__ import annotations | |
| import os | |
| import json | |
| import time | |
| from dataclasses import dataclass, field | |
| from pathlib import Path | |
| from typing import Dict, List, Union, Any, Optional | |
| import numpy as np | |
| from safetensors.numpy import load_file | |
| from tokenizers import Tokenizer | |
| from huggingface_hub import hf_hub_download | |
| # --- Decision Primitives --- | |
| class Noul: | |
| """Binary decision / Gating condition (Yes / No).""" | |
| def __init__(self, question: str, positive_anchor: str = "yes positive true confirm affirmative request", negative_anchor: str = "no negative false ignore denial safe", threshold: float = 0.5): | |
| self.question = question | |
| self.positive_anchor = positive_anchor | |
| self.negative_anchor = negative_anchor | |
| self.threshold = threshold | |
| class Choice: | |
| """Categorical routing / Selection among options.""" | |
| def __init__(self, question: str, options: Union[List[str], Dict[str, str]]): | |
| self.question = question | |
| if isinstance(options, list): | |
| self.options = {opt: opt for opt in options} | |
| else: | |
| self.options = options | |
| class Score: | |
| """Ordinal priority / Severity calibration.""" | |
| def __init__(self, question: str, scale: List[str]): | |
| self.question = question | |
| self.scale = scale | |
| # --- Response Containers --- | |
| class NoulResult: | |
| noul: bool | |
| probability: float | |
| confidence: float | |
| def to_dict(self) -> Dict[str, Any]: | |
| return { | |
| "noul": self.noul, | |
| "probability": round(self.probability, 4), | |
| "confidence": round(self.confidence, 4), | |
| } | |
| class ChoiceResult: | |
| choice: str | |
| confidence: float | |
| distribution: Dict[str, float] | |
| def to_dict(self) -> Dict[str, Any]: | |
| return { | |
| "choice": self.choice, | |
| "confidence": round(self.confidence, 4), | |
| "distribution": {k: round(v, 4) for k, v in self.distribution.items()}, | |
| } | |
| class ScoreResult: | |
| score: str | |
| expected_index: float | |
| distribution: Dict[str, float] | |
| def to_dict(self) -> Dict[str, Any]: | |
| return { | |
| "score": self.score, | |
| "expected_index": round(self.expected_index, 4), | |
| "distribution": {k: round(v, 4) for k, v in self.distribution.items()}, | |
| } | |
| class SystemOneResponse: | |
| nouls: Dict[str, NoulResult] = field(default_factory=dict) | |
| choices: Dict[str, ChoiceResult] = field(default_factory=dict) | |
| scores: Dict[str, ScoreResult] = field(default_factory=dict) | |
| latency_ms: float = 0.0 | |
| def to_dict(self) -> Dict[str, Any]: | |
| return { | |
| "nouls": {k: v.to_dict() for k, v in self.nouls.items()}, | |
| "choices": {k: v.to_dict() for k, v in self.choices.items()}, | |
| "scores": {k: v.to_dict() for k, v in self.scores.items()}, | |
| "latency_ms": round(self.latency_ms, 3), | |
| } | |
| # --- JevClient Core --- | |
| class JevClient: | |
| """JEV-2 Ultra-fast System 1 Decision & Routing Client.""" | |
| def __init__(self, weights: np.ndarray, tokenizer: Tokenizer, config: dict): | |
| self.weights = weights # (V, D) normalized token embeddings | |
| self.tokenizer = tokenizer | |
| self.config = config | |
| self.dim = weights.shape[1] | |
| self.vocab_size = weights.shape[0] | |
| def from_pretrained(cls, model_name_or_path: str) -> "JevClient": | |
| """Load JevClient from local folder or Hugging Face Hub.""" | |
| path = Path(model_name_or_path) | |
| if path.exists() and (path / "model.safetensors").exists(): | |
| model_file = str(path / "model.safetensors") | |
| tok_file = str(path / "tokenizer.json") | |
| cfg_file = str(path / "config.json") if (path / "config.json").exists() else None | |
| else: | |
| model_file = hf_hub_download(model_name_or_path, "model.safetensors") | |
| tok_file = hf_hub_download(model_name_or_path, "tokenizer.json") | |
| try: | |
| cfg_file = hf_hub_download(model_name_or_path, "config.json") | |
| except Exception: | |
| cfg_file = None | |
| tensors = load_file(model_file) | |
| weight_key = "embeddings" if "embeddings" in tensors else list(tensors.keys())[0] | |
| weights = tensors[weight_key].astype(np.float32) | |
| # Normalize token embedding table once | |
| norms = np.linalg.norm(weights, axis=-1, keepdims=True) | |
| weights = weights / np.maximum(norms, 1e-12) | |
| tokenizer = Tokenizer.from_file(tok_file) | |
| cfg = json.loads(Path(cfg_file).read_text()) if cfg_file else {} | |
| return cls(weights, tokenizer, cfg) | |
| def encode(self, texts: List[str]) -> np.ndarray: | |
| """Fast vectorized token lookup and mean pooling on CPU.""" | |
| encoded = self.tokenizer.encode_batch(texts) | |
| res = np.zeros((len(texts), self.dim), dtype=np.float32) | |
| for i, item in enumerate(encoded): | |
| ids = [tid for tid in item.ids if 0 <= tid < self.vocab_size] | |
| if ids: | |
| tok_vecs = self.weights[ids] | |
| # Mean pool | |
| pooled = np.mean(tok_vecs, axis=0) | |
| norm = np.linalg.norm(pooled) | |
| if norm > 1e-9: | |
| res[i] = pooled / norm | |
| else: | |
| res[i] = 0.0 | |
| return res | |
| def system_one(self, state: str, questions: Dict[str, Union[Noul, Choice, Score]]) -> SystemOneResponse: | |
| """Execute typed System 1 non-autoregressive decision primitives concurrently.""" | |
| t0 = time.perf_counter() | |
| response = SystemOneResponse() | |
| # Gather texts to encode in a single optimized pass | |
| batch_texts = [state] | |
| query_map = {} | |
| for key, q in questions.items(): | |
| if isinstance(q, Noul): | |
| # Combined query, positive anchor, negative anchor | |
| idx_q = len(batch_texts) | |
| batch_texts.append(f"{q.question} {state}") | |
| idx_pos = len(batch_texts) | |
| batch_texts.append(f"{q.question} {q.positive_anchor}") | |
| idx_neg = len(batch_texts) | |
| batch_texts.append(f"{q.question} {q.negative_anchor}") | |
| query_map[key] = ("noul", q, (idx_q, idx_pos, idx_neg)) | |
| elif isinstance(q, Choice): | |
| idx_start = len(batch_texts) | |
| keys = list(q.options.keys()) | |
| for k in keys: | |
| batch_texts.append(f"{q.question} {q.options[k]}") | |
| query_map[key] = ("choice", q, (idx_start, keys)) | |
| elif isinstance(q, Score): | |
| idx_start = len(batch_texts) | |
| for opt in q.scale: | |
| batch_texts.append(f"{q.question} {opt}") | |
| query_map[key] = ("score", q, (idx_start, q.scale)) | |
| # Single batch CPU encode | |
| all_vecs = self.encode(batch_texts) | |
| state_vec = all_vecs[0] | |
| # Process each primitive result | |
| for key, info in query_map.items(): | |
| q_type = info[0] | |
| if q_type == "noul": | |
| _, q, (idx_q, idx_pos, idx_neg) = info | |
| q_vec = all_vecs[idx_q] | |
| pos_vec = all_vecs[idx_pos] | |
| neg_vec = all_vecs[idx_neg] | |
| sim_pos = float(np.dot(q_vec, pos_vec)) | |
| sim_neg = float(np.dot(q_vec, neg_vec)) | |
| diff = sim_pos - sim_neg | |
| prob = float(1.0 / (1.0 + np.exp(-diff * 12.0))) | |
| decision = bool(prob >= q.threshold) | |
| conf = float(prob if decision else (1.0 - prob)) | |
| response.nouls[key] = NoulResult(noul=decision, probability=prob, confidence=conf) | |
| elif q_type == "choice": | |
| _, q, (idx_start, keys) = info | |
| cand_vecs = all_vecs[idx_start : idx_start + len(keys)] | |
| # Affinities with state and question context | |
| sims = np.dot(cand_vecs, state_vec) | |
| # Softmax with temperature | |
| exp_s = np.exp(sims * 15.0) | |
| probs = exp_s / np.sum(exp_s) | |
| best_idx = int(np.argmax(probs)) | |
| response.choices[key] = ChoiceResult( | |
| choice=keys[best_idx], | |
| confidence=float(probs[best_idx]), | |
| distribution={k: float(p) for k, p in zip(keys, probs)}, | |
| ) | |
| elif q_type == "score": | |
| _, q, (idx_start, scale) = info | |
| cand_vecs = all_vecs[idx_start : idx_start + len(scale)] | |
| sims = np.dot(cand_vecs, state_vec) | |
| exp_s = np.exp(sims * 15.0) | |
| probs = exp_s / np.sum(exp_s) | |
| best_idx = int(np.argmax(probs)) | |
| exp_idx = float(np.sum(np.arange(len(scale)) * probs)) | |
| response.scores[key] = ScoreResult( | |
| score=scale[best_idx], | |
| expected_index=exp_idx, | |
| distribution={k: float(p) for k, p in zip(scale, probs)}, | |
| ) | |
| response.latency_ms = (time.perf_counter() - t0) * 1000 | |
| return response | |