import json import gradio as gr import spaces import torch from transformers import AutoModelForSequenceClassification, AutoTokenizer MODEL_ID = "AlexWortega/openjev" SUBFOLDER = "qwen3.5-4b-nli-v2" LABELS = ["contradiction", "entailment", "neutral"] # Load at module level on cuda — ZeroGPU emulates CUDA at startup and # attaches a real GPU inside @spaces.GPU functions. tokenizer = AutoTokenizer.from_pretrained(MODEL_ID, subfolder=SUBFOLDER) model = AutoModelForSequenceClassification.from_pretrained( MODEL_ID, subfolder=SUBFOLDER, trust_remote_code=True, torch_dtype=torch.bfloat16 ).to("cuda").eval() @spaces.GPU(duration=60) def classify(premise: str, hypothesis: str) -> str: """Run openjev NLI: returns contradiction / entailment / neutral probabilities.""" text = model.config.nli_template.format(premise=premise, hypothesis=hypothesis) inputs = tokenizer(text, return_tensors="pt").to("cuda") with torch.no_grad(): probs = model(**inputs).logits.softmax(-1)[0].float().cpu() result = {label: round(float(p), 4) for label, p in zip(LABELS, probs)} result["prediction"] = LABELS[int(probs.argmax())] return json.dumps(result, indent=2) @spaces.GPU(duration=60) def rerank(question: str, options: str) -> str: """Pick the option with the highest entailment against the question. Options are one per line.""" best_idx, best_score, scores = -1, -1.0, [] for i, option in enumerate([o.strip() for o in options.splitlines() if o.strip()]): text = model.config.nli_template.format(premise=question, hypothesis=option) inputs = tokenizer(text, return_tensors="pt").to("cuda") with torch.no_grad(): probs = model(**inputs).logits.softmax(-1)[0].float().cpu() score = float(probs[1]) # entailment scores.append((option, round(score, 4))) if score > best_score: best_idx, best_score = i, score return json.dumps({"answer_index": best_idx, "scores": scores}, indent=2) demo = gr.Workflow( graph="workflow.json", bind={"jev-classify": classify, "jev-rerank": rerank}, ) if __name__ == "__main__": demo.launch()