Spaces:
Running on Zero
Running on Zero
Download run.py from akhaliq/openjev: direct link, hf CLI and curl.
- Browser
- Download file 2.17 kB
-
https://huggingface.co/spaces/akhaliq/openjev/resolve/main/run.py
- Command line
-
hf download hf://spaces/akhaliq/openjev/run.py
-
curl -L -o run.py https://huggingface.co/spaces/akhaliq/openjev/resolve/main/run.py
2.17 kB
| import json | |
| import gradio as gr | |
| import spaces | |
| import torch | |
| from transformers import AutoModelForSequenceClassification, AutoTokenizer | |
| MODEL_ID = "AlexWortega/openjev" | |
| SUBFOLDER = "qwen3.5-4b-nli-v2" | |
| LABELS = ["contradiction", "entailment", "neutral"] | |
| # Load at module level on cuda — ZeroGPU emulates CUDA at startup and | |
| # attaches a real GPU inside @spaces.GPU functions. | |
| tokenizer = AutoTokenizer.from_pretrained(MODEL_ID, subfolder=SUBFOLDER) | |
| model = AutoModelForSequenceClassification.from_pretrained( | |
| MODEL_ID, subfolder=SUBFOLDER, trust_remote_code=True, torch_dtype=torch.bfloat16 | |
| ).to("cuda").eval() | |
| def classify(premise: str, hypothesis: str) -> str: | |
| """Run openjev NLI: returns contradiction / entailment / neutral probabilities.""" | |
| text = model.config.nli_template.format(premise=premise, hypothesis=hypothesis) | |
| inputs = tokenizer(text, return_tensors="pt").to("cuda") | |
| with torch.no_grad(): | |
| probs = model(**inputs).logits.softmax(-1)[0].float().cpu() | |
| result = {label: round(float(p), 4) for label, p in zip(LABELS, probs)} | |
| result["prediction"] = LABELS[int(probs.argmax())] | |
| return json.dumps(result, indent=2) | |
| def rerank(question: str, options: str) -> str: | |
| """Pick the option with the highest entailment against the question. | |
| Options are one per line.""" | |
| best_idx, best_score, scores = -1, -1.0, [] | |
| for i, option in enumerate([o.strip() for o in options.splitlines() if o.strip()]): | |
| text = model.config.nli_template.format(premise=question, hypothesis=option) | |
| inputs = tokenizer(text, return_tensors="pt").to("cuda") | |
| with torch.no_grad(): | |
| probs = model(**inputs).logits.softmax(-1)[0].float().cpu() | |
| score = float(probs[1]) # entailment | |
| scores.append((option, round(score, 4))) | |
| if score > best_score: | |
| best_idx, best_score = i, score | |
| return json.dumps({"answer_index": best_idx, "scores": scores}, indent=2) | |
| demo = gr.Workflow( | |
| graph="workflow.json", | |
| bind={"jev-classify": classify, "jev-rerank": rerank}, | |
| ) | |
| if __name__ == "__main__": | |
| demo.launch() | |