Download source/jev_harness.py from andyshu/opensysone: direct link, hf CLI and curl.
- Browser
- Download file 16.9 kB
-
https://huggingface.co/andyshu/opensysone/resolve/58f289696f58962a8ec98293d7b1abf9fd0c6b8b/source/jev_harness.py
- Command line
-
hf download hf://andyshu/opensysone@58f289696f58962a8ec98293d7b1abf9fd0c6b8b/source/jev_harness.py
-
curl -L -o jev_harness.py https://huggingface.co/andyshu/opensysone/resolve/58f289696f58962a8ec98293d7b1abf9fd0c6b8b/source/jev_harness.py
16.9 kB
| """Shared Jev HTTP schema for local scoring, hosted calls and comparisons. | |
| Reference: https://docs.typesafe.ai/api (checked 2026-09-16). | |
| Local confidence is normalized entropy, not an assertion that TypeSafe uses that | |
| formula. Score is an expectation over rubric indices. Questions are independent. | |
| No remote calls occur unless a caller explicitly chooses the hosted backend. | |
| """ | |
| import argparse | |
| import hmac | |
| from http.server import BaseHTTPRequestHandler, HTTPServer | |
| import json | |
| import math | |
| import os | |
| from pathlib import Path | |
| import sys | |
| import time | |
| import urllib.error | |
| import urllib.parse | |
| import urllib.request | |
| LOCAL_MODEL = "opensysone" | |
| API_REFERENCE = "https://docs.typesafe.ai/api" | |
| def description(value, name): | |
| if isinstance(value, str) and value.strip(): | |
| return value | |
| if isinstance(value, (dict, list)) and value: | |
| return json.dumps(value, ensure_ascii=False, sort_keys=True, allow_nan=False) | |
| raise ValueError(f"{name} must be a nonempty string, object or array") | |
| def compile_request(payload): | |
| if not isinstance(payload, dict): | |
| raise ValueError("request must be an object") | |
| state = description(payload.get("state"), "state") | |
| if not isinstance(payload.get("model"), str) or not payload["model"].strip(): | |
| raise ValueError("model must be a nonempty string") | |
| questions = payload.get("questions") | |
| if not isinstance(questions, dict) or not 1 <= len(questions) <= 64: | |
| raise ValueError("questions must contain 1–64 named questions") | |
| compiled = [] | |
| for name, question in questions.items(): | |
| if not isinstance(name, str) or not name.strip() or not isinstance(question, dict): | |
| raise ValueError("question IDs must be nonempty strings and questions must be objects") | |
| instructions = description(question.get("instructions"), f"{name}.instructions") | |
| kind, criteria = question.get("type"), question.get("criteria") | |
| if kind == "choice": | |
| if not isinstance(criteria, dict) or not 2 <= len(criteria) <= 255: | |
| raise ValueError(f"{name}.criteria must be a map with 2–255 choices") | |
| keys, choices = [], [] | |
| for key, value in criteria.items(): | |
| if not isinstance(key, str) or not key.strip() or (value is not None and not isinstance(value, str)): | |
| raise ValueError(f"{name}: choice keys must be nonempty strings; descriptions string or null") | |
| keys.append(key) | |
| choices.append(key if value is None or not value.strip() else f"{key}: {value}") | |
| elif kind == "score": | |
| if not isinstance(criteria, list) or not 2 <= len(criteria) <= 255: | |
| raise ValueError(f"{name}.criteria must be an ordered array with 2–255 levels") | |
| keys = [str(i) for i in range(len(criteria))] | |
| choices = [description(value, f"{name}.criteria[{i}]") for i, value in enumerate(criteria)] | |
| if len(set(choices)) != len(choices): | |
| raise ValueError(f"{name}: rubric levels must have distinct descriptions") | |
| elif kind == "noul": | |
| keys, choices = ["false", "true"], ["no", "yes"] | |
| if criteria is not None: | |
| if not isinstance(criteria, dict) or set(criteria) - {"true", "false"}: | |
| raise ValueError(f"{name}: noul criteria may describe only true and false") | |
| for key, value in criteria.items(): | |
| if not isinstance(value, str): | |
| raise ValueError(f"{name}: noul criterion descriptions must be strings") | |
| choices = [f"{answer}: {criteria[key]}" if criteria.get(key) else answer | |
| for key, answer in zip(keys, choices)] | |
| else: | |
| raise ValueError(f"{name}: type must be choice, score or noul") | |
| compiled.append((name, kind, keys, choices, | |
| {"state": state, "question": instructions, "choices": choices})) | |
| if sum(len(item[2]) for item in compiled) > 512: | |
| raise ValueError("request exceeds the local 512-candidate allocation budget") | |
| return compiled | |
| def entropy_confidence(probabilities): | |
| entropy = -sum(p * math.log(p) for p in probabilities if p > 0) | |
| return min(1.0, max(0.0, 1.0 - entropy / math.log(len(probabilities)))) | |
| def format_response(compiled, distributions, model=LOCAL_MODEL, input_tokens=0): | |
| if len(compiled) != len(distributions): | |
| raise ValueError("backend returned the wrong number of questions") | |
| answers = {} | |
| for (name, kind, keys, choices, _), probabilities in zip(compiled, distributions): | |
| if len(probabilities) != len(keys) or any(not math.isfinite(p) or not 0 <= p <= 1 for p in probabilities): | |
| raise ValueError("backend returned invalid probabilities") | |
| if abs(sum(probabilities) - 1) > 1e-5: | |
| raise ValueError("backend probabilities must sum to one") | |
| if kind == "noul": | |
| answer = {"type": kind, "noul": probabilities[1]} | |
| else: | |
| answer = {"type": kind, "probabilities": dict(zip(keys, probabilities)), | |
| "confidence": entropy_confidence(probabilities)} | |
| if kind == "choice": | |
| answer["choice"] = keys[max(range(len(keys)), key=probabilities.__getitem__)] | |
| else: | |
| answer["score"] = sum(i * p for i, p in enumerate(probabilities)) | |
| answer["legend"] = dict(zip(keys, choices)) | |
| answers[name] = answer | |
| return {"model": model, "answers": answers, "usage": {"input_tokens": input_tokens, "output_tokens": 0}} | |
| class LocalBackend: | |
| def __init__(self, checkpoint, device="cuda", max_tokens=None): | |
| # Import GPU packages only for the local backend; hosted-only use is stdlib. | |
| import torch | |
| from experiment import load_artifact, guard_memory | |
| guard_memory(device) | |
| self.scorer, artifact = load_artifact(checkpoint, device=device) | |
| if max_tokens is not None: | |
| if not 0 < max_tokens <= self.scorer.lm.config.max_position_embeddings: | |
| raise ValueError('inference token limit must fit the base model context limit') | |
| self.scorer.max_tokens = max_tokens | |
| self.max_tokens = self.scorer.max_tokens | |
| self.scorer.eval() | |
| self.temperature = artifact.get("temperature", 1.0) | |
| self.checkpoint = str(Path(checkpoint).resolve()) | |
| self.calibrated = "temperature" in artifact | |
| self.model_name = "opensysone-" + artifact["model_provenance"]["model_id"].split("/")[-1].lower() | |
| def __call__(self, payload): | |
| import torch | |
| compiled = compile_request(payload) | |
| if payload["model"] not in (LOCAL_MODEL, self.model_name, "jev-latest"): | |
| raise ValueError("Unknown local model; opensysone and jev-latest are client compatibility aliases") | |
| # Validate/encode the entire request before starting GPU work. | |
| encoded = [] | |
| for item in compiled: | |
| row = dict(item[4]) | |
| row["_sequences"] = self.scorer.sequences(row) | |
| encoded.append(row) | |
| probabilities = [] | |
| with torch.inference_mode(): | |
| for row in encoded: | |
| score = self.scorer.score_examples([row])[0] | |
| probabilities.append((score.float() / self.temperature).softmax(0).tolist()) | |
| return format_response(compiled, probabilities, model=self.model_name, | |
| input_tokens=sum(len(ids) for row in encoded for ids in row["_sequences"])) | |
| class RemoteBackend: | |
| def __init__(self, base_url=None, api_key=None, timeout=60, attempts=3): | |
| if not math.isfinite(timeout) or timeout <= 0 or not isinstance(attempts,int) or not 1 <= attempts <= 5: | |
| raise ValueError('timeout must be finite and positive; attempts must be 1–5') | |
| self.base_url = (base_url or os.environ.get("TYPESAFE_BASE_URL", "https://api.typesafe.ai")).rstrip("/") | |
| self.api_key = api_key if api_key is not None else os.environ.get("TYPESAFE_API_KEY") | |
| if not self.api_key: | |
| raise ValueError("TYPESAFE_API_KEY is required for hosted Jev calls") | |
| parsed = urllib.parse.urlsplit(self.base_url) | |
| if parsed.scheme != "https" and not (parsed.scheme == "http" and parsed.hostname in ("127.0.0.1", "localhost", "::1")): | |
| raise ValueError("API keys require HTTPS, or loopback HTTP for a local server") | |
| if parsed.username or parsed.password or parsed.query or parsed.fragment: | |
| raise ValueError("base URL must not contain credentials, query or fragment") | |
| self.timeout, self.attempts = timeout, attempts | |
| def __call__(self, payload): | |
| compile_request(payload) | |
| body = json.dumps(payload, allow_nan=False).encode() | |
| for attempt in range(self.attempts): | |
| request = urllib.request.Request(self.base_url + "/v1/systemone", body, | |
| {"Authorization": "Bearer " + self.api_key, "Content-Type": "application/json"}) | |
| try: | |
| # Do not forward credentials across a vendor redirect. | |
| opener = urllib.request.build_opener(NoRedirect()) | |
| with opener.open(request, timeout=self.timeout) as response: | |
| result = json.load(response) | |
| validate_response(payload, result) | |
| return result | |
| except urllib.error.HTTPError as error: | |
| if error.code not in (429, 500, 502, 503, 504, 529) or attempt + 1 == self.attempts: | |
| raise RuntimeError(f"Jev HTTP {error.code}; request failed") from None | |
| delay = 2 ** attempt | |
| retry = error.headers.get("Retry-After", "") | |
| if retry.isdigit(): | |
| delay = min(30, int(retry)) | |
| time.sleep(delay) | |
| except (urllib.error.URLError, TimeoutError): | |
| if attempt + 1 == self.attempts: | |
| raise RuntimeError("Jev connection failed") from None | |
| time.sleep(2 ** attempt) | |
| class NoRedirect(urllib.request.HTTPRedirectHandler): | |
| def redirect_request(self, *args, **kwargs): | |
| return None | |
| def validate_response(payload, result): | |
| if not isinstance(result, dict) or set(result.get("answers", {})) != set(payload["questions"]): | |
| raise ValueError("API response has mismatched question IDs") | |
| for name, kind, keys, _, _ in compile_request(payload): | |
| answer = result["answers"][name] | |
| if not isinstance(answer, dict) or answer.get("type") != kind: | |
| raise ValueError(f"{name}: wrong answer type") | |
| if kind == "noul": | |
| number = answer.get("noul") | |
| if not isinstance(number, (float, int)) or not math.isfinite(number) or not 0 <= number <= 1: | |
| raise ValueError(f"{name}: invalid noul probability") | |
| else: | |
| probabilities = answer.get("probabilities") | |
| if not isinstance(probabilities, dict) or set(probabilities) != set(keys): | |
| raise ValueError(f"{name}: mismatched probabilities") | |
| values = list(probabilities.values()) | |
| if any(not isinstance(v, (float, int)) or not math.isfinite(v) or not 0 <= v <= 1 for v in values) or abs(sum(values) - 1) > 1e-5: | |
| raise ValueError(f"{name}: invalid distribution") | |
| if kind == "choice" and answer.get("choice") not in keys: | |
| raise ValueError(f"{name}: unknown choice") | |
| if kind == "score": | |
| score = answer.get("score") | |
| if not isinstance(score, (float, int)) or not math.isfinite(score) or not 0 <= score <= len(keys) - 1: | |
| raise ValueError(f"{name}: invalid score") | |
| confidence = answer.get("confidence") | |
| if not isinstance(confidence, (float, int)) or not math.isfinite(confidence) or not 0 <= confidence <= 1: | |
| raise ValueError(f"{name}: invalid confidence") | |
| def create_server(backend, port=18081, api_key=None): | |
| class Handler(BaseHTTPRequestHandler): | |
| def setup(self): | |
| super().setup() | |
| self.connection.settimeout(30) | |
| def send_json(self, status, value): | |
| body = json.dumps(value, allow_nan=False).encode() | |
| self.send_response(status) | |
| self.send_header("Content-Type", "application/json") | |
| self.send_header("Content-Length", str(len(body))) | |
| self.end_headers() | |
| self.wfile.write(body) | |
| def authorized(self): | |
| return api_key is None or hmac.compare_digest(self.headers.get("Authorization", ""), "Bearer " + api_key) | |
| def do_GET(self): | |
| if not self.authorized(): | |
| return self.send_json(401, {"error": "unauthorized"}) | |
| if self.path == "/health": | |
| self.send_json(200, {"status": "ready", "model": getattr(backend,"model_name",LOCAL_MODEL), | |
| "checkpoint": getattr(backend, "checkpoint", None), | |
| "max_tokens": getattr(backend,"max_tokens",None), | |
| "temperature_fitted": getattr(backend, "calibrated", False)}) | |
| elif self.path == "/v1/models": | |
| self.send_json(200, {"models": [{"id": getattr(backend,"model_name",LOCAL_MODEL)}]}) | |
| else: | |
| self.send_json(404, {"error": "not found"}) | |
| def do_POST(self): | |
| if not self.authorized(): | |
| return self.send_json(401, {"error": "unauthorized"}) | |
| if self.path != "/v1/systemone": | |
| return self.send_json(404, {"error": "not found"}) | |
| try: | |
| length = int(self.headers.get("Content-Length", "0")) | |
| if not 0 < length <= 1024 * 1024 or self.headers.get("Transfer-Encoding"): | |
| return self.send_json(413, {"error": "body must be 1 byte–1 MiB with Content-Length"}) | |
| payload = json.loads(self.rfile.read(length)) | |
| value = backend(payload) | |
| self.send_json(200, value) | |
| except (ValueError, TypeError, KeyError) as error: | |
| self.send_json(422, {"error": str(error)}) | |
| except Exception: | |
| # Do not return paths, credentials or source state in a traceback. | |
| self.send_json(500, {"error": "local scoring failed"}) | |
| def log_message(self, format, *args): | |
| # HTTP method/status only, never state or authorization headers. | |
| print(json.dumps({"http": args[0], "status": args[1] if len(args) > 1 else None}), flush=True) | |
| return HTTPServer(("127.0.0.1", port), Handler) | |
| def serve(backend, port=18081, api_key=None): | |
| server = create_server(backend, port, api_key) | |
| print(json.dumps({"listening": f"http://127.0.0.1:{server.server_address[1]}", "pid": os.getpid()}), flush=True) | |
| server.serve_forever() | |
| def main(): | |
| parser = argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument("--backend", choices=["local", "jev", "compare", "serve"], required=True) | |
| parser.add_argument("--request", help="JSON request file, or - for stdin") | |
| parser.add_argument("--checkpoint", help="Trusted project artifact") | |
| parser.add_argument("--device", default="cuda", choices=["cuda", "cpu"]) | |
| parser.add_argument("--base-url") | |
| parser.add_argument("--port", type=int, default=18081) | |
| parser.add_argument("--timeout", type=float, default=60, help="Hosted/HTTP request timeout in seconds") | |
| parser.add_argument("--max-tokens",type=int,help="Local inference complete-candidate token limit; defaults to artifact") | |
| args = parser.parse_args() | |
| local = None | |
| if args.backend != "jev": | |
| if not args.checkpoint: | |
| parser.error("local/compare/serve require --checkpoint") | |
| local = LocalBackend(args.checkpoint, args.device,args.max_tokens) | |
| if args.backend == "serve": | |
| return serve(local, args.port, os.environ.get("OPENSYSONE_API_KEY")) | |
| if not args.request: | |
| parser.error("--request is required") | |
| payload = json.load(sys.stdin) if args.request == "-" else json.loads(Path(args.request).read_text()) | |
| if args.backend == "local": | |
| result = local(payload) | |
| elif args.backend == "jev": | |
| result = RemoteBackend(args.base_url,timeout=args.timeout)(payload) | |
| else: | |
| result = {} | |
| for name, backend in (("local", local), ("jev", RemoteBackend(args.base_url,timeout=args.timeout))): | |
| tick = time.perf_counter() | |
| response = backend(payload) | |
| result[name] = {"response": response, "end_to_end_seconds": time.perf_counter() - tick} | |
| # Model agreement is not a ground-truth quality metric. | |
| print(json.dumps(result, indent=2, allow_nan=False)) | |
| if __name__ == "__main__": | |
| main() | |