diff --git a/Dockerfile.runtime b/Dockerfile.runtime deleted file mode 100644 index 3c4c4b158e42012a453272ed954be8d24b4a7cc5..0000000000000000000000000000000000000000 --- a/Dockerfile.runtime +++ /dev/null @@ -1,12 +0,0 @@ -# Public base verified by manifest digest and critical PyTorch file hashes. -# CPU build/import validation is separate from model GPU qualification; see RUNTIME.md. -FROM vllm/vllm-openai-rocm@sha256:1fd21abe66455b4df5a2e83629e97cdcc9d58913b16052d8118b92b239792339 -COPY runtime-fla-requirements.lock /tmp/runtime-fla-requirements.lock -RUN python3 -m pip install --no-cache-dir --no-index --no-deps --require-hashes --target /opt/decision-fla -r /tmp/runtime-fla-requirements.lock -ENV PYTHONPATH=/opt/decision-fla -COPY pyproject.toml /opt/decision-wrapper/pyproject.toml -COPY src/ /opt/decision-wrapper/src/ -RUN python3 -m pip install --no-cache-dir --no-deps --no-build-isolation /opt/decision-wrapper -WORKDIR /model -ENTRYPOINT [] -CMD ["/bin/bash"] diff --git a/MATERIALS.json b/MATERIALS.json deleted file mode 100644 index 5db4c6b40b947924a0f938a4e2475f5da0167063..0000000000000000000000000000000000000000 --- a/MATERIALS.json +++ /dev/null @@ -1,31 +0,0 @@ -{ - "kind": "latest-public-comparison-documents", - "public_models": 15, - "scored_questions": 3766, - "tasks": 54, - "source_statistics_sha256": "8406aea215dc1c4ee2645130c94472b336bb065ba46c6efac4f742589499463b", - "files": { - ".gitattributes": "8ec513d4464879c383554c84c1acff60508f0748470daa16f475d05ac804e2ba", - "DIAGNOSTICS.md": "cc8c0a378013d1cfb87ec775344180047136109cbc236a23da02201c0a7c5baa", - "EVALUATION.md": "10744aeed1a113597e644d0b3f4787a91299f055701f22c8cd34843276a7df2a", - "README.md": "fe6acfe6328b3e1b3b49fc31a77c009c1e42aa492933c53aa29888719c984278", - "SENSITIVITY.md": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5", - "TASKS.md": "573086057e6d87135f119ad95fda16a46f9e8a5e1109430f0a1b8a5da7d99247", - "USAGE.md": "398c56cbb43047629551827fc5c0ba24f4dac2d7b01a5e6e3aeeb85b951706a0", - "WEIGHTING.md": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5", - "assets/decision-matrix.pdf": "3b4f10c3f00820d4b82b11a48401a8c360304c4e7cda90a6e7133c76d18ca41b", - "assets/decision-matrix.png": "cb0c583975fc6848f0d3ca7ed11da48347f7431f91f4debf699b3fb039a0a696", - "assets/decision-matrix.svg": "fc83d1c36264098fce20ca9f000aedc04473c7a16295fa5d16a82afeee2342fc", - "assets/decision-ranking.pdf": "e2d6ecdab0599e65db77b520669e90b32fad6d10c85bb15dcd77bec1ed6ec51f", - "assets/decision-ranking.png": "9bb5e1487f2b66215624e67a72bda0d1c3f4ad6cd931089aa5bd7211f3be02b9", - "assets/decision-ranking.svg": "c8600115c3fafbf03e0b06cc6e3da016e4f74473df4f2d4575ae1c9721c99b8e", - "assets/decision-sol-2b-header.png": "9616b7828774561692b642236d74b72de548e2dfc2ae83f1cae2b70bb3b69b08", - "metrics/benchmark.json": "86dc156356ee80eb849808f6e0d5237157f63d121ae6992c9a833691912ef76c", - "metrics/evaluation-provenance.json": "f9c087bd92e6cdfdc4c4dabc0ec335bb837a79715577a7355012d3a21f81c8e9", - "QUESTION-SCALING.md": "a2aaad17e779958942a02bf5692c871fccc8b2195ea1b68c86e81f7cd45c1355", - "metrics/question-scaling.json": "6b76d2fd372171ed3d08d2bdc960e209bfcb6a20fc3552be272668198917ffe1", - "assets/decision-question-scaling.svg": "e0f5a6c4546c204081fd1ed0a4208c2739b76075b0379b0110d2511c71335930", - "assets/decision-question-scaling.png": "7fe441983a5af90f8832ead12c5f41c30c6181e0d0e537bdc65a3a0c15799b24", - "assets/decision-question-scaling.pdf": "d4eeee05ee71af4cd783d6729da96b863df2487501ac68fb33248f0d67ee0f00" - } -} diff --git a/NORMALIZATION_RUNTIME.md b/NORMALIZATION_RUNTIME.md deleted file mode 100644 index bde3dd98981c3945e3679a2dffd0b82751505ca5..0000000000000000000000000000000000000000 --- a/NORMALIZATION_RUNTIME.md +++ /dev/null @@ -1,5 +0,0 @@ -# Validated normalization runtime - -This Sol bundle installs its recorded FLA normalization profile through the default public entrypoint. It requires the pinned ROCm runtime on gfx942 and a fresh process. The profile covers dimension 128, BF16 input/output, FP32 reciprocal norms, and buckets 1–32 for the actual 16 normalized key heads, batch size at most 8 and complete inputs at most 16,384 tokens. Unknown keys fail closed. Weights, tokenizer, prompt, readout and temperature remain unchanged from the source export. - -No private compilation cache is required. Other FLA kernels retain normal runtime behavior. Numerical validation of this newly bound bundle is still required. Earlier latency measurements do not measure this runtime or candidate. diff --git a/README.md b/README.md index bd298c9e5c935ccc84ac4830b5addea1ac92d15d..f86227103c018289b136fa1e255c8928df8712b9 100644 --- a/README.md +++ b/README.md @@ -7,7 +7,6 @@ tags: - decision-model - classification - qwen3_5 -- custom-code - pytorch - rocm --- @@ -58,73 +57,38 @@ Accuracy (%). Overall weights: Decisions **30%**, Composition **25%**, Reading * ![Capability matrix](assets/decision-matrix.png) -[All 54 tasks](TASKS.md) · [Probability, order and missing-evidence diagnostics](DIAGNOSTICS.md) · [Methods and uncertainty](EVALUATION.md) +[All 54 tasks](evaluation/TASKS.md) · [Probability, order and missing-evidence diagnostics](evaluation/DIAGNOSTICS.md) · [Methods and uncertainty](evaluation/EVALUATION.md) ## More questions, measured ![Question-count latency](assets/decision-question-scaling.png) -Distinct Choice questions at a fixed **499 input tokens per question**. Thirty measurements per point across six independently loaded processes on an otherwise idle AMD gfx942 GPU. Python request latency includes tokenization and inference; loading and network are excluded. [p50, p95 and memory](QUESTION-SCALING.md). +Distinct Choice questions at a fixed **499 input tokens per question**. Thirty measurements per point across six independently loaded processes on an otherwise idle AMD gfx942 GPU. Python request latency includes tokenization and inference; loading and network are excluded. [p50, p95 and memory](evaluation/QUESTION-SCALING.md). -## Use Sol-2B - -Use the [official TypeSafe Python SDK](https://docs.typesafe.ai/sdk/python/usage) with your SystemOne-compatible endpoint, configured to serve `Decision-1.0-Sol-2B`. Replace the example URL and API key with your own. +## Download the complete model repository ```bash -pip install typesafe-sdk +hf download llm-semantic-router/Decision-1.0-Sol-2B --local-dir Decision-1.0-Sol-2B ``` -```python -from typesafe_sdk import Choice, Noul, TypeSafeClient - -with TypeSafeClient( - api_key="YOUR_ENDPOINT_API_KEY", - base_url="https://your-decision-endpoint.example", - model="Decision-1.0-Sol-2B", -) as client: - result = client.system_one( - state="Customer reports a duplicate charge and asks for a refund.", - questions={ - "route": Choice( - instructions="Which team should handle this request?", - criteria={"billing": "Payments and refunds", "technical": "Product faults"}, - ), - "refund_requested": Noul(instructions="Did the customer request a refund?"), - }, - ) - print(result.choices["route"].choice) - print(result.nouls["refund_requested"].noul) -``` +This downloads the complete model release. The root `config.json` lists the backbone, tokenizer, decision head, and calibration files. + +## Serve with vLLM Semantic Router + +This repository contains model data only. Use the vLLM Semantic Router Decision runtime to load `llm-semantic-router/Decision-1.0-Sol-2B` and serve Choice, Noul, and Score requests. The serving implementation and its dependencies live in vLLM Semantic Router; this release does not bundle executable model code. `transformers.AutoModel.from_pretrained` cannot load the custom Decision head directly. -The same request with curl: +After configuring a compatible Decision endpoint, send a [SystemOne request](https://docs.typesafe.ai/api) (replace the placeholder URL and key): ```bash curl -X POST 'https://your-decision-endpoint.example/v1/systemone' \ -H 'Authorization: Bearer YOUR_ENDPOINT_API_KEY' \ -H 'Content-Type: application/json' \ - --data-raw '{ - "model": "Decision-1.0-Sol-2B", - "state": "Customer reports a duplicate charge and asks for a refund.", - "questions": { - "route": { - "type": "choice", - "instructions": "Which team should handle this request?", - "criteria": { - "billing": "Payments and refunds", - "technical": "Product faults" - } - }, - "refund_requested": { - "type": "noul", - "instructions": "Did the customer request a refund?" - } - } -}' + --data-raw '{"model":"Decision-1.0-Sol-2B","state":"Customer requests a refund.","questions":{"route":{"type":"choice","instructions":"Which team should handle this?","criteria":{"billing":"Payments and refunds","technical":"Product faults"}}}}' ``` -[Typed request and response guide](USAGE.md) · [Model runtime requirements](RUNTIME.md) +The Hugging Face repository is a model download, not a hosted inference endpoint. -The complete state, question and candidates must fit 16,384 tokens; overflow is rejected. The bundled normalization profile loads automatically. AMD gfx942 is validated; CPU/MPS are unsupported and NVIDIA is unqualified. Use a fresh Python process when switching profiles. +The published model's complete state, question, and candidates have a 16,384-token input limit. See the [evaluation scope](evaluation/EVALUATION.md) for measured conditions. ## Architecture @@ -132,6 +96,6 @@ The complete state, question and candidates must fit 16,384 tokens; overflow is A causal Qwen3.5 text backbone combines gated linear and full attention. A shared candidate head reads candidate endpoints and the final query vector. Each question uses one forward pass; questions run independently in batches of eight. -[Candidate head](assets/readout.png) · [Vector architecture](assets/architecture.svg) · [Inference code](code/decision_model.py) +[Candidate head](assets/readout.png) · [Vector architecture](assets/architecture.svg) Adapted from [Qwen3.5-2B](https://huggingface.co/Qwen/Qwen3.5-2B). It evaluates supplied evidence without live retrieval; confidence does not guarantee correctness. [License](LICENSE) · [Attributions](ATTRIBUTIONS.md). diff --git a/RUNTIME-RELEASE.json b/RUNTIME-RELEASE.json deleted file mode 100644 index 30583b8ac092da189d39292e8138cd838719ce87..0000000000000000000000000000000000000000 --- a/RUNTIME-RELEASE.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "format": "decision-runtime-patch-v1", - "family": "Sol", - "release_tag": "v1.3.1", - "change_kind": "runtime_only", - "weights_revision": "2412d9470d3aa125b346262dad80f6161847aa09", - "source_bundle_manifest_sha256": "6f0970bfbe5594feecffc14b4824a92080ef09318e12567bd22264c3844e2fc4", - "bundle_manifest_sha256": "1498cc7aa42f5884ab6ca828b5c23f23d4965b7e950a2ef86d530eed99ad78f1", - "changed_inference_files": [ - "code/decision_api.py" - ], - "weights_tokenizer_prompts_calibration_profile_unchanged": true, - "default_public_parity": { - "requests": 2856, - "answers": 3160, - "fixtures": 58, - "raw_logits_probabilities_typed_outputs_exact": true, - "receipt_sha256": "bf686f17361b3fda7f2e46b20bbc331e222ddc7d4ae3bdfea53cb5388782542e" - }, - "timing": { - "receipt_sha256": "501f25efd1813b566d77e98278229440b7a3b4a2358def1462c0171a33e9f3be", - "shape": "distinct fixed-length questions; Q=32; 499 tokens per question", - "reference_ms": 208.943, - "candidate_ms": 203.602, - "same_measured_api_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152", - "no_shared_state_neural_cache_claim": true - }, - "capability_scores_unchanged": true, - "downloaded_package_offline_proof_required_before_promotion": true -} diff --git a/RUNTIME.md b/RUNTIME.md deleted file mode 100644 index 2106cc049a44e2ec2493aa5b84148b32e716492d..0000000000000000000000000000000000000000 --- a/RUNTIME.md +++ /dev/null @@ -1,52 +0,0 @@ -# Public ROCm runtime - -The package has a public, digest-pinned installation path. `Dockerfile.runtime` starts from `vllm/vllm-openai-rocm@sha256:1fd21abe66455b4df5a2e83629e97cdcc9d58913b16052d8118b92b239792339`, adds two hash-checked FLA wheels, and installs this repository's loading wrapper. It keeps the base image's ROCm PyTorch and Triton builds. The vLLM server is not used by Decision inference. - -**Validation boundary:** the public registry manifest, base-image ancestry, package metadata and critical PyTorch binary hashes have been checked. The recipe built successfully and passed CPU imports and real AMD ROCm gfx942 GPU GPU inference for both released bundles. On the packaged three-question Choice/Noul/Score example, its complete responses matched the qualified research runtime exactly; the wrapper matched the direct engine and rejected an oversized complete input. Evidence is in `runtime-build-provenance.json`. This example establishes a working public installation path; it is not a full rerun of the quality or timing benchmark. Published benchmark results use the qualified runtime in each bundle's `runtime.json`. - -## Build and run - -Use an AMD ROCm-compatible Linux host with Docker and the required GPU driver. Check the [AMD PyTorch installation and host prerequisites](https://rocm.docs.amd.com/projects/ai-ecosystem/en/latest/frameworks/pytorch/install.html). CPU and MPS inference are not supported by this model engine; no NVIDIA validation is claimed. - -Run these commands from the downloaded model repository containing `Dockerfile.runtime`, `runtime-fla-requirements.lock`, `pyproject.toml`, `src/`, and the model bundle: - -```bash -docker build --pull -f Dockerfile.runtime -t decision-runtime:1.0 . -mkdir -p runtime-output -docker run --rm \ - --device=/dev/kfd --device=/dev/dri --group-add video --ipc=host \ - -v "$PWD":/model:ro -v "$PWD/runtime-output":/output \ - decision-runtime:1.0 \ - python3 -m decision.example /model --local-files-only --output /output/example.json -``` - -This loads locally without a Hub token. The example tests Choice, Noul and Score through the wrapper and compares them with the frozen direct engine; inspect its actual result rather than assuming a predicted answer. Keep the runtime checks enabled. If they report a mismatch, resolve the cause before using this environment to reproduce benchmark claims. The explicit `allow_unvalidated_runtime=True` option is for separately labeled experiments, not benchmark reproduction. - -The image is large because its public base includes the vLLM development environment. No Decision weights, dataset, user credentials or private runtime image are required to build it. Model weights are mounted at execution time. - -## What is pinned - -The public base locks the existing OS, ROCm libraries and Python dependency environment by content digest. Its exact observed core versions are: - -| Component | Observed value | -|---|---| -| Python | 3.12.13 | -| PyTorch | 2.12.0+git6bbd260 | -| PyTorch commit | 6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5 | -| ROCm userspace / HIP build | 7.2.3 / 7.2.53211 | -| Triton distribution / imported version | 3.7.1+gitf0b55c07 / 3.7.1 | -| Transformers | 5.17.0 | -| Tokenizers / Safetensors | 0.23.2 / 0.8.0 | -| NumPy / Einops | 2.3.5 / 0.8.2 | -| Hugging Face Hub | 1.31.0 | -| FLA core / Flash Linear Attention | 0.5.2 / 0.5.2 | - -`runtime-provenance.json` records the live registry manifest response, 36 shared base layers, selected actual binary SHA256 values, observed versions, wheel URLs and wheel hashes. The FLA wheels were downloaded and their SHA256 values matched the qualified overlay. `runtime-fla-requirements.lock` intentionally covers only that overlay; it is not a standalone dependency lock for an arbitrary system. - -The official [FLA installation guide](https://github.com/fla-org/flash-linear-attention/blob/main/INSTALL.md) separates backend PyTorch installation from FLA and documents `--no-deps` for pre-release/custom Torch builds. This recipe uses that boundary and never asks pip to replace Torch or Triton. Transformers is supplied by the pinned base; see its [official installation documentation](https://huggingface.co/docs/transformers/installation) for the general package installation model. - -## Portability boundary - -The public [PyTorch ROCm 7.2 wheel index](https://download.pytorch.org/whl/rocm7.2/torch/) contains ordinary release wheels such as `2.12.0+rocm7.2`. They are a different artifact from the qualified `2.12.0+git6bbd260` build and are not interchangeable evidence. The latter's [source commit is public](https://github.com/pytorch/pytorch/commit/6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5), but a source commit alone does not reproduce compiler flags, linked libraries and binary behavior. - -An alternative runtime must recheck the installed versions, actual FLA Gated DeltaNet dispatch, BF16 backbone plus FP32 head, fixed prompt rendering, batch size eight, and model output agreement. Runtime changes can shift probabilities near a decision boundary even with identical weights. The supplied engine uses FLA Gated DeltaNet, reference PyTorch causal convolution and SDPA; selecting a different kernel is a new runtime configuration. diff --git a/RUNTIME_BINDING.json b/RUNTIME_BINDING.json deleted file mode 100644 index 2758654dc861d293adc4e9f1865f7ea390d174fc..0000000000000000000000000000000000000000 --- a/RUNTIME_BINDING.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "source_bundle_manifest_sha256": "782fc4a3dc2a003d33f1b388e2db0193ca9253aa48b5c5b99e53e1253274413e", - "profile_sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f", - "profile_validation_receipt_sha256": "be9b75af4c2dec541f565c57566290d6a1d95c85a3ccdb8a3e5c364a8e53218f", - "unchanged_files": [ - { - "file": "backbone/config.json", - "bytes": 1789, - "sha256": "e7bed6f1a1a4d5f029a28040e8687a122b323798d4674666443f6d78e9475a05" - }, - { - "file": "backbone/model.safetensors", - "bytes": 3763685328, - "sha256": "99f97c2564acc1d43b89a1c547b63ce1d87f062637d0a8ed41eb40d752849e36" - }, - { - "file": "chat_template.jinja", - "bytes": 7755, - "sha256": "273d8e0e683b885071fb17e08d71e5f2a5ddfb5309756181681de4f5a1822d80" - }, - { - "file": "code/decision_model.py", - "bytes": 10114, - "sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646" - }, - { - "file": "code/predict.py", - "bytes": 3164, - "sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee" - }, - { - "file": "decision_config.json", - "bytes": 753, - "sha256": "9fa5b80f0df63964410b8498b5473d8cba4aa82e506aae83619d5423c0c0e67a" - }, - { - "file": "decision_head.safetensors", - "bytes": 8424272, - "sha256": "5c3dce45115fba20192c99a32fa89a379c2a6617f02e3647dc8f05e48dcaf215" - }, - { - "file": "temperature.json", - "bytes": 6056, - "sha256": "f0cbe7323441ceaf6abc8dd3f9b1832f5d4121a803f18e9643406aadd53c6efc" - }, - { - "file": "tokenizer.json", - "bytes": 19989325, - "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523" - }, - { - "file": "tokenizer_config.json", - "bytes": 1123, - "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87" - } - ], - "training_selection_calibration_unchanged": true, - "recomputed_or_recalibrated": false, - "private_compiled_cache_required": false, - "new_bundle_offline_proof_required": true -} diff --git a/SERVING_OPTIMIZATION.json b/SERVING_OPTIMIZATION.json deleted file mode 100644 index 00a1a069255d90ab793d99854d679d1990a1229f..0000000000000000000000000000000000000000 --- a/SERVING_OPTIMIZATION.json +++ /dev/null @@ -1,34 +0,0 @@ -{ - "format": "joint-serving-runtime-candidate-v1", - "family": "Sol", - "source_publication_revision": "2412d9470d3aa125b346262dad80f6161847aa09", - "source_bundle_manifest_sha256": "6f0970bfbe5594feecffc14b4824a92080ef09318e12567bd22264c3844e2fc4", - "source_release_manifest_sha256": "5c0a79cb720e8e4e61cef4521cabec3b96e048dc310e54ed80d9735c57f10777", - "source_api_sha256": "1b068eccdffd3c3b67bfa52f8f526e6b482d92551c927668ad28a794767f8a40", - "candidate_api_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152", - "changed_inference_files": [ - "code/decision_api.py" - ], - "held_fixed": [ - "weights", - "tokenizer", - "prompt", - "temperature", - "normalization_profile", - "BF16_backbone_FP32_head", - "batch8", - "input_limit16384", - "public_wrapper" - ], - "shared_state_neural_cache": false, - "cross_request_cache": false, - "timing_evidence": { - "analysis/decoder-joint-serving-v1/COMPLETED-PARITY.json": "4187fb76432eb69b263cb6d5ad55f10a09aef384fe405f3ad204acf1edc761b5", - "analysis/decoder-joint-timing-v1/INDEPENDENTLY-REVIEWED.json": "1b416c8556c2305ffd18bd1e3cd523964820a5e2a3029cf447f20b2829ba82f7", - "results/joint-serving-timing-v1/summary/SUMMARY.json": "501f25efd1813b566d77e98278229440b7a3b4a2358def1462c0171a33e9f3be", - "analysis/decoder4b/joint-timing-actual-review-v1/REVIEW.json": "e94fec9d9c6d487200ef7ea1cf82319b33c4836c08d83f0b856bf8590f2fbee6" - }, - "candidate_default_entrypoint_proof_required": true, - "published": false, - "adoption_authorized": false -} diff --git a/SOURCE_BUNDLE_MANIFEST.json b/SOURCE_BUNDLE_MANIFEST.json deleted file mode 100644 index 802d73801ddd9118b922d5de122665495a4c2770..0000000000000000000000000000000000000000 --- a/SOURCE_BUNDLE_MANIFEST.json +++ /dev/null @@ -1,129 +0,0 @@ -{ - "format": "research-pointer-bundle-v1", - "status": "candidate-export-awaiting-independent-reload-and-quality-gates", - "files": [ - { - "file": "backbone/config.json", - "bytes": 1789, - "sha256": "e7bed6f1a1a4d5f029a28040e8687a122b323798d4674666443f6d78e9475a05" - }, - { - "file": "backbone/model.safetensors", - "bytes": 3763685328, - "sha256": "99f97c2564acc1d43b89a1c547b63ce1d87f062637d0a8ed41eb40d752849e36" - }, - { - "file": "chat_template.jinja", - "bytes": 7755, - "sha256": "273d8e0e683b885071fb17e08d71e5f2a5ddfb5309756181681de4f5a1822d80" - }, - { - "file": "code/decision_api.py", - "bytes": 6724, - "sha256": "147b2fec32cbbbbb1b92cf2a19bb887f9d945e4974e141ca7ea93b8c542f5b21" - }, - { - "file": "code/decision_model.py", - "bytes": 10114, - "sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646" - }, - { - "file": "code/predict.py", - "bytes": 3164, - "sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee" - }, - { - "file": "decision_config.json", - "bytes": 753, - "sha256": "9fa5b80f0df63964410b8498b5473d8cba4aa82e506aae83619d5423c0c0e67a" - }, - { - "file": "decision_head.safetensors", - "bytes": 8424272, - "sha256": "5c3dce45115fba20192c99a32fa89a379c2a6617f02e3647dc8f05e48dcaf215" - }, - { - "file": "runtime.json", - "bytes": 378, - "sha256": "c5d3521358b2817f4e56ea8150c5c612139b4e2b2c5bb09f512d9a7c0b5298b9" - }, - { - "file": "temperature.json", - "bytes": 6056, - "sha256": "f0cbe7323441ceaf6abc8dd3f9b1832f5d4121a803f18e9643406aadd53c6efc" - }, - { - "file": "tokenizer.json", - "bytes": 19989325, - "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523" - }, - { - "file": "tokenizer_config.json", - "bytes": 1123, - "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87" - } - ], - "tensors": [ - { - "file": "backbone/model.safetensors", - "elements": 1881825088, - "elements_by_dtype": { - "BF16": 1881825088 - } - }, - { - "file": "decision_head.safetensors", - "elements": 2105856, - "elements_by_dtype": { - "F32": 2105856 - } - } - ], - "source_checkpoint_files": [ - { - "file": "backbone/config.json", - "bytes": 1788, - "sha256": "071e97d8291168ba712237744c9a60e733acce554e6164322d22550dc96de19b" - }, - { - "file": "backbone/model-00001-of-00002.safetensors", - "bytes": 3999855496, - "sha256": "889eebae42f339b53d110bf17b0c3eb5e691749e614a17ebb8d99a9a8d184f21" - }, - { - "file": "backbone/model-00002-of-00002.safetensors", - "bytes": 3527479552, - "sha256": "db7df0cc1733ba7810d1aafdb74193ab19bdbd0b8ed75a1aaa4033948393e77d" - }, - { - "file": "decision_config.json", - "bytes": 1061, - "sha256": "f0af0ff507fe75eca7644be21b726365c3aba727aabdfcd16616aa85587d282c" - }, - { - "file": "decision_head.safetensors", - "bytes": 8424272, - "sha256": "5c3dce45115fba20192c99a32fa89a379c2a6617f02e3647dc8f05e48dcaf215" - }, - { - "file": "tokenizer.json", - "bytes": 19989325, - "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523" - }, - { - "file": "tokenizer_config.json", - "bytes": 1123, - "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87" - } - ], - "source_model_code_sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646", - "source_api_code_sha256": "147b2fec32cbbbbb1b92cf2a19bb887f9d945e4974e141ca7ea93b8c542f5b21", - "dev_sha256": "45b4cd46acc4be53b95c7ed8f333b3533972296a4d0557c7b8c48c9cab6ced61", - "production_predictions_sha256": "e2b3aa3960ed6811d0eb9ffd23807daabd5eb3550bc42c887691e321de8ebaf6", - "temperature_sha256": "f0cbe7323441ceaf6abc8dd3f9b1832f5d4121a803f18e9643406aadd53c6efc", - "runtime_sha256": "c5d3521358b2817f4e56ea8150c5c612139b4e2b2c5bb09f512d9a7c0b5298b9", - "production_batch_size": 8, - "input_length_limit": 16384, - "original_checkpoint_name": "winner", - "no_publication_performed": true -} diff --git a/USAGE.md b/USAGE.md deleted file mode 100644 index 9c6524eda532c1b6de26b710fc3b69319d50b93f..0000000000000000000000000000000000000000 --- a/USAGE.md +++ /dev/null @@ -1,59 +0,0 @@ -# Use Sol-2B - -The examples below use the [official TypeSafe SDK](https://docs.typesafe.ai/sdk/python/usage) and the standard [SystemOne HTTP request](https://docs.typesafe.ai/api). Configure your endpoint to serve `Decision-1.0-Sol-2B`, then replace the example URL and API key. A Hugging Face model repository is a weights download, not an inference endpoint. - -```bash -pip install typesafe-sdk -``` - -```python -from typesafe_sdk import Choice, Noul, TypeSafeClient - -with TypeSafeClient( - api_key="YOUR_ENDPOINT_API_KEY", - base_url="https://your-decision-endpoint.example", - model="Decision-1.0-Sol-2B", -) as client: - result = client.system_one( - state="Customer reports a duplicate charge and asks for a refund.", - questions={ - "route": Choice( - instructions="Which team should handle this request?", - criteria={"billing": "Payments and refunds", "technical": "Product faults"}, - ), - "refund_requested": Noul(instructions="Did the customer request a refund?"), - }, - ) - print(result.choices["route"].choice) - print(result.nouls["refund_requested"].noul) -``` - -```bash -curl -X POST 'https://your-decision-endpoint.example/v1/systemone' \ - -H 'Authorization: Bearer YOUR_ENDPOINT_API_KEY' \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "model": "Decision-1.0-Sol-2B", - "state": "Customer reports a duplicate charge and asks for a refund.", - "questions": { - "route": { - "type": "choice", - "instructions": "Which team should handle this request?", - "criteria": { - "billing": "Payments and refunds", - "technical": "Product faults" - } - }, - "refund_requested": { - "type": "noul", - "instructions": "Did the customer request a refund?" - } - } -}' -``` - -The state can be text or JSON-compatible structured data. Question IDs and Choice IDs are preserved in the response. Choice uses 2–255 options; Noul returns `noul`, the probability of a condition being true; Score uses 2–10 rubric descriptions ordered from index zero. A request can contain many questions. - -Choice returns `choice`, `probabilities` and `confidence`. Score returns an expected zero-based `score`, a probability distribution and `legend`. Local Decision confidence is normalized maximum probability, `(K × max(p) − 1)/(K − 1)`; it does not reproduce an unpublished provider confidence statistic. An HTTP integration must supply `usage.input_tokens` and `usage.output_tokens`; counting generated tokens as zero is appropriate for this non-generative model, not a claim about provider billing. - -The native runtime processes independent complete questions in batches of eight. Each complete rendered question, including state, instructions and criteria, must fit 16,384 tokens; overflow is rejected. [Runtime and hardware requirements](RUNTIME.md). diff --git a/WEIGHTING.md b/WEIGHTING.md deleted file mode 100644 index 3529addb17dcb361719a418da32687cc96cdf39e..0000000000000000000000000000000000000000 --- a/WEIGHTING.md +++ /dev/null @@ -1,21 +0,0 @@ -# Weight sensitivity - -The current product-priority weights were chosen after observing results. This comparison holds every model and prediction fixed; reweighting is not a training improvement. - -| Model | Current 30/25/15/15/15 | Prior 25/25/15/15/20 | Original four-panel mean | -|---|---:|---:|---:| -| Lux-9B | 77.40 | 77.07 | 79.69 | -| Nox-4B | 73.09 | 72.42 | 75.03 | -| Kev-9B | 71.89 | 72.01 | 73.19 | -| Kev-4B | 70.09 | 70.30 | 71.73 | -| Qwen3.5-9B | 69.73 | 69.70 | 71.99 | -| Decider | 67.71 | 67.97 | 71.75 | -| Qwen3.5-4B | 67.29 | 67.24 | 70.25 | -| Sol-2B | 66.32 | 65.48 | 70.14 | -| Eos-0.8B | 61.89 | 61.19 | 65.99 | -| Kev-0.8B | 58.28 | 58.33 | 59.75 | -| Qwen3.5-2B | 57.24 | 57.20 | 60.54 | -| Kai-0.6B | 53.52 | 53.05 | 55.82 | -| Laya · English | 51.03 | 50.85 | 51.76 | -| Laya · Multilingual | 47.19 | 47.18 | 48.56 | -| Jev | 81.05 | 81.45 | 82.45 | diff --git a/assets/decision-expanded-old_core-600px.png b/assets/decision-expanded-old_core-600px.png deleted file mode 100644 index 014ddf56817eecf4fb91c392d45778c9291fad97..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-old_core-600px.png +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:6d36e595bea7465903ee787d217ef07ad648472d336655e88e51263325d0a7e2 -size 167493 diff --git a/assets/decision-expanded-old_core.pdf b/assets/decision-expanded-old_core.pdf deleted file mode 100644 index 436253de798e442a8abdcd8aa8ce013c6e4a05bb..0000000000000000000000000000000000000000 Binary files a/assets/decision-expanded-old_core.pdf and /dev/null differ diff --git a/assets/decision-expanded-old_core.png b/assets/decision-expanded-old_core.png deleted file mode 100644 index ab21dba4965b334892e3391e48361ba0e4e5234f..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-old_core.png +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2744aafba8afb133b1c92d7b1813d3e769da8babe9439e3d966f94ac28963bbd -size 312133 diff --git a/assets/decision-expanded-old_core.svg b/assets/decision-expanded-old_core.svg deleted file mode 100644 index 32f8dedd8e59ce9fcd3de1480695f5a36e721574..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-old_core.svg +++ /dev/null @@ -1,1620 +0,0 @@ - - - - - - - - 2026-09-22T09:43:52.988078 - image/svg+xml - - - Decision expanded aggregate renderer - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - Sol - - - 2B · v1.3 - - - Nox - - - 4B · v1.3 - - - Kev - - - 9B - - - Decider - - - 2B - - - Qwen - - - 4B · - untuned - - - Qwen - - - 2B · - untuned - - - Laya - - - Upstream - default - - - Jev - - - 1.13.0 · - frontier - - - News classification - - - 83.6 - - - 85.2 - - - 88.3 - - - 86.7 - - - 84.4 - - - 80.5 - - - 91.4 - - - 85.2 - - - Boolean constraints - - - 50.0 - - - 93.8 - - - 92.2 - - - 71.9 - - - 62.5 - - - 37.5 - - - 43.8 - - - 100.0 - - - Entity classification - - - 95.5 - - - 96.4 - - - 98.2 - - - 98.2 - - - 97.3 - - - 92.9 - - - 83.9 - - - 96.4 - - - Intent routing - - - 100.0 - - - 100.0 - - - 100.0 - - - 100.0 - - - 100.0 - - - 96.9 - - - 85.9 - - - 100.0 - - - Evidence placement - - - 99.0 - - - 100.0 - - - 50.0 - - - 8.3 - - - 72.9 - - - 9.4 - - - 95.8 - - - 30.2 - - - Ordered rubric - - - 65.6 - - - 89.1 - - - 96.9 - - - 84.4 - - - 90.6 - - - 81.2 - - - 18.8 - - - 100.0 - - - Relation composition - - - 37.5 - - - 52.1 - - - 52.1 - - - 54.2 - - - 51.0 - - - 49.0 - - - 25.0 - - - 56.2 - - - Scoped evidence - - - 79.2 - - - 78.1 - - - 63.5 - - - 47.9 - - - 51.0 - - - 36.5 - - - 37.5 - - - 89.6 - - - State tracking - - - 27.1 - - - 35.4 - - - 36.5 - - - 29.2 - - - 28.1 - - - 25.0 - - - 24.0 - - - 33.3 - - - In / out of menu - - - 100.0 - - - 100.0 - - - 85.9 - - - 59.4 - - - 60.9 - - - 62.5 - - - 64.1 - - - 100.0 - - - PANEL MEAN - - - 73.7 - - - 83.0 - - - 76.4 - - - 64.0 - - - 69.9 - - - 57.1 - - - 57.0 - - - 79.1 - - - - - - - D E C I S I O N 1 . 0 - - - General decisions - - - 880 decisions · accuracy (%) - - - - - - - - - - - diff --git a/assets/decision-expanded-overview-600px.png b/assets/decision-expanded-overview-600px.png deleted file mode 100644 index bd1b8776926d4fa7fdd06943fe1e8d6155241a1d..0000000000000000000000000000000000000000 Binary files a/assets/decision-expanded-overview-600px.png and /dev/null differ diff --git a/assets/decision-expanded-overview.pdf b/assets/decision-expanded-overview.pdf deleted file mode 100644 index b4948dafcafbd13d552f5fb2c72ca8741317cfdf..0000000000000000000000000000000000000000 Binary files a/assets/decision-expanded-overview.pdf and /dev/null differ diff --git a/assets/decision-expanded-overview.png b/assets/decision-expanded-overview.png deleted file mode 100644 index 5ad5ddac7871eb0072b3d827ca456c19d36c2589..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-overview.png +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:2f12699c29ecde8146482e63925b9e80e3bab4047f79fbec17a87e50a1e64ec1 -size 187355 diff --git a/assets/decision-expanded-overview.svg b/assets/decision-expanded-overview.svg deleted file mode 100644 index 2a8b19f3abf6d5405dc9dddaa1deb8bb8d0c7b3d..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-overview.svg +++ /dev/null @@ -1,804 +0,0 @@ - - - - - - - - 2026-09-22T09:43:52.487083 - image/svg+xml - - - Decision expanded aggregate renderer - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - Sol - - - 2B · v1.3 - - - Nox - - - 4B · v1.3 - - - Kev - - - 9B - - - Decider - - - 2B - - - Qwen - - - 4B · - untuned - - - Qwen - - - 2B · - untuned - - - Laya - - - Upstream - default - - - Jev - - - 1.13.0 · - frontier - - - General decisions - - - 73.7 - - - 83.0 - - - 76.4 - - - 64.0 - - - 69.9 - - - 57.1 - - - 57.0 - - - 79.1 - - - Compositional tasks - - - 46.1 - - - 51.8 - - - 45.5 - - - 46.6 - - - 43.3 - - - 39.0 - - - 37.8 - - - 66.4 - - - Natural reading - - - 76.6 - - - 79.1 - - - 86.7 - - - 92.0 - - - 88.0 - - - 73.8 - - - 51.2 - - - 94.5 - - - Reading and inference - - - 84.2 - - - 86.2 - - - 83.8 - - - 84.4 - - - 79.8 - - - 72.3 - - - 63.7 - - - 89.8 - - - FOUR-PANEL MEAN - - - 70.1 - - - 75.0 - - - 73.1 - - - 71.8 - - - 70.2 - - - 60.5 - - - 52.4 - - - 82.4 - - - - - - - D E C I S I O N 1 . 0 - - - Capability overview - - - 2,720 decisions · accuracy (%) - - - - - - - - - - - diff --git a/assets/decision-expanded-ranking-600px.png b/assets/decision-expanded-ranking-600px.png deleted file mode 100644 index 1b463837047f275ed5ca71d9f26155b041d91c9b..0000000000000000000000000000000000000000 Binary files a/assets/decision-expanded-ranking-600px.png and /dev/null differ diff --git a/assets/decision-expanded-ranking.pdf b/assets/decision-expanded-ranking.pdf deleted file mode 100644 index 5320271adc76281b303baa0aa2a49731fdbcd4c4..0000000000000000000000000000000000000000 Binary files a/assets/decision-expanded-ranking.pdf and /dev/null differ diff --git a/assets/decision-expanded-ranking.png b/assets/decision-expanded-ranking.png deleted file mode 100644 index 56bbd5f0a2f2c9be0b5070a8572e78cf586d5b96..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-ranking.png +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a78d29ea6a73eab02cc8897140614d92b0d7bb4e1295c43c9a2ece60d03ec155 -size 186098 diff --git a/assets/decision-expanded-ranking.svg b/assets/decision-expanded-ranking.svg deleted file mode 100644 index 8885b11311e4944fdb0abbe8389272687703dea1..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-ranking.svg +++ /dev/null @@ -1,501 +0,0 @@ - - - - - - - - 2026-09-22T09:43:52.086108 - image/svg+xml - - - Decision expanded aggregate renderer - - - - - - - - - - - - - - - - - - - - - - - - - - - - - 0 - - - - - - - - - 25 - - - - - - - - - 50 - - - - - - - - - 75 - - - - - - - - - 100 - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - D E C I S I O N 1 . 0 - - - Decision comparison - - - Four-panel mean (%) · 95% confidence intervals - - - MODEL - - - MEAN [95% CI] - - - Sol - - - 2B · v1.3 - - - 70.14% - - - [68.46, 71.80] - - - Nox - - - 4B · v1.3 - - - 75.03% - - - [73.30, 76.71] - - - Kev - - - 9B - - - 73.09% - - - [71.49, 74.68] - - - Decider - - - 2B - - - 71.75% - - - [70.12, 73.37] - - - Qwen3.5 - - - 4B · untuned - - - 70.25% - - - [68.73, 71.76] - - - Qwen3.5 - - - 2B · untuned - - - 60.54% - - - [58.82, 62.31] - - - Laya - - - Upstream default - - - 52.44% - - - [50.66, 54.26] - - - Jev - - - 1.13.0 · frontier - - - 82.45% - - - [81.05, 83.79] - - - - - - - - - - - diff --git a/assets/decision-expanded-v3_core-600px.png b/assets/decision-expanded-v3_core-600px.png deleted file mode 100644 index fef6ee8bf543b8792137780dcbee2cb0befeafff..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-v3_core-600px.png +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:df7099ba80f0a000f7c64cfe6b8aca7ead06d5311e6967a49ae8de32317a2077 -size 168267 diff --git a/assets/decision-expanded-v3_core.pdf b/assets/decision-expanded-v3_core.pdf deleted file mode 100644 index de829ec22c44601ad85e1fae4080d919311df113..0000000000000000000000000000000000000000 Binary files a/assets/decision-expanded-v3_core.pdf and /dev/null differ diff --git a/assets/decision-expanded-v3_core.png b/assets/decision-expanded-v3_core.png deleted file mode 100644 index 91ca4f8e4e984a4410a807134bd27c8b6e142dc3..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-v3_core.png +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:a621a7ffee11dc4909e89a200b7c1207a75bff57a9a237b66e8ed8f2fbac6983 -size 300452 diff --git a/assets/decision-expanded-v3_core.svg b/assets/decision-expanded-v3_core.svg deleted file mode 100644 index 93fbef775c666042f7ad4c92a589f13785fa70fe..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-v3_core.svg +++ /dev/null @@ -1,1621 +0,0 @@ - - - - - - - - 2026-09-22T09:43:53.835405 - image/svg+xml - - - Decision expanded aggregate renderer - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - Sol - - - 2B · v1.3 - - - Nox - - - 4B · v1.3 - - - Kev - - - 9B - - - Decider - - - 2B - - - Qwen - - - 4B · - untuned - - - Qwen - - - 2B · - untuned - - - Laya - - - Upstream - default - - - Jev - - - 1.13.0 · - frontier - - - Record identity - - - 50.0 - - - 53.8 - - - 46.2 - - - 56.2 - - - 50.0 - - - 46.2 - - - 47.5 - - - 76.2 - - - Capacity assignment - - - 47.5 - - - 46.2 - - - 50.0 - - - 51.2 - - - 50.0 - - - 50.0 - - - 63.7 - - - 68.8 - - - Constraint assignment - - - 26.2 - - - 41.2 - - - 25.0 - - - 23.8 - - - 23.8 - - - 21.2 - - - 22.5 - - - 55.0 - - - Intent routing · EN - - - 90.8 - - - 91.7 - - - 86.7 - - - 90.0 - - - 91.7 - - - 70.8 - - - 74.2 - - - 91.7 - - - Intent routing · ZH - - - 87.5 - - - 87.5 - - - 85.0 - - - 88.3 - - - 86.7 - - - 74.2 - - - 73.3 - - - 88.3 - - - Multiset reconciliation - - - 28.7 - - - 28.7 - - - 35.0 - - - 23.8 - - - 27.5 - - - 27.5 - - - 26.2 - - - 58.8 - - - Ordered service loss - - - 25.0 - - - 30.0 - - - 21.2 - - - 22.5 - - - 21.2 - - - 20.0 - - - 20.0 - - - 42.5 - - - Conflicting rule - closure - - - 25.0 - - - 33.8 - - - 28.7 - - - 26.2 - - - 25.0 - - - 27.5 - - - 18.8 - - - 66.2 - - - Temporal exclusion - - - 41.2 - - - 42.5 - - - 32.5 - - - 42.5 - - - 31.2 - - - 28.7 - - - 12.5 - - - 41.2 - - - Transaction recovery - - - 38.8 - - - 62.5 - - - 45.0 - - - 41.2 - - - 26.2 - - - 23.8 - - - 18.8 - - - 75.0 - - - PANEL MEAN - - - 46.1 - - - 51.8 - - - 45.5 - - - 46.6 - - - 43.3 - - - 39.0 - - - 37.8 - - - 66.4 - - - - - - - D E C I S I O N 1 . 0 - - - Compositional tasks - - - 880 decisions · accuracy (%) - - - - - - - - - - - diff --git a/assets/decision-expanded-v4-600px.png b/assets/decision-expanded-v4-600px.png deleted file mode 100644 index 3f56586c44e021ae6e8f7ff4689fd6ee827c0159..0000000000000000000000000000000000000000 Binary files a/assets/decision-expanded-v4-600px.png and /dev/null differ diff --git a/assets/decision-expanded-v4.pdf b/assets/decision-expanded-v4.pdf deleted file mode 100644 index eacc75567ac99b215a1f41e316a07fbe2568ff03..0000000000000000000000000000000000000000 Binary files a/assets/decision-expanded-v4.pdf and /dev/null differ diff --git a/assets/decision-expanded-v4.png b/assets/decision-expanded-v4.png deleted file mode 100644 index 31a63a5bff54bbe29cb9839e7cf82e5eb1ba8a01..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-v4.png +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:287486dbbdd9e8f89ae5ea358d20170ef2496ccf6ac79b3b0b6169990ff14da8 -size 152677 diff --git a/assets/decision-expanded-v4.svg b/assets/decision-expanded-v4.svg deleted file mode 100644 index 3ce023ba3f921e0c6aff35fedf41d2272a7a7afe..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-v4.svg +++ /dev/null @@ -1,668 +0,0 @@ - - - - - - - - 2026-09-22T09:43:54.833008 - image/svg+xml - - - Decision expanded aggregate renderer - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - Sol - - - 2B · v1.3 - - - Nox - - - 4B · v1.3 - - - Kev - - - 9B - - - Decider - - - 2B - - - Qwen - - - 4B · - untuned - - - Qwen - - - 2B · - untuned - - - Laya - - - Upstream - default - - - Jev - - - 1.13.0 · - frontier - - - Yes / no reading - - - 83.1 - - - 86.2 - - - 90.6 - - - 91.9 - - - 83.8 - - - 65.0 - - - 69.4 - - - 92.5 - - - Reading · EN - - - 70.0 - - - 73.1 - - - 83.8 - - - 93.1 - - - 93.1 - - - 83.8 - - - 38.1 - - - 96.9 - - - Reading · ZH - - - 70.0 - - - 70.6 - - - 81.9 - - - 91.2 - - - 91.2 - - - 81.2 - - - 28.1 - - - 96.2 - - - PANEL MEAN - - - 76.6 - - - 79.1 - - - 86.7 - - - 92.0 - - - 88.0 - - - 73.8 - - - 51.2 - - - 94.5 - - - - - - - D E C I S I O N 1 . 0 - - - Natural reading - - - 480 decisions · accuracy (%) - - - - - - - - - - - diff --git a/assets/decision-expanded-v5-600px.png b/assets/decision-expanded-v5-600px.png deleted file mode 100644 index 628512c8f0534b66dac9d2846cb0d91e529e5ffb..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-v5-600px.png +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:8fb899a4acca41fd9cc4cb6885e5e7f6b77a5c962af78f4ad00e99929fb8ce54 -size 102382 diff --git a/assets/decision-expanded-v5.pdf b/assets/decision-expanded-v5.pdf deleted file mode 100644 index a5448c92d93d4872de3560b20cf939850633ddda..0000000000000000000000000000000000000000 Binary files a/assets/decision-expanded-v5.pdf and /dev/null differ diff --git a/assets/decision-expanded-v5.png b/assets/decision-expanded-v5.png deleted file mode 100644 index 191ecd4279e2ecff02b4343af6238bb2c8afb263..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-v5.png +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:bf0d8f84d12b6f1e46ae02a70e0ed42adf45a818d8b4b594b1461ddcb635a13f -size 187722 diff --git a/assets/decision-expanded-v5.svg b/assets/decision-expanded-v5.svg deleted file mode 100644 index 593c88b0b11c57dd0199a0eb07b5b83650edba36..0000000000000000000000000000000000000000 --- a/assets/decision-expanded-v5.svg +++ /dev/null @@ -1,804 +0,0 @@ - - - - - - - - 2026-09-22T09:43:55.456875 - image/svg+xml - - - Decision expanded aggregate renderer - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - Sol - - - 2B · v1.3 - - - Nox - - - 4B · v1.3 - - - Kev - - - 9B - - - Decider - - - 2B - - - Qwen - - - 4B · - untuned - - - Qwen - - - 2B · - untuned - - - Laya - - - Upstream - default - - - Jev - - - 1.13.0 · - frontier - - - Contextual reasoning - - - 69.2 - - - 70.8 - - - 67.5 - - - 66.7 - - - 60.8 - - - 56.7 - - - 30.0 - - - 86.7 - - - Answerability - - - 83.3 - - - 88.3 - - - 80.8 - - - 84.2 - - - 81.7 - - - 82.5 - - - 57.5 - - - 90.8 - - - Textual entailment - - - 88.3 - - - 89.2 - - - 90.0 - - - 90.8 - - - 80.0 - - - 64.2 - - - 72.5 - - - 82.5 - - - Scientific inference - - - 95.8 - - - 96.7 - - - 96.7 - - - 95.8 - - - 96.7 - - - 85.8 - - - 95.0 - - - 99.2 - - - PANEL MEAN - - - 84.2 - - - 86.2 - - - 83.8 - - - 84.4 - - - 79.8 - - - 72.3 - - - 63.7 - - - 89.8 - - - - - - - D E C I S I O N 1 . 0 - - - Reading and inference - - - 480 decisions · accuracy (%) - - - - - - - - - - - diff --git a/assets/decision-family-header.png b/assets/decision-family-header.png deleted file mode 100644 index cd9a02142b9a601f362e25141ba3c590d672f6ac..0000000000000000000000000000000000000000 --- a/assets/decision-family-header.png +++ /dev/null @@ -1,3 +0,0 @@ -version https://git-lfs.github.com/spec/v1 -oid sha256:213511289ce8df038d938ac470e803c427ed57f0f85cc397dd4d79964b866541 -size 2962868 diff --git a/assets/decision-matrix.pdf b/assets/decision-matrix.pdf deleted file mode 100644 index 7baf2a13c6c2103bf17c5fb30a62b391935467f0..0000000000000000000000000000000000000000 Binary files a/assets/decision-matrix.pdf and /dev/null differ diff --git a/assets/decision-matrix.svg b/assets/decision-matrix.svg deleted file mode 100644 index a2be2da08586290d9c2d1392041f434522366997..0000000000000000000000000000000000000000 --- a/assets/decision-matrix.svg +++ /dev/null @@ -1,1338 +0,0 @@ - - - - - - - - 2026-09-23T00:35:01.607165 - image/svg+xml - - - Matplotlib v3.9.4, https://matplotlib.org/ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - Decisions - - - Composition - - - Reading - - - Inference - - - Transfer - - - Mean - - - Jev - - - Hosted frontier - - - 79.10 - - - 66.38 - - - 94.53 - - - 89.79 - - - 87.19 - - - 81.05 - - - Lux-9B - - - Decision family - - - 84.38 - - - 52.75 - - - 90.16 - - - 91.46 - - - 77.72 - - - 77.40 - - - Nox-4B - - - Decision family - - - 83.00 - - - 51.79 - - - 79.06 - - - 86.25 - - - 69.60 - - - 73.09 - - - Kev-9B - - - Qwen3.5 base - - - 76.75 - - - 45.75 - - - 86.72 - - - 83.54 - - - 79.25 - - - 71.89 - - - Kev-4B - - - Qwen3.5 base - - - 71.90 - - - 48.54 - - - 81.88 - - - 84.58 - - - 76.10 - - - 70.09 - - - Qwen3.5-9B - - - Untuned reference - - - 73.91 - - - 44.62 - - - 89.84 - - - 79.58 - - - 73.23 - - - 69.73 - - - Decider - - - 2B - - - 64.01 - - - 46.58 - - - 92.03 - - - 84.38 - - - 69.31 - - - 67.71 - - - Qwen3.5-4B - - - Untuned reference - - - 69.89 - - - 43.33 - - - 87.97 - - - 79.79 - - - 68.83 - - - 67.29 - - - Sol-2B - - - Decision family - - - 73.75 - - - 46.08 - - - 76.56 - - - 84.17 - - - 57.07 - - - 66.32 - - - Eos-0.8B - - - Decision family - - - 65.94 - - - 46.04 - - - 70.31 - - - 81.67 - - - 52.01 - - - 61.89 - - - Kev-0.8B - - - Qwen3.5 base - - - 60.14 - - - 42.29 - - - 67.81 - - - 68.75 - - - 61.19 - - - 58.28 - - - Qwen3.5-2B - - - Untuned reference - - - 57.12 - - - 39.00 - - - 73.75 - - - 72.29 - - - 56.31 - - - 57.24 - - - Kai-0.6B - - - Decision family - - - 57.96 - - - 40.83 - - - 54.69 - - - 69.79 - - - 48.37 - - - 53.52 - - - Laya · English - - - 0.421B encoder - - - 56.54 - - - 35.33 - - - 51.41 - - - 63.75 - - - 53.06 - - - 51.03 - - - Laya · Multilingual - - - 0.322B encoder - - - 47.25 - - - 38.92 - - - 50.78 - - - 57.29 - - - 47.13 - - - 47.19 - - - - Capability matrix - - - Decision-focused accuracy (%) - - - - - - - - diff --git a/assets/decision-question-scaling-600px.png b/assets/decision-question-scaling-600px.png deleted file mode 100644 index dee8e2eee54a6f8e6d087e8aacc48ecc775d7f1a..0000000000000000000000000000000000000000 Binary files a/assets/decision-question-scaling-600px.png and /dev/null differ diff --git a/assets/decision-question-scaling.pdf b/assets/decision-question-scaling.pdf deleted file mode 100644 index 7ad3935b99878c4b4f17cb7960b9465a011ac023..0000000000000000000000000000000000000000 Binary files a/assets/decision-question-scaling.pdf and /dev/null differ diff --git a/assets/decision-question-scaling.svg b/assets/decision-question-scaling.svg deleted file mode 100644 index 489314d125c40c6b02531ce3ffa5e2ccb98f74f1..0000000000000000000000000000000000000000 --- a/assets/decision-question-scaling.svg +++ /dev/null @@ -1,239 +0,0 @@ - - - - - - - - 2026-09-22T18:53:30.157856 - image/svg+xml - - - Matplotlib v3.9.4, https://matplotlib.org/ - - - - - - - - - - - - - - - - - - - - - 1 - - - - - - 2 - - - - - - 4 - - - - - - 8 - - - - - - 16 - - - - - - 32 - - - - Questions per request - - - - - - - - - - 0 - - - - - - - - - 50 - - - - - - - - - 100 - - - - - - - - - 150 - - - - - - - - - 200 - - - - Request latency (ms) - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - p95 - - - - - - - - - p50 - - - - - More questions, one request - - - Sol-2B · fixed 499 tokens per question - - - - - - - - diff --git a/assets/decision-ranking.pdf b/assets/decision-ranking.pdf deleted file mode 100644 index df2b0e93e581e3c9f784ee62166a01c31325bcab..0000000000000000000000000000000000000000 Binary files a/assets/decision-ranking.pdf and /dev/null differ diff --git a/assets/decision-ranking.svg b/assets/decision-ranking.svg deleted file mode 100644 index 0dea2bee45503a4e3a9c760492b82a53a419c26b..0000000000000000000000000000000000000000 --- a/assets/decision-ranking.svg +++ /dev/null @@ -1,369 +0,0 @@ - - - - - - - - 2026-09-23T00:35:01.176315 - image/svg+xml - - - Matplotlib v3.9.4, https://matplotlib.org/ - - - - - - - - - - - - - - - - - - - - - - - - 0% - - - - - - - - - 25% - - - - - - - - - 50% - - - - - - - - - 75% - - - - - - - - - 100% - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - 81.05% - - - 77.40% - - - 73.09% - - - 71.89% - - - 70.09% - - - 69.73% - - - 67.71% - - - 67.29% - - - 66.32% - - - 61.89% - - - 58.28% - - - 57.24% - - - 53.52% - - - 51.03% - - - 47.19% - - - - Decision models - - - Decision-focused accuracy (%) - - - Decision family - - - Jev - - - Hosted frontier - - - Lux-9B - - - Decision family - - - Nox-4B - - - Decision family - - - Kev-9B - - - Qwen3.5 base - - - Kev-4B - - - Qwen3.5 base - - - Qwen3.5-9B - - - Untuned reference - - - Decider - - - 2B - - - Qwen3.5-4B - - - Untuned reference - - - Sol-2B - - - Decision family - - - Eos-0.8B - - - Decision family - - - Kev-0.8B - - - Qwen3.5 base - - - Qwen3.5-2B - - - Untuned reference - - - Kai-0.6B - - - Decision family - - - Laya · English - - - 0.421B encoder - - - Laya · Multilingual - - - 0.322B encoder - - - - - - - - diff --git a/bundle-manifest.json b/bundle-manifest.json deleted file mode 100644 index 4f5ea1f71b47608e586c50990a4e27929b0c91bb..0000000000000000000000000000000000000000 --- a/bundle-manifest.json +++ /dev/null @@ -1,194 +0,0 @@ -{ - "format": "research-pointer-bundle-v1", - "status": "joint-serving-candidate-awaiting-default-public-parity", - "files": [ - { - "file": "NORMALIZATION_RUNTIME.md", - "bytes": 754, - "sha256": "b3fbb8b37961676c15b37790f73cf28c27fb3d2eb844b3d6e467b55637b441bf" - }, - { - "file": "RUNTIME_BINDING.json", - "bytes": 2069, - "sha256": "b3d582d738223adea3cd33b8be4079d3ea4e57a2f5322128e374793f9a61f820" - }, - { - "file": "SERVING_OPTIMIZATION.json", - "bytes": 1548, - "sha256": "09ba5f0c60a57c5fed1592dcd559c3e33e4e4bd948ace3a2e036f2c4e15f383e" - }, - { - "file": "SOURCE_BUNDLE_MANIFEST.json", - "bytes": 4194, - "sha256": "782fc4a3dc2a003d33f1b388e2db0193ca9253aa48b5c5b99e53e1253274413e" - }, - { - "file": "backbone/config.json", - "bytes": 1789, - "sha256": "e7bed6f1a1a4d5f029a28040e8687a122b323798d4674666443f6d78e9475a05" - }, - { - "file": "backbone/model.safetensors", - "bytes": 3763685328, - "sha256": "99f97c2564acc1d43b89a1c547b63ce1d87f062637d0a8ed41eb40d752849e36" - }, - { - "file": "chat_template.jinja", - "bytes": 7755, - "sha256": "273d8e0e683b885071fb17e08d71e5f2a5ddfb5309756181681de4f5a1822d80" - }, - { - "file": "code/decision_api.py", - "bytes": 10952, - "sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152" - }, - { - "file": "code/decision_model.py", - "bytes": 10114, - "sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646" - }, - { - "file": "code/predict.py", - "bytes": 3164, - "sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee" - }, - { - "file": "code/profile_guard.py", - "bytes": 6131, - "sha256": "1603c39038ff783b9d5a5f69110d1695bbb7e28accffd0c03525854e8f258452" - }, - { - "file": "code/runtime_profile.py", - "bytes": 2857, - "sha256": "afb59dda5e4c3890ee5c971549799e69f2e4fa2eafe5893de2168299fd368bda" - }, - { - "file": "decision_config.json", - "bytes": 753, - "sha256": "9fa5b80f0df63964410b8498b5473d8cba4aa82e506aae83619d5423c0c0e67a" - }, - { - "file": "decision_head.safetensors", - "bytes": 8424272, - "sha256": "5c3dce45115fba20192c99a32fa89a379c2a6617f02e3647dc8f05e48dcaf215" - }, - { - "file": "pyproject.toml", - "bytes": 436, - "sha256": "135a9516e87ca2fb41ef54a974e29132256b6c2d528a1cbdea1001e28f306946" - }, - { - "file": "runtime-profile/l2norm_fwd_kernel.json", - "bytes": 13210, - "sha256": "a67f3b4624edc07e2c3f9f1d05553c654a3ff005c6b96a1831d62e323dea9edc" - }, - { - "file": "runtime-profile/profile.json", - "bytes": 18101, - "sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f" - }, - { - "file": "runtime.json", - "bytes": 1112, - "sha256": "8cd739efce9cee371cda1d3b769413a3c0842968fe2b4d511cebab11f071af85" - }, - { - "file": "src/decision/__init__.py", - "bytes": 167, - "sha256": "70de37df98b6fc8e3b9f9d43935ba31a32350496214c11adf8c5fba72c433313" - }, - { - "file": "src/decision/example.py", - "bytes": 3581, - "sha256": "a54dec885f92c2d38d07ef2333dff51965be52bc119f772cc74200ed314e69b6" - }, - { - "file": "src/decision/model.py", - "bytes": 9415, - "sha256": "ba240d7493fc29203fe036966f0ab911200cd4a9252b50b05423409977639ee0" - }, - { - "file": "temperature.json", - "bytes": 6056, - "sha256": "f0cbe7323441ceaf6abc8dd3f9b1832f5d4121a803f18e9643406aadd53c6efc" - }, - { - "file": "tokenizer.json", - "bytes": 19989325, - "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523" - }, - { - "file": "tokenizer_config.json", - "bytes": 1123, - "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87" - } - ], - "tensors": [ - { - "file": "backbone/model.safetensors", - "elements": 1881825088, - "elements_by_dtype": { - "BF16": 1881825088 - } - }, - { - "file": "decision_head.safetensors", - "elements": 2105856, - "elements_by_dtype": { - "F32": 2105856 - } - } - ], - "source_checkpoint_files": [ - { - "file": "backbone/config.json", - "bytes": 1788, - "sha256": "071e97d8291168ba712237744c9a60e733acce554e6164322d22550dc96de19b" - }, - { - "file": "backbone/model-00001-of-00002.safetensors", - "bytes": 3999855496, - "sha256": "889eebae42f339b53d110bf17b0c3eb5e691749e614a17ebb8d99a9a8d184f21" - }, - { - "file": "backbone/model-00002-of-00002.safetensors", - "bytes": 3527479552, - "sha256": "db7df0cc1733ba7810d1aafdb74193ab19bdbd0b8ed75a1aaa4033948393e77d" - }, - { - "file": "decision_config.json", - "bytes": 1061, - "sha256": "f0af0ff507fe75eca7644be21b726365c3aba727aabdfcd16616aa85587d282c" - }, - { - "file": "decision_head.safetensors", - "bytes": 8424272, - "sha256": "5c3dce45115fba20192c99a32fa89a379c2a6617f02e3647dc8f05e48dcaf215" - }, - { - "file": "tokenizer.json", - "bytes": 19989325, - "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523" - }, - { - "file": "tokenizer_config.json", - "bytes": 1123, - "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87" - } - ], - "source_model_code_sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646", - "source_api_code_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152", - "dev_sha256": "45b4cd46acc4be53b95c7ed8f333b3533972296a4d0557c7b8c48c9cab6ced61", - "production_predictions_sha256": "e2b3aa3960ed6811d0eb9ffd23807daabd5eb3550bc42c887691e321de8ebaf6", - "temperature_sha256": "f0cbe7323441ceaf6abc8dd3f9b1832f5d4121a803f18e9643406aadd53c6efc", - "runtime_sha256": "8cd739efce9cee371cda1d3b769413a3c0842968fe2b4d511cebab11f071af85", - "production_batch_size": 8, - "input_length_limit": 16384, - "original_checkpoint_name": "winner", - "no_publication_performed": true, - "source_bundle_manifest_sha256": "782fc4a3dc2a003d33f1b388e2db0193ca9253aa48b5c5b99e53e1253274413e", - "normalization_profile_sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f", - "public_wrapper_included": true, - "source_published_bundle_manifest_sha256": "6f0970bfbe5594feecffc14b4824a92080ef09318e12567bd22264c3844e2fc4", - "serving_optimization_sha256": "09ba5f0c60a57c5fed1592dcd559c3e33e4e4bd948ace3a2e036f2c4e15f383e" -} diff --git a/code/decision_api.py b/code/decision_api.py deleted file mode 100644 index 6619eb00a296890ba4418b0d0ac0e5c8cb572120..0000000000000000000000000000000000000000 --- a/code/decision_api.py +++ /dev/null @@ -1,198 +0,0 @@ -"""Typed local inference adapter for the research decision checkpoints. - -The response schema resembles TypeSafe's primitives. Confidence uses this -implementation's documented normalized maximum probability, not a claimed -reimplementation of TypeSafe's unpublished statistic. No text generation. -""" -from __future__ import annotations -import importlib.util -import math -from pathlib import Path - - -def prepare_runtime_profile(checkpoint, device='cuda:0'): - # This profile is verified before any dependency import can choose kernels. - import hashlib, json, sys - root=Path(checkpoint);runtime=json.loads((root/'runtime.json').read_text()) - spec=runtime.get('normalization_profile') - if spec is None: - if '_decision_process_normalization_profile_v1' in sys.modules: - raise RuntimeError('Use separate processes for profiled and unprofiled models') - return None - import torch - target=torch.device(device) - if target.type!='cuda' or not torch.cuda.is_available(): - raise RuntimeError('The bound profile requires a ROCm CUDA device') - arch=getattr(torch.cuda.get_device_properties(target),'gcnArchName','').split(':')[0] - if arch!=spec['validated_arch']: - raise RuntimeError('Target GPU architecture does not match the bound profile: '+arch) - relative=Path(spec['loader_file']) - if relative.is_absolute() or '..' in relative.parts:raise ValueError('Unsafe profile loader path') - path=root/relative - if hashlib.sha256(path.read_bytes()).hexdigest()!=spec['loader_sha256']: - raise ValueError('Bound runtime profile loader changed') - definition=importlib.util.spec_from_file_location('decision_bundle_runtime_profile',path) - module=importlib.util.module_from_spec(definition);definition.loader.exec_module(module) - return module.ensure_profile(root) - - -def question_row(state, name, question): - kind=question.get('type') - if kind not in {'choice','noul','score'}:raise ValueError('Unknown question type') - if 'instructions' not in question:raise ValueError('instructions is required') - criteria=question.get('criteria') - if kind=='noul': - criteria={} if criteria is None else criteria - if not isinstance(criteria,dict) or set(criteria)-{'true','false'}: - raise ValueError('noul criteria may contain only true and false') - options=[{'key':'false','description':criteria.get('false','The answer to the question is no.')}, - {'key':'true','description':criteria.get('true','The answer to the question is yes.')}] - elif kind=='score': - if not isinstance(criteria,list) or not 2<=len(criteria)<=10: - raise ValueError('score requires an ordered list of 2..10 criteria') - options=[{'key':str(i),'description':value} for i,value in enumerate(criteria)] - else: - if not isinstance(criteria,dict) or not 2<=len(criteria)<=255: - raise ValueError('choice requires a mapping of 2..255 criteria') - if not all(isinstance(k,str) for k in criteria):raise ValueError('Choice keys must be strings') - options=[{'key':key,'description':value} for key,value in criteria.items()] - # The question name is used for bookkeeping only; encoders never render id. - return {'id':name,'state':state,'instructions':question['instructions'], - 'options':options,'task_type':kind,'family':'inference'} - - -def typed_answer(row, probabilities): - p=[float(v) for v in probabilities];k=len(row['options']) - if len(p)!=k or any(not math.isfinite(v) or v<0 for v in p): - raise ValueError('Invalid probability vector') - total=sum(p) - if total<=0 or abs(total-1)>1e-4:raise ValueError('Probabilities must sum to one') - p=[v/total for v in p];selected=max(range(k),key=p.__getitem__) - kind=row['task_type'] - if kind=='noul': - keys=[o['key'] for o in row['options']] - if set(keys)!={'false','true'}:raise ValueError('Native noul rows require false/true keys') - return {'type':'noul','noul':p[keys.index('true')]} - answer={'type':kind,'probabilities':{o['key']:v for o,v in zip(row['options'],p)}, - 'confidence':max(0.,min(1.,(k*max(p)-1)/(k-1)))} - if kind=='choice':answer['choice']=row['options'][selected]['key'] - else: - if [o['key'] for o in row['options']] != [str(i) for i in range(k)]: - raise ValueError('Native score rows require ordered numeric level keys') - answer['score']=sum(i*v for i,v in enumerate(p)) - answer['legend']={str(i):o['description'] for i,o in enumerate(row['options'])} - return answer - - -class DecisionEngine: - def __init__(self, checkpoint, model_code, *, device='cuda:0', max_length=16384, - batch_size=8, temperatures=None, model_name='local-decision-research'): - self.normalization_profile=prepare_runtime_profile(checkpoint, device=device) - import torch - path=Path(model_code)/'decision_model.py' - spec=importlib.util.spec_from_file_location('research_decision_runtime',path) - module=importlib.util.module_from_spec(spec);spec.loader.exec_module(module) - model,tokenizer=module.DecisionModel.from_checkpoint(checkpoint,dtype=torch.bfloat16) - self.model=model.to(device).eval();self.tokenizer=tokenizer;self.module=module - self.device=device;self.max_length=max_length;self.batch_size=batch_size - self.temperatures=temperatures or {};self.model_name=model_name - if batch_size<1 or max_length<1:raise ValueError('Positive batch_size/max_length required') - if any(not math.isfinite(v) or v<=0 for v in self.temperatures.values()): - raise ValueError('Temperatures must be finite positive numbers') - - def predict_rows(self, rows): - import torch - encoded=encode_request(rows,self.tokenizer,self.module,self.max_length) - pad=self.tokenizer.pad_token_id if self.tokenizer.pad_token_id is not None else self.tokenizer.eos_token_id - records=[] - with torch.inference_mode(): - for start in range(0,len(rows),self.batch_size): - items=encoded[start:start+self.batch_size] - batch={key:value.to(self.device) if torch.is_tensor(value) else value - for key,value in self.module.collate(items,pad).items()} - with torch.autocast('cuda',dtype=torch.bfloat16):logits=self.model(**batch) - # Preserve each row's original float/temperature/softmax math, - # but defer host synchronization until the complete batch. - staged=[];transfers=[] - for row,item,values in zip(rows[start:start+self.batch_size],items,logits): - k=len(row['options']);values=values[:k].float() - temperature=self.temperatures.get(row['task_type'],1.) - probabilities=(values/temperature).softmax(-1) - staged.append((row,item,k,temperature)) - transfers.extend((values,probabilities)) - host_values=torch.cat(transfers).tolist() - offset=0 - for row,item,k,temperature in staged: - values=host_values[offset:offset+k] - probabilities=host_values[offset+k:offset+2*k] - offset+=2*k - answer=typed_answer(row,probabilities) - prediction=max(range(k),key=probabilities.__getitem__) - if row['task_type']=='noul': - chosen='true' if answer['noul']>=.5 else 'false' - prediction=[o['key'] for o in row['options']].index(chosen) - rec={'id':row['id'],'status':'ok','prediction':prediction, - 'probabilities':probabilities,'logits':values,'temperature':temperature, - 'native_contract':True,'truncated':False,'input_tokens':len(item['ids']), - 'prompt_sha256':item['prompt_sha256'],'answer':answer} - if row['task_type']=='noul':rec['native_noul']=answer['noul'] - if row['task_type']=='score':rec['native_score']=answer['score'] - records.append(rec) - return records - - def decide(self, state, questions): - if not isinstance(questions,dict) or not questions: - raise ValueError('questions must be a nonempty mapping') - if not all(isinstance(name,str) for name in questions):raise ValueError('Question names must be strings') - rows=[question_row(state,name,q) for name,q in questions.items()] - result=self.predict_rows(rows) - return {'model':self.model_name,'answers':{r['id']:r['answer'] for r in result}, - 'usage':{'input_tokens':sum(r['input_tokens'] for r in result),'scored_questions':len(result)}} - - -"""Experimental request-local exact-segment tokenization. - -Original encode/segments functions remain authoritative. Batch tokenize exact -whole segments, never split a BPE prefix at a new boundary. No cross-request -cache, GPU change, prompt change or change to the eight-row inference groups. -""" - - -class SegmentLookup: - def __init__(self, tokenizer, cache): - self.tokenizer, self.cache = tokenizer, cache - - def encode(self, text, **kwargs): - if kwargs == {'add_special_tokens': False} and text in self.cache: - # Original encode extends its prefix list in place. - return list(self.cache[text]) - return self.tokenizer.encode(text, **kwargs) - - -def encode_request(rows, tokenizer, module, max_length=16384, - max_cached_characters=8_000_000, segment_batch_size=64): - if max_cached_characters < 0 or segment_batch_size < 1: - raise ValueError('Invalid tokenizer resource bound') - unique = {} - characters = 0 - for row in rows: - prefix, options, suffix = module.segments(row) - for segment in (prefix, *options, suffix): - if segment not in unique: - unique[segment] = None - characters += len(segment) - if characters > max_cached_characters: - # Preserve the original behavior under the resource cap. - return [module.encode(r, tokenizer, max_length) for r in rows] - strings = list(unique) - for start in range(0, len(strings), segment_batch_size): - batch = strings[start:start + segment_batch_size] - result = tokenizer(batch, add_special_tokens=False, padding=False, - truncation=False, return_attention_mask=False, - return_token_type_ids=False)['input_ids'] - if len(result) != len(batch): - raise ValueError('Batch tokenizer output count differs') - for segment, ids in zip(batch, result): - unique[segment] = tuple(ids) - lookup = SegmentLookup(tokenizer, unique) - return [module.encode(row, lookup, max_length) for row in rows] diff --git a/code/decision_model.py b/code/decision_model.py deleted file mode 100644 index 6b13fc95f31332587d5024b872c837e3c2984dca..0000000000000000000000000000000000000000 --- a/code/decision_model.py +++ /dev/null @@ -1,173 +0,0 @@ -"""Dynamic candidate readout over a causal Qwen3.5 text backbone. - -Candidate endpoints retain their contextual vectors. A final global-query -vector can incorporate all options before a shared bilinear + MLP scorer -scores every candidate. This is a research architecture, not a Jev claim. -""" -import hashlib -import json -import math -from pathlib import Path - -import torch -from torch import nn -import torch.nn.functional as F -from safetensors.torch import load_file, save_file -from transformers import AutoTokenizer, Qwen3_5ForConditionalGeneration -from transformers.models.qwen3_5.modeling_qwen3_5 import Qwen3_5TextModel - -PROMPT_VERSION = "structured-segmented-candidate-endpoints-global-query-v2" -MAX_OPTIONS = 255 - - -def canonical(value): - return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":")) - - -def payload(value): - return value if isinstance(value, str) else canonical(value) - - -def segments(row): - opts = row["options"] - if not 2 <= len(opts) <= MAX_OPTIONS: - raise ValueError(f"{row['id']}: expected 2..255 options") - if not all(isinstance(o["key"], str) for o in opts): - raise ValueError("Option keys must be strings") - if len({o["key"] for o in opts}) != len(opts): - raise ValueError("Duplicate option keys") - prefix = f"Context:\n{payload(row['state'])}\n\nTask type: {row.get('task_type', 'choice')}\nQuestion:\n{payload(row['instructions'])}\nOptions:" - # Tokenize each part separately. This deliberately fixes boundaries and - # avoids guessing endpoint indices from merged BPE character offsets. - options = ["\n" for o in opts] - suffix = "\n\nSelect the single option best supported by the context and instructions.\nDecision:" - return prefix, options, suffix - - -def render(row): - prefix, opts, suffix = segments(row) - return prefix + "".join(opts) + suffix - - -def encode(row, tokenizer, max_length=16384): - prefix, opts, suffix = segments(row) - ids = tokenizer.encode(prefix, add_special_tokens=False) - candidate_positions = [] - for option in opts: - part = tokenizer.encode(option, add_special_tokens=False) - if not part: - raise ValueError("Empty tokenized candidate") - ids.extend(part) - candidate_positions.append(len(ids) - 1) - ids.extend(tokenizer.encode(suffix, add_special_tokens=False)) - if len(ids) > max_length: - raise ValueError(f"{row['id']}: {len(ids)} tokens exceeds max_length={max_length}; no truncation allowed") - label = row.get("label", -1) - if label != -1 and not 0 <= label < len(opts): - raise ValueError("Invalid label") - prompt = prefix + "".join(opts) + suffix - return {"id": row["id"], "ids": ids, "label": label, "nopts": len(opts), "family": row.get("family", "unspecified"), "candidate_positions": candidate_positions, "query_position": len(ids) - 1, "target_probs": row.get("target_probs"), "prompt_sha256": hashlib.sha256(prompt.encode()).hexdigest(), "token_ids_sha256": hashlib.sha256(canonical(ids).encode()).hexdigest(), "segmented_tokenization": True} - - -def collate(items, pad_id): - length = ((max(len(x["ids"]) for x in items) + 31) // 32) * 32 - nopts = max(x["nopts"] for x in items) - ids = torch.full((len(items), length), pad_id, dtype=torch.long) - mask = torch.zeros_like(ids) - positions = torch.zeros((len(items), nopts), dtype=torch.long) - candidate_mask = torch.zeros((len(items), nopts), dtype=torch.bool) - for i, item in enumerate(items): - if len(item["candidate_positions"]) != item["nopts"]: - raise ValueError("Candidate count does not match endpoint count") - if not all(0 <= p < item["query_position"] < len(item["ids"]) for p in item["candidate_positions"]): - raise ValueError("Candidate endpoints must precede global query") - if len(set(item["candidate_positions"])) != item["nopts"]: - raise ValueError("Duplicate candidate endpoint") - ids[i, :len(item["ids"])] = torch.tensor(item["ids"]) - mask[i, :len(item["ids"])] = 1 - positions[i, :item["nopts"]] = torch.tensor(item["candidate_positions"]) - candidate_mask[i, :item["nopts"]] = True - return {"input_ids": ids, "attention_mask": mask, "candidate_positions": positions, "candidate_mask": candidate_mask, "query_positions": torch.tensor([x["query_position"] for x in items]), "labels": torch.tensor([x["label"] for x in items]), "nopts": torch.tensor([x["nopts"] for x in items]), "ids": [x["id"] for x in items], "families": [x["family"] for x in items]} - - -class CandidateHead(nn.Module): - def __init__(self, hidden_size, head_dim=256): - super().__init__() - self.head_dim = head_dim - self.candidate_norm = nn.LayerNorm(hidden_size) - self.query_norm = nn.LayerNorm(hidden_size) - self.key = nn.Linear(hidden_size, head_dim, bias=False) - self.query = nn.Linear(hidden_size, head_dim, bias=False) - self.candidate_mlp = nn.Linear(hidden_size, head_dim, bias=True) - self.query_mlp = nn.Linear(hidden_size, head_dim, bias=False) - self.scalar = nn.Linear(head_dim, 1, bias=False) - nn.init.normal_(self.scalar.weight, mean=0., std=0.01) - - def forward(self, candidates, query): - # Keep the small shared head in FP32 even when the backbone uses BF16. - # The v1 letter head exhibited BF16 ties sensitive to batch padding. - with torch.autocast(device_type=candidates.device.type, enabled=False): - c = self.candidate_norm(candidates.float()) - q = self.query_norm(query.float()) - bilinear = (self.key(c) * self.query(q)[:, None, :]).sum(-1) / math.sqrt(self.head_dim) - interaction = self.scalar(F.gelu(self.candidate_mlp(c) + self.query_mlp(q)[:, None, :])).squeeze(-1) - return bilinear + interaction - - -class DecisionModel(nn.Module): - def __init__(self, backbone, head, metadata): - super().__init__() - self.backbone, self.head, self.metadata = backbone, head, metadata - - @classmethod - def from_base(cls, path, revision="local", dtype=torch.bfloat16, attention="sdpa", head_dim=256): - tokenizer = AutoTokenizer.from_pretrained(path, local_files_only=True) - full, info = Qwen3_5ForConditionalGeneration.from_pretrained(path, dtype=dtype, local_files_only=True, attn_implementation=attention, output_loading_info=True) - if any(info.get(k) for k in ("missing_keys", "mismatched_keys", "error_msgs")): - raise RuntimeError(f"Incomplete base loading: {info}") - backbone = full.model.language_model - backbone.config.use_cache = False - head = CandidateHead(backbone.config.hidden_size, head_dim) - metadata = {"base_revision": revision, "text_parameter_count": sum(p.numel() for p in backbone.parameters()), "prompt_version": PROMPT_VERSION, "attention": attention, "head_dim": head_dim, "max_options": MAX_OPTIONS, "architecture": "contextual-candidate-endpoint-plus-global-query-shared-bilinear-mlp", "head_initialization": "random-shared-content-scorer", "head_precision": "float32-outside-autocast"} - return cls(backbone, head, metadata), tokenizer - - @classmethod - def from_decision_checkpoint(cls, path, dtype=torch.bfloat16, attention="sdpa", head_dim=256): - """Warm-start the backbone of a trained v1 model; initialize a new head.""" - path = Path(path) - metadata = json.loads((path / "decision_config.json").read_text()) - backbone = Qwen3_5TextModel.from_pretrained(path / "backbone", dtype=dtype, local_files_only=True, attn_implementation=attention) - backbone.config.use_cache = False - head = CandidateHead(backbone.config.hidden_size, head_dim) - metadata.update({"prompt_version": PROMPT_VERSION, "head_dim": head_dim, "max_options": MAX_OPTIONS, "architecture": "contextual-candidate-endpoint-plus-global-query-shared-bilinear-mlp", "head_initialization": "random-shared-content-scorer", "warm_start": "trained-v1-text-backbone", "head_precision": "float32-outside-autocast"}) - return cls(backbone, head, metadata), AutoTokenizer.from_pretrained(path, local_files_only=True) - - @classmethod - def from_checkpoint(cls, path, dtype=torch.bfloat16, attention="sdpa"): - path = Path(path) - metadata = json.loads((path / "decision_config.json").read_text()) - if metadata["prompt_version"] != PROMPT_VERSION: - raise ValueError("Not a pointer-v2 checkpoint; use from_decision_checkpoint for warm start") - backbone = Qwen3_5TextModel.from_pretrained(path / "backbone", dtype=dtype, local_files_only=True, attn_implementation=attention) - head = CandidateHead(backbone.config.hidden_size, metadata["head_dim"]) - head.load_state_dict(load_file(path / "decision_head.safetensors")) - return cls(backbone, head, metadata), AutoTokenizer.from_pretrained(path, local_files_only=True) - - def forward(self, input_ids, attention_mask, candidate_positions, candidate_mask, query_positions, **unused): - hidden = self.backbone(input_ids=input_ids, attention_mask=attention_mask, use_cache=False).last_hidden_state - batches = torch.arange(hidden.shape[0], device=hidden.device) - candidates = hidden[batches[:, None], candidate_positions] - query = hidden[batches, query_positions] - scores = self.head(candidates, query).float() - return scores.masked_fill(~candidate_mask, -float("inf")) - - def save(self, path, tokenizer): - path = Path(path); path.mkdir(parents=True, exist_ok=True) - self.backbone.save_pretrained(path / "backbone", safe_serialization=True, max_shard_size="4GB") - save_file({n: v.detach().cpu().contiguous() for n, v in self.head.state_dict().items()}, str(path / "decision_head.safetensors")) - tokenizer.save_pretrained(path) - (path / "decision_config.json").write_text(json.dumps(self.metadata, indent=2) + "\n") - - -def classification_loss(logits, labels): - return F.cross_entropy(logits, labels) diff --git a/code/predict.py b/code/predict.py deleted file mode 100644 index 93d7512c197d13d0f567b0eb2f753b2db80dbf51..0000000000000000000000000000000000000000 --- a/code/predict.py +++ /dev/null @@ -1,43 +0,0 @@ -"""Portable one-GPU JSONL inference for frozen dynamic-option checkpoints.""" -import argparse -import hashlib -import json -from pathlib import Path -import time -import torch -from decision_model import DecisionModel, encode, collate - -p = argparse.ArgumentParser() -p.add_argument("--model", required=True) -p.add_argument("--base", action="store_true") -p.add_argument("--revision", default="local-checkpoint") -p.add_argument("--input", required=True) -p.add_argument("--output", required=True) -p.add_argument("--batch-size", type=int, default=8) -p.add_argument("--max-length", type=int, default=4096) -p.add_argument("--temperature", type=float, default=1.0) -a = p.parse_args() -assert a.temperature > 0 -torch.cuda.set_device(0) -model, tokenizer = DecisionModel.from_base(a.model, a.revision) if a.base else DecisionModel.from_checkpoint(a.model) -model = model.cuda().eval() -rows = [json.loads(line) for line in Path(a.input).read_text().splitlines() if line.strip()] -encoded = [encode(row, tokenizer, a.max_length) for row in rows] -assert len({row["id"] for row in rows}) == len(rows) -output = Path(a.output); output.parent.mkdir(parents=True, exist_ok=True) -if output.exists(): raise RuntimeError("Refusing to overwrite predictions") -with torch.inference_mode(), output.open("w") as f: - for start in range(0, len(encoded), a.batch_size): - examples = encoded[start:start + a.batch_size] - batch = {key: value.cuda() if torch.is_tensor(value) else value for key, value in collate(examples, tokenizer.pad_token_id if tokenizer.pad_token_id is not None else tokenizer.eos_token_id).items()} - torch.cuda.synchronize(); tick = time.perf_counter() - with torch.autocast("cuda", dtype=torch.bfloat16): logits = model(**batch) - torch.cuda.synchronize(); elapsed = time.perf_counter() - tick - probabilities = (logits / a.temperature).softmax(-1).cpu().tolist(); scores = logits.cpu().tolist() - for row, example, prob, score in zip(rows[start:start + a.batch_size], examples, probabilities, scores): - k = example["nopts"]; pred = max(range(k), key=lambda i: prob[i]) - rec = {"id": row["id"], "family": row.get("family"), "label": row.get("label"), "prediction": pred, "prediction_key": row["options"][pred]["key"], "probabilities": prob[:k], "logits": score[:k], "temperature": a.temperature, "input_tokens": len(example["ids"]), "prompt_sha256": example["prompt_sha256"], "batch_elapsed_seconds": elapsed, "batch_size": len(examples)} - if "score_values" in row: rec["expected_score"] = sum(value * probability for value, probability in zip(row["score_values"], prob[:k])) - f.write(json.dumps(rec, ensure_ascii=False) + "\n") - f.flush(); print(json.dumps({"completed": min(start + a.batch_size, len(rows)), "total": len(rows)}), flush=True) -Path(str(output) + ".metadata.json").write_text(json.dumps({"model": a.model, "revision": a.revision, "temperature": a.temperature, "input_sha256": hashlib.sha256(Path(a.input).read_bytes()).hexdigest(), "predictions_sha256": hashlib.sha256(output.read_bytes()).hexdigest(), "rows": len(rows), "model_metadata": model.metadata}, indent=2)) diff --git a/code/profile_guard.py b/code/profile_guard.py deleted file mode 100644 index 7425cd3497c5584121f91dda038e07c95a3c3c2d..0000000000000000000000000000000000000000 --- a/code/profile_guard.py +++ /dev/null @@ -1,82 +0,0 @@ -"""Fail-closed official FLA strict-config setup for one isolated process. - -No model/prompt/head changes. The guard prevents FLA STRICT's ordinary missing -key fallback and records actual configured calls. This is a diagnostic module, -not an installed change to the published wrapper or dependency environment. -""" -import hashlib,importlib,json,os,sys -from pathlib import Path - -def sha(p):return hashlib.sha256(Path(p).read_bytes()).hexdigest() -def serialized(key):return json.dumps(key,separators=(',',':'),sort_keys=True) -def config_fields(config): - if isinstance(config,dict):return {k:config.get(k) for k in ['kwargs','num_warps','num_stages','num_ctas','maxnreg','ir_override']} - return {k:getattr(config,k,None) for k in ['kwargs','num_warps','num_stages','num_ctas','maxnreg','ir_override']} -def validate_profile(path,expected_sha): - path=Path(path).resolve() - if sha(path)!=expected_sha:raise ValueError('Profile hash changed') - profile=json.loads(path.read_text()) - if profile['format']!='decision-fla-l2norm-profile-v1' or profile['cache_mode']!='strict':raise ValueError('Profile format/mode unsupported') - if len(profile['files'])!=1 or profile['files'][0]['file']!='l2norm_fwd_kernel.json':raise ValueError('Unexpected profile file set') - f=path.parent/'l2norm_fwd_kernel.json' - if sha(f)!=profile['files'][0]['sha256']:raise ValueError('Explicit kernel config changed') - data=json.loads(f.read_text());entries={} - if data.get('default_config') is not None:raise ValueError('Implicit fallback defaults forbidden') - for h,item in data['autotune_entries'].items(): - key=item['autotune_key'];encoded=serialized(key) - if hashlib.md5(encoded.encode()).hexdigest()!=h or encoded in entries:raise ValueError('Invalid/duplicate key') - if len(key)!=5 or key[0]!=128 or type(key[1]) is not int or not 1<=key[1]<=32 or key[2:]!=['torch.bfloat16','torch.bfloat16','torch.float32']:raise ValueError('Unsupported numerical key') - c=item['config'] - if c['kwargs'].keys()!={'BT'} or c['kwargs']['BT'] not in [8,16,32] or c['num_warps'] not in [1,2,4,8,16] or c['num_stages']!=3 or c['num_ctas']!=1 or any(c.get(x) is not None for x in ['maxnreg','pre_hook','ir_override']):raise ValueError('Unexpected launch configuration') - entries[encoded]=c - if {json.loads(k)[1] for k in entries}!=set(range(1,33)):raise ValueError('Incomplete legal NB coverage') - return path,profile,entries - -def attach_guard(kernel,cache_module,entries,telemetry): - original=kernel.run - def guarded(*args,**kwargs): - if cache_module.FLA_CACHE_MODE is not cache_module.FlaCacheMode.STRICT:raise RuntimeError('FLA strict mode changed') - key=cache_module.AutotuneKey.build(kernel.arg_names,kernel.keys,args,kwargs);encoded=serialized(list(key.autotune_key)) - if encoded not in entries:raise RuntimeError('Uncontracted FLA l2norm key: '+encoded) - expected=entries[encoded];loaded=cache_module.load_cached_config(kernel.kernel_name,key) - if config_fields(loaded)!=config_fields(expected):raise RuntimeError('FLA exact config lookup mismatch') - if key.autotune_key in kernel.cache and config_fields(kernel.cache[key.autotune_key])!=config_fields(expected):raise RuntimeError('A conflicting in-process kernel cache exists') - # Explicit official configuration load guarantees that the following original - # run finds this exact cache entry and cannot perform timing-based autotune. - kernel.maybe_load_cached_config(key) - if key.autotune_key not in kernel.cache or config_fields(kernel.cache[key.autotune_key])!=config_fields(expected):raise RuntimeError('Official strict config did not load') - result=original(*args,**kwargs) - if config_fields(kernel.cache[key.autotune_key])!=config_fields(expected):raise RuntimeError('Kernel config changed during call') - telemetry['calls']+=1;telemetry['keys'][encoded]=telemetry['keys'].get(encoded,0)+1 - return result - kernel.run=guarded - return original - -def install(profile_path,expected_sha): - path,profile,entries=validate_profile(profile_path,expected_sha) - if any(n=='fla' or n.startswith('fla.') for n in sys.modules):raise RuntimeError('Install profile before importing FLA; use a fresh isolated process') - for name,wanted in {'FLA_CACHE_MODE':'strict','FLA_CONFIG_DIR':str(path.parent)}.items(): - actual=os.environ.get(name) - if actual is not None and actual!=wanted:raise RuntimeError('Conflicting '+name) - os.environ[name]=wanted - import torch,triton,fla - actual={'torch':str(torch.__version__),'hip':torch.version.hip,'triton':triton.__version__,'fla':fla.__version__} - for name,value in actual.items(): - if value!=profile['runtime'][name]:raise RuntimeError('Runtime mismatch: '+name) - if not torch.cuda.is_available():raise RuntimeError('Profile is only qualified for the specified ROCm GPU') - arch=torch.cuda.get_device_properties(0).gcnArchName.split(':')[0] - if arch!=profile['runtime']['gpu_arch']:raise RuntimeError('Unsupported GPU architecture '+arch) - module=importlib.import_module('fla.modules.l2norm');cache_module=importlib.import_module('fla.ops.utils.cache');root=Path(fla.__file__).parent - for name,value in profile['fla_source_sha256'].items(): - if sha(root/name)!=value:raise RuntimeError('Pinned FLA source changed: '+name) - kernel=module.l2norm_fwd_kernel - if kernel.kernel_name!='l2norm_fwd_kernel' or kernel.keys!=['D','NB'] or kernel.cache:raise RuntimeError('Kernel identity or fresh-cache precondition failed') - telemetry={'profile_sha256':expected_sha,'status':'installed','calls':0,'keys':{},'strict_guard':True,'unknown_keys':'raise','autotune_fallback_permitted':False,'runtime':actual,'gpu_arch':arch,'process_scope':'one explicitly profiled Sol model; other model loading in this process is not supported'} - attach_guard(kernel,cache_module,entries,telemetry) - # The validated inference path only uses the vectorized D128 forward kernel. - # Other dimensions/backward must not silently enter a different autotuner. - for name in ['l2norm_fwd_kernel1','l2norm_bwd_kernel','l2norm_bwd_kernel1']: - other=getattr(module,name) - def reject(*args,_name=name,**kwargs):raise RuntimeError('Uncontracted normalization kernel: '+_name) - other.run=reject - return telemetry diff --git a/code/runtime_profile.py b/code/runtime_profile.py deleted file mode 100644 index 57aaebdc18a329a833979939497fd5377a699d55..0000000000000000000000000000000000000000 --- a/code/runtime_profile.py +++ /dev/null @@ -1,36 +0,0 @@ -"""Bundle-local automatic launch-profile binding, before importing FLA. - -One profile is active per Python process. Repeated loading of the same verified -profile is allowed; mixing with an unprofiled/different-profile model is not. -""" -import hashlib,importlib.util,json,os,sys,types -from pathlib import Path -STATE='_decision_process_normalization_profile_v1' -def sha(p):return hashlib.sha256(Path(p).read_bytes()).hexdigest() -def safe(root,relative): - p=Path(relative) - if p.is_absolute() or '..' in p.parts:raise ValueError('Unsafe runtime-profile path') - return root/p - -def ensure_profile(bundle): - bundle=Path(bundle).resolve();runtime=json.loads((bundle/'runtime.json').read_text());spec=runtime.get('normalization_profile');active=sys.modules.get(STATE) - if spec is None: - if active is not None:raise RuntimeError('Load unprofiled and profiled Decision models in separate processes') - return None - if spec.get('kind')!='decision-fla-l2norm-profile-v1' or spec.get('validated_arch')!='gfx942':raise ValueError('Unknown normalization profile contract') - profile=safe(bundle,spec['profile_file']);guard=safe(bundle,spec['guard_file']) - if sha(profile)!=spec['profile_sha256'] or sha(guard)!=spec['guard_sha256']:raise ValueError('Bound runtime profile bytes changed') - if json.loads((bundle/'decision_config.json').read_text()).get('base_model')!='Qwen/Qwen3.5-2B':raise ValueError('This bundle profile is bound to the validated Sol family only') - if active is not None: - if active.profile_sha256!=spec['profile_sha256'] or active.guard_sha256!=spec['guard_sha256']:raise RuntimeError('Different Decision normalization profile is already active; use a separate process') - active.guard.validate_profile(profile,spec['profile_sha256']) - if os.environ.get('FLA_CACHE_MODE')!='strict' or os.environ.get('FLA_CONFIG_DIR')!=active.profile_dir:raise RuntimeError('Active FLA profile environment changed') - return {'profile_sha256':active.profile_sha256,'guard_sha256':active.guard_sha256,'validated_arch':'gfx942','automatic_bundle_binding':True,'scope':'single profile per process'} - module_spec=importlib.util.spec_from_file_location('decision_profile_guard_'+spec['guard_sha256'][:16],guard);module=importlib.util.module_from_spec(module_spec);module_spec.loader.exec_module(module) - telemetry=module.install(profile,spec['profile_sha256']) - state=types.ModuleType(STATE);state.profile_sha256=spec['profile_sha256'];state.guard_sha256=spec['guard_sha256'];state.profile_dir=str(profile.parent);state.guard=module;state.telemetry=telemetry;sys.modules[STATE]=state - return {'profile_sha256':state.profile_sha256,'guard_sha256':state.guard_sha256,'validated_arch':'gfx942','automatic_bundle_binding':True,'scope':'single profile per process'} - -def active_telemetry(): - state=sys.modules.get(STATE) - return None if state is None else dict(state.telemetry) diff --git a/config.json b/config.json new file mode 100644 index 0000000000000000000000000000000000000000..1cfc6d5f54be6b54eb652cdd1ab39d111d1b9536 --- /dev/null +++ b/config.json @@ -0,0 +1,22 @@ +{ + "decision_format": "vllm-sr-decision", + "format_version": 1, + "model_name": "Decision-1.0-Sol-2B", + "runtime_family": "qwen3.5-decision", + "model_config": "decision_config.json", + "backbone": { + "config": "backbone/config.json", + "weights": ["backbone/model.safetensors"] + }, + "tokenizer": { + "json": "tokenizer.json", + "config": "tokenizer_config.json", + "chat_template": "chat_template.jinja" + }, + "decision_weights": { + "decision_head": "decision_head.safetensors" + }, + "calibration": { + "temperature_file": "temperature.json" + } +} diff --git a/decision_config.json b/decision_config.json index 2a6351a91837797dec06481027a51983d5c03dd9..28388a9d207dd28ff9808fb1602dc4692b643a0e 100644 --- a/decision_config.json +++ b/decision_config.json @@ -14,6 +14,5 @@ "training_master_parameter_dtype": "float32", "backbone_autocast_dtype": "bfloat16", "base_model": "Qwen/Qwen3.5-2B", - "calibration_file": "temperature.json", - "runtime_file": "runtime.json" + "calibration_file": "temperature.json" } diff --git a/DIAGNOSTICS.md b/evaluation/DIAGNOSTICS.md similarity index 100% rename from DIAGNOSTICS.md rename to evaluation/DIAGNOSTICS.md diff --git a/EVALUATION.md b/evaluation/EVALUATION.md similarity index 95% rename from EVALUATION.md rename to evaluation/EVALUATION.md index 28c016d0fc24cdc63b57b7bdaabcc13fd3f29b84..296dc717471061aab02f3b5bcf771433ce205bba 100644 --- a/EVALUATION.md +++ b/evaluation/EVALUATION.md @@ -32,10 +32,10 @@ These are **outcome-informed product-priority weights, chosen after observing be ## Full results -[All 54 task rows](TASKS.md) preserve every original decision, composition, reading and inference task plus all 27 transfer tasks. [Diagnostics](DIAGNOSTICS.md) separately report probability quality, option-order sensitivity, missing evidence, native coverage and uncertainty. [Exact statistics](metrics/benchmark.json) include counts and confidence intervals. +[All 54 task rows](TASKS.md) preserve every original decision, composition, reading and inference task plus all 27 transfer tasks. [Diagnostics](DIAGNOSTICS.md) separately report probability quality, option-order sensitivity, missing evidence, native coverage and uncertainty. [Exact statistics](benchmark.json) include counts and confidence intervals. ## Model and API scope Decision models return typed Choice, Noul and Score answers in the SystemOne format. Encoder and decoder references are evaluated through their published native interfaces. Untuned Qwen models use the frozen letter-logit readout, with no generated-text parsing. Kev uses the matched BF16 backbone with FP32 head and shipped temperature; date-fact injection is disabled. This differs from the authors’ default FP32 reproduction. Laya English and multilingual are measured separately. Kai-0.6B and Eos-0.8B use their independently verified published native interfaces on the same 54 tasks and requested denominators. Jev is a recorded hosted-service snapshot. -The tests assess decisions from the supplied state and fixed choices. They do not establish live fact retrieval, universally superior reasoning or cross-hardware speed rankings. [Immutable identities and evaluation provenance](metrics/evaluation-provenance.json). +The tests assess decisions from the supplied state and fixed choices. They do not establish live fact retrieval, universally superior reasoning or cross-hardware speed rankings. [Immutable identities and evaluation provenance](evaluation-provenance.json). diff --git a/QUESTION-SCALING.md b/evaluation/QUESTION-SCALING.md similarity index 74% rename from QUESTION-SCALING.md rename to evaluation/QUESTION-SCALING.md index 0bd365b2b22ec57d700eaeacfb9bc516d0f4cfed..c9b7b388c1b7f95b00b86636e4d0d94d457dee92 100644 --- a/QUESTION-SCALING.md +++ b/evaluation/QUESTION-SCALING.md @@ -2,7 +2,7 @@ Sol-2B · distinct Choice questions with **499 input tokens per question**. The state and individual question length stay fixed as the question count grows. -![Question scaling](assets/decision-question-scaling.png) +![Question scaling](../assets/decision-question-scaling.png) | Questions | p50 ms ↓ | p95 ms ↓ | Peak allocated GiB | |---:|---:|---:|---:| @@ -15,6 +15,6 @@ Sol-2B · distinct Choice questions with **499 input tokens per question**. The Six independently loaded process blocks supply 30 measured requests per point after warmup on an otherwise idle AMD gfx942 GPU. End-to-end Python request latency includes rendering, tokenization, inference, output assembly and final synchronization. Model loading, JSON serialization and network are excluded. Questions run in their original order in batches of eight. Requests are sequential, with no neural prefix cache. -These existing measurements use the exact API implementation and inference bundle still published. The runtime subsequently passed an offline Hub-download proof and the full 3,160-answer regression. [Measurements and immutable runtime identity](metrics/question-scaling.json). +These measurements used the original API implementation and inference bundle. That runtime passed an offline Hub-download proof and the full 3,160-answer regression; this historical result does not validate a different serving runtime. [Measurements and immutable runtime identity](question-scaling.json). This fixed short-input Choice workload does not establish concurrent HTTP throughput, long-context scaling, other question-type performance or a cross-hardware speed ranking. diff --git a/SENSITIVITY.md b/evaluation/SENSITIVITY.md similarity index 100% rename from SENSITIVITY.md rename to evaluation/SENSITIVITY.md diff --git a/TASKS.md b/evaluation/TASKS.md similarity index 100% rename from TASKS.md rename to evaluation/TASKS.md diff --git a/metrics/benchmark.json b/evaluation/benchmark.json similarity index 100% rename from metrics/benchmark.json rename to evaluation/benchmark.json diff --git a/metrics/evaluation-provenance.json b/evaluation/evaluation-provenance.json similarity index 100% rename from metrics/evaluation-provenance.json rename to evaluation/evaluation-provenance.json diff --git a/metrics/question-scaling.json b/evaluation/question-scaling.json similarity index 100% rename from metrics/question-scaling.json rename to evaluation/question-scaling.json diff --git a/metrics/comparator-coverage.json b/metrics/comparator-coverage.json deleted file mode 100644 index 6fb64e2acf0a9f2556d71a4bebacdd1d6ea48ba9..0000000000000000000000000000000000000000 --- a/metrics/comparator-coverage.json +++ /dev/null @@ -1,458 +0,0 @@ -{ - "Jev \u00b7 1.13.0": { - "old_core": { - "requested": 880, - "recorded_outputs": 880, - "ok": 880, - "truncated_true": 0, - "retention_unknown": 880, - "raw_sha256": "15fc57a65b0571eafcb4a1ad695fb224aeae44b65e1a61ecd4d348f2e8fd3a60", - "normalized_sha256": "069e48e1fdf1b5f3d95cc0097e4179ac3faab97bb357b0ce897cfb463b87cbaf", - "requests_sha256": "7c46df786001f7bd49c4e43d0998c494c169b40164e500e5c523016350db312f", - "raw_to_normalized_reverified": true, - "service_versions": [ - "jev-1.13.0" - ], - "first_request_utc": "2026-09-21T09:49:11.121634+00:00", - "last_request_utc": "2026-09-21T09:55:51.282211+00:00", - "exact_weight_snapshot_available": false - }, - "v3_core": { - "requested": 880, - "recorded_outputs": 880, - "ok": 880, - "truncated_true": 0, - "retention_unknown": 880, - "raw_sha256": "48c1b0d8729c88fdc4a4f89fbf5eda045925243c997c4aae1e6172ba63ab13b9", - "normalized_sha256": "54707354bfe5d96b5de56e6dc1b38cba0a610eea516b28fbff013b94f422b397", - "requests_sha256": "bd5dfe8ab0aff17435b5826126ebcee502c4cc44673dbe0be308bff489833095", - "raw_to_normalized_reverified": true, - "service_versions": [ - "jev-1.13.0" - ], - "first_request_utc": "2026-09-21T14:53:06.592910+00:00", - "last_request_utc": "2026-09-21T15:06:45.698712+00:00", - "exact_weight_snapshot_available": false - }, - "v4": { - "requested": 480, - "recorded_outputs": 480, - "ok": 480, - "truncated_true": 0, - "retention_unknown": 480, - "raw_sha256": "246ddbff360a15adc3e4679c86aed44df8541afee53b0205da1892eb0fe8a811", - "normalized_sha256": "d6583f903e974d70747fae2944014828954dc4d7344ceddcbb2cd1b06fbfc799", - "requests_sha256": "8e7397c7086fc867d2aafaac40f49db156be0f10e2b2728ef041fa04afd59dbe", - "raw_to_normalized_reverified": true, - "service_versions": [ - "jev-1.13.0" - ], - "first_request_utc": "2026-09-21T16:22:16.227785+00:00", - "last_request_utc": "2026-09-21T16:28:05.090381+00:00", - "exact_weight_snapshot_available": false - }, - "v5": { - "requested": 480, - "recorded_outputs": 480, - "ok": 480, - "truncated_true": 0, - "retention_unknown": 480, - "raw_sha256": "a0a6d0699aed767bad897b71191d18f9ee59a49bc0a71c5fbb5003f6d966133b", - "normalized_sha256": "f883164c4c83db8e42ebe03f7e165afbd2f81f2b541bcf8008a2cc06a60e1ee7", - "requests_sha256": "c83a7b9611c740c5232ebeb15b00377798c76806c509e62e2f63b08369c898bc", - "raw_to_normalized_reverified": true, - "service_versions": [ - "jev-1.13.0" - ], - "first_request_utc": "2026-09-21T19:47:20.496709+00:00", - "last_request_utc": "2026-09-21T19:50:20.927928+00:00", - "exact_weight_snapshot_available": false - } - }, - "Laya \u00b7 EN/ML": { - "old_core": { - "requested": 880, - "recorded_outputs": 880, - "ok": 880, - "truncated_true": 112, - "retention_unknown": 0, - "raw_sha256": "cc1df0126bd7d790b55b40c39028a3dcea2bb20d8c8115b7a7f4042ec1314d6e", - "normalized_sha256": "2503adb3d7a3476a9b84f6647f2a69e7986e96eea878c191568a33597800a3dc", - "requests_sha256": "7c46df786001f7bd49c4e43d0998c494c169b40164e500e5c523016350db312f", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "repo": "convaiinnovations/laya", - "revision": "1c5edc17a7acd8701df6fc341c0d179f1c62c982", - "router_source_revision": "42626c348753fbb17572a813127df2278a1ec527", - "adapter_sha256": "a5484e30d9fad04ca578265e5f18df2a10731f5055bef070311102c3b35643fc", - "mode": "routed", - "weight_fingerprints": { - "english": "5bf5d0f9462d860e4c09000fce3131a070578d3574777ae7981077f11cd68185", - "multilingual": "a472bfcc4aa52a25dda0b2864b3e7c7c1267ddd1752b04a1e70bf6ba7e4b514c" - }, - "routing": "Unmodified upstream Router.route(state,questions), no explicit lang, no automatic workflow detection, default english" - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - }, - "v3_core": { - "requested": 880, - "recorded_outputs": 880, - "ok": 880, - "truncated_true": 40, - "retention_unknown": 0, - "raw_sha256": "c5e7703af1d200cd662d1eef8e87f6d30a237c451233893b295e0eac0f92e0fa", - "normalized_sha256": "fd5053fb037d43de5383140b33cb3a363ad5cab56387354f1bac0414d52f7da2", - "requests_sha256": "bd5dfe8ab0aff17435b5826126ebcee502c4cc44673dbe0be308bff489833095", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "repo": "convaiinnovations/laya", - "revision": "1c5edc17a7acd8701df6fc341c0d179f1c62c982", - "router_source_revision": "42626c348753fbb17572a813127df2278a1ec527", - "adapter_sha256": "a5484e30d9fad04ca578265e5f18df2a10731f5055bef070311102c3b35643fc", - "mode": "routed", - "weight_fingerprints": { - "english": "5bf5d0f9462d860e4c09000fce3131a070578d3574777ae7981077f11cd68185", - "multilingual": "a472bfcc4aa52a25dda0b2864b3e7c7c1267ddd1752b04a1e70bf6ba7e4b514c" - }, - "routing": "Unmodified upstream Router.route(state,questions), no explicit lang, no automatic workflow detection, default english" - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - }, - "v4": { - "requested": 480, - "recorded_outputs": 480, - "ok": 480, - "truncated_true": 0, - "retention_unknown": 0, - "raw_sha256": "550cc73b98da4ff7efc4b4cf07703f606831d59ad90ea5ea223ac5f587d51fe5", - "normalized_sha256": "a10922609e2223ca0a818667bc1425d476aec58a4d881002e0e322596c1d0b8d", - "requests_sha256": "8e7397c7086fc867d2aafaac40f49db156be0f10e2b2728ef041fa04afd59dbe", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "repo": "convaiinnovations/laya", - "revision": "1c5edc17a7acd8701df6fc341c0d179f1c62c982", - "router_source_revision": "42626c348753fbb17572a813127df2278a1ec527", - "adapter_sha256": "a5484e30d9fad04ca578265e5f18df2a10731f5055bef070311102c3b35643fc", - "mode": "routed", - "weight_fingerprints": { - "english": "5bf5d0f9462d860e4c09000fce3131a070578d3574777ae7981077f11cd68185", - "multilingual": "a472bfcc4aa52a25dda0b2864b3e7c7c1267ddd1752b04a1e70bf6ba7e4b514c" - }, - "routing": "Unmodified upstream Router.route(state,questions), no explicit lang, no automatic workflow detection, default english" - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - }, - "v5": { - "requested": 480, - "recorded_outputs": 480, - "ok": 480, - "truncated_true": 2, - "retention_unknown": 0, - "raw_sha256": "31064d69fd8a2b1bb86745a0525af40ef3f8feb91a0aa462f3035770fc84e6b3", - "normalized_sha256": "d1334fc4b9106621aa754a8567079201f3ae8bb270ceab981f7351a4a4479c62", - "requests_sha256": "c83a7b9611c740c5232ebeb15b00377798c76806c509e62e2f63b08369c898bc", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "repo": "convaiinnovations/laya", - "revision": "1c5edc17a7acd8701df6fc341c0d179f1c62c982", - "router_source_revision": "42626c348753fbb17572a813127df2278a1ec527", - "adapter_sha256": "a5484e30d9fad04ca578265e5f18df2a10731f5055bef070311102c3b35643fc", - "mode": "routed", - "weight_fingerprints": { - "english": "5bf5d0f9462d860e4c09000fce3131a070578d3574777ae7981077f11cd68185", - "multilingual": "a472bfcc4aa52a25dda0b2864b3e7c7c1267ddd1752b04a1e70bf6ba7e4b514c" - }, - "routing": "Unmodified upstream Router.route(state,questions), no explicit lang, no automatic workflow detection, default english" - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - } - }, - "Decider \u00b7 2B": { - "old_core": { - "requested": 880, - "recorded_outputs": 880, - "ok": 880, - "truncated_true": 0, - "retention_unknown": 0, - "raw_sha256": "5b6e107f9b0eea7686d0e2e3e983f5950254a7dffd875ecaca0a6a9ec5e4e590", - "normalized_sha256": "e1b14f0f9ec521bffbde5945684bc2410a413b6b6418d65ac4fc702a7b5e7937", - "requests_sha256": "7c46df786001f7bd49c4e43d0998c494c169b40164e500e5c523016350db312f", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "model_id": "Mapika/decider-2b", - "original_native_adapter_sha256": "983a2a0d582a1c5e7e58fe3fe4e7c1cd254dd0fe52f65449c68549108d0af387", - "wrapper_sha256": "95b9fcc132d1ae2f0818ea8bb4720f4b5342450e741fefb4315827225652e7f6", - "model_fingerprint": "35f7335c92cf1766c11f52fb4bcdb9c3894d3edad7408a00ae7b59c104b2d318", - "temperature": 1.3, - "layout": "state_first", - "independent": true, - "revision": "b37f7e1ba3fbc9238004cf531fabbee2619973fd" - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - }, - "v3_core": { - "requested": 880, - "recorded_outputs": 880, - "ok": 880, - "truncated_true": 0, - "retention_unknown": 0, - "raw_sha256": "eceb70e8c88117c288599b4a46eaea961f01a75f15fd48fe5b5075c070d680ef", - "normalized_sha256": "12a1f554decf1aff608743e7b4a44681289ea084ae25f383e519b1f34917d46f", - "requests_sha256": "bd5dfe8ab0aff17435b5826126ebcee502c4cc44673dbe0be308bff489833095", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "model_id": "Mapika/decider-2b", - "original_native_adapter_sha256": "983a2a0d582a1c5e7e58fe3fe4e7c1cd254dd0fe52f65449c68549108d0af387", - "wrapper_sha256": "95b9fcc132d1ae2f0818ea8bb4720f4b5342450e741fefb4315827225652e7f6", - "model_fingerprint": "35f7335c92cf1766c11f52fb4bcdb9c3894d3edad7408a00ae7b59c104b2d318", - "temperature": 1.3, - "layout": "state_first", - "independent": true, - "revision": "b37f7e1ba3fbc9238004cf531fabbee2619973fd" - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - }, - "v4": { - "requested": 480, - "recorded_outputs": 480, - "ok": 480, - "truncated_true": 0, - "retention_unknown": 0, - "raw_sha256": "6f4f5be5b606a0dd13cd2303d0562f1e7538bba92288f121ed9886ac5dc873ac", - "normalized_sha256": "626ef366ae471ec10dfb89ef2ff2f7b29a6c879976d3ff14fcbb2bafd7d9b045", - "requests_sha256": "8e7397c7086fc867d2aafaac40f49db156be0f10e2b2728ef041fa04afd59dbe", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "model_id": "Mapika/decider-2b", - "original_native_adapter_sha256": "983a2a0d582a1c5e7e58fe3fe4e7c1cd254dd0fe52f65449c68549108d0af387", - "wrapper_sha256": "95b9fcc132d1ae2f0818ea8bb4720f4b5342450e741fefb4315827225652e7f6", - "model_fingerprint": "35f7335c92cf1766c11f52fb4bcdb9c3894d3edad7408a00ae7b59c104b2d318", - "temperature": 1.3, - "layout": "state_first", - "independent": true, - "revision": "b37f7e1ba3fbc9238004cf531fabbee2619973fd" - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - }, - "v5": { - "requested": 480, - "recorded_outputs": 480, - "ok": 480, - "truncated_true": 0, - "retention_unknown": 0, - "raw_sha256": "90c05f48c5da7caeea4c4acfd58b1f1ff82c9e7af73e06036e30a4680fdc6c9a", - "normalized_sha256": "ae16b17ac8e4208940c0c3e9e04f25e94686b5d16bf46f64f53669baf41aeb71", - "requests_sha256": "c83a7b9611c740c5232ebeb15b00377798c76806c509e62e2f63b08369c898bc", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "model_id": "Mapika/decider-2b", - "original_native_adapter_sha256": "983a2a0d582a1c5e7e58fe3fe4e7c1cd254dd0fe52f65449c68549108d0af387", - "wrapper_sha256": "95b9fcc132d1ae2f0818ea8bb4720f4b5342450e741fefb4315827225652e7f6", - "model_fingerprint": "35f7335c92cf1766c11f52fb4bcdb9c3894d3edad7408a00ae7b59c104b2d318", - "temperature": 1.3, - "layout": "state_first", - "independent": true, - "revision": "b37f7e1ba3fbc9238004cf531fabbee2619973fd" - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - } - }, - "Qwen3.5 \u00b7 2B, untuned": { - "old_core": { - "requested": 880, - "recorded_outputs": 880, - "ok": 880, - "truncated_true": 0, - "retention_unknown": 0, - "raw_sha256": "0f2e105e7204db3e94fce26f57af68d0fbc64fc47097c97207f5feb9c22ecb37", - "normalized_sha256": "c11a94d8f833d83427ea984a4f069aefaa3fa10a2b1085fb3a03542727f819cf", - "requests_sha256": "7c46df786001f7bd49c4e43d0998c494c169b40164e500e5c523016350db312f", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "base_repo": "Qwen/Qwen3.5-2B", - "base_revision": "15852e8c16360a2fea060d615a32b45270f8a8fc", - "adapter_sha256": "e08336ecd148e9240792dc3d915536fc52998364f248f9f2016b8b83e289b989", - "prompt_version": "decision-letter-v1", - "temperature": 1.0, - "readout": "Frozen pretrained next-token LM logits restricted to A..Z; conditional softmax over listed candidates", - "new_trainable_head": false, - "task_finetuning": false - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - }, - "v3_core": { - "requested": 880, - "recorded_outputs": 880, - "ok": 880, - "truncated_true": 0, - "retention_unknown": 0, - "raw_sha256": "75b93056650b33336800cb4ef4a2a1a230bb4d2aa95fdaef7d3a369d9c264a65", - "normalized_sha256": "cc51471716342ecd466b50d5dd9b4ba805f5a6873cd96075b56d7c2e8336175e", - "requests_sha256": "bd5dfe8ab0aff17435b5826126ebcee502c4cc44673dbe0be308bff489833095", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "base_repo": "Qwen/Qwen3.5-2B", - "base_revision": "15852e8c16360a2fea060d615a32b45270f8a8fc", - "adapter_sha256": "e08336ecd148e9240792dc3d915536fc52998364f248f9f2016b8b83e289b989", - "prompt_version": "decision-letter-v1", - "temperature": 1.0, - "readout": "Frozen pretrained next-token LM logits restricted to A..Z; conditional softmax over listed candidates", - "new_trainable_head": false, - "task_finetuning": false - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - }, - "v4": { - "requested": 480, - "recorded_outputs": 480, - "ok": 480, - "truncated_true": 0, - "retention_unknown": 0, - "raw_sha256": "780936e46e6296cd07a6bbefb774144a17464421c1be4251c0c3d0411057bc63", - "normalized_sha256": "93d84cdbe607691e2cef35ece3aa2dd5900f6410b0d72796b86d74c9156f3cdd", - "requests_sha256": "8e7397c7086fc867d2aafaac40f49db156be0f10e2b2728ef041fa04afd59dbe", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "base_repo": "Qwen/Qwen3.5-2B", - "base_revision": "15852e8c16360a2fea060d615a32b45270f8a8fc", - "adapter_sha256": "e08336ecd148e9240792dc3d915536fc52998364f248f9f2016b8b83e289b989", - "prompt_version": "decision-letter-v1", - "temperature": 1.0, - "readout": "Frozen pretrained next-token LM logits restricted to A..Z; conditional softmax over listed candidates", - "new_trainable_head": false, - "task_finetuning": false - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - }, - "v5": { - "requested": 480, - "recorded_outputs": 480, - "ok": 480, - "truncated_true": 0, - "retention_unknown": 0, - "raw_sha256": "deeeee3fa6a37cb9ebb0b891421159e7cdcb2137b38634e79e996b9b67d6bf7e", - "normalized_sha256": "75740c4279eb8dcc9c90b55f155916b3748c887e8c9d9e1d3a47edf7f4dbc34b", - "requests_sha256": "c83a7b9611c740c5232ebeb15b00377798c76806c509e62e2f63b08369c898bc", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "base_repo": "Qwen/Qwen3.5-2B", - "base_revision": "15852e8c16360a2fea060d615a32b45270f8a8fc", - "adapter_sha256": "e08336ecd148e9240792dc3d915536fc52998364f248f9f2016b8b83e289b989", - "prompt_version": "decision-letter-v1", - "temperature": 1.0, - "readout": "Frozen pretrained next-token LM logits restricted to A..Z; conditional softmax over listed candidates", - "new_trainable_head": false, - "task_finetuning": false - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - } - }, - "Qwen3.5 \u00b7 4B, untuned": { - "old_core": { - "requested": 880, - "recorded_outputs": 880, - "ok": 880, - "truncated_true": 0, - "retention_unknown": 0, - "raw_sha256": "0f24febd8ea56174d29f3cf9642f1595905b6d55a29c2dce9258fec616c1cc68", - "normalized_sha256": "d20ad845066ff81caaf60ebee44df941872bdc642da78b5f3b3da4a33dd9a08d", - "requests_sha256": "7c46df786001f7bd49c4e43d0998c494c169b40164e500e5c523016350db312f", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "base_repo": "Qwen/Qwen3.5-4B", - "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", - "adapter_sha256": "e08336ecd148e9240792dc3d915536fc52998364f248f9f2016b8b83e289b989", - "prompt_version": "decision-letter-v1", - "temperature": 1.0, - "readout": "Frozen pretrained next-token LM logits restricted to A..Z; conditional softmax over listed candidates", - "new_trainable_head": false, - "task_finetuning": false - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - }, - "v3_core": { - "requested": 880, - "recorded_outputs": 880, - "ok": 880, - "truncated_true": 0, - "retention_unknown": 0, - "raw_sha256": "f27e7008cb7e7a73248ba119dd6d4279600c61f0169ed4514fb99ccec605dec3", - "normalized_sha256": "7541893d2b2ec0ec2ffc513f50c568eaef860b65bcef9a1747996c116fff9661", - "requests_sha256": "bd5dfe8ab0aff17435b5826126ebcee502c4cc44673dbe0be308bff489833095", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "base_repo": "Qwen/Qwen3.5-4B", - "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", - "adapter_sha256": "e08336ecd148e9240792dc3d915536fc52998364f248f9f2016b8b83e289b989", - "prompt_version": "decision-letter-v1", - "temperature": 1.0, - "readout": "Frozen pretrained next-token LM logits restricted to A..Z; conditional softmax over listed candidates", - "new_trainable_head": false, - "task_finetuning": false - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - }, - "v4": { - "requested": 480, - "recorded_outputs": 480, - "ok": 480, - "truncated_true": 0, - "retention_unknown": 0, - "raw_sha256": "1e09cd713ce91f9c4a5cc81939554a7351634865dc262d93db879de75537f2e5", - "normalized_sha256": "7a6d252ccaa72bf54b94445190bbad023fb465a3eaf4cf657532e2cdd4f750f5", - "requests_sha256": "8e7397c7086fc867d2aafaac40f49db156be0f10e2b2728ef041fa04afd59dbe", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "base_repo": "Qwen/Qwen3.5-4B", - "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", - "adapter_sha256": "e08336ecd148e9240792dc3d915536fc52998364f248f9f2016b8b83e289b989", - "prompt_version": "decision-letter-v1", - "temperature": 1.0, - "readout": "Frozen pretrained next-token LM logits restricted to A..Z; conditional softmax over listed candidates", - "new_trainable_head": false, - "task_finetuning": false - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - }, - "v5": { - "requested": 480, - "recorded_outputs": 480, - "ok": 480, - "truncated_true": 0, - "retention_unknown": 0, - "raw_sha256": "1f7068b586c6ca0dca6f7928acef2d539ba4746a2e7eeb73fd3a1a1ccb8697a5", - "normalized_sha256": "c01ad5daed65267e9aec40ed9e97ece1b6511e4cab17ebff09339b41524fb7f1", - "requests_sha256": "c83a7b9611c740c5232ebeb15b00377798c76806c509e62e2f63b08369c898bc", - "raw_to_normalized_reverified": true, - "adapter_identity": { - "base_repo": "Qwen/Qwen3.5-4B", - "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", - "adapter_sha256": "e08336ecd148e9240792dc3d915536fc52998364f248f9f2016b8b83e289b989", - "prompt_version": "decision-letter-v1", - "temperature": 1.0, - "readout": "Frozen pretrained next-token LM logits restricted to A..Z; conditional softmax over listed candidates", - "new_trainable_head": false, - "task_finetuning": false - }, - "runtime_fingerprint_sha256": "b5af3d997c95d0a6d2088a64f341726fc0daee5eebc730f851b2b65bf0de8d1b" - } - }, - "kev-9b": { - "repo_id": "jaredpalmer/kev-9b", - "revision": "6281032426a9ca3a08374d137bf9dfd8afd78e8a", - "source_revision": "e0bcf50153f1bda4ca6a8be5e12cbd5f9ebbce1c", - "weight_fingerprint": "2b29b6a7e43e94b467a0d516e5d17008cba0f5c6e36c2793c57140604accc4a9", - "four_panel_rows": 2720, - "native_supplement_success_counts": [ - [ - 217, - 220 - ], - [ - 217, - 220 - ] - ], - "adapter": "upstream native adapter; matched FLA0.5.2 / Transformers5.17" - } -} diff --git a/metrics/expanded-quality.json b/metrics/expanded-quality.json deleted file mode 100644 index 3ee28d954020bfb0594557a5de77c72a94b9cfe3..0000000000000000000000000000000000000000 --- a/metrics/expanded-quality.json +++ /dev/null @@ -1,2642 +0,0 @@ -{ - "models": { - "Nox-retention-selected": { - "old_core": { - "point": 0.8300223214285715, - "exact_fraction": "7437/8960", - "ci95": [ - 0.7974693080357143, - 0.862016369047619 - ] - }, - "v3_core": { - "point": 0.5179166666666667, - "exact_fraction": "1243/2400", - "ci95": [ - 0.4920833333333333, - 0.545 - ] - }, - "v4": { - "point": 0.790625, - "exact_fraction": "253/320", - "ci95": [ - 0.7453125, - 0.8331463068181818 - ] - }, - "v5": { - "point": 0.8625, - "exact_fraction": "69/80", - "ci95": [ - 0.8321482316646345, - 0.8915620903779563 - ] - }, - "v3_joint": { - "point": 0.6739694940476191, - "exact_fraction": "181163/268800", - "ci95": [ - 0.652767857142857, - 0.6950187872023811 - ] - }, - "expanded_joint": { - "point": 0.7502659970238095, - "exact_fraction": "403343/537600", - "ci95": [ - 0.7330469621683191, - 0.7670605596068275 - ] - }, - "families": { - "old_core": { - "ag_news": { - "correct": 109, - "requested": 128, - "accuracy_ci95": [ - 0.7890625, - 0.90625 - ] - }, - "boolean_constraints": { - "correct": 60, - "requested": 64, - "accuracy_ci95": [ - 0.859375, - 1.0 - ] - }, - "dbpedia_14": { - "correct": 108, - "requested": 112, - "accuracy_ci95": [ - 0.9285714285714286, - 0.9910714285714286 - ] - }, - "natural_intents": { - "correct": 64, - "requested": 64, - "accuracy_ci95": [ - 1.0, - 1.0 - ] - }, - "option_carrier": { - "correct": 96, - "requested": 96, - "accuracy_ci95": [ - 1.0, - 1.0 - ] - }, - "ordinal_rubric": { - "correct": 57, - "requested": 64, - "accuracy_ci95": [ - 0.78125, - 0.984375 - ] - }, - "relational_composition": { - "correct": 50, - "requested": 96, - "accuracy_ci95": [ - 0.34375, - 0.6979166666666666 - ] - }, - "scoped_evidence": { - "correct": 75, - "requested": 96, - "accuracy_ci95": [ - 0.6145833333333334, - 0.9270833333333334 - ] - }, - "state_tracking": { - "correct": 34, - "requested": 96, - "accuracy_ci95": [ - 0.19791666666666666, - 0.53125 - ] - }, - "unknown_rejection": { - "correct": 64, - "requested": 64, - "accuracy_ci95": [ - 1.0, - 1.0 - ] - } - }, - "v3_core": { - "canonical_record_identity": { - "correct": 43, - "requested": 80, - "accuracy_ci95": [ - 0.5, - 0.575 - ] - }, - "capacitated_assignment": { - "correct": 37, - "requested": 80, - "accuracy_ci95": [ - 0.4125, - 0.5125 - ] - }, - "constraint_assignment": { - "correct": 33, - "requested": 80, - "accuracy_ci95": [ - 0.3375, - 0.475 - ] - }, - "massive_en": { - "correct": 110, - "requested": 120, - "accuracy_ci95": [ - 0.8666666666666667, - 0.9583333333333334 - ] - }, - "massive_zh": { - "correct": 105, - "requested": 120, - "accuracy_ci95": [ - 0.8166666666666667, - 0.925 - ] - }, - "multiset_reconciliation": { - "correct": 23, - "requested": 80, - "accuracy_ci95": [ - 0.2125, - 0.375 - ] - }, - "ordinal_service_loss": { - "correct": 24, - "requested": 80, - "accuracy_ci95": [ - 0.225, - 0.375 - ] - }, - "paraconsistent_rule_closure": { - "correct": 27, - "requested": 80, - "accuracy_ci95": [ - 0.275, - 0.425 - ] - }, - "temporal_exclusion": { - "correct": 34, - "requested": 80, - "accuracy_ci95": [ - 0.3, - 0.55 - ] - }, - "transaction_recovery": { - "correct": 50, - "requested": 80, - "accuracy_ci95": [ - 0.4875, - 0.7625 - ] - } - }, - "v4": { - "boolq": { - "correct": 138, - "requested": 160, - "accuracy_ci95": [ - 0.80625, - 0.9125 - ] - }, - "belebele_en": { - "correct": 117, - "requested": 160, - "accuracy_ci95": [ - 0.6585365853658537, - 0.802547770700637 - ] - }, - "belebele_zh": { - "correct": 113, - "requested": 160, - "accuracy_ci95": [ - 0.632258064516129, - 0.779874213836478 - ] - } - }, - "v5": { - "cosmos_qa": { - "correct": 85, - "requested": 120, - "accuracy_ci95": [ - 0.625, - 0.7833333333333333 - ] - }, - "squad2_answerability": { - "correct": 106, - "requested": 120, - "accuracy_ci95": [ - 0.8214285714285714, - 0.9407026836158189 - ] - }, - "snli": { - "correct": 107, - "requested": 120, - "accuracy_ci95": [ - 0.8333333333333334, - 0.9416666666666667 - ] - }, - "qasc": { - "correct": 116, - "requested": 120, - "accuracy_ci95": [ - 0.9327731092436975, - 0.991869918699187 - ] - } - } - }, - "proper_scores": { - "old_core": 0.22558726136256252, - "v3_core": 0.6082926735018094, - "v4": 0.3013823815596702, - "v5": 0.19389530242905464, - "expanded_weighted_brier": 0.3322894047132742 - }, - "probability_coverage": { - "old_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v3_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v4": { - "requested": 480, - "valid_probability_rows": 480 - }, - "v5": { - "requested": 480, - "valid_probability_rows": 480 - } - }, - "brier_ci95": { - "old_core": [ - 0.1846602761421124, - 0.2688592667224405 - ], - "v3_core": [ - 0.5803874232294962, - 0.6368345885571338 - ], - "v4": [ - 0.24180651680296794, - 0.3645626677453216 - ], - "v5": [ - 0.16179407555902875, - 0.22847239437277253 - ], - "expanded_weighted_brier": [ - 0.3111201217342611, - 0.35396540259832066 - ] - } - }, - "Sol-composition-selected": { - "old_core": { - "point": 0.7374627976190476, - "exact_fraction": "19823/26880", - "ci95": [ - 0.6992931547619048, - 0.7742950148809524 - ] - }, - "v3_core": { - "point": 0.4608333333333333, - "exact_fraction": "553/1200", - "ci95": [ - 0.43875000000000003, - 0.48291666666666666 - ] - }, - "v4": { - "point": 0.765625, - "exact_fraction": "49/64", - "ci95": [ - 0.7210721817484662, - 0.8086081288343558 - ] - }, - "v5": { - "point": 0.8416666666666667, - "exact_fraction": "101/120", - "ci95": [ - 0.8111792524213076, - 0.8710143196177259 - ] - }, - "v3_joint": { - "point": 0.5991480654761905, - "exact_fraction": "161051/268800", - "ci95": [ - 0.5773286830357143, - 0.6207081473214285 - ] - }, - "expanded_joint": { - "point": 0.7013969494047619, - "exact_fraction": "377071/537600", - "ci95": [ - 0.6845734141379205, - 0.7180071456587547 - ] - }, - "families": { - "old_core": { - "ag_news": { - "correct": 107, - "requested": 128, - "accuracy_ci95": [ - 0.7734375, - 0.8984375 - ] - }, - "boolean_constraints": { - "correct": 32, - "requested": 64, - "accuracy_ci95": [ - 0.328125, - 0.671875 - ] - }, - "dbpedia_14": { - "correct": 107, - "requested": 112, - "accuracy_ci95": [ - 0.9107142857142857, - 0.9910714285714286 - ] - }, - "natural_intents": { - "correct": 64, - "requested": 64, - "accuracy_ci95": [ - 1.0, - 1.0 - ] - }, - "option_carrier": { - "correct": 95, - "requested": 96, - "accuracy_ci95": [ - 0.96875, - 1.0 - ] - }, - "ordinal_rubric": { - "correct": 42, - "requested": 64, - "accuracy_ci95": [ - 0.484375, - 0.8125 - ] - }, - "relational_composition": { - "correct": 36, - "requested": 96, - "accuracy_ci95": [ - 0.19791666666666666, - 0.5625 - ] - }, - "scoped_evidence": { - "correct": 76, - "requested": 96, - "accuracy_ci95": [ - 0.6354166666666666, - 0.9375 - ] - }, - "state_tracking": { - "correct": 26, - "requested": 96, - "accuracy_ci95": [ - 0.125, - 0.4270833333333333 - ] - }, - "unknown_rejection": { - "correct": 64, - "requested": 64, - "accuracy_ci95": [ - 1.0, - 1.0 - ] - } - }, - "v3_core": { - "canonical_record_identity": { - "correct": 40, - "requested": 80, - "accuracy_ci95": [ - 0.5, - 0.5 - ] - }, - "capacitated_assignment": { - "correct": 38, - "requested": 80, - "accuracy_ci95": [ - 0.4375, - 0.5 - ] - }, - "constraint_assignment": { - "correct": 21, - "requested": 80, - "accuracy_ci95": [ - 0.1875, - 0.3125 - ] - }, - "massive_en": { - "correct": 109, - "requested": 120, - "accuracy_ci95": [ - 0.8500000000000001, - 0.9583333333333333 - ] - }, - "massive_zh": { - "correct": 105, - "requested": 120, - "accuracy_ci95": [ - 0.8166666666666667, - 0.9333333333333332 - ] - }, - "multiset_reconciliation": { - "correct": 23, - "requested": 80, - "accuracy_ci95": [ - 0.25, - 0.325 - ] - }, - "ordinal_service_loss": { - "correct": 20, - "requested": 80, - "accuracy_ci95": [ - 0.1875, - 0.3125 - ] - }, - "paraconsistent_rule_closure": { - "correct": 20, - "requested": 80, - "accuracy_ci95": [ - 0.25, - 0.25 - ] - }, - "temporal_exclusion": { - "correct": 33, - "requested": 80, - "accuracy_ci95": [ - 0.275, - 0.5625 - ] - }, - "transaction_recovery": { - "correct": 31, - "requested": 80, - "accuracy_ci95": [ - 0.3, - 0.475 - ] - } - }, - "v4": { - "boolq": { - "correct": 133, - "requested": 160, - "accuracy_ci95": [ - 0.775, - 0.8875 - ] - }, - "belebele_en": { - "correct": 112, - "requested": 160, - "accuracy_ci95": [ - 0.6280452353942144, - 0.7692307692307693 - ] - }, - "belebele_zh": { - "correct": 112, - "requested": 160, - "accuracy_ci95": [ - 0.6289261100440628, - 0.7682926829268293 - ] - } - }, - "v5": { - "cosmos_qa": { - "correct": 83, - "requested": 120, - "accuracy_ci95": [ - 0.6083333333333333, - 0.775 - ] - }, - "squad2_answerability": { - "correct": 100, - "requested": 120, - "accuracy_ci95": [ - 0.7786885245901639, - 0.8859649122807017 - ] - }, - "snli": { - "correct": 106, - "requested": 120, - "accuracy_ci95": [ - 0.825, - 0.9416666666666667 - ] - }, - "qasc": { - "correct": 115, - "requested": 120, - "accuracy_ci95": [ - 0.912, - 0.9918032786885246 - ] - } - } - }, - "proper_scores": { - "old_core": 0.42193820645312435, - "v3_core": 0.6913074500248156, - "v4": 0.33935305266273397, - "v5": 0.250760098789168, - "expanded_weighted_brier": 0.42583970198246046 - }, - "probability_coverage": { - "old_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v3_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v4": { - "requested": 480, - "valid_probability_rows": 480 - }, - "v5": { - "requested": 480, - "valid_probability_rows": 480 - } - }, - "brier_ci95": { - "old_core": [ - 0.365622760488921, - 0.48051917201614724 - ], - "v3_core": [ - 0.6695390244240492, - 0.7144598051610135 - ], - "v4": [ - 0.28386912550452903, - 0.3966449902668765 - ], - "v5": [ - 0.21482273267248006, - 0.28856469231607446 - ], - "expanded_weighted_brier": [ - 0.40383487736748497, - 0.44863844813421033 - ] - } - }, - "kev-9b": { - "old_core": { - "point": 0.7635788690476191, - "exact_fraction": "4105/5376", - "ci95": [ - 0.729052269345238, - 0.798104538690476 - ] - }, - "v3_core": { - "point": 0.4554166666666667, - "exact_fraction": "1093/2400", - "ci95": [ - 0.4304166666666666, - 0.48 - ] - }, - "v4": { - "point": 0.8671875, - "exact_fraction": "111/128", - "ci95": [ - 0.8295129243827161, - 0.9025080271804062 - ] - }, - "v5": { - "point": 0.8375, - "exact_fraction": "67/80", - "ci95": [ - 0.8058128867311815, - 0.8681624928267716 - ] - }, - "v3_joint": { - "point": 0.6094977678571428, - "exact_fraction": "54611/89600", - "ci95": [ - 0.58812109375, - 0.6303536086309525 - ] - }, - "expanded_joint": { - "point": 0.7309207589285714, - "exact_fraction": "130981/179200", - "ci95": [ - 0.7149211829146931, - 0.7468337863817799 - ] - }, - "families": { - "old_core": { - "ag_news": { - "correct": 113, - "requested": 128, - "accuracy_ci95": [ - 0.828125, - 0.9375 - ] - }, - "boolean_constraints": { - "correct": 59, - "requested": 64, - "accuracy_ci95": [ - 0.84375, - 0.984375 - ] - }, - "dbpedia_14": { - "correct": 110, - "requested": 112, - "accuracy_ci95": [ - 0.9553571428571429, - 1.0 - ] - }, - "natural_intents": { - "correct": 64, - "requested": 64, - "accuracy_ci95": [ - 1.0, - 1.0 - ] - }, - "option_carrier": { - "correct": 48, - "requested": 96, - "accuracy_ci95": [ - 0.5, - 0.5 - ] - }, - "ordinal_rubric": { - "correct": 62, - "requested": 64, - "accuracy_ci95": [ - 0.921875, - 1.0 - ] - }, - "relational_composition": { - "correct": 50, - "requested": 96, - "accuracy_ci95": [ - 0.3229166666666667, - 0.7083333333333334 - ] - }, - "scoped_evidence": { - "correct": 61, - "requested": 96, - "accuracy_ci95": [ - 0.4583333333333333, - 0.8125 - ] - }, - "state_tracking": { - "correct": 35, - "requested": 96, - "accuracy_ci95": [ - 0.19791666666666666, - 0.53125 - ] - }, - "unknown_rejection": { - "correct": 55, - "requested": 64, - "accuracy_ci95": [ - 0.75, - 0.953125 - ] - } - }, - "v3_core": { - "canonical_record_identity": { - "correct": 37, - "requested": 80, - "accuracy_ci95": [ - 0.375, - 0.55 - ] - }, - "capacitated_assignment": { - "correct": 40, - "requested": 80, - "accuracy_ci95": [ - 0.5, - 0.5 - ] - }, - "constraint_assignment": { - "correct": 20, - "requested": 80, - "accuracy_ci95": [ - 0.15, - 0.35 - ] - }, - "massive_en": { - "correct": 104, - "requested": 120, - "accuracy_ci95": [ - 0.8, - 0.925 - ] - }, - "massive_zh": { - "correct": 102, - "requested": 120, - "accuracy_ci95": [ - 0.7833333333333333, - 0.9083333333333334 - ] - }, - "multiset_reconciliation": { - "correct": 28, - "requested": 80, - "accuracy_ci95": [ - 0.25, - 0.45 - ] - }, - "ordinal_service_loss": { - "correct": 17, - "requested": 80, - "accuracy_ci95": [ - 0.175, - 0.25 - ] - }, - "paraconsistent_rule_closure": { - "correct": 23, - "requested": 80, - "accuracy_ci95": [ - 0.225, - 0.3625 - ] - }, - "temporal_exclusion": { - "correct": 26, - "requested": 80, - "accuracy_ci95": [ - 0.2375, - 0.4 - ] - }, - "transaction_recovery": { - "correct": 36, - "requested": 80, - "accuracy_ci95": [ - 0.3625, - 0.5375 - ] - } - }, - "v4": { - "boolq": { - "correct": 145, - "requested": 160, - "accuracy_ci95": [ - 0.85625, - 0.95 - ] - }, - "belebele_en": { - "correct": 134, - "requested": 160, - "accuracy_ci95": [ - 0.7751442307692308, - 0.8954248366013072 - ] - }, - "belebele_zh": { - "correct": 131, - "requested": 160, - "accuracy_ci95": [ - 0.7564102564102564, - 0.8773006134969326 - ] - } - }, - "v5": { - "cosmos_qa": { - "correct": 81, - "requested": 120, - "accuracy_ci95": [ - 0.5916666666666667, - 0.7583333333333333 - ] - }, - "squad2_answerability": { - "correct": 97, - "requested": 120, - "accuracy_ci95": [ - 0.7410061813186815, - 0.8769262295081967 - ] - }, - "snli": { - "correct": 108, - "requested": 120, - "accuracy_ci95": [ - 0.8416666666666667, - 0.95 - ] - }, - "qasc": { - "correct": 116, - "requested": 120, - "accuracy_ci95": [ - 0.9333333333333333, - 0.991869918699187 - ] - } - } - }, - "proper_scores": { - "old_core": 0.3579333453812052, - "v3_core": 0.6866233781387137, - "v4": 0.19851043272472813, - "v5": 0.24294908882289837, - "expanded_weighted_brier": 0.37150406126688634 - }, - "probability_coverage": { - "old_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v3_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v4": { - "requested": 480, - "valid_probability_rows": 480 - }, - "v5": { - "requested": 480, - "valid_probability_rows": 480 - } - }, - "brier_ci95": { - "old_core": [ - 0.3081633137767088, - 0.40859251388765916 - ], - "v3_core": [ - 0.656917870943704, - 0.7174003704260583 - ], - "v4": [ - 0.1515888351649323, - 0.24959797650555063 - ], - "v5": [ - 0.20024454898292451, - 0.2894231406637825 - ], - "expanded_weighted_brier": [ - 0.34973259999393813, - 0.39395437787789234 - ] - } - }, - "Decider \u00b7 2B": { - "old_core": { - "point": 0.6401413690476191, - "exact_fraction": "17207/26880", - "ci95": [ - 0.5992559523809524, - 0.6806184895833334 - ] - }, - "v3_core": { - "point": 0.4658333333333333, - "exact_fraction": "559/1200", - "ci95": [ - 0.4362500000000001, - 0.49499999999999994 - ] - }, - "v4": { - "point": 0.9203125, - "exact_fraction": "589/640", - "ci95": [ - 0.8903095518867924, - 0.9479299363057325 - ] - }, - "v5": { - "point": 0.84375, - "exact_fraction": "27/32", - "ci95": [ - 0.8130660699474849, - 0.8734622752789641 - ] - }, - "v3_joint": { - "point": 0.5529873511904762, - "exact_fraction": "148643/268800", - "ci95": [ - 0.5282024739583334, - 0.5777246093749999 - ] - }, - "expanded_joint": { - "point": 0.7175093005952381, - "exact_fraction": "385733/537600", - "ci95": [ - 0.7012097728950374, - 0.7336543123969403 - ] - }, - "families": { - "old_core": { - "ag_news": { - "correct": 111, - "requested": 128, - "accuracy_ci95": [ - 0.8046875, - 0.921875 - ] - }, - "boolean_constraints": { - "correct": 46, - "requested": 64, - "accuracy_ci95": [ - 0.578125, - 0.84375 - ] - }, - "dbpedia_14": { - "correct": 110, - "requested": 112, - "accuracy_ci95": [ - 0.9553571428571429, - 1.0 - ] - }, - "natural_intents": { - "correct": 64, - "requested": 64, - "accuracy_ci95": [ - 1.0, - 1.0 - ] - }, - "option_carrier": { - "correct": 8, - "requested": 96, - "accuracy_ci95": [ - 0.041666666666666664, - 0.13541666666666666 - ] - }, - "ordinal_rubric": { - "correct": 54, - "requested": 64, - "accuracy_ci95": [ - 0.71875, - 0.9375 - ] - }, - "relational_composition": { - "correct": 52, - "requested": 96, - "accuracy_ci95": [ - 0.3854166666666667, - 0.6979166666666666 - ] - }, - "scoped_evidence": { - "correct": 46, - "requested": 96, - "accuracy_ci95": [ - 0.2916666666666667, - 0.6666666666666666 - ] - }, - "state_tracking": { - "correct": 28, - "requested": 96, - "accuracy_ci95": [ - 0.13541666666666666, - 0.4583333333333333 - ] - }, - "unknown_rejection": { - "correct": 38, - "requested": 64, - "accuracy_ci95": [ - 0.375, - 0.796875 - ] - } - }, - "v3_core": { - "canonical_record_identity": { - "correct": 45, - "requested": 80, - "accuracy_ci95": [ - 0.45, - 0.675 - ] - }, - "capacitated_assignment": { - "correct": 41, - "requested": 80, - "accuracy_ci95": [ - 0.4375, - 0.5875 - ] - }, - "constraint_assignment": { - "correct": 19, - "requested": 80, - "accuracy_ci95": [ - 0.1625, - 0.3 - ] - }, - "massive_en": { - "correct": 108, - "requested": 120, - "accuracy_ci95": [ - 0.8416666666666667, - 0.9500000000000001 - ] - }, - "massive_zh": { - "correct": 106, - "requested": 120, - "accuracy_ci95": [ - 0.825, - 0.9333333333333333 - ] - }, - "multiset_reconciliation": { - "correct": 19, - "requested": 80, - "accuracy_ci95": [ - 0.1875, - 0.2875 - ] - }, - "ordinal_service_loss": { - "correct": 18, - "requested": 80, - "accuracy_ci95": [ - 0.175, - 0.275 - ] - }, - "paraconsistent_rule_closure": { - "correct": 21, - "requested": 80, - "accuracy_ci95": [ - 0.1875, - 0.3375 - ] - }, - "temporal_exclusion": { - "correct": 34, - "requested": 80, - "accuracy_ci95": [ - 0.2625, - 0.6125 - ] - }, - "transaction_recovery": { - "correct": 33, - "requested": 80, - "accuracy_ci95": [ - 0.3125, - 0.5125 - ] - } - }, - "v4": { - "boolq": { - "correct": 147, - "requested": 160, - "accuracy_ci95": [ - 0.875, - 0.95625 - ] - }, - "belebele_en": { - "correct": 149, - "requested": 160, - "accuracy_ci95": [ - 0.8910256410256411, - 0.9683544303797469 - ] - }, - "belebele_zh": { - "correct": 146, - "requested": 160, - "accuracy_ci95": [ - 0.8622754491017964, - 0.9559748427672956 - ] - } - }, - "v5": { - "cosmos_qa": { - "correct": 80, - "requested": 120, - "accuracy_ci95": [ - 0.5833333333333334, - 0.75 - ] - }, - "squad2_answerability": { - "correct": 101, - "requested": 120, - "accuracy_ci95": [ - 0.7857142857142857, - 0.896551724137931 - ] - }, - "snli": { - "correct": 109, - "requested": 120, - "accuracy_ci95": [ - 0.85, - 0.9583333333333334 - ] - }, - "qasc": { - "correct": 115, - "requested": 120, - "accuracy_ci95": [ - 0.9112903225806451, - 0.991869918699187 - ] - } - } - }, - "proper_scores": { - "old_core": 0.44246049270704674, - "v3_core": 0.5974556966463511, - "v4": 0.13737783476250795, - "v5": 0.23664932218586204, - "expanded_weighted_brier": 0.35348583657544197 - }, - "probability_coverage": { - "old_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v3_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v4": { - "requested": 480, - "valid_probability_rows": 480 - }, - "v5": { - "requested": 480, - "valid_probability_rows": 480 - } - }, - "brier_ci95": { - "old_core": [ - 0.40579693831982216, - 0.4787313843802038 - ], - "v3_core": [ - 0.5767822346729764, - 0.6189924929703139 - ], - "v4": [ - 0.10073496035335808, - 0.17666521563140353 - ], - "v5": [ - 0.2031026857962821, - 0.27234790035750356 - ], - "expanded_weighted_brier": [ - 0.3373790190368837, - 0.37038546676401063 - ] - } - }, - "Laya \u00b7 EN/ML": { - "old_core": { - "point": 0.5701264880952381, - "exact_fraction": "3065/5376", - "ci95": [ - 0.5321056547619049, - 0.6094494047619048 - ] - }, - "v3_core": { - "point": 0.3775, - "exact_fraction": "151/400", - "ci95": [ - 0.35458333333333336, - 0.39958333333333335 - ] - }, - "v4": { - "point": 0.5125, - "exact_fraction": "41/80", - "ci95": [ - 0.4661671277501309, - 0.557454268292683 - ] - }, - "v5": { - "point": 0.6375, - "exact_fraction": "51/80", - "ci95": [ - 0.6032948178983685, - 0.6716809713450388 - ] - }, - "v3_joint": { - "point": 0.47381324404761904, - "exact_fraction": "127361/268800", - "ci95": [ - 0.4519083891369048, - 0.4964252232142857 - ] - }, - "expanded_joint": { - "point": 0.5244066220238095, - "exact_fraction": "281921/537600", - "ci95": [ - 0.5065901244344904, - 0.5426481205211884 - ] - }, - "families": { - "old_core": { - "ag_news": { - "correct": 117, - "requested": 128, - "accuracy_ci95": [ - 0.859375, - 0.9609375 - ] - }, - "boolean_constraints": { - "correct": 28, - "requested": 64, - "accuracy_ci95": [ - 0.28125, - 0.609375 - ] - }, - "dbpedia_14": { - "correct": 94, - "requested": 112, - "accuracy_ci95": [ - 0.7678571428571429, - 0.9017857142857143 - ] - }, - "natural_intents": { - "correct": 55, - "requested": 64, - "accuracy_ci95": [ - 0.75, - 0.9375 - ] - }, - "option_carrier": { - "correct": 92, - "requested": 96, - "accuracy_ci95": [ - 0.8854166666666666, - 1.0 - ] - }, - "ordinal_rubric": { - "correct": 12, - "requested": 64, - "accuracy_ci95": [ - 0.0625, - 0.34375 - ] - }, - "relational_composition": { - "correct": 24, - "requested": 96, - "accuracy_ci95": [ - 0.17708333333333334, - 0.3333333333333333 - ] - }, - "scoped_evidence": { - "correct": 36, - "requested": 96, - "accuracy_ci95": [ - 0.23958333333333334, - 0.5208333333333334 - ] - }, - "state_tracking": { - "correct": 23, - "requested": 96, - "accuracy_ci95": [ - 0.11458333333333333, - 0.3854166666666667 - ] - }, - "unknown_rejection": { - "correct": 41, - "requested": 64, - "accuracy_ci95": [ - 0.4375, - 0.84375 - ] - } - }, - "v3_core": { - "canonical_record_identity": { - "correct": 38, - "requested": 80, - "accuracy_ci95": [ - 0.425, - 0.525 - ] - }, - "capacitated_assignment": { - "correct": 51, - "requested": 80, - "accuracy_ci95": [ - 0.55, - 0.725 - ] - }, - "constraint_assignment": { - "correct": 18, - "requested": 80, - "accuracy_ci95": [ - 0.15, - 0.3125 - ] - }, - "massive_en": { - "correct": 89, - "requested": 120, - "accuracy_ci95": [ - 0.6583333333333333, - 0.8166666666666667 - ] - }, - "massive_zh": { - "correct": 88, - "requested": 120, - "accuracy_ci95": [ - 0.6583333333333332, - 0.8083333333333333 - ] - }, - "multiset_reconciliation": { - "correct": 21, - "requested": 80, - "accuracy_ci95": [ - 0.25, - 0.2875 - ] - }, - "ordinal_service_loss": { - "correct": 16, - "requested": 80, - "accuracy_ci95": [ - 0.15, - 0.25 - ] - }, - "paraconsistent_rule_closure": { - "correct": 15, - "requested": 80, - "accuracy_ci95": [ - 0.1375, - 0.2375 - ] - }, - "temporal_exclusion": { - "correct": 10, - "requested": 80, - "accuracy_ci95": [ - 0.05, - 0.2125 - ] - }, - "transaction_recovery": { - "correct": 15, - "requested": 80, - "accuracy_ci95": [ - 0.1, - 0.2625 - ] - } - }, - "v4": { - "boolq": { - "correct": 111, - "requested": 160, - "accuracy_ci95": [ - 0.61875, - 0.7625 - ] - }, - "belebele_en": { - "correct": 61, - "requested": 160, - "accuracy_ci95": [ - 0.3062370621019109, - 0.4585987261146497 - ] - }, - "belebele_zh": { - "correct": 45, - "requested": 160, - "accuracy_ci95": [ - 0.2138364779874214, - 0.3503184713375796 - ] - } - }, - "v5": { - "cosmos_qa": { - "correct": 36, - "requested": 120, - "accuracy_ci95": [ - 0.21666666666666667, - 0.38333333333333336 - ] - }, - "squad2_answerability": { - "correct": 69, - "requested": 120, - "accuracy_ci95": [ - 0.5084745762711864, - 0.6416666666666667 - ] - }, - "snli": { - "correct": 87, - "requested": 120, - "accuracy_ci95": [ - 0.6416666666666667, - 0.8 - ] - }, - "qasc": { - "correct": 114, - "requested": 120, - "accuracy_ci95": [ - 0.9083333333333333, - 0.983739837398374 - ] - } - } - }, - "proper_scores": { - "old_core": 0.5709760997718798, - "v3_core": 0.700479457442883, - "v4": 0.6149664146731222, - "v5": 0.4747850267787647, - "expanded_weighted_brier": 0.5903017496666624 - }, - "probability_coverage": { - "old_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v3_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v4": { - "requested": 480, - "valid_probability_rows": 480 - }, - "v5": { - "requested": 480, - "valid_probability_rows": 480 - } - }, - "brier_ci95": { - "old_core": [ - 0.5307603405543191, - 0.611972788595099 - ], - "v3_core": [ - 0.677830213791377, - 0.7242813502662896 - ], - "v4": [ - 0.5601239092953858, - 0.6718388215324546 - ], - "v5": [ - 0.4441678879307976, - 0.5061559610657542 - ], - "expanded_weighted_brier": [ - 0.5709748439741565, - 0.6102902589236013 - ] - } - }, - "Qwen3.5 \u00b7 2B, untuned": { - "old_core": { - "point": 0.5712425595238095, - "exact_fraction": "3071/5376", - "ci95": [ - 0.535453869047619, - 0.607627418154762 - ] - }, - "v3_core": { - "point": 0.39, - "exact_fraction": "39/100", - "ci95": [ - 0.3733333333333334, - 0.40625 - ] - }, - "v4": { - "point": 0.7375, - "exact_fraction": "59/80", - "ci95": [ - 0.6928571428571428, - 0.782292158018868 - ] - }, - "v5": { - "point": 0.7229166666666667, - "exact_fraction": "347/480", - "ci95": [ - 0.6848579953997695, - 0.760121271215623 - ] - }, - "v3_joint": { - "point": 0.4806212797619048, - "exact_fraction": "129191/268800", - "ci95": [ - 0.46085509672619046, - 0.5003615141369048 - ] - }, - "expanded_joint": { - "point": 0.605414806547619, - "exact_fraction": "325471/537600", - "ci95": [ - 0.5881936025503637, - 0.6231347666599296 - ] - }, - "families": { - "old_core": { - "ag_news": { - "correct": 103, - "requested": 128, - "accuracy_ci95": [ - 0.734375, - 0.8671875 - ] - }, - "boolean_constraints": { - "correct": 24, - "requested": 64, - "accuracy_ci95": [ - 0.21875, - 0.53125 - ] - }, - "dbpedia_14": { - "correct": 104, - "requested": 112, - "accuracy_ci95": [ - 0.875, - 0.9732142857142857 - ] - }, - "natural_intents": { - "correct": 62, - "requested": 64, - "accuracy_ci95": [ - 0.921875, - 1.0 - ] - }, - "option_carrier": { - "correct": 9, - "requested": 96, - "accuracy_ci95": [ - 0.041666666666666664, - 0.15625 - ] - }, - "ordinal_rubric": { - "correct": 52, - "requested": 64, - "accuracy_ci95": [ - 0.671875, - 0.9375 - ] - }, - "relational_composition": { - "correct": 47, - "requested": 96, - "accuracy_ci95": [ - 0.375, - 0.6145833333333334 - ] - }, - "scoped_evidence": { - "correct": 35, - "requested": 96, - "accuracy_ci95": [ - 0.23958333333333334, - 0.5 - ] - }, - "state_tracking": { - "correct": 24, - "requested": 96, - "accuracy_ci95": [ - 0.15625, - 0.3541666666666667 - ] - }, - "unknown_rejection": { - "correct": 40, - "requested": 64, - "accuracy_ci95": [ - 0.4375, - 0.8125 - ] - } - }, - "v3_core": { - "canonical_record_identity": { - "correct": 37, - "requested": 80, - "accuracy_ci95": [ - 0.425, - 0.5 - ] - }, - "capacitated_assignment": { - "correct": 40, - "requested": 80, - "accuracy_ci95": [ - 0.5, - 0.5 - ] - }, - "constraint_assignment": { - "correct": 17, - "requested": 80, - "accuracy_ci95": [ - 0.15, - 0.275 - ] - }, - "massive_en": { - "correct": 85, - "requested": 120, - "accuracy_ci95": [ - 0.625, - 0.7833333333333334 - ] - }, - "massive_zh": { - "correct": 89, - "requested": 120, - "accuracy_ci95": [ - 0.6666666666666666, - 0.8166666666666667 - ] - }, - "multiset_reconciliation": { - "correct": 22, - "requested": 80, - "accuracy_ci95": [ - 0.25, - 0.3125 - ] - }, - "ordinal_service_loss": { - "correct": 16, - "requested": 80, - "accuracy_ci95": [ - 0.2, - 0.2 - ] - }, - "paraconsistent_rule_closure": { - "correct": 22, - "requested": 80, - "accuracy_ci95": [ - 0.25, - 0.3125 - ] - }, - "temporal_exclusion": { - "correct": 23, - "requested": 80, - "accuracy_ci95": [ - 0.2375, - 0.3375 - ] - }, - "transaction_recovery": { - "correct": 19, - "requested": 80, - "accuracy_ci95": [ - 0.2125, - 0.25 - ] - } - }, - "v4": { - "boolq": { - "correct": 104, - "requested": 160, - "accuracy_ci95": [ - 0.575, - 0.725 - ] - }, - "belebele_en": { - "correct": 134, - "requested": 160, - "accuracy_ci95": [ - 0.7791411042944786, - 0.8924050632911392 - ] - }, - "belebele_zh": { - "correct": 130, - "requested": 160, - "accuracy_ci95": [ - 0.7484276729559748, - 0.8742138364779874 - ] - } - }, - "v5": { - "cosmos_qa": { - "correct": 68, - "requested": 120, - "accuracy_ci95": [ - 0.48333333333333334, - 0.65 - ] - }, - "squad2_answerability": { - "correct": 99, - "requested": 120, - "accuracy_ci95": [ - 0.7631578947368421, - 0.8833333333333333 - ] - }, - "snli": { - "correct": 77, - "requested": 120, - "accuracy_ci95": [ - 0.55, - 0.725 - ] - }, - "qasc": { - "correct": 103, - "requested": 120, - "accuracy_ci95": [ - 0.7950819672131147, - 0.9173553719008265 - ] - } - } - }, - "proper_scores": { - "old_core": 0.5723777597574953, - "v3_core": 0.7477724847216414, - "v4": 0.31678065323351606, - "v5": 0.4151003191775817, - "expanded_weighted_brier": 0.5130078042225586 - }, - "probability_coverage": { - "old_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v3_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v4": { - "requested": 480, - "valid_probability_rows": 480 - }, - "v5": { - "requested": 480, - "valid_probability_rows": 480 - } - }, - "brier_ci95": { - "old_core": [ - 0.5306199339702083, - 0.613544919176592 - ], - "v3_core": [ - 0.7296525923390248, - 0.7668117066670981 - ], - "v4": [ - 0.27446950339258713, - 0.3607698891917115 - ], - "v5": [ - 0.37186021348012016, - 0.45924139988994434 - ], - "expanded_weighted_brier": [ - 0.4934107964402564, - 0.5319870117266131 - ] - } - }, - "Qwen3.5 \u00b7 4B, untuned": { - "old_core": { - "point": 0.6988839285714286, - "exact_fraction": "3131/4480", - "ci95": [ - 0.6632803199404762, - 0.7337797619047619 - ] - }, - "v3_core": { - "point": 0.43333333333333335, - "exact_fraction": "13/30", - "ci95": [ - 0.4179166666666667, - 0.4479166666666667 - ] - }, - "v4": { - "point": 0.8796875, - "exact_fraction": "563/640", - "ci95": [ - 0.8461463032298809, - 0.9125000000000001 - ] - }, - "v5": { - "point": 0.7979166666666667, - "exact_fraction": "383/480", - "ci95": [ - 0.7641944237372369, - 0.830579286387442 - ] - }, - "v3_joint": { - "point": 0.566108630952381, - "exact_fraction": "15217/26880", - "ci95": [ - 0.5467666480654761, - 0.5849108072916667 - ] - }, - "expanded_joint": { - "point": 0.7024553571428571, - "exact_fraction": "3147/4480", - "ci95": [ - 0.6873480253417569, - 0.7176086887612844 - ] - }, - "families": { - "old_core": { - "ag_news": { - "correct": 108, - "requested": 128, - "accuracy_ci95": [ - 0.78125, - 0.90625 - ] - }, - "boolean_constraints": { - "correct": 40, - "requested": 64, - "accuracy_ci95": [ - 0.484375, - 0.765625 - ] - }, - "dbpedia_14": { - "correct": 109, - "requested": 112, - "accuracy_ci95": [ - 0.9375, - 1.0 - ] - }, - "natural_intents": { - "correct": 64, - "requested": 64, - "accuracy_ci95": [ - 1.0, - 1.0 - ] - }, - "option_carrier": { - "correct": 70, - "requested": 96, - "accuracy_ci95": [ - 0.6354166666666666, - 0.8229166666666666 - ] - }, - "ordinal_rubric": { - "correct": 58, - "requested": 64, - "accuracy_ci95": [ - 0.8125, - 0.984375 - ] - }, - "relational_composition": { - "correct": 49, - "requested": 96, - "accuracy_ci95": [ - 0.3854166666666667, - 0.6354166666666666 - ] - }, - "scoped_evidence": { - "correct": 49, - "requested": 96, - "accuracy_ci95": [ - 0.3645833333333333, - 0.65625 - ] - }, - "state_tracking": { - "correct": 27, - "requested": 96, - "accuracy_ci95": [ - 0.16666666666666666, - 0.3958333333333333 - ] - }, - "unknown_rejection": { - "correct": 39, - "requested": 64, - "accuracy_ci95": [ - 0.421875, - 0.7816406249999943 - ] - } - }, - "v3_core": { - "canonical_record_identity": { - "correct": 40, - "requested": 80, - "accuracy_ci95": [ - 0.5, - 0.5 - ] - }, - "capacitated_assignment": { - "correct": 40, - "requested": 80, - "accuracy_ci95": [ - 0.5, - 0.5 - ] - }, - "constraint_assignment": { - "correct": 19, - "requested": 80, - "accuracy_ci95": [ - 0.1625, - 0.3125 - ] - }, - "massive_en": { - "correct": 110, - "requested": 120, - "accuracy_ci95": [ - 0.8666666666666667, - 0.9583333333333334 - ] - }, - "massive_zh": { - "correct": 104, - "requested": 120, - "accuracy_ci95": [ - 0.8, - 0.925 - ] - }, - "multiset_reconciliation": { - "correct": 22, - "requested": 80, - "accuracy_ci95": [ - 0.225, - 0.325 - ] - }, - "ordinal_service_loss": { - "correct": 17, - "requested": 80, - "accuracy_ci95": [ - 0.2, - 0.2375 - ] - }, - "paraconsistent_rule_closure": { - "correct": 20, - "requested": 80, - "accuracy_ci95": [ - 0.2125, - 0.2875 - ] - }, - "temporal_exclusion": { - "correct": 25, - "requested": 80, - "accuracy_ci95": [ - 0.25, - 0.375 - ] - }, - "transaction_recovery": { - "correct": 21, - "requested": 80, - "accuracy_ci95": [ - 0.25, - 0.2875 - ] - } - }, - "v4": { - "boolq": { - "correct": 134, - "requested": 160, - "accuracy_ci95": [ - 0.78125, - 0.89375 - ] - }, - "belebele_en": { - "correct": 149, - "requested": 160, - "accuracy_ci95": [ - 0.8910256410256411, - 0.967948717948718 - ] - }, - "belebele_zh": { - "correct": 146, - "requested": 160, - "accuracy_ci95": [ - 0.8633475672877846, - 0.95625 - ] - } - }, - "v5": { - "cosmos_qa": { - "correct": 73, - "requested": 120, - "accuracy_ci95": [ - 0.5166666666666667, - 0.6916666666666667 - ] - }, - "squad2_answerability": { - "correct": 98, - "requested": 120, - "accuracy_ci95": [ - 0.7540983606557377, - 0.8790322580645161 - ] - }, - "snli": { - "correct": 96, - "requested": 120, - "accuracy_ci95": [ - 0.725, - 0.8666666666666667 - ] - }, - "qasc": { - "correct": 116, - "requested": 120, - "accuracy_ci95": [ - 0.9333333333333333, - 0.991869918699187 - ] - } - } - }, - "proper_scores": { - "old_core": 0.427127994572904, - "v3_core": 0.6845537392617043, - "v4": 0.19639894537679964, - "v5": 0.2904012063572707, - "expanded_weighted_brier": 0.39962047139216966 - }, - "probability_coverage": { - "old_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v3_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v4": { - "requested": 480, - "valid_probability_rows": 480 - }, - "v5": { - "requested": 480, - "valid_probability_rows": 480 - } - }, - "brier_ci95": { - "old_core": [ - 0.377528048361006, - 0.4790590681446513 - ], - "v3_core": [ - 0.6674510320346829, - 0.7036301143486854 - ], - "v4": [ - 0.1623582012987519, - 0.23277732548629693 - ], - "v5": [ - 0.24711830906655044, - 0.33490954796052225 - ], - "expanded_weighted_brier": [ - 0.38062758349672393, - 0.4193732417566481 - ] - } - }, - "Jev \u00b7 1.13.0": { - "old_core": { - "point": 0.7909598214285715, - "exact_fraction": "7087/8960", - "ci95": [ - 0.7616062127976191, - 0.819271763392857 - ] - }, - "v3_core": { - "point": 0.66375, - "exact_fraction": "531/800", - "ci95": [ - 0.6316666666666667, - 0.6945833333333333 - ] - }, - "v4": { - "point": 0.9453125, - "exact_fraction": "121/128", - "ci95": [ - 0.9197115384615384, - 0.9685135514495712 - ] - }, - "v5": { - "point": 0.8979166666666667, - "exact_fraction": "431/480", - "ci95": [ - 0.8713251366120218, - 0.9229854264081012 - ] - }, - "v3_joint": { - "point": 0.7273549107142857, - "exact_fraction": "65171/89600", - "ci95": [ - 0.7058627232142857, - 0.7482447916666666 - ] - }, - "expanded_joint": { - "point": 0.8244847470238095, - "exact_fraction": "443243/537600", - "ci95": [ - 0.8104711862785249, - 0.8378880015726611 - ] - }, - "families": { - "old_core": { - "ag_news": { - "correct": 109, - "requested": 128, - "accuracy_ci95": [ - 0.7890625, - 0.90625 - ] - }, - "boolean_constraints": { - "correct": 64, - "requested": 64, - "accuracy_ci95": [ - 1.0, - 1.0 - ] - }, - "dbpedia_14": { - "correct": 108, - "requested": 112, - "accuracy_ci95": [ - 0.9285714285714286, - 0.9910714285714286 - ] - }, - "natural_intents": { - "correct": 64, - "requested": 64, - "accuracy_ci95": [ - 1.0, - 1.0 - ] - }, - "option_carrier": { - "correct": 29, - "requested": 96, - "accuracy_ci95": [ - 0.2604166666666667, - 0.34375 - ] - }, - "ordinal_rubric": { - "correct": 64, - "requested": 64, - "accuracy_ci95": [ - 1.0, - 1.0 - ] - }, - "relational_composition": { - "correct": 54, - "requested": 96, - "accuracy_ci95": [ - 0.3645833333333333, - 0.75 - ] - }, - "scoped_evidence": { - "correct": 86, - "requested": 96, - "accuracy_ci95": [ - 0.78125, - 0.9895833333333334 - ] - }, - "state_tracking": { - "correct": 32, - "requested": 96, - "accuracy_ci95": [ - 0.16640625000000023, - 0.5106770833333295 - ] - }, - "unknown_rejection": { - "correct": 64, - "requested": 64, - "accuracy_ci95": [ - 1.0, - 1.0 - ] - } - }, - "v3_core": { - "canonical_record_identity": { - "correct": 61, - "requested": 80, - "accuracy_ci95": [ - 0.6375, - 0.875 - ] - }, - "capacitated_assignment": { - "correct": 55, - "requested": 80, - "accuracy_ci95": [ - 0.575, - 0.7875 - ] - }, - "constraint_assignment": { - "correct": 44, - "requested": 80, - "accuracy_ci95": [ - 0.4, - 0.675 - ] - }, - "massive_en": { - "correct": 110, - "requested": 120, - "accuracy_ci95": [ - 0.8666666666666667, - 0.9583333333333334 - ] - }, - "massive_zh": { - "correct": 106, - "requested": 120, - "accuracy_ci95": [ - 0.825, - 0.9416666666666667 - ] - }, - "multiset_reconciliation": { - "correct": 47, - "requested": 80, - "accuracy_ci95": [ - 0.5125, - 0.65 - ] - }, - "ordinal_service_loss": { - "correct": 34, - "requested": 80, - "accuracy_ci95": [ - 0.325, - 0.525 - ] - }, - "paraconsistent_rule_closure": { - "correct": 53, - "requested": 80, - "accuracy_ci95": [ - 0.5875, - 0.725 - ] - }, - "temporal_exclusion": { - "correct": 33, - "requested": 80, - "accuracy_ci95": [ - 0.3125, - 0.5125 - ] - }, - "transaction_recovery": { - "correct": 60, - "requested": 80, - "accuracy_ci95": [ - 0.625, - 0.8625 - ] - } - }, - "v4": { - "boolq": { - "correct": 148, - "requested": 160, - "accuracy_ci95": [ - 0.88125, - 0.9625 - ] - }, - "belebele_en": { - "correct": 155, - "requested": 160, - "accuracy_ci95": [ - 0.9390243902439024, - 0.9937106918238994 - ] - }, - "belebele_zh": { - "correct": 154, - "requested": 160, - "accuracy_ci95": [ - 0.930379746835443, - 0.9877300613496932 - ] - } - }, - "v5": { - "cosmos_qa": { - "correct": 104, - "requested": 120, - "accuracy_ci95": [ - 0.8, - 0.925 - ] - }, - "squad2_answerability": { - "correct": 109, - "requested": 120, - "accuracy_ci95": [ - 0.8620689655172413, - 0.9508196721311475 - ] - }, - "snli": { - "correct": 99, - "requested": 120, - "accuracy_ci95": [ - 0.75, - 0.8916666666666667 - ] - }, - "qasc": { - "correct": 119, - "requested": 120, - "accuracy_ci95": [ - 0.9745762711864406, - 1.0 - ] - } - } - }, - "proper_scores": { - "old_core": 0.24806216931976088, - "v3_core": 0.4403896043771044, - "v4": 0.07270531250000001, - "v5": 0.15378333333333333, - "expanded_weighted_brier": 0.22873510488254967 - }, - "probability_coverage": { - "old_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v3_core": { - "requested": 880, - "valid_probability_rows": 880 - }, - "v4": { - "requested": 480, - "valid_probability_rows": 480 - }, - "v5": { - "requested": 480, - "valid_probability_rows": 480 - } - }, - "brier_ci95": { - "old_core": [ - 0.21912044641352113, - 0.27809132807795134 - ], - "v3_core": [ - 0.4130360873769258, - 0.4688325815363228 - ], - "v4": [ - 0.04802243933725408, - 0.100621887515886 - ], - "v5": [ - 0.12116203971412033, - 0.18851406142759564 - ], - "expanded_weighted_brier": [ - 0.21435641903529326, - 0.2433957477840872 - ] - } - } - }, - "method": { - "seed": 202609225801, - "repeats": 10000, - "weights": { - "old_core": "1/4", - "v3_core": "1/4", - "v4": "1/4", - "v5": "1/4" - }, - "within_panel_weights": { - "old_core": { - "ag_news": "1/10", - "boolean_constraints": "1/10", - "dbpedia_14": "1/10", - "natural_intents": "1/10", - "option_carrier": "1/10", - "ordinal_rubric": "1/10", - "relational_composition": "1/10", - "scoped_evidence": "1/10", - "state_tracking": "1/10", - "unknown_rejection": "1/10" - }, - "v3_core": { - "canonical_record_identity": "1/10", - "capacitated_assignment": "1/10", - "constraint_assignment": "1/10", - "massive_en": "1/10", - "massive_zh": "1/10", - "multiset_reconciliation": "1/10", - "ordinal_service_loss": "1/10", - "paraconsistent_rule_closure": "1/10", - "temporal_exclusion": "1/10", - "transaction_recovery": "1/10" - }, - "v4": { - "boolq": "1/2", - "belebele_en": "1/4", - "belebele_zh": "1/4" - }, - "v5": { - "cosmos_qa": "1/4", - "squad2_answerability": "1/4", - "snli": "1/4", - "qasc": "1/4" - } - }, - "v3_joint_unchanged": "(old_core+v3_core)/2", - "exact_accuracy": "Reduced Fraction from integer family correct/total counts; no rounded eligibility.", - "native_decision": "Use recorded prediction; never replace it with FP64-softmax argmax.", - "brier": "Recorded valid probability vector normalized once as specified by shared score.assess; sum squared category errors, no division by K. Complete coverage required for a panel proper-score point.", - "clusters": "Original V3 within-family groups and coupled MASSIVE translations; original V4 shared passages/translations; original V5 complete source/near components. Identical panel seeds reused across models and accuracy/Brier. Panels sampled independently.", - "interval_role": "Report only; no new CI/family gate.", - "native_supplements": "Both 220-answer native panels remain outside the quality average, fully reported." - } -} diff --git a/metrics/figure-evidence.json b/metrics/figure-evidence.json deleted file mode 100644 index 6fb76b65d32ba39bd7305c8b98a7160d1284d792..0000000000000000000000000000000000000000 --- a/metrics/figure-evidence.json +++ /dev/null @@ -1,278 +0,0 @@ -{ - "format": "decision-expanded-figure-evidence-v1", - "kind": "measured", - "manifest_sha256": "182dfbad245c160bddd7ad6627d076e8a3dd9c7117bac9c6da2c2e8240163fbc", - "statistics_sha256": "7b8936b43bce9d19d9010ab61225a6460160b3aa76c37f5d6df5ba650b433c6c", - "public_projection_sha256": "69477b6a3f6693d8af367f2ecbcc14db3a485871c0db58ee9bb07d752ed4f971", - "base_style_sha256": "93fa22aa96a6dd5f06fe9a5670d84691f2092f84a91bac600c9777a29a36123a", - "branding_sha256": "d83cf7878c3839f3085feb8c02334217ad2fef0a615b805e1726b292552c46e8", - "source_sha256": { - "render.py": "ce19f3a2e3c9ca38b79373ebcec9ad4fcbde05d6acb6089cfb3817e1057985c2", - "contract.py": "dbddb4094f8c412d52aa4e195c59a669e7b9c084f6c5b9105f1114e6d12ef445" - }, - "display_order": [ - "Sol-composition-selected", - "Nox-retention-selected", - "kev-9b", - "Decider \u00b7 2B", - "Qwen3.5 \u00b7 4B, untuned", - "Qwen3.5 \u00b7 2B, untuned", - "Laya \u00b7 EN/ML", - "Jev \u00b7 1.13.0" - ], - "display_labels": { - "Nox-retention-selected": { - "label": "Nox", - "detail": "4B \u00b7 v1.3", - "color": "#965837" - }, - "Sol-composition-selected": { - "label": "Sol", - "detail": "2B \u00b7 v1.3", - "color": "#B07B51" - }, - "kev-9b": { - "label": "Kev", - "detail": "9B" - }, - "Decider \u00b7 2B": { - "label": "Decider", - "detail": "2B" - }, - "Laya \u00b7 EN/ML": { - "label": "Laya", - "detail": "Upstream default" - }, - "Qwen3.5 \u00b7 2B, untuned": { - "label": "Qwen3.5", - "detail": "2B \u00b7 untuned" - }, - "Qwen3.5 \u00b7 4B, untuned": { - "label": "Qwen3.5", - "detail": "4B \u00b7 untuned" - }, - "Jev \u00b7 1.13.0": { - "label": "Jev", - "detail": "1.13.0 \u00b7 frontier", - "color": "#879095" - } - }, - "model_versions": { - "Nox-retention-selected": { - "repo_id": "llm-semantic-router/Decision-1.0-Nox", - "revision": "ad089ad3a5dc9a7a21e6d96db546bb53e2212654", - "release_tag": "v1.3", - "bundle_manifest_sha256": "92d7f5be5e1ef21ee682f574b35cc01de4dd8ab16b8aa6edf0dfbdfcfcfabba3", - "publication_receipt_sha256": "fe15d9e49608c6d2d89301272a993e7d17cffe0e5b0966c32e9bb2740f3ad1e6", - "statistics_sha256": "0eb70b3dbc8a5ab8d78279beadd3ab9796623dfe64218dae06ca5edb2795f46d" - }, - "Sol-composition-selected": { - "repo_id": "llm-semantic-router/Decision-1.0-Sol", - "revision": "2412d9470d3aa125b346262dad80f6161847aa09", - "release_tag": "v1.3", - "bundle_manifest_sha256": "6f0970bfbe5594feecffc14b4824a92080ef09318e12567bd22264c3844e2fc4", - "publication_receipt_sha256": "7316ce19ffc5bbdd440316f7899dc33a8eecb9c36ebdd0cd9f9f8b49f48a446f", - "statistics_sha256": "fe2be9746c4302b829b5aa373c027910cfa84881b80c2068af7c69db76c4659c" - }, - "kev-9b": { - "repo_id": "jaredpalmer/kev-9b", - "revision": "6281032426a9ca3a08374d137bf9dfd8afd78e8a", - "source_revision": "e0bcf50153f1bda4ca6a8be5e12cbd5f9ebbce1c", - "weight_fingerprint": "2b29b6a7e43e94b467a0d516e5d17008cba0f5c6e36c2793c57140604accc4a9" - } - }, - "weights": { - "old_core": "1/4", - "v3_core": "1/4", - "v4": "1/4", - "v5": "1/4" - }, - "within_panel_weights": { - "old_core": { - "ag_news": "1/10", - "boolean_constraints": "1/10", - "dbpedia_14": "1/10", - "natural_intents": "1/10", - "option_carrier": "1/10", - "ordinal_rubric": "1/10", - "relational_composition": "1/10", - "scoped_evidence": "1/10", - "state_tracking": "1/10", - "unknown_rejection": "1/10" - }, - "v3_core": { - "canonical_record_identity": "1/10", - "capacitated_assignment": "1/10", - "constraint_assignment": "1/10", - "massive_en": "1/10", - "massive_zh": "1/10", - "multiset_reconciliation": "1/10", - "ordinal_service_loss": "1/10", - "paraconsistent_rule_closure": "1/10", - "temporal_exclusion": "1/10", - "transaction_recovery": "1/10" - }, - "v4": { - "boolq": "1/2", - "belebele_en": "1/4", - "belebele_zh": "1/4" - }, - "v5": { - "cosmos_qa": "1/4", - "squad2_answerability": "1/4", - "snli": "1/4", - "qasc": "1/4" - } - }, - "families": 27, - "quality_decisions_per_model": 2720, - "strict_external_reference_bolding": true, - "ties_not_bold": true, - "frontier_not_bold": true, - "no_raw_examples_or_predictions_read": true, - "background": "white", - "footnotes_in_image": false, - "qualification_or_publication_performed": false, - "files": [ - { - "file": "TABLES.md", - "bytes": 5093, - "sha256": "2f078e7339457e04a8832c068a9cf9b67aaff39b147794006a17ddd1a72bf41a" - }, - { - "file": "decision-expanded-old_core-600px.png", - "bytes": 167493, - "sha256": "6d36e595bea7465903ee787d217ef07ad648472d336655e88e51263325d0a7e2" - }, - { - "file": "decision-expanded-old_core.pdf", - "bytes": 44384, - "sha256": "cbd80145027dd79b37ed799a64b4c173d44973b358bc05414da80355ba7072a2" - }, - { - "file": "decision-expanded-old_core.png", - "bytes": 312133, - "sha256": "2744aafba8afb133b1c92d7b1813d3e769da8babe9439e3d966f94ac28963bbd" - }, - { - "file": "decision-expanded-old_core.svg", - "bytes": 79821, - "sha256": "0648a4f0ed3720dcc9dc0f19dd43c2f46e08b2dce1cf7a057dba957fa475f288" - }, - { - "file": "decision-expanded-overview-600px.png", - "bytes": 94850, - "sha256": "b8e8361345b814083eef97dc08b9a4fd173bcef83faa72c5f66eac5d0cc1c10b" - }, - { - "file": "decision-expanded-overview.pdf", - "bytes": 41563, - "sha256": "4f008072cb5b8268a31e058d0839d68e2416c7f63795ff40cd931906cf74552e" - }, - { - "file": "decision-expanded-overview.png", - "bytes": 187355, - "sha256": "2f12699c29ecde8146482e63925b9e80e3bab4047f79fbec17a87e50a1e64ec1" - }, - { - "file": "decision-expanded-overview.svg", - "bytes": 48619, - "sha256": "6565620295efd3180f1c2079ea7fec332f17b977fcc9781101e9a82001eff743" - }, - { - "file": "decision-expanded-ranking-600px.png", - "bytes": 83184, - "sha256": "ddbb71c528b3741f1258367e5436b35148461c628aaf4d03d209ae6c4eccdee1" - }, - { - "file": "decision-expanded-ranking.pdf", - "bytes": 37133, - "sha256": "bac1d9bf032cd06d4d81170fd10e033e56c1a793e958e9599f8bc570700bf400" - }, - { - "file": "decision-expanded-ranking.png", - "bytes": 186098, - "sha256": "a78d29ea6a73eab02cc8897140614d92b0d7bb4e1295c43c9a2ece60d03ec155" - }, - { - "file": "decision-expanded-ranking.svg", - "bytes": 38118, - "sha256": "8de46cff93bdf4b899e74b6b7baddafae6956b6e32b9713702472c8aace0a164" - }, - { - "file": "decision-expanded-v3_core-600px.png", - "bytes": 168267, - "sha256": "df7099ba80f0a000f7c64cfe6b8aca7ead06d5311e6967a49ae8de32317a2077" - }, - { - "file": "decision-expanded-v3_core.pdf", - "bytes": 44464, - "sha256": "2ffe3a8fdad5dc957f524dbb03d437aaa50a098920358322575f84d65e95e09f" - }, - { - "file": "decision-expanded-v3_core.png", - "bytes": 300452, - "sha256": "a621a7ffee11dc4909e89a200b7c1207a75bff57a9a237b66e8ed8f2fbac6983" - }, - { - "file": "decision-expanded-v3_core.svg", - "bytes": 79901, - "sha256": "75c74babb4214337caff4b3a5ec36bb1de5824f486312bde33ab3264d2c50f48" - }, - { - "file": "decision-expanded-v4-600px.png", - "bytes": 76498, - "sha256": "613e8f9c4bc1a6e398190414ae6bf0a70e784cd36e278beec10becbf6c5e71cc" - }, - { - "file": "decision-expanded-v4.pdf", - "bytes": 37680, - "sha256": "b56a46b9f26ebd18280b73f5bdb381cbc207312f477a82c002a662409e86e32f" - }, - { - "file": "decision-expanded-v4.png", - "bytes": 152677, - "sha256": "287486dbbdd9e8f89ae5ea358d20170ef2496ccf6ac79b3b0b6169990ff14da8" - }, - { - "file": "decision-expanded-v4.svg", - "bytes": 42942, - "sha256": "8abbe97a67a7a89291dea776f93990a42b0f2fafe47330bbafcfe6c14c4cc404" - }, - { - "file": "decision-expanded-v5-600px.png", - "bytes": 102382, - "sha256": "8fb899a4acca41fd9cc4cb6885e5e7f6b77a5c962af78f4ad00e99929fb8ce54" - }, - { - "file": "decision-expanded-v5.pdf", - "bytes": 40179, - "sha256": "b08d52344bd383766b19e967d4ab9ab0783a50ea423de93140dce26c7c91354a" - }, - { - "file": "decision-expanded-v5.png", - "bytes": 187722, - "sha256": "bf0d8f84d12b6f1e46ae02a70e0ed42adf45a818d8b4b594b1461ddcb635a13f" - }, - { - "file": "decision-expanded-v5.svg", - "bytes": 48613, - "sha256": "a5fba4d5176a8edb6668051678b5926cd0aaa40eabf192d9cefea03e6a64338f" - } - ], - "post_render_finalizer_sha256": "4818899220640baf74e6f87915dfe4ca5f77aedeca40e63c6f95ff9db1cf3d3e", - "preview_reencoding": { - "asset": "decision-expanded-v5-600px.png", - "reason": "Compressed pixel stream accidentally matched generic private-host text pattern; no textual host metadata", - "before_sha256": "afd80c6dc022fe4020885ec6eed87838854cda45d87ae99a0794d86be9ac8da9", - "after_sha256": "8fb899a4acca41fd9cc4cb6885e5e7f6b77a5c962af78f4ad00e99929fb8ce54", - "decoded_pixels_sha256": "dd1453b3d518a513f067473100b45aa76b91737035c026311d28f76c78fa4d13", - "mode": "RGBA", - "size": [ - 600, - 379 - ], - "pixels_exact": true, - "scanner_unchanged": true - } -} diff --git a/metrics/materials-provenance.json b/metrics/materials-provenance.json deleted file mode 100644 index 2c1d8afd339d2e883c38afe3a08b3fb7fae25085..0000000000000000000000000000000000000000 --- a/metrics/materials-provenance.json +++ /dev/null @@ -1,27 +0,0 @@ -{ - "docs_only": true, - "public_release_bindings": { - "Nox": { - "repo_id": "llm-semantic-router/Decision-1.0-Nox", - "revision": "ad089ad3a5dc9a7a21e6d96db546bb53e2212654", - "release_tag": "v1.3", - "bundle_manifest_sha256": "92d7f5be5e1ef21ee682f574b35cc01de4dd8ab16b8aa6edf0dfbdfcfcfabba3", - "publication_receipt_sha256": "fe15d9e49608c6d2d89301272a993e7d17cffe0e5b0966c32e9bb2740f3ad1e6", - "statistics_sha256": "0eb70b3dbc8a5ab8d78279beadd3ab9796623dfe64218dae06ca5edb2795f46d" - }, - "Sol": { - "repo_id": "llm-semantic-router/Decision-1.0-Sol", - "revision": "2412d9470d3aa125b346262dad80f6161847aa09", - "release_tag": "v1.3", - "bundle_manifest_sha256": "6f0970bfbe5594feecffc14b4824a92080ef09318e12567bd22264c3844e2fc4", - "publication_receipt_sha256": "7316ce19ffc5bbdd440316f7899dc33a8eecb9c36ebdd0cd9f9f8b49f48a446f", - "statistics_sha256": "fe2be9746c4302b829b5aa373c027910cfa84881b80c2068af7c69db76c4659c" - } - }, - "aggregate_source_sha256": "af8274ce0ba2c1dd1fe35a36dc77d347872443ce144e95b709a5e050680d3fda", - "source_evidence_sha256": "8ef8f73315fc185dca5c887728c44ed517a97623cbe25710c2dc83480204d3a8", - "original_weights_and_qualification_unchanged": true, - "displayed_latest_only": true, - "no_new_model_calls": true, - "no_APUS_no_V6_mean": true -} diff --git a/metrics/semantic-consistency.json b/metrics/semantic-consistency.json deleted file mode 100644 index 4dbcd51f021745de615bfcc525738c56bbf288ce..0000000000000000000000000000000000000000 --- a/metrics/semantic-consistency.json +++ /dev/null @@ -1,702 +0,0 @@ -{ - "metric": "semantic_prediction_consistency_group_fraction", - "display_metric": "Same-semantics group consistency", - "panel": "Observed original core (880 rows)", - "in_quality_average": false, - "unit": "semantic groups, not individual questions", - "rows": [ - { - "model_key": "Nox-retention-selected", - "display_name": "Nox v1.3", - "consistent_groups": 174, - "eligible_groups": 192, - "excluded_groups": 240, - "fraction": 0.90625, - "all_groups": 432, - "families": { - "ag_news": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 128, - "fraction": null, - "all_groups": 128 - }, - "boolean_constraints": { - "consistent_groups": 30, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 0.9375, - "all_groups": 32 - }, - "dbpedia_14": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 112, - "fraction": null, - "all_groups": 112 - }, - "natural_intents": { - "consistent_groups": 16, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 1.0, - "all_groups": 16 - }, - "option_carrier": { - "consistent_groups": 24, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 1.0, - "all_groups": 24 - }, - "ordinal_rubric": { - "consistent_groups": 31, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 0.96875, - "all_groups": 32 - }, - "relational_composition": { - "consistent_groups": 15, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.625, - "all_groups": 24 - }, - "scoped_evidence": { - "consistent_groups": 23, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.9583333333333334, - "all_groups": 24 - }, - "state_tracking": { - "consistent_groups": 19, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.7916666666666666, - "all_groups": 24 - }, - "unknown_rejection": { - "consistent_groups": 16, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 1.0, - "all_groups": 16 - } - } - }, - { - "model_key": "Sol-composition-selected", - "display_name": "Sol v1.3", - "consistent_groups": 167, - "eligible_groups": 192, - "excluded_groups": 240, - "fraction": 0.8697916666666666, - "all_groups": 432, - "families": { - "ag_news": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 128, - "fraction": null, - "all_groups": 128 - }, - "boolean_constraints": { - "consistent_groups": 30, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 0.9375, - "all_groups": 32 - }, - "dbpedia_14": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 112, - "fraction": null, - "all_groups": 112 - }, - "natural_intents": { - "consistent_groups": 16, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 1.0, - "all_groups": 16 - }, - "option_carrier": { - "consistent_groups": 23, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.9583333333333334, - "all_groups": 24 - }, - "ordinal_rubric": { - "consistent_groups": 29, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 0.90625, - "all_groups": 32 - }, - "relational_composition": { - "consistent_groups": 17, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.7083333333333334, - "all_groups": 24 - }, - "scoped_evidence": { - "consistent_groups": 21, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.875, - "all_groups": 24 - }, - "state_tracking": { - "consistent_groups": 15, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.625, - "all_groups": 24 - }, - "unknown_rejection": { - "consistent_groups": 16, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 1.0, - "all_groups": 16 - } - } - }, - { - "model_key": "kev-9b", - "display_name": "Kev 9B", - "consistent_groups": 143, - "eligible_groups": 192, - "excluded_groups": 240, - "fraction": 0.7447916666666666, - "all_groups": 432, - "families": { - "ag_news": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 128, - "fraction": null, - "all_groups": 128 - }, - "boolean_constraints": { - "consistent_groups": 29, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 0.90625, - "all_groups": 32 - }, - "dbpedia_14": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 112, - "fraction": null, - "all_groups": 112 - }, - "natural_intents": { - "consistent_groups": 16, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 1.0, - "all_groups": 16 - }, - "option_carrier": { - "consistent_groups": 0, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.0, - "all_groups": 24 - }, - "ordinal_rubric": { - "consistent_groups": 30, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 0.9375, - "all_groups": 32 - }, - "relational_composition": { - "consistent_groups": 21, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.875, - "all_groups": 24 - }, - "scoped_evidence": { - "consistent_groups": 20, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.8333333333333334, - "all_groups": 24 - }, - "state_tracking": { - "consistent_groups": 16, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.6666666666666666, - "all_groups": 24 - }, - "unknown_rejection": { - "consistent_groups": 11, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 0.6875, - "all_groups": 16 - } - } - }, - { - "model_key": "Decider · 2B", - "display_name": "Decider 2B", - "consistent_groups": 126, - "eligible_groups": 192, - "excluded_groups": 240, - "fraction": 0.65625, - "all_groups": 432, - "families": { - "ag_news": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 128, - "fraction": null, - "all_groups": 128 - }, - "boolean_constraints": { - "consistent_groups": 24, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 0.75, - "all_groups": 32 - }, - "dbpedia_14": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 112, - "fraction": null, - "all_groups": 112 - }, - "natural_intents": { - "consistent_groups": 16, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 1.0, - "all_groups": 16 - }, - "option_carrier": { - "consistent_groups": 0, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.0, - "all_groups": 24 - }, - "ordinal_rubric": { - "consistent_groups": 28, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 0.875, - "all_groups": 32 - }, - "relational_composition": { - "consistent_groups": 14, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.5833333333333334, - "all_groups": 24 - }, - "scoped_evidence": { - "consistent_groups": 22, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.9166666666666666, - "all_groups": 24 - }, - "state_tracking": { - "consistent_groups": 14, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.5833333333333334, - "all_groups": 24 - }, - "unknown_rejection": { - "consistent_groups": 8, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 0.5, - "all_groups": 16 - } - } - }, - { - "model_key": "Laya · EN/ML", - "display_name": "Laya · upstream default", - "consistent_groups": 115, - "eligible_groups": 192, - "excluded_groups": 240, - "fraction": 0.5989583333333334, - "all_groups": 432, - "families": { - "ag_news": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 128, - "fraction": null, - "all_groups": 128 - }, - "boolean_constraints": { - "consistent_groups": 28, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 0.875, - "all_groups": 32 - }, - "dbpedia_14": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 112, - "fraction": null, - "all_groups": 112 - }, - "natural_intents": { - "consistent_groups": 9, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 0.5625, - "all_groups": 16 - }, - "option_carrier": { - "consistent_groups": 22, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.9166666666666666, - "all_groups": 24 - }, - "ordinal_rubric": { - "consistent_groups": 25, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 0.78125, - "all_groups": 32 - }, - "relational_composition": { - "consistent_groups": 2, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.08333333333333333, - "all_groups": 24 - }, - "scoped_evidence": { - "consistent_groups": 9, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.375, - "all_groups": 24 - }, - "state_tracking": { - "consistent_groups": 12, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.5, - "all_groups": 24 - }, - "unknown_rejection": { - "consistent_groups": 8, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 0.5, - "all_groups": 16 - } - } - }, - { - "model_key": "Qwen3.5 · 2B, untuned", - "display_name": "Qwen3.5 2B · untuned", - "consistent_groups": 94, - "eligible_groups": 192, - "excluded_groups": 240, - "fraction": 0.4895833333333333, - "all_groups": 432, - "families": { - "ag_news": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 128, - "fraction": null, - "all_groups": 128 - }, - "boolean_constraints": { - "consistent_groups": 32, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 1.0, - "all_groups": 32 - }, - "dbpedia_14": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 112, - "fraction": null, - "all_groups": 112 - }, - "natural_intents": { - "consistent_groups": 14, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 0.875, - "all_groups": 16 - }, - "option_carrier": { - "consistent_groups": 0, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.0, - "all_groups": 24 - }, - "ordinal_rubric": { - "consistent_groups": 30, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 0.9375, - "all_groups": 32 - }, - "relational_composition": { - "consistent_groups": 3, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.125, - "all_groups": 24 - }, - "scoped_evidence": { - "consistent_groups": 6, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.25, - "all_groups": 24 - }, - "state_tracking": { - "consistent_groups": 0, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.0, - "all_groups": 24 - }, - "unknown_rejection": { - "consistent_groups": 9, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 0.5625, - "all_groups": 16 - } - } - }, - { - "model_key": "Qwen3.5 · 4B, untuned", - "display_name": "Qwen3.5 4B · untuned", - "consistent_groups": 98, - "eligible_groups": 192, - "excluded_groups": 240, - "fraction": 0.5104166666666666, - "all_groups": 432, - "families": { - "ag_news": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 128, - "fraction": null, - "all_groups": 128 - }, - "boolean_constraints": { - "consistent_groups": 22, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 0.6875, - "all_groups": 32 - }, - "dbpedia_14": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 112, - "fraction": null, - "all_groups": 112 - }, - "natural_intents": { - "consistent_groups": 16, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 1.0, - "all_groups": 16 - }, - "option_carrier": { - "consistent_groups": 10, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.4166666666666667, - "all_groups": 24 - }, - "ordinal_rubric": { - "consistent_groups": 30, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 0.9375, - "all_groups": 32 - }, - "relational_composition": { - "consistent_groups": 4, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.16666666666666666, - "all_groups": 24 - }, - "scoped_evidence": { - "consistent_groups": 6, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.25, - "all_groups": 24 - }, - "state_tracking": { - "consistent_groups": 1, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.041666666666666664, - "all_groups": 24 - }, - "unknown_rejection": { - "consistent_groups": 9, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 0.5625, - "all_groups": 16 - } - } - }, - { - "model_key": "Jev · 1.13.0", - "display_name": "Jev 1.13.0", - "consistent_groups": 158, - "eligible_groups": 192, - "excluded_groups": 240, - "fraction": 0.8229166666666666, - "all_groups": 432, - "families": { - "ag_news": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 128, - "fraction": null, - "all_groups": 128 - }, - "boolean_constraints": { - "consistent_groups": 32, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 1.0, - "all_groups": 32 - }, - "dbpedia_14": { - "consistent_groups": null, - "eligible_groups": 0, - "excluded_groups": 112, - "fraction": null, - "all_groups": 112 - }, - "natural_intents": { - "consistent_groups": 16, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 1.0, - "all_groups": 16 - }, - "option_carrier": { - "consistent_groups": 0, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.0, - "all_groups": 24 - }, - "ordinal_rubric": { - "consistent_groups": 32, - "eligible_groups": 32, - "excluded_groups": 0, - "fraction": 1.0, - "all_groups": 32 - }, - "relational_composition": { - "consistent_groups": 21, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.875, - "all_groups": 24 - }, - "scoped_evidence": { - "consistent_groups": 21, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.875, - "all_groups": 24 - }, - "state_tracking": { - "consistent_groups": 20, - "eligible_groups": 24, - "excluded_groups": 0, - "fraction": 0.8333333333333334, - "all_groups": 24 - }, - "unknown_rejection": { - "consistent_groups": 16, - "eligible_groups": 16, - "excluded_groups": 0, - "fraction": 1.0, - "all_groups": 16 - } - } - } - ], - "eligible_rule": "Group has more than one row; every row has unique identifiable option semantics and consistency_expected=true; all rows share one semantic gold. Success requires every output valid and one identical semantic prediction across the group.", - "coverage": "192 eligible groups across 8 families (640 rows); 240 singleton groups from AG News and DBpedia excluded. Groups are weighted equally, not families.", - "limits": [ - "Mixed variations of language, opaque keys, option order and evidence-carrier position/role; not an isolated order intervention.", - "Consistent wrong answers count as consistent. This is not accuracy, semantic correctness, or a new quality/release gate.", - "Bilingual Boolean/ordinal groups contain language variation but no Choice option permutation.", - "The same group count can hide different errors; retain task accuracy, probability calibration and input-support disclosures.", - "These are already-observed regression data. V6 no-op, combined key/order and genuine reversal are separate diagnostics." - ], - "count_proof": "Each numerator recovered from its stored aggregate ratio, verified by exact Python integer division equal to the stored float; family integer sums equal overall. No new scoring or inference.", - "sources": [ - { - "artifact": "aggregate-details.json", - "sha256": "428275514d77ca32ac5ae81826aa284136b70ff5eeb919b8dd66aa8d6aca5563" - }, - { - "artifact": "statistics.json", - "sha256": "af8274ce0ba2c1dd1fe35a36dc77d347872443ce144e95b709a5e050680d3fda" - }, - { - "artifact": "identities.json", - "sha256": "89bd6fde6da572da6f631d9b45216377fc5716009ff43b1ef2bd7fa394de029b" - }, - { - "artifact": "score.py", - "sha256": "36a4f1ea811fcb40f299ed311636322482b28047034b5e6e75b0a1c70864a9d7" - }, - { - "artifact": "build_suite.py", - "sha256": "12def3caa03277120b5fb27be7bb6afa4a86ac39adc4cffbe755a7cfb735297f" - }, - { - "artifact": "PUBLICATION-COMPLETE.json", - "sha256": "46a25490dbacdbe07d1587c617b2115c4fde5f7a15233d442fa7f3238864fea9" - }, - { - "artifact": "PUBLICATION-COMPLETE.json", - "sha256": "9e9b777ad714b2f31cc36b25be9aa6e457b8ec51cdc9795d0bafd25789b63b34" - }, - { - "artifact": "PUBLISHED-NOX-v1.3-ADDENDUM.json", - "sha256": "6682e9881cc6ab707f775323c2807f940a94838507b63212bc81f032c0fd9ba7" - } - ], - "source_sha256": "d78e0762e7bc0b12c774510acbaa7d135c4103abdd2305901fa3d7cbc6238e02" -} diff --git a/model-card-example.json b/model-card-example.json deleted file mode 100644 index e9550cc2f4db0cc77c5d1bac4c01d64b307584fa..0000000000000000000000000000000000000000 --- a/model-card-example.json +++ /dev/null @@ -1,113 +0,0 @@ -{ - "request": { - "state": "The customer reports that the same invoice was charged twice. They ask for a refund. There is no product outage.", - "questions": { - "destination": { - "type": "choice", - "instructions": "Choose the team that handles this request.", - "criteria": { - "billing": "Invoices, payments, refunds and duplicate charges", - "technical": "Product errors and troubleshooting" - } - }, - "refund_requested": { - "type": "noul", - "instructions": "Does the customer explicitly ask for a refund?" - }, - "urgency": { - "type": "score", - "instructions": "Rate urgency using only these ordered levels.", - "criteria": [ - "Routine information request with no payment problem or outage", - "A payment or billing problem, with no product outage", - "An active product outage stopping the customer from working" - ] - } - } - }, - "response": { - "model": "Decision-1.0-Sol", - "answers": { - "destination": { - "type": "choice", - "probabilities": { - "billing": 0.9994137709479778, - "technical": 0.0005862290520222329 - }, - "confidence": 0.9988275418959556, - "choice": "billing" - }, - "refund_requested": { - "type": "noul", - "noul": 0.9981037460809301 - }, - "urgency": { - "type": "score", - "probabilities": { - "0": 0.09484846481689971, - "1": 0.9038855158357278, - "2": 0.0012660193473724532 - }, - "confidence": 0.8558282737535918, - "score": 0.9064175545304728, - "legend": { - "0": "Routine information request with no payment problem or outage", - "1": "A payment or billing problem, with no product outage", - "2": "An active product outage stopping the customer from working" - } - } - }, - "usage": { - "input_tokens": 355, - "scored_questions": 3 - } - }, - "direct_engine_exact_response": true, - "overflow_rejected": true, - "overflow_message": "check: 40083 tokens exceeds max_length=16384; no truncation allowed", - "bundle_manifest_sha256": "1498cc7aa42f5884ab6ca828b5c23f23d4965b7e950a2ef86d530eed99ad78f1", - "runtime": { - "actual": { - "torch": "2.12.0+git6bbd260", - "hip": "7.2.53211", - "transformers": "5.17.0", - "fla": "0.5.2", - "tokenizers": "0.23.2", - "safetensors": "0.8.0", - "triton": "3.7.1", - "gated_delta": "fla.ops.gated_delta_rule.chunk" - }, - "differences": {}, - "matches_validated_runtime": true, - "normalization_profile": { - "profile_sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f", - "guard_sha256": "1603c39038ff783b9d5a5f69110d1695bbb7e28accffd0c03525854e8f258452", - "validated_arch": "gfx942", - "automatic_bundle_binding": true, - "scope": "single profile per process" - } - }, - "revision": null, - "device": "cuda:0", - "model_name": "Decision-1.0-Sol", - "example_source_sha256": "a54dec885f92c2d38d07ef2333dff51965be52bc119f772cc74200ed314e69b6", - "normalization_telemetry": { - "profile_sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f", - "status": "installed", - "calls": 72, - "keys": { - "[128,1,\"torch.bfloat16\",\"torch.bfloat16\",\"torch.float32\"]": 72 - }, - "strict_guard": true, - "unknown_keys": "raise", - "autotune_fallback_permitted": false, - "runtime": { - "torch": "2.12.0+git6bbd260", - "hip": "7.2.53211", - "triton": "3.7.1", - "fla": "0.5.2" - }, - "gpu_arch": "gfx942", - "process_scope": "one explicitly profiled Sol model; other model loading in this process is not supported" - } -} diff --git a/pyproject.toml b/pyproject.toml deleted file mode 100644 index 414ada1d87dcac10beb7ed3f47972dd7838f569d..0000000000000000000000000000000000000000 --- a/pyproject.toml +++ /dev/null @@ -1,19 +0,0 @@ -[build-system] -requires = ["setuptools>=68"] -build-backend = "setuptools.build_meta" - -[project] -name = "decision-local" -version = "1.1.0" -description = "Local typed inference for exported Decision decoder models" -requires-python = ">=3.10" -dependencies = [] - -[project.optional-dependencies] -hub = ["huggingface-hub==1.31.0"] - -[project.scripts] -decision-example = "decision.example:main" - -[tool.setuptools.packages.find] -where = ["src"] diff --git a/release-manifest.json b/release-manifest.json deleted file mode 100644 index f6b89c0c8019277bb1facad484af68e9f6cc862d..0000000000000000000000000000000000000000 --- a/release-manifest.json +++ /dev/null @@ -1,559 +0,0 @@ -{ - "format": "decision-public-release-v1", - "status": "current-documents-assembled", - "bundle_manifest_sha256": "1498cc7aa42f5884ab6ca828b5c23f23d4965b7e950a2ef86d530eed99ad78f1", - "readiness_sha256": "8627fe4a471c8588bf17c0913da08692a31af01757ebfb8c5be400f13e21ed7f", - "model_card_sha256": "fe6acfe6328b3e1b3b49fc31a77c009c1e42aa492933c53aa29888719c984278", - "repo_id": "llm-semantic-router/Decision-1.0-Sol-2B", - "assembly_script_sha256": "a8f72b8a67a64034d6183f89c945bab611056816ba0cba43670e9becb6eb78c4", - "original_bundle_manifest_preserved": false, - "files_exclude_this_manifest": true, - "files": [ - { - "file": ".gitattributes", - "bytes": 3132, - "sha256": "8ec513d4464879c383554c84c1acff60508f0748470daa16f475d05ac804e2ba" - }, - { - "file": "ATTRIBUTIONS.md", - "bytes": 6606, - "sha256": "e9ed3c41423a6d15edbf82e14b80897fe5ef6cd454bcd9957d91b31460461718" - }, - { - "file": "DIAGNOSTICS.md", - "bytes": 6327, - "sha256": "cc8c0a378013d1cfb87ec775344180047136109cbc236a23da02201c0a7c5baa" - }, - { - "file": "Dockerfile.runtime", - "bytes": 751, - "sha256": "36e81318bed2b6005ff46d6aa949533924b17d88ab8013a56caab77b1a0b5cf7" - }, - { - "file": "EVALUATION.md", - "bytes": 4079, - "sha256": "10744aeed1a113597e644d0b3f4787a91299f055701f22c8cd34843276a7df2a" - }, - { - "file": "LICENSE", - "bytes": 11544, - "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" - }, - { - "file": "MATERIALS.json", - "bytes": 2400, - "sha256": "48e950fed869536eb9befe67378662aa8ecaf95576c2ef0d0b981f318250a999" - }, - { - "file": "NORMALIZATION_RUNTIME.md", - "bytes": 754, - "sha256": "b3fbb8b37961676c15b37790f73cf28c27fb3d2eb844b3d6e467b55637b441bf" - }, - { - "file": "QUESTION-SCALING.md", - "bytes": 1395, - "sha256": "a2aaad17e779958942a02bf5692c871fccc8b2195ea1b68c86e81f7cd45c1355" - }, - { - "file": "QWEN-LICENSE", - "bytes": 11544, - "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" - }, - { - "file": "README.md", - "bytes": 5942, - "sha256": "fe6acfe6328b3e1b3b49fc31a77c009c1e42aa492933c53aa29888719c984278" - }, - { - "file": "RUNTIME-RELEASE.json", - "bytes": 1264, - "sha256": "8627fe4a471c8588bf17c0913da08692a31af01757ebfb8c5be400f13e21ed7f" - }, - { - "file": "RUNTIME.md", - "bytes": 5205, - "sha256": "be35b2f8a9c977f38be9b5eb892fdc809db5167c327575c9311355aae3e781c7" - }, - { - "file": "RUNTIME_BINDING.json", - "bytes": 2069, - "sha256": "b3d582d738223adea3cd33b8be4079d3ea4e57a2f5322128e374793f9a61f820" - }, - { - "file": "SENSITIVITY.md", - "bytes": 866, - "sha256": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5" - }, - { - "file": "SERVING_OPTIMIZATION.json", - "bytes": 1548, - "sha256": "09ba5f0c60a57c5fed1592dcd559c3e33e4e4bd948ace3a2e036f2c4e15f383e" - }, - { - "file": "SOURCE_BUNDLE_MANIFEST.json", - "bytes": 4194, - "sha256": "782fc4a3dc2a003d33f1b388e2db0193ca9253aa48b5c5b99e53e1253274413e" - }, - { - "file": "TASKS.md", - "bytes": 10393, - "sha256": "573086057e6d87135f119ad95fda16a46f9e8a5e1109430f0a1b8a5da7d99247" - }, - { - "file": "USAGE.md", - "bytes": 2886, - "sha256": "398c56cbb43047629551827fc5c0ba24f4dac2d7b01a5e6e3aeeb85b951706a0" - }, - { - "file": "WEIGHTING.md", - "bytes": 866, - "sha256": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5" - }, - { - "file": "assets/architecture.png", - "bytes": 511058, - "sha256": "8fcfa10b3941ee4aec1f5fba20a25e382a7d5de83074071c9ef68e627996de40" - }, - { - "file": "assets/architecture.svg", - "bytes": 14718, - "sha256": "215d9af6f94cc24847fc1d23a0b2a287b36afdc1e10ea4f5c82895b2fe103002" - }, - { - "file": "assets/decision-expanded-old_core-600px.png", - "bytes": 167493, - "sha256": "6d36e595bea7465903ee787d217ef07ad648472d336655e88e51263325d0a7e2" - }, - { - "file": "assets/decision-expanded-old_core.pdf", - "bytes": 44384, - "sha256": "cbd80145027dd79b37ed799a64b4c173d44973b358bc05414da80355ba7072a2" - }, - { - "file": "assets/decision-expanded-old_core.png", - "bytes": 312133, - "sha256": "2744aafba8afb133b1c92d7b1813d3e769da8babe9439e3d966f94ac28963bbd" - }, - { - "file": "assets/decision-expanded-old_core.svg", - "bytes": 79821, - "sha256": "0648a4f0ed3720dcc9dc0f19dd43c2f46e08b2dce1cf7a057dba957fa475f288" - }, - { - "file": "assets/decision-expanded-overview-600px.png", - "bytes": 94850, - "sha256": "b8e8361345b814083eef97dc08b9a4fd173bcef83faa72c5f66eac5d0cc1c10b" - }, - { - "file": "assets/decision-expanded-overview.pdf", - "bytes": 41563, - "sha256": "4f008072cb5b8268a31e058d0839d68e2416c7f63795ff40cd931906cf74552e" - }, - { - "file": "assets/decision-expanded-overview.png", - "bytes": 187355, - "sha256": "2f12699c29ecde8146482e63925b9e80e3bab4047f79fbec17a87e50a1e64ec1" - }, - { - "file": "assets/decision-expanded-overview.svg", - "bytes": 48619, - "sha256": "6565620295efd3180f1c2079ea7fec332f17b977fcc9781101e9a82001eff743" - }, - { - "file": "assets/decision-expanded-ranking-600px.png", - "bytes": 83184, - "sha256": "ddbb71c528b3741f1258367e5436b35148461c628aaf4d03d209ae6c4eccdee1" - }, - { - "file": "assets/decision-expanded-ranking.pdf", - "bytes": 37133, - "sha256": "bac1d9bf032cd06d4d81170fd10e033e56c1a793e958e9599f8bc570700bf400" - }, - { - "file": "assets/decision-expanded-ranking.png", - "bytes": 186098, - "sha256": "a78d29ea6a73eab02cc8897140614d92b0d7bb4e1295c43c9a2ece60d03ec155" - }, - { - "file": "assets/decision-expanded-ranking.svg", - "bytes": 38118, - "sha256": "8de46cff93bdf4b899e74b6b7baddafae6956b6e32b9713702472c8aace0a164" - }, - { - "file": "assets/decision-expanded-v3_core-600px.png", - "bytes": 168267, - "sha256": "df7099ba80f0a000f7c64cfe6b8aca7ead06d5311e6967a49ae8de32317a2077" - }, - { - "file": "assets/decision-expanded-v3_core.pdf", - "bytes": 44464, - "sha256": "2ffe3a8fdad5dc957f524dbb03d437aaa50a098920358322575f84d65e95e09f" - }, - { - "file": "assets/decision-expanded-v3_core.png", - "bytes": 300452, - "sha256": "a621a7ffee11dc4909e89a200b7c1207a75bff57a9a237b66e8ed8f2fbac6983" - }, - { - "file": "assets/decision-expanded-v3_core.svg", - "bytes": 79901, - "sha256": "75c74babb4214337caff4b3a5ec36bb1de5824f486312bde33ab3264d2c50f48" - }, - { - "file": "assets/decision-expanded-v4-600px.png", - "bytes": 76498, - "sha256": "613e8f9c4bc1a6e398190414ae6bf0a70e784cd36e278beec10becbf6c5e71cc" - }, - { - "file": "assets/decision-expanded-v4.pdf", - "bytes": 37680, - "sha256": "b56a46b9f26ebd18280b73f5bdb381cbc207312f477a82c002a662409e86e32f" - }, - { - "file": "assets/decision-expanded-v4.png", - "bytes": 152677, - "sha256": "287486dbbdd9e8f89ae5ea358d20170ef2496ccf6ac79b3b0b6169990ff14da8" - }, - { - "file": "assets/decision-expanded-v4.svg", - "bytes": 42942, - "sha256": "8abbe97a67a7a89291dea776f93990a42b0f2fafe47330bbafcfe6c14c4cc404" - }, - { - "file": "assets/decision-expanded-v5-600px.png", - "bytes": 102382, - "sha256": "8fb899a4acca41fd9cc4cb6885e5e7f6b77a5c962af78f4ad00e99929fb8ce54" - }, - { - "file": "assets/decision-expanded-v5.pdf", - "bytes": 40179, - "sha256": "b08d52344bd383766b19e967d4ab9ab0783a50ea423de93140dce26c7c91354a" - }, - { - "file": "assets/decision-expanded-v5.png", - "bytes": 187722, - "sha256": "bf0d8f84d12b6f1e46ae02a70e0ed42adf45a818d8b4b594b1461ddcb635a13f" - }, - { - "file": "assets/decision-expanded-v5.svg", - "bytes": 48613, - "sha256": "a5fba4d5176a8edb6668051678b5926cd0aaa40eabf192d9cefea03e6a64338f" - }, - { - "file": "assets/decision-family-header.png", - "bytes": 2962868, - "sha256": "213511289ce8df038d938ac470e803c427ed57f0f85cc397dd4d79964b866541" - }, - { - "file": "assets/decision-matrix.pdf", - "bytes": 29410, - "sha256": "3b4f10c3f00820d4b82b11a48401a8c360304c4e7cda90a6e7133c76d18ca41b" - }, - { - "file": "assets/decision-matrix.png", - "bytes": 375860, - "sha256": "cb0c583975fc6848f0d3ca7ed11da48347f7431f91f4debf699b3fb039a0a696" - }, - { - "file": "assets/decision-matrix.svg", - "bytes": 50107, - "sha256": "fc83d1c36264098fce20ca9f000aedc04473c7a16295fa5d16a82afeee2342fc" - }, - { - "file": "assets/decision-question-scaling-600px.png", - "bytes": 40784, - "sha256": "3de11b7e5d1b49ca874b50ef6ae8ced1dcdc3a83dae4889170a0134f0ca3f8cc" - }, - { - "file": "assets/decision-question-scaling.pdf", - "bytes": 19269, - "sha256": "d4eeee05ee71af4cd783d6729da96b863df2487501ac68fb33248f0d67ee0f00" - }, - { - "file": "assets/decision-question-scaling.png", - "bytes": 105784, - "sha256": "7fe441983a5af90f8832ead12c5f41c30c6181e0d0e537bdc65a3a0c15799b24" - }, - { - "file": "assets/decision-question-scaling.svg", - "bytes": 9275, - "sha256": "e0f5a6c4546c204081fd1ed0a4208c2739b76075b0379b0110d2511c71335930" - }, - { - "file": "assets/decision-ranking.pdf", - "bytes": 25376, - "sha256": "e2d6ecdab0599e65db77b520669e90b32fad6d10c85bb15dcd77bec1ed6ec51f" - }, - { - "file": "assets/decision-ranking.png", - "bytes": 248285, - "sha256": "9bb5e1487f2b66215624e67a72bda0d1c3f4ad6cd931089aa5bd7211f3be02b9" - }, - { - "file": "assets/decision-ranking.svg", - "bytes": 15728, - "sha256": "c8600115c3fafbf03e0b06cc6e3da016e4f74473df4f2d4575ae1c9721c99b8e" - }, - { - "file": "assets/decision-sol-2b-header.png", - "bytes": 2398842, - "sha256": "9616b7828774561692b642236d74b72de548e2dfc2ae83f1cae2b70bb3b69b08" - }, - { - "file": "assets/readout.png", - "bytes": 285481, - "sha256": "fb6b3d1cf149a1a3ccb584eb1c213df73c07bcd328732b4238928e2a2a4af085" - }, - { - "file": "backbone/config.json", - "bytes": 1789, - "sha256": "e7bed6f1a1a4d5f029a28040e8687a122b323798d4674666443f6d78e9475a05" - }, - { - "file": "backbone/model.safetensors", - "bytes": 3763685328, - "sha256": "99f97c2564acc1d43b89a1c547b63ce1d87f062637d0a8ed41eb40d752849e36" - }, - { - "file": "bundle-manifest.json", - "bytes": 6549, - "sha256": "1498cc7aa42f5884ab6ca828b5c23f23d4965b7e950a2ef86d530eed99ad78f1" - }, - { - "file": "chat_template.jinja", - "bytes": 7755, - "sha256": "273d8e0e683b885071fb17e08d71e5f2a5ddfb5309756181681de4f5a1822d80" - }, - { - "file": "code/decision_api.py", - "bytes": 10952, - "sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152" - }, - { - "file": "code/decision_model.py", - "bytes": 10114, - "sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646" - }, - { - "file": "code/predict.py", - "bytes": 3164, - "sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee" - }, - { - "file": "code/profile_guard.py", - "bytes": 6131, - "sha256": "1603c39038ff783b9d5a5f69110d1695bbb7e28accffd0c03525854e8f258452" - }, - { - "file": "code/runtime_profile.py", - "bytes": 2857, - "sha256": "afb59dda5e4c3890ee5c971549799e69f2e4fa2eafe5893de2168299fd368bda" - }, - { - "file": "decision_config.json", - "bytes": 753, - "sha256": "9fa5b80f0df63964410b8498b5473d8cba4aa82e506aae83619d5423c0c0e67a" - }, - { - "file": "decision_head.safetensors", - "bytes": 8424272, - "sha256": "5c3dce45115fba20192c99a32fa89a379c2a6617f02e3647dc8f05e48dcaf215" - }, - { - "file": "metrics/benchmark.json", - "bytes": 2569391, - "sha256": "86dc156356ee80eb849808f6e0d5237157f63d121ae6992c9a833691912ef76c" - }, - { - "file": "metrics/comparator-coverage.json", - "bytes": 21708, - "sha256": "7ec368a3aefedff14af32d94a320c5f57d9c0bea2779272e2d171a4afc1d0fa9" - }, - { - "file": "metrics/evaluation-provenance.json", - "bytes": 8510, - "sha256": "f9c087bd92e6cdfdc4c4dabc0ec335bb837a79715577a7355012d3a21f81c8e9" - }, - { - "file": "metrics/expanded-quality.json", - "bytes": 64295, - "sha256": "7b8936b43bce9d19d9010ab61225a6460160b3aa76c37f5d6df5ba650b433c6c" - }, - { - "file": "metrics/figure-evidence.json", - "bytes": 9339, - "sha256": "1fde66bcafc9e602b030e7492a01d72fca510cc5fbd690f4fdef34696db38f23" - }, - { - "file": "metrics/materials-provenance.json", - "bytes": 1333, - "sha256": "3f052b894ab0780825ef698556483a5900d1e024cfcfaa1e89919261d812adb4" - }, - { - "file": "metrics/question-scaling.json", - "bytes": 3509, - "sha256": "6b76d2fd372171ed3d08d2bdc960e209bfcb6a20fc3552be272668198917ffe1" - }, - { - "file": "metrics/semantic-consistency.json", - "bytes": 20784, - "sha256": "5a389236a22e1068be862b2daf88a5caf5a384a9260f0cafbaca9a37b18d8d26" - }, - { - "file": "model-card-example.json", - "bytes": 3773, - "sha256": "cd65908ac162719d351887a204b662425e0dd232ce229b86b12e95ac7d14e465" - }, - { - "file": "pyproject.toml", - "bytes": 436, - "sha256": "135a9516e87ca2fb41ef54a974e29132256b6c2d528a1cbdea1001e28f306946" - }, - { - "file": "runtime-build-provenance.json", - "bytes": 2894, - "sha256": "61d064b0db35d0ebd04a038e376896d2aab8b69e6525a8ec1f317f73b5284345" - }, - { - "file": "runtime-fla-requirements.lock", - "bytes": 654, - "sha256": "35e1fedff9ca49092a7ada277bbd794abe7e8474e14a678a7e4bcec6406d0eb5" - }, - { - "file": "runtime-profile/l2norm_fwd_kernel.json", - "bytes": 13210, - "sha256": "a67f3b4624edc07e2c3f9f1d05553c654a3ff005c6b96a1831d62e323dea9edc" - }, - { - "file": "runtime-profile/profile.json", - "bytes": 18101, - "sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f" - }, - { - "file": "runtime-provenance.json", - "bytes": 5805, - "sha256": "307e865c47e9f6120b50d8be4f7c5071a116e6f8a50d40c5424ab0232b6fe690" - }, - { - "file": "runtime.json", - "bytes": 1112, - "sha256": "8cd739efce9cee371cda1d3b769413a3c0842968fe2b4d511cebab11f071af85" - }, - { - "file": "src/decision/__init__.py", - "bytes": 167, - "sha256": "70de37df98b6fc8e3b9f9d43935ba31a32350496214c11adf8c5fba72c433313" - }, - { - "file": "src/decision/example.py", - "bytes": 3581, - "sha256": "a54dec885f92c2d38d07ef2333dff51965be52bc119f772cc74200ed314e69b6" - }, - { - "file": "src/decision/model.py", - "bytes": 9415, - "sha256": "ba240d7493fc29203fe036966f0ab911200cd4a9252b50b05423409977639ee0" - }, - { - "file": "temperature.json", - "bytes": 6056, - "sha256": "f0cbe7323441ceaf6abc8dd3f9b1832f5d4121a803f18e9643406aadd53c6efc" - }, - { - "file": "tokenizer.json", - "bytes": 19989325, - "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523" - }, - { - "file": "tokenizer_config.json", - "bytes": 1123, - "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87" - } - ], - "staging_methods": { - "ATTRIBUTIONS.md": "copy", - "Dockerfile.runtime": "copy", - "EVALUATION.md": "copy", - "LICENSE": "copy", - "NORMALIZATION_RUNTIME.md": "copy", - "QUESTION-SCALING.md": "copy", - "QWEN-LICENSE": "copy", - "README.md": "copy", - "RUNTIME.md": "copy", - "RUNTIME_BINDING.json": "copy", - "SOURCE_BUNDLE_MANIFEST.json": "copy", - "USAGE.md": "copy", - "assets/architecture.png": "copy", - "assets/architecture.svg": "copy", - "assets/decision-expanded-old_core-600px.png": "copy", - "assets/decision-expanded-old_core.pdf": "copy", - "assets/decision-expanded-old_core.png": "copy", - "assets/decision-expanded-old_core.svg": "copy", - "assets/decision-expanded-overview-600px.png": "copy", - "assets/decision-expanded-overview.pdf": "copy", - "assets/decision-expanded-overview.png": "copy", - "assets/decision-expanded-overview.svg": "copy", - "assets/decision-expanded-ranking-600px.png": "copy", - "assets/decision-expanded-ranking.pdf": "copy", - "assets/decision-expanded-ranking.png": "copy", - "assets/decision-expanded-ranking.svg": "copy", - "assets/decision-expanded-v3_core-600px.png": "copy", - "assets/decision-expanded-v3_core.pdf": "copy", - "assets/decision-expanded-v3_core.png": "copy", - "assets/decision-expanded-v3_core.svg": "copy", - "assets/decision-expanded-v4-600px.png": "copy", - "assets/decision-expanded-v4.pdf": "copy", - "assets/decision-expanded-v4.png": "copy", - "assets/decision-expanded-v4.svg": "copy", - "assets/decision-expanded-v5-600px.png": "copy", - "assets/decision-expanded-v5.pdf": "copy", - "assets/decision-expanded-v5.png": "copy", - "assets/decision-expanded-v5.svg": "copy", - "assets/decision-family-header.png": "copy", - "assets/decision-question-scaling-600px.png": "copy", - "assets/decision-question-scaling.pdf": "copy", - "assets/decision-question-scaling.png": "copy", - "assets/decision-question-scaling.svg": "copy", - "assets/readout.png": "copy", - "backbone/config.json": "copy", - "backbone/model.safetensors": "hardlink", - "bundle-manifest.json": "copy", - "chat_template.jinja": "copy", - "code/decision_api.py": "copy", - "code/decision_model.py": "copy", - "code/predict.py": "copy", - "code/profile_guard.py": "copy", - "code/runtime_profile.py": "copy", - "decision_config.json": "copy", - "decision_head.safetensors": "hardlink", - "metrics/comparator-coverage.json": "copy", - "metrics/expanded-quality.json": "copy", - "metrics/figure-evidence.json": "copy", - "metrics/materials-provenance.json": "copy", - "metrics/question-scaling.json": "copy", - "model-card-example.json": "copy", - "pyproject.toml": "copy", - "runtime-build-provenance.json": "copy", - "runtime-fla-requirements.lock": "copy", - "runtime-profile/l2norm_fwd_kernel.json": "copy", - "runtime-profile/profile.json": "copy", - "runtime-provenance.json": "copy", - "runtime.json": "copy", - "src/decision/__init__.py": "copy", - "src/decision/example.py": "copy", - "src/decision/model.py": "copy", - "temperature.json": "copy", - "tokenizer.json": "copy", - "tokenizer_config.json": "copy", - "MATERIALS.json": "copy", - "metrics/semantic-consistency.json": "copy" - }, - "scope": "Latest qualified Lux comparison and54 task diagnostics; all other model rows and inference files unchanged.", - "release_tag": "v1.3.1", - "change_kind": "latest-comparison-documents-only", - "previous_main_revision": "bf0248376030f4ef2aca9130f0aa2f64fc255958", - "previous_release_manifest_sha256": "5485a589879013f2cdde5addaf768b00c4b3f1a6046e0fcbdb09a9dc2f4943e2", - "presentation_amendment_sha256": "4fadc67b276d2b1e79e19ccacd383482cb0580996450d0103ffe2a90cb494a56", - "statistics_sha256": "8406aea215dc1c4ee2645130c94472b336bb065ba46c6efac4f742589499463b", - "latency_correction_sha256": "9727f7a0768d3888ed8cd22f95b25598e558b447adc8f9f4f5a0bb24959269b1" -} diff --git a/runtime-build-provenance.json b/runtime-build-provenance.json deleted file mode 100644 index 11431cf3821a0f44e76123d06cedfa9945d16f8c..0000000000000000000000000000000000000000 --- a/runtime-build-provenance.json +++ /dev/null @@ -1,65 +0,0 @@ -{ - "image_id": "sha256:ef8192e3dbaadf97a03f1d0bf87ee63ae16d9ffdc5ee209b199ed17e773b438e", - "created": "2026-09-21T12:56:59.399496065Z", - "layer_count": 42, - "cpu_import_probe": { - "exit_code": 0, - "result": { - "torch": "2.12.0+git6bbd260", - "torch_git": "6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5", - "hip": "7.2.53211", - "cuda_initialized": false, - "transformers": "5.17.0", - "wrapper_imported": true, - "packages": { - "triton": "3.7.1+gitf0b55c07", - "fla-core": "0.5.2", - "flash-linear-attention": "0.5.2", - "decision-local": "1.0.0" - } - }, - "stderr_last_line": [] - }, - "schema_version": 1, - "public_base": "vllm/vllm-openai-rocm@sha256:1fd21abe66455b4df5a2e83629e97cdcc9d58913b16052d8118b92b239792339", - "build_exit_code": 0, - "gpu_validation": { - "status": "passed", - "scope": "Real packaged three-question example and explicit complete-input overflow rejection for each bundle. Exact response equality against qualified runtime on this request; not a replacement for full benchmark rerun.", - "models": { - "Decision-1.0-Sol": { - "bundle_manifest_sha256": "076558961011e924bf17084f940401150245bae6e135d4b1b0a6ee16be01f13c", - "example_source_sha256": "d52c20b602bd1049aecda55244e71c27d8ce44bd1aa8b1941db62ae41d31ba12", - "gpu": "AMD ROCm gfx942", - "request_count": 1, - "typed_questions": 3, - "response_bit_exact_against_qualified_runtime": true, - "matches_validated_runtime": true, - "wrapper_equals_direct_engine": true, - "overflow_rejected": true, - "board_product_name_verified": false - }, - "Decision-1.0-Nox": { - "bundle_manifest_sha256": "adf5d88146d4f4a7fdcf0b265aa0c804a6116537c1e92bd620e40dfdf6803550", - "example_source_sha256": "d52c20b602bd1049aecda55244e71c27d8ce44bd1aa8b1941db62ae41d31ba12", - "gpu": "AMD ROCm gfx942", - "request_count": 1, - "typed_questions": 3, - "response_bit_exact_against_qualified_runtime": true, - "matches_validated_runtime": true, - "wrapper_equals_direct_engine": true, - "overflow_rejected": true, - "board_product_name_verified": false - } - } - }, - "built_utc": "2026-09-21T12:57:54.147566+00:00", - "files": { - "Dockerfile.runtime": "36e81318bed2b6005ff46d6aa949533924b17d88ab8013a56caab77b1a0b5cf7", - "runtime-fla-requirements.lock": "35e1fedff9ca49092a7ada277bbd794abe7e8474e14a678a7e4bcec6406d0eb5", - "pyproject.toml": "6a557fbe472027af103ca2cfc07e977caab3e257de930aa57cf9fea8cf90f8a0", - "src/decision/__init__.py": "70de37df98b6fc8e3b9f9d43935ba31a32350496214c11adf8c5fba72c433313", - "src/decision/example.py": "d52c20b602bd1049aecda55244e71c27d8ce44bd1aa8b1941db62ae41d31ba12", - "src/decision/model.py": "ab428234d508a1245d3723c5a5bb0c796685abebd0447f77c7fbed9113954b21" - } -} diff --git a/runtime-fla-requirements.lock b/runtime-fla-requirements.lock deleted file mode 100644 index 874a8ddd500c525a122cf00fd5c42981541ae7c1..0000000000000000000000000000000000000000 --- a/runtime-fla-requirements.lock +++ /dev/null @@ -1,4 +0,0 @@ -# Only the FLA overlay; all other dependencies are fixed by the public base digest. -# Install with --no-deps --require-hashes to preserve the ROCm Torch/Triton builds. -fla-core @ https://files.pythonhosted.org/packages/2d/ed/dfe19c4da779957eb6a42a26812f9b4e2280bf757a17a71933ff59ffcb98/fla_core-0.5.2-py3-none-any.whl --hash=sha256:5e830c85bad3d0d34677f98ac7074d08687a3756f0f0499d95ceb96eb6920761 -flash-linear-attention @ https://files.pythonhosted.org/packages/90/d2/2070e3cf2148c5cce99ca4876633c4b7b89ec085323600bd3d47aeacd306/flash_linear_attention-0.5.2-py3-none-any.whl --hash=sha256:dcf405d81f5426393b59037097aa700d0f4a841465d5028d5aa543f4502f2400 diff --git a/runtime-profile/l2norm_fwd_kernel.json b/runtime-profile/l2norm_fwd_kernel.json deleted file mode 100644 index 975e1af621e81c2ef3a8f9511a61c376336eb4ec..0000000000000000000000000000000000000000 --- a/runtime-profile/l2norm_fwd_kernel.json +++ /dev/null @@ -1,646 +0,0 @@ -{ - "kernel_name": "l2norm_fwd_kernel", - "triton_version": "3.7.1", - "autotune_entries": { - "878a9638751a7375a0fb15d3a7767659": { - "autotune_key": [ - 128, - 1, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "eca743e0790be43b6867d648e9430f67": { - "autotune_key": [ - 128, - 2, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 16 - }, - "num_warps": 1, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "7e2d02ce45e4f072828b4c73e16fe0af": { - "autotune_key": [ - 128, - 3, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 16 - }, - "num_warps": 2, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "424d41ef00a3161badd7f960c34e6ece": { - "autotune_key": [ - 128, - 4, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "6bc2b57d9035425140554c225e667aad": { - "autotune_key": [ - 128, - 5, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "91b9289c43a2d1b21042009d44c57b8a": { - "autotune_key": [ - 128, - 6, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "f44702dfc86a96df1de10439f5c5575a": { - "autotune_key": [ - 128, - 7, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "87c993ac4858bff6042f319bec1647f7": { - "autotune_key": [ - 128, - 8, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "b35c6bfa2cd85845ba58306ee63e28b1": { - "autotune_key": [ - 128, - 9, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 8 - }, - "num_warps": 4, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "3527d130e8f6f1415c6d24df9821e638": { - "autotune_key": [ - 128, - 10, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 8 - }, - "num_warps": 2, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "fe2e43a7c21c6a4814d2636f3004d7dc": { - "autotune_key": [ - 128, - 11, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 8 - }, - "num_warps": 4, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "c39c159634377c459eccde8a272f74fe": { - "autotune_key": [ - 128, - 12, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "a10453829fb545fd2e01acfb4a809d17": { - "autotune_key": [ - 128, - 13, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 8 - }, - "num_warps": 4, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "bd8734e29b9704051fe467ede8e8cc03": { - "autotune_key": [ - 128, - 14, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "d5ffd2f70a70e13ac03901bd22a150a7": { - "autotune_key": [ - 128, - 15, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "d522f744092358110c9ad2cc5f54b501": { - "autotune_key": [ - 128, - 16, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "6003bee20af205e16410f1983eadd67c": { - "autotune_key": [ - 128, - 17, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "89d2cb2e8a818ce3ecb80bd340ba42b6": { - "autotune_key": [ - 128, - 18, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "157376310735556f015b8ac2dc6e9e2e": { - "autotune_key": [ - 128, - 19, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "e19965e082b861f637bf4bab6abe025b": { - "autotune_key": [ - 128, - 20, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "49feda728414835e54048468abd808ab": { - "autotune_key": [ - 128, - 21, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "6b419788330d3468ef22dc741ec9d7f7": { - "autotune_key": [ - 128, - 22, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "8827081d0dda52c9376b3a1d97a82a8e": { - "autotune_key": [ - 128, - 23, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "49e16fa1a715a8913c54c4ff9ce61aa9": { - "autotune_key": [ - 128, - 24, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "7bf76d12f90b95bbddff57fbc8fe3c39": { - "autotune_key": [ - 128, - 25, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "3f2cd7a3a429540e08f0e4dd8ac9afc5": { - "autotune_key": [ - 128, - 26, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "837afa436bfab6256a535e9360403f34": { - "autotune_key": [ - 128, - 27, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "d4e411f29afbb396dcfe1455427e5178": { - "autotune_key": [ - 128, - 28, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "1b96f1802f4f63f5c872de3d1972ad8a": { - "autotune_key": [ - 128, - 29, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "aec76b66508d5e884ffcc743e8f55a92": { - "autotune_key": [ - 128, - 30, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "439d493be8fe1159bd65368245a4aa60": { - "autotune_key": [ - 128, - 31, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - "801422dba59f6c5e144328aabdd135a3": { - "autotune_key": [ - 128, - 32, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - } - } -} diff --git a/runtime-profile/profile.json b/runtime-profile/profile.json deleted file mode 100644 index 14b33c53e914c63169e30ab44a7f1a331ee170a1..0000000000000000000000000000000000000000 --- a/runtime-profile/profile.json +++ /dev/null @@ -1,822 +0,0 @@ -{ - "format": "decision-fla-l2norm-profile-v1", - "status": "diagnostic-only-awaiting-clean-room-proof", - "target": "Decision Sol Qwen3.5-2B pointer Stage4; one profile per process", - "model_changes": false, - "numerical_tolerance_unchanged": 1e-06, - "cache_mode": "strict", - "supported": { - "head_dimension": 128, - "key_heads": 16, - "body_dtype": "bfloat16", - "output_dtype": "bfloat16", - "rstd_dtype": "float32", - "batch_size_min": 1, - "batch_size_max": 8, - "padded_tokens_max": 16384, - "NB_min": 1, - "NB_max": 32, - "NB_formula": "ceil(B*padded_tokens*16/65536)", - "unknown_key": "raise before kernel/autotune" - }, - "runtime": { - "torch": "2.12.0+git6bbd260", - "hip": "7.2.53211", - "triton": "3.7.1", - "fla": "0.5.2", - "gpu_arch": "gfx942" - }, - "files": [ - { - "file": "l2norm_fwd_kernel.json", - "sha256": "a67f3b4624edc07e2c3f9f1d05553c654a3ff005c6b96a1831d62e323dea9edc" - } - ], - "fla_source_sha256": { - "modules/l2norm.py": "30f7feebdbfa87b8e90c143ec62a7fa0781bfd736c856c0ec69452f0c778be3f", - "ops/utils/cache.py": "a64b09ffd3f51aed547ad72559dee94cb9aeafd2a5f02c6c8e60a6e020147a9c" - }, - "source_evidence": [ - { - "key": [ - 128, - 1, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "030be4b902759f8a02c66edfe9724c8a04d320c32eb8298ef42da42f8ff02a54", - "chosen_config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "key": [ - 128, - 7, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "b868fb0bcb06cfdc5b4432392e35ade0daeeb40374645dbfd65476996362dc48", - "chosen_config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "key": [ - 128, - 8, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "ee3b7b3a9facff1f50feacbc32427f005e6d344b1e452e9daabc68b19a9fed0d", - "chosen_config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "key": [ - 128, - 9, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "fd0b189927eb2475fd73e9a71899a6655ae4d73a010c618b00316f9eb15e429e", - "chosen_config": { - "kwargs": { - "BT": 8 - }, - "num_warps": 4, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "key": [ - 128, - 5, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "a8ad1053d222fb9d5751a6ebe11fc58679a9e44366be3dc9a2c214042485778c", - "chosen_config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "key": [ - 128, - 2, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "00499403ea55da8fc3e1110dfb41a94204f19813ac8302dc2cd0894b2e1f51a9", - "chosen_config": { - "kwargs": { - "BT": 16 - }, - "num_warps": 1, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "key": [ - 128, - 12, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "bad146284201145ccafd77500eba52f4b20483596065a3da91e6b1bccb0548e3", - "chosen_config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "key": [ - 128, - 10, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "cb8bd713aee627cb50fcca3ed750e7ac9976ee56f9d6d3d44ef469325049822c", - "chosen_config": { - "kwargs": { - "BT": 8 - }, - "num_warps": 2, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "key": [ - 128, - 13, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "f2c9fe94fbf5ff3f48b4ec875cbfb8016148c772fe085b69d6e42dd8053f9549", - "chosen_config": { - "kwargs": { - "BT": 8 - }, - "num_warps": 4, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "key": [ - 128, - 11, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "0d2cc4c185b6c73918b1397271e98276a6ee8aa510d345506b8fd30343f4e38d", - "chosen_config": { - "kwargs": { - "BT": 8 - }, - "num_warps": 4, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "key": [ - 128, - 3, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "a426f2ff58b25d5f0e7b23876254d3f4fc255ac596a902eeb14e47917688f508", - "chosen_config": { - "kwargs": { - "BT": 16 - }, - "num_warps": 2, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "key": [ - 128, - 4, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "615e132647ea6735a52e431485cd2fd4706dd754e7d4aec356d0a006e18d9166", - "chosen_config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "key": [ - 128, - 18, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "ef64a3156f4dd428d7b9f80f9435c25eabbcfc192882c199c6fe5fc1c2316be8", - "chosen_config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "key": [ - 128, - 6, - "torch.bfloat16", - "torch.bfloat16", - "torch.float32" - ], - "source_autotune_sha256": "adca9d81bb4df7c1a64dc7e525544b05c362d943c95446f4871bf92f09fb603b", - "chosen_config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - } - ], - "per_NB_policy": [ - { - "NB": 1, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 2, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 16 - }, - "num_warps": 1, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 3, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 16 - }, - "num_warps": 2, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 4, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 5, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 6, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 7, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 8, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 16, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 9, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 8 - }, - "num_warps": 4, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 10, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 8 - }, - "num_warps": 2, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 11, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 8 - }, - "num_warps": 4, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 12, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 13, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 8 - }, - "num_warps": 4, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 14, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 15, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 16, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 17, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 18, - "basis": "original-production-chosen", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 19, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 20, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 21, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 22, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 23, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 24, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 25, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 26, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 27, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 28, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 29, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 30, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 31, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - }, - { - "NB": 32, - "basis": "prospective-fixed-original-NB18-setting", - "config": { - "kwargs": { - "BT": 32 - }, - "num_warps": 8, - "num_ctas": 1, - "num_stages": 3, - "maxnreg": null, - "pre_hook": null, - "ir_override": null - } - } - ], - "unseen_NB_claim": "Deterministic prospective settings, not proof of matching an undefined prior autotune choice. No speed claim.", - "other_kernels": "Unmodified FLA/Transformers dispatch; this profile changes only l2norm_fwd_kernel launch settings.", - "compiled_binaries_included": false, - "private_paths_included": false, - "builder_sha256": "833d5dcfad43a94c239ea2f5ad68d69b0c501e82d9a3b3ef530b5b85b72db379" -} diff --git a/runtime-provenance.json b/runtime-provenance.json deleted file mode 100644 index 696ba51b23c2d868b8b88622d574cf4379e0439e..0000000000000000000000000000000000000000 --- a/runtime-provenance.json +++ /dev/null @@ -1,172 +0,0 @@ -{ - "public_base": "vllm/vllm-openai-rocm@sha256:1fd21abe66455b4df5a2e83629e97cdcc9d58913b16052d8118b92b239792339", - "registry": { - "status": 200, - "manifest_digest": "sha256:1fd21abe66455b4df5a2e83629e97cdcc9d58913b16052d8118b92b239792339", - "config_digest": "sha256:72b270afcf4b9b45f6ad0d76917415d206efc8c9d05399af065b53d6891c781a", - "layers": 36 - }, - "common_prefix_layers": 36, - "qualified_total_layers": 48, - "cpu_read_only_file_audits": [ - { - "kind": "qualified_image", - "returncode": 0, - "data": { - "python": "3.12.13", - "versions": { - "__all__": [ - "__version__", - "debug", - "cuda", - "git_version", - "hip", - "rocm", - "xpu" - ], - "__version__": "2.12.0+git6bbd260", - "debug": false, - "cuda": null, - "git_version": "6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5", - "hip": "7.2.53211", - "rocm": "7.2.3", - "xpu": null - }, - "files": { - "version.py": { - "sha256": "94650c6ec9f5dc786a2b36ee2b7485b4d68e5c6bf1fdf01d0b12da19fde8bc70", - "bytes": 330 - }, - "__init__.py": { - "sha256": "d9dfff4b75d46e4c75572200a3466b70231d05b0318e38ac1bd121789165fb49", - "bytes": 109312 - }, - "_C.cpython-312-x86_64-linux-gnu.so": { - "sha256": "ca9f553cbb03d07a28f9de4a4e7a14510a26f1611bea2559aaa8ceb93065c1ce", - "bytes": 22632 - }, - "lib/libtorch_hip.so": { - "sha256": "72d4b50fef7ee355ad49b7cbaea3ebe982187b05116a9ef7b50321449cffa0ce", - "bytes": 415569576 - }, - "lib/libtorch_cpu.so": { - "sha256": "c0c8bb597e689b8e67266a21c7a65f0eac3089cb94e0c4ed8274b7d30db804ea", - "bytes": 340291896 - } - }, - "packages": { - "torch": "2.12.0+git6bbd260", - "triton": "3.7.1+gitf0b55c07", - "transformers": "5.17.0", - "tokenizers": "0.23.2", - "safetensors": "0.8.0", - "numpy": "2.3.5", - "einops": "0.8.2", - "packaging": "26.3", - "huggingface-hub": "1.31.0", - "fla-core": null, - "flash-linear-attention": null, - "ninja": "1.13.2", - "PyYAML": "6.0.3", - "regex": "2026.9.10", - "tqdm": "4.70.1", - "filelock": "3.32.7" - } - }, - "error_tail": [] - }, - { - "kind": "public_base", - "returncode": 0, - "data": { - "python": "3.12.13", - "versions": { - "__all__": [ - "__version__", - "debug", - "cuda", - "git_version", - "hip", - "rocm", - "xpu" - ], - "__version__": "2.12.0+git6bbd260", - "debug": false, - "cuda": null, - "git_version": "6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5", - "hip": "7.2.53211", - "rocm": "7.2.3", - "xpu": null - }, - "files": { - "version.py": { - "sha256": "94650c6ec9f5dc786a2b36ee2b7485b4d68e5c6bf1fdf01d0b12da19fde8bc70", - "bytes": 330 - }, - "__init__.py": { - "sha256": "d9dfff4b75d46e4c75572200a3466b70231d05b0318e38ac1bd121789165fb49", - "bytes": 109312 - }, - "_C.cpython-312-x86_64-linux-gnu.so": { - "sha256": "ca9f553cbb03d07a28f9de4a4e7a14510a26f1611bea2559aaa8ceb93065c1ce", - "bytes": 22632 - }, - "lib/libtorch_hip.so": { - "sha256": "72d4b50fef7ee355ad49b7cbaea3ebe982187b05116a9ef7b50321449cffa0ce", - "bytes": 415569576 - }, - "lib/libtorch_cpu.so": { - "sha256": "c0c8bb597e689b8e67266a21c7a65f0eac3089cb94e0c4ed8274b7d30db804ea", - "bytes": 340291896 - } - }, - "packages": { - "torch": "2.12.0+git6bbd260", - "triton": "3.7.1+gitf0b55c07", - "transformers": "5.17.0", - "tokenizers": "0.23.2", - "safetensors": "0.8.0", - "numpy": "2.3.5", - "einops": "0.8.2", - "packaging": "26.3", - "huggingface-hub": "1.31.0", - "fla-core": null, - "flash-linear-attention": null, - "ninja": "1.13.2", - "PyYAML": "6.0.3", - "regex": "2026.9.10", - "tqdm": "4.70.1", - "filelock": "3.32.7" - } - }, - "error_tail": [] - } - ], - "history_source_pins": { - "rocm_dev": "rocm/dev-ubuntu-22.04:7.2.3-complete", - "torch_commit": "6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5", - "triton_short_commit": "f0b55c0" - }, - "validation_boundary": "Public base provenance and CPU metadata/file comparison only; no GPU inference or rebuilt-image qualification in this audit.", - "fla_wheels": [ - { - "name": "fla-core", - "version": "0.5.2", - "filename": "fla_core-0.5.2-py3-none-any.whl", - "url": "https://files.pythonhosted.org/packages/2d/ed/dfe19c4da779957eb6a42a26812f9b4e2280bf757a17a71933ff59ffcb98/fla_core-0.5.2-py3-none-any.whl", - "sha256": "5e830c85bad3d0d34677f98ac7074d08687a3756f0f0499d95ceb96eb6920761", - "download_sha256_verified": true, - "bytes": 819225 - }, - { - "name": "flash-linear-attention", - "version": "0.5.2", - "filename": "flash_linear_attention-0.5.2-py3-none-any.whl", - "url": "https://files.pythonhosted.org/packages/90/d2/2070e3cf2148c5cce99ca4876633c4b7b89ec085323600bd3d47aeacd306/flash_linear_attention-0.5.2-py3-none-any.whl", - "sha256": "dcf405d81f5426393b59037097aa700d0f4a841465d5028d5aa543f4502f2400", - "download_sha256_verified": true, - "bytes": 399590 - } - ], - "verified_utc": "2026-09-21" -} diff --git a/runtime.json b/runtime.json deleted file mode 100644 index abd054c1adaeb1b3441e7770695ec3db9003ebf2..0000000000000000000000000000000000000000 --- a/runtime.json +++ /dev/null @@ -1,25 +0,0 @@ -{ - "torch": "2.12.0+git6bbd260", - "torch_git": "6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5", - "hip": "7.2.53211", - "transformers": "5.17.0", - "triton": "3.7.1", - "fla": "0.5.2", - "gdn_implementation": "fla.ops.gated_delta_rule.chunk", - "gdn_new_implementation": true, - "tokenizers": "0.23.2", - "safetensors": "0.8.0", - "gated_delta": "fla.ops.gated_delta_rule.chunk", - "normalization_profile": { - "kind": "decision-fla-l2norm-profile-v1", - "profile_file": "runtime-profile/profile.json", - "profile_sha256": "6b03450d42dbb68f0ffe14945ffcf3e6ea043e1033a819fb7211a8176a51722f", - "guard_file": "code/profile_guard.py", - "guard_sha256": "1603c39038ff783b9d5a5f69110d1695bbb7e28accffd0c03525854e8f258452", - "loader_file": "code/runtime_profile.py", - "loader_sha256": "afb59dda5e4c3890ee5c971549799e69f2e4fa2eafe5893de2168299fd368bda", - "validated_arch": "gfx942", - "activation": "automatic before FLA import by public from_pretrained and bundle DecisionEngine", - "process_scope": "one profile per fresh Python process; same-profile loads allowed; mixed profiles rejected" - } -} diff --git a/src/decision/__init__.py b/src/decision/__init__.py deleted file mode 100644 index 5d2239b526f70c91b4b49d18e9518d57ac0fd1ad..0000000000000000000000000000000000000000 --- a/src/decision/__init__.py +++ /dev/null @@ -1,5 +0,0 @@ -"""The installable wrapper delegates numerical inference to the bundled engine.""" -from .model import DecisionModel - -__all__ = ["DecisionModel"] -__version__ = "1.0.0" diff --git a/src/decision/example.py b/src/decision/example.py deleted file mode 100644 index 7bcc62c41334fd27f6b8b67628b674f14d912bad..0000000000000000000000000000000000000000 --- a/src/decision/example.py +++ /dev/null @@ -1,64 +0,0 @@ -"""Run the real model-card request and save outputs; no invented predictions.""" -import argparse -import hashlib -import json -from pathlib import Path - -from .model import DecisionModel - -REQUEST = { - 'state': 'The customer reports that the same invoice was charged twice. They ask for a refund. There is no product outage.', - 'questions': { - 'destination': {'type': 'choice', 'instructions': 'Choose the team that handles this request.', - 'criteria': {'billing': 'Invoices, payments, refunds and duplicate charges', - 'technical': 'Product errors and troubleshooting'}}, - 'refund_requested': {'type': 'noul', 'instructions': 'Does the customer explicitly ask for a refund?'}, - 'urgency': {'type': 'score', 'instructions': 'Rate urgency using only these ordered levels.', - 'criteria': ['Routine information request with no payment problem or outage', - 'A payment or billing problem, with no product outage', - 'An active product outage stopping the customer from working']}, - }, -} - - -def main(): - parser = argparse.ArgumentParser() - parser.add_argument('model', help='Exported local directory or Hugging Face repo ID') - parser.add_argument('--revision', help='Required for a repo ID; prefer a full commit SHA') - parser.add_argument('--device', default='cuda:0') - parser.add_argument('--local-files-only', action='store_true') - parser.add_argument('--allow-unvalidated-runtime', action='store_true') - parser.add_argument('--output', type=Path, required=True) - args = parser.parse_args() - model = DecisionModel.from_pretrained(args.model, revision=args.revision, device=args.device, - local_files_only=args.local_files_only, - allow_unvalidated_runtime=args.allow_unvalidated_runtime) - response = model.decide(**REQUEST) - # Prove wrapper pass-through against the unchanged engine on this exact request. - reference = model._engine.decide(**REQUEST) - if response != reference: - raise AssertionError('Wrapper/direct-engine response mismatch') - overflow_message = None - try: - model.decide('overflow-test ' * 20000, {'check': {'type': 'noul', 'instructions': 'Is this a test?'}}) - except ValueError as exc: - if 'no truncation allowed' not in str(exc): - raise - overflow_message = str(exc) - if overflow_message is None: - raise AssertionError('Oversized complete input was not rejected') - record = {'request': REQUEST, 'response': response, 'direct_engine_exact_response': True, - 'overflow_rejected': True, 'overflow_message': overflow_message, - 'bundle_manifest_sha256': hashlib.sha256((model.bundle_path / 'bundle-manifest.json').read_bytes()).hexdigest(), - 'runtime': model.runtime, 'revision': args.revision, 'device': args.device, - 'model_name': response['model'], 'example_source_sha256': hashlib.sha256(Path(__file__).read_bytes()).hexdigest()} - if model.runtime.get('normalization_profile') is not None: - import sys - record['normalization_telemetry'] = dict(sys.modules['_decision_process_normalization_profile_v1'].telemetry) - args.output.parent.mkdir(parents=True, exist_ok=True) - args.output.write_text(json.dumps(record, ensure_ascii=False, indent=2) + '\n') - print(json.dumps(response, ensure_ascii=False, indent=2)) - - -if __name__ == '__main__': - main() diff --git a/src/decision/model.py b/src/decision/model.py deleted file mode 100644 index 395a9f77fa6d7f2c550b34fec732e77d957fe51d..0000000000000000000000000000000000000000 --- a/src/decision/model.py +++ /dev/null @@ -1,175 +0,0 @@ -from __future__ import annotations - -import hashlib -import importlib.util -import inspect -import json -import math -from pathlib import Path -import warnings - - -def _sha(path): - digest = hashlib.sha256() - with path.open('rb') as stream: - for block in iter(lambda: stream.read(8 << 20), b''): - digest.update(block) - return digest.hexdigest() - - -def _resolve(source, revision, local_files_only, cache_dir): - path = Path(source).expanduser() - if path.is_dir(): - return path.resolve() - if path.is_absolute() or str(source).startswith(('.', '~')): - raise FileNotFoundError(f'Local model directory does not exist: {source}') - if not revision: - raise ValueError('A Hugging Face repo requires revision=; use a commit SHA for reproducibility.') - try: - from huggingface_hub import snapshot_download - except ImportError as exc: - raise RuntimeError('Hub loading requires huggingface_hub; install this package with [hub].') from exc - return Path(snapshot_download(repo_id=str(source), revision=revision, - local_files_only=local_files_only, cache_dir=cache_dir)).resolve() - - -def _verify_bundle(path): - manifest_path = path / 'bundle-manifest.json' - manifest = json.loads(manifest_path.read_text()) - if manifest.get('format') != 'research-pointer-bundle-v1': - raise ValueError('Unsupported Decision bundle format') - seen = set() - for item in manifest['files']: - relative = Path(item['file']) - if relative.is_absolute() or '..' in relative.parts or item['file'] in seen: - raise ValueError('Invalid or duplicate bundle manifest path') - seen.add(item['file']) - file = path / relative - if not file.is_file() or file.stat().st_size != item['bytes'] or _sha(file) != item['sha256']: - raise ValueError(f'Bundle integrity check failed: {item["file"]}') - required = {'code/decision_api.py', 'code/decision_model.py', 'decision_config.json', - 'decision_head.safetensors', 'temperature.json', 'runtime.json', - 'backbone/config.json', 'tokenizer.json', 'tokenizer_config.json'} - if not required <= seen or not any(name.startswith('backbone/') and name.endswith('.safetensors') for name in seen): - raise ValueError('Bundle manifest omits required inference files') - return manifest - - -def _runtime_report(expected, device, allow_unvalidated_runtime): - if not str(device).startswith('cuda'): - raise RuntimeError('CPU/MPS inference is not implemented by this engine; use a GPU cuda device, including ROCm.') - try: - import torch - import transformers - import fla - import tokenizers - import safetensors - import triton - except ImportError as exc: - raise RuntimeError('Install the inference runtime recorded in runtime.json before loading weights; ' - 'pip install . installs only the lightweight wrapper.') from exc - if not torch.cuda.is_available(): - raise RuntimeError('This release supports GPU inference through PyTorch cuda devices, including ROCm. ' - 'CPU/MPS inference is not implemented by the packaged numerical engine.') - from transformers.models.qwen3_5 import modeling_qwen3_5 as qwen - closure = inspect.getclosurevars(qwen.torch_chunk_gated_delta_rule).nonlocals - implementation = closure.get('implementation') - dispatch = getattr(implementation, '__module__', None) - current = {'torch': str(torch.__version__), 'hip': torch.version.hip, - 'transformers': transformers.__version__, 'fla': fla.__version__, - 'tokenizers': tokenizers.__version__, 'safetensors': safetensors.__version__, - 'triton': triton.__version__, 'gated_delta': dispatch} - differences = {key: {'expected': expected.get(key), 'actual': value} - for key, value in current.items() if expected.get(key) != value} - if not closure.get('is_new_implementation'): - differences['gdn_new_implementation'] = {'expected': True, 'actual': False} - if differences: - message = 'Runtime differs from the validated bundle: ' + json.dumps(differences, sort_keys=True) - if not allow_unvalidated_runtime: - raise RuntimeError(message + '; explicitly set allow_unvalidated_runtime=True for exploratory use.') - warnings.warn(message + '. Numerical or runtime compatibility is not established.', RuntimeWarning, stacklevel=2) - return {'actual': current, 'differences': differences, 'matches_validated_runtime': not differences} - - -def _load_api(path): - api_path = path / 'code/decision_api.py' - spec = importlib.util.spec_from_file_location('decision_bundle_api_' + _sha(api_path)[:16], api_path) - api = importlib.util.module_from_spec(spec) - spec.loader.exec_module(api) - return api - - -class DecisionModel: - """Load an exported bundle without modifying its prompt, head, or calibration.""" - - def __init__(self, engine, bundle_path, manifest, runtime): - self._engine = engine - self.bundle_path = bundle_path - self.manifest = manifest - self.runtime = runtime - - @classmethod - def from_pretrained(cls, source, *, revision=None, device='cuda:0', local_files_only=False, - cache_dir=None, model_name=None, max_length=None, - allow_unvalidated_runtime=False): - """Local paths never download; Hub sources require an explicit revision. - - The manifest fixes the batch size and maximum complete-question length. - A lower max_length may be selected to reject larger requests. Calibration - and parameter dtypes come from the bundle, with no caller override. - """ - path = _resolve(source, revision, local_files_only, cache_dir) - manifest = _verify_bundle(path) - limit = manifest['input_length_limit'] - maximum = limit if max_length is None else max_length - if isinstance(maximum, bool) or not isinstance(maximum, int) or not 1 <= maximum <= limit: - raise ValueError(f'max_length must be an integer in 1..{limit}') - config = json.loads((path / 'decision_config.json').read_text()) - calibration = json.loads((path / 'temperature.json').read_text()) - temperatures = calibration['temperatures'] - if set(temperatures) != {'choice', 'noul', 'score'} or any( - isinstance(v, bool) or not isinstance(v, (int, float)) or not math.isfinite(v) or v <= 0 - for v in temperatures.values()): - raise ValueError('Bundle must contain finite positive temperatures for all three types') - api = _load_api(path) - expected_runtime = json.loads((path / 'runtime.json').read_text()) - if expected_runtime.get('normalization_profile') is not None: - if not hasattr(api, 'prepare_runtime_profile'): - raise RuntimeError('Profiled bundle omits its automatic runtime entrypoint') - profile = api.prepare_runtime_profile(path, device=device) - else: - import sys - if '_decision_process_normalization_profile_v1' in sys.modules: - raise RuntimeError('Use separate processes for profiled and unprofiled models') - profile = None - runtime = _runtime_report(expected_runtime, device, allow_unvalidated_runtime) - runtime['normalization_profile'] = profile - names = {'Qwen/Qwen3.5-2B': 'Decision-1.0-Sol', 'Qwen/Qwen3.5-4B': 'Decision-1.0-Nox'} - name = model_name or config.get('model_name') or names.get(config.get('base_model'), path.name) - engine = api.DecisionEngine(path, path / 'code', device=device, max_length=maximum, - batch_size=manifest['production_batch_size'], temperatures=temperatures, - model_name=name) - import torch - if {p.dtype for p in engine.model.backbone.parameters()} != {torch.bfloat16}: - raise RuntimeError('Backbone must remain BF16') - if {p.dtype for p in engine.model.head.parameters()} != {torch.float32}: - raise RuntimeError('Candidate head must remain FP32') - return cls(engine, path, manifest, runtime) - - def decide(self, state, questions): - """Return native Choice, Noul, and Score answers from the frozen engine. - - Every question includes the complete state, instructions and candidate - descriptions. Any overflowing question raises ValueError before forward; - there is no truncation. Question names and candidate order are preserved. - """ - if not isinstance(questions, dict) or not questions: - raise ValueError('questions must be a nonempty mapping') - if not all(isinstance(name, str) and isinstance(question, dict) for name, question in questions.items()): - raise ValueError('Question names must be strings and questions must be mappings') - # Reject non-JSON inputs/NaN instead of accepting implementation-dependent text. - try: - json.dumps({'state': state, 'questions': questions}, ensure_ascii=False, allow_nan=False) - except (TypeError, ValueError) as exc: - raise ValueError('state and questions must be finite JSON-compatible values') from exc - return self._engine.decide(state, questions)