Upload 45 files
Browse files- LICENSE +176 -0
- THIRD_PARTY.md +13 -0
- USAGE.md +48 -0
- adapter.safetensors +3 -0
- adapter_config.json +25 -0
- ariadne_bench/__init__.py +3 -0
- ariadne_bench/__main__.py +3 -0
- ariadne_bench/adapters.py +154 -0
- ariadne_bench/cli.py +132 -0
- ariadne_bench/datasets.py +532 -0
- ariadne_bench/experiments/__init__.py +1 -0
- ariadne_bench/experiments/align.py +336 -0
- ariadne_bench/experiments/prepare_alignment.py +43 -0
- ariadne_bench/experiments/spectrum.py +84 -0
- ariadne_bench/frozen_input_interface.py +64 -0
- ariadne_bench/full_finetune.py +104 -0
- ariadne_bench/full_input_interface.py +77 -0
- ariadne_bench/hybrid.py +342 -0
- ariadne_bench/hybrid_control.py +169 -0
- ariadne_bench/interfaces.py +121 -0
- ariadne_bench/metrics.py +229 -0
- ariadne_bench/portable.py +88 -0
- ariadne_bench/reproducibility.py +48 -0
- ariadne_bench/runner.py +257 -0
- ariadne_bench/schema.py +172 -0
- benchmarks/sources.lock.json +102 -0
- environment.json +44 -0
- evidence/adapter-equivalence.json +39 -0
- evidence/audit.json +59 -0
- evidence/candidate-uncertainty.json +76 -0
- evidence/identity-check.json +6 -0
- evidence/replay-check.json +32 -0
- evidence/seed-0-history.json +416 -0
- evidence/seed-1-history.json +314 -0
- evidence/seed-2-history.json +331 -0
- evidence/seed-3-history.json +484 -0
- evidence/seed-4-history.json +195 -0
- export-verification.json +42 -0
- licenses/jev-benchmarks-LICENSE +201 -0
- licenses/laya-LICENSE +176 -0
- load_adapter.py +14 -0
- metrics.json +984 -0
- requirements.txt +8 -0
- training_protocol.json +92 -0
- upload-manifest.json +55 -0
LICENSE
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Apache License
|
| 2 |
+
Version 2.0, January 2004
|
| 3 |
+
http://www.apache.org/licenses/
|
| 4 |
+
|
| 5 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 6 |
+
|
| 7 |
+
1. Definitions.
|
| 8 |
+
|
| 9 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 10 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 11 |
+
|
| 12 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 13 |
+
the copyright owner that is granting the License.
|
| 14 |
+
|
| 15 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 16 |
+
other entities that control, are controlled by, or are under common
|
| 17 |
+
control with that entity. For the purposes of this definition,
|
| 18 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 19 |
+
direction or management of such entity, whether by contract or
|
| 20 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 21 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 22 |
+
|
| 23 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 24 |
+
exercising permissions granted by this License.
|
| 25 |
+
|
| 26 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 27 |
+
including but not limited to software source code, documentation
|
| 28 |
+
source, and configuration files.
|
| 29 |
+
|
| 30 |
+
"Object" form shall mean any form resulting from mechanical
|
| 31 |
+
transformation or translation of a Source form, including but
|
| 32 |
+
not limited to compiled object code, generated documentation,
|
| 33 |
+
and conversions to other media types.
|
| 34 |
+
|
| 35 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 36 |
+
Object form, made available under the License, as indicated by a
|
| 37 |
+
copyright notice that is included in or attached to the work
|
| 38 |
+
(an example is provided in the Appendix below).
|
| 39 |
+
|
| 40 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 41 |
+
form, that is based on (or derived from) the Work and for which the
|
| 42 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 43 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 44 |
+
of this License, Derivative Works shall not include works that remain
|
| 45 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 46 |
+
the Work and Derivative Works thereof.
|
| 47 |
+
|
| 48 |
+
"Contribution" shall mean any work of authorship, including
|
| 49 |
+
the original version of the Work and any modifications or additions
|
| 50 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 51 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 52 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 53 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 54 |
+
means any form of electronic, verbal, or written communication sent
|
| 55 |
+
to the Licensor or its representatives, including but not limited to
|
| 56 |
+
communication on electronic mailing lists, source code control systems,
|
| 57 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 58 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 59 |
+
excluding communication that is conspicuously marked or otherwise
|
| 60 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 61 |
+
|
| 62 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 63 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 64 |
+
subsequently incorporated within the Work.
|
| 65 |
+
|
| 66 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 67 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 68 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 69 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 70 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 71 |
+
Work and such Derivative Works in Source or Object form.
|
| 72 |
+
|
| 73 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 74 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 75 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 76 |
+
(except as stated in this section) patent license to make, have made,
|
| 77 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 78 |
+
where such license applies only to those patent claims licensable
|
| 79 |
+
by such Contributor that are necessarily infringed by their
|
| 80 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 81 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 82 |
+
institute patent litigation against any entity (including a
|
| 83 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 84 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 85 |
+
or contributory patent infringement, then any patent licenses
|
| 86 |
+
granted to You under this License for that Work shall terminate
|
| 87 |
+
as of the date such litigation is filed.
|
| 88 |
+
|
| 89 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 90 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 91 |
+
modifications, and in Source or Object form, provided that You
|
| 92 |
+
meet the following conditions:
|
| 93 |
+
|
| 94 |
+
(a) You must give any other recipients of the Work or
|
| 95 |
+
Derivative Works a copy of this License; and
|
| 96 |
+
|
| 97 |
+
(b) You must cause any modified files to carry prominent notices
|
| 98 |
+
stating that You changed the files; and
|
| 99 |
+
|
| 100 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 101 |
+
that You distribute, all copyright, patent, trademark, and
|
| 102 |
+
attribution notices from the Source form of the Work,
|
| 103 |
+
excluding those notices that do not pertain to any part of
|
| 104 |
+
the Derivative Works; and
|
| 105 |
+
|
| 106 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 107 |
+
distribution, then any Derivative Works that You distribute must
|
| 108 |
+
include a readable copy of the attribution notices contained
|
| 109 |
+
within such NOTICE file, excluding those notices that do not
|
| 110 |
+
pertain to any part of the Derivative Works, in at least one
|
| 111 |
+
of the following places: within a NOTICE text file distributed
|
| 112 |
+
as part of the Derivative Works; within the Source form or
|
| 113 |
+
documentation, if provided along with the Derivative Works; or,
|
| 114 |
+
within a display generated by the Derivative Works, if and
|
| 115 |
+
wherever such third-party notices normally appear. The contents
|
| 116 |
+
of the NOTICE file are for informational purposes only and
|
| 117 |
+
do not modify the License. You may add Your own attribution
|
| 118 |
+
notices within Derivative Works that You distribute, alongside
|
| 119 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 120 |
+
that such additional attribution notices cannot be construed
|
| 121 |
+
as modifying the License.
|
| 122 |
+
|
| 123 |
+
You may add Your own copyright statement to Your modifications and
|
| 124 |
+
may provide additional or different license terms and conditions
|
| 125 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 126 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 127 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 128 |
+
the conditions stated in this License.
|
| 129 |
+
|
| 130 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 131 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 132 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 133 |
+
this License, without any additional terms or conditions.
|
| 134 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 135 |
+
the terms of any separate license agreement you may have executed
|
| 136 |
+
with Licensor regarding such Contributions.
|
| 137 |
+
|
| 138 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 139 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 140 |
+
except as required for reasonable and customary use in describing the
|
| 141 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 142 |
+
|
| 143 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 144 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 145 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 146 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 147 |
+
implied, including, without limitation, any warranties or conditions
|
| 148 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 149 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 150 |
+
appropriateness of using or redistributing the Work and assume any
|
| 151 |
+
risks associated with Your exercise of permissions under this License.
|
| 152 |
+
|
| 153 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 154 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 155 |
+
unless required by applicable law (such as deliberate and grossly
|
| 156 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 157 |
+
liable to You for damages, including any direct, indirect, special,
|
| 158 |
+
incidental, or consequential damages of any character arising as a
|
| 159 |
+
result of this License or out of the use or inability to use the
|
| 160 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 161 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 162 |
+
other commercial damages or losses), even if such Contributor
|
| 163 |
+
has been advised of the possibility of such damages.
|
| 164 |
+
|
| 165 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 166 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 167 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 168 |
+
or other liability obligations and/or rights consistent with this
|
| 169 |
+
License. However, in accepting such obligations, You may act only
|
| 170 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 171 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 172 |
+
defend, and hold each Contributor harmless for any liability
|
| 173 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 174 |
+
of your accepting any such warranty or additional liability.
|
| 175 |
+
|
| 176 |
+
END OF TERMS AND CONDITIONS
|
THIRD_PARTY.md
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Third-party sources
|
| 2 |
+
|
| 3 |
+
Ariadne's prompts and preparation recipes were adapted from:
|
| 4 |
+
|
| 5 |
+
- NandhaKishorM/laya, commit `4066d5d5fbf08b66c6757ddeedbd797bd7655bc0`, Apache-2.0. Source: https://github.com/NandhaKishorM/laya
|
| 6 |
+
- AbdelStark/jev-benchmarks, commit `0d610cc53e79bcbec691312b0c4adb4a0e371642`, Apache-2.0. The BTZSC grouping and balanced sample recipe follows this project. Source: https://github.com/AbdelStark/jev-benchmarks
|
| 7 |
+
|
| 8 |
+
Copyright and license notices are retained in `licenses/`. The code is adapted for this repository's canonical case format, adapters and scorer. No upstream claims of affiliation or endorsement apply.
|
| 9 |
+
|
| 10 |
+
Datasets and model weights keep their own licenses. Their source URLs, immutable revisions and model-card license metadata appear in `benchmarks/sources.lock.json` and model configs. Data is downloaded on request into ignored local directories; the harness does not relicense it. See each source card for terms, provenance and training exposure.
|
| 11 |
+
|
| 12 |
+
TypeSafe's public documentation defines the native System One request/response contract. The official evaluation website is cited for research; its examples are not redistributed.
|
| 13 |
+
|
USAGE.md
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Load the Laya linear adapter
|
| 2 |
+
|
| 3 |
+
This package contains the validation-selected **seed 1, epoch 14** checkpoint from the deterministic frozen-base experiment. It trained 1,049,600 affine parameters at LR 3e-4 while all base Laya weights remained frozen. Its 4.2 MB weight file requires the pinned base model, which the loader downloads separately.
|
| 4 |
+
|
| 5 |
+
The ongoing lower-learning-rate experiment is separate. These are the already verified weights, not a claim that the training stability investigation is complete.
|
| 6 |
+
|
| 7 |
+
## Upload
|
| 8 |
+
|
| 9 |
+
Extract the ZIP and upload the files and folders **at the root of your existing Hugging Face model repository**. There is deliberately no README.md in this bundle, so it can coexist with your existing model card. The ZIP is a transport archive; upload its extracted contents for normal use.
|
| 10 |
+
|
| 11 |
+
This is a custom input adapter. Loading requires the bundled loader; bare `laya.load(repo_id)` and `AutoModel.from_pretrained(repo_id)` do not install the extra affine layer.
|
| 12 |
+
|
| 13 |
+
## Load a downloaded copy
|
| 14 |
+
|
| 15 |
+
Install an appropriate PyTorch build and the versions in `requirements.txt`. The measured GPU environment is recorded in `environment.json`. From the downloaded repository directory:
|
| 16 |
+
|
| 17 |
+
```python
|
| 18 |
+
from load_adapter import load
|
| 19 |
+
|
| 20 |
+
model = load(device="cuda:0")
|
| 21 |
+
answer = model.predict("The delivery arrived damaged.", {
|
| 22 |
+
"route": {
|
| 23 |
+
"type": "choice",
|
| 24 |
+
"instructions": "Choose the customer-support queue.",
|
| 25 |
+
"criteria": {"delivery": "Delivery and damaged items", "billing": "Payments and invoices"},
|
| 26 |
+
}
|
| 27 |
+
})
|
| 28 |
+
print(answer)
|
| 29 |
+
model.close()
|
| 30 |
+
```
|
| 31 |
+
|
| 32 |
+
After you upload, first download your repository with `huggingface_hub.snapshot_download("YOUR_ACCOUNT/YOUR_REPOSITORY")`, then run the example from that downloaded directory. Use `device="cpu"` for CPU inference. `base_path` can point to the pinned base snapshot already on disk; `local_files_only=True` uses cached base files.
|
| 33 |
+
|
| 34 |
+
The loader verifies the adapter checksum, base configuration/tokenizer hashes and pretrained tensor hash. Strict determinism is enabled by default and requires the CUBLAS workspace environment to be set before CUDA is initialized. A fresh process handles this automatically. Numerical agreement across other hardware or software is not guaranteed.
|
| 35 |
+
|
| 36 |
+
## Measured performance and limits
|
| 37 |
+
|
| 38 |
+
Selected checkpoint accuracy: typed decisions **76.95%**, AG News **93.17%**, BoolQ **79.83%**, Emotion **57.33%**, prompt injections **71.55%**, SST-5 **42.00%**, MASSIVE EN **74.33%**, XNLI EN **87.67%**. Native base typed accuracy was 36.25% under the matched protocol.
|
| 39 |
+
|
| 40 |
+
Across all five training seeds, typed accuracy was **70.23% ± 6.68 pp** (sample SD). Three seeds had severe retention losses. The selected model also lost 3.17 pp on BoolQ and 4.33 pp on MASSIVE versus base. Full results and paired uncertainty are in `metrics.json` and `evidence/candidate-uncertainty.json`. Other seeds' scores are included for transparency; this bundle contains only seed 1's weights.
|
| 41 |
+
|
| 42 |
+
The benchmark uses 2,000 typed decisions plus fixed subsets of seven other tasks, totaling 5,116 decisions. MASSIVE uses 20 candidate intents. The same test subsets were inspected in earlier experiments. These results describe an experimental task adapter, not established broad task improvement.
|
| 43 |
+
|
| 44 |
+
The base checkpoint's choice:11+ temperature is clamped from 0.1005828 to 0.5 by Laya 0.3.20. This expected load warning concerns confidence calibration; raw-logit training and class ordering are unaffected. Calibration was not refitted after adaptation.
|
| 45 |
+
|
| 46 |
+
All 5,116 answer objects matched the original selected checkpoint when the portable export was tested from outside the workspace. `export-verification.json` records that check. `upload-manifest.json` verifies that this bundle keeps the same weights and inference code, with the other seeds removed from its loading configuration.
|
| 47 |
+
|
| 48 |
+
License: Apache-2.0. Base model: `convaiinnovations/laya`, pinned to revision `55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851`. Attribution is in `THIRD_PARTY.md` and `licenses/`.
|
adapter.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e5ec5e1d494509f63960b6949af15be8390442271805ba45d796a1911b5c239a
|
| 3 |
+
size 4198616
|
adapter_config.json
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"format_version": 1,
|
| 3 |
+
"interface": "exact_linear",
|
| 4 |
+
"position": "after native embeddings, before ModernBERT block 0",
|
| 5 |
+
"base_model": "convaiinnovations/laya",
|
| 6 |
+
"base_revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851",
|
| 7 |
+
"candidate_seed": 1,
|
| 8 |
+
"max_len": 1024,
|
| 9 |
+
"head_max_len": 256,
|
| 10 |
+
"trainable_parameters_during_training": 1049600,
|
| 11 |
+
"frozen_state_sha256": "7560d83b1e1c17cce7eb67b25674fc8e7f0958bac4b47f559c725f8695c4d59b",
|
| 12 |
+
"base_config_sha256": {
|
| 13 |
+
"rl_agent_config.json": "ae287b56bbcf5f8c4f4541ae9dfd00c914c4c48b940b8398c3058af37ba92bbd",
|
| 14 |
+
"encoder/config.json": "bf3ab80598fdccf414855a2ce80f22859e4492d06ca8a62ddd1cfb63972f8979",
|
| 15 |
+
"tokenizer/tokenizer.json": "6c8aaa9a542084f2457eab775d4eeb51f92a70c0fd9de28d5edb0ddec3c08d30",
|
| 16 |
+
"tokenizer/tokenizer_config.json": "50044de60daaa73df97d262e15a40d4faf0160e7d742df64b377877a1320dd12"
|
| 17 |
+
},
|
| 18 |
+
"seeds": {
|
| 19 |
+
"1": {
|
| 20 |
+
"weights_file": "adapter.safetensors",
|
| 21 |
+
"weights_sha256": "e5ec5e1d494509f63960b6949af15be8390442271805ba45d796a1911b5c239a",
|
| 22 |
+
"selected_epoch": 14
|
| 23 |
+
}
|
| 24 |
+
}
|
| 25 |
+
}
|
ariadne_bench/__init__.py
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Typed-decision benchmark harness."""
|
| 2 |
+
|
| 3 |
+
__version__ = "0.1.0"
|
ariadne_bench/__main__.py
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from .cli import main
|
| 2 |
+
|
| 3 |
+
raise SystemExit(main())
|
ariadne_bench/adapters.py
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Adapters receive only state and questions; labels never enter model input."""
|
| 2 |
+
|
| 3 |
+
import importlib
|
| 4 |
+
import os
|
| 5 |
+
import urllib.error
|
| 6 |
+
import urllib.request
|
| 7 |
+
|
| 8 |
+
from .schema import dumps, labels
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
class Uniform:
|
| 12 |
+
def predict(self, state, questions):
|
| 13 |
+
answers = {}
|
| 14 |
+
for qid, q in questions.items():
|
| 15 |
+
keys = labels(q)
|
| 16 |
+
a = {"type": q["type"], "probabilities": {k: 1 / len(keys) for k in keys}}
|
| 17 |
+
if q["type"] == "noul":
|
| 18 |
+
a["noul"] = 0.5
|
| 19 |
+
elif q["type"] == "choice":
|
| 20 |
+
a["choice"] = keys[0]
|
| 21 |
+
else:
|
| 22 |
+
a["score"] = (len(keys) - 1) / 2
|
| 23 |
+
answers[qid] = a
|
| 24 |
+
return {"model": "uniform", "answers": answers}
|
| 25 |
+
|
| 26 |
+
|
| 27 |
+
class NativeHTTP:
|
| 28 |
+
def __init__(self, config):
|
| 29 |
+
self.endpoint = config["endpoint"]
|
| 30 |
+
self.model = config["model"]
|
| 31 |
+
self.timeout = config.get("timeout_seconds", 60)
|
| 32 |
+
self.key = os.environ.get(config.get("api_key_env", ""))
|
| 33 |
+
if config.get("api_key_env") and not self.key:
|
| 34 |
+
raise ValueError(f"set {config['api_key_env']} before running this adapter")
|
| 35 |
+
if not self.endpoint.startswith(("http://", "https://")):
|
| 36 |
+
raise ValueError("endpoint must be an HTTP(S) URL")
|
| 37 |
+
|
| 38 |
+
# Refuse redirects so a bearer credential cannot move to a different host.
|
| 39 |
+
class NoRedirect(urllib.request.HTTPRedirectHandler):
|
| 40 |
+
def redirect_request(self, req, fp, code, msg, headers, newurl):
|
| 41 |
+
return None
|
| 42 |
+
|
| 43 |
+
self.opener = urllib.request.build_opener(NoRedirect())
|
| 44 |
+
|
| 45 |
+
def predict(self, state, questions):
|
| 46 |
+
import json
|
| 47 |
+
|
| 48 |
+
headers = {"Content-Type": "application/json"}
|
| 49 |
+
if self.key:
|
| 50 |
+
headers["Authorization"] = "Bearer " + self.key
|
| 51 |
+
payload = {"model": self.model, "state": state, "questions": questions}
|
| 52 |
+
request = urllib.request.Request(
|
| 53 |
+
self.endpoint, data=dumps(payload).encode(), headers=headers
|
| 54 |
+
)
|
| 55 |
+
try:
|
| 56 |
+
with self.opener.open(request, timeout=self.timeout) as response:
|
| 57 |
+
return json.load(response)
|
| 58 |
+
except urllib.error.HTTPError as exc:
|
| 59 |
+
# Response bodies may contain credentials or echoed inputs. Store status only.
|
| 60 |
+
raise RuntimeError(f"HTTP {exc.code}") from None
|
| 61 |
+
except urllib.error.URLError:
|
| 62 |
+
raise RuntimeError("HTTP connection failed") from None
|
| 63 |
+
|
| 64 |
+
|
| 65 |
+
class Laya:
|
| 66 |
+
def __init__(self, config):
|
| 67 |
+
os.environ.setdefault("USE_TF", "0")
|
| 68 |
+
try:
|
| 69 |
+
import laya
|
| 70 |
+
except ImportError as exc:
|
| 71 |
+
raise RuntimeError("install the laya extra to use the local adapter") from exc
|
| 72 |
+
from pathlib import Path
|
| 73 |
+
|
| 74 |
+
model_path = Path(config["model"])
|
| 75 |
+
revision = None
|
| 76 |
+
if not model_path.is_dir():
|
| 77 |
+
from huggingface_hub import HfApi, snapshot_download
|
| 78 |
+
|
| 79 |
+
revision = config.get("revision")
|
| 80 |
+
if not (
|
| 81 |
+
isinstance(revision, str)
|
| 82 |
+
and len(revision) == 40
|
| 83 |
+
and all(c in "0123456789abcdef" for c in revision)
|
| 84 |
+
):
|
| 85 |
+
revision = HfApi().model_info(config["model"], revision=revision).sha
|
| 86 |
+
prefix = config.get("subfolder", "").strip("/")
|
| 87 |
+
prefix = prefix + "/" if prefix else ""
|
| 88 |
+
model_path = Path(
|
| 89 |
+
snapshot_download(
|
| 90 |
+
config["model"],
|
| 91 |
+
revision=revision,
|
| 92 |
+
allow_patterns=[
|
| 93 |
+
prefix + name
|
| 94 |
+
for name in (
|
| 95 |
+
"rl_agent_config.json",
|
| 96 |
+
"model.safetensors",
|
| 97 |
+
"tokenizer/*",
|
| 98 |
+
"encoder/*",
|
| 99 |
+
)
|
| 100 |
+
],
|
| 101 |
+
)
|
| 102 |
+
)
|
| 103 |
+
if config.get("subfolder"):
|
| 104 |
+
model_path = model_path / config["subfolder"]
|
| 105 |
+
self.agent = laya.load(str(model_path), device=config.get("device"))
|
| 106 |
+
self.max_len = config.get("max_len")
|
| 107 |
+
if config.get("freeze_parameters", False):
|
| 108 |
+
self.agent.model.requires_grad_(False)
|
| 109 |
+
import torch
|
| 110 |
+
|
| 111 |
+
self.metadata = {
|
| 112 |
+
"resolved_revision": revision,
|
| 113 |
+
"checkpoint_path": str(model_path.resolve()),
|
| 114 |
+
"device": str(self.agent.device),
|
| 115 |
+
"laya_version": laya.__version__,
|
| 116 |
+
"model_config": self.agent.cfg,
|
| 117 |
+
"dtype": str(self.agent.dtype),
|
| 118 |
+
"torch_version": torch.__version__,
|
| 119 |
+
"cpu_threads": torch.get_num_threads(),
|
| 120 |
+
"gpu_name": torch.cuda.get_device_name(self.agent.device)
|
| 121 |
+
if self.agent.device.type == "cuda"
|
| 122 |
+
else None,
|
| 123 |
+
"frozen_parameters": bool(config.get("freeze_parameters", False)),
|
| 124 |
+
"served_temperatures": [float(t) for t in self.agent.temperature],
|
| 125 |
+
"served_temperature_by_options": dict(self.agent.temperature_by_options),
|
| 126 |
+
}
|
| 127 |
+
|
| 128 |
+
def synchronize(self):
|
| 129 |
+
import torch
|
| 130 |
+
|
| 131 |
+
if self.agent.device.type == "cuda":
|
| 132 |
+
torch.cuda.synchronize(self.agent.device)
|
| 133 |
+
elif self.agent.device.type == "mps":
|
| 134 |
+
torch.mps.synchronize()
|
| 135 |
+
elif self.agent.device.type == "xpu":
|
| 136 |
+
torch.xpu.synchronize(self.agent.device)
|
| 137 |
+
|
| 138 |
+
def predict(self, state, questions):
|
| 139 |
+
kw = {} if self.max_len is None else {"max_len": self.max_len}
|
| 140 |
+
return self.agent.predict(state, questions, **kw)
|
| 141 |
+
|
| 142 |
+
|
| 143 |
+
def create_adapter(config):
|
| 144 |
+
kind = config["adapter"]
|
| 145 |
+
if kind == "uniform":
|
| 146 |
+
return Uniform()
|
| 147 |
+
if kind == "http":
|
| 148 |
+
return NativeHTTP(config)
|
| 149 |
+
if kind == "laya":
|
| 150 |
+
return Laya(config)
|
| 151 |
+
if kind == "python":
|
| 152 |
+
module, name = config["factory"].split(":", 1)
|
| 153 |
+
return getattr(importlib.import_module(module), name)(config.get("options", {}))
|
| 154 |
+
raise ValueError(f"unknown adapter: {kind}")
|
ariadne_bench/cli.py
ADDED
|
@@ -0,0 +1,132 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import argparse
|
| 2 |
+
import json
|
| 3 |
+
import sys
|
| 4 |
+
from pathlib import Path
|
| 5 |
+
|
| 6 |
+
from .datasets import APPS, CORE, prepare
|
| 7 |
+
from .runner import compare, run, score
|
| 8 |
+
from .schema import read_jsonl, save_bundle
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
def positive(value):
|
| 12 |
+
number = int(value)
|
| 13 |
+
if number <= 0:
|
| 14 |
+
raise argparse.ArgumentTypeError("must be positive")
|
| 15 |
+
return number
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
def nonnegative(value):
|
| 19 |
+
number = int(value)
|
| 20 |
+
if number < 0:
|
| 21 |
+
raise argparse.ArgumentTypeError("must be nonnegative")
|
| 22 |
+
return number
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
def tolerance_value(value):
|
| 26 |
+
number = float(value)
|
| 27 |
+
if not 0 <= number <= 0.1:
|
| 28 |
+
raise argparse.ArgumentTypeError("tolerance must be in [0, 0.1]")
|
| 29 |
+
return number
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def main(argv=None):
|
| 33 |
+
parser = argparse.ArgumentParser(
|
| 34 |
+
description="Freeze data, evaluate typed decisions, compare models."
|
| 35 |
+
)
|
| 36 |
+
sub = parser.add_subparsers(dest="command", required=True)
|
| 37 |
+
sub.add_parser("list", help="list available suites")
|
| 38 |
+
p = sub.add_parser("prepare", help="download pinned sources and freeze prompts")
|
| 39 |
+
p.add_argument(
|
| 40 |
+
"--profile",
|
| 41 |
+
choices=["smoke", "latency", "laya-core", "laya-apps", "laya-multilingual", "jev-btzsc"],
|
| 42 |
+
default="laya-core",
|
| 43 |
+
)
|
| 44 |
+
p.add_argument(
|
| 45 |
+
"--suites", help="comma-separated suites; multilingual names include language, e.g. xnli.en"
|
| 46 |
+
)
|
| 47 |
+
p.add_argument("--out", required=True)
|
| 48 |
+
p.add_argument("--lock", default="benchmarks/sources.lock.json")
|
| 49 |
+
p.add_argument("--cache", default=".cache/datasets")
|
| 50 |
+
p.add_argument("--limit", type=positive, help="cases per suite; changes the protocol")
|
| 51 |
+
p.add_argument("--seed", type=int, default=13)
|
| 52 |
+
p.add_argument("--permutations", type=nonnegative, default=0)
|
| 53 |
+
p = sub.add_parser("import", help="validate and freeze your own canonical JSONL")
|
| 54 |
+
p.add_argument("--cases", required=True)
|
| 55 |
+
p.add_argument("--out", required=True)
|
| 56 |
+
p = sub.add_parser("run", help="run a model; nonzero exit if any decisions fail")
|
| 57 |
+
p.add_argument("--data", required=True)
|
| 58 |
+
p.add_argument("--config", required=True)
|
| 59 |
+
p.add_argument("--out", required=True)
|
| 60 |
+
p.add_argument("--warmup", type=nonnegative, default=0)
|
| 61 |
+
p.add_argument("--resume", action="store_true")
|
| 62 |
+
p.add_argument("--ece-bins", type=positive, default=15)
|
| 63 |
+
p.add_argument("--probability-tolerance", type=tolerance_value, default=0.02)
|
| 64 |
+
p = sub.add_parser("score", help="score saved predictions without model calls")
|
| 65 |
+
p.add_argument("--data", required=True)
|
| 66 |
+
p.add_argument("--predictions", required=True)
|
| 67 |
+
p.add_argument("--out", required=True)
|
| 68 |
+
p.add_argument("--ece-bins", type=positive, default=15)
|
| 69 |
+
p.add_argument("--probability-tolerance", type=tolerance_value, default=0.02)
|
| 70 |
+
p = sub.add_parser("compare", help="compare reports from identical frozen cases")
|
| 71 |
+
p.add_argument("reports", nargs="+")
|
| 72 |
+
p.add_argument("--out")
|
| 73 |
+
args = parser.parse_args(argv)
|
| 74 |
+
try:
|
| 75 |
+
if args.command == "list":
|
| 76 |
+
print(
|
| 77 |
+
"\n".join(
|
| 78 |
+
dict.fromkeys(
|
| 79 |
+
CORE
|
| 80 |
+
+ APPS
|
| 81 |
+
+ [
|
| 82 |
+
"massive-intent.<language>",
|
| 83 |
+
"massive-scenario.<language>",
|
| 84 |
+
"xnli.<language>",
|
| 85 |
+
"btzsc.agnews",
|
| 86 |
+
"btzsc.emotiondair",
|
| 87 |
+
"btzsc.banking77",
|
| 88 |
+
]
|
| 89 |
+
)
|
| 90 |
+
)
|
| 91 |
+
)
|
| 92 |
+
elif args.command == "prepare":
|
| 93 |
+
result = prepare(
|
| 94 |
+
args.profile,
|
| 95 |
+
args.suites,
|
| 96 |
+
args.out,
|
| 97 |
+
args.lock,
|
| 98 |
+
args.cache,
|
| 99 |
+
args.limit,
|
| 100 |
+
args.seed,
|
| 101 |
+
args.permutations,
|
| 102 |
+
)
|
| 103 |
+
print(json.dumps({k: result[k] for k in ("cases", "decisions", "sha256")}, indent=2))
|
| 104 |
+
elif args.command == "import":
|
| 105 |
+
result = save_bundle(args.out, read_jsonl(args.cases), {"profile": "custom"})
|
| 106 |
+
print(f"Frozen {result['cases']} cases")
|
| 107 |
+
elif args.command == "run":
|
| 108 |
+
result = run(
|
| 109 |
+
args.data,
|
| 110 |
+
args.config,
|
| 111 |
+
args.out,
|
| 112 |
+
args.ece_bins,
|
| 113 |
+
args.probability_tolerance,
|
| 114 |
+
args.warmup,
|
| 115 |
+
args.resume,
|
| 116 |
+
)
|
| 117 |
+
print(f"Report: {Path(args.out) / 'report.md'}")
|
| 118 |
+
return 2 if result["overall"]["failed"] else 0
|
| 119 |
+
elif args.command == "score":
|
| 120 |
+
result = score(
|
| 121 |
+
args.data, args.predictions, args.out, args.ece_bins, args.probability_tolerance
|
| 122 |
+
)
|
| 123 |
+
return 2 if result["overall"]["failed"] else 0
|
| 124 |
+
else:
|
| 125 |
+
table = compare(args.reports)
|
| 126 |
+
if args.out:
|
| 127 |
+
Path(args.out).write_text(table + "\n")
|
| 128 |
+
print(table)
|
| 129 |
+
except (ValueError, KeyError, OSError, ImportError) as exc:
|
| 130 |
+
print(f"Error: {exc}", file=sys.stderr)
|
| 131 |
+
return 1
|
| 132 |
+
return 0
|
ariadne_bench/datasets.py
ADDED
|
@@ -0,0 +1,532 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Dataset preparation is separate from inference and freezes all prompt bytes."""
|
| 2 |
+
|
| 3 |
+
import copy
|
| 4 |
+
import json
|
| 5 |
+
import random
|
| 6 |
+
from collections import defaultdict
|
| 7 |
+
from pathlib import Path
|
| 8 |
+
|
| 9 |
+
from .schema import labels, save_bundle
|
| 10 |
+
|
| 11 |
+
LAYA_REVISION = "4066d5d5fbf08b66c6757ddeedbd797bd7655bc0"
|
| 12 |
+
BTZSC_REVISION = "fef2a2ac62b69c58670047dddf045c53d7c3cb5e"
|
| 13 |
+
CORE = ["typed-decisions", "ag-news", "emotion", "banking77", "boolq", "sst5", "prompt-injections"]
|
| 14 |
+
APPS = [
|
| 15 |
+
"ag-news",
|
| 16 |
+
"emotion",
|
| 17 |
+
"banking77",
|
| 18 |
+
"enron-spam",
|
| 19 |
+
"phishing",
|
| 20 |
+
"toxic-chat",
|
| 21 |
+
"jailbreak",
|
| 22 |
+
"support-triage",
|
| 23 |
+
"rag-relevance",
|
| 24 |
+
"model-routing",
|
| 25 |
+
]
|
| 26 |
+
MASSIVE_LANGS = "en,de,fr,es,pt,ru,tr,ar,hi,ta,zh-CN,ja,ko,sw".split(",")
|
| 27 |
+
XNLI_LANGS = "en,de,fr,es,ru,tr,ar,hi,ur,vi,th,el,bg,zh,sw".split(",")
|
| 28 |
+
SPECS = {
|
| 29 |
+
"typed-decisions": ("LocalLLaMA/typed-decisions", "all", "test", None),
|
| 30 |
+
"ag-news": ("fancyzhx/ag_news", None, "test", 600),
|
| 31 |
+
"emotion": ("dair-ai/emotion", "split", "test", 600),
|
| 32 |
+
"banking77": ("mteb/banking77", None, "test", 500),
|
| 33 |
+
"boolq": ("google/boolq", None, "validation", 600),
|
| 34 |
+
"sst5": ("SetFit/sst5", None, "test", 600),
|
| 35 |
+
"prompt-injections": ("deepset/prompt-injections", None, "test", None),
|
| 36 |
+
"enron-spam": ("SetFit/enron_spam", None, "test", 400),
|
| 37 |
+
"phishing": ("zefang-liu/phishing-email-dataset", None, "train", 400),
|
| 38 |
+
"toxic-chat": ("lmsys/toxic-chat", "toxicchat0124", "test", 400),
|
| 39 |
+
"jailbreak": ("lmsys/toxic-chat", "toxicchat0124", "test", 400),
|
| 40 |
+
"support-triage": ("Tobi-Bueck/customer-support-tickets", None, "train", 400),
|
| 41 |
+
"rag-relevance": ("microsoft/ms_marco", "v1.1", "validation", 400),
|
| 42 |
+
"model-routing": ("openai/gsm8k", "main", "test", 400),
|
| 43 |
+
}
|
| 44 |
+
AG_CRITERIA = {
|
| 45 |
+
"world": "world news and international politics",
|
| 46 |
+
"sports": "sports",
|
| 47 |
+
"business": "business and economy",
|
| 48 |
+
"sci_tech": "science and technology",
|
| 49 |
+
}
|
| 50 |
+
NLI_CRITERIA = {
|
| 51 |
+
"entailment": "the premise implies the hypothesis is true",
|
| 52 |
+
"neutral": "the premise neither implies nor contradicts the hypothesis",
|
| 53 |
+
"contradiction": "the premise implies the hypothesis is false",
|
| 54 |
+
}
|
| 55 |
+
QUEUES = {
|
| 56 |
+
"Technical Support": "technical problems, bugs, outages, integrations",
|
| 57 |
+
"Product Support": "help using a product or feature",
|
| 58 |
+
"Customer Service": "general account or service questions",
|
| 59 |
+
"IT Support": "internal IT, devices, access, networks",
|
| 60 |
+
"Billing and Payments": "invoices, charges, refunds, payment methods",
|
| 61 |
+
"Returns and Exchanges": "returning or exchanging an item",
|
| 62 |
+
"Service Outages and Maintenance": "downtime, outages, scheduled maintenance",
|
| 63 |
+
"Sales and Pre-Sales": "pricing, quotes, buying",
|
| 64 |
+
"Human Resources": "employment, payroll, leave, hiring",
|
| 65 |
+
"General Inquiry": "anything else",
|
| 66 |
+
}
|
| 67 |
+
|
| 68 |
+
|
| 69 |
+
def parsed(value):
|
| 70 |
+
if isinstance(value, str):
|
| 71 |
+
try:
|
| 72 |
+
return json.loads(value)
|
| 73 |
+
except json.JSONDecodeError:
|
| 74 |
+
pass
|
| 75 |
+
return value
|
| 76 |
+
|
| 77 |
+
|
| 78 |
+
def single(suite, index, state, qid, question, target, **metadata):
|
| 79 |
+
keys = labels(question)
|
| 80 |
+
gold = {"label": keys[int(target)]}
|
| 81 |
+
if question["type"] == "score":
|
| 82 |
+
gold["score"] = float(target)
|
| 83 |
+
return {
|
| 84 |
+
"id": f"{suite}:{index}",
|
| 85 |
+
"suite": suite,
|
| 86 |
+
"state": state,
|
| 87 |
+
"questions": {qid: question},
|
| 88 |
+
"gold": {qid: gold},
|
| 89 |
+
**metadata,
|
| 90 |
+
}
|
| 91 |
+
|
| 92 |
+
|
| 93 |
+
def question(kind, instructions, criteria=None):
|
| 94 |
+
q = {"type": kind, "instructions": instructions}
|
| 95 |
+
if criteria is not None:
|
| 96 |
+
q["criteria"] = criteria
|
| 97 |
+
return q
|
| 98 |
+
|
| 99 |
+
|
| 100 |
+
class Sources:
|
| 101 |
+
def __init__(self, lock, cache):
|
| 102 |
+
self.lock = lock
|
| 103 |
+
self.cache = cache
|
| 104 |
+
self.used = {}
|
| 105 |
+
|
| 106 |
+
def load(self, repo, config=None, split="test"):
|
| 107 |
+
from datasets import load_dataset
|
| 108 |
+
|
| 109 |
+
revision = self.lock[repo]["revision"]
|
| 110 |
+
if len(revision) != 40 or any(c not in "0123456789abcdef" for c in revision):
|
| 111 |
+
raise ValueError(f"source lock must pin an immutable commit: {repo}")
|
| 112 |
+
key = f"{repo}:{config or 'default'}:{split}"
|
| 113 |
+
self.used[key] = {"repository": repo, "config": config, "split": split, **self.lock[repo]}
|
| 114 |
+
return load_dataset(repo, name=config, split=split, revision=revision, cache_dir=self.cache)
|
| 115 |
+
|
| 116 |
+
|
| 117 |
+
def typed(rows):
|
| 118 |
+
result = []
|
| 119 |
+
for r in rows:
|
| 120 |
+
qs, gold = parsed(r["questions"]), parsed(r["gold"])
|
| 121 |
+
gs = {}
|
| 122 |
+
for qid, q in qs.items():
|
| 123 |
+
g = gold[qid]
|
| 124 |
+
label = str(g["label"]).lower() if q["type"] == "noul" else str(g["label"])
|
| 125 |
+
gs[qid] = {"label": label, "probabilities": g["probabilities"]}
|
| 126 |
+
if q["type"] == "score":
|
| 127 |
+
gs[qid]["score"] = float(g["score"])
|
| 128 |
+
result.append(
|
| 129 |
+
{
|
| 130 |
+
"id": "typed-decisions:" + r["id"],
|
| 131 |
+
"suite": "typed-decisions",
|
| 132 |
+
"workflow": r["workflow"],
|
| 133 |
+
"state": parsed(r["state"]),
|
| 134 |
+
"questions": qs,
|
| 135 |
+
"gold": gs,
|
| 136 |
+
}
|
| 137 |
+
)
|
| 138 |
+
return result
|
| 139 |
+
|
| 140 |
+
|
| 141 |
+
def build_core(name, rows):
|
| 142 |
+
if name == "typed-decisions":
|
| 143 |
+
return typed(rows)
|
| 144 |
+
bank_labels = sorted(set(rows["label_text"])) if name == "banking77" else None
|
| 145 |
+
output = []
|
| 146 |
+
for i, r in enumerate(rows):
|
| 147 |
+
if name == "ag-news":
|
| 148 |
+
state, qid, q, target = (
|
| 149 |
+
{"article": r["text"]},
|
| 150 |
+
"topic",
|
| 151 |
+
question("choice", "What is the topic of `article`?", AG_CRITERIA),
|
| 152 |
+
r["label"],
|
| 153 |
+
)
|
| 154 |
+
elif name == "emotion":
|
| 155 |
+
state, qid, q, target = (
|
| 156 |
+
{"text": r["text"]},
|
| 157 |
+
"emotion",
|
| 158 |
+
question(
|
| 159 |
+
"choice",
|
| 160 |
+
"Which emotion is most strongly expressed in `text`?",
|
| 161 |
+
dict.fromkeys(["sadness", "joy", "love", "anger", "fear", "surprise"]),
|
| 162 |
+
),
|
| 163 |
+
r["label"],
|
| 164 |
+
)
|
| 165 |
+
elif name == "banking77":
|
| 166 |
+
state, qid, q, target = (
|
| 167 |
+
{"message": r["text"]},
|
| 168 |
+
"intent",
|
| 169 |
+
question(
|
| 170 |
+
"choice",
|
| 171 |
+
"Which banking intent does `message` express?",
|
| 172 |
+
{s.replace("_", " "): None for s in bank_labels},
|
| 173 |
+
),
|
| 174 |
+
bank_labels.index(r["label_text"]),
|
| 175 |
+
)
|
| 176 |
+
elif name == "boolq":
|
| 177 |
+
state, qid, q, target = (
|
| 178 |
+
{"passage": r["passage"], "question": r["question"]},
|
| 179 |
+
"answer",
|
| 180 |
+
question("noul", "Based on `passage`, is the answer to `question` yes?"),
|
| 181 |
+
int(r["answer"]),
|
| 182 |
+
)
|
| 183 |
+
elif name == "sst5":
|
| 184 |
+
state, qid, q, target = (
|
| 185 |
+
{"text": r["text"]},
|
| 186 |
+
"sentiment",
|
| 187 |
+
question(
|
| 188 |
+
"score",
|
| 189 |
+
"How positive is the sentiment of `text`?",
|
| 190 |
+
["very negative", "negative", "neutral", "positive", "very positive"],
|
| 191 |
+
),
|
| 192 |
+
r["label"],
|
| 193 |
+
)
|
| 194 |
+
else:
|
| 195 |
+
state, qid, q, target = (
|
| 196 |
+
{"text": r["text"]},
|
| 197 |
+
"injection",
|
| 198 |
+
question(
|
| 199 |
+
"noul",
|
| 200 |
+
"Does `text` try to inject or override instructions given to an AI system?",
|
| 201 |
+
),
|
| 202 |
+
r["label"],
|
| 203 |
+
)
|
| 204 |
+
output.append(single(name, i, state, qid, q, target))
|
| 205 |
+
return output
|
| 206 |
+
|
| 207 |
+
|
| 208 |
+
def build_multilingual(name, rows, n, seed=13):
|
| 209 |
+
family, language = name.split(".", 1)
|
| 210 |
+
rng = random.Random(seed)
|
| 211 |
+
pool = sorted(set(rows["label_text"])) if family.startswith("massive") else []
|
| 212 |
+
output = []
|
| 213 |
+
for i, r in enumerate(rows):
|
| 214 |
+
if i >= n:
|
| 215 |
+
break
|
| 216 |
+
if family == "xnli":
|
| 217 |
+
output.append(
|
| 218 |
+
single(
|
| 219 |
+
name,
|
| 220 |
+
i,
|
| 221 |
+
{"premise": r["premise"], "hypothesis": r["hypothesis"]},
|
| 222 |
+
"relation",
|
| 223 |
+
question(
|
| 224 |
+
"choice",
|
| 225 |
+
"What is the relationship between `premise` and `hypothesis`?",
|
| 226 |
+
NLI_CRITERIA,
|
| 227 |
+
),
|
| 228 |
+
r["label"],
|
| 229 |
+
language=language,
|
| 230 |
+
)
|
| 231 |
+
)
|
| 232 |
+
else:
|
| 233 |
+
gold = r["label_text"]
|
| 234 |
+
distractors = [x for x in pool if x != gold]
|
| 235 |
+
options = [gold] + rng.sample(distractors, min(19, len(distractors)))
|
| 236 |
+
rng.shuffle(options)
|
| 237 |
+
instructions = (
|
| 238 |
+
"What is the user asking for in `utterance`?"
|
| 239 |
+
if family == "massive-intent"
|
| 240 |
+
else "Which domain does `utterance` belong to?"
|
| 241 |
+
)
|
| 242 |
+
q = question(
|
| 243 |
+
"choice", instructions, {k: k.replace("_", " ").replace(".", ": ") for k in options}
|
| 244 |
+
)
|
| 245 |
+
output.append(
|
| 246 |
+
single(
|
| 247 |
+
name,
|
| 248 |
+
i,
|
| 249 |
+
{"utterance": r["text"]},
|
| 250 |
+
"label",
|
| 251 |
+
q,
|
| 252 |
+
options.index(gold),
|
| 253 |
+
language=language,
|
| 254 |
+
)
|
| 255 |
+
)
|
| 256 |
+
return output
|
| 257 |
+
|
| 258 |
+
|
| 259 |
+
def balanced_indices(targets, limit, seed):
|
| 260 |
+
groups = defaultdict(list)
|
| 261 |
+
for i, target in enumerate(targets):
|
| 262 |
+
groups[target].append(i)
|
| 263 |
+
rng = random.Random(seed)
|
| 264 |
+
for indices in groups.values():
|
| 265 |
+
rng.shuffle(indices)
|
| 266 |
+
chosen = []
|
| 267 |
+
while len(chosen) < min(limit, len(targets)):
|
| 268 |
+
for cls in sorted(groups):
|
| 269 |
+
if groups[cls] and len(chosen) < limit:
|
| 270 |
+
chosen.append(groups[cls].pop())
|
| 271 |
+
return sorted(chosen)
|
| 272 |
+
|
| 273 |
+
|
| 274 |
+
def build_btzsc(name, rows, n=100):
|
| 275 |
+
task = name.split(".", 1)[1]
|
| 276 |
+
texts = list(rows["text"])
|
| 277 |
+
k = next(i for i in range(1, len(texts)) if texts[i] != texts[0])
|
| 278 |
+
options = [str(rows[i]["hypothesis"]) for i in range(k)]
|
| 279 |
+
if len(rows) % k:
|
| 280 |
+
raise ValueError("incomplete BTZSC candidate group")
|
| 281 |
+
valid, targets = [], []
|
| 282 |
+
for start in range(0, len(rows), k):
|
| 283 |
+
group = rows[start : start + k]
|
| 284 |
+
if len(set(group["text"])) != 1 or list(group["hypothesis"]) != options:
|
| 285 |
+
raise ValueError("BTZSC candidate order changed")
|
| 286 |
+
y = [int(v) for v in group["labels"]]
|
| 287 |
+
if sum(y) == 1:
|
| 288 |
+
valid.append(start)
|
| 289 |
+
targets.append(y.index(1))
|
| 290 |
+
offset = ["agnews", "emotiondair", "banking77"].index(task)
|
| 291 |
+
q = question(
|
| 292 |
+
"choice",
|
| 293 |
+
"Which single label best describes the input text?",
|
| 294 |
+
{f"label_{i:03d}": option for i, option in enumerate(options)},
|
| 295 |
+
)
|
| 296 |
+
return [
|
| 297 |
+
single(name, valid[pos] // k, {"text": texts[valid[pos]]}, "label", q, targets[pos])
|
| 298 |
+
for pos in balanced_indices(targets, n, 20260917 + offset)
|
| 299 |
+
]
|
| 300 |
+
|
| 301 |
+
|
| 302 |
+
def build_app(name, rows, n, seed, sources):
|
| 303 |
+
rng = random.Random(seed)
|
| 304 |
+
indexed = list(enumerate(rows))
|
| 305 |
+
if name == "phishing":
|
| 306 |
+
indexed = [
|
| 307 |
+
(i, r)
|
| 308 |
+
for i, r in indexed[:6000]
|
| 309 |
+
if (r.get("Email Text") or "").strip()
|
| 310 |
+
and r.get("Email Type") in ("Safe Email", "Phishing Email")
|
| 311 |
+
]
|
| 312 |
+
rng.shuffle(indexed)
|
| 313 |
+
if name in ("toxic-chat", "jailbreak"):
|
| 314 |
+
field = "toxicity" if name == "toxic-chat" else "jailbreaking"
|
| 315 |
+
indexed = [(i, r) for i, r in indexed if (r.get("user_input") or "").strip()]
|
| 316 |
+
positive = [(i, r) for i, r in indexed if int(r[field]) == 1][: n // 2]
|
| 317 |
+
indexed = positive + [(i, r) for i, r in indexed if int(r[field]) == 0][: n - len(positive)]
|
| 318 |
+
rng.shuffle(indexed)
|
| 319 |
+
if name == "model-routing":
|
| 320 |
+
domains = {
|
| 321 |
+
"code": "software engineering, programming, refactoring, architecture, debugging",
|
| 322 |
+
"math_or_logic": "mathematics, logic puzzles, proofs, complex calculation",
|
| 323 |
+
"writing": "creative writing, essays, emails, blog posts, copywriting",
|
| 324 |
+
"factual_lookup": "facts, definitions, trivia, history",
|
| 325 |
+
"data_analysis": "statistics, SQL, data manipulation, metrics",
|
| 326 |
+
"chitchat": "casual conversation, greetings, small talk",
|
| 327 |
+
}
|
| 328 |
+
code = sources.load("google-research-datasets/mbpp", "full")
|
| 329 |
+
news = sources.load("fancyzhx/ag_news")
|
| 330 |
+
pool = [(r["question"], "math_or_logic") for r in list(rows)[: n // 3]]
|
| 331 |
+
pool += [(r["text"], "code") for r in list(code)[: n // 3]]
|
| 332 |
+
pool += [(r["text"][:400], "factual_lookup") for r in list(news)[: n // 3]]
|
| 333 |
+
rng.shuffle(pool)
|
| 334 |
+
q = question("choice", "What domain does `request` belong to?", domains)
|
| 335 |
+
return [
|
| 336 |
+
single(name, i, {"request": text}, "domain", q, list(domains).index(label))
|
| 337 |
+
for i, (text, label) in enumerate(pool)
|
| 338 |
+
]
|
| 339 |
+
output = []
|
| 340 |
+
for i, r in indexed:
|
| 341 |
+
if len(output) >= n:
|
| 342 |
+
break
|
| 343 |
+
if name == "enron-spam":
|
| 344 |
+
state = {"subject": r.get("subject") or "", "body": (r.get("message") or "")[:3000]}
|
| 345 |
+
qid, q, target = (
|
| 346 |
+
"is_spam",
|
| 347 |
+
question("noul", "Is this email unsolicited spam or bulk marketing?"),
|
| 348 |
+
r["label"],
|
| 349 |
+
)
|
| 350 |
+
elif name == "phishing":
|
| 351 |
+
state = {"email": r["Email Text"][:3000]}
|
| 352 |
+
qid, q, target = (
|
| 353 |
+
"is_phishing",
|
| 354 |
+
question(
|
| 355 |
+
"noul",
|
| 356 |
+
"Is this email a phishing or scam attempt to steal money, credentials, or personal data?",
|
| 357 |
+
{
|
| 358 |
+
"true": "phishing, scam, or fraud",
|
| 359 |
+
"false": "a legitimate email (even if promotional)",
|
| 360 |
+
},
|
| 361 |
+
),
|
| 362 |
+
int(r["Email Type"] == "Phishing Email"),
|
| 363 |
+
)
|
| 364 |
+
elif name in ("toxic-chat", "jailbreak"):
|
| 365 |
+
toxic = name == "toxic-chat"
|
| 366 |
+
state = {"post" if toxic else "prompt": r["user_input"][:3000]}
|
| 367 |
+
qid = "toxic" if toxic else "jailbreak"
|
| 368 |
+
q = question(
|
| 369 |
+
"noul",
|
| 370 |
+
"Is `post` toxic: rude, disrespectful or likely to make someone leave the discussion?"
|
| 371 |
+
if toxic
|
| 372 |
+
else "Does `prompt` try to make an AI assistant ignore its rules, policies or system instructions?",
|
| 373 |
+
)
|
| 374 |
+
target = int(r["toxicity" if toxic else "jailbreaking"])
|
| 375 |
+
elif name == "support-triage":
|
| 376 |
+
if r.get("language") != "en" or r.get("queue") not in QUEUES or not r.get("body"):
|
| 377 |
+
continue
|
| 378 |
+
state = {"subject": r["subject"] or "", "body": r["body"].replace("\\n", "\n")[:3000]}
|
| 379 |
+
qid, q, target = (
|
| 380 |
+
"queue",
|
| 381 |
+
question("choice", "Which support queue should handle this ticket?", QUEUES),
|
| 382 |
+
list(QUEUES).index(r["queue"]),
|
| 383 |
+
)
|
| 384 |
+
else:
|
| 385 |
+
passages = r["passages"]
|
| 386 |
+
pos = [t for t, s in zip(passages["passage_text"], passages["is_selected"]) if s == 1]
|
| 387 |
+
neg = [t for t, s in zip(passages["passage_text"], passages["is_selected"]) if s == 0]
|
| 388 |
+
if not pos or not neg:
|
| 389 |
+
continue
|
| 390 |
+
target = int(len(output) % 2 == 0)
|
| 391 |
+
state = {"query": r["query"], "passage": rng.choice(pos if target else neg)}
|
| 392 |
+
qid, q = "relevant", question("noul", "Does `passage` help answer `query`?")
|
| 393 |
+
output.append(single(name, i, state, qid, q, target))
|
| 394 |
+
return output
|
| 395 |
+
|
| 396 |
+
|
| 397 |
+
def smoke():
|
| 398 |
+
return [
|
| 399 |
+
{
|
| 400 |
+
"id": "smoke:0",
|
| 401 |
+
"suite": "smoke",
|
| 402 |
+
"state": "The package is late. Please refund me.",
|
| 403 |
+
"questions": {
|
| 404 |
+
"refund": question("noul", "Is a refund requested?"),
|
| 405 |
+
"department": question(
|
| 406 |
+
"choice",
|
| 407 |
+
"Which team should handle this?",
|
| 408 |
+
{"billing": "refunds", "technical": "bugs"},
|
| 409 |
+
),
|
| 410 |
+
"urgency": question(
|
| 411 |
+
"score", "How urgent is the request?", ["low", "medium", "high"]
|
| 412 |
+
),
|
| 413 |
+
},
|
| 414 |
+
"gold": {
|
| 415 |
+
"refund": {"label": "true"},
|
| 416 |
+
"department": {"label": "billing"},
|
| 417 |
+
"urgency": {"label": "1", "score": 1.0},
|
| 418 |
+
},
|
| 419 |
+
}
|
| 420 |
+
]
|
| 421 |
+
|
| 422 |
+
|
| 423 |
+
def latency():
|
| 424 |
+
result = []
|
| 425 |
+
for n in (1, 5, 10, 50):
|
| 426 |
+
for i in range(30):
|
| 427 |
+
base = smoke()[0]
|
| 428 |
+
result.append(
|
| 429 |
+
{
|
| 430 |
+
"id": f"latency:{n}:{i}",
|
| 431 |
+
"suite": f"latency-{n}",
|
| 432 |
+
"state": base["state"],
|
| 433 |
+
"questions": {f"q{j}": base["questions"]["department"] for j in range(n)},
|
| 434 |
+
"gold": {f"q{j}": {"label": "billing"} for j in range(n)},
|
| 435 |
+
}
|
| 436 |
+
)
|
| 437 |
+
return result
|
| 438 |
+
|
| 439 |
+
|
| 440 |
+
def permute(cases, count, seed):
|
| 441 |
+
result = list(cases)
|
| 442 |
+
rng = random.Random(seed)
|
| 443 |
+
for case in cases:
|
| 444 |
+
if not any(q["type"] == "choice" for q in case["questions"].values()):
|
| 445 |
+
continue
|
| 446 |
+
for variant in range(1, count + 1):
|
| 447 |
+
c = copy.deepcopy(case)
|
| 448 |
+
c.update(
|
| 449 |
+
id=f"{case['id']}:permutation:{variant}", base_id=case["id"], permutation=variant
|
| 450 |
+
)
|
| 451 |
+
for q in c["questions"].values():
|
| 452 |
+
if q["type"] == "choice":
|
| 453 |
+
keys = list(q["criteria"])
|
| 454 |
+
rng.shuffle(keys)
|
| 455 |
+
q["criteria"] = {k: q["criteria"][k] for k in keys}
|
| 456 |
+
result.append(c)
|
| 457 |
+
return result
|
| 458 |
+
|
| 459 |
+
|
| 460 |
+
def prepare(profile, suites, output, lock_path, cache, limit=None, seed=13, permutations=0):
|
| 461 |
+
sources = (
|
| 462 |
+
Sources(json.loads(Path(lock_path).read_text()), cache)
|
| 463 |
+
if profile not in ("smoke", "latency")
|
| 464 |
+
else None
|
| 465 |
+
)
|
| 466 |
+
if profile == "smoke":
|
| 467 |
+
cases = smoke()
|
| 468 |
+
elif profile == "latency":
|
| 469 |
+
cases = latency()
|
| 470 |
+
else:
|
| 471 |
+
if suites:
|
| 472 |
+
names = suites.split(",")
|
| 473 |
+
elif profile == "laya-core":
|
| 474 |
+
names = CORE
|
| 475 |
+
elif profile == "laya-apps":
|
| 476 |
+
names = APPS
|
| 477 |
+
elif profile == "jev-btzsc":
|
| 478 |
+
names = ["btzsc.agnews", "btzsc.emotiondair", "btzsc.banking77"]
|
| 479 |
+
elif profile == "laya-multilingual":
|
| 480 |
+
names = [
|
| 481 |
+
f"{family}.{lang}"
|
| 482 |
+
for family in ("massive-intent", "massive-scenario")
|
| 483 |
+
for lang in MASSIVE_LANGS
|
| 484 |
+
]
|
| 485 |
+
names += [f"xnli.{lang}" for lang in XNLI_LANGS]
|
| 486 |
+
else:
|
| 487 |
+
raise ValueError(f"unknown profile: {profile}")
|
| 488 |
+
if len(set(names)) != len(names):
|
| 489 |
+
raise ValueError("duplicate suites")
|
| 490 |
+
cases = []
|
| 491 |
+
for name in names:
|
| 492 |
+
print(f"Preparing {name} ...", flush=True)
|
| 493 |
+
if name.startswith("btzsc."):
|
| 494 |
+
rows = sources.load("btzsc/btzsc", name.split(".", 1)[1])
|
| 495 |
+
built = build_btzsc(name, rows, limit or 100)
|
| 496 |
+
elif name.startswith(("massive-intent.", "massive-scenario.", "xnli.")):
|
| 497 |
+
family, lang = name.split(".", 1)
|
| 498 |
+
repo = {
|
| 499 |
+
"massive-intent": "mteb/amazon_massive_intent",
|
| 500 |
+
"massive-scenario": "mteb/amazon_massive_scenario",
|
| 501 |
+
"xnli": "facebook/xnli",
|
| 502 |
+
}[family]
|
| 503 |
+
rows = sources.load(repo, lang)
|
| 504 |
+
built = build_multilingual(name, rows, limit or 300, seed)
|
| 505 |
+
else:
|
| 506 |
+
if name not in SPECS:
|
| 507 |
+
raise ValueError(f"unknown suite: {name}")
|
| 508 |
+
repo, config, split, default_n = SPECS[name]
|
| 509 |
+
rows = sources.load(repo, config, split)
|
| 510 |
+
n = limit or (400 if profile == "laya-apps" else default_n) or len(rows)
|
| 511 |
+
if name in CORE:
|
| 512 |
+
# Build label inventories on the full split before taking the prefix.
|
| 513 |
+
built = build_core(name, rows)[:n]
|
| 514 |
+
else:
|
| 515 |
+
built = build_app(name, rows, n, seed, sources)
|
| 516 |
+
if not built:
|
| 517 |
+
raise ValueError(f"suite produced no cases: {name}")
|
| 518 |
+
cases.extend(built)
|
| 519 |
+
cases = permute(cases, permutations, seed)
|
| 520 |
+
return save_bundle(
|
| 521 |
+
output,
|
| 522 |
+
cases,
|
| 523 |
+
{
|
| 524 |
+
"profile": profile,
|
| 525 |
+
"seed": seed,
|
| 526 |
+
"limit": limit,
|
| 527 |
+
"permutations": permutations,
|
| 528 |
+
"sources": sources.used if sources else {},
|
| 529 |
+
"prompt_source": f"https://github.com/NandhaKishorM/laya/tree/{LAYA_REVISION}/research",
|
| 530 |
+
"protocol_note": "See docs/BENCHMARKS.md for differences from historical runs; freeze and reuse this bundle.",
|
| 531 |
+
},
|
| 532 |
+
)
|
ariadne_bench/experiments/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
"""Experimental training entry points, separate from benchmark evaluation."""
|
ariadne_bench/experiments/align.py
ADDED
|
@@ -0,0 +1,336 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Train the hybrid connection using a separate case-level validation split."""
|
| 2 |
+
|
| 3 |
+
import argparse
|
| 4 |
+
import copy
|
| 5 |
+
import itertools
|
| 6 |
+
import json
|
| 7 |
+
import math
|
| 8 |
+
import random
|
| 9 |
+
import shutil
|
| 10 |
+
import time
|
| 11 |
+
from pathlib import Path
|
| 12 |
+
|
| 13 |
+
import torch
|
| 14 |
+
|
| 15 |
+
from ariadne_bench.hybrid import collate
|
| 16 |
+
from ariadne_bench.adapters import create_adapter
|
| 17 |
+
from ariadne_bench.schema import distribution, load_bundle, write_json
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
def examples(adapter, cases):
|
| 21 |
+
result = []
|
| 22 |
+
for case in cases:
|
| 23 |
+
items = adapter.prepare(case["state"], case["questions"])
|
| 24 |
+
for (qid, _), item in zip(case["questions"].items(), items):
|
| 25 |
+
gold = case["gold"][qid]
|
| 26 |
+
keys = item["keys"]
|
| 27 |
+
item["target"] = (
|
| 28 |
+
distribution(gold["probabilities"], keys)[0]
|
| 29 |
+
if "probabilities" in gold
|
| 30 |
+
else [float(k == gold["label"]) for k in keys]
|
| 31 |
+
)
|
| 32 |
+
result.append(item)
|
| 33 |
+
return result
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def forward_loss(adapter, items):
|
| 37 |
+
batch = collate(items, adapter.pad_id, adapter.device)
|
| 38 |
+
target = torch.zeros(batch["marker_mask"].shape, device=adapter.device)
|
| 39 |
+
for row, item in enumerate(items):
|
| 40 |
+
target[row, : len(item["target"])] = torch.tensor(item["target"], device=adapter.device)
|
| 41 |
+
with adapter.autocast():
|
| 42 |
+
logits, _ = adapter.model(**batch)
|
| 43 |
+
loss = -(target * logits.log_softmax(-1)).sum(-1).mean()
|
| 44 |
+
return loss, logits, target
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
@torch.no_grad()
|
| 48 |
+
def evaluate(adapter, items, batch_size):
|
| 49 |
+
adapter.model.eval()
|
| 50 |
+
total_loss, correct, brier, count = 0.0, 0, 0.0, 0
|
| 51 |
+
for start in range(0, len(items), batch_size):
|
| 52 |
+
chunk = items[start : start + batch_size]
|
| 53 |
+
loss, logits, target = forward_loss(adapter, chunk)
|
| 54 |
+
total_loss += loss.item() * len(chunk)
|
| 55 |
+
correct += (logits.argmax(-1) == target.argmax(-1)).sum().item()
|
| 56 |
+
brier += ((logits.softmax(-1) - target) ** 2).sum().item()
|
| 57 |
+
count += len(chunk)
|
| 58 |
+
return {
|
| 59 |
+
"soft_cross_entropy": total_loss / count,
|
| 60 |
+
"accuracy": correct / count,
|
| 61 |
+
"brier_soft": brier / count,
|
| 62 |
+
"decisions": count,
|
| 63 |
+
}
|
| 64 |
+
|
| 65 |
+
|
| 66 |
+
class ValidationPatience:
|
| 67 |
+
"""Count consecutive epochs that fail to beat the best validation loss."""
|
| 68 |
+
|
| 69 |
+
def __init__(self, patience=None, min_epochs=10):
|
| 70 |
+
if patience is not None and patience < 1:
|
| 71 |
+
raise ValueError("patience must be positive")
|
| 72 |
+
if min_epochs < 1:
|
| 73 |
+
raise ValueError("minimum epochs must be positive")
|
| 74 |
+
self.patience = patience
|
| 75 |
+
self.min_epochs = min_epochs
|
| 76 |
+
self.epochs = 0
|
| 77 |
+
self.best = float("inf")
|
| 78 |
+
self.stale = 0
|
| 79 |
+
|
| 80 |
+
def observe(self, loss):
|
| 81 |
+
if not math.isfinite(loss):
|
| 82 |
+
raise ValueError("validation loss must be finite")
|
| 83 |
+
self.epochs += 1
|
| 84 |
+
improved = loss < self.best
|
| 85 |
+
if improved:
|
| 86 |
+
self.best = loss
|
| 87 |
+
self.stale = 0
|
| 88 |
+
else:
|
| 89 |
+
self.stale += 1
|
| 90 |
+
return improved, (
|
| 91 |
+
self.patience is not None
|
| 92 |
+
and self.stale >= self.patience
|
| 93 |
+
and self.epochs >= self.min_epochs
|
| 94 |
+
)
|
| 95 |
+
|
| 96 |
+
|
| 97 |
+
def main():
|
| 98 |
+
p = argparse.ArgumentParser(description=__doc__)
|
| 99 |
+
p.add_argument("--config", required=True)
|
| 100 |
+
p.add_argument("--train", required=True)
|
| 101 |
+
p.add_argument("--validation", required=True)
|
| 102 |
+
p.add_argument("--out", required=True)
|
| 103 |
+
p.add_argument(
|
| 104 |
+
"--epochs", type=int, default=3, help="epoch limit; 0 means unlimited with early stopping"
|
| 105 |
+
)
|
| 106 |
+
p.add_argument(
|
| 107 |
+
"--early-stopping-patience",
|
| 108 |
+
type=int,
|
| 109 |
+
help="stop after this many consecutive epochs without a new lowest validation loss",
|
| 110 |
+
)
|
| 111 |
+
p.add_argument(
|
| 112 |
+
"--min-epochs",
|
| 113 |
+
type=int,
|
| 114 |
+
default=10,
|
| 115 |
+
help="minimum completed epochs before early stopping can fire",
|
| 116 |
+
)
|
| 117 |
+
p.add_argument("--batch-size", type=int, default=8)
|
| 118 |
+
p.add_argument("--accumulation", type=int, default=4)
|
| 119 |
+
p.add_argument("--lr", type=float, default=3e-4)
|
| 120 |
+
p.add_argument("--device", help="override config device, e.g. cuda:1")
|
| 121 |
+
p.add_argument("--head-lr", type=float, help="full fine-tuning head learning rate")
|
| 122 |
+
p.add_argument(
|
| 123 |
+
"--keep-best-only",
|
| 124 |
+
action="store_true",
|
| 125 |
+
help="remove superseded checkpoints created by this run",
|
| 126 |
+
)
|
| 127 |
+
args = p.parse_args()
|
| 128 |
+
if (
|
| 129 |
+
min(args.batch_size, args.accumulation) < 1
|
| 130 |
+
or args.epochs < 0
|
| 131 |
+
or args.min_epochs < 1
|
| 132 |
+
or (args.early_stopping_patience is not None and 0 < args.epochs < args.min_epochs)
|
| 133 |
+
or (args.epochs == 0 and args.early_stopping_patience is None)
|
| 134 |
+
or (args.early_stopping_patience is not None and args.early_stopping_patience < 1)
|
| 135 |
+
or not math.isfinite(args.lr)
|
| 136 |
+
or args.lr <= 0
|
| 137 |
+
):
|
| 138 |
+
p.error(
|
| 139 |
+
"use positive batch/accumulation/lr/patience and a positive epoch limit, or epochs=0 with patience"
|
| 140 |
+
)
|
| 141 |
+
train, train_manifest = load_bundle(args.train)
|
| 142 |
+
validation, validation_manifest = load_bundle(args.validation)
|
| 143 |
+
if train_manifest.get("role") != "train" or validation_manifest.get("role") != "validation":
|
| 144 |
+
raise ValueError(
|
| 145 |
+
"explicit train and validation bundle roles are required; do not use benchmark test data"
|
| 146 |
+
)
|
| 147 |
+
if {c["id"] for c in train} & {c["id"] for c in validation}:
|
| 148 |
+
raise ValueError("train and validation case IDs overlap")
|
| 149 |
+
config = json.loads(Path(args.config).read_text())
|
| 150 |
+
if args.device:
|
| 151 |
+
config["options"]["device"] = args.device
|
| 152 |
+
scope = config["options"].get("training_scope", "bridge")
|
| 153 |
+
if scope not in ("bridge", "full"):
|
| 154 |
+
raise ValueError("supported training scopes are bridge and full")
|
| 155 |
+
if args.head_lr is not None and (not math.isfinite(args.head_lr) or args.head_lr <= 0):
|
| 156 |
+
p.error("head learning rate must be positive")
|
| 157 |
+
out = Path(args.out)
|
| 158 |
+
out.mkdir(parents=True, exist_ok=False)
|
| 159 |
+
adapter = create_adapter(config)
|
| 160 |
+
train_items, validation_items = examples(adapter, train), examples(adapter, validation)
|
| 161 |
+
parameters = [p for p in adapter.model.parameters() if p.requires_grad]
|
| 162 |
+
frozen_versions = {
|
| 163 |
+
name: value._version
|
| 164 |
+
for name, value in adapter.model.named_parameters()
|
| 165 |
+
if not value.requires_grad
|
| 166 |
+
}
|
| 167 |
+
if scope == "full":
|
| 168 |
+
optimizer_parameters = [
|
| 169 |
+
{
|
| 170 |
+
"params": [
|
| 171 |
+
p
|
| 172 |
+
for n, p in adapter.model.named_parameters()
|
| 173 |
+
if p.requires_grad and n.startswith("encoder.")
|
| 174 |
+
],
|
| 175 |
+
"lr": args.lr,
|
| 176 |
+
},
|
| 177 |
+
{
|
| 178 |
+
"params": [
|
| 179 |
+
p
|
| 180 |
+
for n, p in adapter.model.named_parameters()
|
| 181 |
+
if p.requires_grad and not n.startswith("encoder.")
|
| 182 |
+
],
|
| 183 |
+
"lr": args.head_lr or args.lr,
|
| 184 |
+
},
|
| 185 |
+
]
|
| 186 |
+
else:
|
| 187 |
+
optimizer_parameters = parameters
|
| 188 |
+
optimizer = torch.optim.AdamW(optimizer_parameters, lr=args.lr, weight_decay=0.01)
|
| 189 |
+
rng = random.Random(config["options"].get("seed", 42))
|
| 190 |
+
log = {
|
| 191 |
+
"configuration": vars(args),
|
| 192 |
+
"model": adapter.metadata,
|
| 193 |
+
"train_manifest": train_manifest,
|
| 194 |
+
"validation_manifest": validation_manifest,
|
| 195 |
+
"trainable_names": [n for n, p in adapter.model.named_parameters() if p.requires_grad],
|
| 196 |
+
"initial_validation": evaluate(adapter, validation_items, args.batch_size),
|
| 197 |
+
"epochs": [],
|
| 198 |
+
}
|
| 199 |
+
write_json(out / "training.json", log)
|
| 200 |
+
print(
|
| 201 |
+
json.dumps(
|
| 202 |
+
{
|
| 203 |
+
"initial_validation": log["initial_validation"],
|
| 204 |
+
"trainable_parameters": sum(p.numel() for p in parameters),
|
| 205 |
+
}
|
| 206 |
+
),
|
| 207 |
+
flush=True,
|
| 208 |
+
)
|
| 209 |
+
start_time = time.perf_counter()
|
| 210 |
+
best_checkpoint = None
|
| 211 |
+
stopping = ValidationPatience(args.early_stopping_patience, args.min_epochs)
|
| 212 |
+
epochs = range(1, args.epochs + 1) if args.epochs else itertools.count(1)
|
| 213 |
+
for epoch in epochs:
|
| 214 |
+
rng.shuffle(train_items)
|
| 215 |
+
adapter.training_mode()
|
| 216 |
+
total_loss, completed = 0.0, 0
|
| 217 |
+
effective_batch = args.batch_size * args.accumulation
|
| 218 |
+
for start in range(0, len(train_items), effective_batch):
|
| 219 |
+
group = train_items[start : start + effective_batch]
|
| 220 |
+
optimizer.zero_grad(set_to_none=True)
|
| 221 |
+
for micro in range(0, len(group), args.batch_size):
|
| 222 |
+
chunk = group[micro : micro + args.batch_size]
|
| 223 |
+
loss, _, _ = forward_loss(adapter, chunk)
|
| 224 |
+
if not torch.isfinite(loss):
|
| 225 |
+
raise RuntimeError("nonfinite alignment loss")
|
| 226 |
+
(loss * (len(chunk) / len(group))).backward()
|
| 227 |
+
total_loss += loss.item() * len(chunk)
|
| 228 |
+
completed += len(chunk)
|
| 229 |
+
norm = torch.nn.utils.clip_grad_norm_(parameters, 1.0, error_if_nonfinite=True)
|
| 230 |
+
optimizer.step()
|
| 231 |
+
if start == 0 or (start // effective_batch + 1) % 20 == 0:
|
| 232 |
+
adapter.synchronize()
|
| 233 |
+
elapsed = time.perf_counter() - start_time
|
| 234 |
+
processed = (epoch - 1) * len(train_items) + completed
|
| 235 |
+
remaining = (
|
| 236 |
+
((args.epochs * len(train_items) - processed) * elapsed / processed)
|
| 237 |
+
if args.epochs
|
| 238 |
+
else None
|
| 239 |
+
)
|
| 240 |
+
print(
|
| 241 |
+
json.dumps(
|
| 242 |
+
{
|
| 243 |
+
"epoch": epoch,
|
| 244 |
+
"decisions": completed,
|
| 245 |
+
"train_ce": total_loss / completed,
|
| 246 |
+
"gradient_norm": norm.item(),
|
| 247 |
+
"elapsed_s": elapsed,
|
| 248 |
+
"estimated_remaining_s": remaining,
|
| 249 |
+
"peak_gpu_allocated_gb": torch.cuda.max_memory_allocated(adapter.device)
|
| 250 |
+
/ 1e9
|
| 251 |
+
if adapter.device.type == "cuda"
|
| 252 |
+
else None,
|
| 253 |
+
}
|
| 254 |
+
),
|
| 255 |
+
flush=True,
|
| 256 |
+
)
|
| 257 |
+
stats = {
|
| 258 |
+
"epoch": epoch,
|
| 259 |
+
"train_soft_cross_entropy": total_loss / completed,
|
| 260 |
+
"validation": evaluate(adapter, validation_items, args.batch_size),
|
| 261 |
+
}
|
| 262 |
+
log["epochs"].append(stats)
|
| 263 |
+
improved, should_stop = stopping.observe(stats["validation"]["soft_cross_entropy"])
|
| 264 |
+
if improved:
|
| 265 |
+
previous_checkpoint = best_checkpoint
|
| 266 |
+
best_checkpoint = out / f"epoch-{epoch}"
|
| 267 |
+
adapter.save_checkpoint(
|
| 268 |
+
best_checkpoint,
|
| 269 |
+
{
|
| 270 |
+
"scope": "full decision path; unused auxiliary action head frozen"
|
| 271 |
+
if scope == "full"
|
| 272 |
+
else "interface only; pretrained source parameters frozen",
|
| 273 |
+
"epoch": epoch,
|
| 274 |
+
"train_sha256": train_manifest["sha256"],
|
| 275 |
+
"validation_sha256": validation_manifest["sha256"],
|
| 276 |
+
"selection": "lowest validation soft cross entropy",
|
| 277 |
+
"validation": stats["validation"],
|
| 278 |
+
},
|
| 279 |
+
)
|
| 280 |
+
if args.keep_best_only and previous_checkpoint is not None:
|
| 281 |
+
# Only delete an earlier checkpoint created within this new run.
|
| 282 |
+
if previous_checkpoint.parent != out or not previous_checkpoint.name.startswith(
|
| 283 |
+
"epoch-"
|
| 284 |
+
):
|
| 285 |
+
raise ValueError("refusing to remove a checkpoint outside this run")
|
| 286 |
+
shutil.rmtree(previous_checkpoint)
|
| 287 |
+
if args.early_stopping_patience is not None:
|
| 288 |
+
stats["early_stopping"] = {
|
| 289 |
+
"patience": stopping.patience,
|
| 290 |
+
"min_epochs": stopping.min_epochs,
|
| 291 |
+
"epochs_without_improvement": stopping.stale,
|
| 292 |
+
"best_validation_loss": stopping.best,
|
| 293 |
+
"improved": improved,
|
| 294 |
+
}
|
| 295 |
+
write_json(out / "training.json", log)
|
| 296 |
+
print(json.dumps(stats), flush=True)
|
| 297 |
+
if should_stop:
|
| 298 |
+
break
|
| 299 |
+
log["stopping"] = {
|
| 300 |
+
"reason": "early_stopping" if should_stop else "epoch_limit",
|
| 301 |
+
"epochs_completed": len(log["epochs"]),
|
| 302 |
+
"patience": stopping.patience,
|
| 303 |
+
"min_epochs": stopping.min_epochs,
|
| 304 |
+
"epochs_without_improvement": stopping.stale,
|
| 305 |
+
"metric": "validation.soft_cross_entropy",
|
| 306 |
+
"min_delta": 0.0,
|
| 307 |
+
}
|
| 308 |
+
assert all(
|
| 309 |
+
value._version == frozen_versions[name] and value.grad is None
|
| 310 |
+
for name, value in adapter.model.named_parameters()
|
| 311 |
+
if name in frozen_versions
|
| 312 |
+
)
|
| 313 |
+
log.update(
|
| 314 |
+
{
|
| 315 |
+
"elapsed_s": time.perf_counter() - start_time,
|
| 316 |
+
"frozen_parameters_unchanged": True,
|
| 317 |
+
"best_checkpoint": str(best_checkpoint.resolve()),
|
| 318 |
+
}
|
| 319 |
+
)
|
| 320 |
+
write_json(out / "training.json", log)
|
| 321 |
+
trained_config = copy.deepcopy(config)
|
| 322 |
+
trained_config["name"] = config["name"].replace("(UNTRAINED)", "(alignment only)")
|
| 323 |
+
if scope == "full":
|
| 324 |
+
trained_config["name"] = config["name"].replace("(UNTRAINED)", "(full fine-tuned)")
|
| 325 |
+
trained_config["options"]["inference_only"] = True
|
| 326 |
+
trained_config["mode"] = "specialist"
|
| 327 |
+
trained_config["options"]["checkpoint"] = str(best_checkpoint.resolve())
|
| 328 |
+
write_json(out / "benchmark-config.json", trained_config)
|
| 329 |
+
print(
|
| 330 |
+
json.dumps({"best_checkpoint": str(best_checkpoint), "elapsed_s": log["elapsed_s"]}),
|
| 331 |
+
flush=True,
|
| 332 |
+
)
|
| 333 |
+
|
| 334 |
+
|
| 335 |
+
if __name__ == "__main__":
|
| 336 |
+
main()
|
ariadne_bench/experiments/prepare_alignment.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Freeze disjoint train/validation/test bundles for the interface experiment."""
|
| 2 |
+
|
| 3 |
+
import argparse
|
| 4 |
+
import random
|
| 5 |
+
from pathlib import Path
|
| 6 |
+
|
| 7 |
+
from ariadne_bench.datasets import typed
|
| 8 |
+
from ariadne_bench.schema import save_bundle
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
def main():
|
| 12 |
+
from datasets import load_dataset
|
| 13 |
+
|
| 14 |
+
parser = argparse.ArgumentParser(description=__doc__)
|
| 15 |
+
parser.add_argument("--out", required=True)
|
| 16 |
+
parser.add_argument("--cache", default=".cache/datasets")
|
| 17 |
+
args = parser.parse_args()
|
| 18 |
+
out = Path(args.out)
|
| 19 |
+
if out.exists():
|
| 20 |
+
raise ValueError("choose a new output directory")
|
| 21 |
+
revision = "f7a2487edd7a043a5441a5e9ccc7fe5ddbd9ebe8"
|
| 22 |
+
ds = load_dataset("LocalLLaMA/typed-decisions", "all", revision=revision, cache_dir=args.cache)
|
| 23 |
+
train, test = typed(ds["train"]), typed(ds["test"])
|
| 24 |
+
if {c["id"] for c in train} & {c["id"] for c in test}:
|
| 25 |
+
raise ValueError("source training and test IDs overlap")
|
| 26 |
+
random.Random(42).shuffle(train)
|
| 27 |
+
for role, rows in [("train", train[120:]), ("validation", train[:120]), ("test", test)]:
|
| 28 |
+
save_bundle(
|
| 29 |
+
out / role,
|
| 30 |
+
rows,
|
| 31 |
+
{
|
| 32 |
+
"profile": "hybrid-" + role,
|
| 33 |
+
"role": role,
|
| 34 |
+
"source": "LocalLLaMA/typed-decisions",
|
| 35 |
+
"revision": revision,
|
| 36 |
+
"source_split": "test" if role == "test" else "train",
|
| 37 |
+
"split_seed": 42,
|
| 38 |
+
},
|
| 39 |
+
)
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
if __name__ == "__main__":
|
| 43 |
+
main()
|
ariadne_bench/experiments/spectrum.py
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Analyze both the complete learned linear map and its departure from identity."""
|
| 2 |
+
|
| 3 |
+
import argparse
|
| 4 |
+
import json
|
| 5 |
+
from pathlib import Path
|
| 6 |
+
|
| 7 |
+
import torch
|
| 8 |
+
from safetensors.torch import load_file, save_file
|
| 9 |
+
|
| 10 |
+
from ariadne_bench.schema import write_json
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
def describe(matrix):
|
| 14 |
+
u, s, vh = torch.linalg.svd(matrix.double(), full_matrices=False)
|
| 15 |
+
energy = s.square()
|
| 16 |
+
cumulative = energy.cumsum(0) / energy.sum()
|
| 17 |
+
probability = s / s.sum()
|
| 18 |
+
entropy = -(probability * probability.clamp_min(1e-300).log()).sum()
|
| 19 |
+
eigen = torch.linalg.eigvals(matrix.double())
|
| 20 |
+
result = {
|
| 21 |
+
"frobenius_norm": matrix.double().norm().item(),
|
| 22 |
+
"singular_values": s.tolist(),
|
| 23 |
+
"cumulative_squared_singular_value_fraction": cumulative.tolist(),
|
| 24 |
+
"singular_value_entropy_effective_rank": entropy.exp().item(),
|
| 25 |
+
"stable_rank": (energy.sum() / energy[0]).item(),
|
| 26 |
+
"numerical_rank": int(torch.linalg.matrix_rank(matrix.double())),
|
| 27 |
+
"condition_number": (s[0] / s[-1]).item(),
|
| 28 |
+
"ranks_for_energy": {
|
| 29 |
+
str(p): int(torch.searchsorted(cumulative, torch.tensor(p, dtype=cumulative.dtype))) + 1
|
| 30 |
+
for p in [0.5, 0.9, 0.95, 0.99]
|
| 31 |
+
},
|
| 32 |
+
"eigenvalues": [[z.real.item(), z.imag.item()] for z in eigen],
|
| 33 |
+
}
|
| 34 |
+
return result, {"u": u.float(), "s": s.float(), "vh": vh.float()}
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
def main():
|
| 38 |
+
parser = argparse.ArgumentParser(description=__doc__)
|
| 39 |
+
parser.add_argument("--checkpoint", default="runs/hybrid-laya-control/epoch-3")
|
| 40 |
+
parser.add_argument("--out", default="runs/rank-ablation/spectrum")
|
| 41 |
+
args = parser.parse_args()
|
| 42 |
+
source = Path(args.checkpoint)
|
| 43 |
+
out = Path(args.out)
|
| 44 |
+
out.mkdir(parents=True, exist_ok=False)
|
| 45 |
+
state = load_file(source / "adapter.safetensors")
|
| 46 |
+
weight = state["encoder.bridge.1.weight"]
|
| 47 |
+
identity = torch.eye(weight.shape[0], dtype=weight.dtype)
|
| 48 |
+
result = {
|
| 49 |
+
"source_manifest": json.loads((source / "manifest.json").read_text()),
|
| 50 |
+
"width": weight.shape[0],
|
| 51 |
+
"interpretation": "Singular energy describes the matrix, not data-weighted activation variance; W and W-I answer different questions.",
|
| 52 |
+
}
|
| 53 |
+
factors = {}
|
| 54 |
+
for name, matrix in [("weight", weight), ("delta", weight - identity)]:
|
| 55 |
+
result[name], values = describe(matrix)
|
| 56 |
+
factors.update({name + "." + k: v for k, v in values.items()})
|
| 57 |
+
result["relative_delta_frobenius"] = (weight - identity).norm().item() / identity.norm().item()
|
| 58 |
+
result["layernorm_scale_change_l2"] = (state["encoder.bridge.0.weight"] - 1).norm().item()
|
| 59 |
+
result["layernorm_bias_l2"] = state["encoder.bridge.0.bias"].norm().item()
|
| 60 |
+
result["linear_bias_l2"] = state["encoder.bridge.1.bias"].norm().item()
|
| 61 |
+
save_file({k:v.contiguous() for k,v in factors.items()}, out / "svd.safetensors")
|
| 62 |
+
write_json(out / "analysis.json", result)
|
| 63 |
+
print(
|
| 64 |
+
json.dumps(
|
| 65 |
+
{
|
| 66 |
+
name: {
|
| 67 |
+
k: v
|
| 68 |
+
for k, v in result[name].items()
|
| 69 |
+
if k
|
| 70 |
+
not in (
|
| 71 |
+
"singular_values",
|
| 72 |
+
"eigenvalues",
|
| 73 |
+
"cumulative_squared_singular_value_fraction",
|
| 74 |
+
)
|
| 75 |
+
}
|
| 76 |
+
for name in ("weight", "delta")
|
| 77 |
+
},
|
| 78 |
+
indent=2,
|
| 79 |
+
)
|
| 80 |
+
)
|
| 81 |
+
|
| 82 |
+
|
| 83 |
+
if __name__ == "__main__":
|
| 84 |
+
main()
|
ariadne_bench/frozen_input_interface.py
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Train only the exact-identity input affine map on frozen native Laya."""
|
| 2 |
+
|
| 3 |
+
import copy
|
| 4 |
+
import hashlib
|
| 5 |
+
|
| 6 |
+
from .full_input_interface import FullInputInterface
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
class FrozenInputInterface(FullInputInterface):
|
| 10 |
+
def __init__(self, options):
|
| 11 |
+
if not options.get("add_interface", True):
|
| 12 |
+
raise ValueError("the frozen-interface adapter requires its linear interface")
|
| 13 |
+
initial = copy.deepcopy(options)
|
| 14 |
+
initial.pop("checkpoint", None)
|
| 15 |
+
initial.pop("inference_only", None)
|
| 16 |
+
super().__init__(initial)
|
| 17 |
+
self.options = copy.deepcopy(options)
|
| 18 |
+
self.model.requires_grad_(False)
|
| 19 |
+
self.model.encoder.embeddings.interface.requires_grad_(True)
|
| 20 |
+
self.training_scope = "bridge"
|
| 21 |
+
self.identity["training_scope"] = "bridge"
|
| 22 |
+
self.metadata.update(
|
| 23 |
+
**self.identity,
|
| 24 |
+
trainable_parameters=sum(p.numel() for p in self.model.parameters() if p.requires_grad),
|
| 25 |
+
frozen_pretrained_model=True,
|
| 26 |
+
)
|
| 27 |
+
self.initial_frozen_sha256 = self.frozen_state_sha256()
|
| 28 |
+
self.metadata["frozen_state_sha256"] = self.initial_frozen_sha256
|
| 29 |
+
if options.get("checkpoint"):
|
| 30 |
+
self.load_checkpoint(options["checkpoint"])
|
| 31 |
+
if options.get("inference_only", False):
|
| 32 |
+
self.model.requires_grad_(False)
|
| 33 |
+
self.model.eval()
|
| 34 |
+
|
| 35 |
+
def training_mode(self):
|
| 36 |
+
# Autograd still traverses the frozen stack to compute interface gradients.
|
| 37 |
+
self.model.eval()
|
| 38 |
+
|
| 39 |
+
def checkpoint_tensor_names(self):
|
| 40 |
+
return {
|
| 41 |
+
name
|
| 42 |
+
for name in self.model.state_dict()
|
| 43 |
+
if name.startswith("encoder.embeddings.interface.")
|
| 44 |
+
}
|
| 45 |
+
|
| 46 |
+
def frozen_state_sha256(self):
|
| 47 |
+
digest = hashlib.sha256()
|
| 48 |
+
for name, value in sorted(self.model.state_dict().items()):
|
| 49 |
+
if name.startswith("encoder.embeddings.interface."):
|
| 50 |
+
continue
|
| 51 |
+
digest.update(name.encode())
|
| 52 |
+
digest.update(str((value.dtype, tuple(value.shape))).encode())
|
| 53 |
+
digest.update(value.detach().cpu().contiguous().numpy().tobytes())
|
| 54 |
+
return digest.hexdigest()
|
| 55 |
+
|
| 56 |
+
def save_checkpoint(self, directory, training):
|
| 57 |
+
actual = self.frozen_state_sha256()
|
| 58 |
+
if actual != self.initial_frozen_sha256:
|
| 59 |
+
raise RuntimeError("frozen pretrained tensors changed")
|
| 60 |
+
super().save_checkpoint(directory, {**training, "frozen_state_sha256": actual})
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
def create(options):
|
| 64 |
+
return FrozenInputInterface(options)
|
ariadne_bench/full_finetune.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Full task-model fine-tuning of the native Laya specialist."""
|
| 2 |
+
|
| 3 |
+
import copy
|
| 4 |
+
from pathlib import Path
|
| 5 |
+
|
| 6 |
+
import torch
|
| 7 |
+
|
| 8 |
+
from .hybrid import Hybrid
|
| 9 |
+
from .hybrid_control import LayaControl
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
class FullFineTune(Hybrid):
|
| 13 |
+
prepare = LayaControl.prepare
|
| 14 |
+
|
| 15 |
+
def __init__(self, options):
|
| 16 |
+
import laya
|
| 17 |
+
import transformers
|
| 18 |
+
|
| 19 |
+
self.options = copy.deepcopy(options)
|
| 20 |
+
if options.get("deterministic", False):
|
| 21 |
+
from .reproducibility import configure_determinism
|
| 22 |
+
|
| 23 |
+
configure_determinism(options.get("seed", 42))
|
| 24 |
+
torch.manual_seed(options.get("seed", 42))
|
| 25 |
+
self.device = torch.device(options.get("device", "cpu"))
|
| 26 |
+
if self.device.type == "cuda":
|
| 27 |
+
torch.cuda.set_device(self.device)
|
| 28 |
+
path = Path(options["laya_model"]).resolve(strict=True)
|
| 29 |
+
self.agent = laya.load(str(path), device=str(self.device))
|
| 30 |
+
if self.agent.device != self.device:
|
| 31 |
+
raise RuntimeError("native Laya loaded on an unexpected device")
|
| 32 |
+
self.model, self.tokenizer = self.agent.model, self.agent.tok
|
| 33 |
+
self.dtype = self.agent.dtype
|
| 34 |
+
self.pad_id = self.tokenizer.pad_token_id
|
| 35 |
+
self.max_len = int(options.get("max_len", self.agent.cfg.get("max_len", 512)))
|
| 36 |
+
self.head_max_len = self.agent.cfg.get("head_max_len", 192)
|
| 37 |
+
self.batch_size = int(options.get("batch_size", 16))
|
| 38 |
+
self.training_scope = "full"
|
| 39 |
+
self.native_prediction = True
|
| 40 |
+
self.model.requires_grad_(True)
|
| 41 |
+
# The supervised decision loss has no action/reward target. All parameters
|
| 42 |
+
# on the decision path train; the unused auxiliary action head stays frozen.
|
| 43 |
+
self.model.act_head.requires_grad_(False)
|
| 44 |
+
if options.get("gradient_checkpointing", False):
|
| 45 |
+
self.model.encoder.gradient_checkpointing_enable(
|
| 46 |
+
gradient_checkpointing_kwargs={"use_reentrant": False}
|
| 47 |
+
)
|
| 48 |
+
self.model.head_checkpointing = True
|
| 49 |
+
self.identity = {
|
| 50 |
+
"format_version": 1,
|
| 51 |
+
"laya_model": str(path),
|
| 52 |
+
"training_scope": "full",
|
| 53 |
+
"prompt_format": "original-laya-v1",
|
| 54 |
+
"max_len": self.max_len,
|
| 55 |
+
"head_max_len": self.head_max_len,
|
| 56 |
+
"native_prediction": True,
|
| 57 |
+
}
|
| 58 |
+
self.metadata = {
|
| 59 |
+
**self.identity,
|
| 60 |
+
"device": str(self.device),
|
| 61 |
+
"gpu_name": torch.cuda.get_device_name(self.device)
|
| 62 |
+
if self.device.type == "cuda"
|
| 63 |
+
else None,
|
| 64 |
+
"dtype": str(self.dtype),
|
| 65 |
+
"torch_version": torch.__version__,
|
| 66 |
+
"transformers_version": transformers.__version__,
|
| 67 |
+
"laya_version": laya.__version__,
|
| 68 |
+
"parameters": sum(p.numel() for p in self.model.parameters()),
|
| 69 |
+
"trainable_parameters": sum(
|
| 70 |
+
p.numel() for p in self.model.parameters() if p.requires_grad
|
| 71 |
+
),
|
| 72 |
+
"frozen_auxiliary_action_head": True,
|
| 73 |
+
"gradient_checkpointing": bool(options.get("gradient_checkpointing", False)),
|
| 74 |
+
"temperature": list(self.agent.temperature),
|
| 75 |
+
"temperature_by_options": dict(self.agent.temperature_by_options),
|
| 76 |
+
}
|
| 77 |
+
if options.get("deterministic", False):
|
| 78 |
+
from .reproducibility import settings
|
| 79 |
+
|
| 80 |
+
self.metadata["reproducibility"] = settings()
|
| 81 |
+
if options.get("checkpoint"):
|
| 82 |
+
self.load_checkpoint(options["checkpoint"])
|
| 83 |
+
if options.get("inference_only", False):
|
| 84 |
+
self.model.requires_grad_(False)
|
| 85 |
+
self.model.eval()
|
| 86 |
+
|
| 87 |
+
def close(self):
|
| 88 |
+
if hasattr(self, "agent"):
|
| 89 |
+
del self.agent
|
| 90 |
+
super().close()
|
| 91 |
+
|
| 92 |
+
def training_mode(self):
|
| 93 |
+
self.model.train()
|
| 94 |
+
|
| 95 |
+
def checkpoint_tensor_names(self):
|
| 96 |
+
return set(self.model.state_dict())
|
| 97 |
+
|
| 98 |
+
def predict(self, state, questions):
|
| 99 |
+
self.model.eval()
|
| 100 |
+
return self.agent.predict(state, questions, max_len=self.max_len)
|
| 101 |
+
|
| 102 |
+
|
| 103 |
+
def create(options):
|
| 104 |
+
return FullFineTune(options)
|
ariadne_bench/full_input_interface.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Jointly fine-tune native Laya and an identity-initialized input affine map."""
|
| 2 |
+
|
| 3 |
+
import copy
|
| 4 |
+
|
| 5 |
+
import torch
|
| 6 |
+
from torch import nn
|
| 7 |
+
|
| 8 |
+
from .full_finetune import FullFineTune
|
| 9 |
+
from .interfaces import IdentityAffine
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
class TrainableInputInterface(nn.Module):
|
| 13 |
+
"""Keep gradients through the original embedding, normalization, and map."""
|
| 14 |
+
|
| 15 |
+
def __init__(self, original, width):
|
| 16 |
+
super().__init__()
|
| 17 |
+
self.original = original
|
| 18 |
+
self.interface = IdentityAffine(width)
|
| 19 |
+
|
| 20 |
+
@property
|
| 21 |
+
def tok_embeddings(self):
|
| 22 |
+
return self.original.tok_embeddings
|
| 23 |
+
|
| 24 |
+
def forward(self, input_ids=None, inputs_embeds=None):
|
| 25 |
+
hidden = self.original(input_ids=input_ids, inputs_embeds=inputs_embeds)
|
| 26 |
+
return self.interface(hidden)
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
class FullInputInterface(FullFineTune):
|
| 30 |
+
def __init__(self, options):
|
| 31 |
+
initial_options = copy.deepcopy(options)
|
| 32 |
+
initial_options.pop("checkpoint", None)
|
| 33 |
+
initial_options.pop("inference_only", None)
|
| 34 |
+
super().__init__(initial_options)
|
| 35 |
+
self.options = copy.deepcopy(options)
|
| 36 |
+
self.head_max_len = int(options.get("head_max_len", self.head_max_len))
|
| 37 |
+
if self.head_max_len < 1:
|
| 38 |
+
raise ValueError("head_max_len must be positive")
|
| 39 |
+
torch.set_float32_matmul_precision("highest")
|
| 40 |
+
add_interface = options.get("add_interface", True)
|
| 41 |
+
if add_interface:
|
| 42 |
+
encoder = self.model.encoder
|
| 43 |
+
encoder.embeddings = TrainableInputInterface(
|
| 44 |
+
encoder.embeddings, encoder.config.hidden_size
|
| 45 |
+
).to(self.device)
|
| 46 |
+
self.identity.update(
|
| 47 |
+
head_max_len=self.head_max_len,
|
| 48 |
+
interface={"kind": "exact_linear", "position": "after native embeddings"}
|
| 49 |
+
if add_interface
|
| 50 |
+
else None,
|
| 51 |
+
)
|
| 52 |
+
self.metadata.update(
|
| 53 |
+
**self.identity,
|
| 54 |
+
parameters=sum(p.numel() for p in self.model.parameters()),
|
| 55 |
+
trainable_parameters=sum(p.numel() for p in self.model.parameters() if p.requires_grad),
|
| 56 |
+
float32_matmul_precision=torch.get_float32_matmul_precision(),
|
| 57 |
+
interface_parameters=sum(
|
| 58 |
+
p.numel() for p in self.model.encoder.embeddings.interface.parameters()
|
| 59 |
+
)
|
| 60 |
+
if add_interface
|
| 61 |
+
else 0,
|
| 62 |
+
)
|
| 63 |
+
if options.get("checkpoint"):
|
| 64 |
+
self.load_checkpoint(options["checkpoint"])
|
| 65 |
+
if options.get("inference_only", False):
|
| 66 |
+
self.model.requires_grad_(False)
|
| 67 |
+
self.model.eval()
|
| 68 |
+
|
| 69 |
+
def predict(self, state, questions):
|
| 70 |
+
self.model.eval()
|
| 71 |
+
return self.agent.predict(
|
| 72 |
+
state, questions, max_len=self.max_len, head_max_len=self.head_max_len
|
| 73 |
+
)
|
| 74 |
+
|
| 75 |
+
|
| 76 |
+
def create(options):
|
| 77 |
+
return FullInputInterface(options)
|
ariadne_bench/hybrid.py
ADDED
|
@@ -0,0 +1,342 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Experimental small causal LM -> ModernBERT -> Laya decision model.
|
| 2 |
+
|
| 3 |
+
Optional torch/transformers/laya dependencies are imported only by this module.
|
| 4 |
+
No generation, vocabulary output head, or persistent KV cache is used.
|
| 5 |
+
"""
|
| 6 |
+
|
| 7 |
+
import copy
|
| 8 |
+
import hashlib
|
| 9 |
+
import json
|
| 10 |
+
from contextlib import nullcontext
|
| 11 |
+
from pathlib import Path
|
| 12 |
+
|
| 13 |
+
import torch
|
| 14 |
+
from torch import nn
|
| 15 |
+
|
| 16 |
+
from .schema import labels
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
class ContextualEmbeddings(nn.Module):
|
| 20 |
+
"""Replace ModernBERT's entire token lookup / normalization / dropout module."""
|
| 21 |
+
|
| 22 |
+
def forward(self, input_ids=None, inputs_embeds=None):
|
| 23 |
+
if input_ids is not None or inputs_embeds is None:
|
| 24 |
+
raise ValueError("the hybrid encoder accepts contextual embeddings only")
|
| 25 |
+
return inputs_embeds
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
class SemanticEncoder(nn.Module):
|
| 29 |
+
def __init__(self, llm, bert, layer_indices):
|
| 30 |
+
super().__init__()
|
| 31 |
+
indices = list(layer_indices)
|
| 32 |
+
if not indices or indices != sorted(set(indices)):
|
| 33 |
+
raise ValueError("bert_layers must be a nonempty, increasing list of distinct indices")
|
| 34 |
+
if indices[0] < 0 or indices[-1] >= len(bert.layers):
|
| 35 |
+
raise ValueError("bert_layers contains an out-of-range index")
|
| 36 |
+
self.llm, self.bert = llm, bert
|
| 37 |
+
self.layer_indices = indices
|
| 38 |
+
self.config = bert.config
|
| 39 |
+
self.bert.layers = nn.ModuleList([bert.layers[i] for i in indices])
|
| 40 |
+
self.bert.config.num_hidden_layers = len(indices)
|
| 41 |
+
# Preserve each retained block's original attention type and rotary settings.
|
| 42 |
+
self.bert.config.layer_types = [layer.attention_type for layer in self.bert.layers]
|
| 43 |
+
self.bert.embeddings = ContextualEmbeddings()
|
| 44 |
+
width = llm.config.hidden_size
|
| 45 |
+
self.bridge = nn.Sequential(nn.LayerNorm(width), nn.Linear(width, self.config.hidden_size))
|
| 46 |
+
if width == self.config.hidden_size:
|
| 47 |
+
nn.init.eye_(self.bridge[1].weight)
|
| 48 |
+
nn.init.zeros_(self.bridge[1].bias)
|
| 49 |
+
self.llm.requires_grad_(False).eval()
|
| 50 |
+
|
| 51 |
+
def train(self, mode=True):
|
| 52 |
+
super().train(mode)
|
| 53 |
+
self.llm.eval()
|
| 54 |
+
return self
|
| 55 |
+
|
| 56 |
+
def forward(self, input_ids, attention_mask, **kwargs):
|
| 57 |
+
with torch.no_grad():
|
| 58 |
+
h = self.llm(
|
| 59 |
+
input_ids=input_ids,
|
| 60 |
+
attention_mask=attention_mask,
|
| 61 |
+
use_cache=False,
|
| 62 |
+
output_hidden_states=False,
|
| 63 |
+
return_dict=True,
|
| 64 |
+
).last_hidden_state
|
| 65 |
+
h = self.bridge(h)
|
| 66 |
+
return self.bert(inputs_embeds=h, attention_mask=attention_mask, return_dict=True)
|
| 67 |
+
|
| 68 |
+
|
| 69 |
+
def build_item(tokenizer, state_ids, question, max_len=1024, min_state_tokens=32):
|
| 70 |
+
"""Use the LM tokenizer throughout; gather the final token of each whole option.
|
| 71 |
+
|
| 72 |
+
State comes first so causal option features can attend to it. Only the state
|
| 73 |
+
may be truncated. Reject an overlong rubric instead of silently deleting options.
|
| 74 |
+
"""
|
| 75 |
+
from laya.common import QTYPES, render_options
|
| 76 |
+
|
| 77 |
+
keys = labels(question)
|
| 78 |
+
q = {"t": question["type"], "ins": question["instructions"], "crit": question.get("criteria")}
|
| 79 |
+
if "labels" in question:
|
| 80 |
+
q["labels"] = question["labels"]
|
| 81 |
+
|
| 82 |
+
def encode(text):
|
| 83 |
+
return tokenizer.encode(text, add_special_tokens=False)
|
| 84 |
+
|
| 85 |
+
prefix = encode("State:\n")
|
| 86 |
+
heading = encode(f"\nQuestion ({q['t']}): {q['ins']}\nOptions:")
|
| 87 |
+
options = [encode("\nOption: " + text) for text in render_options(q)]
|
| 88 |
+
if len(options) != len(keys) or any(not option for option in options):
|
| 89 |
+
raise ValueError("every option must have a token span")
|
| 90 |
+
room = max_len - len(prefix) - len(heading) - sum(map(len, options))
|
| 91 |
+
if room < min(len(state_ids), min_state_tokens):
|
| 92 |
+
raise ValueError("question and full options exceed the token budget; increase max_len")
|
| 93 |
+
ids = prefix + state_ids[:room] + heading
|
| 94 |
+
markers = []
|
| 95 |
+
for option in options:
|
| 96 |
+
ids.extend(option)
|
| 97 |
+
markers.append(len(ids) - 1)
|
| 98 |
+
return {
|
| 99 |
+
"ids": ids,
|
| 100 |
+
"markers": markers,
|
| 101 |
+
"qtype": QTYPES[question["type"]],
|
| 102 |
+
"keys": keys,
|
| 103 |
+
"state_tokens_original": len(state_ids),
|
| 104 |
+
"state_tokens_used": min(len(state_ids), room),
|
| 105 |
+
}
|
| 106 |
+
|
| 107 |
+
|
| 108 |
+
def collate(items, pad_token_id, device="cpu"):
|
| 109 |
+
if not items:
|
| 110 |
+
raise ValueError("cannot collate an empty decision batch")
|
| 111 |
+
batch, length, options = (
|
| 112 |
+
len(items),
|
| 113 |
+
max(len(i["ids"]) for i in items),
|
| 114 |
+
max(len(i["markers"]) for i in items),
|
| 115 |
+
)
|
| 116 |
+
ids = torch.full((batch, length), pad_token_id, dtype=torch.long, device=device)
|
| 117 |
+
attention = torch.zeros_like(ids)
|
| 118 |
+
positions = torch.zeros((batch, options), dtype=torch.long, device=device)
|
| 119 |
+
marker_mask = torch.zeros_like(positions, dtype=torch.bool)
|
| 120 |
+
for row, item in enumerate(items):
|
| 121 |
+
n, k = len(item["ids"]), len(item["markers"])
|
| 122 |
+
ids[row, :n] = torch.tensor(item["ids"], device=device)
|
| 123 |
+
attention[row, :n] = 1
|
| 124 |
+
positions[row, :k] = torch.tensor(item["markers"], device=device)
|
| 125 |
+
marker_mask[row, :k] = True
|
| 126 |
+
return {
|
| 127 |
+
"input_ids": ids,
|
| 128 |
+
"attention_mask": attention,
|
| 129 |
+
"marker_pos": positions,
|
| 130 |
+
"marker_mask": marker_mask,
|
| 131 |
+
"qtype": torch.tensor([i["qtype"] for i in items], device=device),
|
| 132 |
+
}
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
class Hybrid:
|
| 136 |
+
"""Harness adapter; local source checkpoints remain unchanged."""
|
| 137 |
+
|
| 138 |
+
def __init__(self, options):
|
| 139 |
+
import laya
|
| 140 |
+
import transformers
|
| 141 |
+
from transformers import AutoModel, AutoTokenizer
|
| 142 |
+
|
| 143 |
+
self.options = copy.deepcopy(options)
|
| 144 |
+
torch.manual_seed(options.get("seed", 42))
|
| 145 |
+
self.device = torch.device(options.get("device", "cpu"))
|
| 146 |
+
if self.device.type not in ("cpu", "cuda"):
|
| 147 |
+
raise ValueError("the prototype supports CPU and CUDA")
|
| 148 |
+
self.dtype = torch.bfloat16 if self.device.type == "cuda" else torch.float32
|
| 149 |
+
llm_path = Path(options["llm_model"]).resolve(strict=True)
|
| 150 |
+
laya_path = Path(options["laya_model"]).resolve(strict=True)
|
| 151 |
+
self.tokenizer = AutoTokenizer.from_pretrained(llm_path, local_files_only=True)
|
| 152 |
+
self.pad_id = self.tokenizer.pad_token_id
|
| 153 |
+
if self.pad_id is None:
|
| 154 |
+
self.pad_id = self.tokenizer.eos_token_id
|
| 155 |
+
if self.pad_id is None:
|
| 156 |
+
raise ValueError("the small LM needs a pad or EOS token")
|
| 157 |
+
llm = AutoModel.from_pretrained(
|
| 158 |
+
llm_path, local_files_only=True, dtype=self.dtype, attn_implementation="sdpa"
|
| 159 |
+
)
|
| 160 |
+
if sum(p.numel() for p in llm.parameters()) >= 1_000_000_000:
|
| 161 |
+
raise ValueError("this prototype requires a small LM with fewer than 1B parameters")
|
| 162 |
+
agent = laya.load(str(laya_path), device="cpu")
|
| 163 |
+
self.model = agent.model
|
| 164 |
+
indices = options.get("bert_layers", list(range(28)))
|
| 165 |
+
self.model.encoder = SemanticEncoder(llm, self.model.encoder, indices)
|
| 166 |
+
self.training_scope = options.get("training_scope", "bridge")
|
| 167 |
+
self.model.requires_grad_(False)
|
| 168 |
+
if self.training_scope == "bridge":
|
| 169 |
+
self.model.encoder.bridge.requires_grad_(True)
|
| 170 |
+
elif self.training_scope == "downstream":
|
| 171 |
+
for name, parameter in self.model.named_parameters():
|
| 172 |
+
parameter.requires_grad_(not name.startswith(("encoder.llm.", "act_head.")))
|
| 173 |
+
else:
|
| 174 |
+
raise ValueError("training_scope must be bridge or downstream")
|
| 175 |
+
self.max_len = int(options.get("max_len", 1024))
|
| 176 |
+
self.min_state_tokens = int(options.get("min_state_tokens", 32))
|
| 177 |
+
self.batch_size = int(options.get("batch_size", 16))
|
| 178 |
+
if self.batch_size < 1 or self.max_len < 1 or self.min_state_tokens < 1:
|
| 179 |
+
raise ValueError("batch_size, max_len, and min_state_tokens must be positive")
|
| 180 |
+
if self.max_len > min(
|
| 181 |
+
llm.config.max_position_embeddings, self.model.encoder.config.max_position_embeddings
|
| 182 |
+
):
|
| 183 |
+
raise ValueError("max_len exceeds a backbone's positional limit")
|
| 184 |
+
self.identity = {
|
| 185 |
+
"format_version": 1,
|
| 186 |
+
"llm_model": str(llm_path),
|
| 187 |
+
"laya_model": str(laya_path),
|
| 188 |
+
"bert_layers": list(indices),
|
| 189 |
+
"max_len": self.max_len,
|
| 190 |
+
"min_state_tokens": self.min_state_tokens,
|
| 191 |
+
"prompt_format": "state-question-whole-options-v1",
|
| 192 |
+
"training_scope": self.training_scope,
|
| 193 |
+
}
|
| 194 |
+
self.metadata = {
|
| 195 |
+
**self.identity,
|
| 196 |
+
"training_status": "untrained hybrid connection",
|
| 197 |
+
"device": str(self.device),
|
| 198 |
+
"gpu_name": torch.cuda.get_device_name(self.device)
|
| 199 |
+
if self.device.type == "cuda"
|
| 200 |
+
else None,
|
| 201 |
+
"dtype": str(self.dtype),
|
| 202 |
+
"torch_version": torch.__version__,
|
| 203 |
+
"transformers_version": transformers.__version__,
|
| 204 |
+
"batch_size": self.batch_size,
|
| 205 |
+
"llm_layers": llm.config.num_hidden_layers,
|
| 206 |
+
"llm_width": llm.config.hidden_size,
|
| 207 |
+
"bert_width": self.model.encoder.config.hidden_size,
|
| 208 |
+
"head_layers": len(self.model.head.layers) if self.model.head else 0,
|
| 209 |
+
"parameters": sum(p.numel() for p in self.model.parameters()),
|
| 210 |
+
"llm_parameters": sum(p.numel() for p in llm.parameters()),
|
| 211 |
+
"trainable_parameters": sum(
|
| 212 |
+
p.numel() for p in self.model.parameters() if p.requires_grad
|
| 213 |
+
),
|
| 214 |
+
"temperature": 1.0,
|
| 215 |
+
"use_cache": False,
|
| 216 |
+
}
|
| 217 |
+
if options.get("checkpoint"):
|
| 218 |
+
self.load_checkpoint(options["checkpoint"])
|
| 219 |
+
self.model.to(self.device).eval()
|
| 220 |
+
|
| 221 |
+
def training_mode(self):
|
| 222 |
+
# Frozen modules remain deterministic, but autograd still traverses BERT
|
| 223 |
+
# and the scorer to train the connection below them.
|
| 224 |
+
self.model.train(self.training_scope == "downstream")
|
| 225 |
+
|
| 226 |
+
def checkpoint_tensor_names(self):
|
| 227 |
+
if self.training_scope == "bridge":
|
| 228 |
+
return {k for k in self.model.state_dict() if k.startswith("encoder.bridge.")}
|
| 229 |
+
return {k for k in self.model.state_dict() if not k.startswith("encoder.llm.")}
|
| 230 |
+
|
| 231 |
+
def autocast(self):
|
| 232 |
+
return (
|
| 233 |
+
torch.autocast("cuda", dtype=self.dtype)
|
| 234 |
+
if self.device.type == "cuda"
|
| 235 |
+
else nullcontext()
|
| 236 |
+
)
|
| 237 |
+
|
| 238 |
+
def prepare(self, state, questions):
|
| 239 |
+
from laya.common import serialize_state
|
| 240 |
+
|
| 241 |
+
if not questions:
|
| 242 |
+
raise ValueError("at least one question is required")
|
| 243 |
+
state_ids = self.tokenizer.encode(serialize_state(state), add_special_tokens=False)
|
| 244 |
+
return [
|
| 245 |
+
build_item(self.tokenizer, state_ids, q, self.max_len, self.min_state_tokens)
|
| 246 |
+
for q in questions.values()
|
| 247 |
+
]
|
| 248 |
+
|
| 249 |
+
@torch.inference_mode()
|
| 250 |
+
def predict(self, state, questions):
|
| 251 |
+
self.model.eval()
|
| 252 |
+
items = self.prepare(state, questions)
|
| 253 |
+
probabilities = []
|
| 254 |
+
for start in range(0, len(items), self.batch_size):
|
| 255 |
+
chunk = items[start : start + self.batch_size]
|
| 256 |
+
with self.autocast():
|
| 257 |
+
logits, _ = self.model(**collate(chunk, self.pad_id, self.device))
|
| 258 |
+
p = logits.softmax(-1).cpu().tolist()
|
| 259 |
+
probabilities.extend([row[: len(item["keys"])] for row, item in zip(p, chunk)])
|
| 260 |
+
answers = {}
|
| 261 |
+
for (qid, q), item, p in zip(questions.items(), items, probabilities):
|
| 262 |
+
keys = item["keys"]
|
| 263 |
+
answer = {"type": q["type"], "probabilities": dict(zip(keys, p)), "confidence": max(p)}
|
| 264 |
+
if q["type"] == "choice":
|
| 265 |
+
answer["choice"] = keys[max(range(len(p)), key=p.__getitem__)]
|
| 266 |
+
elif q["type"] == "noul":
|
| 267 |
+
answer["noul"] = p[1]
|
| 268 |
+
else:
|
| 269 |
+
answer["score"] = sum(i * value for i, value in enumerate(p))
|
| 270 |
+
answers[qid] = answer
|
| 271 |
+
return {
|
| 272 |
+
"model": "experimental-lm-laya-hybrid",
|
| 273 |
+
"answers": answers,
|
| 274 |
+
"usage": {
|
| 275 |
+
"llm_forward_calls": (len(items) + self.batch_size - 1) // self.batch_size,
|
| 276 |
+
"generated_tokens": 0,
|
| 277 |
+
"input_tokens": sum(len(i["ids"]) for i in items),
|
| 278 |
+
"truncated_state_questions": sum(
|
| 279 |
+
i["state_tokens_used"] < i["state_tokens_original"] for i in items
|
| 280 |
+
),
|
| 281 |
+
},
|
| 282 |
+
}
|
| 283 |
+
|
| 284 |
+
def synchronize(self):
|
| 285 |
+
if self.device.type == "cuda":
|
| 286 |
+
torch.cuda.synchronize(self.device)
|
| 287 |
+
|
| 288 |
+
def close(self):
|
| 289 |
+
if hasattr(self, "model"):
|
| 290 |
+
del self.model
|
| 291 |
+
if self.device.type == "cuda":
|
| 292 |
+
torch.cuda.empty_cache()
|
| 293 |
+
|
| 294 |
+
def save_checkpoint(self, directory, training):
|
| 295 |
+
from safetensors.torch import save_file
|
| 296 |
+
from .schema import write_json
|
| 297 |
+
|
| 298 |
+
directory = Path(directory)
|
| 299 |
+
directory.mkdir(parents=True, exist_ok=False)
|
| 300 |
+
# Frozen source weights are referenced; only the selected training scope is saved.
|
| 301 |
+
names = self.checkpoint_tensor_names()
|
| 302 |
+
state = {
|
| 303 |
+
k: v.detach().cpu().contiguous()
|
| 304 |
+
for k, v in self.model.state_dict().items()
|
| 305 |
+
if k in names
|
| 306 |
+
}
|
| 307 |
+
weights = directory / "adapter.safetensors"
|
| 308 |
+
save_file(state, weights)
|
| 309 |
+
write_json(
|
| 310 |
+
directory / "manifest.json",
|
| 311 |
+
{
|
| 312 |
+
"identity": self.identity,
|
| 313 |
+
"training": training,
|
| 314 |
+
"weights_sha256": hashlib.sha256(weights.read_bytes()).hexdigest(),
|
| 315 |
+
},
|
| 316 |
+
)
|
| 317 |
+
|
| 318 |
+
def load_checkpoint(self, directory):
|
| 319 |
+
from safetensors.torch import load_file
|
| 320 |
+
|
| 321 |
+
directory = Path(directory)
|
| 322 |
+
manifest = json.loads((directory / "manifest.json").read_text())
|
| 323 |
+
if manifest["identity"] != self.identity:
|
| 324 |
+
raise ValueError("hybrid checkpoint architecture or tokenizer provenance differs")
|
| 325 |
+
weights = directory / "adapter.safetensors"
|
| 326 |
+
actual_hash = hashlib.sha256(weights.read_bytes()).hexdigest()
|
| 327 |
+
if actual_hash != manifest["weights_sha256"]:
|
| 328 |
+
raise ValueError("hybrid checkpoint hash mismatch")
|
| 329 |
+
state = load_file(weights)
|
| 330 |
+
expected = self.checkpoint_tensor_names()
|
| 331 |
+
if set(state) != expected:
|
| 332 |
+
raise ValueError(
|
| 333 |
+
"hybrid checkpoint must contain exactly the tensors for its training scope"
|
| 334 |
+
)
|
| 335 |
+
self.model.load_state_dict(state, strict=False)
|
| 336 |
+
self.metadata.update(
|
| 337 |
+
{"training_status": manifest["training"], "checkpoint_sha256": actual_hash}
|
| 338 |
+
)
|
| 339 |
+
|
| 340 |
+
|
| 341 |
+
def create(options):
|
| 342 |
+
return Hybrid(options)
|
ariadne_bench/hybrid_control.py
ADDED
|
@@ -0,0 +1,169 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Matched trainable-interface control using Laya's original token embeddings."""
|
| 2 |
+
|
| 3 |
+
import copy
|
| 4 |
+
from pathlib import Path
|
| 5 |
+
|
| 6 |
+
import torch
|
| 7 |
+
from torch import nn
|
| 8 |
+
from transformers.modeling_outputs import BaseModelOutput
|
| 9 |
+
|
| 10 |
+
from .hybrid import Hybrid, SemanticEncoder
|
| 11 |
+
from .interfaces import make_interface
|
| 12 |
+
from .schema import labels
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
class FrozenTokenEncoder(nn.Module):
|
| 16 |
+
def __init__(self, embeddings, config):
|
| 17 |
+
super().__init__()
|
| 18 |
+
self.embeddings = embeddings
|
| 19 |
+
self.config = copy.deepcopy(config)
|
| 20 |
+
self.config.num_hidden_layers = 0
|
| 21 |
+
|
| 22 |
+
def forward(self, input_ids, **kwargs):
|
| 23 |
+
return BaseModelOutput(last_hidden_state=self.embeddings(input_ids=input_ids))
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
class LayaControl(Hybrid):
|
| 27 |
+
def __init__(self, options):
|
| 28 |
+
import laya
|
| 29 |
+
import transformers
|
| 30 |
+
|
| 31 |
+
self.options = copy.deepcopy(options)
|
| 32 |
+
torch.manual_seed(options.get("seed", 42))
|
| 33 |
+
self.device = torch.device(options.get("device", "cpu"))
|
| 34 |
+
if self.device.type not in ("cpu", "cuda"):
|
| 35 |
+
raise ValueError("the prototype supports CPU and CUDA")
|
| 36 |
+
self.dtype = torch.bfloat16 if self.device.type == "cuda" else torch.float32
|
| 37 |
+
self.training_scope = "bridge"
|
| 38 |
+
laya_path = Path(options["laya_model"]).resolve(strict=True)
|
| 39 |
+
self.native_prediction = bool(options.get("native_prediction", False))
|
| 40 |
+
agent = laya.load(
|
| 41 |
+
str(laya_path), device=str(self.device) if self.native_prediction else "cpu"
|
| 42 |
+
)
|
| 43 |
+
if self.native_prediction:
|
| 44 |
+
if agent.device != self.device:
|
| 45 |
+
raise RuntimeError("native Laya loaded on an unexpected device")
|
| 46 |
+
self.agent = agent
|
| 47 |
+
self.dtype = agent.dtype
|
| 48 |
+
self.tokenizer, self.model = agent.tok, agent.model
|
| 49 |
+
self.pad_id = self.tokenizer.pad_token_id
|
| 50 |
+
bert = self.model.encoder
|
| 51 |
+
tokens = FrozenTokenEncoder(bert.embeddings, bert.config)
|
| 52 |
+
indices = options.get("bert_layers", list(range(len(bert.layers))))
|
| 53 |
+
self.model.encoder = SemanticEncoder(tokens, bert, indices)
|
| 54 |
+
if "interface" in options:
|
| 55 |
+
specification = options["interface"]
|
| 56 |
+
if specification.get("kind", "").startswith("exact_linear"):
|
| 57 |
+
torch.set_float32_matmul_precision("highest")
|
| 58 |
+
self.model.encoder.bridge = make_interface(bert.config.hidden_size, specification)
|
| 59 |
+
self.model.requires_grad_(False)
|
| 60 |
+
self.model.encoder.bridge.requires_grad_(True)
|
| 61 |
+
self.max_len = int(options.get("max_len", 512))
|
| 62 |
+
self.head_max_len = agent.cfg.get("head_max_len", 192)
|
| 63 |
+
self.batch_size = int(options.get("batch_size", 16))
|
| 64 |
+
if self.max_len < 1 or self.batch_size < 1:
|
| 65 |
+
raise ValueError("max_len and batch_size must be positive")
|
| 66 |
+
self.identity = {
|
| 67 |
+
"format_version": 1,
|
| 68 |
+
"feature_source": "original Laya token embeddings",
|
| 69 |
+
"laya_model": str(laya_path),
|
| 70 |
+
"bert_layers": list(indices),
|
| 71 |
+
"max_len": self.max_len,
|
| 72 |
+
"head_max_len": self.head_max_len,
|
| 73 |
+
"prompt_format": "original-laya-v1",
|
| 74 |
+
"training_scope": "bridge",
|
| 75 |
+
}
|
| 76 |
+
# Preserve the identity schema of existing saved LayerNorm+Linear checkpoints.
|
| 77 |
+
if "interface" in options:
|
| 78 |
+
self.identity["interface"] = copy.deepcopy(options["interface"])
|
| 79 |
+
if self.native_prediction:
|
| 80 |
+
self.identity["native_prediction"] = True
|
| 81 |
+
self.metadata = {
|
| 82 |
+
**self.identity,
|
| 83 |
+
"training_status": "untrained control interface",
|
| 84 |
+
"device": str(self.device),
|
| 85 |
+
"gpu_name": torch.cuda.get_device_name(self.device)
|
| 86 |
+
if self.device.type == "cuda"
|
| 87 |
+
else None,
|
| 88 |
+
"dtype": str(self.dtype),
|
| 89 |
+
"torch_version": torch.__version__,
|
| 90 |
+
"transformers_version": transformers.__version__,
|
| 91 |
+
"batch_size": self.batch_size,
|
| 92 |
+
"parameters": sum(p.numel() for p in self.model.parameters()),
|
| 93 |
+
"trainable_parameters": sum(
|
| 94 |
+
p.numel() for p in self.model.parameters() if p.requires_grad
|
| 95 |
+
),
|
| 96 |
+
"temperature": 1.0,
|
| 97 |
+
}
|
| 98 |
+
if "interface" in options:
|
| 99 |
+
self.metadata["float32_matmul_precision"] = torch.get_float32_matmul_precision()
|
| 100 |
+
if self.native_prediction:
|
| 101 |
+
self.metadata.update(
|
| 102 |
+
{
|
| 103 |
+
"prediction_path": "native Laya SDK",
|
| 104 |
+
"temperature": list(agent.temperature),
|
| 105 |
+
"temperature_by_options": dict(agent.temperature_by_options),
|
| 106 |
+
"laya_version": laya.__version__,
|
| 107 |
+
}
|
| 108 |
+
)
|
| 109 |
+
if options.get("checkpoint"):
|
| 110 |
+
self.load_checkpoint(options["checkpoint"])
|
| 111 |
+
self.model.to(self.device).eval()
|
| 112 |
+
|
| 113 |
+
def prepare(self, state, questions):
|
| 114 |
+
from laya.common import QTYPES, build_sequence, serialize_state
|
| 115 |
+
|
| 116 |
+
if not questions:
|
| 117 |
+
raise ValueError("at least one question is required")
|
| 118 |
+
state_ids = self.tokenizer.encode(
|
| 119 |
+
serialize_state(state).replace(self.tokenizer.mask_token, " "), add_special_tokens=False
|
| 120 |
+
)
|
| 121 |
+
result = []
|
| 122 |
+
for question in questions.values():
|
| 123 |
+
q = {
|
| 124 |
+
"t": question["type"],
|
| 125 |
+
"ins": question["instructions"],
|
| 126 |
+
"crit": question.get("criteria"),
|
| 127 |
+
}
|
| 128 |
+
if "labels" in question:
|
| 129 |
+
q["labels"] = question["labels"]
|
| 130 |
+
ids, markers = build_sequence(
|
| 131 |
+
self.tokenizer, state, q, self.max_len, self.head_max_len, state_ids=state_ids
|
| 132 |
+
)
|
| 133 |
+
keys = labels(question)
|
| 134 |
+
if len(keys) != len(markers):
|
| 135 |
+
raise ValueError("the Laya token budget removed an option")
|
| 136 |
+
# build_sequence appends state after its final head SEP and leaves a final SEP.
|
| 137 |
+
last_head_sep = next(
|
| 138 |
+
i for i in range(markers[-1] + 1, len(ids)) if ids[i] == self.tokenizer.sep_token_id
|
| 139 |
+
)
|
| 140 |
+
state_used = min(len(state_ids), max(0, len(ids) - last_head_sep - 2))
|
| 141 |
+
result.append(
|
| 142 |
+
{
|
| 143 |
+
"ids": ids,
|
| 144 |
+
"markers": markers,
|
| 145 |
+
"qtype": QTYPES[q["t"]],
|
| 146 |
+
"keys": keys,
|
| 147 |
+
"state_tokens_original": len(state_ids),
|
| 148 |
+
"state_tokens_used": state_used,
|
| 149 |
+
}
|
| 150 |
+
)
|
| 151 |
+
return result
|
| 152 |
+
|
| 153 |
+
def predict(self, state, questions):
|
| 154 |
+
if self.native_prediction:
|
| 155 |
+
self.model.eval()
|
| 156 |
+
return self.agent.predict(state, questions, max_len=self.max_len)
|
| 157 |
+
response = super().predict(state, questions)
|
| 158 |
+
response["model"] = "laya-interface-control"
|
| 159 |
+
response["usage"]["llm_forward_calls"] = 0
|
| 160 |
+
return response
|
| 161 |
+
|
| 162 |
+
def close(self):
|
| 163 |
+
if hasattr(self, "agent"):
|
| 164 |
+
del self.agent
|
| 165 |
+
super().close()
|
| 166 |
+
|
| 167 |
+
|
| 168 |
+
def create(options):
|
| 169 |
+
return LayaControl(options)
|
ariadne_bench/interfaces.py
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Input transformations with explicit initialization and low-rank factors."""
|
| 2 |
+
|
| 3 |
+
import math
|
| 4 |
+
|
| 5 |
+
import torch
|
| 6 |
+
from torch import nn
|
| 7 |
+
from torch.nn import functional as F
|
| 8 |
+
|
| 9 |
+
|
| 10 |
+
class IdentityAffine(nn.Module):
|
| 11 |
+
"""Affine map initialized as identity, preserving input precision under AMP.
|
| 12 |
+
|
| 13 |
+
Autocasting an identity GEMM to BF16 would round FP32 embeddings before
|
| 14 |
+
BERT sees them. Keep this small interface in FP32 and return the input dtype.
|
| 15 |
+
CUDA callers must use highest float32 matmul precision (no TF32 rounding).
|
| 16 |
+
"""
|
| 17 |
+
|
| 18 |
+
def __init__(self, width, residual=False):
|
| 19 |
+
super().__init__()
|
| 20 |
+
self.residual = residual
|
| 21 |
+
self.proj = nn.Linear(width, width)
|
| 22 |
+
if residual:
|
| 23 |
+
nn.init.zeros_(self.proj.weight)
|
| 24 |
+
else:
|
| 25 |
+
nn.init.eye_(self.proj.weight)
|
| 26 |
+
nn.init.zeros_(self.proj.bias)
|
| 27 |
+
|
| 28 |
+
def forward(self, x):
|
| 29 |
+
with torch.autocast(device_type=x.device.type, enabled=False):
|
| 30 |
+
value = F.linear(x.float(), self.proj.weight.float(), self.proj.bias.float())
|
| 31 |
+
if self.residual:
|
| 32 |
+
value = x.float() + value
|
| 33 |
+
return value.to(x.dtype)
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
class LowRankLinear(nn.Module):
|
| 37 |
+
def __init__(self, width, rank, residual=False):
|
| 38 |
+
super().__init__()
|
| 39 |
+
if not 1 <= rank <= width:
|
| 40 |
+
raise ValueError("rank must be between one and the hidden width")
|
| 41 |
+
self.rank, self.residual = rank, residual
|
| 42 |
+
self.a = nn.Parameter(torch.empty(rank, width))
|
| 43 |
+
self.b = nn.Parameter(torch.empty(width, rank))
|
| 44 |
+
self.bias = nn.Parameter(torch.zeros(width))
|
| 45 |
+
if residual:
|
| 46 |
+
nn.init.kaiming_uniform_(self.a, a=math.sqrt(5))
|
| 47 |
+
nn.init.zeros_(self.b)
|
| 48 |
+
else:
|
| 49 |
+
# A pure rank-r map cannot start at identity. Start as an orthogonal
|
| 50 |
+
# rank-r projection instead, and record this in experiment metadata.
|
| 51 |
+
q, _ = torch.linalg.qr(torch.randn(width, rank), mode="reduced")
|
| 52 |
+
with torch.no_grad():
|
| 53 |
+
self.a.copy_(q.T)
|
| 54 |
+
self.b.copy_(q)
|
| 55 |
+
|
| 56 |
+
def forward(self, x):
|
| 57 |
+
value = F.linear(F.linear(x, self.a), self.b, self.bias)
|
| 58 |
+
return x + value if self.residual else value
|
| 59 |
+
|
| 60 |
+
def dense_weight(self):
|
| 61 |
+
weight = self.b @ self.a
|
| 62 |
+
if self.residual:
|
| 63 |
+
weight = weight + torch.eye(weight.shape[0], device=weight.device, dtype=weight.dtype)
|
| 64 |
+
return weight
|
| 65 |
+
|
| 66 |
+
@classmethod
|
| 67 |
+
def from_svd(cls, u, s, vh, bias, rank, residual=False):
|
| 68 |
+
layer = cls(u.shape[0], rank, residual=residual)
|
| 69 |
+
with torch.no_grad():
|
| 70 |
+
root = s[:rank].sqrt()
|
| 71 |
+
layer.a.copy_(root[:, None] * vh[:rank])
|
| 72 |
+
layer.b.copy_(u[:, :rank] * root[None, :])
|
| 73 |
+
layer.bias.copy_(bias)
|
| 74 |
+
return layer
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
class Diagonal(nn.Module):
|
| 78 |
+
def __init__(self, width):
|
| 79 |
+
super().__init__()
|
| 80 |
+
self.scale = nn.Parameter(torch.ones(width))
|
| 81 |
+
self.bias = nn.Parameter(torch.zeros(width))
|
| 82 |
+
|
| 83 |
+
def forward(self, x):
|
| 84 |
+
return x * self.scale + self.bias
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
class ResidualInput(nn.Module):
|
| 88 |
+
def __init__(self, width):
|
| 89 |
+
super().__init__()
|
| 90 |
+
self.norm = nn.LayerNorm(width)
|
| 91 |
+
self.linear = nn.Linear(width, width)
|
| 92 |
+
nn.init.zeros_(self.linear.weight)
|
| 93 |
+
nn.init.zeros_(self.linear.bias)
|
| 94 |
+
|
| 95 |
+
def forward(self, x):
|
| 96 |
+
return x + self.linear(self.norm(x))
|
| 97 |
+
|
| 98 |
+
|
| 99 |
+
def make_interface(width, specification):
|
| 100 |
+
kind = specification.get("kind", "norm_linear")
|
| 101 |
+
if kind in ("exact_linear", "exact_linear_residual"):
|
| 102 |
+
return IdentityAffine(width, residual=kind == "exact_linear_residual")
|
| 103 |
+
if kind == "identity":
|
| 104 |
+
return nn.Identity()
|
| 105 |
+
if kind == "norm":
|
| 106 |
+
return nn.LayerNorm(width)
|
| 107 |
+
if kind == "diagonal":
|
| 108 |
+
return Diagonal(width)
|
| 109 |
+
if kind == "residual":
|
| 110 |
+
return ResidualInput(width)
|
| 111 |
+
if kind in ("low_rank", "low_rank_delta"):
|
| 112 |
+
return nn.Sequential(
|
| 113 |
+
nn.LayerNorm(width),
|
| 114 |
+
LowRankLinear(width, specification["rank"], residual=kind == "low_rank_delta"),
|
| 115 |
+
)
|
| 116 |
+
if kind in ("linear", "norm_linear"):
|
| 117 |
+
linear = nn.Linear(width, width)
|
| 118 |
+
nn.init.eye_(linear.weight)
|
| 119 |
+
nn.init.zeros_(linear.bias)
|
| 120 |
+
return linear if kind == "linear" else nn.Sequential(nn.LayerNorm(width), linear)
|
| 121 |
+
raise ValueError(f"unknown interface kind: {kind}")
|
ariadne_bench/metrics.py
ADDED
|
@@ -0,0 +1,229 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Explicit scoring conventions; no fitted temperatures or test-set thresholds."""
|
| 2 |
+
|
| 3 |
+
import math
|
| 4 |
+
from collections import defaultdict
|
| 5 |
+
from statistics import mean
|
| 6 |
+
|
| 7 |
+
from .schema import distribution, labels
|
| 8 |
+
|
| 9 |
+
METRIC_VERSION = "1:sum-brier:top-label-ece:right-closed:eps-1e-12"
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
def decision(case, qid, answer):
|
| 13 |
+
q, gold = case["questions"][qid], case["gold"][qid]
|
| 14 |
+
keys = labels(q)
|
| 15 |
+
p = [answer["probabilities"][k] for k in keys]
|
| 16 |
+
target = keys.index(gold["label"])
|
| 17 |
+
row = {
|
| 18 |
+
"case_id": case["id"],
|
| 19 |
+
"suite": case["suite"],
|
| 20 |
+
"question": qid,
|
| 21 |
+
"workflow": case.get("workflow", ""),
|
| 22 |
+
"language": case.get("language", ""),
|
| 23 |
+
"type": q["type"],
|
| 24 |
+
"gold": gold["label"],
|
| 25 |
+
"prediction": answer["label"],
|
| 26 |
+
"correct": float(answer["label"] == gold["label"]),
|
| 27 |
+
"confidence": answer["confidence"],
|
| 28 |
+
"brier_hard": sum((v - float(i == target)) ** 2 for i, v in enumerate(p)),
|
| 29 |
+
"nll_hard": -math.log(max(p[target], 1e-12)),
|
| 30 |
+
"zero_probability_gold": float(p[target] == 0),
|
| 31 |
+
"raw_probability_sum": answer["raw_probability_sum"],
|
| 32 |
+
}
|
| 33 |
+
if "probabilities" in gold:
|
| 34 |
+
g, _ = distribution(gold["probabilities"], keys)
|
| 35 |
+
row.update(
|
| 36 |
+
soft_accuracy=sum(a * b for a, b in zip(p, g)),
|
| 37 |
+
brier_soft=sum((a - b) ** 2 for a, b in zip(p, g)),
|
| 38 |
+
kl_gold_to_prediction=sum(
|
| 39 |
+
b * math.log(b / max(a, 1e-12)) for a, b in zip(p, g) if b > 0
|
| 40 |
+
),
|
| 41 |
+
total_variation=0.5 * sum(abs(a - b) for a, b in zip(p, g)),
|
| 42 |
+
)
|
| 43 |
+
if q["type"] == "score":
|
| 44 |
+
score = sum(i * v for i, v in enumerate(p))
|
| 45 |
+
error = abs(score - gold.get("score", float(gold["label"])))
|
| 46 |
+
row.update(score_mae=error, within_one_level=float(error <= 1))
|
| 47 |
+
return row
|
| 48 |
+
|
| 49 |
+
|
| 50 |
+
def ece(rows, bins):
|
| 51 |
+
# [0, 1/B], (1/B, 2/B], ..., ((B-1)/B, 1]; includes confidence 0 and 1.
|
| 52 |
+
buckets = defaultdict(list)
|
| 53 |
+
for row in rows:
|
| 54 |
+
index = max(0, min(bins - 1, math.ceil(row["confidence"] * bins) - 1))
|
| 55 |
+
buckets[index].append(row)
|
| 56 |
+
return sum(
|
| 57 |
+
len(rs) * abs(mean(r["confidence"] for r in rs) - mean(r["correct"] for r in rs))
|
| 58 |
+
for rs in buckets.values()
|
| 59 |
+
) / len(rows)
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
def summarize(rows, attempted, bins=15):
|
| 63 |
+
result = {
|
| 64 |
+
"attempted": attempted,
|
| 65 |
+
"valid": len(rows),
|
| 66 |
+
"failed": attempted - len(rows),
|
| 67 |
+
"coverage": len(rows) / attempted if attempted else None,
|
| 68 |
+
"accuracy_all": sum(r["correct"] for r in rows) / attempted if attempted else None,
|
| 69 |
+
}
|
| 70 |
+
if not rows:
|
| 71 |
+
result.update(accuracy_valid=None, ece_top_label=None, macro_f1=None)
|
| 72 |
+
return result
|
| 73 |
+
result.update(
|
| 74 |
+
accuracy_valid=mean(r["correct"] for r in rows),
|
| 75 |
+
ece_top_label=ece(rows, bins),
|
| 76 |
+
mean_confidence=mean(r["confidence"] for r in rows),
|
| 77 |
+
)
|
| 78 |
+
for metric in (
|
| 79 |
+
"brier_hard",
|
| 80 |
+
"nll_hard",
|
| 81 |
+
"zero_probability_gold",
|
| 82 |
+
"soft_accuracy",
|
| 83 |
+
"brier_soft",
|
| 84 |
+
"kl_gold_to_prediction",
|
| 85 |
+
"total_variation",
|
| 86 |
+
"score_mae",
|
| 87 |
+
"within_one_level",
|
| 88 |
+
):
|
| 89 |
+
values = [r[metric] for r in rows if metric in r]
|
| 90 |
+
if values:
|
| 91 |
+
result[metric] = mean(values)
|
| 92 |
+
result[metric + "_n"] = len(values)
|
| 93 |
+
# Semantic label namespaces prevent conflating e.g. ordinal 0 across unrelated tasks.
|
| 94 |
+
confusion = defaultdict(lambda: [0, 0, 0])
|
| 95 |
+
for r in rows:
|
| 96 |
+
ns = (r["suite"], r["workflow"], r["question"])
|
| 97 |
+
gold, pred = ns + (r["gold"],), ns + (r["prediction"],)
|
| 98 |
+
if gold == pred:
|
| 99 |
+
confusion[gold][0] += 1
|
| 100 |
+
else:
|
| 101 |
+
confusion[pred][1] += 1
|
| 102 |
+
confusion[gold][2] += 1
|
| 103 |
+
result["macro_f1"] = mean(2 * tp / (2 * tp + fp + fn) for tp, fp, fn in confusion.values())
|
| 104 |
+
# Include complete confidence ties; these are descriptive curves, not chosen thresholds.
|
| 105 |
+
ranked = sorted(rows, key=lambda r: -r["confidence"])
|
| 106 |
+
curve, errors = [], 0
|
| 107 |
+
for i, row in enumerate(ranked, 1):
|
| 108 |
+
errors += 1 - row["correct"]
|
| 109 |
+
if i == len(ranked) or ranked[i]["confidence"] != row["confidence"]:
|
| 110 |
+
curve.append(
|
| 111 |
+
{
|
| 112 |
+
"threshold": row["confidence"],
|
| 113 |
+
"coverage": i / attempted,
|
| 114 |
+
"risk": errors / i,
|
| 115 |
+
"accepted": i,
|
| 116 |
+
}
|
| 117 |
+
)
|
| 118 |
+
result["risk_coverage"] = curve
|
| 119 |
+
return result
|
| 120 |
+
|
| 121 |
+
|
| 122 |
+
def percentile(values, q):
|
| 123 |
+
if not values:
|
| 124 |
+
return None
|
| 125 |
+
values = sorted(values)
|
| 126 |
+
pos = (len(values) - 1) * q
|
| 127 |
+
lo, hi = math.floor(pos), math.ceil(pos)
|
| 128 |
+
return values[lo] + (values[hi] - values[lo]) * (pos - lo)
|
| 129 |
+
|
| 130 |
+
|
| 131 |
+
def timing(records):
|
| 132 |
+
values = [r["latency_ms"] for r in records if r.get("latency_ms") is not None]
|
| 133 |
+
seconds = sum(values) / 1000
|
| 134 |
+
return {
|
| 135 |
+
"requests": len(records),
|
| 136 |
+
"timed_requests": len(values),
|
| 137 |
+
"p50_ms": percentile(values, 0.5),
|
| 138 |
+
"p95_ms": percentile(values, 0.95),
|
| 139 |
+
"p99_ms": percentile(values, 0.99),
|
| 140 |
+
"total_request_seconds": seconds,
|
| 141 |
+
"attempted_decisions_per_second": sum(r["question_count"] for r in records) / seconds
|
| 142 |
+
if seconds and len(values) == len(records)
|
| 143 |
+
else None,
|
| 144 |
+
}
|
| 145 |
+
|
| 146 |
+
|
| 147 |
+
def report(cases, records, bins=15, tolerance=0.02):
|
| 148 |
+
from .schema import normalize_answer
|
| 149 |
+
|
| 150 |
+
by_id = {r["case_id"]: r for r in records}
|
| 151 |
+
if len(by_id) != len(records) or set(by_id) != {c["id"] for c in cases}:
|
| 152 |
+
raise ValueError("records must contain each case exactly once")
|
| 153 |
+
rows, errors = [], []
|
| 154 |
+
counts = {field: defaultdict(int) for field in ("suite", "workflow", "language", "type")}
|
| 155 |
+
for case in cases:
|
| 156 |
+
record = by_id[case["id"]]
|
| 157 |
+
response = record.get("response") or {}
|
| 158 |
+
answers = response.get("answers", {}) if isinstance(response, dict) else {}
|
| 159 |
+
if not isinstance(answers, dict):
|
| 160 |
+
answers = {}
|
| 161 |
+
extra_keys = set(answers) - set(case["questions"])
|
| 162 |
+
for qid, q in case["questions"].items():
|
| 163 |
+
for field in counts:
|
| 164 |
+
counts[field][q["type"] if field == "type" else case.get(field, "")] += 1
|
| 165 |
+
try:
|
| 166 |
+
if record.get("error"):
|
| 167 |
+
raise ValueError(record["error"])
|
| 168 |
+
if extra_keys:
|
| 169 |
+
raise ValueError("response contains unexpected question keys")
|
| 170 |
+
a = normalize_answer(q, answers.get(qid), tolerance)
|
| 171 |
+
rows.append(decision(case, qid, a))
|
| 172 |
+
except (ValueError, TypeError, KeyError) as exc:
|
| 173 |
+
errors.append({"case_id": case["id"], "question": qid, "error": str(exc)})
|
| 174 |
+
total = sum(len(c["questions"]) for c in cases)
|
| 175 |
+
result = {
|
| 176 |
+
"metric_version": METRIC_VERSION,
|
| 177 |
+
"ece_bins": bins,
|
| 178 |
+
"probability_sum_tolerance": tolerance,
|
| 179 |
+
"overall": summarize(rows, total, bins),
|
| 180 |
+
"latency": timing(records),
|
| 181 |
+
"errors": errors,
|
| 182 |
+
}
|
| 183 |
+
for field, sizes in counts.items():
|
| 184 |
+
result["by_" + field] = {
|
| 185 |
+
key: summarize([r for r in rows if r[field] == key], n, bins)
|
| 186 |
+
for key, n in sizes.items()
|
| 187 |
+
if key
|
| 188 |
+
}
|
| 189 |
+
result["latency_by_question_count"] = {
|
| 190 |
+
str(n): timing([r for r in records if r["question_count"] == n])
|
| 191 |
+
for n in sorted({r["question_count"] for r in records})
|
| 192 |
+
}
|
| 193 |
+
usages = [
|
| 194 |
+
r.get("response", {}).get("usage", {})
|
| 195 |
+
for r in records
|
| 196 |
+
if isinstance(r.get("response"), dict)
|
| 197 |
+
]
|
| 198 |
+
usages = [
|
| 199 |
+
u
|
| 200 |
+
for u in usages
|
| 201 |
+
if isinstance(u, dict)
|
| 202 |
+
and all(
|
| 203 |
+
isinstance(u.get(k, 0), int) and not isinstance(u.get(k, 0), bool) and u.get(k, 0) >= 0
|
| 204 |
+
for k in ("input_tokens", "output_tokens")
|
| 205 |
+
)
|
| 206 |
+
]
|
| 207 |
+
result["usage"] = {
|
| 208 |
+
"requests_with_usage": sum(bool(u) for u in usages),
|
| 209 |
+
"input_tokens_known": sum(u.get("input_tokens", 0) or 0 for u in usages),
|
| 210 |
+
"output_tokens_known": sum(u.get("output_tokens", 0) or 0 for u in usages),
|
| 211 |
+
}
|
| 212 |
+
paired = defaultdict(dict)
|
| 213 |
+
case_map = {c["id"]: c for c in cases}
|
| 214 |
+
for row in rows:
|
| 215 |
+
case = case_map[row["case_id"]]
|
| 216 |
+
base = case.get("base_id", case["id"])
|
| 217 |
+
paired[(base, row["question"])][case.get("permutation", 0)] = row["prediction"]
|
| 218 |
+
flips = [
|
| 219 |
+
pred != variants[0]
|
| 220 |
+
for variants in paired.values()
|
| 221 |
+
if 0 in variants
|
| 222 |
+
for variant, pred in variants.items()
|
| 223 |
+
if variant != 0
|
| 224 |
+
]
|
| 225 |
+
result["option_order"] = {
|
| 226 |
+
"valid_pairs": len(flips),
|
| 227 |
+
"flip_rate": mean(flips) if flips else None,
|
| 228 |
+
}
|
| 229 |
+
return result
|
ariadne_bench/portable.py
ADDED
|
@@ -0,0 +1,88 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Load a compact affine adapter with pinned, independently verified base files."""
|
| 2 |
+
|
| 3 |
+
import hashlib
|
| 4 |
+
import json
|
| 5 |
+
from pathlib import Path
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
def sha256(path):
|
| 9 |
+
return hashlib.sha256(Path(path).read_bytes()).hexdigest()
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
def load_release(
|
| 13 |
+
directory, device="cpu", seed=None, base_path=None, local_files_only=False, deterministic=True
|
| 14 |
+
):
|
| 15 |
+
from huggingface_hub import snapshot_download
|
| 16 |
+
from safetensors.torch import load_file
|
| 17 |
+
from .frozen_input_interface import FrozenInputInterface
|
| 18 |
+
|
| 19 |
+
root = Path(directory).resolve()
|
| 20 |
+
config = json.loads((root / "adapter_config.json").read_text())
|
| 21 |
+
if config["format_version"] != 1 or config["interface"] != "exact_linear":
|
| 22 |
+
raise ValueError("unsupported adapter format")
|
| 23 |
+
selected = config["candidate_seed"] if seed is None else int(seed)
|
| 24 |
+
entry = config["seeds"][str(selected)]
|
| 25 |
+
weights = (root / entry["weights_file"]).resolve()
|
| 26 |
+
if not weights.is_relative_to(root):
|
| 27 |
+
raise ValueError("adapter weights must be inside the release directory")
|
| 28 |
+
if sha256(weights) != entry["weights_sha256"]:
|
| 29 |
+
raise ValueError("adapter checkpoint checksum mismatch")
|
| 30 |
+
source = (
|
| 31 |
+
Path(base_path)
|
| 32 |
+
if base_path
|
| 33 |
+
else Path(
|
| 34 |
+
snapshot_download(
|
| 35 |
+
config["base_model"],
|
| 36 |
+
revision=config["base_revision"],
|
| 37 |
+
local_files_only=local_files_only,
|
| 38 |
+
allow_patterns=[
|
| 39 |
+
"model.safetensors",
|
| 40 |
+
"rl_agent_config.json",
|
| 41 |
+
"encoder/*",
|
| 42 |
+
"tokenizer/*",
|
| 43 |
+
],
|
| 44 |
+
)
|
| 45 |
+
)
|
| 46 |
+
)
|
| 47 |
+
for name, expected in config["base_config_sha256"].items():
|
| 48 |
+
if sha256(source / name) != expected:
|
| 49 |
+
raise ValueError(f"base tokenizer/configuration mismatch: {name}")
|
| 50 |
+
adapter = FrozenInputInterface(
|
| 51 |
+
{
|
| 52 |
+
"laya_model": str(source.resolve()),
|
| 53 |
+
"device": device,
|
| 54 |
+
"max_len": config["max_len"],
|
| 55 |
+
"head_max_len": config["head_max_len"],
|
| 56 |
+
"seed": selected,
|
| 57 |
+
"training_scope": "bridge",
|
| 58 |
+
"add_interface": True,
|
| 59 |
+
"deterministic": deterministic,
|
| 60 |
+
}
|
| 61 |
+
)
|
| 62 |
+
try:
|
| 63 |
+
if adapter.initial_frozen_sha256 != config["frozen_state_sha256"]:
|
| 64 |
+
raise ValueError("base model tensors do not match the pinned training source")
|
| 65 |
+
state = load_file(str(weights))
|
| 66 |
+
if set(state) != adapter.checkpoint_tensor_names():
|
| 67 |
+
raise ValueError("checkpoint must contain exactly the affine weight and bias")
|
| 68 |
+
adapter.model.load_state_dict(state, strict=False)
|
| 69 |
+
adapter.model.requires_grad_(False)
|
| 70 |
+
adapter.model.eval()
|
| 71 |
+
adapter.metadata.update(
|
| 72 |
+
trainable_parameters=0,
|
| 73 |
+
checkpoint_sha256=entry["weights_sha256"],
|
| 74 |
+
portable_adapter={
|
| 75 |
+
"base_model": config["base_model"],
|
| 76 |
+
"base_revision": config["base_revision"],
|
| 77 |
+
"seed": selected,
|
| 78 |
+
"selected_epoch": entry["selected_epoch"],
|
| 79 |
+
},
|
| 80 |
+
)
|
| 81 |
+
except Exception:
|
| 82 |
+
adapter.close()
|
| 83 |
+
raise
|
| 84 |
+
return adapter
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
def create(options):
|
| 88 |
+
return load_release(**options)
|
ariadne_bench/reproducibility.py
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Opt-in deterministic execution for a fixed GPU, runtime, and data order."""
|
| 2 |
+
|
| 3 |
+
import os
|
| 4 |
+
import random
|
| 5 |
+
|
| 6 |
+
import numpy as np
|
| 7 |
+
import torch
|
| 8 |
+
|
| 9 |
+
|
| 10 |
+
def configure_determinism(seed):
|
| 11 |
+
workspace = os.environ.get("CUBLAS_WORKSPACE_CONFIG")
|
| 12 |
+
if workspace is None:
|
| 13 |
+
if torch.cuda.is_initialized():
|
| 14 |
+
raise RuntimeError("set CUBLAS_WORKSPACE_CONFIG=:4096:8 before CUDA initialization")
|
| 15 |
+
os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
|
| 16 |
+
elif workspace not in {":4096:8", ":16:8"}:
|
| 17 |
+
raise ValueError("deterministic execution requires a supported cuBLAS workspace config")
|
| 18 |
+
random.seed(seed)
|
| 19 |
+
np.random.seed(seed)
|
| 20 |
+
torch.manual_seed(seed)
|
| 21 |
+
torch.use_deterministic_algorithms(True, warn_only=False)
|
| 22 |
+
torch.backends.cudnn.benchmark = False
|
| 23 |
+
torch.backends.cudnn.deterministic = True
|
| 24 |
+
torch.backends.cudnn.allow_tf32 = False
|
| 25 |
+
torch.set_float32_matmul_precision("highest")
|
| 26 |
+
# Avoid a backend whose deterministic backward support varies by cuDNN version.
|
| 27 |
+
torch.backends.cuda.enable_cudnn_sdp(False)
|
| 28 |
+
return settings()
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
def settings():
|
| 32 |
+
return {
|
| 33 |
+
"deterministic_algorithms": torch.are_deterministic_algorithms_enabled(),
|
| 34 |
+
"warn_only": torch.is_deterministic_algorithms_warn_only_enabled(),
|
| 35 |
+
"cublas_workspace_config": os.environ.get("CUBLAS_WORKSPACE_CONFIG"),
|
| 36 |
+
"python_hash_seed": os.environ.get("PYTHONHASHSEED"),
|
| 37 |
+
"cudnn_benchmark": torch.backends.cudnn.benchmark,
|
| 38 |
+
"cudnn_deterministic": torch.backends.cudnn.deterministic,
|
| 39 |
+
"cudnn_allow_tf32": torch.backends.cudnn.allow_tf32,
|
| 40 |
+
"float32_matmul_precision": torch.get_float32_matmul_precision(),
|
| 41 |
+
"cudnn_sdp_enabled": torch.backends.cuda.cudnn_sdp_enabled(),
|
| 42 |
+
"flash_sdp_enabled": torch.backends.cuda.flash_sdp_enabled(),
|
| 43 |
+
"mem_efficient_sdp_enabled": torch.backends.cuda.mem_efficient_sdp_enabled(),
|
| 44 |
+
"math_sdp_enabled": torch.backends.cuda.math_sdp_enabled(),
|
| 45 |
+
"cuda_version": torch.version.cuda,
|
| 46 |
+
"cudnn_version": torch.backends.cudnn.version(),
|
| 47 |
+
"scope": "fixed hardware/runtime; cross-platform bitwise agreement is not promised",
|
| 48 |
+
}
|
ariadne_bench/runner.py
ADDED
|
@@ -0,0 +1,257 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import copy
|
| 2 |
+
import importlib.metadata
|
| 3 |
+
import json
|
| 4 |
+
import platform
|
| 5 |
+
import time
|
| 6 |
+
from datetime import datetime, timezone
|
| 7 |
+
from pathlib import Path
|
| 8 |
+
|
| 9 |
+
from . import __version__
|
| 10 |
+
from .adapters import create_adapter
|
| 11 |
+
from .metrics import report
|
| 12 |
+
from .schema import digest, dumps, load_bundle, read_jsonl, write_json
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
def environment():
|
| 16 |
+
versions = {}
|
| 17 |
+
for name in ("laya", "torch", "transformers", "datasets", "huggingface-hub"):
|
| 18 |
+
try:
|
| 19 |
+
versions[name] = importlib.metadata.version(name)
|
| 20 |
+
except importlib.metadata.PackageNotFoundError:
|
| 21 |
+
pass
|
| 22 |
+
return {
|
| 23 |
+
"python": platform.python_version(),
|
| 24 |
+
"platform": platform.platform(),
|
| 25 |
+
"processor": platform.processor(),
|
| 26 |
+
"packages": versions,
|
| 27 |
+
"harness": __version__,
|
| 28 |
+
}
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
def check_config(config):
|
| 32 |
+
# Config files are copied into artifacts. Credentials must stay in the environment.
|
| 33 |
+
def visit(value):
|
| 34 |
+
if isinstance(value, dict):
|
| 35 |
+
for key, child in value.items():
|
| 36 |
+
if key.lower() in ("api_key", "token", "password", "authorization", "secret"):
|
| 37 |
+
raise ValueError(
|
| 38 |
+
"put credentials in environment variables, not the saved config"
|
| 39 |
+
)
|
| 40 |
+
visit(child)
|
| 41 |
+
elif isinstance(value, list):
|
| 42 |
+
for child in value:
|
| 43 |
+
visit(child)
|
| 44 |
+
|
| 45 |
+
visit(config)
|
| 46 |
+
if config.get("mode") not in ("generalist", "specialist", "reference"):
|
| 47 |
+
raise ValueError("config mode must be generalist, specialist or reference")
|
| 48 |
+
if not config.get("name"):
|
| 49 |
+
raise ValueError("config requires a display name")
|
| 50 |
+
if "endpoint" in config:
|
| 51 |
+
from urllib.parse import urlsplit
|
| 52 |
+
|
| 53 |
+
url = urlsplit(config["endpoint"])
|
| 54 |
+
if url.username or url.password or url.query or url.fragment:
|
| 55 |
+
raise ValueError("endpoint must not contain credentials, a query or a fragment")
|
| 56 |
+
|
| 57 |
+
|
| 58 |
+
def invoke(adapter, case):
|
| 59 |
+
start = time.perf_counter()
|
| 60 |
+
result = {"case_id": case["id"], "question_count": len(case["questions"])}
|
| 61 |
+
try:
|
| 62 |
+
if hasattr(adapter, "synchronize"):
|
| 63 |
+
adapter.synchronize()
|
| 64 |
+
start = time.perf_counter()
|
| 65 |
+
# Copy protects the frozen bundle from model code that mutates its arguments.
|
| 66 |
+
response = adapter.predict(copy.deepcopy(case["state"]), copy.deepcopy(case["questions"]))
|
| 67 |
+
if hasattr(adapter, "synchronize"):
|
| 68 |
+
adapter.synchronize()
|
| 69 |
+
result["latency_ms"] = (time.perf_counter() - start) * 1000
|
| 70 |
+
dumps(response)
|
| 71 |
+
result["response"] = response
|
| 72 |
+
except Exception as exc:
|
| 73 |
+
result["latency_ms"] = (time.perf_counter() - start) * 1000
|
| 74 |
+
# Arbitrary plugin errors can embed secret configuration. Keep their type only.
|
| 75 |
+
result["error"] = type(exc).__name__
|
| 76 |
+
if isinstance(exc, RuntimeError) and str(exc).startswith("HTTP "):
|
| 77 |
+
result["error"] = str(exc)
|
| 78 |
+
return result
|
| 79 |
+
|
| 80 |
+
|
| 81 |
+
def render_report(result):
|
| 82 |
+
meta = result["run"]
|
| 83 |
+
lines = [
|
| 84 |
+
f"# {meta['config']['name']}",
|
| 85 |
+
"",
|
| 86 |
+
f"Mode: **{meta['config']['mode']}**. Profile: `{meta['manifest']['profile']}`.",
|
| 87 |
+
"",
|
| 88 |
+
f"Cases SHA-256: `{meta['manifest']['sha256']}`.",
|
| 89 |
+
"",
|
| 90 |
+
"| Suite | Valid / attempted | Accuracy (all) | Brier (hard) | Brier (soft) | ECE |",
|
| 91 |
+
"|---|---:|---:|---:|---:|---:|",
|
| 92 |
+
]
|
| 93 |
+
for suite, r in result["by_suite"].items():
|
| 94 |
+
|
| 95 |
+
def fmt(v):
|
| 96 |
+
return "—" if v is None else f"{v:.4f}"
|
| 97 |
+
|
| 98 |
+
lines.append(
|
| 99 |
+
f"| {suite} | {r['valid']} / {r['attempted']} | {fmt(r['accuracy_all'])} | "
|
| 100 |
+
f"{fmt(r.get('brier_hard'))} | {fmt(r.get('brier_soft'))} | {fmt(r.get('ece_top_label'))} |"
|
| 101 |
+
)
|
| 102 |
+
t = result["latency"]
|
| 103 |
+
lines.extend(
|
| 104 |
+
[
|
| 105 |
+
"",
|
| 106 |
+
f"Request p50: {t['p50_ms']} ms; p95: {t['p95_ms']} ms.",
|
| 107 |
+
"",
|
| 108 |
+
"Accuracy (all) counts invalid or missing decisions as incorrect. Proper scores and ECE use valid decisions; check coverage.",
|
| 109 |
+
"ECE uses the predicted option's probability, not the provider's confidence field. Brier sums over classes.",
|
| 110 |
+
"Latency includes the adapter call and local device synchronization; model loading and explicit warmups are separate.",
|
| 111 |
+
"Published reference numbers and local timing runs require matching protocols before comparison.",
|
| 112 |
+
"",
|
| 113 |
+
]
|
| 114 |
+
)
|
| 115 |
+
return "\n".join(lines)
|
| 116 |
+
|
| 117 |
+
|
| 118 |
+
def run(bundle, config_path, output, bins=15, tolerance=0.02, warmup=0, resume=False):
|
| 119 |
+
cases, manifest = load_bundle(bundle)
|
| 120 |
+
config = json.loads(Path(config_path).read_text())
|
| 121 |
+
check_config(config)
|
| 122 |
+
directory = Path(output)
|
| 123 |
+
identity = digest(
|
| 124 |
+
{
|
| 125 |
+
"manifest": manifest,
|
| 126 |
+
"config": config,
|
| 127 |
+
"bins": bins,
|
| 128 |
+
"tolerance": tolerance,
|
| 129 |
+
"warmup": warmup,
|
| 130 |
+
"harness": __version__,
|
| 131 |
+
}
|
| 132 |
+
)
|
| 133 |
+
if resume:
|
| 134 |
+
saved = json.loads((directory / "run.json").read_text())
|
| 135 |
+
if saved["identity"] != identity:
|
| 136 |
+
raise ValueError("resume requires identical data, config and scoring settings")
|
| 137 |
+
records = (
|
| 138 |
+
read_jsonl(directory / "predictions.jsonl")
|
| 139 |
+
if (directory / "predictions.jsonl").exists()
|
| 140 |
+
else []
|
| 141 |
+
)
|
| 142 |
+
ids = [r["case_id"] for r in records]
|
| 143 |
+
if ids != [c["id"] for c in cases[: len(ids)]]:
|
| 144 |
+
raise ValueError("saved records are not the expected case prefix")
|
| 145 |
+
metadata = saved
|
| 146 |
+
else:
|
| 147 |
+
if directory.exists():
|
| 148 |
+
raise ValueError("run directory exists; choose a new path or --resume")
|
| 149 |
+
records = []
|
| 150 |
+
metadata = {
|
| 151 |
+
"identity": identity,
|
| 152 |
+
"created_utc": datetime.now(timezone.utc).isoformat(),
|
| 153 |
+
"config": config,
|
| 154 |
+
"manifest": manifest,
|
| 155 |
+
"environment": environment(),
|
| 156 |
+
"ece_bins": bins,
|
| 157 |
+
"probability_sum_tolerance": tolerance,
|
| 158 |
+
"warmup_per_question_count": warmup,
|
| 159 |
+
"sessions": [],
|
| 160 |
+
}
|
| 161 |
+
if len(records) < len(cases):
|
| 162 |
+
load_start = time.perf_counter()
|
| 163 |
+
adapter = create_adapter(config)
|
| 164 |
+
session = {
|
| 165 |
+
"model_load_seconds": time.perf_counter() - load_start,
|
| 166 |
+
"adapter_metadata": getattr(adapter, "metadata", {}),
|
| 167 |
+
"start_index": len(records),
|
| 168 |
+
"utc": datetime.now(timezone.utc).isoformat(),
|
| 169 |
+
}
|
| 170 |
+
metadata["sessions"].append(session)
|
| 171 |
+
directory.mkdir(parents=True, exist_ok=True)
|
| 172 |
+
write_json(directory / "run.json", metadata)
|
| 173 |
+
try:
|
| 174 |
+
shapes = {}
|
| 175 |
+
for case in cases:
|
| 176 |
+
shapes.setdefault(len(case["questions"]), case)
|
| 177 |
+
with (directory / "warmup.jsonl").open("a") as f:
|
| 178 |
+
for case in shapes.values():
|
| 179 |
+
for _ in range(warmup):
|
| 180 |
+
record = invoke(adapter, case)
|
| 181 |
+
f.write(dumps(record) + "\n")
|
| 182 |
+
f.flush()
|
| 183 |
+
if record.get("error"):
|
| 184 |
+
raise ValueError(f"warmup failed: {record['error']}; details saved")
|
| 185 |
+
with (directory / "predictions.jsonl").open("a") as f:
|
| 186 |
+
for case in cases[len(records) :]:
|
| 187 |
+
record = invoke(adapter, case)
|
| 188 |
+
f.write(dumps(record) + "\n")
|
| 189 |
+
f.flush()
|
| 190 |
+
records.append(record)
|
| 191 |
+
if len(records) % 50 == 0 or len(records) == len(cases):
|
| 192 |
+
print(f"Completed {len(records)}/{len(cases)} requests", flush=True)
|
| 193 |
+
finally:
|
| 194 |
+
if hasattr(adapter, "close"):
|
| 195 |
+
adapter.close()
|
| 196 |
+
result = report(cases, records, bins, tolerance)
|
| 197 |
+
result["run"] = metadata
|
| 198 |
+
warmups = read_jsonl(directory / "warmup.jsonl")
|
| 199 |
+
result["warmup"] = {
|
| 200 |
+
"requests": len(warmups),
|
| 201 |
+
"usage": [r.get("response", {}).get("usage") for r in warmups],
|
| 202 |
+
}
|
| 203 |
+
result["resolved_models"] = sorted(
|
| 204 |
+
{str(r["response"].get("model")) for r in records if isinstance(r.get("response"), dict)}
|
| 205 |
+
)
|
| 206 |
+
write_json(directory / "report.json", result)
|
| 207 |
+
(directory / "report.md").write_text(render_report(result))
|
| 208 |
+
return result
|
| 209 |
+
|
| 210 |
+
|
| 211 |
+
def score(bundle, predictions, output, bins=15, tolerance=0.02):
|
| 212 |
+
cases, manifest = load_bundle(bundle)
|
| 213 |
+
records = read_jsonl(predictions)
|
| 214 |
+
mapping = {r["case_id"]: r for r in records}
|
| 215 |
+
expected = {c["id"] for c in cases}
|
| 216 |
+
if len(mapping) != len(records) or set(mapping) - expected:
|
| 217 |
+
raise ValueError("duplicate or unknown prediction IDs")
|
| 218 |
+
complete = []
|
| 219 |
+
for case in cases:
|
| 220 |
+
r = dict(mapping.get(case["id"], {"case_id": case["id"], "error": "missing prediction"}))
|
| 221 |
+
r["question_count"] = len(case["questions"])
|
| 222 |
+
complete.append(r)
|
| 223 |
+
result = report(cases, complete, bins, tolerance)
|
| 224 |
+
result["manifest"] = manifest
|
| 225 |
+
write_json(output, result)
|
| 226 |
+
return result
|
| 227 |
+
|
| 228 |
+
|
| 229 |
+
def compare(paths):
|
| 230 |
+
reports = [json.loads(Path(p).read_text()) for p in paths]
|
| 231 |
+
protocols = {
|
| 232 |
+
(
|
| 233 |
+
r["run"]["manifest"]["sha256"],
|
| 234 |
+
r["metric_version"],
|
| 235 |
+
r["ece_bins"],
|
| 236 |
+
r["probability_sum_tolerance"],
|
| 237 |
+
)
|
| 238 |
+
for r in reports
|
| 239 |
+
}
|
| 240 |
+
if len(protocols) != 1:
|
| 241 |
+
raise ValueError("comparison requires identical frozen cases and scoring conventions")
|
| 242 |
+
lines = [
|
| 243 |
+
"| Model | Mode | Valid / attempted | Accuracy (all) | ECE | Request p50 ms |",
|
| 244 |
+
"|---|---|---:|---:|---:|---:|",
|
| 245 |
+
]
|
| 246 |
+
for r in reports:
|
| 247 |
+
c, s = r["run"]["config"], r["overall"]
|
| 248 |
+
e = "—" if s["ece_top_label"] is None else f"{s['ece_top_label']:.4f}"
|
| 249 |
+
lines.append(
|
| 250 |
+
f"| {c['name']} | {c['mode']} | {s['valid']}/{s['attempted']} | "
|
| 251 |
+
f"{s['accuracy_all']:.4f} | {e} | {r['latency']['p50_ms']} |"
|
| 252 |
+
)
|
| 253 |
+
lines += [
|
| 254 |
+
"",
|
| 255 |
+
"Training modes, runtime versions, hardware and network location remain material differences.",
|
| 256 |
+
]
|
| 257 |
+
return "\n".join(lines)
|
ariadne_bench/schema.py
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""One ordered question schema for data, native APIs and local model plugins."""
|
| 2 |
+
|
| 3 |
+
import hashlib
|
| 4 |
+
import json
|
| 5 |
+
import math
|
| 6 |
+
from pathlib import Path
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
def dumps(value):
|
| 10 |
+
# Ordering is intentional: changing option order changes the model's input.
|
| 11 |
+
return json.dumps(value, ensure_ascii=False, separators=(",", ":"), allow_nan=False)
|
| 12 |
+
|
| 13 |
+
|
| 14 |
+
def digest(value):
|
| 15 |
+
return hashlib.sha256(dumps(value).encode()).hexdigest()
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
def read_jsonl(path):
|
| 19 |
+
return [json.loads(line) for line in Path(path).read_text().splitlines() if line.strip()]
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
def write_json(path, value):
|
| 23 |
+
Path(path).write_text(json.dumps(value, ensure_ascii=False, indent=2, allow_nan=False) + "\n")
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
def labels(question):
|
| 27 |
+
kind = question["type"]
|
| 28 |
+
if kind == "noul":
|
| 29 |
+
return ["false", "true"]
|
| 30 |
+
if kind == "choice":
|
| 31 |
+
criteria = question["criteria"]
|
| 32 |
+
if not isinstance(criteria, dict) or not criteria:
|
| 33 |
+
raise ValueError("choice requires a nonempty criteria map")
|
| 34 |
+
return list(criteria)
|
| 35 |
+
if kind == "score":
|
| 36 |
+
criteria = question["criteria"]
|
| 37 |
+
if not isinstance(criteria, list) or len(criteria) < 2:
|
| 38 |
+
raise ValueError("score requires at least two ordered criteria")
|
| 39 |
+
return [str(i) for i in range(len(criteria))]
|
| 40 |
+
raise ValueError(f"unknown question type: {kind}")
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
def distribution(raw, keys, tolerance=0.02):
|
| 44 |
+
if not isinstance(raw, dict) or set(raw) != set(keys):
|
| 45 |
+
raise ValueError("probability keys must match every option exactly")
|
| 46 |
+
if any(isinstance(raw[k], bool) or not isinstance(raw[k], (int, float)) for k in keys):
|
| 47 |
+
raise ValueError("probabilities must be numbers")
|
| 48 |
+
values = [float(raw[k]) for k in keys]
|
| 49 |
+
if any(not math.isfinite(p) or p < 0 or p > 1 for p in values):
|
| 50 |
+
raise ValueError("probabilities must be finite and in [0, 1]")
|
| 51 |
+
total = sum(values)
|
| 52 |
+
if total <= 0 or abs(total - 1) > tolerance + 1e-12:
|
| 53 |
+
raise ValueError(f"probabilities sum to {total}, outside rounding tolerance {tolerance}")
|
| 54 |
+
return [p / total for p in values], total
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
def validate_case(case):
|
| 58 |
+
if not isinstance(case.get("id"), str) or not case["id"]:
|
| 59 |
+
raise ValueError("case requires a nonempty string id")
|
| 60 |
+
if not isinstance(case.get("suite"), str) or not case["suite"]:
|
| 61 |
+
raise ValueError("case requires a suite")
|
| 62 |
+
if not isinstance(case.get("state"), (str, dict, list)):
|
| 63 |
+
raise ValueError("state must be text, an object or an array")
|
| 64 |
+
questions, gold = case.get("questions"), case.get("gold")
|
| 65 |
+
if not isinstance(questions, dict) or not questions or set(questions) != set(gold or {}):
|
| 66 |
+
raise ValueError("each question needs exactly one gold entry")
|
| 67 |
+
for qid, question in questions.items():
|
| 68 |
+
if "instructions" not in question:
|
| 69 |
+
raise ValueError(f"missing instructions: {qid}")
|
| 70 |
+
keys = labels(question)
|
| 71 |
+
if gold[qid]["label"] not in keys:
|
| 72 |
+
raise ValueError(f"gold label outside options: {qid}")
|
| 73 |
+
if "probabilities" in gold[qid]:
|
| 74 |
+
distribution(gold[qid]["probabilities"], keys)
|
| 75 |
+
if "score" in gold[qid]:
|
| 76 |
+
score = gold[qid]["score"]
|
| 77 |
+
if not math.isfinite(score) or not 0 <= score <= len(keys) - 1:
|
| 78 |
+
raise ValueError(f"invalid gold score: {qid}")
|
| 79 |
+
dumps(case)
|
| 80 |
+
|
| 81 |
+
|
| 82 |
+
def normalize_answer(question, answer, tolerance=0.02):
|
| 83 |
+
if not isinstance(answer, dict) or answer.get("type") != question["type"]:
|
| 84 |
+
raise ValueError("answer type does not match question")
|
| 85 |
+
keys = labels(question)
|
| 86 |
+
if question["type"] == "noul":
|
| 87 |
+
p = answer.get("noul")
|
| 88 |
+
if isinstance(p, bool) or not isinstance(p, (int, float)):
|
| 89 |
+
raise ValueError("noul must be a probability")
|
| 90 |
+
raw = {"false": 1 - p, "true": p}
|
| 91 |
+
else:
|
| 92 |
+
raw = answer.get("probabilities")
|
| 93 |
+
probs, total = distribution(raw, keys, tolerance)
|
| 94 |
+
idx = max(range(len(keys)), key=probs.__getitem__)
|
| 95 |
+
if question["type"] == "choice":
|
| 96 |
+
chosen = answer.get("choice")
|
| 97 |
+
if chosen not in keys or probs[keys.index(chosen)] < max(probs) - 1e-12:
|
| 98 |
+
raise ValueError("choice is missing or disagrees with the distribution")
|
| 99 |
+
idx = keys.index(chosen) # retain the provider's tie break
|
| 100 |
+
if question["type"] == "score":
|
| 101 |
+
score = answer.get("score")
|
| 102 |
+
if isinstance(score, bool) or not isinstance(score, (int, float)):
|
| 103 |
+
raise ValueError("score is missing or nonnumeric")
|
| 104 |
+
if not math.isfinite(score) or not 0 <= score <= len(keys) - 1:
|
| 105 |
+
raise ValueError("score is outside its rubric")
|
| 106 |
+
expected = sum(i * p for i, p in enumerate(probs))
|
| 107 |
+
if abs(score - expected) > max(0.05, tolerance * (len(keys) - 1)):
|
| 108 |
+
raise ValueError("score disagrees with the distribution's expectation")
|
| 109 |
+
if "confidence" in answer:
|
| 110 |
+
c = answer["confidence"]
|
| 111 |
+
if (
|
| 112 |
+
isinstance(c, bool)
|
| 113 |
+
or not isinstance(c, (int, float))
|
| 114 |
+
or not math.isfinite(c)
|
| 115 |
+
or not 0 <= c <= 1
|
| 116 |
+
):
|
| 117 |
+
raise ValueError("invalid provider confidence")
|
| 118 |
+
return {
|
| 119 |
+
"label": keys[idx],
|
| 120 |
+
"probabilities": dict(zip(keys, probs)),
|
| 121 |
+
"confidence": probs[idx],
|
| 122 |
+
"raw_probability_sum": total,
|
| 123 |
+
"provider_confidence": answer.get("confidence"),
|
| 124 |
+
}
|
| 125 |
+
|
| 126 |
+
|
| 127 |
+
def save_bundle(directory, cases, metadata):
|
| 128 |
+
directory = Path(directory)
|
| 129 |
+
if directory.exists():
|
| 130 |
+
raise ValueError(f"output already exists: {directory}; use a new directory")
|
| 131 |
+
ids = set()
|
| 132 |
+
for case in cases:
|
| 133 |
+
validate_case(case)
|
| 134 |
+
if case["id"] in ids:
|
| 135 |
+
raise ValueError(f"duplicate case id: {case['id']}")
|
| 136 |
+
ids.add(case["id"])
|
| 137 |
+
if not cases:
|
| 138 |
+
raise ValueError("empty benchmark")
|
| 139 |
+
content = "".join(dumps(c) + "\n" for c in cases).encode()
|
| 140 |
+
directory.mkdir(parents=True)
|
| 141 |
+
(directory / "cases.jsonl").write_bytes(content)
|
| 142 |
+
manifest = {
|
| 143 |
+
"format_version": 1,
|
| 144 |
+
"sha256": hashlib.sha256(content).hexdigest(),
|
| 145 |
+
"cases": len(cases),
|
| 146 |
+
"decisions": sum(len(c["questions"]) for c in cases),
|
| 147 |
+
**metadata,
|
| 148 |
+
}
|
| 149 |
+
write_json(directory / "manifest.json", manifest)
|
| 150 |
+
return manifest
|
| 151 |
+
|
| 152 |
+
|
| 153 |
+
def load_bundle(directory):
|
| 154 |
+
directory = Path(directory)
|
| 155 |
+
manifest = json.loads((directory / "manifest.json").read_text())
|
| 156 |
+
content = (directory / "cases.jsonl").read_bytes()
|
| 157 |
+
if hashlib.sha256(content).hexdigest() != manifest["sha256"]:
|
| 158 |
+
raise ValueError("case file does not match its manifest hash")
|
| 159 |
+
cases = read_jsonl(directory / "cases.jsonl")
|
| 160 |
+
ids = set()
|
| 161 |
+
for case in cases:
|
| 162 |
+
validate_case(case)
|
| 163 |
+
if case["id"] in ids:
|
| 164 |
+
raise ValueError("duplicate case id")
|
| 165 |
+
ids.add(case["id"])
|
| 166 |
+
if (
|
| 167 |
+
not cases
|
| 168 |
+
or len(cases) != manifest["cases"]
|
| 169 |
+
or sum(len(c["questions"]) for c in cases) != manifest["decisions"]
|
| 170 |
+
):
|
| 171 |
+
raise ValueError("manifest counts disagree with cases")
|
| 172 |
+
return cases, manifest
|
benchmarks/sources.lock.json
ADDED
|
@@ -0,0 +1,102 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"LocalLLaMA/typed-decisions": {
|
| 3 |
+
"revision": "f7a2487edd7a043a5441a5e9ccc7fe5ddbd9ebe8",
|
| 4 |
+
"license": "apache-2.0",
|
| 5 |
+
"url": "https://huggingface.co/datasets/LocalLLaMA/typed-decisions"
|
| 6 |
+
},
|
| 7 |
+
"SetFit/enron_spam": {
|
| 8 |
+
"revision": "1916f66c89d52221ae33eb57d44498b4f3a5df22",
|
| 9 |
+
"license": "see source dataset card",
|
| 10 |
+
"url": "https://huggingface.co/datasets/SetFit/enron_spam"
|
| 11 |
+
},
|
| 12 |
+
"SetFit/sst5": {
|
| 13 |
+
"revision": "e51bdcd8cd3a30da231967c1a249ba59361279a3",
|
| 14 |
+
"license": "see source dataset card",
|
| 15 |
+
"url": "https://huggingface.co/datasets/SetFit/sst5"
|
| 16 |
+
},
|
| 17 |
+
"Tobi-Bueck/customer-support-tickets": {
|
| 18 |
+
"revision": "ddf1c81a5475992c4fa6752bf1e8b4e31f07bbeb",
|
| 19 |
+
"license": "cc-by-nc-4.0",
|
| 20 |
+
"url": "https://huggingface.co/datasets/Tobi-Bueck/customer-support-tickets"
|
| 21 |
+
},
|
| 22 |
+
"btzsc/btzsc": {
|
| 23 |
+
"revision": "fef2a2ac62b69c58670047dddf045c53d7c3cb5e",
|
| 24 |
+
"license": "see source dataset card",
|
| 25 |
+
"url": "https://huggingface.co/datasets/btzsc/btzsc"
|
| 26 |
+
},
|
| 27 |
+
"dair-ai/emotion": {
|
| 28 |
+
"revision": "cab853a1dbdf4c42c2b3ef2173804746df8825fe",
|
| 29 |
+
"license": [
|
| 30 |
+
"other"
|
| 31 |
+
],
|
| 32 |
+
"url": "https://huggingface.co/datasets/dair-ai/emotion"
|
| 33 |
+
},
|
| 34 |
+
"deepset/prompt-injections": {
|
| 35 |
+
"revision": "4f61ecb038e9c3fb77e21034b22511b523772cdd",
|
| 36 |
+
"license": "apache-2.0",
|
| 37 |
+
"url": "https://huggingface.co/datasets/deepset/prompt-injections"
|
| 38 |
+
},
|
| 39 |
+
"facebook/xnli": {
|
| 40 |
+
"revision": "b8dd5d7af51114dbda02c0e3f6133f332186418e",
|
| 41 |
+
"license": "see source dataset card",
|
| 42 |
+
"url": "https://huggingface.co/datasets/facebook/xnli"
|
| 43 |
+
},
|
| 44 |
+
"fancyzhx/ag_news": {
|
| 45 |
+
"revision": "eb185aade064a813bc0b7f42de02595523103ca4",
|
| 46 |
+
"license": [
|
| 47 |
+
"unknown"
|
| 48 |
+
],
|
| 49 |
+
"url": "https://huggingface.co/datasets/fancyzhx/ag_news"
|
| 50 |
+
},
|
| 51 |
+
"google-research-datasets/mbpp": {
|
| 52 |
+
"revision": "4bb6404fdc6cacfda99d4ac4205087b89d32030c",
|
| 53 |
+
"license": [
|
| 54 |
+
"cc-by-4.0"
|
| 55 |
+
],
|
| 56 |
+
"url": "https://huggingface.co/datasets/google-research-datasets/mbpp"
|
| 57 |
+
},
|
| 58 |
+
"google/boolq": {
|
| 59 |
+
"revision": "35b264d03638db9f4ce671b711558bf7ff0f80d5",
|
| 60 |
+
"license": [
|
| 61 |
+
"cc-by-sa-3.0"
|
| 62 |
+
],
|
| 63 |
+
"url": "https://huggingface.co/datasets/google/boolq"
|
| 64 |
+
},
|
| 65 |
+
"lmsys/toxic-chat": {
|
| 66 |
+
"revision": "29df8e4dba60e1f4af4b4075c0705c5b313548a8",
|
| 67 |
+
"license": "cc-by-nc-4.0",
|
| 68 |
+
"url": "https://huggingface.co/datasets/lmsys/toxic-chat"
|
| 69 |
+
},
|
| 70 |
+
"microsoft/ms_marco": {
|
| 71 |
+
"revision": "a47ee7aae8d7d466ba15f9f0bfac3b3681087b3a",
|
| 72 |
+
"license": "see source dataset card",
|
| 73 |
+
"url": "https://huggingface.co/datasets/microsoft/ms_marco"
|
| 74 |
+
},
|
| 75 |
+
"mteb/amazon_massive_intent": {
|
| 76 |
+
"revision": "940fd47a81eaa7f2cc7b129674d945d618ac38c2",
|
| 77 |
+
"license": "apache-2.0",
|
| 78 |
+
"url": "https://huggingface.co/datasets/mteb/amazon_massive_intent"
|
| 79 |
+
},
|
| 80 |
+
"mteb/amazon_massive_scenario": {
|
| 81 |
+
"revision": "58871793b91addb7c5f7afff26ccf08737fb6697",
|
| 82 |
+
"license": "apache-2.0",
|
| 83 |
+
"url": "https://huggingface.co/datasets/mteb/amazon_massive_scenario"
|
| 84 |
+
},
|
| 85 |
+
"mteb/banking77": {
|
| 86 |
+
"revision": "18072d2685ea682290f7b8924d94c62acc19c0b2",
|
| 87 |
+
"license": "mit",
|
| 88 |
+
"url": "https://huggingface.co/datasets/mteb/banking77"
|
| 89 |
+
},
|
| 90 |
+
"openai/gsm8k": {
|
| 91 |
+
"revision": "740312add88f781978c0658806c59bc2815b9866",
|
| 92 |
+
"license": [
|
| 93 |
+
"mit"
|
| 94 |
+
],
|
| 95 |
+
"url": "https://huggingface.co/datasets/openai/gsm8k"
|
| 96 |
+
},
|
| 97 |
+
"zefang-liu/phishing-email-dataset": {
|
| 98 |
+
"revision": "34085a032c123ca237f314a01a67909cdea35e34",
|
| 99 |
+
"license": "lgpl-3.0",
|
| 100 |
+
"url": "https://huggingface.co/datasets/zefang-liu/phishing-email-dataset"
|
| 101 |
+
}
|
| 102 |
+
}
|
environment.json
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"torch_version": "2.13.0+cu130",
|
| 3 |
+
"transformers_version": "5.17.0",
|
| 4 |
+
"laya_version": "0.3.20",
|
| 5 |
+
"dtype": "torch.bfloat16",
|
| 6 |
+
"gpu_name": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
|
| 7 |
+
"reproducibility": {
|
| 8 |
+
"deterministic_algorithms": true,
|
| 9 |
+
"warn_only": false,
|
| 10 |
+
"cublas_workspace_config": ":4096:8",
|
| 11 |
+
"python_hash_seed": "0",
|
| 12 |
+
"cudnn_benchmark": false,
|
| 13 |
+
"cudnn_deterministic": true,
|
| 14 |
+
"cudnn_allow_tf32": false,
|
| 15 |
+
"float32_matmul_precision": "highest",
|
| 16 |
+
"cudnn_sdp_enabled": false,
|
| 17 |
+
"flash_sdp_enabled": true,
|
| 18 |
+
"mem_efficient_sdp_enabled": true,
|
| 19 |
+
"math_sdp_enabled": true,
|
| 20 |
+
"cuda_version": "13.0",
|
| 21 |
+
"cudnn_version": 92000,
|
| 22 |
+
"scope": "fixed hardware/runtime; cross-platform bitwise agreement is not promised"
|
| 23 |
+
},
|
| 24 |
+
"benchmark_runtime": {
|
| 25 |
+
"python": "3.12.3",
|
| 26 |
+
"platform": "Linux-7.0.0-34-generic-x86_64-with-glibc2.39",
|
| 27 |
+
"processor": "x86_64",
|
| 28 |
+
"packages": {
|
| 29 |
+
"laya": "0.3.20",
|
| 30 |
+
"torch": "2.13.0+cu130",
|
| 31 |
+
"transformers": "5.17.0",
|
| 32 |
+
"datasets": "5.0.1",
|
| 33 |
+
"huggingface-hub": "1.33.0"
|
| 34 |
+
},
|
| 35 |
+
"harness": "0.1.0"
|
| 36 |
+
},
|
| 37 |
+
"process_environment": {
|
| 38 |
+
"PYTHONHASHSEED": "0",
|
| 39 |
+
"CUBLAS_WORKSPACE_CONFIG": ":4096:8",
|
| 40 |
+
"USE_TF": "0",
|
| 41 |
+
"OMP_NUM_THREADS": "4",
|
| 42 |
+
"TOKENIZERS_PARALLELISM": "false"
|
| 43 |
+
}
|
| 44 |
+
}
|
evidence/adapter-equivalence.json
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"device": "cuda:1",
|
| 3 |
+
"deterministic_algorithms": true,
|
| 4 |
+
"trained_checkpoint_reference": "runs/base-frozen-linear-seeds/seed-0/training/training.json",
|
| 5 |
+
"diagnostic_only_no_optimizer_updates": true,
|
| 6 |
+
"gpu": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
|
| 7 |
+
"forward_passed": true,
|
| 8 |
+
"all_frozen_modules_in_eval_mode": true,
|
| 9 |
+
"native_attention_implementation": "sdpa",
|
| 10 |
+
"all_gradients_exact": true,
|
| 11 |
+
"comparisons": [
|
| 12 |
+
{
|
| 13 |
+
"stage": "initial",
|
| 14 |
+
"decisions": 8,
|
| 15 |
+
"loss_difference": 0.0,
|
| 16 |
+
"max_logit_difference": 0.0,
|
| 17 |
+
"max_gradient_difference": 0.0,
|
| 18 |
+
"same_wrapper_repeat_gradient_difference": 0.0,
|
| 19 |
+
"legacy_repeat_gradient_difference": 0.0,
|
| 20 |
+
"max_gradient_magnitude": 0.4774176776409149,
|
| 21 |
+
"cuda_rng_unchanged": true,
|
| 22 |
+
"forward_exact": true,
|
| 23 |
+
"exact_match": true
|
| 24 |
+
},
|
| 25 |
+
{
|
| 26 |
+
"stage": "trained_seed_0",
|
| 27 |
+
"decisions": 8,
|
| 28 |
+
"loss_difference": 0.0,
|
| 29 |
+
"max_logit_difference": 0.0,
|
| 30 |
+
"max_gradient_difference": 0.0,
|
| 31 |
+
"same_wrapper_repeat_gradient_difference": 0.0,
|
| 32 |
+
"legacy_repeat_gradient_difference": 0.0,
|
| 33 |
+
"max_gradient_magnitude": 0.01077677309513092,
|
| 34 |
+
"cuda_rng_unchanged": true,
|
| 35 |
+
"forward_exact": true,
|
| 36 |
+
"exact_match": true
|
| 37 |
+
}
|
| 38 |
+
]
|
| 39 |
+
}
|
evidence/audit.json
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"frozen_state_sha256": "7560d83b1e1c17cce7eb67b25674fc8e7f0958bac4b47f559c725f8695c4d59b",
|
| 3 |
+
"verified_decisions": 25580,
|
| 4 |
+
"initial_identity_verified": true,
|
| 5 |
+
"runs": [
|
| 6 |
+
{
|
| 7 |
+
"seed": 0,
|
| 8 |
+
"checkpoint_sha256": "bcb37755de54782bc4d1f79615539f48940aedb2b733dadc1033fba2f2304cf8",
|
| 9 |
+
"checkpoint_bytes": 4198616,
|
| 10 |
+
"all_pretrained_tensors_unchanged": true,
|
| 11 |
+
"independent_scoring_matches": true,
|
| 12 |
+
"min_epochs": 10,
|
| 13 |
+
"stop_epoch": 23,
|
| 14 |
+
"minimum_epoch_rule_verified": true
|
| 15 |
+
},
|
| 16 |
+
{
|
| 17 |
+
"seed": 1,
|
| 18 |
+
"checkpoint_sha256": "e5ec5e1d494509f63960b6949af15be8390442271805ba45d796a1911b5c239a",
|
| 19 |
+
"checkpoint_bytes": 4198616,
|
| 20 |
+
"all_pretrained_tensors_unchanged": true,
|
| 21 |
+
"independent_scoring_matches": true,
|
| 22 |
+
"min_epochs": 10,
|
| 23 |
+
"stop_epoch": 17,
|
| 24 |
+
"minimum_epoch_rule_verified": true
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"seed": 2,
|
| 28 |
+
"checkpoint_sha256": "61391c3869c1f4a8449444db6a9e3b660692eefa25e3c149cee0d6af5daf264c",
|
| 29 |
+
"checkpoint_bytes": 4198616,
|
| 30 |
+
"all_pretrained_tensors_unchanged": true,
|
| 31 |
+
"independent_scoring_matches": true,
|
| 32 |
+
"min_epochs": 10,
|
| 33 |
+
"stop_epoch": 18,
|
| 34 |
+
"minimum_epoch_rule_verified": true
|
| 35 |
+
},
|
| 36 |
+
{
|
| 37 |
+
"seed": 3,
|
| 38 |
+
"checkpoint_sha256": "26ed2443b8ea146d3fab98e94ab55356557b9ef10ac5b8ba656b1a37cf4f0f64",
|
| 39 |
+
"checkpoint_bytes": 4198616,
|
| 40 |
+
"all_pretrained_tensors_unchanged": true,
|
| 41 |
+
"independent_scoring_matches": true,
|
| 42 |
+
"min_epochs": 10,
|
| 43 |
+
"stop_epoch": 27,
|
| 44 |
+
"minimum_epoch_rule_verified": true
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"seed": 4,
|
| 48 |
+
"checkpoint_sha256": "603a83ca4996e32b4753250ea6b07856e7cc5196b41e635cdad35ee4270eca38",
|
| 49 |
+
"checkpoint_bytes": 4198616,
|
| 50 |
+
"all_pretrained_tensors_unchanged": true,
|
| 51 |
+
"independent_scoring_matches": true,
|
| 52 |
+
"min_epochs": 10,
|
| 53 |
+
"stop_epoch": 10,
|
| 54 |
+
"minimum_epoch_rule_verified": true
|
| 55 |
+
}
|
| 56 |
+
],
|
| 57 |
+
"verified_runs": 5,
|
| 58 |
+
"statistics_verified": true
|
| 59 |
+
}
|
evidence/candidate-uncertainty.json
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"candidate_seed": 1,
|
| 3 |
+
"selected_epoch": 14,
|
| 4 |
+
"benchmark_cases_sha256": "10fb671245cdac1ee09e9c4d2b9b09829fbf1abd06e0954fddee0b9ec8519854",
|
| 5 |
+
"resamples": 10000,
|
| 6 |
+
"bootstrap_seed_per_suite": 42,
|
| 7 |
+
"unit": "cases, preserving all decisions in each case",
|
| 8 |
+
"interval": "paired percentile 95%; percentage points",
|
| 9 |
+
"limits": "descriptive resampling of the measured cases for a fixed model; no multiple-comparison correction or training-seed uncertainty; not used for checkpoint selection",
|
| 10 |
+
"suites": {
|
| 11 |
+
"typed-decisions": {
|
| 12 |
+
"cases": 400,
|
| 13 |
+
"accuracy_difference_pp": 40.7,
|
| 14 |
+
"case_bootstrap_95_interval_pp": [
|
| 15 |
+
37.3,
|
| 16 |
+
44.101249999999986
|
| 17 |
+
]
|
| 18 |
+
},
|
| 19 |
+
"ag-news": {
|
| 20 |
+
"cases": 600,
|
| 21 |
+
"accuracy_difference_pp": -1.6666666666666667,
|
| 22 |
+
"case_bootstrap_95_interval_pp": [
|
| 23 |
+
-3.5,
|
| 24 |
+
0.16666666666666666
|
| 25 |
+
]
|
| 26 |
+
},
|
| 27 |
+
"boolq": {
|
| 28 |
+
"cases": 600,
|
| 29 |
+
"accuracy_difference_pp": -3.1666666666666665,
|
| 30 |
+
"case_bootstrap_95_interval_pp": [
|
| 31 |
+
-5.833333333333333,
|
| 32 |
+
-0.6666666666666666
|
| 33 |
+
]
|
| 34 |
+
},
|
| 35 |
+
"emotion": {
|
| 36 |
+
"cases": 600,
|
| 37 |
+
"accuracy_difference_pp": 0.0,
|
| 38 |
+
"case_bootstrap_95_interval_pp": [
|
| 39 |
+
-2.5,
|
| 40 |
+
2.6666666666666665
|
| 41 |
+
]
|
| 42 |
+
},
|
| 43 |
+
"prompt-injections": {
|
| 44 |
+
"cases": 116,
|
| 45 |
+
"accuracy_difference_pp": 2.586206896551724,
|
| 46 |
+
"case_bootstrap_95_interval_pp": [
|
| 47 |
+
-5.172413793103448,
|
| 48 |
+
10.344827586206897
|
| 49 |
+
]
|
| 50 |
+
},
|
| 51 |
+
"sst5": {
|
| 52 |
+
"cases": 600,
|
| 53 |
+
"accuracy_difference_pp": 5.0,
|
| 54 |
+
"case_bootstrap_95_interval_pp": [
|
| 55 |
+
1.1666666666666667,
|
| 56 |
+
9.0
|
| 57 |
+
]
|
| 58 |
+
},
|
| 59 |
+
"massive-intent.en": {
|
| 60 |
+
"cases": 300,
|
| 61 |
+
"accuracy_difference_pp": -4.333333333333333,
|
| 62 |
+
"case_bootstrap_95_interval_pp": [
|
| 63 |
+
-8.333333333333334,
|
| 64 |
+
-0.3333333333333333
|
| 65 |
+
]
|
| 66 |
+
},
|
| 67 |
+
"xnli.en": {
|
| 68 |
+
"cases": 300,
|
| 69 |
+
"accuracy_difference_pp": 1.6666666666666667,
|
| 70 |
+
"case_bootstrap_95_interval_pp": [
|
| 71 |
+
-1.0,
|
| 72 |
+
4.666666666666667
|
| 73 |
+
]
|
| 74 |
+
}
|
| 75 |
+
}
|
| 76 |
+
}
|
evidence/identity-check.json
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"passed": true,
|
| 3 |
+
"cases": 3516,
|
| 4 |
+
"decisions": 5116,
|
| 5 |
+
"mismatched_cases": []
|
| 6 |
+
}
|
evidence/replay-check.json
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"passed": true,
|
| 3 |
+
"seed": 0,
|
| 4 |
+
"separate_processes": true,
|
| 5 |
+
"epochs_per_process": 2,
|
| 6 |
+
"optimizer_updates_per_process": 338,
|
| 7 |
+
"initial_validation_exact": true,
|
| 8 |
+
"epoch_losses_and_metrics_exact": true,
|
| 9 |
+
"selected_checkpoint_tensors_bitwise_equal": true,
|
| 10 |
+
"checkpoint_sha256": [
|
| 11 |
+
"3a30ce9881420cebc61829e52b41a9e669a114b4dffe0f8f836831572bd70331",
|
| 12 |
+
"3a30ce9881420cebc61829e52b41a9e669a114b4dffe0f8f836831572bd70331"
|
| 13 |
+
],
|
| 14 |
+
"reproducibility": {
|
| 15 |
+
"deterministic_algorithms": true,
|
| 16 |
+
"warn_only": false,
|
| 17 |
+
"cublas_workspace_config": ":4096:8",
|
| 18 |
+
"python_hash_seed": "0",
|
| 19 |
+
"cudnn_benchmark": false,
|
| 20 |
+
"cudnn_deterministic": true,
|
| 21 |
+
"cudnn_allow_tf32": false,
|
| 22 |
+
"float32_matmul_precision": "highest",
|
| 23 |
+
"cudnn_sdp_enabled": false,
|
| 24 |
+
"flash_sdp_enabled": true,
|
| 25 |
+
"mem_efficient_sdp_enabled": true,
|
| 26 |
+
"math_sdp_enabled": true,
|
| 27 |
+
"cuda_version": "13.0",
|
| 28 |
+
"cudnn_version": 92000,
|
| 29 |
+
"scope": "fixed hardware/runtime; cross-platform bitwise agreement is not promised"
|
| 30 |
+
},
|
| 31 |
+
"limit": "replay verified for this data, device, runtime, and two-epoch prefix"
|
| 32 |
+
}
|
evidence/seed-0-history.json
ADDED
|
@@ -0,0 +1,416 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"seed": 0,
|
| 3 |
+
"initial_validation": {
|
| 4 |
+
"soft_cross_entropy": 1.544869564374288,
|
| 5 |
+
"accuracy": 0.36833333333333335,
|
| 6 |
+
"brier_soft": 0.43051501592000324,
|
| 7 |
+
"decisions": 600
|
| 8 |
+
},
|
| 9 |
+
"epochs": [
|
| 10 |
+
{
|
| 11 |
+
"epoch": 1,
|
| 12 |
+
"train_soft_cross_entropy": 1.1371864740936843,
|
| 13 |
+
"validation": {
|
| 14 |
+
"soft_cross_entropy": 1.195652896563212,
|
| 15 |
+
"accuracy": 0.3433333333333333,
|
| 16 |
+
"brier_soft": 0.2543176457285881,
|
| 17 |
+
"decisions": 600
|
| 18 |
+
},
|
| 19 |
+
"early_stopping": {
|
| 20 |
+
"patience": 3,
|
| 21 |
+
"min_epochs": 10,
|
| 22 |
+
"epochs_without_improvement": 0,
|
| 23 |
+
"best_validation_loss": 1.195652896563212,
|
| 24 |
+
"improved": true
|
| 25 |
+
}
|
| 26 |
+
},
|
| 27 |
+
{
|
| 28 |
+
"epoch": 2,
|
| 29 |
+
"train_soft_cross_entropy": 1.1867035720966481,
|
| 30 |
+
"validation": {
|
| 31 |
+
"soft_cross_entropy": 1.1503166087468466,
|
| 32 |
+
"accuracy": 0.43666666666666665,
|
| 33 |
+
"brier_soft": 0.2273359453678131,
|
| 34 |
+
"decisions": 600
|
| 35 |
+
},
|
| 36 |
+
"early_stopping": {
|
| 37 |
+
"patience": 3,
|
| 38 |
+
"min_epochs": 10,
|
| 39 |
+
"epochs_without_improvement": 0,
|
| 40 |
+
"best_validation_loss": 1.1503166087468466,
|
| 41 |
+
"improved": true
|
| 42 |
+
}
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"epoch": 3,
|
| 46 |
+
"train_soft_cross_entropy": 1.1479326088340194,
|
| 47 |
+
"validation": {
|
| 48 |
+
"soft_cross_entropy": 1.1359616955121359,
|
| 49 |
+
"accuracy": 0.4583333333333333,
|
| 50 |
+
"brier_soft": 0.21933125893274943,
|
| 51 |
+
"decisions": 600
|
| 52 |
+
},
|
| 53 |
+
"early_stopping": {
|
| 54 |
+
"patience": 3,
|
| 55 |
+
"min_epochs": 10,
|
| 56 |
+
"epochs_without_improvement": 0,
|
| 57 |
+
"best_validation_loss": 1.1359616955121359,
|
| 58 |
+
"improved": true
|
| 59 |
+
}
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"epoch": 4,
|
| 63 |
+
"train_soft_cross_entropy": 1.17192115077266,
|
| 64 |
+
"validation": {
|
| 65 |
+
"soft_cross_entropy": 1.1432366434733072,
|
| 66 |
+
"accuracy": 0.4533333333333333,
|
| 67 |
+
"brier_soft": 0.22181885927915573,
|
| 68 |
+
"decisions": 600
|
| 69 |
+
},
|
| 70 |
+
"early_stopping": {
|
| 71 |
+
"patience": 3,
|
| 72 |
+
"min_epochs": 10,
|
| 73 |
+
"epochs_without_improvement": 1,
|
| 74 |
+
"best_validation_loss": 1.1359616955121359,
|
| 75 |
+
"improved": false
|
| 76 |
+
}
|
| 77 |
+
},
|
| 78 |
+
{
|
| 79 |
+
"epoch": 5,
|
| 80 |
+
"train_soft_cross_entropy": 1.1521123818114951,
|
| 81 |
+
"validation": {
|
| 82 |
+
"soft_cross_entropy": 1.1532048479715984,
|
| 83 |
+
"accuracy": 0.42,
|
| 84 |
+
"brier_soft": 0.22427557816108068,
|
| 85 |
+
"decisions": 600
|
| 86 |
+
},
|
| 87 |
+
"early_stopping": {
|
| 88 |
+
"patience": 3,
|
| 89 |
+
"min_epochs": 10,
|
| 90 |
+
"epochs_without_improvement": 2,
|
| 91 |
+
"best_validation_loss": 1.1359616955121359,
|
| 92 |
+
"improved": false
|
| 93 |
+
}
|
| 94 |
+
},
|
| 95 |
+
{
|
| 96 |
+
"epoch": 6,
|
| 97 |
+
"train_soft_cross_entropy": 1.139600450904281,
|
| 98 |
+
"validation": {
|
| 99 |
+
"soft_cross_entropy": 1.1171702599525453,
|
| 100 |
+
"accuracy": 0.465,
|
| 101 |
+
"brier_soft": 0.21267332146565118,
|
| 102 |
+
"decisions": 600
|
| 103 |
+
},
|
| 104 |
+
"early_stopping": {
|
| 105 |
+
"patience": 3,
|
| 106 |
+
"min_epochs": 10,
|
| 107 |
+
"epochs_without_improvement": 0,
|
| 108 |
+
"best_validation_loss": 1.1171702599525453,
|
| 109 |
+
"improved": true
|
| 110 |
+
}
|
| 111 |
+
},
|
| 112 |
+
{
|
| 113 |
+
"epoch": 7,
|
| 114 |
+
"train_soft_cross_entropy": 1.1084874327094467,
|
| 115 |
+
"validation": {
|
| 116 |
+
"soft_cross_entropy": 1.101010274887085,
|
| 117 |
+
"accuracy": 0.48333333333333334,
|
| 118 |
+
"brier_soft": 0.20341493268807728,
|
| 119 |
+
"decisions": 600
|
| 120 |
+
},
|
| 121 |
+
"early_stopping": {
|
| 122 |
+
"patience": 3,
|
| 123 |
+
"min_epochs": 10,
|
| 124 |
+
"epochs_without_improvement": 0,
|
| 125 |
+
"best_validation_loss": 1.101010274887085,
|
| 126 |
+
"improved": true
|
| 127 |
+
}
|
| 128 |
+
},
|
| 129 |
+
{
|
| 130 |
+
"epoch": 8,
|
| 131 |
+
"train_soft_cross_entropy": 1.0972168815577472,
|
| 132 |
+
"validation": {
|
| 133 |
+
"soft_cross_entropy": 1.0927943936983744,
|
| 134 |
+
"accuracy": 0.4866666666666667,
|
| 135 |
+
"brier_soft": 0.20212342927853266,
|
| 136 |
+
"decisions": 600
|
| 137 |
+
},
|
| 138 |
+
"early_stopping": {
|
| 139 |
+
"patience": 3,
|
| 140 |
+
"min_epochs": 10,
|
| 141 |
+
"epochs_without_improvement": 0,
|
| 142 |
+
"best_validation_loss": 1.0927943936983744,
|
| 143 |
+
"improved": true
|
| 144 |
+
}
|
| 145 |
+
},
|
| 146 |
+
{
|
| 147 |
+
"epoch": 9,
|
| 148 |
+
"train_soft_cross_entropy": 1.0959268622045164,
|
| 149 |
+
"validation": {
|
| 150 |
+
"soft_cross_entropy": 1.0886747964223227,
|
| 151 |
+
"accuracy": 0.485,
|
| 152 |
+
"brier_soft": 0.1981955200433731,
|
| 153 |
+
"decisions": 600
|
| 154 |
+
},
|
| 155 |
+
"early_stopping": {
|
| 156 |
+
"patience": 3,
|
| 157 |
+
"min_epochs": 10,
|
| 158 |
+
"epochs_without_improvement": 0,
|
| 159 |
+
"best_validation_loss": 1.0886747964223227,
|
| 160 |
+
"improved": true
|
| 161 |
+
}
|
| 162 |
+
},
|
| 163 |
+
{
|
| 164 |
+
"epoch": 10,
|
| 165 |
+
"train_soft_cross_entropy": 1.0879168816849036,
|
| 166 |
+
"validation": {
|
| 167 |
+
"soft_cross_entropy": 1.0776302735010783,
|
| 168 |
+
"accuracy": 0.5133333333333333,
|
| 169 |
+
"brier_soft": 0.19176260660092037,
|
| 170 |
+
"decisions": 600
|
| 171 |
+
},
|
| 172 |
+
"early_stopping": {
|
| 173 |
+
"patience": 3,
|
| 174 |
+
"min_epochs": 10,
|
| 175 |
+
"epochs_without_improvement": 0,
|
| 176 |
+
"best_validation_loss": 1.0776302735010783,
|
| 177 |
+
"improved": true
|
| 178 |
+
}
|
| 179 |
+
},
|
| 180 |
+
{
|
| 181 |
+
"epoch": 11,
|
| 182 |
+
"train_soft_cross_entropy": 1.0794475747037817,
|
| 183 |
+
"validation": {
|
| 184 |
+
"soft_cross_entropy": 1.0989113728205362,
|
| 185 |
+
"accuracy": 0.49666666666666665,
|
| 186 |
+
"brier_soft": 0.20668872609734534,
|
| 187 |
+
"decisions": 600
|
| 188 |
+
},
|
| 189 |
+
"early_stopping": {
|
| 190 |
+
"patience": 3,
|
| 191 |
+
"min_epochs": 10,
|
| 192 |
+
"epochs_without_improvement": 1,
|
| 193 |
+
"best_validation_loss": 1.0776302735010783,
|
| 194 |
+
"improved": false
|
| 195 |
+
}
|
| 196 |
+
},
|
| 197 |
+
{
|
| 198 |
+
"epoch": 12,
|
| 199 |
+
"train_soft_cross_entropy": 1.0683085640271506,
|
| 200 |
+
"validation": {
|
| 201 |
+
"soft_cross_entropy": 1.0758668931325277,
|
| 202 |
+
"accuracy": 0.5016666666666667,
|
| 203 |
+
"brier_soft": 0.19317058285077413,
|
| 204 |
+
"decisions": 600
|
| 205 |
+
},
|
| 206 |
+
"early_stopping": {
|
| 207 |
+
"patience": 3,
|
| 208 |
+
"min_epochs": 10,
|
| 209 |
+
"epochs_without_improvement": 0,
|
| 210 |
+
"best_validation_loss": 1.0758668931325277,
|
| 211 |
+
"improved": true
|
| 212 |
+
}
|
| 213 |
+
},
|
| 214 |
+
{
|
| 215 |
+
"epoch": 13,
|
| 216 |
+
"train_soft_cross_entropy": 1.052984880871243,
|
| 217 |
+
"validation": {
|
| 218 |
+
"soft_cross_entropy": 1.0600487383206685,
|
| 219 |
+
"accuracy": 0.5316666666666666,
|
| 220 |
+
"brier_soft": 0.18257314254840215,
|
| 221 |
+
"decisions": 600
|
| 222 |
+
},
|
| 223 |
+
"early_stopping": {
|
| 224 |
+
"patience": 3,
|
| 225 |
+
"min_epochs": 10,
|
| 226 |
+
"epochs_without_improvement": 0,
|
| 227 |
+
"best_validation_loss": 1.0600487383206685,
|
| 228 |
+
"improved": true
|
| 229 |
+
}
|
| 230 |
+
},
|
| 231 |
+
{
|
| 232 |
+
"epoch": 14,
|
| 233 |
+
"train_soft_cross_entropy": 1.0401919462062694,
|
| 234 |
+
"validation": {
|
| 235 |
+
"soft_cross_entropy": 1.0525298221906025,
|
| 236 |
+
"accuracy": 0.5366666666666666,
|
| 237 |
+
"brier_soft": 0.1811353324353695,
|
| 238 |
+
"decisions": 600
|
| 239 |
+
},
|
| 240 |
+
"early_stopping": {
|
| 241 |
+
"patience": 3,
|
| 242 |
+
"min_epochs": 10,
|
| 243 |
+
"epochs_without_improvement": 0,
|
| 244 |
+
"best_validation_loss": 1.0525298221906025,
|
| 245 |
+
"improved": true
|
| 246 |
+
}
|
| 247 |
+
},
|
| 248 |
+
{
|
| 249 |
+
"epoch": 15,
|
| 250 |
+
"train_soft_cross_entropy": 1.0262903751267327,
|
| 251 |
+
"validation": {
|
| 252 |
+
"soft_cross_entropy": 1.0581993921597799,
|
| 253 |
+
"accuracy": 0.5233333333333333,
|
| 254 |
+
"brier_soft": 0.18550985043247542,
|
| 255 |
+
"decisions": 600
|
| 256 |
+
},
|
| 257 |
+
"early_stopping": {
|
| 258 |
+
"patience": 3,
|
| 259 |
+
"min_epochs": 10,
|
| 260 |
+
"epochs_without_improvement": 1,
|
| 261 |
+
"best_validation_loss": 1.0525298221906025,
|
| 262 |
+
"improved": false
|
| 263 |
+
}
|
| 264 |
+
},
|
| 265 |
+
{
|
| 266 |
+
"epoch": 16,
|
| 267 |
+
"train_soft_cross_entropy": 1.0127459985238534,
|
| 268 |
+
"validation": {
|
| 269 |
+
"soft_cross_entropy": 1.0545752588907877,
|
| 270 |
+
"accuracy": 0.54,
|
| 271 |
+
"brier_soft": 0.18113683501879374,
|
| 272 |
+
"decisions": 600
|
| 273 |
+
},
|
| 274 |
+
"early_stopping": {
|
| 275 |
+
"patience": 3,
|
| 276 |
+
"min_epochs": 10,
|
| 277 |
+
"epochs_without_improvement": 2,
|
| 278 |
+
"best_validation_loss": 1.0525298221906025,
|
| 279 |
+
"improved": false
|
| 280 |
+
}
|
| 281 |
+
},
|
| 282 |
+
{
|
| 283 |
+
"epoch": 17,
|
| 284 |
+
"train_soft_cross_entropy": 1.0096984642523306,
|
| 285 |
+
"validation": {
|
| 286 |
+
"soft_cross_entropy": 1.0468252456188203,
|
| 287 |
+
"accuracy": 0.5483333333333333,
|
| 288 |
+
"brier_soft": 0.17509076982736588,
|
| 289 |
+
"decisions": 600
|
| 290 |
+
},
|
| 291 |
+
"early_stopping": {
|
| 292 |
+
"patience": 3,
|
| 293 |
+
"min_epochs": 10,
|
| 294 |
+
"epochs_without_improvement": 0,
|
| 295 |
+
"best_validation_loss": 1.0468252456188203,
|
| 296 |
+
"improved": true
|
| 297 |
+
}
|
| 298 |
+
},
|
| 299 |
+
{
|
| 300 |
+
"epoch": 18,
|
| 301 |
+
"train_soft_cross_entropy": 1.0022709012914588,
|
| 302 |
+
"validation": {
|
| 303 |
+
"soft_cross_entropy": 1.0376164364814757,
|
| 304 |
+
"accuracy": 0.5766666666666667,
|
| 305 |
+
"brier_soft": 0.17226109830041728,
|
| 306 |
+
"decisions": 600
|
| 307 |
+
},
|
| 308 |
+
"early_stopping": {
|
| 309 |
+
"patience": 3,
|
| 310 |
+
"min_epochs": 10,
|
| 311 |
+
"epochs_without_improvement": 0,
|
| 312 |
+
"best_validation_loss": 1.0376164364814757,
|
| 313 |
+
"improved": true
|
| 314 |
+
}
|
| 315 |
+
},
|
| 316 |
+
{
|
| 317 |
+
"epoch": 19,
|
| 318 |
+
"train_soft_cross_entropy": 0.9934269437083492,
|
| 319 |
+
"validation": {
|
| 320 |
+
"soft_cross_entropy": 1.0526774628957112,
|
| 321 |
+
"accuracy": 0.5516666666666666,
|
| 322 |
+
"brier_soft": 0.17908830145994822,
|
| 323 |
+
"decisions": 600
|
| 324 |
+
},
|
| 325 |
+
"early_stopping": {
|
| 326 |
+
"patience": 3,
|
| 327 |
+
"min_epochs": 10,
|
| 328 |
+
"epochs_without_improvement": 1,
|
| 329 |
+
"best_validation_loss": 1.0376164364814757,
|
| 330 |
+
"improved": false
|
| 331 |
+
}
|
| 332 |
+
},
|
| 333 |
+
{
|
| 334 |
+
"epoch": 20,
|
| 335 |
+
"train_soft_cross_entropy": 0.9915926403911025,
|
| 336 |
+
"validation": {
|
| 337 |
+
"soft_cross_entropy": 1.0292588464419048,
|
| 338 |
+
"accuracy": 0.57,
|
| 339 |
+
"brier_soft": 0.16918850486477216,
|
| 340 |
+
"decisions": 600
|
| 341 |
+
},
|
| 342 |
+
"early_stopping": {
|
| 343 |
+
"patience": 3,
|
| 344 |
+
"min_epochs": 10,
|
| 345 |
+
"epochs_without_improvement": 0,
|
| 346 |
+
"best_validation_loss": 1.0292588464419048,
|
| 347 |
+
"improved": true
|
| 348 |
+
}
|
| 349 |
+
},
|
| 350 |
+
{
|
| 351 |
+
"epoch": 21,
|
| 352 |
+
"train_soft_cross_entropy": 0.9833765982698511,
|
| 353 |
+
"validation": {
|
| 354 |
+
"soft_cross_entropy": 1.0493251442909242,
|
| 355 |
+
"accuracy": 0.555,
|
| 356 |
+
"brier_soft": 0.17917159788310527,
|
| 357 |
+
"decisions": 600
|
| 358 |
+
},
|
| 359 |
+
"early_stopping": {
|
| 360 |
+
"patience": 3,
|
| 361 |
+
"min_epochs": 10,
|
| 362 |
+
"epochs_without_improvement": 1,
|
| 363 |
+
"best_validation_loss": 1.0292588464419048,
|
| 364 |
+
"improved": false
|
| 365 |
+
}
|
| 366 |
+
},
|
| 367 |
+
{
|
| 368 |
+
"epoch": 22,
|
| 369 |
+
"train_soft_cross_entropy": 0.9786927858988445,
|
| 370 |
+
"validation": {
|
| 371 |
+
"soft_cross_entropy": 1.0412022503217062,
|
| 372 |
+
"accuracy": 0.555,
|
| 373 |
+
"brier_soft": 0.17517984464764594,
|
| 374 |
+
"decisions": 600
|
| 375 |
+
},
|
| 376 |
+
"early_stopping": {
|
| 377 |
+
"patience": 3,
|
| 378 |
+
"min_epochs": 10,
|
| 379 |
+
"epochs_without_improvement": 2,
|
| 380 |
+
"best_validation_loss": 1.0292588464419048,
|
| 381 |
+
"improved": false
|
| 382 |
+
}
|
| 383 |
+
},
|
| 384 |
+
{
|
| 385 |
+
"epoch": 23,
|
| 386 |
+
"train_soft_cross_entropy": 0.9695542351404826,
|
| 387 |
+
"validation": {
|
| 388 |
+
"soft_cross_entropy": 1.0376375365257262,
|
| 389 |
+
"accuracy": 0.5583333333333333,
|
| 390 |
+
"brier_soft": 0.17371407074232897,
|
| 391 |
+
"decisions": 600
|
| 392 |
+
},
|
| 393 |
+
"early_stopping": {
|
| 394 |
+
"patience": 3,
|
| 395 |
+
"min_epochs": 10,
|
| 396 |
+
"epochs_without_improvement": 3,
|
| 397 |
+
"best_validation_loss": 1.0292588464419048,
|
| 398 |
+
"improved": false
|
| 399 |
+
}
|
| 400 |
+
}
|
| 401 |
+
],
|
| 402 |
+
"stopping": {
|
| 403 |
+
"reason": "early_stopping",
|
| 404 |
+
"epochs_completed": 23,
|
| 405 |
+
"patience": 3,
|
| 406 |
+
"min_epochs": 10,
|
| 407 |
+
"epochs_without_improvement": 3,
|
| 408 |
+
"metric": "validation.soft_cross_entropy",
|
| 409 |
+
"min_delta": 0.0
|
| 410 |
+
},
|
| 411 |
+
"trainable_names": [
|
| 412 |
+
"encoder.embeddings.interface.proj.weight",
|
| 413 |
+
"encoder.embeddings.interface.proj.bias"
|
| 414 |
+
],
|
| 415 |
+
"frozen_parameters_unchanged": true
|
| 416 |
+
}
|
evidence/seed-1-history.json
ADDED
|
@@ -0,0 +1,314 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"seed": 1,
|
| 3 |
+
"initial_validation": {
|
| 4 |
+
"soft_cross_entropy": 1.544869564374288,
|
| 5 |
+
"accuracy": 0.36833333333333335,
|
| 6 |
+
"brier_soft": 0.43051501592000324,
|
| 7 |
+
"decisions": 600
|
| 8 |
+
},
|
| 9 |
+
"epochs": [
|
| 10 |
+
{
|
| 11 |
+
"epoch": 1,
|
| 12 |
+
"train_soft_cross_entropy": 1.0002122403074194,
|
| 13 |
+
"validation": {
|
| 14 |
+
"soft_cross_entropy": 0.9555757478872935,
|
| 15 |
+
"accuracy": 0.665,
|
| 16 |
+
"brier_soft": 0.13395084381103517,
|
| 17 |
+
"decisions": 600
|
| 18 |
+
},
|
| 19 |
+
"early_stopping": {
|
| 20 |
+
"patience": 3,
|
| 21 |
+
"min_epochs": 10,
|
| 22 |
+
"epochs_without_improvement": 0,
|
| 23 |
+
"best_validation_loss": 0.9555757478872935,
|
| 24 |
+
"improved": true
|
| 25 |
+
}
|
| 26 |
+
},
|
| 27 |
+
{
|
| 28 |
+
"epoch": 2,
|
| 29 |
+
"train_soft_cross_entropy": 0.895550852616628,
|
| 30 |
+
"validation": {
|
| 31 |
+
"soft_cross_entropy": 0.8875669745604197,
|
| 32 |
+
"accuracy": 0.7216666666666667,
|
| 33 |
+
"brier_soft": 0.08987226173281669,
|
| 34 |
+
"decisions": 600
|
| 35 |
+
},
|
| 36 |
+
"early_stopping": {
|
| 37 |
+
"patience": 3,
|
| 38 |
+
"min_epochs": 10,
|
| 39 |
+
"epochs_without_improvement": 0,
|
| 40 |
+
"best_validation_loss": 0.8875669745604197,
|
| 41 |
+
"improved": true
|
| 42 |
+
}
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"epoch": 3,
|
| 46 |
+
"train_soft_cross_entropy": 0.8399523215382187,
|
| 47 |
+
"validation": {
|
| 48 |
+
"soft_cross_entropy": 0.8776892550786336,
|
| 49 |
+
"accuracy": 0.725,
|
| 50 |
+
"brier_soft": 0.0829127719004949,
|
| 51 |
+
"decisions": 600
|
| 52 |
+
},
|
| 53 |
+
"early_stopping": {
|
| 54 |
+
"patience": 3,
|
| 55 |
+
"min_epochs": 10,
|
| 56 |
+
"epochs_without_improvement": 0,
|
| 57 |
+
"best_validation_loss": 0.8776892550786336,
|
| 58 |
+
"improved": true
|
| 59 |
+
}
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"epoch": 4,
|
| 63 |
+
"train_soft_cross_entropy": 0.8150597869908368,
|
| 64 |
+
"validation": {
|
| 65 |
+
"soft_cross_entropy": 0.8523939752578735,
|
| 66 |
+
"accuracy": 0.7666666666666667,
|
| 67 |
+
"brier_soft": 0.07098765720923741,
|
| 68 |
+
"decisions": 600
|
| 69 |
+
},
|
| 70 |
+
"early_stopping": {
|
| 71 |
+
"patience": 3,
|
| 72 |
+
"min_epochs": 10,
|
| 73 |
+
"epochs_without_improvement": 0,
|
| 74 |
+
"best_validation_loss": 0.8523939752578735,
|
| 75 |
+
"improved": true
|
| 76 |
+
}
|
| 77 |
+
},
|
| 78 |
+
{
|
| 79 |
+
"epoch": 5,
|
| 80 |
+
"train_soft_cross_entropy": 0.7998251422246297,
|
| 81 |
+
"validation": {
|
| 82 |
+
"soft_cross_entropy": 0.853373521566391,
|
| 83 |
+
"accuracy": 0.745,
|
| 84 |
+
"brier_soft": 0.07141184127579132,
|
| 85 |
+
"decisions": 600
|
| 86 |
+
},
|
| 87 |
+
"early_stopping": {
|
| 88 |
+
"patience": 3,
|
| 89 |
+
"min_epochs": 10,
|
| 90 |
+
"epochs_without_improvement": 1,
|
| 91 |
+
"best_validation_loss": 0.8523939752578735,
|
| 92 |
+
"improved": false
|
| 93 |
+
}
|
| 94 |
+
},
|
| 95 |
+
{
|
| 96 |
+
"epoch": 6,
|
| 97 |
+
"train_soft_cross_entropy": 0.7924317064991704,
|
| 98 |
+
"validation": {
|
| 99 |
+
"soft_cross_entropy": 0.8612222145001094,
|
| 100 |
+
"accuracy": 0.7516666666666667,
|
| 101 |
+
"brier_soft": 0.07590873730679353,
|
| 102 |
+
"decisions": 600
|
| 103 |
+
},
|
| 104 |
+
"early_stopping": {
|
| 105 |
+
"patience": 3,
|
| 106 |
+
"min_epochs": 10,
|
| 107 |
+
"epochs_without_improvement": 2,
|
| 108 |
+
"best_validation_loss": 0.8523939752578735,
|
| 109 |
+
"improved": false
|
| 110 |
+
}
|
| 111 |
+
},
|
| 112 |
+
{
|
| 113 |
+
"epoch": 7,
|
| 114 |
+
"train_soft_cross_entropy": 0.7880261039733887,
|
| 115 |
+
"validation": {
|
| 116 |
+
"soft_cross_entropy": 0.8462035346031189,
|
| 117 |
+
"accuracy": 0.7466666666666667,
|
| 118 |
+
"brier_soft": 0.06817000946650903,
|
| 119 |
+
"decisions": 600
|
| 120 |
+
},
|
| 121 |
+
"early_stopping": {
|
| 122 |
+
"patience": 3,
|
| 123 |
+
"min_epochs": 10,
|
| 124 |
+
"epochs_without_improvement": 0,
|
| 125 |
+
"best_validation_loss": 0.8462035346031189,
|
| 126 |
+
"improved": true
|
| 127 |
+
}
|
| 128 |
+
},
|
| 129 |
+
{
|
| 130 |
+
"epoch": 8,
|
| 131 |
+
"train_soft_cross_entropy": 0.7816576555923179,
|
| 132 |
+
"validation": {
|
| 133 |
+
"soft_cross_entropy": 0.8440792632102966,
|
| 134 |
+
"accuracy": 0.7633333333333333,
|
| 135 |
+
"brier_soft": 0.06713677939027547,
|
| 136 |
+
"decisions": 600
|
| 137 |
+
},
|
| 138 |
+
"early_stopping": {
|
| 139 |
+
"patience": 3,
|
| 140 |
+
"min_epochs": 10,
|
| 141 |
+
"epochs_without_improvement": 0,
|
| 142 |
+
"best_validation_loss": 0.8440792632102966,
|
| 143 |
+
"improved": true
|
| 144 |
+
}
|
| 145 |
+
},
|
| 146 |
+
{
|
| 147 |
+
"epoch": 9,
|
| 148 |
+
"train_soft_cross_entropy": 0.7783506816404837,
|
| 149 |
+
"validation": {
|
| 150 |
+
"soft_cross_entropy": 0.8438087296485901,
|
| 151 |
+
"accuracy": 0.7683333333333333,
|
| 152 |
+
"brier_soft": 0.06672837336858113,
|
| 153 |
+
"decisions": 600
|
| 154 |
+
},
|
| 155 |
+
"early_stopping": {
|
| 156 |
+
"patience": 3,
|
| 157 |
+
"min_epochs": 10,
|
| 158 |
+
"epochs_without_improvement": 0,
|
| 159 |
+
"best_validation_loss": 0.8438087296485901,
|
| 160 |
+
"improved": true
|
| 161 |
+
}
|
| 162 |
+
},
|
| 163 |
+
{
|
| 164 |
+
"epoch": 10,
|
| 165 |
+
"train_soft_cross_entropy": 0.7754799175262451,
|
| 166 |
+
"validation": {
|
| 167 |
+
"soft_cross_entropy": 0.8468067542711893,
|
| 168 |
+
"accuracy": 0.755,
|
| 169 |
+
"brier_soft": 0.06876023932515334,
|
| 170 |
+
"decisions": 600
|
| 171 |
+
},
|
| 172 |
+
"early_stopping": {
|
| 173 |
+
"patience": 3,
|
| 174 |
+
"min_epochs": 10,
|
| 175 |
+
"epochs_without_improvement": 1,
|
| 176 |
+
"best_validation_loss": 0.8438087296485901,
|
| 177 |
+
"improved": false
|
| 178 |
+
}
|
| 179 |
+
},
|
| 180 |
+
{
|
| 181 |
+
"epoch": 11,
|
| 182 |
+
"train_soft_cross_entropy": 0.7734219932114637,
|
| 183 |
+
"validation": {
|
| 184 |
+
"soft_cross_entropy": 0.8521546524763107,
|
| 185 |
+
"accuracy": 0.7783333333333333,
|
| 186 |
+
"brier_soft": 0.07055049358556668,
|
| 187 |
+
"decisions": 600
|
| 188 |
+
},
|
| 189 |
+
"early_stopping": {
|
| 190 |
+
"patience": 3,
|
| 191 |
+
"min_epochs": 10,
|
| 192 |
+
"epochs_without_improvement": 2,
|
| 193 |
+
"best_validation_loss": 0.8438087296485901,
|
| 194 |
+
"improved": false
|
| 195 |
+
}
|
| 196 |
+
},
|
| 197 |
+
{
|
| 198 |
+
"epoch": 12,
|
| 199 |
+
"train_soft_cross_entropy": 0.7719193545094243,
|
| 200 |
+
"validation": {
|
| 201 |
+
"soft_cross_entropy": 0.842797059615453,
|
| 202 |
+
"accuracy": 0.7766666666666666,
|
| 203 |
+
"brier_soft": 0.06609537469533583,
|
| 204 |
+
"decisions": 600
|
| 205 |
+
},
|
| 206 |
+
"early_stopping": {
|
| 207 |
+
"patience": 3,
|
| 208 |
+
"min_epochs": 10,
|
| 209 |
+
"epochs_without_improvement": 0,
|
| 210 |
+
"best_validation_loss": 0.842797059615453,
|
| 211 |
+
"improved": true
|
| 212 |
+
}
|
| 213 |
+
},
|
| 214 |
+
{
|
| 215 |
+
"epoch": 13,
|
| 216 |
+
"train_soft_cross_entropy": 0.7722371352601934,
|
| 217 |
+
"validation": {
|
| 218 |
+
"soft_cross_entropy": 0.848033101161321,
|
| 219 |
+
"accuracy": 0.7633333333333333,
|
| 220 |
+
"brier_soft": 0.06775808438037832,
|
| 221 |
+
"decisions": 600
|
| 222 |
+
},
|
| 223 |
+
"early_stopping": {
|
| 224 |
+
"patience": 3,
|
| 225 |
+
"min_epochs": 10,
|
| 226 |
+
"epochs_without_improvement": 1,
|
| 227 |
+
"best_validation_loss": 0.842797059615453,
|
| 228 |
+
"improved": false
|
| 229 |
+
}
|
| 230 |
+
},
|
| 231 |
+
{
|
| 232 |
+
"epoch": 14,
|
| 233 |
+
"train_soft_cross_entropy": 0.7689926949695305,
|
| 234 |
+
"validation": {
|
| 235 |
+
"soft_cross_entropy": 0.8377540612220764,
|
| 236 |
+
"accuracy": 0.765,
|
| 237 |
+
"brier_soft": 0.06276483290052662,
|
| 238 |
+
"decisions": 600
|
| 239 |
+
},
|
| 240 |
+
"early_stopping": {
|
| 241 |
+
"patience": 3,
|
| 242 |
+
"min_epochs": 10,
|
| 243 |
+
"epochs_without_improvement": 0,
|
| 244 |
+
"best_validation_loss": 0.8377540612220764,
|
| 245 |
+
"improved": true
|
| 246 |
+
}
|
| 247 |
+
},
|
| 248 |
+
{
|
| 249 |
+
"epoch": 15,
|
| 250 |
+
"train_soft_cross_entropy": 0.7658979249000549,
|
| 251 |
+
"validation": {
|
| 252 |
+
"soft_cross_entropy": 0.8519912085930507,
|
| 253 |
+
"accuracy": 0.765,
|
| 254 |
+
"brier_soft": 0.07007166295622785,
|
| 255 |
+
"decisions": 600
|
| 256 |
+
},
|
| 257 |
+
"early_stopping": {
|
| 258 |
+
"patience": 3,
|
| 259 |
+
"min_epochs": 10,
|
| 260 |
+
"epochs_without_improvement": 1,
|
| 261 |
+
"best_validation_loss": 0.8377540612220764,
|
| 262 |
+
"improved": false
|
| 263 |
+
}
|
| 264 |
+
},
|
| 265 |
+
{
|
| 266 |
+
"epoch": 16,
|
| 267 |
+
"train_soft_cross_entropy": 0.7651108237107594,
|
| 268 |
+
"validation": {
|
| 269 |
+
"soft_cross_entropy": 0.8408213845888773,
|
| 270 |
+
"accuracy": 0.7633333333333333,
|
| 271 |
+
"brier_soft": 0.06544813718336324,
|
| 272 |
+
"decisions": 600
|
| 273 |
+
},
|
| 274 |
+
"early_stopping": {
|
| 275 |
+
"patience": 3,
|
| 276 |
+
"min_epochs": 10,
|
| 277 |
+
"epochs_without_improvement": 2,
|
| 278 |
+
"best_validation_loss": 0.8377540612220764,
|
| 279 |
+
"improved": false
|
| 280 |
+
}
|
| 281 |
+
},
|
| 282 |
+
{
|
| 283 |
+
"epoch": 17,
|
| 284 |
+
"train_soft_cross_entropy": 0.7817914198063038,
|
| 285 |
+
"validation": {
|
| 286 |
+
"soft_cross_entropy": 0.8468167595068614,
|
| 287 |
+
"accuracy": 0.7683333333333333,
|
| 288 |
+
"brier_soft": 0.06789324807934463,
|
| 289 |
+
"decisions": 600
|
| 290 |
+
},
|
| 291 |
+
"early_stopping": {
|
| 292 |
+
"patience": 3,
|
| 293 |
+
"min_epochs": 10,
|
| 294 |
+
"epochs_without_improvement": 3,
|
| 295 |
+
"best_validation_loss": 0.8377540612220764,
|
| 296 |
+
"improved": false
|
| 297 |
+
}
|
| 298 |
+
}
|
| 299 |
+
],
|
| 300 |
+
"stopping": {
|
| 301 |
+
"reason": "early_stopping",
|
| 302 |
+
"epochs_completed": 17,
|
| 303 |
+
"patience": 3,
|
| 304 |
+
"min_epochs": 10,
|
| 305 |
+
"epochs_without_improvement": 3,
|
| 306 |
+
"metric": "validation.soft_cross_entropy",
|
| 307 |
+
"min_delta": 0.0
|
| 308 |
+
},
|
| 309 |
+
"trainable_names": [
|
| 310 |
+
"encoder.embeddings.interface.proj.weight",
|
| 311 |
+
"encoder.embeddings.interface.proj.bias"
|
| 312 |
+
],
|
| 313 |
+
"frozen_parameters_unchanged": true
|
| 314 |
+
}
|
evidence/seed-2-history.json
ADDED
|
@@ -0,0 +1,331 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"seed": 2,
|
| 3 |
+
"initial_validation": {
|
| 4 |
+
"soft_cross_entropy": 1.544869564374288,
|
| 5 |
+
"accuracy": 0.36833333333333335,
|
| 6 |
+
"brier_soft": 0.43051501592000324,
|
| 7 |
+
"decisions": 600
|
| 8 |
+
},
|
| 9 |
+
"epochs": [
|
| 10 |
+
{
|
| 11 |
+
"epoch": 1,
|
| 12 |
+
"train_soft_cross_entropy": 1.1849676999339351,
|
| 13 |
+
"validation": {
|
| 14 |
+
"soft_cross_entropy": 1.1535689290364584,
|
| 15 |
+
"accuracy": 0.4266666666666667,
|
| 16 |
+
"brier_soft": 0.2362905572851499,
|
| 17 |
+
"decisions": 600
|
| 18 |
+
},
|
| 19 |
+
"early_stopping": {
|
| 20 |
+
"patience": 3,
|
| 21 |
+
"min_epochs": 10,
|
| 22 |
+
"epochs_without_improvement": 0,
|
| 23 |
+
"best_validation_loss": 1.1535689290364584,
|
| 24 |
+
"improved": true
|
| 25 |
+
}
|
| 26 |
+
},
|
| 27 |
+
{
|
| 28 |
+
"epoch": 2,
|
| 29 |
+
"train_soft_cross_entropy": 1.1660638781830117,
|
| 30 |
+
"validation": {
|
| 31 |
+
"soft_cross_entropy": 1.1240008862813313,
|
| 32 |
+
"accuracy": 0.44666666666666666,
|
| 33 |
+
"brier_soft": 0.2163119477033615,
|
| 34 |
+
"decisions": 600
|
| 35 |
+
},
|
| 36 |
+
"early_stopping": {
|
| 37 |
+
"patience": 3,
|
| 38 |
+
"min_epochs": 10,
|
| 39 |
+
"epochs_without_improvement": 0,
|
| 40 |
+
"best_validation_loss": 1.1240008862813313,
|
| 41 |
+
"improved": true
|
| 42 |
+
}
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"epoch": 3,
|
| 46 |
+
"train_soft_cross_entropy": 1.1152002820262203,
|
| 47 |
+
"validation": {
|
| 48 |
+
"soft_cross_entropy": 1.111793461640676,
|
| 49 |
+
"accuracy": 0.4816666666666667,
|
| 50 |
+
"brier_soft": 0.21397013902664186,
|
| 51 |
+
"decisions": 600
|
| 52 |
+
},
|
| 53 |
+
"early_stopping": {
|
| 54 |
+
"patience": 3,
|
| 55 |
+
"min_epochs": 10,
|
| 56 |
+
"epochs_without_improvement": 0,
|
| 57 |
+
"best_validation_loss": 1.111793461640676,
|
| 58 |
+
"improved": true
|
| 59 |
+
}
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"epoch": 4,
|
| 63 |
+
"train_soft_cross_entropy": 1.0905909954177009,
|
| 64 |
+
"validation": {
|
| 65 |
+
"soft_cross_entropy": 1.0822320612271628,
|
| 66 |
+
"accuracy": 0.49833333333333335,
|
| 67 |
+
"brier_soft": 0.19861485362052916,
|
| 68 |
+
"decisions": 600
|
| 69 |
+
},
|
| 70 |
+
"early_stopping": {
|
| 71 |
+
"patience": 3,
|
| 72 |
+
"min_epochs": 10,
|
| 73 |
+
"epochs_without_improvement": 0,
|
| 74 |
+
"best_validation_loss": 1.0822320612271628,
|
| 75 |
+
"improved": true
|
| 76 |
+
}
|
| 77 |
+
},
|
| 78 |
+
{
|
| 79 |
+
"epoch": 5,
|
| 80 |
+
"train_soft_cross_entropy": 1.0488388820047732,
|
| 81 |
+
"validation": {
|
| 82 |
+
"soft_cross_entropy": 1.0466791025797526,
|
| 83 |
+
"accuracy": 0.5383333333333333,
|
| 84 |
+
"brier_soft": 0.18209601615866025,
|
| 85 |
+
"decisions": 600
|
| 86 |
+
},
|
| 87 |
+
"early_stopping": {
|
| 88 |
+
"patience": 3,
|
| 89 |
+
"min_epochs": 10,
|
| 90 |
+
"epochs_without_improvement": 0,
|
| 91 |
+
"best_validation_loss": 1.0466791025797526,
|
| 92 |
+
"improved": true
|
| 93 |
+
}
|
| 94 |
+
},
|
| 95 |
+
{
|
| 96 |
+
"epoch": 6,
|
| 97 |
+
"train_soft_cross_entropy": 1.0179926924352292,
|
| 98 |
+
"validation": {
|
| 99 |
+
"soft_cross_entropy": 1.0302869494756062,
|
| 100 |
+
"accuracy": 0.5633333333333334,
|
| 101 |
+
"brier_soft": 0.1717192947367827,
|
| 102 |
+
"decisions": 600
|
| 103 |
+
},
|
| 104 |
+
"early_stopping": {
|
| 105 |
+
"patience": 3,
|
| 106 |
+
"min_epochs": 10,
|
| 107 |
+
"epochs_without_improvement": 0,
|
| 108 |
+
"best_validation_loss": 1.0302869494756062,
|
| 109 |
+
"improved": true
|
| 110 |
+
}
|
| 111 |
+
},
|
| 112 |
+
{
|
| 113 |
+
"epoch": 7,
|
| 114 |
+
"train_soft_cross_entropy": 1.0011521806540313,
|
| 115 |
+
"validation": {
|
| 116 |
+
"soft_cross_entropy": 1.0306965279579163,
|
| 117 |
+
"accuracy": 0.5533333333333333,
|
| 118 |
+
"brier_soft": 0.17337830337385338,
|
| 119 |
+
"decisions": 600
|
| 120 |
+
},
|
| 121 |
+
"early_stopping": {
|
| 122 |
+
"patience": 3,
|
| 123 |
+
"min_epochs": 10,
|
| 124 |
+
"epochs_without_improvement": 1,
|
| 125 |
+
"best_validation_loss": 1.0302869494756062,
|
| 126 |
+
"improved": false
|
| 127 |
+
}
|
| 128 |
+
},
|
| 129 |
+
{
|
| 130 |
+
"epoch": 8,
|
| 131 |
+
"train_soft_cross_entropy": 0.9898686430189345,
|
| 132 |
+
"validation": {
|
| 133 |
+
"soft_cross_entropy": 1.0352703229586284,
|
| 134 |
+
"accuracy": 0.5533333333333333,
|
| 135 |
+
"brier_soft": 0.17590213686227799,
|
| 136 |
+
"decisions": 600
|
| 137 |
+
},
|
| 138 |
+
"early_stopping": {
|
| 139 |
+
"patience": 3,
|
| 140 |
+
"min_epochs": 10,
|
| 141 |
+
"epochs_without_improvement": 2,
|
| 142 |
+
"best_validation_loss": 1.0302869494756062,
|
| 143 |
+
"improved": false
|
| 144 |
+
}
|
| 145 |
+
},
|
| 146 |
+
{
|
| 147 |
+
"epoch": 9,
|
| 148 |
+
"train_soft_cross_entropy": 0.9732429999775357,
|
| 149 |
+
"validation": {
|
| 150 |
+
"soft_cross_entropy": 1.0375931040445963,
|
| 151 |
+
"accuracy": 0.56,
|
| 152 |
+
"brier_soft": 0.1798627228786548,
|
| 153 |
+
"decisions": 600
|
| 154 |
+
},
|
| 155 |
+
"early_stopping": {
|
| 156 |
+
"patience": 3,
|
| 157 |
+
"min_epochs": 10,
|
| 158 |
+
"epochs_without_improvement": 3,
|
| 159 |
+
"best_validation_loss": 1.0302869494756062,
|
| 160 |
+
"improved": false
|
| 161 |
+
}
|
| 162 |
+
},
|
| 163 |
+
{
|
| 164 |
+
"epoch": 10,
|
| 165 |
+
"train_soft_cross_entropy": 0.9616305458987201,
|
| 166 |
+
"validation": {
|
| 167 |
+
"soft_cross_entropy": 1.0139233843485513,
|
| 168 |
+
"accuracy": 0.6116666666666667,
|
| 169 |
+
"brier_soft": 0.16307140870640674,
|
| 170 |
+
"decisions": 600
|
| 171 |
+
},
|
| 172 |
+
"early_stopping": {
|
| 173 |
+
"patience": 3,
|
| 174 |
+
"min_epochs": 10,
|
| 175 |
+
"epochs_without_improvement": 0,
|
| 176 |
+
"best_validation_loss": 1.0139233843485513,
|
| 177 |
+
"improved": true
|
| 178 |
+
}
|
| 179 |
+
},
|
| 180 |
+
{
|
| 181 |
+
"epoch": 11,
|
| 182 |
+
"train_soft_cross_entropy": 0.9440543747831274,
|
| 183 |
+
"validation": {
|
| 184 |
+
"soft_cross_entropy": 1.0138032054901123,
|
| 185 |
+
"accuracy": 0.6166666666666667,
|
| 186 |
+
"brier_soft": 0.165894419302543,
|
| 187 |
+
"decisions": 600
|
| 188 |
+
},
|
| 189 |
+
"early_stopping": {
|
| 190 |
+
"patience": 3,
|
| 191 |
+
"min_epochs": 10,
|
| 192 |
+
"epochs_without_improvement": 0,
|
| 193 |
+
"best_validation_loss": 1.0138032054901123,
|
| 194 |
+
"improved": true
|
| 195 |
+
}
|
| 196 |
+
},
|
| 197 |
+
{
|
| 198 |
+
"epoch": 12,
|
| 199 |
+
"train_soft_cross_entropy": 0.9309376364284091,
|
| 200 |
+
"validation": {
|
| 201 |
+
"soft_cross_entropy": 0.9929217569033305,
|
| 202 |
+
"accuracy": 0.6266666666666667,
|
| 203 |
+
"brier_soft": 0.15395031906664372,
|
| 204 |
+
"decisions": 600
|
| 205 |
+
},
|
| 206 |
+
"early_stopping": {
|
| 207 |
+
"patience": 3,
|
| 208 |
+
"min_epochs": 10,
|
| 209 |
+
"epochs_without_improvement": 0,
|
| 210 |
+
"best_validation_loss": 0.9929217569033305,
|
| 211 |
+
"improved": true
|
| 212 |
+
}
|
| 213 |
+
},
|
| 214 |
+
{
|
| 215 |
+
"epoch": 13,
|
| 216 |
+
"train_soft_cross_entropy": 0.9144250082969666,
|
| 217 |
+
"validation": {
|
| 218 |
+
"soft_cross_entropy": 0.9964303588867187,
|
| 219 |
+
"accuracy": 0.6266666666666667,
|
| 220 |
+
"brier_soft": 0.1560174826408426,
|
| 221 |
+
"decisions": 600
|
| 222 |
+
},
|
| 223 |
+
"early_stopping": {
|
| 224 |
+
"patience": 3,
|
| 225 |
+
"min_epochs": 10,
|
| 226 |
+
"epochs_without_improvement": 1,
|
| 227 |
+
"best_validation_loss": 0.9929217569033305,
|
| 228 |
+
"improved": false
|
| 229 |
+
}
|
| 230 |
+
},
|
| 231 |
+
{
|
| 232 |
+
"epoch": 14,
|
| 233 |
+
"train_soft_cross_entropy": 0.9063218560925237,
|
| 234 |
+
"validation": {
|
| 235 |
+
"soft_cross_entropy": 1.005851571559906,
|
| 236 |
+
"accuracy": 0.6133333333333333,
|
| 237 |
+
"brier_soft": 0.1608497215807438,
|
| 238 |
+
"decisions": 600
|
| 239 |
+
},
|
| 240 |
+
"early_stopping": {
|
| 241 |
+
"patience": 3,
|
| 242 |
+
"min_epochs": 10,
|
| 243 |
+
"epochs_without_improvement": 2,
|
| 244 |
+
"best_validation_loss": 0.9929217569033305,
|
| 245 |
+
"improved": false
|
| 246 |
+
}
|
| 247 |
+
},
|
| 248 |
+
{
|
| 249 |
+
"epoch": 15,
|
| 250 |
+
"train_soft_cross_entropy": 0.9014872776137458,
|
| 251 |
+
"validation": {
|
| 252 |
+
"soft_cross_entropy": 0.9867499772707621,
|
| 253 |
+
"accuracy": 0.6316666666666667,
|
| 254 |
+
"brier_soft": 0.14927835414807003,
|
| 255 |
+
"decisions": 600
|
| 256 |
+
},
|
| 257 |
+
"early_stopping": {
|
| 258 |
+
"patience": 3,
|
| 259 |
+
"min_epochs": 10,
|
| 260 |
+
"epochs_without_improvement": 0,
|
| 261 |
+
"best_validation_loss": 0.9867499772707621,
|
| 262 |
+
"improved": true
|
| 263 |
+
}
|
| 264 |
+
},
|
| 265 |
+
{
|
| 266 |
+
"epoch": 16,
|
| 267 |
+
"train_soft_cross_entropy": 0.8920391595805133,
|
| 268 |
+
"validation": {
|
| 269 |
+
"soft_cross_entropy": 0.9926939264933268,
|
| 270 |
+
"accuracy": 0.6283333333333333,
|
| 271 |
+
"brier_soft": 0.15533705055713654,
|
| 272 |
+
"decisions": 600
|
| 273 |
+
},
|
| 274 |
+
"early_stopping": {
|
| 275 |
+
"patience": 3,
|
| 276 |
+
"min_epochs": 10,
|
| 277 |
+
"epochs_without_improvement": 1,
|
| 278 |
+
"best_validation_loss": 0.9867499772707621,
|
| 279 |
+
"improved": false
|
| 280 |
+
}
|
| 281 |
+
},
|
| 282 |
+
{
|
| 283 |
+
"epoch": 17,
|
| 284 |
+
"train_soft_cross_entropy": 0.8795346831833875,
|
| 285 |
+
"validation": {
|
| 286 |
+
"soft_cross_entropy": 1.02633438428243,
|
| 287 |
+
"accuracy": 0.6583333333333333,
|
| 288 |
+
"brier_soft": 0.16615503872434298,
|
| 289 |
+
"decisions": 600
|
| 290 |
+
},
|
| 291 |
+
"early_stopping": {
|
| 292 |
+
"patience": 3,
|
| 293 |
+
"min_epochs": 10,
|
| 294 |
+
"epochs_without_improvement": 2,
|
| 295 |
+
"best_validation_loss": 0.9867499772707621,
|
| 296 |
+
"improved": false
|
| 297 |
+
}
|
| 298 |
+
},
|
| 299 |
+
{
|
| 300 |
+
"epoch": 18,
|
| 301 |
+
"train_soft_cross_entropy": 0.879185232453876,
|
| 302 |
+
"validation": {
|
| 303 |
+
"soft_cross_entropy": 1.0007199597358705,
|
| 304 |
+
"accuracy": 0.65,
|
| 305 |
+
"brier_soft": 0.15312831775595745,
|
| 306 |
+
"decisions": 600
|
| 307 |
+
},
|
| 308 |
+
"early_stopping": {
|
| 309 |
+
"patience": 3,
|
| 310 |
+
"min_epochs": 10,
|
| 311 |
+
"epochs_without_improvement": 3,
|
| 312 |
+
"best_validation_loss": 0.9867499772707621,
|
| 313 |
+
"improved": false
|
| 314 |
+
}
|
| 315 |
+
}
|
| 316 |
+
],
|
| 317 |
+
"stopping": {
|
| 318 |
+
"reason": "early_stopping",
|
| 319 |
+
"epochs_completed": 18,
|
| 320 |
+
"patience": 3,
|
| 321 |
+
"min_epochs": 10,
|
| 322 |
+
"epochs_without_improvement": 3,
|
| 323 |
+
"metric": "validation.soft_cross_entropy",
|
| 324 |
+
"min_delta": 0.0
|
| 325 |
+
},
|
| 326 |
+
"trainable_names": [
|
| 327 |
+
"encoder.embeddings.interface.proj.weight",
|
| 328 |
+
"encoder.embeddings.interface.proj.bias"
|
| 329 |
+
],
|
| 330 |
+
"frozen_parameters_unchanged": true
|
| 331 |
+
}
|
evidence/seed-3-history.json
ADDED
|
@@ -0,0 +1,484 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"seed": 3,
|
| 3 |
+
"initial_validation": {
|
| 4 |
+
"soft_cross_entropy": 1.544869564374288,
|
| 5 |
+
"accuracy": 0.36833333333333335,
|
| 6 |
+
"brier_soft": 0.43051501592000324,
|
| 7 |
+
"decisions": 600
|
| 8 |
+
},
|
| 9 |
+
"epochs": [
|
| 10 |
+
{
|
| 11 |
+
"epoch": 1,
|
| 12 |
+
"train_soft_cross_entropy": 1.1980749452555621,
|
| 13 |
+
"validation": {
|
| 14 |
+
"soft_cross_entropy": 1.16634921391805,
|
| 15 |
+
"accuracy": 0.4266666666666667,
|
| 16 |
+
"brier_soft": 0.2427163557211558,
|
| 17 |
+
"decisions": 600
|
| 18 |
+
},
|
| 19 |
+
"early_stopping": {
|
| 20 |
+
"patience": 3,
|
| 21 |
+
"min_epochs": 10,
|
| 22 |
+
"epochs_without_improvement": 0,
|
| 23 |
+
"best_validation_loss": 1.16634921391805,
|
| 24 |
+
"improved": true
|
| 25 |
+
}
|
| 26 |
+
},
|
| 27 |
+
{
|
| 28 |
+
"epoch": 2,
|
| 29 |
+
"train_soft_cross_entropy": 1.1646321240177862,
|
| 30 |
+
"validation": {
|
| 31 |
+
"soft_cross_entropy": 1.1217502697308859,
|
| 32 |
+
"accuracy": 0.4533333333333333,
|
| 33 |
+
"brier_soft": 0.2186164912581444,
|
| 34 |
+
"decisions": 600
|
| 35 |
+
},
|
| 36 |
+
"early_stopping": {
|
| 37 |
+
"patience": 3,
|
| 38 |
+
"min_epochs": 10,
|
| 39 |
+
"epochs_without_improvement": 0,
|
| 40 |
+
"best_validation_loss": 1.1217502697308859,
|
| 41 |
+
"improved": true
|
| 42 |
+
}
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"epoch": 3,
|
| 46 |
+
"train_soft_cross_entropy": 1.1056555556367944,
|
| 47 |
+
"validation": {
|
| 48 |
+
"soft_cross_entropy": 1.093972578048706,
|
| 49 |
+
"accuracy": 0.505,
|
| 50 |
+
"brier_soft": 0.199787005285422,
|
| 51 |
+
"decisions": 600
|
| 52 |
+
},
|
| 53 |
+
"early_stopping": {
|
| 54 |
+
"patience": 3,
|
| 55 |
+
"min_epochs": 10,
|
| 56 |
+
"epochs_without_improvement": 0,
|
| 57 |
+
"best_validation_loss": 1.093972578048706,
|
| 58 |
+
"improved": true
|
| 59 |
+
}
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"epoch": 4,
|
| 63 |
+
"train_soft_cross_entropy": 1.0671771099832323,
|
| 64 |
+
"validation": {
|
| 65 |
+
"soft_cross_entropy": 1.0549713468551636,
|
| 66 |
+
"accuracy": 0.5516666666666666,
|
| 67 |
+
"brier_soft": 0.18068428387244542,
|
| 68 |
+
"decisions": 600
|
| 69 |
+
},
|
| 70 |
+
"early_stopping": {
|
| 71 |
+
"patience": 3,
|
| 72 |
+
"min_epochs": 10,
|
| 73 |
+
"epochs_without_improvement": 0,
|
| 74 |
+
"best_validation_loss": 1.0549713468551636,
|
| 75 |
+
"improved": true
|
| 76 |
+
}
|
| 77 |
+
},
|
| 78 |
+
{
|
| 79 |
+
"epoch": 5,
|
| 80 |
+
"train_soft_cross_entropy": 1.0494003568755257,
|
| 81 |
+
"validation": {
|
| 82 |
+
"soft_cross_entropy": 1.0567963059743246,
|
| 83 |
+
"accuracy": 0.54,
|
| 84 |
+
"brier_soft": 0.18257476242880027,
|
| 85 |
+
"decisions": 600
|
| 86 |
+
},
|
| 87 |
+
"early_stopping": {
|
| 88 |
+
"patience": 3,
|
| 89 |
+
"min_epochs": 10,
|
| 90 |
+
"epochs_without_improvement": 1,
|
| 91 |
+
"best_validation_loss": 1.0549713468551636,
|
| 92 |
+
"improved": false
|
| 93 |
+
}
|
| 94 |
+
},
|
| 95 |
+
{
|
| 96 |
+
"epoch": 6,
|
| 97 |
+
"train_soft_cross_entropy": 1.038703513851872,
|
| 98 |
+
"validation": {
|
| 99 |
+
"soft_cross_entropy": 1.0705021890004476,
|
| 100 |
+
"accuracy": 0.525,
|
| 101 |
+
"brier_soft": 0.19120527582863966,
|
| 102 |
+
"decisions": 600
|
| 103 |
+
},
|
| 104 |
+
"early_stopping": {
|
| 105 |
+
"patience": 3,
|
| 106 |
+
"min_epochs": 10,
|
| 107 |
+
"epochs_without_improvement": 2,
|
| 108 |
+
"best_validation_loss": 1.0549713468551636,
|
| 109 |
+
"improved": false
|
| 110 |
+
}
|
| 111 |
+
},
|
| 112 |
+
{
|
| 113 |
+
"epoch": 7,
|
| 114 |
+
"train_soft_cross_entropy": 1.0313762316880404,
|
| 115 |
+
"validation": {
|
| 116 |
+
"soft_cross_entropy": 1.0447989773750306,
|
| 117 |
+
"accuracy": 0.5466666666666666,
|
| 118 |
+
"brier_soft": 0.17501757830381393,
|
| 119 |
+
"decisions": 600
|
| 120 |
+
},
|
| 121 |
+
"early_stopping": {
|
| 122 |
+
"patience": 3,
|
| 123 |
+
"min_epochs": 10,
|
| 124 |
+
"epochs_without_improvement": 0,
|
| 125 |
+
"best_validation_loss": 1.0447989773750306,
|
| 126 |
+
"improved": true
|
| 127 |
+
}
|
| 128 |
+
},
|
| 129 |
+
{
|
| 130 |
+
"epoch": 8,
|
| 131 |
+
"train_soft_cross_entropy": 1.017818791601393,
|
| 132 |
+
"validation": {
|
| 133 |
+
"soft_cross_entropy": 1.024161856174469,
|
| 134 |
+
"accuracy": 0.5816666666666667,
|
| 135 |
+
"brier_soft": 0.1630909529576699,
|
| 136 |
+
"decisions": 600
|
| 137 |
+
},
|
| 138 |
+
"early_stopping": {
|
| 139 |
+
"patience": 3,
|
| 140 |
+
"min_epochs": 10,
|
| 141 |
+
"epochs_without_improvement": 0,
|
| 142 |
+
"best_validation_loss": 1.024161856174469,
|
| 143 |
+
"improved": true
|
| 144 |
+
}
|
| 145 |
+
},
|
| 146 |
+
{
|
| 147 |
+
"epoch": 9,
|
| 148 |
+
"train_soft_cross_entropy": 0.9899537578335514,
|
| 149 |
+
"validation": {
|
| 150 |
+
"soft_cross_entropy": 1.0265704361597696,
|
| 151 |
+
"accuracy": 0.6083333333333333,
|
| 152 |
+
"brier_soft": 0.16808087141563496,
|
| 153 |
+
"decisions": 600
|
| 154 |
+
},
|
| 155 |
+
"early_stopping": {
|
| 156 |
+
"patience": 3,
|
| 157 |
+
"min_epochs": 10,
|
| 158 |
+
"epochs_without_improvement": 1,
|
| 159 |
+
"best_validation_loss": 1.024161856174469,
|
| 160 |
+
"improved": false
|
| 161 |
+
}
|
| 162 |
+
},
|
| 163 |
+
{
|
| 164 |
+
"epoch": 10,
|
| 165 |
+
"train_soft_cross_entropy": 0.9632394627288535,
|
| 166 |
+
"validation": {
|
| 167 |
+
"soft_cross_entropy": 1.0183856010437011,
|
| 168 |
+
"accuracy": 0.5983333333333334,
|
| 169 |
+
"brier_soft": 0.1608257148663203,
|
| 170 |
+
"decisions": 600
|
| 171 |
+
},
|
| 172 |
+
"early_stopping": {
|
| 173 |
+
"patience": 3,
|
| 174 |
+
"min_epochs": 10,
|
| 175 |
+
"epochs_without_improvement": 0,
|
| 176 |
+
"best_validation_loss": 1.0183856010437011,
|
| 177 |
+
"improved": true
|
| 178 |
+
}
|
| 179 |
+
},
|
| 180 |
+
{
|
| 181 |
+
"epoch": 11,
|
| 182 |
+
"train_soft_cross_entropy": 0.9342622081438701,
|
| 183 |
+
"validation": {
|
| 184 |
+
"soft_cross_entropy": 1.0024159566561381,
|
| 185 |
+
"accuracy": 0.6333333333333333,
|
| 186 |
+
"brier_soft": 0.14996731283764045,
|
| 187 |
+
"decisions": 600
|
| 188 |
+
},
|
| 189 |
+
"early_stopping": {
|
| 190 |
+
"patience": 3,
|
| 191 |
+
"min_epochs": 10,
|
| 192 |
+
"epochs_without_improvement": 0,
|
| 193 |
+
"best_validation_loss": 1.0024159566561381,
|
| 194 |
+
"improved": true
|
| 195 |
+
}
|
| 196 |
+
},
|
| 197 |
+
{
|
| 198 |
+
"epoch": 12,
|
| 199 |
+
"train_soft_cross_entropy": 0.9137524432606168,
|
| 200 |
+
"validation": {
|
| 201 |
+
"soft_cross_entropy": 0.9873138956228892,
|
| 202 |
+
"accuracy": 0.6516666666666666,
|
| 203 |
+
"brier_soft": 0.14277802929282188,
|
| 204 |
+
"decisions": 600
|
| 205 |
+
},
|
| 206 |
+
"early_stopping": {
|
| 207 |
+
"patience": 3,
|
| 208 |
+
"min_epochs": 10,
|
| 209 |
+
"epochs_without_improvement": 0,
|
| 210 |
+
"best_validation_loss": 0.9873138956228892,
|
| 211 |
+
"improved": true
|
| 212 |
+
}
|
| 213 |
+
},
|
| 214 |
+
{
|
| 215 |
+
"epoch": 13,
|
| 216 |
+
"train_soft_cross_entropy": 0.8881368139496556,
|
| 217 |
+
"validation": {
|
| 218 |
+
"soft_cross_entropy": 0.9626709421475729,
|
| 219 |
+
"accuracy": 0.6716666666666666,
|
| 220 |
+
"brier_soft": 0.12283751085400581,
|
| 221 |
+
"decisions": 600
|
| 222 |
+
},
|
| 223 |
+
"early_stopping": {
|
| 224 |
+
"patience": 3,
|
| 225 |
+
"min_epochs": 10,
|
| 226 |
+
"epochs_without_improvement": 0,
|
| 227 |
+
"best_validation_loss": 0.9626709421475729,
|
| 228 |
+
"improved": true
|
| 229 |
+
}
|
| 230 |
+
},
|
| 231 |
+
{
|
| 232 |
+
"epoch": 14,
|
| 233 |
+
"train_soft_cross_entropy": 0.865500467883216,
|
| 234 |
+
"validation": {
|
| 235 |
+
"soft_cross_entropy": 0.9580995849768321,
|
| 236 |
+
"accuracy": 0.685,
|
| 237 |
+
"brier_soft": 0.12383397127191226,
|
| 238 |
+
"decisions": 600
|
| 239 |
+
},
|
| 240 |
+
"early_stopping": {
|
| 241 |
+
"patience": 3,
|
| 242 |
+
"min_epochs": 10,
|
| 243 |
+
"epochs_without_improvement": 0,
|
| 244 |
+
"best_validation_loss": 0.9580995849768321,
|
| 245 |
+
"improved": true
|
| 246 |
+
}
|
| 247 |
+
},
|
| 248 |
+
{
|
| 249 |
+
"epoch": 15,
|
| 250 |
+
"train_soft_cross_entropy": 0.8549449095461104,
|
| 251 |
+
"validation": {
|
| 252 |
+
"soft_cross_entropy": 0.9462178750832876,
|
| 253 |
+
"accuracy": 0.6916666666666667,
|
| 254 |
+
"brier_soft": 0.1187205430244406,
|
| 255 |
+
"decisions": 600
|
| 256 |
+
},
|
| 257 |
+
"early_stopping": {
|
| 258 |
+
"patience": 3,
|
| 259 |
+
"min_epochs": 10,
|
| 260 |
+
"epochs_without_improvement": 0,
|
| 261 |
+
"best_validation_loss": 0.9462178750832876,
|
| 262 |
+
"improved": true
|
| 263 |
+
}
|
| 264 |
+
},
|
| 265 |
+
{
|
| 266 |
+
"epoch": 16,
|
| 267 |
+
"train_soft_cross_entropy": 0.8394836834183446,
|
| 268 |
+
"validation": {
|
| 269 |
+
"soft_cross_entropy": 0.9463119049866994,
|
| 270 |
+
"accuracy": 0.6933333333333334,
|
| 271 |
+
"brier_soft": 0.11875337022046248,
|
| 272 |
+
"decisions": 600
|
| 273 |
+
},
|
| 274 |
+
"early_stopping": {
|
| 275 |
+
"patience": 3,
|
| 276 |
+
"min_epochs": 10,
|
| 277 |
+
"epochs_without_improvement": 1,
|
| 278 |
+
"best_validation_loss": 0.9462178750832876,
|
| 279 |
+
"improved": false
|
| 280 |
+
}
|
| 281 |
+
},
|
| 282 |
+
{
|
| 283 |
+
"epoch": 17,
|
| 284 |
+
"train_soft_cross_entropy": 0.8286281617040987,
|
| 285 |
+
"validation": {
|
| 286 |
+
"soft_cross_entropy": 0.941375896135966,
|
| 287 |
+
"accuracy": 0.685,
|
| 288 |
+
"brier_soft": 0.11455494280904531,
|
| 289 |
+
"decisions": 600
|
| 290 |
+
},
|
| 291 |
+
"early_stopping": {
|
| 292 |
+
"patience": 3,
|
| 293 |
+
"min_epochs": 10,
|
| 294 |
+
"epochs_without_improvement": 0,
|
| 295 |
+
"best_validation_loss": 0.941375896135966,
|
| 296 |
+
"improved": true
|
| 297 |
+
}
|
| 298 |
+
},
|
| 299 |
+
{
|
| 300 |
+
"epoch": 18,
|
| 301 |
+
"train_soft_cross_entropy": 0.8222652976601212,
|
| 302 |
+
"validation": {
|
| 303 |
+
"soft_cross_entropy": 0.9331588689486185,
|
| 304 |
+
"accuracy": 0.6866666666666666,
|
| 305 |
+
"brier_soft": 0.11237604923546314,
|
| 306 |
+
"decisions": 600
|
| 307 |
+
},
|
| 308 |
+
"early_stopping": {
|
| 309 |
+
"patience": 3,
|
| 310 |
+
"min_epochs": 10,
|
| 311 |
+
"epochs_without_improvement": 0,
|
| 312 |
+
"best_validation_loss": 0.9331588689486185,
|
| 313 |
+
"improved": true
|
| 314 |
+
}
|
| 315 |
+
},
|
| 316 |
+
{
|
| 317 |
+
"epoch": 19,
|
| 318 |
+
"train_soft_cross_entropy": 0.8160834203826056,
|
| 319 |
+
"validation": {
|
| 320 |
+
"soft_cross_entropy": 0.9387085942427317,
|
| 321 |
+
"accuracy": 0.6916666666666667,
|
| 322 |
+
"brier_soft": 0.11558652246991793,
|
| 323 |
+
"decisions": 600
|
| 324 |
+
},
|
| 325 |
+
"early_stopping": {
|
| 326 |
+
"patience": 3,
|
| 327 |
+
"min_epochs": 10,
|
| 328 |
+
"epochs_without_improvement": 1,
|
| 329 |
+
"best_validation_loss": 0.9331588689486185,
|
| 330 |
+
"improved": false
|
| 331 |
+
}
|
| 332 |
+
},
|
| 333 |
+
{
|
| 334 |
+
"epoch": 20,
|
| 335 |
+
"train_soft_cross_entropy": 0.8087538785846146,
|
| 336 |
+
"validation": {
|
| 337 |
+
"soft_cross_entropy": 0.9408066284656524,
|
| 338 |
+
"accuracy": 0.6783333333333333,
|
| 339 |
+
"brier_soft": 0.11930533437679211,
|
| 340 |
+
"decisions": 600
|
| 341 |
+
},
|
| 342 |
+
"early_stopping": {
|
| 343 |
+
"patience": 3,
|
| 344 |
+
"min_epochs": 10,
|
| 345 |
+
"epochs_without_improvement": 2,
|
| 346 |
+
"best_validation_loss": 0.9331588689486185,
|
| 347 |
+
"improved": false
|
| 348 |
+
}
|
| 349 |
+
},
|
| 350 |
+
{
|
| 351 |
+
"epoch": 21,
|
| 352 |
+
"train_soft_cross_entropy": 0.8048860243956248,
|
| 353 |
+
"validation": {
|
| 354 |
+
"soft_cross_entropy": 0.9236933688322703,
|
| 355 |
+
"accuracy": 0.7016666666666667,
|
| 356 |
+
"brier_soft": 0.10792215374608835,
|
| 357 |
+
"decisions": 600
|
| 358 |
+
},
|
| 359 |
+
"early_stopping": {
|
| 360 |
+
"patience": 3,
|
| 361 |
+
"min_epochs": 10,
|
| 362 |
+
"epochs_without_improvement": 0,
|
| 363 |
+
"best_validation_loss": 0.9236933688322703,
|
| 364 |
+
"improved": true
|
| 365 |
+
}
|
| 366 |
+
},
|
| 367 |
+
{
|
| 368 |
+
"epoch": 22,
|
| 369 |
+
"train_soft_cross_entropy": 0.8001213801790167,
|
| 370 |
+
"validation": {
|
| 371 |
+
"soft_cross_entropy": 0.9347231006622314,
|
| 372 |
+
"accuracy": 0.6983333333333334,
|
| 373 |
+
"brier_soft": 0.11313647958139579,
|
| 374 |
+
"decisions": 600
|
| 375 |
+
},
|
| 376 |
+
"early_stopping": {
|
| 377 |
+
"patience": 3,
|
| 378 |
+
"min_epochs": 10,
|
| 379 |
+
"epochs_without_improvement": 1,
|
| 380 |
+
"best_validation_loss": 0.9236933688322703,
|
| 381 |
+
"improved": false
|
| 382 |
+
}
|
| 383 |
+
},
|
| 384 |
+
{
|
| 385 |
+
"epoch": 23,
|
| 386 |
+
"train_soft_cross_entropy": 0.7940637088263476,
|
| 387 |
+
"validation": {
|
| 388 |
+
"soft_cross_entropy": 0.9299098292986552,
|
| 389 |
+
"accuracy": 0.7033333333333334,
|
| 390 |
+
"brier_soft": 0.11475709093113741,
|
| 391 |
+
"decisions": 600
|
| 392 |
+
},
|
| 393 |
+
"early_stopping": {
|
| 394 |
+
"patience": 3,
|
| 395 |
+
"min_epochs": 10,
|
| 396 |
+
"epochs_without_improvement": 2,
|
| 397 |
+
"best_validation_loss": 0.9236933688322703,
|
| 398 |
+
"improved": false
|
| 399 |
+
}
|
| 400 |
+
},
|
| 401 |
+
{
|
| 402 |
+
"epoch": 24,
|
| 403 |
+
"train_soft_cross_entropy": 0.7884197377717054,
|
| 404 |
+
"validation": {
|
| 405 |
+
"soft_cross_entropy": 0.9105557099978129,
|
| 406 |
+
"accuracy": 0.715,
|
| 407 |
+
"brier_soft": 0.1013037375236551,
|
| 408 |
+
"decisions": 600
|
| 409 |
+
},
|
| 410 |
+
"early_stopping": {
|
| 411 |
+
"patience": 3,
|
| 412 |
+
"min_epochs": 10,
|
| 413 |
+
"epochs_without_improvement": 0,
|
| 414 |
+
"best_validation_loss": 0.9105557099978129,
|
| 415 |
+
"improved": true
|
| 416 |
+
}
|
| 417 |
+
},
|
| 418 |
+
{
|
| 419 |
+
"epoch": 25,
|
| 420 |
+
"train_soft_cross_entropy": 0.784954049454795,
|
| 421 |
+
"validation": {
|
| 422 |
+
"soft_cross_entropy": 0.923290346066157,
|
| 423 |
+
"accuracy": 0.71,
|
| 424 |
+
"brier_soft": 0.10530036373684803,
|
| 425 |
+
"decisions": 600
|
| 426 |
+
},
|
| 427 |
+
"early_stopping": {
|
| 428 |
+
"patience": 3,
|
| 429 |
+
"min_epochs": 10,
|
| 430 |
+
"epochs_without_improvement": 1,
|
| 431 |
+
"best_validation_loss": 0.9105557099978129,
|
| 432 |
+
"improved": false
|
| 433 |
+
}
|
| 434 |
+
},
|
| 435 |
+
{
|
| 436 |
+
"epoch": 26,
|
| 437 |
+
"train_soft_cross_entropy": 0.7828840752442677,
|
| 438 |
+
"validation": {
|
| 439 |
+
"soft_cross_entropy": 0.9299345835049947,
|
| 440 |
+
"accuracy": 0.7066666666666667,
|
| 441 |
+
"brier_soft": 0.11173659774164359,
|
| 442 |
+
"decisions": 600
|
| 443 |
+
},
|
| 444 |
+
"early_stopping": {
|
| 445 |
+
"patience": 3,
|
| 446 |
+
"min_epochs": 10,
|
| 447 |
+
"epochs_without_improvement": 2,
|
| 448 |
+
"best_validation_loss": 0.9105557099978129,
|
| 449 |
+
"improved": false
|
| 450 |
+
}
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"epoch": 27,
|
| 454 |
+
"train_soft_cross_entropy": 0.7790651953661883,
|
| 455 |
+
"validation": {
|
| 456 |
+
"soft_cross_entropy": 0.9137676254908244,
|
| 457 |
+
"accuracy": 0.7216666666666667,
|
| 458 |
+
"brier_soft": 0.10172271362195412,
|
| 459 |
+
"decisions": 600
|
| 460 |
+
},
|
| 461 |
+
"early_stopping": {
|
| 462 |
+
"patience": 3,
|
| 463 |
+
"min_epochs": 10,
|
| 464 |
+
"epochs_without_improvement": 3,
|
| 465 |
+
"best_validation_loss": 0.9105557099978129,
|
| 466 |
+
"improved": false
|
| 467 |
+
}
|
| 468 |
+
}
|
| 469 |
+
],
|
| 470 |
+
"stopping": {
|
| 471 |
+
"reason": "early_stopping",
|
| 472 |
+
"epochs_completed": 27,
|
| 473 |
+
"patience": 3,
|
| 474 |
+
"min_epochs": 10,
|
| 475 |
+
"epochs_without_improvement": 3,
|
| 476 |
+
"metric": "validation.soft_cross_entropy",
|
| 477 |
+
"min_delta": 0.0
|
| 478 |
+
},
|
| 479 |
+
"trainable_names": [
|
| 480 |
+
"encoder.embeddings.interface.proj.weight",
|
| 481 |
+
"encoder.embeddings.interface.proj.bias"
|
| 482 |
+
],
|
| 483 |
+
"frozen_parameters_unchanged": true
|
| 484 |
+
}
|
evidence/seed-4-history.json
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"seed": 4,
|
| 3 |
+
"initial_validation": {
|
| 4 |
+
"soft_cross_entropy": 1.544869564374288,
|
| 5 |
+
"accuracy": 0.36833333333333335,
|
| 6 |
+
"brier_soft": 0.43051501592000324,
|
| 7 |
+
"decisions": 600
|
| 8 |
+
},
|
| 9 |
+
"epochs": [
|
| 10 |
+
{
|
| 11 |
+
"epoch": 1,
|
| 12 |
+
"train_soft_cross_entropy": 1.002756692568461,
|
| 13 |
+
"validation": {
|
| 14 |
+
"soft_cross_entropy": 0.9602992224693299,
|
| 15 |
+
"accuracy": 0.6783333333333333,
|
| 16 |
+
"brier_soft": 0.1296700432151556,
|
| 17 |
+
"decisions": 600
|
| 18 |
+
},
|
| 19 |
+
"early_stopping": {
|
| 20 |
+
"patience": 3,
|
| 21 |
+
"min_epochs": 10,
|
| 22 |
+
"epochs_without_improvement": 0,
|
| 23 |
+
"best_validation_loss": 0.9602992224693299,
|
| 24 |
+
"improved": true
|
| 25 |
+
}
|
| 26 |
+
},
|
| 27 |
+
{
|
| 28 |
+
"epoch": 2,
|
| 29 |
+
"train_soft_cross_entropy": 0.8854288237624698,
|
| 30 |
+
"validation": {
|
| 31 |
+
"soft_cross_entropy": 0.8732180511951446,
|
| 32 |
+
"accuracy": 0.7316666666666667,
|
| 33 |
+
"brier_soft": 0.08001670623819034,
|
| 34 |
+
"decisions": 600
|
| 35 |
+
},
|
| 36 |
+
"early_stopping": {
|
| 37 |
+
"patience": 3,
|
| 38 |
+
"min_epochs": 10,
|
| 39 |
+
"epochs_without_improvement": 0,
|
| 40 |
+
"best_validation_loss": 0.8732180511951446,
|
| 41 |
+
"improved": true
|
| 42 |
+
}
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"epoch": 3,
|
| 46 |
+
"train_soft_cross_entropy": 0.8375967553809837,
|
| 47 |
+
"validation": {
|
| 48 |
+
"soft_cross_entropy": 0.8618760740756989,
|
| 49 |
+
"accuracy": 0.7566666666666667,
|
| 50 |
+
"brier_soft": 0.07448753335823616,
|
| 51 |
+
"decisions": 600
|
| 52 |
+
},
|
| 53 |
+
"early_stopping": {
|
| 54 |
+
"patience": 3,
|
| 55 |
+
"min_epochs": 10,
|
| 56 |
+
"epochs_without_improvement": 0,
|
| 57 |
+
"best_validation_loss": 0.8618760740756989,
|
| 58 |
+
"improved": true
|
| 59 |
+
}
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"epoch": 4,
|
| 63 |
+
"train_soft_cross_entropy": 0.8120107175244226,
|
| 64 |
+
"validation": {
|
| 65 |
+
"soft_cross_entropy": 0.8502214312553406,
|
| 66 |
+
"accuracy": 0.7683333333333333,
|
| 67 |
+
"brier_soft": 0.06676056392490864,
|
| 68 |
+
"decisions": 600
|
| 69 |
+
},
|
| 70 |
+
"early_stopping": {
|
| 71 |
+
"patience": 3,
|
| 72 |
+
"min_epochs": 10,
|
| 73 |
+
"epochs_without_improvement": 0,
|
| 74 |
+
"best_validation_loss": 0.8502214312553406,
|
| 75 |
+
"improved": true
|
| 76 |
+
}
|
| 77 |
+
},
|
| 78 |
+
{
|
| 79 |
+
"epoch": 5,
|
| 80 |
+
"train_soft_cross_entropy": 0.7984476970301734,
|
| 81 |
+
"validation": {
|
| 82 |
+
"soft_cross_entropy": 0.8527251331011454,
|
| 83 |
+
"accuracy": 0.765,
|
| 84 |
+
"brier_soft": 0.07089023986210426,
|
| 85 |
+
"decisions": 600
|
| 86 |
+
},
|
| 87 |
+
"early_stopping": {
|
| 88 |
+
"patience": 3,
|
| 89 |
+
"min_epochs": 10,
|
| 90 |
+
"epochs_without_improvement": 1,
|
| 91 |
+
"best_validation_loss": 0.8502214312553406,
|
| 92 |
+
"improved": false
|
| 93 |
+
}
|
| 94 |
+
},
|
| 95 |
+
{
|
| 96 |
+
"epoch": 6,
|
| 97 |
+
"train_soft_cross_entropy": 0.7902782445042221,
|
| 98 |
+
"validation": {
|
| 99 |
+
"soft_cross_entropy": 0.8496846203009287,
|
| 100 |
+
"accuracy": 0.765,
|
| 101 |
+
"brier_soft": 0.06887457605761786,
|
| 102 |
+
"decisions": 600
|
| 103 |
+
},
|
| 104 |
+
"early_stopping": {
|
| 105 |
+
"patience": 3,
|
| 106 |
+
"min_epochs": 10,
|
| 107 |
+
"epochs_without_improvement": 0,
|
| 108 |
+
"best_validation_loss": 0.8496846203009287,
|
| 109 |
+
"improved": true
|
| 110 |
+
}
|
| 111 |
+
},
|
| 112 |
+
{
|
| 113 |
+
"epoch": 7,
|
| 114 |
+
"train_soft_cross_entropy": 0.7847388537283297,
|
| 115 |
+
"validation": {
|
| 116 |
+
"soft_cross_entropy": 0.8448993345101674,
|
| 117 |
+
"accuracy": 0.7533333333333333,
|
| 118 |
+
"brier_soft": 0.0668264798882107,
|
| 119 |
+
"decisions": 600
|
| 120 |
+
},
|
| 121 |
+
"early_stopping": {
|
| 122 |
+
"patience": 3,
|
| 123 |
+
"min_epochs": 10,
|
| 124 |
+
"epochs_without_improvement": 0,
|
| 125 |
+
"best_validation_loss": 0.8448993345101674,
|
| 126 |
+
"improved": true
|
| 127 |
+
}
|
| 128 |
+
},
|
| 129 |
+
{
|
| 130 |
+
"epoch": 8,
|
| 131 |
+
"train_soft_cross_entropy": 0.780297756018462,
|
| 132 |
+
"validation": {
|
| 133 |
+
"soft_cross_entropy": 0.8471182523171107,
|
| 134 |
+
"accuracy": 0.7683333333333333,
|
| 135 |
+
"brier_soft": 0.06819427679603299,
|
| 136 |
+
"decisions": 600
|
| 137 |
+
},
|
| 138 |
+
"early_stopping": {
|
| 139 |
+
"patience": 3,
|
| 140 |
+
"min_epochs": 10,
|
| 141 |
+
"epochs_without_improvement": 1,
|
| 142 |
+
"best_validation_loss": 0.8448993345101674,
|
| 143 |
+
"improved": false
|
| 144 |
+
}
|
| 145 |
+
},
|
| 146 |
+
{
|
| 147 |
+
"epoch": 9,
|
| 148 |
+
"train_soft_cross_entropy": 0.7801684511590887,
|
| 149 |
+
"validation": {
|
| 150 |
+
"soft_cross_entropy": 0.8466147363185883,
|
| 151 |
+
"accuracy": 0.775,
|
| 152 |
+
"brier_soft": 0.06677873468647401,
|
| 153 |
+
"decisions": 600
|
| 154 |
+
},
|
| 155 |
+
"early_stopping": {
|
| 156 |
+
"patience": 3,
|
| 157 |
+
"min_epochs": 10,
|
| 158 |
+
"epochs_without_improvement": 2,
|
| 159 |
+
"best_validation_loss": 0.8448993345101674,
|
| 160 |
+
"improved": false
|
| 161 |
+
}
|
| 162 |
+
},
|
| 163 |
+
{
|
| 164 |
+
"epoch": 10,
|
| 165 |
+
"train_soft_cross_entropy": 0.7775638208565888,
|
| 166 |
+
"validation": {
|
| 167 |
+
"soft_cross_entropy": 0.8455785755316416,
|
| 168 |
+
"accuracy": 0.78,
|
| 169 |
+
"brier_soft": 0.06686171248555184,
|
| 170 |
+
"decisions": 600
|
| 171 |
+
},
|
| 172 |
+
"early_stopping": {
|
| 173 |
+
"patience": 3,
|
| 174 |
+
"min_epochs": 10,
|
| 175 |
+
"epochs_without_improvement": 3,
|
| 176 |
+
"best_validation_loss": 0.8448993345101674,
|
| 177 |
+
"improved": false
|
| 178 |
+
}
|
| 179 |
+
}
|
| 180 |
+
],
|
| 181 |
+
"stopping": {
|
| 182 |
+
"reason": "early_stopping",
|
| 183 |
+
"epochs_completed": 10,
|
| 184 |
+
"patience": 3,
|
| 185 |
+
"min_epochs": 10,
|
| 186 |
+
"epochs_without_improvement": 3,
|
| 187 |
+
"metric": "validation.soft_cross_entropy",
|
| 188 |
+
"min_delta": 0.0
|
| 189 |
+
},
|
| 190 |
+
"trainable_names": [
|
| 191 |
+
"encoder.embeddings.interface.proj.weight",
|
| 192 |
+
"encoder.embeddings.interface.proj.bias"
|
| 193 |
+
],
|
| 194 |
+
"frozen_parameters_unchanged": true
|
| 195 |
+
}
|
export-verification.json
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"passed": true,
|
| 3 |
+
"candidate_seed": 1,
|
| 4 |
+
"selected_epoch": 14,
|
| 5 |
+
"cases": 3516,
|
| 6 |
+
"decisions": 5116,
|
| 7 |
+
"benchmark_cases_sha256": "10fb671245cdac1ee09e9c4d2b9b09829fbf1abd06e0954fddee0b9ec8519854",
|
| 8 |
+
"checkpoint_sha256": "e5ec5e1d494509f63960b6949af15be8390442271805ba45d796a1911b5c239a",
|
| 9 |
+
"all_answer_objects_exact": true,
|
| 10 |
+
"all_suite_metrics_exact": true,
|
| 11 |
+
"bundled_code_imported_from_outside_workspace": true,
|
| 12 |
+
"local_model_card_metadata_valid": true,
|
| 13 |
+
"device": "cuda:1",
|
| 14 |
+
"model": {
|
| 15 |
+
"gpu_name": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
|
| 16 |
+
"dtype": "torch.bfloat16",
|
| 17 |
+
"torch_version": "2.13.0+cu130",
|
| 18 |
+
"reproducibility": {
|
| 19 |
+
"deterministic_algorithms": true,
|
| 20 |
+
"warn_only": false,
|
| 21 |
+
"cublas_workspace_config": ":4096:8",
|
| 22 |
+
"python_hash_seed": "0",
|
| 23 |
+
"cudnn_benchmark": false,
|
| 24 |
+
"cudnn_deterministic": true,
|
| 25 |
+
"cudnn_allow_tf32": false,
|
| 26 |
+
"float32_matmul_precision": "highest",
|
| 27 |
+
"cudnn_sdp_enabled": false,
|
| 28 |
+
"flash_sdp_enabled": true,
|
| 29 |
+
"mem_efficient_sdp_enabled": true,
|
| 30 |
+
"math_sdp_enabled": true,
|
| 31 |
+
"cuda_version": "13.0",
|
| 32 |
+
"cudnn_version": 92000,
|
| 33 |
+
"scope": "fixed hardware/runtime; cross-platform bitwise agreement is not promised"
|
| 34 |
+
},
|
| 35 |
+
"portable_adapter": {
|
| 36 |
+
"base_model": "convaiinnovations/laya",
|
| 37 |
+
"base_revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851",
|
| 38 |
+
"seed": 1,
|
| 39 |
+
"selected_epoch": 14
|
| 40 |
+
}
|
| 41 |
+
}
|
| 42 |
+
}
|
licenses/jev-benchmarks-LICENSE
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Apache License
|
| 2 |
+
Version 2.0, January 2004
|
| 3 |
+
http://www.apache.org/licenses/
|
| 4 |
+
|
| 5 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 6 |
+
|
| 7 |
+
1. Definitions.
|
| 8 |
+
|
| 9 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 10 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 11 |
+
|
| 12 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 13 |
+
the copyright owner that is granting the License.
|
| 14 |
+
|
| 15 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 16 |
+
other entities that control, are controlled by, or are under common
|
| 17 |
+
control with that entity. For the purposes of this definition,
|
| 18 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 19 |
+
direction or management of such entity, whether by contract or
|
| 20 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 21 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 22 |
+
|
| 23 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 24 |
+
exercising permissions granted by this License.
|
| 25 |
+
|
| 26 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 27 |
+
including but not limited to software source code, documentation
|
| 28 |
+
source, and configuration files.
|
| 29 |
+
|
| 30 |
+
"Object" form shall mean any form resulting from mechanical
|
| 31 |
+
transformation or translation of a Source form, including but
|
| 32 |
+
not limited to compiled object code, generated documentation,
|
| 33 |
+
and conversions to other media types.
|
| 34 |
+
|
| 35 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 36 |
+
Object form, made available under the License, as indicated by a
|
| 37 |
+
copyright notice that is included in or attached to the work
|
| 38 |
+
(an example is provided in the Appendix below).
|
| 39 |
+
|
| 40 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 41 |
+
form, that is based on (or derived from) the Work and for which the
|
| 42 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 43 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 44 |
+
of this License, Derivative Works shall not include works that remain
|
| 45 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 46 |
+
the Work and Derivative Works thereof.
|
| 47 |
+
|
| 48 |
+
"Contribution" shall mean any work of authorship, including
|
| 49 |
+
the original version of the Work and any modifications or additions
|
| 50 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 51 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 52 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 53 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 54 |
+
means any form of electronic, verbal, or written communication sent
|
| 55 |
+
to the Licensor or its representatives, including but not limited to
|
| 56 |
+
communication on electronic mailing lists, source code control systems,
|
| 57 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 58 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 59 |
+
excluding communication that is conspicuously marked or otherwise
|
| 60 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 61 |
+
|
| 62 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 63 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 64 |
+
subsequently incorporated within the Work.
|
| 65 |
+
|
| 66 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 67 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 68 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 69 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 70 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 71 |
+
Work and such Derivative Works in Source or Object form.
|
| 72 |
+
|
| 73 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 74 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 75 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 76 |
+
(except as stated in this section) patent license to make, have made,
|
| 77 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 78 |
+
where such license applies only to those patent claims licensable
|
| 79 |
+
by such Contributor that are necessarily infringed by their
|
| 80 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 81 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 82 |
+
institute patent litigation against any entity (including a
|
| 83 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 84 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 85 |
+
or contributory patent infringement, then any patent licenses
|
| 86 |
+
granted to You under this License for that Work shall terminate
|
| 87 |
+
as of the date such litigation is filed.
|
| 88 |
+
|
| 89 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 90 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 91 |
+
modifications, and in Source or Object form, provided that You
|
| 92 |
+
meet the following conditions:
|
| 93 |
+
|
| 94 |
+
(a) You must give any other recipients of the Work or
|
| 95 |
+
Derivative Works a copy of this License; and
|
| 96 |
+
|
| 97 |
+
(b) You must cause any modified files to carry prominent notices
|
| 98 |
+
stating that You changed the files; and
|
| 99 |
+
|
| 100 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 101 |
+
that You distribute, all copyright, patent, trademark, and
|
| 102 |
+
attribution notices from the Source form of the Work,
|
| 103 |
+
excluding those notices that do not pertain to any part of
|
| 104 |
+
the Derivative Works; and
|
| 105 |
+
|
| 106 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 107 |
+
distribution, then any Derivative Works that You distribute must
|
| 108 |
+
include a readable copy of the attribution notices contained
|
| 109 |
+
within such NOTICE file, excluding those notices that do not
|
| 110 |
+
pertain to any part of the Derivative Works, in at least one
|
| 111 |
+
of the following places: within a NOTICE text file distributed
|
| 112 |
+
as part of the Derivative Works; within the Source form or
|
| 113 |
+
documentation, if provided along with the Derivative Works; or,
|
| 114 |
+
within a display generated by the Derivative Works, if and
|
| 115 |
+
wherever such third-party notices normally appear. The contents
|
| 116 |
+
of the NOTICE file are for informational purposes only and
|
| 117 |
+
do not modify the License. You may add Your own attribution
|
| 118 |
+
notices within Derivative Works that You distribute, alongside
|
| 119 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 120 |
+
that such additional attribution notices cannot be construed
|
| 121 |
+
as modifying the License.
|
| 122 |
+
|
| 123 |
+
You may add Your own copyright statement to Your modifications and
|
| 124 |
+
may provide additional or different license terms and conditions
|
| 125 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 126 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 127 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 128 |
+
the conditions stated in this License.
|
| 129 |
+
|
| 130 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 131 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 132 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 133 |
+
this License, without any additional terms or conditions.
|
| 134 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 135 |
+
the terms of any separate license agreement you may have executed
|
| 136 |
+
with Licensor regarding such Contributions.
|
| 137 |
+
|
| 138 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 139 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 140 |
+
except as required for reasonable and customary use in describing the
|
| 141 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 142 |
+
|
| 143 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 144 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 145 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 146 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 147 |
+
implied, including, without limitation, any warranties or conditions
|
| 148 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 149 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 150 |
+
appropriateness of using or redistributing the Work and assume any
|
| 151 |
+
risks associated with Your exercise of permissions under this License.
|
| 152 |
+
|
| 153 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 154 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 155 |
+
unless required by applicable law (such as deliberate and grossly
|
| 156 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 157 |
+
liable to You for damages, including any direct, indirect, special,
|
| 158 |
+
incidental, or consequential damages of any character arising as a
|
| 159 |
+
result of this License or out of the use or inability to use the
|
| 160 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 161 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 162 |
+
other commercial damages or losses), even if such Contributor
|
| 163 |
+
has been advised of the possibility of such damages.
|
| 164 |
+
|
| 165 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 166 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 167 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 168 |
+
or other liability obligations and/or rights consistent with this
|
| 169 |
+
License. However, in accepting such obligations, You may act only
|
| 170 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 171 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 172 |
+
defend, and hold each Contributor harmless for any liability
|
| 173 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 174 |
+
of your accepting any such warranty or additional liability.
|
| 175 |
+
|
| 176 |
+
END OF TERMS AND CONDITIONS
|
| 177 |
+
|
| 178 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 179 |
+
|
| 180 |
+
To apply the Apache License to your work, attach the following
|
| 181 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 182 |
+
replaced with your own identifying information. (Don't include
|
| 183 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 184 |
+
comment syntax for the file format. We also recommend that a
|
| 185 |
+
file or class name and description of purpose be included on the
|
| 186 |
+
same "printed page" as the copyright notice for easier
|
| 187 |
+
identification within third-party archives.
|
| 188 |
+
|
| 189 |
+
Copyright [yyyy] [name of copyright owner]
|
| 190 |
+
|
| 191 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 192 |
+
you may not use this file except in compliance with the License.
|
| 193 |
+
You may obtain a copy of the License at
|
| 194 |
+
|
| 195 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 196 |
+
|
| 197 |
+
Unless required by applicable law or agreed to in writing, software
|
| 198 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 199 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 200 |
+
See the License for the specific language governing permissions and
|
| 201 |
+
limitations under the License.
|
licenses/laya-LICENSE
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Apache License
|
| 2 |
+
Version 2.0, January 2004
|
| 3 |
+
http://www.apache.org/licenses/
|
| 4 |
+
|
| 5 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 6 |
+
|
| 7 |
+
1. Definitions.
|
| 8 |
+
|
| 9 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 10 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 11 |
+
|
| 12 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 13 |
+
the copyright owner that is granting the License.
|
| 14 |
+
|
| 15 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 16 |
+
other entities that control, are controlled by, or are under common
|
| 17 |
+
control with that entity. For the purposes of this definition,
|
| 18 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 19 |
+
direction or management of such entity, whether by contract or
|
| 20 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 21 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 22 |
+
|
| 23 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 24 |
+
exercising permissions granted by this License.
|
| 25 |
+
|
| 26 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 27 |
+
including but not limited to software source code, documentation
|
| 28 |
+
source, and configuration files.
|
| 29 |
+
|
| 30 |
+
"Object" form shall mean any form resulting from mechanical
|
| 31 |
+
transformation or translation of a Source form, including but
|
| 32 |
+
not limited to compiled object code, generated documentation,
|
| 33 |
+
and conversions to other media types.
|
| 34 |
+
|
| 35 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 36 |
+
Object form, made available under the License, as indicated by a
|
| 37 |
+
copyright notice that is included in or attached to the work
|
| 38 |
+
(an example is provided in the Appendix below).
|
| 39 |
+
|
| 40 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 41 |
+
form, that is based on (or derived from) the Work and for which the
|
| 42 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 43 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 44 |
+
of this License, Derivative Works shall not include works that remain
|
| 45 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 46 |
+
the Work and Derivative Works thereof.
|
| 47 |
+
|
| 48 |
+
"Contribution" shall mean any work of authorship, including
|
| 49 |
+
the original version of the Work and any modifications or additions
|
| 50 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 51 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 52 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 53 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 54 |
+
means any form of electronic, verbal, or written communication sent
|
| 55 |
+
to the Licensor or its representatives, including but not limited to
|
| 56 |
+
communication on electronic mailing lists, source code control systems,
|
| 57 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 58 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 59 |
+
excluding communication that is conspicuously marked or otherwise
|
| 60 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 61 |
+
|
| 62 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 63 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 64 |
+
subsequently incorporated within the Work.
|
| 65 |
+
|
| 66 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 67 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 68 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 69 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 70 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 71 |
+
Work and such Derivative Works in Source or Object form.
|
| 72 |
+
|
| 73 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 74 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 75 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 76 |
+
(except as stated in this section) patent license to make, have made,
|
| 77 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 78 |
+
where such license applies only to those patent claims licensable
|
| 79 |
+
by such Contributor that are necessarily infringed by their
|
| 80 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 81 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 82 |
+
institute patent litigation against any entity (including a
|
| 83 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 84 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 85 |
+
or contributory patent infringement, then any patent licenses
|
| 86 |
+
granted to You under this License for that Work shall terminate
|
| 87 |
+
as of the date such litigation is filed.
|
| 88 |
+
|
| 89 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 90 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 91 |
+
modifications, and in Source or Object form, provided that You
|
| 92 |
+
meet the following conditions:
|
| 93 |
+
|
| 94 |
+
(a) You must give any other recipients of the Work or
|
| 95 |
+
Derivative Works a copy of this License; and
|
| 96 |
+
|
| 97 |
+
(b) You must cause any modified files to carry prominent notices
|
| 98 |
+
stating that You changed the files; and
|
| 99 |
+
|
| 100 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 101 |
+
that You distribute, all copyright, patent, trademark, and
|
| 102 |
+
attribution notices from the Source form of the Work,
|
| 103 |
+
excluding those notices that do not pertain to any part of
|
| 104 |
+
the Derivative Works; and
|
| 105 |
+
|
| 106 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 107 |
+
distribution, then any Derivative Works that You distribute must
|
| 108 |
+
include a readable copy of the attribution notices contained
|
| 109 |
+
within such NOTICE file, excluding those notices that do not
|
| 110 |
+
pertain to any part of the Derivative Works, in at least one
|
| 111 |
+
of the following places: within a NOTICE text file distributed
|
| 112 |
+
as part of the Derivative Works; within the Source form or
|
| 113 |
+
documentation, if provided along with the Derivative Works; or,
|
| 114 |
+
within a display generated by the Derivative Works, if and
|
| 115 |
+
wherever such third-party notices normally appear. The contents
|
| 116 |
+
of the NOTICE file are for informational purposes only and
|
| 117 |
+
do not modify the License. You may add Your own attribution
|
| 118 |
+
notices within Derivative Works that You distribute, alongside
|
| 119 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 120 |
+
that such additional attribution notices cannot be construed
|
| 121 |
+
as modifying the License.
|
| 122 |
+
|
| 123 |
+
You may add Your own copyright statement to Your modifications and
|
| 124 |
+
may provide additional or different license terms and conditions
|
| 125 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 126 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 127 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 128 |
+
the conditions stated in this License.
|
| 129 |
+
|
| 130 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 131 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 132 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 133 |
+
this License, without any additional terms or conditions.
|
| 134 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 135 |
+
the terms of any separate license agreement you may have executed
|
| 136 |
+
with Licensor regarding such Contributions.
|
| 137 |
+
|
| 138 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 139 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 140 |
+
except as required for reasonable and customary use in describing the
|
| 141 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 142 |
+
|
| 143 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 144 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 145 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 146 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 147 |
+
implied, including, without limitation, any warranties or conditions
|
| 148 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 149 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 150 |
+
appropriateness of using or redistributing the Work and assume any
|
| 151 |
+
risks associated with Your exercise of permissions under this License.
|
| 152 |
+
|
| 153 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 154 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 155 |
+
unless required by applicable law (such as deliberate and grossly
|
| 156 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 157 |
+
liable to You for damages, including any direct, indirect, special,
|
| 158 |
+
incidental, or consequential damages of any character arising as a
|
| 159 |
+
result of this License or out of the use or inability to use the
|
| 160 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 161 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 162 |
+
other commercial damages or losses), even if such Contributor
|
| 163 |
+
has been advised of the possibility of such damages.
|
| 164 |
+
|
| 165 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 166 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 167 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 168 |
+
or other liability obligations and/or rights consistent with this
|
| 169 |
+
License. However, in accepting such obligations, You may act only
|
| 170 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 171 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 172 |
+
defend, and hold each Contributor harmless for any liability
|
| 173 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 174 |
+
of your accepting any such warranty or additional liability.
|
| 175 |
+
|
| 176 |
+
END OF TERMS AND CONDITIONS
|
load_adapter.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Local entry point for the bundled, pinned Laya input adapter."""
|
| 2 |
+
|
| 3 |
+
from pathlib import Path
|
| 4 |
+
import sys
|
| 5 |
+
|
| 6 |
+
PACKAGE = Path(__file__).resolve().parent
|
| 7 |
+
if str(PACKAGE) not in sys.path:
|
| 8 |
+
sys.path.insert(0, str(PACKAGE))
|
| 9 |
+
|
| 10 |
+
from ariadne_bench.portable import load_release # noqa: E402
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
def load(directory=PACKAGE, **options):
|
| 14 |
+
return load_release(directory, **options)
|
metrics.json
ADDED
|
@@ -0,0 +1,984 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"baseline": {
|
| 3 |
+
"typed-decisions": 0.3625,
|
| 4 |
+
"ag-news": 0.9483333333333334,
|
| 5 |
+
"boolq": 0.83,
|
| 6 |
+
"emotion": 0.5733333333333334,
|
| 7 |
+
"prompt-injections": 0.6896551724137931,
|
| 8 |
+
"sst5": 0.37,
|
| 9 |
+
"massive-intent.en": 0.7866666666666666,
|
| 10 |
+
"xnli.en": 0.86
|
| 11 |
+
},
|
| 12 |
+
"runs": {
|
| 13 |
+
"0": {
|
| 14 |
+
"seed": 0,
|
| 15 |
+
"selected_epoch": 20,
|
| 16 |
+
"elapsed_s": 1038.9482859019772,
|
| 17 |
+
"initial_validation": {
|
| 18 |
+
"soft_cross_entropy": 1.544869564374288,
|
| 19 |
+
"accuracy": 0.36833333333333335,
|
| 20 |
+
"brier_soft": 0.43051501592000324,
|
| 21 |
+
"decisions": 600
|
| 22 |
+
},
|
| 23 |
+
"selected_validation": {
|
| 24 |
+
"soft_cross_entropy": 1.0292588464419048,
|
| 25 |
+
"accuracy": 0.57,
|
| 26 |
+
"brier_soft": 0.16918850486477216,
|
| 27 |
+
"decisions": 600
|
| 28 |
+
},
|
| 29 |
+
"stopping": {
|
| 30 |
+
"reason": "early_stopping",
|
| 31 |
+
"epochs_completed": 23,
|
| 32 |
+
"patience": 3,
|
| 33 |
+
"min_epochs": 10,
|
| 34 |
+
"epochs_without_improvement": 3,
|
| 35 |
+
"metric": "validation.soft_cross_entropy",
|
| 36 |
+
"min_delta": 0.0
|
| 37 |
+
},
|
| 38 |
+
"frozen_parameters_unchanged": true,
|
| 39 |
+
"suites": {
|
| 40 |
+
"typed-decisions": {
|
| 41 |
+
"attempted": 2000,
|
| 42 |
+
"valid": 2000,
|
| 43 |
+
"failed": 0,
|
| 44 |
+
"coverage": 1.0,
|
| 45 |
+
"accuracy_all": 0.61,
|
| 46 |
+
"accuracy_valid": 0.61,
|
| 47 |
+
"ece_top_label": 0.15303136435282014,
|
| 48 |
+
"mean_confidence": 0.4574502381919959,
|
| 49 |
+
"brier_hard": 0.549798004153015,
|
| 50 |
+
"brier_hard_n": 2000,
|
| 51 |
+
"nll_hard": 0.9521041848093966,
|
| 52 |
+
"nll_hard_n": 2000,
|
| 53 |
+
"zero_probability_gold": 0.0,
|
| 54 |
+
"zero_probability_gold_n": 2000,
|
| 55 |
+
"soft_accuracy": 0.3883819759829014,
|
| 56 |
+
"soft_accuracy_n": 2000,
|
| 57 |
+
"brier_soft": 0.15223049605383382,
|
| 58 |
+
"brier_soft_n": 2000,
|
| 59 |
+
"kl_gold_to_prediction": 0.2712886346630264,
|
| 60 |
+
"kl_gold_to_prediction_n": 2000,
|
| 61 |
+
"total_variation": 0.2797391333751494,
|
| 62 |
+
"total_variation_n": 2000,
|
| 63 |
+
"score_mae": 0.45095113683604626,
|
| 64 |
+
"score_mae_n": 800,
|
| 65 |
+
"within_one_level": 0.915,
|
| 66 |
+
"within_one_level_n": 800,
|
| 67 |
+
"macro_f1": 0.4150680787086972
|
| 68 |
+
},
|
| 69 |
+
"ag-news": {
|
| 70 |
+
"attempted": 600,
|
| 71 |
+
"valid": 600,
|
| 72 |
+
"failed": 0,
|
| 73 |
+
"coverage": 1.0,
|
| 74 |
+
"accuracy_all": 0.31333333333333335,
|
| 75 |
+
"accuracy_valid": 0.31333333333333335,
|
| 76 |
+
"ece_top_label": 0.025934925132124264,
|
| 77 |
+
"mean_confidence": 0.28915707486787573,
|
| 78 |
+
"brier_hard": 0.7441339007069468,
|
| 79 |
+
"brier_hard_n": 600,
|
| 80 |
+
"nll_hard": 1.3750333163467814,
|
| 81 |
+
"nll_hard_n": 600,
|
| 82 |
+
"zero_probability_gold": 0.0,
|
| 83 |
+
"zero_probability_gold_n": 600,
|
| 84 |
+
"macro_f1": 0.29084523228646375
|
| 85 |
+
},
|
| 86 |
+
"emotion": {
|
| 87 |
+
"attempted": 600,
|
| 88 |
+
"valid": 600,
|
| 89 |
+
"failed": 0,
|
| 90 |
+
"coverage": 1.0,
|
| 91 |
+
"accuracy_all": 0.195,
|
| 92 |
+
"accuracy_valid": 0.195,
|
| 93 |
+
"ece_top_label": 0.08537782470783518,
|
| 94 |
+
"mean_confidence": 0.2764666886168944,
|
| 95 |
+
"brier_hard": 0.8415061960524678,
|
| 96 |
+
"brier_hard_n": 600,
|
| 97 |
+
"nll_hard": 1.810555559185162,
|
| 98 |
+
"nll_hard_n": 600,
|
| 99 |
+
"zero_probability_gold": 0.0,
|
| 100 |
+
"zero_probability_gold_n": 600,
|
| 101 |
+
"macro_f1": 0.12883846057517262
|
| 102 |
+
},
|
| 103 |
+
"boolq": {
|
| 104 |
+
"attempted": 600,
|
| 105 |
+
"valid": 600,
|
| 106 |
+
"failed": 0,
|
| 107 |
+
"coverage": 1.0,
|
| 108 |
+
"accuracy_all": 0.5666666666666667,
|
| 109 |
+
"accuracy_valid": 0.5666666666666667,
|
| 110 |
+
"ece_top_label": 0.08635800000000002,
|
| 111 |
+
"mean_confidence": 0.5258943333333334,
|
| 112 |
+
"brier_hard": 0.5020554209333333,
|
| 113 |
+
"brier_hard_n": 600,
|
| 114 |
+
"nll_hard": 0.6953013099942155,
|
| 115 |
+
"nll_hard_n": 600,
|
| 116 |
+
"zero_probability_gold": 0.0,
|
| 117 |
+
"zero_probability_gold_n": 600,
|
| 118 |
+
"macro_f1": 0.5074762578298646
|
| 119 |
+
},
|
| 120 |
+
"sst5": {
|
| 121 |
+
"attempted": 600,
|
| 122 |
+
"valid": 600,
|
| 123 |
+
"failed": 0,
|
| 124 |
+
"coverage": 1.0,
|
| 125 |
+
"accuracy_all": 0.15,
|
| 126 |
+
"accuracy_valid": 0.15,
|
| 127 |
+
"ece_top_label": 0.11939667407180776,
|
| 128 |
+
"mean_confidence": 0.2693966740718078,
|
| 129 |
+
"brier_hard": 0.8207044411796734,
|
| 130 |
+
"brier_hard_n": 600,
|
| 131 |
+
"nll_hard": 1.6589117409852412,
|
| 132 |
+
"nll_hard_n": 600,
|
| 133 |
+
"zero_probability_gold": 0.0,
|
| 134 |
+
"zero_probability_gold_n": 600,
|
| 135 |
+
"score_mae": 1.2247300886774997,
|
| 136 |
+
"score_mae_n": 600,
|
| 137 |
+
"within_one_level": 0.42,
|
| 138 |
+
"within_one_level_n": 600,
|
| 139 |
+
"macro_f1": 0.10629896865332711
|
| 140 |
+
},
|
| 141 |
+
"prompt-injections": {
|
| 142 |
+
"attempted": 116,
|
| 143 |
+
"valid": 116,
|
| 144 |
+
"failed": 0,
|
| 145 |
+
"coverage": 1.0,
|
| 146 |
+
"accuracy_all": 0.5344827586206896,
|
| 147 |
+
"accuracy_valid": 0.5344827586206896,
|
| 148 |
+
"ece_top_label": 0.02089741379310345,
|
| 149 |
+
"mean_confidence": 0.5135853448275862,
|
| 150 |
+
"brier_hard": 0.5003080198275862,
|
| 151 |
+
"brier_hard_n": 116,
|
| 152 |
+
"nll_hard": 0.6934547870034389,
|
| 153 |
+
"nll_hard_n": 116,
|
| 154 |
+
"zero_probability_gold": 0.0,
|
| 155 |
+
"zero_probability_gold_n": 116,
|
| 156 |
+
"macro_f1": 0.513664596273292
|
| 157 |
+
},
|
| 158 |
+
"massive-intent.en": {
|
| 159 |
+
"attempted": 300,
|
| 160 |
+
"valid": 300,
|
| 161 |
+
"failed": 0,
|
| 162 |
+
"coverage": 1.0,
|
| 163 |
+
"accuracy_all": 0.023333333333333334,
|
| 164 |
+
"accuracy_valid": 0.023333333333333334,
|
| 165 |
+
"ece_top_label": 0.09797495336730408,
|
| 166 |
+
"mean_confidence": 0.12130828670063742,
|
| 167 |
+
"brier_hard": 0.9672041702729498,
|
| 168 |
+
"brier_hard_n": 300,
|
| 169 |
+
"nll_hard": 3.1313647377288074,
|
| 170 |
+
"nll_hard_n": 300,
|
| 171 |
+
"zero_probability_gold": 0.0,
|
| 172 |
+
"zero_probability_gold_n": 300,
|
| 173 |
+
"macro_f1": 0.011215849106652133
|
| 174 |
+
},
|
| 175 |
+
"xnli.en": {
|
| 176 |
+
"attempted": 300,
|
| 177 |
+
"valid": 300,
|
| 178 |
+
"failed": 0,
|
| 179 |
+
"coverage": 1.0,
|
| 180 |
+
"accuracy_all": 0.3433333333333333,
|
| 181 |
+
"accuracy_valid": 0.3433333333333333,
|
| 182 |
+
"ece_top_label": 0.05212130996336308,
|
| 183 |
+
"mean_confidence": 0.39545464329669644,
|
| 184 |
+
"brier_hard": 0.6747449551934047,
|
| 185 |
+
"brier_hard_n": 300,
|
| 186 |
+
"nll_hard": 1.1111148411541347,
|
| 187 |
+
"nll_hard_n": 300,
|
| 188 |
+
"zero_probability_gold": 0.0,
|
| 189 |
+
"zero_probability_gold_n": 300,
|
| 190 |
+
"macro_f1": 0.24805362074756226
|
| 191 |
+
}
|
| 192 |
+
}
|
| 193 |
+
},
|
| 194 |
+
"1": {
|
| 195 |
+
"seed": 1,
|
| 196 |
+
"selected_epoch": 14,
|
| 197 |
+
"elapsed_s": 766.9112328969641,
|
| 198 |
+
"initial_validation": {
|
| 199 |
+
"soft_cross_entropy": 1.544869564374288,
|
| 200 |
+
"accuracy": 0.36833333333333335,
|
| 201 |
+
"brier_soft": 0.43051501592000324,
|
| 202 |
+
"decisions": 600
|
| 203 |
+
},
|
| 204 |
+
"selected_validation": {
|
| 205 |
+
"soft_cross_entropy": 0.8377540612220764,
|
| 206 |
+
"accuracy": 0.765,
|
| 207 |
+
"brier_soft": 0.06276483290052662,
|
| 208 |
+
"decisions": 600
|
| 209 |
+
},
|
| 210 |
+
"stopping": {
|
| 211 |
+
"reason": "early_stopping",
|
| 212 |
+
"epochs_completed": 17,
|
| 213 |
+
"patience": 3,
|
| 214 |
+
"min_epochs": 10,
|
| 215 |
+
"epochs_without_improvement": 3,
|
| 216 |
+
"metric": "validation.soft_cross_entropy",
|
| 217 |
+
"min_delta": 0.0
|
| 218 |
+
},
|
| 219 |
+
"frozen_parameters_unchanged": true,
|
| 220 |
+
"suites": {
|
| 221 |
+
"typed-decisions": {
|
| 222 |
+
"attempted": 2000,
|
| 223 |
+
"valid": 2000,
|
| 224 |
+
"failed": 0,
|
| 225 |
+
"coverage": 1.0,
|
| 226 |
+
"accuracy_all": 0.7695,
|
| 227 |
+
"accuracy_valid": 0.7695,
|
| 228 |
+
"ece_top_label": 0.2142237525419539,
|
| 229 |
+
"mean_confidence": 0.555276247458046,
|
| 230 |
+
"brier_hard": 0.3977842163576973,
|
| 231 |
+
"brier_hard_n": 2000,
|
| 232 |
+
"nll_hard": 0.7027635604099817,
|
| 233 |
+
"nll_hard_n": 2000,
|
| 234 |
+
"zero_probability_gold": 0.0,
|
| 235 |
+
"zero_probability_gold_n": 2000,
|
| 236 |
+
"soft_accuracy": 0.47103605907759827,
|
| 237 |
+
"soft_accuracy_n": 2000,
|
| 238 |
+
"brier_soft": 0.06255403710877039,
|
| 239 |
+
"brier_soft_n": 2000,
|
| 240 |
+
"kl_gold_to_prediction": 0.11643450461088396,
|
| 241 |
+
"kl_gold_to_prediction_n": 2000,
|
| 242 |
+
"total_variation": 0.17146599324503523,
|
| 243 |
+
"total_variation_n": 2000,
|
| 244 |
+
"score_mae": 0.2299944493828981,
|
| 245 |
+
"score_mae_n": 800,
|
| 246 |
+
"within_one_level": 0.98875,
|
| 247 |
+
"within_one_level_n": 800,
|
| 248 |
+
"macro_f1": 0.6487623885802034
|
| 249 |
+
},
|
| 250 |
+
"ag-news": {
|
| 251 |
+
"attempted": 600,
|
| 252 |
+
"valid": 600,
|
| 253 |
+
"failed": 0,
|
| 254 |
+
"coverage": 1.0,
|
| 255 |
+
"accuracy_all": 0.9316666666666666,
|
| 256 |
+
"accuracy_valid": 0.9316666666666666,
|
| 257 |
+
"ece_top_label": 0.03387674147756575,
|
| 258 |
+
"mean_confidence": 0.904708562141314,
|
| 259 |
+
"brier_hard": 0.10911433540423408,
|
| 260 |
+
"brier_hard_n": 600,
|
| 261 |
+
"nll_hard": 0.21361121678114303,
|
| 262 |
+
"nll_hard_n": 600,
|
| 263 |
+
"zero_probability_gold": 0.0,
|
| 264 |
+
"zero_probability_gold_n": 600,
|
| 265 |
+
"macro_f1": 0.9283987145646934
|
| 266 |
+
},
|
| 267 |
+
"emotion": {
|
| 268 |
+
"attempted": 600,
|
| 269 |
+
"valid": 600,
|
| 270 |
+
"failed": 0,
|
| 271 |
+
"coverage": 1.0,
|
| 272 |
+
"accuracy_all": 0.5733333333333334,
|
| 273 |
+
"accuracy_valid": 0.5733333333333334,
|
| 274 |
+
"ece_top_label": 0.3169210441851814,
|
| 275 |
+
"mean_confidence": 0.8880587177380197,
|
| 276 |
+
"brier_hard": 0.7264584486728032,
|
| 277 |
+
"brier_hard_n": 600,
|
| 278 |
+
"nll_hard": 2.146154603203373,
|
| 279 |
+
"nll_hard_n": 600,
|
| 280 |
+
"zero_probability_gold": 0.008333333333333333,
|
| 281 |
+
"zero_probability_gold_n": 600,
|
| 282 |
+
"macro_f1": 0.4862286665693279
|
| 283 |
+
},
|
| 284 |
+
"boolq": {
|
| 285 |
+
"attempted": 600,
|
| 286 |
+
"valid": 600,
|
| 287 |
+
"failed": 0,
|
| 288 |
+
"coverage": 1.0,
|
| 289 |
+
"accuracy_all": 0.7983333333333333,
|
| 290 |
+
"accuracy_valid": 0.7983333333333333,
|
| 291 |
+
"ece_top_label": 0.09755433333333334,
|
| 292 |
+
"mean_confidence": 0.8900093333333333,
|
| 293 |
+
"brier_hard": 0.29822958146666667,
|
| 294 |
+
"brier_hard_n": 600,
|
| 295 |
+
"nll_hard": 0.4711253960780479,
|
| 296 |
+
"nll_hard_n": 600,
|
| 297 |
+
"zero_probability_gold": 0.0,
|
| 298 |
+
"zero_probability_gold_n": 600,
|
| 299 |
+
"macro_f1": 0.7813983878883868
|
| 300 |
+
},
|
| 301 |
+
"sst5": {
|
| 302 |
+
"attempted": 600,
|
| 303 |
+
"valid": 600,
|
| 304 |
+
"failed": 0,
|
| 305 |
+
"coverage": 1.0,
|
| 306 |
+
"accuracy_all": 0.42,
|
| 307 |
+
"accuracy_valid": 0.42,
|
| 308 |
+
"ece_top_label": 0.1643168667044709,
|
| 309 |
+
"mean_confidence": 0.5761336745094923,
|
| 310 |
+
"brier_hard": 0.722245716690302,
|
| 311 |
+
"brier_hard_n": 600,
|
| 312 |
+
"nll_hard": 1.412054753482227,
|
| 313 |
+
"nll_hard_n": 600,
|
| 314 |
+
"zero_probability_gold": 0.0,
|
| 315 |
+
"zero_probability_gold_n": 600,
|
| 316 |
+
"score_mae": 0.7360310433394476,
|
| 317 |
+
"score_mae_n": 600,
|
| 318 |
+
"within_one_level": 0.7466666666666667,
|
| 319 |
+
"within_one_level_n": 600,
|
| 320 |
+
"macro_f1": 0.41387920942262574
|
| 321 |
+
},
|
| 322 |
+
"prompt-injections": {
|
| 323 |
+
"attempted": 116,
|
| 324 |
+
"valid": 116,
|
| 325 |
+
"failed": 0,
|
| 326 |
+
"coverage": 1.0,
|
| 327 |
+
"accuracy_all": 0.7155172413793104,
|
| 328 |
+
"accuracy_valid": 0.7155172413793104,
|
| 329 |
+
"ece_top_label": 0.20010517241379305,
|
| 330 |
+
"mean_confidence": 0.9105,
|
| 331 |
+
"brier_hard": 0.43610344172413795,
|
| 332 |
+
"brier_hard_n": 116,
|
| 333 |
+
"nll_hard": 1.2355666268201304,
|
| 334 |
+
"nll_hard_n": 116,
|
| 335 |
+
"zero_probability_gold": 0.008620689655172414,
|
| 336 |
+
"zero_probability_gold_n": 116,
|
| 337 |
+
"macro_f1": 0.7038756091900673
|
| 338 |
+
},
|
| 339 |
+
"massive-intent.en": {
|
| 340 |
+
"attempted": 300,
|
| 341 |
+
"valid": 300,
|
| 342 |
+
"failed": 0,
|
| 343 |
+
"coverage": 1.0,
|
| 344 |
+
"accuracy_all": 0.7433333333333333,
|
| 345 |
+
"accuracy_valid": 0.7433333333333333,
|
| 346 |
+
"ece_top_label": 0.1906540683935104,
|
| 347 |
+
"mean_confidence": 0.9339874017268438,
|
| 348 |
+
"brier_hard": 0.43706654758878133,
|
| 349 |
+
"brier_hard_n": 300,
|
| 350 |
+
"nll_hard": 2.7801978001907903,
|
| 351 |
+
"nll_hard_n": 300,
|
| 352 |
+
"zero_probability_gold": 0.07,
|
| 353 |
+
"zero_probability_gold_n": 300,
|
| 354 |
+
"macro_f1": 0.44346732036749054
|
| 355 |
+
},
|
| 356 |
+
"xnli.en": {
|
| 357 |
+
"attempted": 300,
|
| 358 |
+
"valid": 300,
|
| 359 |
+
"failed": 0,
|
| 360 |
+
"coverage": 1.0,
|
| 361 |
+
"accuracy_all": 0.8766666666666667,
|
| 362 |
+
"accuracy_valid": 0.8766666666666667,
|
| 363 |
+
"ece_top_label": 0.055158283453726184,
|
| 364 |
+
"mean_confidence": 0.913468962679153,
|
| 365 |
+
"brier_hard": 0.19180470102420566,
|
| 366 |
+
"brier_hard_n": 300,
|
| 367 |
+
"nll_hard": 0.3708980570966955,
|
| 368 |
+
"nll_hard_n": 300,
|
| 369 |
+
"zero_probability_gold": 0.0,
|
| 370 |
+
"zero_probability_gold_n": 300,
|
| 371 |
+
"macro_f1": 0.8772028178860477
|
| 372 |
+
}
|
| 373 |
+
}
|
| 374 |
+
},
|
| 375 |
+
"2": {
|
| 376 |
+
"seed": 2,
|
| 377 |
+
"selected_epoch": 15,
|
| 378 |
+
"elapsed_s": 813.5451362769818,
|
| 379 |
+
"initial_validation": {
|
| 380 |
+
"soft_cross_entropy": 1.544869564374288,
|
| 381 |
+
"accuracy": 0.36833333333333335,
|
| 382 |
+
"brier_soft": 0.43051501592000324,
|
| 383 |
+
"decisions": 600
|
| 384 |
+
},
|
| 385 |
+
"selected_validation": {
|
| 386 |
+
"soft_cross_entropy": 0.9867499772707621,
|
| 387 |
+
"accuracy": 0.6316666666666667,
|
| 388 |
+
"brier_soft": 0.14927835414807003,
|
| 389 |
+
"decisions": 600
|
| 390 |
+
},
|
| 391 |
+
"stopping": {
|
| 392 |
+
"reason": "early_stopping",
|
| 393 |
+
"epochs_completed": 18,
|
| 394 |
+
"patience": 3,
|
| 395 |
+
"min_epochs": 10,
|
| 396 |
+
"epochs_without_improvement": 3,
|
| 397 |
+
"metric": "validation.soft_cross_entropy",
|
| 398 |
+
"min_delta": 0.0
|
| 399 |
+
},
|
| 400 |
+
"frozen_parameters_unchanged": true,
|
| 401 |
+
"suites": {
|
| 402 |
+
"typed-decisions": {
|
| 403 |
+
"attempted": 2000,
|
| 404 |
+
"valid": 2000,
|
| 405 |
+
"failed": 0,
|
| 406 |
+
"coverage": 1.0,
|
| 407 |
+
"accuracy_all": 0.661,
|
| 408 |
+
"accuracy_valid": 0.661,
|
| 409 |
+
"ece_top_label": 0.17250655060654457,
|
| 410 |
+
"mean_confidence": 0.48883668313682976,
|
| 411 |
+
"brier_hard": 0.5040496940863629,
|
| 412 |
+
"brier_hard_n": 2000,
|
| 413 |
+
"nll_hard": 0.8694405673021295,
|
| 414 |
+
"nll_hard_n": 2000,
|
| 415 |
+
"zero_probability_gold": 0.0,
|
| 416 |
+
"zero_probability_gold_n": 2000,
|
| 417 |
+
"soft_accuracy": 0.41427322765683283,
|
| 418 |
+
"soft_accuracy_n": 2000,
|
| 419 |
+
"brier_soft": 0.12329331823446005,
|
| 420 |
+
"brier_soft_n": 2000,
|
| 421 |
+
"kl_gold_to_prediction": 0.2167589625120186,
|
| 422 |
+
"kl_gold_to_prediction_n": 2000,
|
| 423 |
+
"total_variation": 0.24557641844905673,
|
| 424 |
+
"total_variation_n": 2000,
|
| 425 |
+
"score_mae": 0.38257377026959916,
|
| 426 |
+
"score_mae_n": 800,
|
| 427 |
+
"within_one_level": 0.92625,
|
| 428 |
+
"within_one_level_n": 800,
|
| 429 |
+
"macro_f1": 0.511489281675723
|
| 430 |
+
},
|
| 431 |
+
"ag-news": {
|
| 432 |
+
"attempted": 600,
|
| 433 |
+
"valid": 600,
|
| 434 |
+
"failed": 0,
|
| 435 |
+
"coverage": 1.0,
|
| 436 |
+
"accuracy_all": 0.2633333333333333,
|
| 437 |
+
"accuracy_valid": 0.2633333333333333,
|
| 438 |
+
"ece_top_label": 0.0415880758215024,
|
| 439 |
+
"mean_confidence": 0.30492140915483573,
|
| 440 |
+
"brier_hard": 0.7533353201524136,
|
| 441 |
+
"brier_hard_n": 600,
|
| 442 |
+
"nll_hard": 1.392051963909145,
|
| 443 |
+
"nll_hard_n": 600,
|
| 444 |
+
"zero_probability_gold": 0.0,
|
| 445 |
+
"zero_probability_gold_n": 600,
|
| 446 |
+
"macro_f1": 0.2196261943166739
|
| 447 |
+
},
|
| 448 |
+
"emotion": {
|
| 449 |
+
"attempted": 600,
|
| 450 |
+
"valid": 600,
|
| 451 |
+
"failed": 0,
|
| 452 |
+
"coverage": 1.0,
|
| 453 |
+
"accuracy_all": 0.04833333333333333,
|
| 454 |
+
"accuracy_valid": 0.04833333333333333,
|
| 455 |
+
"ece_top_label": 0.21796334391120661,
|
| 456 |
+
"mean_confidence": 0.2662966772445399,
|
| 457 |
+
"brier_hard": 0.9083949711964834,
|
| 458 |
+
"brier_hard_n": 600,
|
| 459 |
+
"nll_hard": 2.0132867760168445,
|
| 460 |
+
"nll_hard_n": 600,
|
| 461 |
+
"zero_probability_gold": 0.0,
|
| 462 |
+
"zero_probability_gold_n": 600,
|
| 463 |
+
"macro_f1": 0.03431938431938432
|
| 464 |
+
},
|
| 465 |
+
"boolq": {
|
| 466 |
+
"attempted": 600,
|
| 467 |
+
"valid": 600,
|
| 468 |
+
"failed": 0,
|
| 469 |
+
"coverage": 1.0,
|
| 470 |
+
"accuracy_all": 0.49833333333333335,
|
| 471 |
+
"accuracy_valid": 0.49833333333333335,
|
| 472 |
+
"ece_top_label": 0.03133150000000007,
|
| 473 |
+
"mean_confidence": 0.5290545,
|
| 474 |
+
"brier_hard": 0.5040687543666666,
|
| 475 |
+
"brier_hard_n": 600,
|
| 476 |
+
"nll_hard": 0.6972355680599187,
|
| 477 |
+
"nll_hard_n": 600,
|
| 478 |
+
"zero_probability_gold": 0.0,
|
| 479 |
+
"zero_probability_gold_n": 600,
|
| 480 |
+
"macro_f1": 0.49597983919356775
|
| 481 |
+
},
|
| 482 |
+
"sst5": {
|
| 483 |
+
"attempted": 600,
|
| 484 |
+
"valid": 600,
|
| 485 |
+
"failed": 0,
|
| 486 |
+
"coverage": 1.0,
|
| 487 |
+
"accuracy_all": 0.22666666666666666,
|
| 488 |
+
"accuracy_valid": 0.22666666666666666,
|
| 489 |
+
"ece_top_label": 0.08382145968032653,
|
| 490 |
+
"mean_confidence": 0.30788872429147607,
|
| 491 |
+
"brier_hard": 0.8111226172703578,
|
| 492 |
+
"brier_hard_n": 600,
|
| 493 |
+
"nll_hard": 1.6364255257258482,
|
| 494 |
+
"nll_hard_n": 600,
|
| 495 |
+
"zero_probability_gold": 0.0,
|
| 496 |
+
"zero_probability_gold_n": 600,
|
| 497 |
+
"score_mae": 1.223873285603664,
|
| 498 |
+
"score_mae_n": 600,
|
| 499 |
+
"within_one_level": 0.39,
|
| 500 |
+
"within_one_level_n": 600,
|
| 501 |
+
"macro_f1": 0.12552908285983264
|
| 502 |
+
},
|
| 503 |
+
"prompt-injections": {
|
| 504 |
+
"attempted": 116,
|
| 505 |
+
"valid": 116,
|
| 506 |
+
"failed": 0,
|
| 507 |
+
"coverage": 1.0,
|
| 508 |
+
"accuracy_all": 0.3448275862068966,
|
| 509 |
+
"accuracy_valid": 0.3448275862068966,
|
| 510 |
+
"ece_top_label": 0.1914112068965517,
|
| 511 |
+
"mean_confidence": 0.5362387931034482,
|
| 512 |
+
"brier_hard": 0.5323824856896552,
|
| 513 |
+
"brier_hard_n": 116,
|
| 514 |
+
"nll_hard": 0.7258214322004152,
|
| 515 |
+
"nll_hard_n": 116,
|
| 516 |
+
"zero_probability_gold": 0.0,
|
| 517 |
+
"zero_probability_gold_n": 116,
|
| 518 |
+
"macro_f1": 0.275
|
| 519 |
+
},
|
| 520 |
+
"massive-intent.en": {
|
| 521 |
+
"attempted": 300,
|
| 522 |
+
"valid": 300,
|
| 523 |
+
"failed": 0,
|
| 524 |
+
"coverage": 1.0,
|
| 525 |
+
"accuracy_all": 0.06333333333333334,
|
| 526 |
+
"accuracy_valid": 0.06333333333333334,
|
| 527 |
+
"ece_top_label": 0.03411377590323972,
|
| 528 |
+
"mean_confidence": 0.09744710923657304,
|
| 529 |
+
"brier_hard": 0.9608971275390831,
|
| 530 |
+
"brier_hard_n": 300,
|
| 531 |
+
"nll_hard": 3.135262799945936,
|
| 532 |
+
"nll_hard_n": 300,
|
| 533 |
+
"zero_probability_gold": 0.0,
|
| 534 |
+
"zero_probability_gold_n": 300,
|
| 535 |
+
"macro_f1": 0.02907709642455108
|
| 536 |
+
},
|
| 537 |
+
"xnli.en": {
|
| 538 |
+
"attempted": 300,
|
| 539 |
+
"valid": 300,
|
| 540 |
+
"failed": 0,
|
| 541 |
+
"coverage": 1.0,
|
| 542 |
+
"accuracy_all": 0.3433333333333333,
|
| 543 |
+
"accuracy_valid": 0.3433333333333333,
|
| 544 |
+
"ece_top_label": 0.04294870734165375,
|
| 545 |
+
"mean_confidence": 0.38628204067498706,
|
| 546 |
+
"brier_hard": 0.6761361630010466,
|
| 547 |
+
"brier_hard_n": 300,
|
| 548 |
+
"nll_hard": 1.1128104653939794,
|
| 549 |
+
"nll_hard_n": 300,
|
| 550 |
+
"zero_probability_gold": 0.0,
|
| 551 |
+
"zero_probability_gold_n": 300,
|
| 552 |
+
"macro_f1": 0.3099922839506173
|
| 553 |
+
}
|
| 554 |
+
}
|
| 555 |
+
},
|
| 556 |
+
"3": {
|
| 557 |
+
"seed": 3,
|
| 558 |
+
"selected_epoch": 24,
|
| 559 |
+
"elapsed_s": 1226.4128511130111,
|
| 560 |
+
"initial_validation": {
|
| 561 |
+
"soft_cross_entropy": 1.544869564374288,
|
| 562 |
+
"accuracy": 0.36833333333333335,
|
| 563 |
+
"brier_soft": 0.43051501592000324,
|
| 564 |
+
"decisions": 600
|
| 565 |
+
},
|
| 566 |
+
"selected_validation": {
|
| 567 |
+
"soft_cross_entropy": 0.9105557099978129,
|
| 568 |
+
"accuracy": 0.715,
|
| 569 |
+
"brier_soft": 0.1013037375236551,
|
| 570 |
+
"decisions": 600
|
| 571 |
+
},
|
| 572 |
+
"stopping": {
|
| 573 |
+
"reason": "early_stopping",
|
| 574 |
+
"epochs_completed": 27,
|
| 575 |
+
"patience": 3,
|
| 576 |
+
"min_epochs": 10,
|
| 577 |
+
"epochs_without_improvement": 3,
|
| 578 |
+
"metric": "validation.soft_cross_entropy",
|
| 579 |
+
"min_delta": 0.0
|
| 580 |
+
},
|
| 581 |
+
"frozen_parameters_unchanged": true,
|
| 582 |
+
"suites": {
|
| 583 |
+
"typed-decisions": {
|
| 584 |
+
"attempted": 2000,
|
| 585 |
+
"valid": 2000,
|
| 586 |
+
"failed": 0,
|
| 587 |
+
"coverage": 1.0,
|
| 588 |
+
"accuracy_all": 0.7145,
|
| 589 |
+
"accuracy_valid": 0.7145,
|
| 590 |
+
"ece_top_label": 0.18500231281217855,
|
| 591 |
+
"mean_confidence": 0.5300184130804106,
|
| 592 |
+
"brier_hard": 0.45082257101028994,
|
| 593 |
+
"brier_hard_n": 2000,
|
| 594 |
+
"nll_hard": 0.7891005479512444,
|
| 595 |
+
"nll_hard_n": 2000,
|
| 596 |
+
"zero_probability_gold": 0.0,
|
| 597 |
+
"zero_probability_gold_n": 2000,
|
| 598 |
+
"soft_accuracy": 0.4447570518885741,
|
| 599 |
+
"soft_accuracy_n": 2000,
|
| 600 |
+
"brier_soft": 0.09325671264191153,
|
| 601 |
+
"brier_soft_n": 2000,
|
| 602 |
+
"kl_gold_to_prediction": 0.17157487793178008,
|
| 603 |
+
"kl_gold_to_prediction_n": 2000,
|
| 604 |
+
"total_variation": 0.21097575386758916,
|
| 605 |
+
"total_variation_n": 2000,
|
| 606 |
+
"score_mae": 0.32394138130337413,
|
| 607 |
+
"score_mae_n": 800,
|
| 608 |
+
"within_one_level": 0.96375,
|
| 609 |
+
"within_one_level_n": 800,
|
| 610 |
+
"macro_f1": 0.5630278027844289
|
| 611 |
+
},
|
| 612 |
+
"ag-news": {
|
| 613 |
+
"attempted": 600,
|
| 614 |
+
"valid": 600,
|
| 615 |
+
"failed": 0,
|
| 616 |
+
"coverage": 1.0,
|
| 617 |
+
"accuracy_all": 0.285,
|
| 618 |
+
"accuracy_valid": 0.285,
|
| 619 |
+
"ece_top_label": 0.03904648151058483,
|
| 620 |
+
"mean_confidence": 0.3240464815105848,
|
| 621 |
+
"brier_hard": 0.7625373960165811,
|
| 622 |
+
"brier_hard_n": 600,
|
| 623 |
+
"nll_hard": 1.4140716749662658,
|
| 624 |
+
"nll_hard_n": 600,
|
| 625 |
+
"zero_probability_gold": 0.0,
|
| 626 |
+
"zero_probability_gold_n": 600,
|
| 627 |
+
"macro_f1": 0.25772655273033623
|
| 628 |
+
},
|
| 629 |
+
"emotion": {
|
| 630 |
+
"attempted": 600,
|
| 631 |
+
"valid": 600,
|
| 632 |
+
"failed": 0,
|
| 633 |
+
"coverage": 1.0,
|
| 634 |
+
"accuracy_all": 0.135,
|
| 635 |
+
"accuracy_valid": 0.135,
|
| 636 |
+
"ece_top_label": 0.31863940813484304,
|
| 637 |
+
"mean_confidence": 0.45363940813484305,
|
| 638 |
+
"brier_hard": 0.9777635802768799,
|
| 639 |
+
"brier_hard_n": 600,
|
| 640 |
+
"nll_hard": 2.137352366621794,
|
| 641 |
+
"nll_hard_n": 600,
|
| 642 |
+
"zero_probability_gold": 0.0,
|
| 643 |
+
"zero_probability_gold_n": 600,
|
| 644 |
+
"macro_f1": 0.06590661903472231
|
| 645 |
+
},
|
| 646 |
+
"boolq": {
|
| 647 |
+
"attempted": 600,
|
| 648 |
+
"valid": 600,
|
| 649 |
+
"failed": 0,
|
| 650 |
+
"coverage": 1.0,
|
| 651 |
+
"accuracy_all": 0.48833333333333334,
|
| 652 |
+
"accuracy_valid": 0.48833333333333334,
|
| 653 |
+
"ece_top_label": 0.10787050000000001,
|
| 654 |
+
"mean_confidence": 0.5912141666666667,
|
| 655 |
+
"brier_hard": 0.5250116893,
|
| 656 |
+
"brier_hard_n": 600,
|
| 657 |
+
"nll_hard": 0.719837552287629,
|
| 658 |
+
"nll_hard_n": 600,
|
| 659 |
+
"zero_probability_gold": 0.0,
|
| 660 |
+
"zero_probability_gold_n": 600,
|
| 661 |
+
"macro_f1": 0.48696381173075903
|
| 662 |
+
},
|
| 663 |
+
"sst5": {
|
| 664 |
+
"attempted": 600,
|
| 665 |
+
"valid": 600,
|
| 666 |
+
"failed": 0,
|
| 667 |
+
"coverage": 1.0,
|
| 668 |
+
"accuracy_all": 0.25666666666666665,
|
| 669 |
+
"accuracy_valid": 0.25666666666666665,
|
| 670 |
+
"ece_top_label": 0.05066152500332892,
|
| 671 |
+
"mean_confidence": 0.3073281916699956,
|
| 672 |
+
"brier_hard": 0.8215936069738192,
|
| 673 |
+
"brier_hard_n": 600,
|
| 674 |
+
"nll_hard": 1.6775697808435297,
|
| 675 |
+
"nll_hard_n": 600,
|
| 676 |
+
"zero_probability_gold": 0.0,
|
| 677 |
+
"zero_probability_gold_n": 600,
|
| 678 |
+
"score_mae": 1.245226132804566,
|
| 679 |
+
"score_mae_n": 600,
|
| 680 |
+
"within_one_level": 0.44333333333333336,
|
| 681 |
+
"within_one_level_n": 600,
|
| 682 |
+
"macro_f1": 0.13260195835186606
|
| 683 |
+
},
|
| 684 |
+
"prompt-injections": {
|
| 685 |
+
"attempted": 116,
|
| 686 |
+
"valid": 116,
|
| 687 |
+
"failed": 0,
|
| 688 |
+
"coverage": 1.0,
|
| 689 |
+
"accuracy_all": 0.5689655172413793,
|
| 690 |
+
"accuracy_valid": 0.5689655172413793,
|
| 691 |
+
"ece_top_label": 0.07609396551724136,
|
| 692 |
+
"mean_confidence": 0.6450594827586207,
|
| 693 |
+
"brier_hard": 0.4881634129310345,
|
| 694 |
+
"brier_hard_n": 116,
|
| 695 |
+
"nll_hard": 0.6835747994579267,
|
| 696 |
+
"nll_hard_n": 116,
|
| 697 |
+
"zero_probability_gold": 0.0,
|
| 698 |
+
"zero_probability_gold_n": 116,
|
| 699 |
+
"macro_f1": 0.5461658841940532
|
| 700 |
+
},
|
| 701 |
+
"massive-intent.en": {
|
| 702 |
+
"attempted": 300,
|
| 703 |
+
"valid": 300,
|
| 704 |
+
"failed": 0,
|
| 705 |
+
"coverage": 1.0,
|
| 706 |
+
"accuracy_all": 0.023333333333333334,
|
| 707 |
+
"accuracy_valid": 0.023333333333333334,
|
| 708 |
+
"ece_top_label": 0.14827139054632427,
|
| 709 |
+
"mean_confidence": 0.1716047238796576,
|
| 710 |
+
"brier_hard": 1.012268483556247,
|
| 711 |
+
"brier_hard_n": 300,
|
| 712 |
+
"nll_hard": 3.540188562309456,
|
| 713 |
+
"nll_hard_n": 300,
|
| 714 |
+
"zero_probability_gold": 0.0,
|
| 715 |
+
"zero_probability_gold_n": 300,
|
| 716 |
+
"macro_f1": 0.014601824457593688
|
| 717 |
+
},
|
| 718 |
+
"xnli.en": {
|
| 719 |
+
"attempted": 300,
|
| 720 |
+
"valid": 300,
|
| 721 |
+
"failed": 0,
|
| 722 |
+
"coverage": 1.0,
|
| 723 |
+
"accuracy_all": 0.33666666666666667,
|
| 724 |
+
"accuracy_valid": 0.33666666666666667,
|
| 725 |
+
"ece_top_label": 0.21980800490067007,
|
| 726 |
+
"mean_confidence": 0.5564746715673368,
|
| 727 |
+
"brier_hard": 0.7709342547858443,
|
| 728 |
+
"brier_hard_n": 300,
|
| 729 |
+
"nll_hard": 1.2650663651832452,
|
| 730 |
+
"nll_hard_n": 300,
|
| 731 |
+
"zero_probability_gold": 0.0,
|
| 732 |
+
"zero_probability_gold_n": 300,
|
| 733 |
+
"macro_f1": 0.20238216957239646
|
| 734 |
+
}
|
| 735 |
+
}
|
| 736 |
+
},
|
| 737 |
+
"4": {
|
| 738 |
+
"seed": 4,
|
| 739 |
+
"selected_epoch": 7,
|
| 740 |
+
"elapsed_s": 452.7137592760264,
|
| 741 |
+
"initial_validation": {
|
| 742 |
+
"soft_cross_entropy": 1.544869564374288,
|
| 743 |
+
"accuracy": 0.36833333333333335,
|
| 744 |
+
"brier_soft": 0.43051501592000324,
|
| 745 |
+
"decisions": 600
|
| 746 |
+
},
|
| 747 |
+
"selected_validation": {
|
| 748 |
+
"soft_cross_entropy": 0.8448993345101674,
|
| 749 |
+
"accuracy": 0.7533333333333333,
|
| 750 |
+
"brier_soft": 0.0668264798882107,
|
| 751 |
+
"decisions": 600
|
| 752 |
+
},
|
| 753 |
+
"stopping": {
|
| 754 |
+
"reason": "early_stopping",
|
| 755 |
+
"epochs_completed": 10,
|
| 756 |
+
"patience": 3,
|
| 757 |
+
"min_epochs": 10,
|
| 758 |
+
"epochs_without_improvement": 3,
|
| 759 |
+
"metric": "validation.soft_cross_entropy",
|
| 760 |
+
"min_delta": 0.0
|
| 761 |
+
},
|
| 762 |
+
"frozen_parameters_unchanged": true,
|
| 763 |
+
"suites": {
|
| 764 |
+
"typed-decisions": {
|
| 765 |
+
"attempted": 2000,
|
| 766 |
+
"valid": 2000,
|
| 767 |
+
"failed": 0,
|
| 768 |
+
"coverage": 1.0,
|
| 769 |
+
"accuracy_all": 0.7565,
|
| 770 |
+
"accuracy_valid": 0.7565,
|
| 771 |
+
"ece_top_label": 0.20311433485058888,
|
| 772 |
+
"mean_confidence": 0.5533856651494111,
|
| 773 |
+
"brier_hard": 0.4063906772103004,
|
| 774 |
+
"brier_hard_n": 2000,
|
| 775 |
+
"nll_hard": 0.7152837956058272,
|
| 776 |
+
"nll_hard_n": 2000,
|
| 777 |
+
"zero_probability_gold": 0.0,
|
| 778 |
+
"zero_probability_gold_n": 2000,
|
| 779 |
+
"soft_accuracy": 0.46913207673904656,
|
| 780 |
+
"soft_accuracy_n": 2000,
|
| 781 |
+
"brier_soft": 0.06497074467877281,
|
| 782 |
+
"brier_soft_n": 2000,
|
| 783 |
+
"kl_gold_to_prediction": 0.12038107186141371,
|
| 784 |
+
"kl_gold_to_prediction_n": 2000,
|
| 785 |
+
"total_variation": 0.17619801386776057,
|
| 786 |
+
"total_variation_n": 2000,
|
| 787 |
+
"score_mae": 0.2447599606823941,
|
| 788 |
+
"score_mae_n": 800,
|
| 789 |
+
"within_one_level": 0.99375,
|
| 790 |
+
"within_one_level_n": 800,
|
| 791 |
+
"macro_f1": 0.6349242079343602
|
| 792 |
+
},
|
| 793 |
+
"ag-news": {
|
| 794 |
+
"attempted": 600,
|
| 795 |
+
"valid": 600,
|
| 796 |
+
"failed": 0,
|
| 797 |
+
"coverage": 1.0,
|
| 798 |
+
"accuracy_all": 0.935,
|
| 799 |
+
"accuracy_valid": 0.935,
|
| 800 |
+
"ece_top_label": 0.0277188639855619,
|
| 801 |
+
"mean_confidence": 0.9106097216911906,
|
| 802 |
+
"brier_hard": 0.10223891825148304,
|
| 803 |
+
"brier_hard_n": 600,
|
| 804 |
+
"nll_hard": 0.2046134786862912,
|
| 805 |
+
"nll_hard_n": 600,
|
| 806 |
+
"zero_probability_gold": 0.0,
|
| 807 |
+
"zero_probability_gold_n": 600,
|
| 808 |
+
"macro_f1": 0.9319475717225341
|
| 809 |
+
},
|
| 810 |
+
"emotion": {
|
| 811 |
+
"attempted": 600,
|
| 812 |
+
"valid": 600,
|
| 813 |
+
"failed": 0,
|
| 814 |
+
"coverage": 1.0,
|
| 815 |
+
"accuracy_all": 0.565,
|
| 816 |
+
"accuracy_valid": 0.565,
|
| 817 |
+
"ece_top_label": 0.3127789762695219,
|
| 818 |
+
"mean_confidence": 0.8750361088494466,
|
| 819 |
+
"brier_hard": 0.7278820816911535,
|
| 820 |
+
"brier_hard_n": 600,
|
| 821 |
+
"nll_hard": 2.179428847187102,
|
| 822 |
+
"nll_hard_n": 600,
|
| 823 |
+
"zero_probability_gold": 0.01,
|
| 824 |
+
"zero_probability_gold_n": 600,
|
| 825 |
+
"macro_f1": 0.47063728412574746
|
| 826 |
+
},
|
| 827 |
+
"boolq": {
|
| 828 |
+
"attempted": 600,
|
| 829 |
+
"valid": 600,
|
| 830 |
+
"failed": 0,
|
| 831 |
+
"coverage": 1.0,
|
| 832 |
+
"accuracy_all": 0.8116666666666666,
|
| 833 |
+
"accuracy_valid": 0.8116666666666666,
|
| 834 |
+
"ece_top_label": 0.08128733333333335,
|
| 835 |
+
"mean_confidence": 0.8882866666666667,
|
| 836 |
+
"brier_hard": 0.2800035268666667,
|
| 837 |
+
"brier_hard_n": 600,
|
| 838 |
+
"nll_hard": 0.4526075086973918,
|
| 839 |
+
"nll_hard_n": 600,
|
| 840 |
+
"zero_probability_gold": 0.0,
|
| 841 |
+
"zero_probability_gold_n": 600,
|
| 842 |
+
"macro_f1": 0.7989317880539385
|
| 843 |
+
},
|
| 844 |
+
"sst5": {
|
| 845 |
+
"attempted": 600,
|
| 846 |
+
"valid": 600,
|
| 847 |
+
"failed": 0,
|
| 848 |
+
"coverage": 1.0,
|
| 849 |
+
"accuracy_all": 0.47333333333333333,
|
| 850 |
+
"accuracy_valid": 0.47333333333333333,
|
| 851 |
+
"ece_top_label": 0.11648579870821457,
|
| 852 |
+
"mean_confidence": 0.5717307193165773,
|
| 853 |
+
"brier_hard": 0.6735168048802583,
|
| 854 |
+
"brier_hard_n": 600,
|
| 855 |
+
"nll_hard": 1.2841517570209144,
|
| 856 |
+
"nll_hard_n": 600,
|
| 857 |
+
"zero_probability_gold": 0.0,
|
| 858 |
+
"zero_probability_gold_n": 600,
|
| 859 |
+
"score_mae": 0.6724125988264632,
|
| 860 |
+
"score_mae_n": 600,
|
| 861 |
+
"within_one_level": 0.7883333333333333,
|
| 862 |
+
"within_one_level_n": 600,
|
| 863 |
+
"macro_f1": 0.46002232794452175
|
| 864 |
+
},
|
| 865 |
+
"prompt-injections": {
|
| 866 |
+
"attempted": 116,
|
| 867 |
+
"valid": 116,
|
| 868 |
+
"failed": 0,
|
| 869 |
+
"coverage": 1.0,
|
| 870 |
+
"accuracy_all": 0.7241379310344828,
|
| 871 |
+
"accuracy_valid": 0.7241379310344828,
|
| 872 |
+
"ece_top_label": 0.22484482758620694,
|
| 873 |
+
"mean_confidence": 0.9346275862068966,
|
| 874 |
+
"brier_hard": 0.44889866172413795,
|
| 875 |
+
"brier_hard_n": 116,
|
| 876 |
+
"nll_hard": 1.682957724359292,
|
| 877 |
+
"nll_hard_n": 116,
|
| 878 |
+
"zero_probability_gold": 0.02586206896551724,
|
| 879 |
+
"zero_probability_gold_n": 116,
|
| 880 |
+
"macro_f1": 0.7118012422360249
|
| 881 |
+
},
|
| 882 |
+
"massive-intent.en": {
|
| 883 |
+
"attempted": 300,
|
| 884 |
+
"valid": 300,
|
| 885 |
+
"failed": 0,
|
| 886 |
+
"coverage": 1.0,
|
| 887 |
+
"accuracy_all": 0.7733333333333333,
|
| 888 |
+
"accuracy_valid": 0.7733333333333333,
|
| 889 |
+
"ece_top_label": 0.1762748719939652,
|
| 890 |
+
"mean_confidence": 0.9448706790799233,
|
| 891 |
+
"brier_hard": 0.3914895786578462,
|
| 892 |
+
"brier_hard_n": 300,
|
| 893 |
+
"nll_hard": 2.594218156353015,
|
| 894 |
+
"nll_hard_n": 300,
|
| 895 |
+
"zero_probability_gold": 0.06333333333333334,
|
| 896 |
+
"zero_probability_gold_n": 300,
|
| 897 |
+
"macro_f1": 0.4611733462473961
|
| 898 |
+
},
|
| 899 |
+
"xnli.en": {
|
| 900 |
+
"attempted": 300,
|
| 901 |
+
"valid": 300,
|
| 902 |
+
"failed": 0,
|
| 903 |
+
"coverage": 1.0,
|
| 904 |
+
"accuracy_all": 0.8766666666666667,
|
| 905 |
+
"accuracy_valid": 0.8766666666666667,
|
| 906 |
+
"ece_top_label": 0.059115758529190925,
|
| 907 |
+
"mean_confidence": 0.9151327612640476,
|
| 908 |
+
"brier_hard": 0.1964252077913328,
|
| 909 |
+
"brier_hard_n": 300,
|
| 910 |
+
"nll_hard": 0.36246740512671244,
|
| 911 |
+
"nll_hard_n": 300,
|
| 912 |
+
"zero_probability_gold": 0.0,
|
| 913 |
+
"zero_probability_gold_n": 300,
|
| 914 |
+
"macro_f1": 0.8760253913421515
|
| 915 |
+
}
|
| 916 |
+
}
|
| 917 |
+
}
|
| 918 |
+
},
|
| 919 |
+
"summary": {
|
| 920 |
+
"typed-decisions": {
|
| 921 |
+
"n": 5,
|
| 922 |
+
"mean_percent": 70.23,
|
| 923 |
+
"sample_variance_pp_squared": 44.56824999999997,
|
| 924 |
+
"sample_std_pp": 6.675945625901995
|
| 925 |
+
},
|
| 926 |
+
"ag-news": {
|
| 927 |
+
"n": 5,
|
| 928 |
+
"mean_percent": 54.56666666666666,
|
| 929 |
+
"sample_variance_pp_squared": 1255.536111111111,
|
| 930 |
+
"sample_std_pp": 35.433544997799906
|
| 931 |
+
},
|
| 932 |
+
"boolq": {
|
| 933 |
+
"n": 5,
|
| 934 |
+
"mean_percent": 63.266666666666666,
|
| 935 |
+
"sample_variance_pp_squared": 256.79999999999995,
|
| 936 |
+
"sample_std_pp": 16.0249804992081
|
| 937 |
+
},
|
| 938 |
+
"emotion": {
|
| 939 |
+
"n": 5,
|
| 940 |
+
"mean_percent": 30.333333333333332,
|
| 941 |
+
"sample_variance_pp_squared": 616.1666666666666,
|
| 942 |
+
"sample_std_pp": 24.82270466058577
|
| 943 |
+
},
|
| 944 |
+
"prompt-injections": {
|
| 945 |
+
"n": 5,
|
| 946 |
+
"mean_percent": 57.758620689655174,
|
| 947 |
+
"sample_variance_pp_squared": 241.5279429250891,
|
| 948 |
+
"sample_std_pp": 15.541169290793055
|
| 949 |
+
},
|
| 950 |
+
"sst5": {
|
| 951 |
+
"n": 5,
|
| 952 |
+
"mean_percent": 30.53333333333333,
|
| 953 |
+
"sample_variance_pp_squared": 185.14444444444447,
|
| 954 |
+
"sample_std_pp": 13.606779356057938
|
| 955 |
+
},
|
| 956 |
+
"massive-intent.en": {
|
| 957 |
+
"n": 5,
|
| 958 |
+
"mean_percent": 32.53333333333333,
|
| 959 |
+
"sample_variance_pp_squared": 1566.1999999999998,
|
| 960 |
+
"sample_std_pp": 39.57524478761944
|
| 961 |
+
},
|
| 962 |
+
"xnli.en": {
|
| 963 |
+
"n": 5,
|
| 964 |
+
"mean_percent": 55.53333333333334,
|
| 965 |
+
"sample_variance_pp_squared": 860.5333333333334,
|
| 966 |
+
"sample_std_pp": 29.33484844571953
|
| 967 |
+
}
|
| 968 |
+
},
|
| 969 |
+
"candidate_seed": 1,
|
| 970 |
+
"variance_definition": "sample variance, n-1, percentage points squared",
|
| 971 |
+
"benchmark_manifest": {
|
| 972 |
+
"format_version": 1,
|
| 973 |
+
"sha256": "10fb671245cdac1ee09e9c4d2b9b09829fbf1abd06e0954fddee0b9ec8519854",
|
| 974 |
+
"cases": 3516,
|
| 975 |
+
"decisions": 5116,
|
| 976 |
+
"profile": "heldout-eight",
|
| 977 |
+
"role": "test",
|
| 978 |
+
"parent_bundles": {
|
| 979 |
+
"core": "477c9cbd4f9eaf8844fa4d6870560e751960735e7441b803306d6e011e689435",
|
| 980 |
+
"language_en": "a3d41f5060826770f0f28b16a0aff26b4294691423624349db6b008789ea9096"
|
| 981 |
+
},
|
| 982 |
+
"purpose": "Base vs official typed-decisions specialist vs trained Laya-only input interface; no new training."
|
| 983 |
+
}
|
| 984 |
+
}
|
requirements.txt
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
torch==2.13.0
|
| 2 |
+
laya==0.3.20
|
| 3 |
+
transformers==5.17.0
|
| 4 |
+
huggingface-hub==1.33.0
|
| 5 |
+
safetensors==0.8.0
|
| 6 |
+
numpy==2.4.4
|
| 7 |
+
tokenizers==0.23.2
|
| 8 |
+
datasets==5.0.1
|
training_protocol.json
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"experiment": "E0i",
|
| 3 |
+
"seeds": [
|
| 4 |
+
0,
|
| 5 |
+
1,
|
| 6 |
+
2,
|
| 7 |
+
3,
|
| 8 |
+
4
|
| 9 |
+
],
|
| 10 |
+
"created_at": "2026-09-27T13:49:50.006241+00:00",
|
| 11 |
+
"data": {
|
| 12 |
+
"train": {
|
| 13 |
+
"format_version": 1,
|
| 14 |
+
"sha256": "5b273d74d4eed80f75763af64cfaf3fabae1d2a31a0b67551a5a753b6972e22e",
|
| 15 |
+
"cases": 1080,
|
| 16 |
+
"decisions": 5400,
|
| 17 |
+
"profile": "hybrid-train",
|
| 18 |
+
"role": "train",
|
| 19 |
+
"source": "LocalLLaMA/typed-decisions",
|
| 20 |
+
"revision": "f7a2487edd7a043a5441a5e9ccc7fe5ddbd9ebe8",
|
| 21 |
+
"source_split": "train",
|
| 22 |
+
"split_seed": 42
|
| 23 |
+
},
|
| 24 |
+
"validation": {
|
| 25 |
+
"format_version": 1,
|
| 26 |
+
"sha256": "95fe23fc9e82f9992c5056e7ab7ff76888aa614f49aa6d48929aba5d4605812a",
|
| 27 |
+
"cases": 120,
|
| 28 |
+
"decisions": 600,
|
| 29 |
+
"profile": "hybrid-validation",
|
| 30 |
+
"role": "validation",
|
| 31 |
+
"source": "LocalLLaMA/typed-decisions",
|
| 32 |
+
"revision": "f7a2487edd7a043a5441a5e9ccc7fe5ddbd9ebe8",
|
| 33 |
+
"source_split": "train",
|
| 34 |
+
"split_seed": 42
|
| 35 |
+
},
|
| 36 |
+
"test": {
|
| 37 |
+
"format_version": 1,
|
| 38 |
+
"sha256": "dd34a1df70013b7835b57aa474e36bf432f6e586a98e960e2b43b80ab7b75d06",
|
| 39 |
+
"cases": 400,
|
| 40 |
+
"decisions": 2000,
|
| 41 |
+
"profile": "typed-decisions-test",
|
| 42 |
+
"role": "test",
|
| 43 |
+
"source_split": "test",
|
| 44 |
+
"parent_sha256": "477c9cbd4f9eaf8844fa4d6870560e751960735e7441b803306d6e011e689435"
|
| 45 |
+
}
|
| 46 |
+
},
|
| 47 |
+
"heldout_manifest": {
|
| 48 |
+
"format_version": 1,
|
| 49 |
+
"sha256": "10fb671245cdac1ee09e9c4d2b9b09829fbf1abd06e0954fddee0b9ec8519854",
|
| 50 |
+
"cases": 3516,
|
| 51 |
+
"decisions": 5116,
|
| 52 |
+
"profile": "heldout-eight",
|
| 53 |
+
"role": "test",
|
| 54 |
+
"parent_bundles": {
|
| 55 |
+
"core": "477c9cbd4f9eaf8844fa4d6870560e751960735e7441b803306d6e011e689435",
|
| 56 |
+
"language_en": "a3d41f5060826770f0f28b16a0aff26b4294691423624349db6b008789ea9096"
|
| 57 |
+
},
|
| 58 |
+
"purpose": "Base vs official typed-decisions specialist vs trained Laya-only input interface; no new training."
|
| 59 |
+
},
|
| 60 |
+
"max_len": 1024,
|
| 61 |
+
"head_max_len": 256,
|
| 62 |
+
"epochs": null,
|
| 63 |
+
"min_epochs": 10,
|
| 64 |
+
"early_stopping_patience": 3,
|
| 65 |
+
"early_stopping_metric": "validation.soft_cross_entropy",
|
| 66 |
+
"early_stopping_min_delta": 0.0,
|
| 67 |
+
"microbatch": 8,
|
| 68 |
+
"accumulation": 4,
|
| 69 |
+
"effective_batch": 32,
|
| 70 |
+
"interface_lr": 0.0003,
|
| 71 |
+
"weight_decay": 0.01,
|
| 72 |
+
"gradient_clip": 1,
|
| 73 |
+
"training_scope": "bridge",
|
| 74 |
+
"training_dropout": false,
|
| 75 |
+
"selection": "minimum validation soft cross-entropy among trained epochs",
|
| 76 |
+
"release_candidate_selection": "lowest selected validation soft cross-entropy across all five seeds; tie goes to the lowest seed; no test-based selection",
|
| 77 |
+
"device": "cuda:1",
|
| 78 |
+
"deterministic": true,
|
| 79 |
+
"native_sdk_calibration": true,
|
| 80 |
+
"publish_policy": "draft locally; upload only after user and assistant agree",
|
| 81 |
+
"baseline_predictions_sha256": "37dff666a00a340cd6e37f775dc60df33a3c6eb78f76bfee7f094443c225b6e8",
|
| 82 |
+
"base_model": "convaiinnovations/laya",
|
| 83 |
+
"base_revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851",
|
| 84 |
+
"training_sources_sha256": {
|
| 85 |
+
"ariadne_bench/reproducibility.py": "5b0c161c596f25277b3326c645df53d5d1f2dc0272da9e77bc4a4c0d56188b6e",
|
| 86 |
+
"ariadne_bench/frozen_input_interface.py": "253f0186dac6e4a7f4b0fcf3c33449c3b3882cf8be538dca6eeca8d9e4ce9b83",
|
| 87 |
+
"ariadne_bench/full_input_interface.py": "a8f440a2185085c64896300a243ac52a9619c4d05a1b4e43c50c2c32aa1da996",
|
| 88 |
+
"ariadne_bench/full_finetune.py": "6067be8cf837d254b919c841c9a9851e1546a413dbf6324ff242e5d2f56c9734",
|
| 89 |
+
"ariadne_bench/interfaces.py": "19700fb8170b2460401900cb96ebdff22bd7bf778ca5132e98d56ba68f344187",
|
| 90 |
+
"ariadne_bench/experiments/align.py": "31e2733dcc223a1e1a6250d41c4158f84091f1cb4067d61388e06b5b01ae34c4"
|
| 91 |
+
}
|
| 92 |
+
}
|
upload-manifest.json
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"published_by_assistant": false,
|
| 3 |
+
"candidate_seed": 1,
|
| 4 |
+
"selected_epoch": 14,
|
| 5 |
+
"format": "compact affine adapter requiring pinned base Laya",
|
| 6 |
+
"existing_model_card_preserved": true,
|
| 7 |
+
"inference_code_and_weights_match_verified_export": true,
|
| 8 |
+
"removed_only_other_seed_loading_entries": true,
|
| 9 |
+
"files_sha256": {
|
| 10 |
+
"LICENSE": "a6cba85bc92e0cff7a450b1d873c0eaa2e9fc96bf472df0247a26bec77bf3ff9",
|
| 11 |
+
"THIRD_PARTY.md": "3dd2683344d45ce64c15f45c723f4b0db720deee84446365fdf4084a73ebeceb",
|
| 12 |
+
"USAGE.md": "e6d76a3a450fd589e52cab5190a065cbd783082a98d3524071bcf60a7015ebef",
|
| 13 |
+
"adapter.safetensors": "e5ec5e1d494509f63960b6949af15be8390442271805ba45d796a1911b5c239a",
|
| 14 |
+
"adapter_config.json": "8d7450b812f696e1dbeb3d7abfa13fbee0cde95aae0349a94e5f22c9b458ed1f",
|
| 15 |
+
"ariadne_bench/__init__.py": "a73c9fdf21413382cb93b523bfb3925aa3fd354e18faeead4650e86a30645411",
|
| 16 |
+
"ariadne_bench/__main__.py": "935a1c1166b0c1ea35a82256345000bf2c73ded718d77773bc27a71ecce28f7d",
|
| 17 |
+
"ariadne_bench/adapters.py": "3bd4820165be077b3b98fe83ac23e649e1ad69186ad5f2a4f5ee7addf1427cf5",
|
| 18 |
+
"ariadne_bench/cli.py": "de267c0502489eb4bb37e4b8f6faa8da8be37a56ced270f754424a40404d1af6",
|
| 19 |
+
"ariadne_bench/datasets.py": "31c722c64895519a4635b501da88862b446384682e3e8ee884495941d6990912",
|
| 20 |
+
"ariadne_bench/experiments/__init__.py": "95ad0a7f30c4e0a194d0a6e2ae8905abae5d622edbb3d4e5d791b4bc460aee5d",
|
| 21 |
+
"ariadne_bench/experiments/align.py": "31e2733dcc223a1e1a6250d41c4158f84091f1cb4067d61388e06b5b01ae34c4",
|
| 22 |
+
"ariadne_bench/experiments/prepare_alignment.py": "c4f39c10ce46ee13c0d89128d18a85ec4a86729e81e070f5834ed01fc1c405a9",
|
| 23 |
+
"ariadne_bench/experiments/spectrum.py": "22d4801b585c539d0d579a9d2a00076e1002fd337bfe31d1906d8ae3fb8d4320",
|
| 24 |
+
"ariadne_bench/frozen_input_interface.py": "253f0186dac6e4a7f4b0fcf3c33449c3b3882cf8be538dca6eeca8d9e4ce9b83",
|
| 25 |
+
"ariadne_bench/full_finetune.py": "6067be8cf837d254b919c841c9a9851e1546a413dbf6324ff242e5d2f56c9734",
|
| 26 |
+
"ariadne_bench/full_input_interface.py": "a8f440a2185085c64896300a243ac52a9619c4d05a1b4e43c50c2c32aa1da996",
|
| 27 |
+
"ariadne_bench/hybrid.py": "90fbbdab9b8157a32854d27c2d93089be6dc88dfe9320426f102eae4e03d9ffd",
|
| 28 |
+
"ariadne_bench/hybrid_control.py": "1b53678bea73fe7608570b78d7215d1dff2e38d6ad3627615f26c06ae74d8b4c",
|
| 29 |
+
"ariadne_bench/interfaces.py": "19700fb8170b2460401900cb96ebdff22bd7bf778ca5132e98d56ba68f344187",
|
| 30 |
+
"ariadne_bench/metrics.py": "fc6fde20c95d5052de52d05d6fb9b6eb69b780c46966f5bf1e517738b276a4b4",
|
| 31 |
+
"ariadne_bench/portable.py": "259f633a8bb6663d0594d9639ea8d80ef493f5a63d4b11444a4003f2f9a318e8",
|
| 32 |
+
"ariadne_bench/reproducibility.py": "5b0c161c596f25277b3326c645df53d5d1f2dc0272da9e77bc4a4c0d56188b6e",
|
| 33 |
+
"ariadne_bench/runner.py": "b20e2af4ba254e2d48a07563356f95728c4807018c3c989b429e9922674c64cd",
|
| 34 |
+
"ariadne_bench/schema.py": "67794f1172c1ef9d9625b84c8096fd48b8fa7cd10e3f88fc13ba3a98ef7c7a76",
|
| 35 |
+
"benchmarks/sources.lock.json": "6e6eaab2778092c8127257fce202c9fa8d4625dd0d029490dd9f137da91d134c",
|
| 36 |
+
"environment.json": "c5b9c796d97f3dec6d242caf79aa5d2ac69c91717f2be6ed5a51627bf6d58c17",
|
| 37 |
+
"evidence/adapter-equivalence.json": "c77faa691bd273d363dbb196155545b4f72697b9d0f4c80451ffb06b0090d200",
|
| 38 |
+
"evidence/audit.json": "1fb474a47923e22374a47e06f545e747420c0a7bd269bbbeaabb5869e89a8d70",
|
| 39 |
+
"evidence/candidate-uncertainty.json": "39eba4a205eaea5f13384da24ca8563de364c9edb6858c1db38eb4111960ce40",
|
| 40 |
+
"evidence/identity-check.json": "0cc8904dc58e1e0ed49496c7ae01980c52e3a1395d61ef91286cec2c9dcdfa25",
|
| 41 |
+
"evidence/replay-check.json": "4f246eb3d6afc0c0b1d306a20b82060ae564d476ec22451322b065edb946d15e",
|
| 42 |
+
"evidence/seed-0-history.json": "bc3de468970bf350426d9c4281456e258984838feb4d8dbbfe54c8752ebb24b7",
|
| 43 |
+
"evidence/seed-1-history.json": "8ee78b5685138712512b39035ef48a491ed52813c2131a8e9b80715871ab9405",
|
| 44 |
+
"evidence/seed-2-history.json": "d9bf7ab3991b85b6e3877b9ebd4ca2fc113cba1716f56da5aa16f4018ac4b197",
|
| 45 |
+
"evidence/seed-3-history.json": "7c0f19638e31f28e438af879ceaa68bbe06480e6f7fb72e8e749b4aabb304341",
|
| 46 |
+
"evidence/seed-4-history.json": "601c5bd91dcbb004ff8a07621f3f125ba578cc4107a9884e8fe69ac17751a741",
|
| 47 |
+
"export-verification.json": "b9f8667626e2facd03f310c0c67c100ddde9481218c51cbead7972fd048d4afb",
|
| 48 |
+
"licenses/jev-benchmarks-LICENSE": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4",
|
| 49 |
+
"licenses/laya-LICENSE": "a6cba85bc92e0cff7a450b1d873c0eaa2e9fc96bf472df0247a26bec77bf3ff9",
|
| 50 |
+
"load_adapter.py": "2f305ef95fef2b2482e9214836fc0071a59cfd6c6d3aaf42cebdbfb131272081",
|
| 51 |
+
"metrics.json": "dd45292f04dc5b81cc958a39f618aa8748b5198e39b5effee6c95520c92d94b6",
|
| 52 |
+
"requirements.txt": "285a8ae6529abeefd058f2e9679e0e314b19fda38169e685ea16aec59b053bd9",
|
| 53 |
+
"training_protocol.json": "dccf81e3d8a70c756a95a88280cbec139eea43bc391d67fa7b40e384a8824f57"
|
| 54 |
+
}
|
| 55 |
+
}
|