GoatHerder commited on
Commit
eaea151
·
verified ·
1 Parent(s): 182590e

Upload 45 files

Browse files
Files changed (45) hide show
  1. LICENSE +176 -0
  2. THIRD_PARTY.md +13 -0
  3. USAGE.md +48 -0
  4. adapter.safetensors +3 -0
  5. adapter_config.json +25 -0
  6. ariadne_bench/__init__.py +3 -0
  7. ariadne_bench/__main__.py +3 -0
  8. ariadne_bench/adapters.py +154 -0
  9. ariadne_bench/cli.py +132 -0
  10. ariadne_bench/datasets.py +532 -0
  11. ariadne_bench/experiments/__init__.py +1 -0
  12. ariadne_bench/experiments/align.py +336 -0
  13. ariadne_bench/experiments/prepare_alignment.py +43 -0
  14. ariadne_bench/experiments/spectrum.py +84 -0
  15. ariadne_bench/frozen_input_interface.py +64 -0
  16. ariadne_bench/full_finetune.py +104 -0
  17. ariadne_bench/full_input_interface.py +77 -0
  18. ariadne_bench/hybrid.py +342 -0
  19. ariadne_bench/hybrid_control.py +169 -0
  20. ariadne_bench/interfaces.py +121 -0
  21. ariadne_bench/metrics.py +229 -0
  22. ariadne_bench/portable.py +88 -0
  23. ariadne_bench/reproducibility.py +48 -0
  24. ariadne_bench/runner.py +257 -0
  25. ariadne_bench/schema.py +172 -0
  26. benchmarks/sources.lock.json +102 -0
  27. environment.json +44 -0
  28. evidence/adapter-equivalence.json +39 -0
  29. evidence/audit.json +59 -0
  30. evidence/candidate-uncertainty.json +76 -0
  31. evidence/identity-check.json +6 -0
  32. evidence/replay-check.json +32 -0
  33. evidence/seed-0-history.json +416 -0
  34. evidence/seed-1-history.json +314 -0
  35. evidence/seed-2-history.json +331 -0
  36. evidence/seed-3-history.json +484 -0
  37. evidence/seed-4-history.json +195 -0
  38. export-verification.json +42 -0
  39. licenses/jev-benchmarks-LICENSE +201 -0
  40. licenses/laya-LICENSE +176 -0
  41. load_adapter.py +14 -0
  42. metrics.json +984 -0
  43. requirements.txt +8 -0
  44. training_protocol.json +92 -0
  45. upload-manifest.json +55 -0
LICENSE ADDED
@@ -0,0 +1,176 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction,
10
+ and distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by
13
+ the copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all
16
+ other entities that control, are controlled by, or are under common
17
+ control with that entity. For the purposes of this definition,
18
+ "control" means (i) the power, direct or indirect, to cause the
19
+ direction or management of such entity, whether by contract or
20
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21
+ outstanding shares, or (iii) beneficial ownership of such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity
24
+ exercising permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation
28
+ source, and configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical
31
+ transformation or translation of a Source form, including but
32
+ not limited to compiled object code, generated documentation,
33
+ and conversions to other media types.
34
+
35
+ "Work" shall mean the work of authorship, whether in Source or
36
+ Object form, made available under the License, as indicated by a
37
+ copyright notice that is included in or attached to the work
38
+ (an example is provided in the Appendix below).
39
+
40
+ "Derivative Works" shall mean any work, whether in Source or Object
41
+ form, that is based on (or derived from) the Work and for which the
42
+ editorial revisions, annotations, elaborations, or other modifications
43
+ represent, as a whole, an original work of authorship. For the purposes
44
+ of this License, Derivative Works shall not include works that remain
45
+ separable from, or merely link (or bind by name) to the interfaces of,
46
+ the Work and Derivative Works thereof.
47
+
48
+ "Contribution" shall mean any work of authorship, including
49
+ the original version of the Work and any modifications or additions
50
+ to that Work or Derivative Works thereof, that is intentionally
51
+ submitted to Licensor for inclusion in the Work by the copyright owner
52
+ or by an individual or Legal Entity authorized to submit on behalf of
53
+ the copyright owner. For the purposes of this definition, "submitted"
54
+ means any form of electronic, verbal, or written communication sent
55
+ to the Licensor or its representatives, including but not limited to
56
+ communication on electronic mailing lists, source code control systems,
57
+ and issue tracking systems that are managed by, or on behalf of, the
58
+ Licensor for the purpose of discussing and improving the Work, but
59
+ excluding communication that is conspicuously marked or otherwise
60
+ designated in writing by the copyright owner as "Not a Contribution."
61
+
62
+ "Contributor" shall mean Licensor and any individual or Legal Entity
63
+ on behalf of whom a Contribution has been received by Licensor and
64
+ subsequently incorporated within the Work.
65
+
66
+ 2. Grant of Copyright License. Subject to the terms and conditions of
67
+ this License, each Contributor hereby grants to You a perpetual,
68
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69
+ copyright license to reproduce, prepare Derivative Works of,
70
+ publicly display, publicly perform, sublicense, and distribute the
71
+ Work and such Derivative Works in Source or Object form.
72
+
73
+ 3. Grant of Patent License. Subject to the terms and conditions of
74
+ this License, each Contributor hereby grants to You a perpetual,
75
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76
+ (except as stated in this section) patent license to make, have made,
77
+ use, offer to sell, sell, import, and otherwise transfer the Work,
78
+ where such license applies only to those patent claims licensable
79
+ by such Contributor that are necessarily infringed by their
80
+ Contribution(s) alone or by combination of their Contribution(s)
81
+ with the Work to which such Contribution(s) was submitted. If You
82
+ institute patent litigation against any entity (including a
83
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84
+ or a Contribution incorporated within the Work constitutes direct
85
+ or contributory patent infringement, then any patent licenses
86
+ granted to You under this License for that Work shall terminate
87
+ as of the date such litigation is filed.
88
+
89
+ 4. Redistribution. You may reproduce and distribute copies of the
90
+ Work or Derivative Works thereof in any medium, with or without
91
+ modifications, and in Source or Object form, provided that You
92
+ meet the following conditions:
93
+
94
+ (a) You must give any other recipients of the Work or
95
+ Derivative Works a copy of this License; and
96
+
97
+ (b) You must cause any modified files to carry prominent notices
98
+ stating that You changed the files; and
99
+
100
+ (c) You must retain, in the Source form of any Derivative Works
101
+ that You distribute, all copyright, patent, trademark, and
102
+ attribution notices from the Source form of the Work,
103
+ excluding those notices that do not pertain to any part of
104
+ the Derivative Works; and
105
+
106
+ (d) If the Work includes a "NOTICE" text file as part of its
107
+ distribution, then any Derivative Works that You distribute must
108
+ include a readable copy of the attribution notices contained
109
+ within such NOTICE file, excluding those notices that do not
110
+ pertain to any part of the Derivative Works, in at least one
111
+ of the following places: within a NOTICE text file distributed
112
+ as part of the Derivative Works; within the Source form or
113
+ documentation, if provided along with the Derivative Works; or,
114
+ within a display generated by the Derivative Works, if and
115
+ wherever such third-party notices normally appear. The contents
116
+ of the NOTICE file are for informational purposes only and
117
+ do not modify the License. You may add Your own attribution
118
+ notices within Derivative Works that You distribute, alongside
119
+ or as an addendum to the NOTICE text from the Work, provided
120
+ that such additional attribution notices cannot be construed
121
+ as modifying the License.
122
+
123
+ You may add Your own copyright statement to Your modifications and
124
+ may provide additional or different license terms and conditions
125
+ for use, reproduction, or distribution of Your modifications, or
126
+ for any such Derivative Works as a whole, provided Your use,
127
+ reproduction, and distribution of the Work otherwise complies with
128
+ the conditions stated in this License.
129
+
130
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131
+ any Contribution intentionally submitted for inclusion in the Work
132
+ by You to the Licensor shall be under the terms and conditions of
133
+ this License, without any additional terms or conditions.
134
+ Notwithstanding the above, nothing herein shall supersede or modify
135
+ the terms of any separate license agreement you may have executed
136
+ with Licensor regarding such Contributions.
137
+
138
+ 6. Trademarks. This License does not grant permission to use the trade
139
+ names, trademarks, service marks, or product names of the Licensor,
140
+ except as required for reasonable and customary use in describing the
141
+ origin of the Work and reproducing the content of the NOTICE file.
142
+
143
+ 7. Disclaimer of Warranty. Unless required by applicable law or
144
+ agreed to in writing, Licensor provides the Work (and each
145
+ Contributor provides its Contributions) on an "AS IS" BASIS,
146
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147
+ implied, including, without limitation, any warranties or conditions
148
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149
+ PARTICULAR PURPOSE. You are solely responsible for determining the
150
+ appropriateness of using or redistributing the Work and assume any
151
+ risks associated with Your exercise of permissions under this License.
152
+
153
+ 8. Limitation of Liability. In no event and under no legal theory,
154
+ whether in tort (including negligence), contract, or otherwise,
155
+ unless required by applicable law (such as deliberate and grossly
156
+ negligent acts) or agreed to in writing, shall any Contributor be
157
+ liable to You for damages, including any direct, indirect, special,
158
+ incidental, or consequential damages of any character arising as a
159
+ result of this License or out of the use or inability to use the
160
+ Work (including but not limited to damages for loss of goodwill,
161
+ work stoppage, computer failure or malfunction, or any and all
162
+ other commercial damages or losses), even if such Contributor
163
+ has been advised of the possibility of such damages.
164
+
165
+ 9. Accepting Warranty or Additional Liability. While redistributing
166
+ the Work or Derivative Works thereof, You may choose to offer,
167
+ and charge a fee for, acceptance of support, warranty, indemnity,
168
+ or other liability obligations and/or rights consistent with this
169
+ License. However, in accepting such obligations, You may act only
170
+ on Your own behalf and on Your sole responsibility, not on behalf
171
+ of any other Contributor, and only if You agree to indemnify,
172
+ defend, and hold each Contributor harmless for any liability
173
+ incurred by, or claims asserted against, such Contributor by reason
174
+ of your accepting any such warranty or additional liability.
175
+
176
+ END OF TERMS AND CONDITIONS
THIRD_PARTY.md ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Third-party sources
2
+
3
+ Ariadne's prompts and preparation recipes were adapted from:
4
+
5
+ - NandhaKishorM/laya, commit `4066d5d5fbf08b66c6757ddeedbd797bd7655bc0`, Apache-2.0. Source: https://github.com/NandhaKishorM/laya
6
+ - AbdelStark/jev-benchmarks, commit `0d610cc53e79bcbec691312b0c4adb4a0e371642`, Apache-2.0. The BTZSC grouping and balanced sample recipe follows this project. Source: https://github.com/AbdelStark/jev-benchmarks
7
+
8
+ Copyright and license notices are retained in `licenses/`. The code is adapted for this repository's canonical case format, adapters and scorer. No upstream claims of affiliation or endorsement apply.
9
+
10
+ Datasets and model weights keep their own licenses. Their source URLs, immutable revisions and model-card license metadata appear in `benchmarks/sources.lock.json` and model configs. Data is downloaded on request into ignored local directories; the harness does not relicense it. See each source card for terms, provenance and training exposure.
11
+
12
+ TypeSafe's public documentation defines the native System One request/response contract. The official evaluation website is cited for research; its examples are not redistributed.
13
+
USAGE.md ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Load the Laya linear adapter
2
+
3
+ This package contains the validation-selected **seed 1, epoch 14** checkpoint from the deterministic frozen-base experiment. It trained 1,049,600 affine parameters at LR 3e-4 while all base Laya weights remained frozen. Its 4.2 MB weight file requires the pinned base model, which the loader downloads separately.
4
+
5
+ The ongoing lower-learning-rate experiment is separate. These are the already verified weights, not a claim that the training stability investigation is complete.
6
+
7
+ ## Upload
8
+
9
+ Extract the ZIP and upload the files and folders **at the root of your existing Hugging Face model repository**. There is deliberately no README.md in this bundle, so it can coexist with your existing model card. The ZIP is a transport archive; upload its extracted contents for normal use.
10
+
11
+ This is a custom input adapter. Loading requires the bundled loader; bare `laya.load(repo_id)` and `AutoModel.from_pretrained(repo_id)` do not install the extra affine layer.
12
+
13
+ ## Load a downloaded copy
14
+
15
+ Install an appropriate PyTorch build and the versions in `requirements.txt`. The measured GPU environment is recorded in `environment.json`. From the downloaded repository directory:
16
+
17
+ ```python
18
+ from load_adapter import load
19
+
20
+ model = load(device="cuda:0")
21
+ answer = model.predict("The delivery arrived damaged.", {
22
+ "route": {
23
+ "type": "choice",
24
+ "instructions": "Choose the customer-support queue.",
25
+ "criteria": {"delivery": "Delivery and damaged items", "billing": "Payments and invoices"},
26
+ }
27
+ })
28
+ print(answer)
29
+ model.close()
30
+ ```
31
+
32
+ After you upload, first download your repository with `huggingface_hub.snapshot_download("YOUR_ACCOUNT/YOUR_REPOSITORY")`, then run the example from that downloaded directory. Use `device="cpu"` for CPU inference. `base_path` can point to the pinned base snapshot already on disk; `local_files_only=True` uses cached base files.
33
+
34
+ The loader verifies the adapter checksum, base configuration/tokenizer hashes and pretrained tensor hash. Strict determinism is enabled by default and requires the CUBLAS workspace environment to be set before CUDA is initialized. A fresh process handles this automatically. Numerical agreement across other hardware or software is not guaranteed.
35
+
36
+ ## Measured performance and limits
37
+
38
+ Selected checkpoint accuracy: typed decisions **76.95%**, AG News **93.17%**, BoolQ **79.83%**, Emotion **57.33%**, prompt injections **71.55%**, SST-5 **42.00%**, MASSIVE EN **74.33%**, XNLI EN **87.67%**. Native base typed accuracy was 36.25% under the matched protocol.
39
+
40
+ Across all five training seeds, typed accuracy was **70.23% ± 6.68 pp** (sample SD). Three seeds had severe retention losses. The selected model also lost 3.17 pp on BoolQ and 4.33 pp on MASSIVE versus base. Full results and paired uncertainty are in `metrics.json` and `evidence/candidate-uncertainty.json`. Other seeds' scores are included for transparency; this bundle contains only seed 1's weights.
41
+
42
+ The benchmark uses 2,000 typed decisions plus fixed subsets of seven other tasks, totaling 5,116 decisions. MASSIVE uses 20 candidate intents. The same test subsets were inspected in earlier experiments. These results describe an experimental task adapter, not established broad task improvement.
43
+
44
+ The base checkpoint's choice:11+ temperature is clamped from 0.1005828 to 0.5 by Laya 0.3.20. This expected load warning concerns confidence calibration; raw-logit training and class ordering are unaffected. Calibration was not refitted after adaptation.
45
+
46
+ All 5,116 answer objects matched the original selected checkpoint when the portable export was tested from outside the workspace. `export-verification.json` records that check. `upload-manifest.json` verifies that this bundle keeps the same weights and inference code, with the other seeds removed from its loading configuration.
47
+
48
+ License: Apache-2.0. Base model: `convaiinnovations/laya`, pinned to revision `55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851`. Attribution is in `THIRD_PARTY.md` and `licenses/`.
adapter.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e5ec5e1d494509f63960b6949af15be8390442271805ba45d796a1911b5c239a
3
+ size 4198616
adapter_config.json ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format_version": 1,
3
+ "interface": "exact_linear",
4
+ "position": "after native embeddings, before ModernBERT block 0",
5
+ "base_model": "convaiinnovations/laya",
6
+ "base_revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851",
7
+ "candidate_seed": 1,
8
+ "max_len": 1024,
9
+ "head_max_len": 256,
10
+ "trainable_parameters_during_training": 1049600,
11
+ "frozen_state_sha256": "7560d83b1e1c17cce7eb67b25674fc8e7f0958bac4b47f559c725f8695c4d59b",
12
+ "base_config_sha256": {
13
+ "rl_agent_config.json": "ae287b56bbcf5f8c4f4541ae9dfd00c914c4c48b940b8398c3058af37ba92bbd",
14
+ "encoder/config.json": "bf3ab80598fdccf414855a2ce80f22859e4492d06ca8a62ddd1cfb63972f8979",
15
+ "tokenizer/tokenizer.json": "6c8aaa9a542084f2457eab775d4eeb51f92a70c0fd9de28d5edb0ddec3c08d30",
16
+ "tokenizer/tokenizer_config.json": "50044de60daaa73df97d262e15a40d4faf0160e7d742df64b377877a1320dd12"
17
+ },
18
+ "seeds": {
19
+ "1": {
20
+ "weights_file": "adapter.safetensors",
21
+ "weights_sha256": "e5ec5e1d494509f63960b6949af15be8390442271805ba45d796a1911b5c239a",
22
+ "selected_epoch": 14
23
+ }
24
+ }
25
+ }
ariadne_bench/__init__.py ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ """Typed-decision benchmark harness."""
2
+
3
+ __version__ = "0.1.0"
ariadne_bench/__main__.py ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ from .cli import main
2
+
3
+ raise SystemExit(main())
ariadne_bench/adapters.py ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Adapters receive only state and questions; labels never enter model input."""
2
+
3
+ import importlib
4
+ import os
5
+ import urllib.error
6
+ import urllib.request
7
+
8
+ from .schema import dumps, labels
9
+
10
+
11
+ class Uniform:
12
+ def predict(self, state, questions):
13
+ answers = {}
14
+ for qid, q in questions.items():
15
+ keys = labels(q)
16
+ a = {"type": q["type"], "probabilities": {k: 1 / len(keys) for k in keys}}
17
+ if q["type"] == "noul":
18
+ a["noul"] = 0.5
19
+ elif q["type"] == "choice":
20
+ a["choice"] = keys[0]
21
+ else:
22
+ a["score"] = (len(keys) - 1) / 2
23
+ answers[qid] = a
24
+ return {"model": "uniform", "answers": answers}
25
+
26
+
27
+ class NativeHTTP:
28
+ def __init__(self, config):
29
+ self.endpoint = config["endpoint"]
30
+ self.model = config["model"]
31
+ self.timeout = config.get("timeout_seconds", 60)
32
+ self.key = os.environ.get(config.get("api_key_env", ""))
33
+ if config.get("api_key_env") and not self.key:
34
+ raise ValueError(f"set {config['api_key_env']} before running this adapter")
35
+ if not self.endpoint.startswith(("http://", "https://")):
36
+ raise ValueError("endpoint must be an HTTP(S) URL")
37
+
38
+ # Refuse redirects so a bearer credential cannot move to a different host.
39
+ class NoRedirect(urllib.request.HTTPRedirectHandler):
40
+ def redirect_request(self, req, fp, code, msg, headers, newurl):
41
+ return None
42
+
43
+ self.opener = urllib.request.build_opener(NoRedirect())
44
+
45
+ def predict(self, state, questions):
46
+ import json
47
+
48
+ headers = {"Content-Type": "application/json"}
49
+ if self.key:
50
+ headers["Authorization"] = "Bearer " + self.key
51
+ payload = {"model": self.model, "state": state, "questions": questions}
52
+ request = urllib.request.Request(
53
+ self.endpoint, data=dumps(payload).encode(), headers=headers
54
+ )
55
+ try:
56
+ with self.opener.open(request, timeout=self.timeout) as response:
57
+ return json.load(response)
58
+ except urllib.error.HTTPError as exc:
59
+ # Response bodies may contain credentials or echoed inputs. Store status only.
60
+ raise RuntimeError(f"HTTP {exc.code}") from None
61
+ except urllib.error.URLError:
62
+ raise RuntimeError("HTTP connection failed") from None
63
+
64
+
65
+ class Laya:
66
+ def __init__(self, config):
67
+ os.environ.setdefault("USE_TF", "0")
68
+ try:
69
+ import laya
70
+ except ImportError as exc:
71
+ raise RuntimeError("install the laya extra to use the local adapter") from exc
72
+ from pathlib import Path
73
+
74
+ model_path = Path(config["model"])
75
+ revision = None
76
+ if not model_path.is_dir():
77
+ from huggingface_hub import HfApi, snapshot_download
78
+
79
+ revision = config.get("revision")
80
+ if not (
81
+ isinstance(revision, str)
82
+ and len(revision) == 40
83
+ and all(c in "0123456789abcdef" for c in revision)
84
+ ):
85
+ revision = HfApi().model_info(config["model"], revision=revision).sha
86
+ prefix = config.get("subfolder", "").strip("/")
87
+ prefix = prefix + "/" if prefix else ""
88
+ model_path = Path(
89
+ snapshot_download(
90
+ config["model"],
91
+ revision=revision,
92
+ allow_patterns=[
93
+ prefix + name
94
+ for name in (
95
+ "rl_agent_config.json",
96
+ "model.safetensors",
97
+ "tokenizer/*",
98
+ "encoder/*",
99
+ )
100
+ ],
101
+ )
102
+ )
103
+ if config.get("subfolder"):
104
+ model_path = model_path / config["subfolder"]
105
+ self.agent = laya.load(str(model_path), device=config.get("device"))
106
+ self.max_len = config.get("max_len")
107
+ if config.get("freeze_parameters", False):
108
+ self.agent.model.requires_grad_(False)
109
+ import torch
110
+
111
+ self.metadata = {
112
+ "resolved_revision": revision,
113
+ "checkpoint_path": str(model_path.resolve()),
114
+ "device": str(self.agent.device),
115
+ "laya_version": laya.__version__,
116
+ "model_config": self.agent.cfg,
117
+ "dtype": str(self.agent.dtype),
118
+ "torch_version": torch.__version__,
119
+ "cpu_threads": torch.get_num_threads(),
120
+ "gpu_name": torch.cuda.get_device_name(self.agent.device)
121
+ if self.agent.device.type == "cuda"
122
+ else None,
123
+ "frozen_parameters": bool(config.get("freeze_parameters", False)),
124
+ "served_temperatures": [float(t) for t in self.agent.temperature],
125
+ "served_temperature_by_options": dict(self.agent.temperature_by_options),
126
+ }
127
+
128
+ def synchronize(self):
129
+ import torch
130
+
131
+ if self.agent.device.type == "cuda":
132
+ torch.cuda.synchronize(self.agent.device)
133
+ elif self.agent.device.type == "mps":
134
+ torch.mps.synchronize()
135
+ elif self.agent.device.type == "xpu":
136
+ torch.xpu.synchronize(self.agent.device)
137
+
138
+ def predict(self, state, questions):
139
+ kw = {} if self.max_len is None else {"max_len": self.max_len}
140
+ return self.agent.predict(state, questions, **kw)
141
+
142
+
143
+ def create_adapter(config):
144
+ kind = config["adapter"]
145
+ if kind == "uniform":
146
+ return Uniform()
147
+ if kind == "http":
148
+ return NativeHTTP(config)
149
+ if kind == "laya":
150
+ return Laya(config)
151
+ if kind == "python":
152
+ module, name = config["factory"].split(":", 1)
153
+ return getattr(importlib.import_module(module), name)(config.get("options", {}))
154
+ raise ValueError(f"unknown adapter: {kind}")
ariadne_bench/cli.py ADDED
@@ -0,0 +1,132 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import argparse
2
+ import json
3
+ import sys
4
+ from pathlib import Path
5
+
6
+ from .datasets import APPS, CORE, prepare
7
+ from .runner import compare, run, score
8
+ from .schema import read_jsonl, save_bundle
9
+
10
+
11
+ def positive(value):
12
+ number = int(value)
13
+ if number <= 0:
14
+ raise argparse.ArgumentTypeError("must be positive")
15
+ return number
16
+
17
+
18
+ def nonnegative(value):
19
+ number = int(value)
20
+ if number < 0:
21
+ raise argparse.ArgumentTypeError("must be nonnegative")
22
+ return number
23
+
24
+
25
+ def tolerance_value(value):
26
+ number = float(value)
27
+ if not 0 <= number <= 0.1:
28
+ raise argparse.ArgumentTypeError("tolerance must be in [0, 0.1]")
29
+ return number
30
+
31
+
32
+ def main(argv=None):
33
+ parser = argparse.ArgumentParser(
34
+ description="Freeze data, evaluate typed decisions, compare models."
35
+ )
36
+ sub = parser.add_subparsers(dest="command", required=True)
37
+ sub.add_parser("list", help="list available suites")
38
+ p = sub.add_parser("prepare", help="download pinned sources and freeze prompts")
39
+ p.add_argument(
40
+ "--profile",
41
+ choices=["smoke", "latency", "laya-core", "laya-apps", "laya-multilingual", "jev-btzsc"],
42
+ default="laya-core",
43
+ )
44
+ p.add_argument(
45
+ "--suites", help="comma-separated suites; multilingual names include language, e.g. xnli.en"
46
+ )
47
+ p.add_argument("--out", required=True)
48
+ p.add_argument("--lock", default="benchmarks/sources.lock.json")
49
+ p.add_argument("--cache", default=".cache/datasets")
50
+ p.add_argument("--limit", type=positive, help="cases per suite; changes the protocol")
51
+ p.add_argument("--seed", type=int, default=13)
52
+ p.add_argument("--permutations", type=nonnegative, default=0)
53
+ p = sub.add_parser("import", help="validate and freeze your own canonical JSONL")
54
+ p.add_argument("--cases", required=True)
55
+ p.add_argument("--out", required=True)
56
+ p = sub.add_parser("run", help="run a model; nonzero exit if any decisions fail")
57
+ p.add_argument("--data", required=True)
58
+ p.add_argument("--config", required=True)
59
+ p.add_argument("--out", required=True)
60
+ p.add_argument("--warmup", type=nonnegative, default=0)
61
+ p.add_argument("--resume", action="store_true")
62
+ p.add_argument("--ece-bins", type=positive, default=15)
63
+ p.add_argument("--probability-tolerance", type=tolerance_value, default=0.02)
64
+ p = sub.add_parser("score", help="score saved predictions without model calls")
65
+ p.add_argument("--data", required=True)
66
+ p.add_argument("--predictions", required=True)
67
+ p.add_argument("--out", required=True)
68
+ p.add_argument("--ece-bins", type=positive, default=15)
69
+ p.add_argument("--probability-tolerance", type=tolerance_value, default=0.02)
70
+ p = sub.add_parser("compare", help="compare reports from identical frozen cases")
71
+ p.add_argument("reports", nargs="+")
72
+ p.add_argument("--out")
73
+ args = parser.parse_args(argv)
74
+ try:
75
+ if args.command == "list":
76
+ print(
77
+ "\n".join(
78
+ dict.fromkeys(
79
+ CORE
80
+ + APPS
81
+ + [
82
+ "massive-intent.<language>",
83
+ "massive-scenario.<language>",
84
+ "xnli.<language>",
85
+ "btzsc.agnews",
86
+ "btzsc.emotiondair",
87
+ "btzsc.banking77",
88
+ ]
89
+ )
90
+ )
91
+ )
92
+ elif args.command == "prepare":
93
+ result = prepare(
94
+ args.profile,
95
+ args.suites,
96
+ args.out,
97
+ args.lock,
98
+ args.cache,
99
+ args.limit,
100
+ args.seed,
101
+ args.permutations,
102
+ )
103
+ print(json.dumps({k: result[k] for k in ("cases", "decisions", "sha256")}, indent=2))
104
+ elif args.command == "import":
105
+ result = save_bundle(args.out, read_jsonl(args.cases), {"profile": "custom"})
106
+ print(f"Frozen {result['cases']} cases")
107
+ elif args.command == "run":
108
+ result = run(
109
+ args.data,
110
+ args.config,
111
+ args.out,
112
+ args.ece_bins,
113
+ args.probability_tolerance,
114
+ args.warmup,
115
+ args.resume,
116
+ )
117
+ print(f"Report: {Path(args.out) / 'report.md'}")
118
+ return 2 if result["overall"]["failed"] else 0
119
+ elif args.command == "score":
120
+ result = score(
121
+ args.data, args.predictions, args.out, args.ece_bins, args.probability_tolerance
122
+ )
123
+ return 2 if result["overall"]["failed"] else 0
124
+ else:
125
+ table = compare(args.reports)
126
+ if args.out:
127
+ Path(args.out).write_text(table + "\n")
128
+ print(table)
129
+ except (ValueError, KeyError, OSError, ImportError) as exc:
130
+ print(f"Error: {exc}", file=sys.stderr)
131
+ return 1
132
+ return 0
ariadne_bench/datasets.py ADDED
@@ -0,0 +1,532 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Dataset preparation is separate from inference and freezes all prompt bytes."""
2
+
3
+ import copy
4
+ import json
5
+ import random
6
+ from collections import defaultdict
7
+ from pathlib import Path
8
+
9
+ from .schema import labels, save_bundle
10
+
11
+ LAYA_REVISION = "4066d5d5fbf08b66c6757ddeedbd797bd7655bc0"
12
+ BTZSC_REVISION = "fef2a2ac62b69c58670047dddf045c53d7c3cb5e"
13
+ CORE = ["typed-decisions", "ag-news", "emotion", "banking77", "boolq", "sst5", "prompt-injections"]
14
+ APPS = [
15
+ "ag-news",
16
+ "emotion",
17
+ "banking77",
18
+ "enron-spam",
19
+ "phishing",
20
+ "toxic-chat",
21
+ "jailbreak",
22
+ "support-triage",
23
+ "rag-relevance",
24
+ "model-routing",
25
+ ]
26
+ MASSIVE_LANGS = "en,de,fr,es,pt,ru,tr,ar,hi,ta,zh-CN,ja,ko,sw".split(",")
27
+ XNLI_LANGS = "en,de,fr,es,ru,tr,ar,hi,ur,vi,th,el,bg,zh,sw".split(",")
28
+ SPECS = {
29
+ "typed-decisions": ("LocalLLaMA/typed-decisions", "all", "test", None),
30
+ "ag-news": ("fancyzhx/ag_news", None, "test", 600),
31
+ "emotion": ("dair-ai/emotion", "split", "test", 600),
32
+ "banking77": ("mteb/banking77", None, "test", 500),
33
+ "boolq": ("google/boolq", None, "validation", 600),
34
+ "sst5": ("SetFit/sst5", None, "test", 600),
35
+ "prompt-injections": ("deepset/prompt-injections", None, "test", None),
36
+ "enron-spam": ("SetFit/enron_spam", None, "test", 400),
37
+ "phishing": ("zefang-liu/phishing-email-dataset", None, "train", 400),
38
+ "toxic-chat": ("lmsys/toxic-chat", "toxicchat0124", "test", 400),
39
+ "jailbreak": ("lmsys/toxic-chat", "toxicchat0124", "test", 400),
40
+ "support-triage": ("Tobi-Bueck/customer-support-tickets", None, "train", 400),
41
+ "rag-relevance": ("microsoft/ms_marco", "v1.1", "validation", 400),
42
+ "model-routing": ("openai/gsm8k", "main", "test", 400),
43
+ }
44
+ AG_CRITERIA = {
45
+ "world": "world news and international politics",
46
+ "sports": "sports",
47
+ "business": "business and economy",
48
+ "sci_tech": "science and technology",
49
+ }
50
+ NLI_CRITERIA = {
51
+ "entailment": "the premise implies the hypothesis is true",
52
+ "neutral": "the premise neither implies nor contradicts the hypothesis",
53
+ "contradiction": "the premise implies the hypothesis is false",
54
+ }
55
+ QUEUES = {
56
+ "Technical Support": "technical problems, bugs, outages, integrations",
57
+ "Product Support": "help using a product or feature",
58
+ "Customer Service": "general account or service questions",
59
+ "IT Support": "internal IT, devices, access, networks",
60
+ "Billing and Payments": "invoices, charges, refunds, payment methods",
61
+ "Returns and Exchanges": "returning or exchanging an item",
62
+ "Service Outages and Maintenance": "downtime, outages, scheduled maintenance",
63
+ "Sales and Pre-Sales": "pricing, quotes, buying",
64
+ "Human Resources": "employment, payroll, leave, hiring",
65
+ "General Inquiry": "anything else",
66
+ }
67
+
68
+
69
+ def parsed(value):
70
+ if isinstance(value, str):
71
+ try:
72
+ return json.loads(value)
73
+ except json.JSONDecodeError:
74
+ pass
75
+ return value
76
+
77
+
78
+ def single(suite, index, state, qid, question, target, **metadata):
79
+ keys = labels(question)
80
+ gold = {"label": keys[int(target)]}
81
+ if question["type"] == "score":
82
+ gold["score"] = float(target)
83
+ return {
84
+ "id": f"{suite}:{index}",
85
+ "suite": suite,
86
+ "state": state,
87
+ "questions": {qid: question},
88
+ "gold": {qid: gold},
89
+ **metadata,
90
+ }
91
+
92
+
93
+ def question(kind, instructions, criteria=None):
94
+ q = {"type": kind, "instructions": instructions}
95
+ if criteria is not None:
96
+ q["criteria"] = criteria
97
+ return q
98
+
99
+
100
+ class Sources:
101
+ def __init__(self, lock, cache):
102
+ self.lock = lock
103
+ self.cache = cache
104
+ self.used = {}
105
+
106
+ def load(self, repo, config=None, split="test"):
107
+ from datasets import load_dataset
108
+
109
+ revision = self.lock[repo]["revision"]
110
+ if len(revision) != 40 or any(c not in "0123456789abcdef" for c in revision):
111
+ raise ValueError(f"source lock must pin an immutable commit: {repo}")
112
+ key = f"{repo}:{config or 'default'}:{split}"
113
+ self.used[key] = {"repository": repo, "config": config, "split": split, **self.lock[repo]}
114
+ return load_dataset(repo, name=config, split=split, revision=revision, cache_dir=self.cache)
115
+
116
+
117
+ def typed(rows):
118
+ result = []
119
+ for r in rows:
120
+ qs, gold = parsed(r["questions"]), parsed(r["gold"])
121
+ gs = {}
122
+ for qid, q in qs.items():
123
+ g = gold[qid]
124
+ label = str(g["label"]).lower() if q["type"] == "noul" else str(g["label"])
125
+ gs[qid] = {"label": label, "probabilities": g["probabilities"]}
126
+ if q["type"] == "score":
127
+ gs[qid]["score"] = float(g["score"])
128
+ result.append(
129
+ {
130
+ "id": "typed-decisions:" + r["id"],
131
+ "suite": "typed-decisions",
132
+ "workflow": r["workflow"],
133
+ "state": parsed(r["state"]),
134
+ "questions": qs,
135
+ "gold": gs,
136
+ }
137
+ )
138
+ return result
139
+
140
+
141
+ def build_core(name, rows):
142
+ if name == "typed-decisions":
143
+ return typed(rows)
144
+ bank_labels = sorted(set(rows["label_text"])) if name == "banking77" else None
145
+ output = []
146
+ for i, r in enumerate(rows):
147
+ if name == "ag-news":
148
+ state, qid, q, target = (
149
+ {"article": r["text"]},
150
+ "topic",
151
+ question("choice", "What is the topic of `article`?", AG_CRITERIA),
152
+ r["label"],
153
+ )
154
+ elif name == "emotion":
155
+ state, qid, q, target = (
156
+ {"text": r["text"]},
157
+ "emotion",
158
+ question(
159
+ "choice",
160
+ "Which emotion is most strongly expressed in `text`?",
161
+ dict.fromkeys(["sadness", "joy", "love", "anger", "fear", "surprise"]),
162
+ ),
163
+ r["label"],
164
+ )
165
+ elif name == "banking77":
166
+ state, qid, q, target = (
167
+ {"message": r["text"]},
168
+ "intent",
169
+ question(
170
+ "choice",
171
+ "Which banking intent does `message` express?",
172
+ {s.replace("_", " "): None for s in bank_labels},
173
+ ),
174
+ bank_labels.index(r["label_text"]),
175
+ )
176
+ elif name == "boolq":
177
+ state, qid, q, target = (
178
+ {"passage": r["passage"], "question": r["question"]},
179
+ "answer",
180
+ question("noul", "Based on `passage`, is the answer to `question` yes?"),
181
+ int(r["answer"]),
182
+ )
183
+ elif name == "sst5":
184
+ state, qid, q, target = (
185
+ {"text": r["text"]},
186
+ "sentiment",
187
+ question(
188
+ "score",
189
+ "How positive is the sentiment of `text`?",
190
+ ["very negative", "negative", "neutral", "positive", "very positive"],
191
+ ),
192
+ r["label"],
193
+ )
194
+ else:
195
+ state, qid, q, target = (
196
+ {"text": r["text"]},
197
+ "injection",
198
+ question(
199
+ "noul",
200
+ "Does `text` try to inject or override instructions given to an AI system?",
201
+ ),
202
+ r["label"],
203
+ )
204
+ output.append(single(name, i, state, qid, q, target))
205
+ return output
206
+
207
+
208
+ def build_multilingual(name, rows, n, seed=13):
209
+ family, language = name.split(".", 1)
210
+ rng = random.Random(seed)
211
+ pool = sorted(set(rows["label_text"])) if family.startswith("massive") else []
212
+ output = []
213
+ for i, r in enumerate(rows):
214
+ if i >= n:
215
+ break
216
+ if family == "xnli":
217
+ output.append(
218
+ single(
219
+ name,
220
+ i,
221
+ {"premise": r["premise"], "hypothesis": r["hypothesis"]},
222
+ "relation",
223
+ question(
224
+ "choice",
225
+ "What is the relationship between `premise` and `hypothesis`?",
226
+ NLI_CRITERIA,
227
+ ),
228
+ r["label"],
229
+ language=language,
230
+ )
231
+ )
232
+ else:
233
+ gold = r["label_text"]
234
+ distractors = [x for x in pool if x != gold]
235
+ options = [gold] + rng.sample(distractors, min(19, len(distractors)))
236
+ rng.shuffle(options)
237
+ instructions = (
238
+ "What is the user asking for in `utterance`?"
239
+ if family == "massive-intent"
240
+ else "Which domain does `utterance` belong to?"
241
+ )
242
+ q = question(
243
+ "choice", instructions, {k: k.replace("_", " ").replace(".", ": ") for k in options}
244
+ )
245
+ output.append(
246
+ single(
247
+ name,
248
+ i,
249
+ {"utterance": r["text"]},
250
+ "label",
251
+ q,
252
+ options.index(gold),
253
+ language=language,
254
+ )
255
+ )
256
+ return output
257
+
258
+
259
+ def balanced_indices(targets, limit, seed):
260
+ groups = defaultdict(list)
261
+ for i, target in enumerate(targets):
262
+ groups[target].append(i)
263
+ rng = random.Random(seed)
264
+ for indices in groups.values():
265
+ rng.shuffle(indices)
266
+ chosen = []
267
+ while len(chosen) < min(limit, len(targets)):
268
+ for cls in sorted(groups):
269
+ if groups[cls] and len(chosen) < limit:
270
+ chosen.append(groups[cls].pop())
271
+ return sorted(chosen)
272
+
273
+
274
+ def build_btzsc(name, rows, n=100):
275
+ task = name.split(".", 1)[1]
276
+ texts = list(rows["text"])
277
+ k = next(i for i in range(1, len(texts)) if texts[i] != texts[0])
278
+ options = [str(rows[i]["hypothesis"]) for i in range(k)]
279
+ if len(rows) % k:
280
+ raise ValueError("incomplete BTZSC candidate group")
281
+ valid, targets = [], []
282
+ for start in range(0, len(rows), k):
283
+ group = rows[start : start + k]
284
+ if len(set(group["text"])) != 1 or list(group["hypothesis"]) != options:
285
+ raise ValueError("BTZSC candidate order changed")
286
+ y = [int(v) for v in group["labels"]]
287
+ if sum(y) == 1:
288
+ valid.append(start)
289
+ targets.append(y.index(1))
290
+ offset = ["agnews", "emotiondair", "banking77"].index(task)
291
+ q = question(
292
+ "choice",
293
+ "Which single label best describes the input text?",
294
+ {f"label_{i:03d}": option for i, option in enumerate(options)},
295
+ )
296
+ return [
297
+ single(name, valid[pos] // k, {"text": texts[valid[pos]]}, "label", q, targets[pos])
298
+ for pos in balanced_indices(targets, n, 20260917 + offset)
299
+ ]
300
+
301
+
302
+ def build_app(name, rows, n, seed, sources):
303
+ rng = random.Random(seed)
304
+ indexed = list(enumerate(rows))
305
+ if name == "phishing":
306
+ indexed = [
307
+ (i, r)
308
+ for i, r in indexed[:6000]
309
+ if (r.get("Email Text") or "").strip()
310
+ and r.get("Email Type") in ("Safe Email", "Phishing Email")
311
+ ]
312
+ rng.shuffle(indexed)
313
+ if name in ("toxic-chat", "jailbreak"):
314
+ field = "toxicity" if name == "toxic-chat" else "jailbreaking"
315
+ indexed = [(i, r) for i, r in indexed if (r.get("user_input") or "").strip()]
316
+ positive = [(i, r) for i, r in indexed if int(r[field]) == 1][: n // 2]
317
+ indexed = positive + [(i, r) for i, r in indexed if int(r[field]) == 0][: n - len(positive)]
318
+ rng.shuffle(indexed)
319
+ if name == "model-routing":
320
+ domains = {
321
+ "code": "software engineering, programming, refactoring, architecture, debugging",
322
+ "math_or_logic": "mathematics, logic puzzles, proofs, complex calculation",
323
+ "writing": "creative writing, essays, emails, blog posts, copywriting",
324
+ "factual_lookup": "facts, definitions, trivia, history",
325
+ "data_analysis": "statistics, SQL, data manipulation, metrics",
326
+ "chitchat": "casual conversation, greetings, small talk",
327
+ }
328
+ code = sources.load("google-research-datasets/mbpp", "full")
329
+ news = sources.load("fancyzhx/ag_news")
330
+ pool = [(r["question"], "math_or_logic") for r in list(rows)[: n // 3]]
331
+ pool += [(r["text"], "code") for r in list(code)[: n // 3]]
332
+ pool += [(r["text"][:400], "factual_lookup") for r in list(news)[: n // 3]]
333
+ rng.shuffle(pool)
334
+ q = question("choice", "What domain does `request` belong to?", domains)
335
+ return [
336
+ single(name, i, {"request": text}, "domain", q, list(domains).index(label))
337
+ for i, (text, label) in enumerate(pool)
338
+ ]
339
+ output = []
340
+ for i, r in indexed:
341
+ if len(output) >= n:
342
+ break
343
+ if name == "enron-spam":
344
+ state = {"subject": r.get("subject") or "", "body": (r.get("message") or "")[:3000]}
345
+ qid, q, target = (
346
+ "is_spam",
347
+ question("noul", "Is this email unsolicited spam or bulk marketing?"),
348
+ r["label"],
349
+ )
350
+ elif name == "phishing":
351
+ state = {"email": r["Email Text"][:3000]}
352
+ qid, q, target = (
353
+ "is_phishing",
354
+ question(
355
+ "noul",
356
+ "Is this email a phishing or scam attempt to steal money, credentials, or personal data?",
357
+ {
358
+ "true": "phishing, scam, or fraud",
359
+ "false": "a legitimate email (even if promotional)",
360
+ },
361
+ ),
362
+ int(r["Email Type"] == "Phishing Email"),
363
+ )
364
+ elif name in ("toxic-chat", "jailbreak"):
365
+ toxic = name == "toxic-chat"
366
+ state = {"post" if toxic else "prompt": r["user_input"][:3000]}
367
+ qid = "toxic" if toxic else "jailbreak"
368
+ q = question(
369
+ "noul",
370
+ "Is `post` toxic: rude, disrespectful or likely to make someone leave the discussion?"
371
+ if toxic
372
+ else "Does `prompt` try to make an AI assistant ignore its rules, policies or system instructions?",
373
+ )
374
+ target = int(r["toxicity" if toxic else "jailbreaking"])
375
+ elif name == "support-triage":
376
+ if r.get("language") != "en" or r.get("queue") not in QUEUES or not r.get("body"):
377
+ continue
378
+ state = {"subject": r["subject"] or "", "body": r["body"].replace("\\n", "\n")[:3000]}
379
+ qid, q, target = (
380
+ "queue",
381
+ question("choice", "Which support queue should handle this ticket?", QUEUES),
382
+ list(QUEUES).index(r["queue"]),
383
+ )
384
+ else:
385
+ passages = r["passages"]
386
+ pos = [t for t, s in zip(passages["passage_text"], passages["is_selected"]) if s == 1]
387
+ neg = [t for t, s in zip(passages["passage_text"], passages["is_selected"]) if s == 0]
388
+ if not pos or not neg:
389
+ continue
390
+ target = int(len(output) % 2 == 0)
391
+ state = {"query": r["query"], "passage": rng.choice(pos if target else neg)}
392
+ qid, q = "relevant", question("noul", "Does `passage` help answer `query`?")
393
+ output.append(single(name, i, state, qid, q, target))
394
+ return output
395
+
396
+
397
+ def smoke():
398
+ return [
399
+ {
400
+ "id": "smoke:0",
401
+ "suite": "smoke",
402
+ "state": "The package is late. Please refund me.",
403
+ "questions": {
404
+ "refund": question("noul", "Is a refund requested?"),
405
+ "department": question(
406
+ "choice",
407
+ "Which team should handle this?",
408
+ {"billing": "refunds", "technical": "bugs"},
409
+ ),
410
+ "urgency": question(
411
+ "score", "How urgent is the request?", ["low", "medium", "high"]
412
+ ),
413
+ },
414
+ "gold": {
415
+ "refund": {"label": "true"},
416
+ "department": {"label": "billing"},
417
+ "urgency": {"label": "1", "score": 1.0},
418
+ },
419
+ }
420
+ ]
421
+
422
+
423
+ def latency():
424
+ result = []
425
+ for n in (1, 5, 10, 50):
426
+ for i in range(30):
427
+ base = smoke()[0]
428
+ result.append(
429
+ {
430
+ "id": f"latency:{n}:{i}",
431
+ "suite": f"latency-{n}",
432
+ "state": base["state"],
433
+ "questions": {f"q{j}": base["questions"]["department"] for j in range(n)},
434
+ "gold": {f"q{j}": {"label": "billing"} for j in range(n)},
435
+ }
436
+ )
437
+ return result
438
+
439
+
440
+ def permute(cases, count, seed):
441
+ result = list(cases)
442
+ rng = random.Random(seed)
443
+ for case in cases:
444
+ if not any(q["type"] == "choice" for q in case["questions"].values()):
445
+ continue
446
+ for variant in range(1, count + 1):
447
+ c = copy.deepcopy(case)
448
+ c.update(
449
+ id=f"{case['id']}:permutation:{variant}", base_id=case["id"], permutation=variant
450
+ )
451
+ for q in c["questions"].values():
452
+ if q["type"] == "choice":
453
+ keys = list(q["criteria"])
454
+ rng.shuffle(keys)
455
+ q["criteria"] = {k: q["criteria"][k] for k in keys}
456
+ result.append(c)
457
+ return result
458
+
459
+
460
+ def prepare(profile, suites, output, lock_path, cache, limit=None, seed=13, permutations=0):
461
+ sources = (
462
+ Sources(json.loads(Path(lock_path).read_text()), cache)
463
+ if profile not in ("smoke", "latency")
464
+ else None
465
+ )
466
+ if profile == "smoke":
467
+ cases = smoke()
468
+ elif profile == "latency":
469
+ cases = latency()
470
+ else:
471
+ if suites:
472
+ names = suites.split(",")
473
+ elif profile == "laya-core":
474
+ names = CORE
475
+ elif profile == "laya-apps":
476
+ names = APPS
477
+ elif profile == "jev-btzsc":
478
+ names = ["btzsc.agnews", "btzsc.emotiondair", "btzsc.banking77"]
479
+ elif profile == "laya-multilingual":
480
+ names = [
481
+ f"{family}.{lang}"
482
+ for family in ("massive-intent", "massive-scenario")
483
+ for lang in MASSIVE_LANGS
484
+ ]
485
+ names += [f"xnli.{lang}" for lang in XNLI_LANGS]
486
+ else:
487
+ raise ValueError(f"unknown profile: {profile}")
488
+ if len(set(names)) != len(names):
489
+ raise ValueError("duplicate suites")
490
+ cases = []
491
+ for name in names:
492
+ print(f"Preparing {name} ...", flush=True)
493
+ if name.startswith("btzsc."):
494
+ rows = sources.load("btzsc/btzsc", name.split(".", 1)[1])
495
+ built = build_btzsc(name, rows, limit or 100)
496
+ elif name.startswith(("massive-intent.", "massive-scenario.", "xnli.")):
497
+ family, lang = name.split(".", 1)
498
+ repo = {
499
+ "massive-intent": "mteb/amazon_massive_intent",
500
+ "massive-scenario": "mteb/amazon_massive_scenario",
501
+ "xnli": "facebook/xnli",
502
+ }[family]
503
+ rows = sources.load(repo, lang)
504
+ built = build_multilingual(name, rows, limit or 300, seed)
505
+ else:
506
+ if name not in SPECS:
507
+ raise ValueError(f"unknown suite: {name}")
508
+ repo, config, split, default_n = SPECS[name]
509
+ rows = sources.load(repo, config, split)
510
+ n = limit or (400 if profile == "laya-apps" else default_n) or len(rows)
511
+ if name in CORE:
512
+ # Build label inventories on the full split before taking the prefix.
513
+ built = build_core(name, rows)[:n]
514
+ else:
515
+ built = build_app(name, rows, n, seed, sources)
516
+ if not built:
517
+ raise ValueError(f"suite produced no cases: {name}")
518
+ cases.extend(built)
519
+ cases = permute(cases, permutations, seed)
520
+ return save_bundle(
521
+ output,
522
+ cases,
523
+ {
524
+ "profile": profile,
525
+ "seed": seed,
526
+ "limit": limit,
527
+ "permutations": permutations,
528
+ "sources": sources.used if sources else {},
529
+ "prompt_source": f"https://github.com/NandhaKishorM/laya/tree/{LAYA_REVISION}/research",
530
+ "protocol_note": "See docs/BENCHMARKS.md for differences from historical runs; freeze and reuse this bundle.",
531
+ },
532
+ )
ariadne_bench/experiments/__init__.py ADDED
@@ -0,0 +1 @@
 
 
1
+ """Experimental training entry points, separate from benchmark evaluation."""
ariadne_bench/experiments/align.py ADDED
@@ -0,0 +1,336 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Train the hybrid connection using a separate case-level validation split."""
2
+
3
+ import argparse
4
+ import copy
5
+ import itertools
6
+ import json
7
+ import math
8
+ import random
9
+ import shutil
10
+ import time
11
+ from pathlib import Path
12
+
13
+ import torch
14
+
15
+ from ariadne_bench.hybrid import collate
16
+ from ariadne_bench.adapters import create_adapter
17
+ from ariadne_bench.schema import distribution, load_bundle, write_json
18
+
19
+
20
+ def examples(adapter, cases):
21
+ result = []
22
+ for case in cases:
23
+ items = adapter.prepare(case["state"], case["questions"])
24
+ for (qid, _), item in zip(case["questions"].items(), items):
25
+ gold = case["gold"][qid]
26
+ keys = item["keys"]
27
+ item["target"] = (
28
+ distribution(gold["probabilities"], keys)[0]
29
+ if "probabilities" in gold
30
+ else [float(k == gold["label"]) for k in keys]
31
+ )
32
+ result.append(item)
33
+ return result
34
+
35
+
36
+ def forward_loss(adapter, items):
37
+ batch = collate(items, adapter.pad_id, adapter.device)
38
+ target = torch.zeros(batch["marker_mask"].shape, device=adapter.device)
39
+ for row, item in enumerate(items):
40
+ target[row, : len(item["target"])] = torch.tensor(item["target"], device=adapter.device)
41
+ with adapter.autocast():
42
+ logits, _ = adapter.model(**batch)
43
+ loss = -(target * logits.log_softmax(-1)).sum(-1).mean()
44
+ return loss, logits, target
45
+
46
+
47
+ @torch.no_grad()
48
+ def evaluate(adapter, items, batch_size):
49
+ adapter.model.eval()
50
+ total_loss, correct, brier, count = 0.0, 0, 0.0, 0
51
+ for start in range(0, len(items), batch_size):
52
+ chunk = items[start : start + batch_size]
53
+ loss, logits, target = forward_loss(adapter, chunk)
54
+ total_loss += loss.item() * len(chunk)
55
+ correct += (logits.argmax(-1) == target.argmax(-1)).sum().item()
56
+ brier += ((logits.softmax(-1) - target) ** 2).sum().item()
57
+ count += len(chunk)
58
+ return {
59
+ "soft_cross_entropy": total_loss / count,
60
+ "accuracy": correct / count,
61
+ "brier_soft": brier / count,
62
+ "decisions": count,
63
+ }
64
+
65
+
66
+ class ValidationPatience:
67
+ """Count consecutive epochs that fail to beat the best validation loss."""
68
+
69
+ def __init__(self, patience=None, min_epochs=10):
70
+ if patience is not None and patience < 1:
71
+ raise ValueError("patience must be positive")
72
+ if min_epochs < 1:
73
+ raise ValueError("minimum epochs must be positive")
74
+ self.patience = patience
75
+ self.min_epochs = min_epochs
76
+ self.epochs = 0
77
+ self.best = float("inf")
78
+ self.stale = 0
79
+
80
+ def observe(self, loss):
81
+ if not math.isfinite(loss):
82
+ raise ValueError("validation loss must be finite")
83
+ self.epochs += 1
84
+ improved = loss < self.best
85
+ if improved:
86
+ self.best = loss
87
+ self.stale = 0
88
+ else:
89
+ self.stale += 1
90
+ return improved, (
91
+ self.patience is not None
92
+ and self.stale >= self.patience
93
+ and self.epochs >= self.min_epochs
94
+ )
95
+
96
+
97
+ def main():
98
+ p = argparse.ArgumentParser(description=__doc__)
99
+ p.add_argument("--config", required=True)
100
+ p.add_argument("--train", required=True)
101
+ p.add_argument("--validation", required=True)
102
+ p.add_argument("--out", required=True)
103
+ p.add_argument(
104
+ "--epochs", type=int, default=3, help="epoch limit; 0 means unlimited with early stopping"
105
+ )
106
+ p.add_argument(
107
+ "--early-stopping-patience",
108
+ type=int,
109
+ help="stop after this many consecutive epochs without a new lowest validation loss",
110
+ )
111
+ p.add_argument(
112
+ "--min-epochs",
113
+ type=int,
114
+ default=10,
115
+ help="minimum completed epochs before early stopping can fire",
116
+ )
117
+ p.add_argument("--batch-size", type=int, default=8)
118
+ p.add_argument("--accumulation", type=int, default=4)
119
+ p.add_argument("--lr", type=float, default=3e-4)
120
+ p.add_argument("--device", help="override config device, e.g. cuda:1")
121
+ p.add_argument("--head-lr", type=float, help="full fine-tuning head learning rate")
122
+ p.add_argument(
123
+ "--keep-best-only",
124
+ action="store_true",
125
+ help="remove superseded checkpoints created by this run",
126
+ )
127
+ args = p.parse_args()
128
+ if (
129
+ min(args.batch_size, args.accumulation) < 1
130
+ or args.epochs < 0
131
+ or args.min_epochs < 1
132
+ or (args.early_stopping_patience is not None and 0 < args.epochs < args.min_epochs)
133
+ or (args.epochs == 0 and args.early_stopping_patience is None)
134
+ or (args.early_stopping_patience is not None and args.early_stopping_patience < 1)
135
+ or not math.isfinite(args.lr)
136
+ or args.lr <= 0
137
+ ):
138
+ p.error(
139
+ "use positive batch/accumulation/lr/patience and a positive epoch limit, or epochs=0 with patience"
140
+ )
141
+ train, train_manifest = load_bundle(args.train)
142
+ validation, validation_manifest = load_bundle(args.validation)
143
+ if train_manifest.get("role") != "train" or validation_manifest.get("role") != "validation":
144
+ raise ValueError(
145
+ "explicit train and validation bundle roles are required; do not use benchmark test data"
146
+ )
147
+ if {c["id"] for c in train} & {c["id"] for c in validation}:
148
+ raise ValueError("train and validation case IDs overlap")
149
+ config = json.loads(Path(args.config).read_text())
150
+ if args.device:
151
+ config["options"]["device"] = args.device
152
+ scope = config["options"].get("training_scope", "bridge")
153
+ if scope not in ("bridge", "full"):
154
+ raise ValueError("supported training scopes are bridge and full")
155
+ if args.head_lr is not None and (not math.isfinite(args.head_lr) or args.head_lr <= 0):
156
+ p.error("head learning rate must be positive")
157
+ out = Path(args.out)
158
+ out.mkdir(parents=True, exist_ok=False)
159
+ adapter = create_adapter(config)
160
+ train_items, validation_items = examples(adapter, train), examples(adapter, validation)
161
+ parameters = [p for p in adapter.model.parameters() if p.requires_grad]
162
+ frozen_versions = {
163
+ name: value._version
164
+ for name, value in adapter.model.named_parameters()
165
+ if not value.requires_grad
166
+ }
167
+ if scope == "full":
168
+ optimizer_parameters = [
169
+ {
170
+ "params": [
171
+ p
172
+ for n, p in adapter.model.named_parameters()
173
+ if p.requires_grad and n.startswith("encoder.")
174
+ ],
175
+ "lr": args.lr,
176
+ },
177
+ {
178
+ "params": [
179
+ p
180
+ for n, p in adapter.model.named_parameters()
181
+ if p.requires_grad and not n.startswith("encoder.")
182
+ ],
183
+ "lr": args.head_lr or args.lr,
184
+ },
185
+ ]
186
+ else:
187
+ optimizer_parameters = parameters
188
+ optimizer = torch.optim.AdamW(optimizer_parameters, lr=args.lr, weight_decay=0.01)
189
+ rng = random.Random(config["options"].get("seed", 42))
190
+ log = {
191
+ "configuration": vars(args),
192
+ "model": adapter.metadata,
193
+ "train_manifest": train_manifest,
194
+ "validation_manifest": validation_manifest,
195
+ "trainable_names": [n for n, p in adapter.model.named_parameters() if p.requires_grad],
196
+ "initial_validation": evaluate(adapter, validation_items, args.batch_size),
197
+ "epochs": [],
198
+ }
199
+ write_json(out / "training.json", log)
200
+ print(
201
+ json.dumps(
202
+ {
203
+ "initial_validation": log["initial_validation"],
204
+ "trainable_parameters": sum(p.numel() for p in parameters),
205
+ }
206
+ ),
207
+ flush=True,
208
+ )
209
+ start_time = time.perf_counter()
210
+ best_checkpoint = None
211
+ stopping = ValidationPatience(args.early_stopping_patience, args.min_epochs)
212
+ epochs = range(1, args.epochs + 1) if args.epochs else itertools.count(1)
213
+ for epoch in epochs:
214
+ rng.shuffle(train_items)
215
+ adapter.training_mode()
216
+ total_loss, completed = 0.0, 0
217
+ effective_batch = args.batch_size * args.accumulation
218
+ for start in range(0, len(train_items), effective_batch):
219
+ group = train_items[start : start + effective_batch]
220
+ optimizer.zero_grad(set_to_none=True)
221
+ for micro in range(0, len(group), args.batch_size):
222
+ chunk = group[micro : micro + args.batch_size]
223
+ loss, _, _ = forward_loss(adapter, chunk)
224
+ if not torch.isfinite(loss):
225
+ raise RuntimeError("nonfinite alignment loss")
226
+ (loss * (len(chunk) / len(group))).backward()
227
+ total_loss += loss.item() * len(chunk)
228
+ completed += len(chunk)
229
+ norm = torch.nn.utils.clip_grad_norm_(parameters, 1.0, error_if_nonfinite=True)
230
+ optimizer.step()
231
+ if start == 0 or (start // effective_batch + 1) % 20 == 0:
232
+ adapter.synchronize()
233
+ elapsed = time.perf_counter() - start_time
234
+ processed = (epoch - 1) * len(train_items) + completed
235
+ remaining = (
236
+ ((args.epochs * len(train_items) - processed) * elapsed / processed)
237
+ if args.epochs
238
+ else None
239
+ )
240
+ print(
241
+ json.dumps(
242
+ {
243
+ "epoch": epoch,
244
+ "decisions": completed,
245
+ "train_ce": total_loss / completed,
246
+ "gradient_norm": norm.item(),
247
+ "elapsed_s": elapsed,
248
+ "estimated_remaining_s": remaining,
249
+ "peak_gpu_allocated_gb": torch.cuda.max_memory_allocated(adapter.device)
250
+ / 1e9
251
+ if adapter.device.type == "cuda"
252
+ else None,
253
+ }
254
+ ),
255
+ flush=True,
256
+ )
257
+ stats = {
258
+ "epoch": epoch,
259
+ "train_soft_cross_entropy": total_loss / completed,
260
+ "validation": evaluate(adapter, validation_items, args.batch_size),
261
+ }
262
+ log["epochs"].append(stats)
263
+ improved, should_stop = stopping.observe(stats["validation"]["soft_cross_entropy"])
264
+ if improved:
265
+ previous_checkpoint = best_checkpoint
266
+ best_checkpoint = out / f"epoch-{epoch}"
267
+ adapter.save_checkpoint(
268
+ best_checkpoint,
269
+ {
270
+ "scope": "full decision path; unused auxiliary action head frozen"
271
+ if scope == "full"
272
+ else "interface only; pretrained source parameters frozen",
273
+ "epoch": epoch,
274
+ "train_sha256": train_manifest["sha256"],
275
+ "validation_sha256": validation_manifest["sha256"],
276
+ "selection": "lowest validation soft cross entropy",
277
+ "validation": stats["validation"],
278
+ },
279
+ )
280
+ if args.keep_best_only and previous_checkpoint is not None:
281
+ # Only delete an earlier checkpoint created within this new run.
282
+ if previous_checkpoint.parent != out or not previous_checkpoint.name.startswith(
283
+ "epoch-"
284
+ ):
285
+ raise ValueError("refusing to remove a checkpoint outside this run")
286
+ shutil.rmtree(previous_checkpoint)
287
+ if args.early_stopping_patience is not None:
288
+ stats["early_stopping"] = {
289
+ "patience": stopping.patience,
290
+ "min_epochs": stopping.min_epochs,
291
+ "epochs_without_improvement": stopping.stale,
292
+ "best_validation_loss": stopping.best,
293
+ "improved": improved,
294
+ }
295
+ write_json(out / "training.json", log)
296
+ print(json.dumps(stats), flush=True)
297
+ if should_stop:
298
+ break
299
+ log["stopping"] = {
300
+ "reason": "early_stopping" if should_stop else "epoch_limit",
301
+ "epochs_completed": len(log["epochs"]),
302
+ "patience": stopping.patience,
303
+ "min_epochs": stopping.min_epochs,
304
+ "epochs_without_improvement": stopping.stale,
305
+ "metric": "validation.soft_cross_entropy",
306
+ "min_delta": 0.0,
307
+ }
308
+ assert all(
309
+ value._version == frozen_versions[name] and value.grad is None
310
+ for name, value in adapter.model.named_parameters()
311
+ if name in frozen_versions
312
+ )
313
+ log.update(
314
+ {
315
+ "elapsed_s": time.perf_counter() - start_time,
316
+ "frozen_parameters_unchanged": True,
317
+ "best_checkpoint": str(best_checkpoint.resolve()),
318
+ }
319
+ )
320
+ write_json(out / "training.json", log)
321
+ trained_config = copy.deepcopy(config)
322
+ trained_config["name"] = config["name"].replace("(UNTRAINED)", "(alignment only)")
323
+ if scope == "full":
324
+ trained_config["name"] = config["name"].replace("(UNTRAINED)", "(full fine-tuned)")
325
+ trained_config["options"]["inference_only"] = True
326
+ trained_config["mode"] = "specialist"
327
+ trained_config["options"]["checkpoint"] = str(best_checkpoint.resolve())
328
+ write_json(out / "benchmark-config.json", trained_config)
329
+ print(
330
+ json.dumps({"best_checkpoint": str(best_checkpoint), "elapsed_s": log["elapsed_s"]}),
331
+ flush=True,
332
+ )
333
+
334
+
335
+ if __name__ == "__main__":
336
+ main()
ariadne_bench/experiments/prepare_alignment.py ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Freeze disjoint train/validation/test bundles for the interface experiment."""
2
+
3
+ import argparse
4
+ import random
5
+ from pathlib import Path
6
+
7
+ from ariadne_bench.datasets import typed
8
+ from ariadne_bench.schema import save_bundle
9
+
10
+
11
+ def main():
12
+ from datasets import load_dataset
13
+
14
+ parser = argparse.ArgumentParser(description=__doc__)
15
+ parser.add_argument("--out", required=True)
16
+ parser.add_argument("--cache", default=".cache/datasets")
17
+ args = parser.parse_args()
18
+ out = Path(args.out)
19
+ if out.exists():
20
+ raise ValueError("choose a new output directory")
21
+ revision = "f7a2487edd7a043a5441a5e9ccc7fe5ddbd9ebe8"
22
+ ds = load_dataset("LocalLLaMA/typed-decisions", "all", revision=revision, cache_dir=args.cache)
23
+ train, test = typed(ds["train"]), typed(ds["test"])
24
+ if {c["id"] for c in train} & {c["id"] for c in test}:
25
+ raise ValueError("source training and test IDs overlap")
26
+ random.Random(42).shuffle(train)
27
+ for role, rows in [("train", train[120:]), ("validation", train[:120]), ("test", test)]:
28
+ save_bundle(
29
+ out / role,
30
+ rows,
31
+ {
32
+ "profile": "hybrid-" + role,
33
+ "role": role,
34
+ "source": "LocalLLaMA/typed-decisions",
35
+ "revision": revision,
36
+ "source_split": "test" if role == "test" else "train",
37
+ "split_seed": 42,
38
+ },
39
+ )
40
+
41
+
42
+ if __name__ == "__main__":
43
+ main()
ariadne_bench/experiments/spectrum.py ADDED
@@ -0,0 +1,84 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Analyze both the complete learned linear map and its departure from identity."""
2
+
3
+ import argparse
4
+ import json
5
+ from pathlib import Path
6
+
7
+ import torch
8
+ from safetensors.torch import load_file, save_file
9
+
10
+ from ariadne_bench.schema import write_json
11
+
12
+
13
+ def describe(matrix):
14
+ u, s, vh = torch.linalg.svd(matrix.double(), full_matrices=False)
15
+ energy = s.square()
16
+ cumulative = energy.cumsum(0) / energy.sum()
17
+ probability = s / s.sum()
18
+ entropy = -(probability * probability.clamp_min(1e-300).log()).sum()
19
+ eigen = torch.linalg.eigvals(matrix.double())
20
+ result = {
21
+ "frobenius_norm": matrix.double().norm().item(),
22
+ "singular_values": s.tolist(),
23
+ "cumulative_squared_singular_value_fraction": cumulative.tolist(),
24
+ "singular_value_entropy_effective_rank": entropy.exp().item(),
25
+ "stable_rank": (energy.sum() / energy[0]).item(),
26
+ "numerical_rank": int(torch.linalg.matrix_rank(matrix.double())),
27
+ "condition_number": (s[0] / s[-1]).item(),
28
+ "ranks_for_energy": {
29
+ str(p): int(torch.searchsorted(cumulative, torch.tensor(p, dtype=cumulative.dtype))) + 1
30
+ for p in [0.5, 0.9, 0.95, 0.99]
31
+ },
32
+ "eigenvalues": [[z.real.item(), z.imag.item()] for z in eigen],
33
+ }
34
+ return result, {"u": u.float(), "s": s.float(), "vh": vh.float()}
35
+
36
+
37
+ def main():
38
+ parser = argparse.ArgumentParser(description=__doc__)
39
+ parser.add_argument("--checkpoint", default="runs/hybrid-laya-control/epoch-3")
40
+ parser.add_argument("--out", default="runs/rank-ablation/spectrum")
41
+ args = parser.parse_args()
42
+ source = Path(args.checkpoint)
43
+ out = Path(args.out)
44
+ out.mkdir(parents=True, exist_ok=False)
45
+ state = load_file(source / "adapter.safetensors")
46
+ weight = state["encoder.bridge.1.weight"]
47
+ identity = torch.eye(weight.shape[0], dtype=weight.dtype)
48
+ result = {
49
+ "source_manifest": json.loads((source / "manifest.json").read_text()),
50
+ "width": weight.shape[0],
51
+ "interpretation": "Singular energy describes the matrix, not data-weighted activation variance; W and W-I answer different questions.",
52
+ }
53
+ factors = {}
54
+ for name, matrix in [("weight", weight), ("delta", weight - identity)]:
55
+ result[name], values = describe(matrix)
56
+ factors.update({name + "." + k: v for k, v in values.items()})
57
+ result["relative_delta_frobenius"] = (weight - identity).norm().item() / identity.norm().item()
58
+ result["layernorm_scale_change_l2"] = (state["encoder.bridge.0.weight"] - 1).norm().item()
59
+ result["layernorm_bias_l2"] = state["encoder.bridge.0.bias"].norm().item()
60
+ result["linear_bias_l2"] = state["encoder.bridge.1.bias"].norm().item()
61
+ save_file({k:v.contiguous() for k,v in factors.items()}, out / "svd.safetensors")
62
+ write_json(out / "analysis.json", result)
63
+ print(
64
+ json.dumps(
65
+ {
66
+ name: {
67
+ k: v
68
+ for k, v in result[name].items()
69
+ if k
70
+ not in (
71
+ "singular_values",
72
+ "eigenvalues",
73
+ "cumulative_squared_singular_value_fraction",
74
+ )
75
+ }
76
+ for name in ("weight", "delta")
77
+ },
78
+ indent=2,
79
+ )
80
+ )
81
+
82
+
83
+ if __name__ == "__main__":
84
+ main()
ariadne_bench/frozen_input_interface.py ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Train only the exact-identity input affine map on frozen native Laya."""
2
+
3
+ import copy
4
+ import hashlib
5
+
6
+ from .full_input_interface import FullInputInterface
7
+
8
+
9
+ class FrozenInputInterface(FullInputInterface):
10
+ def __init__(self, options):
11
+ if not options.get("add_interface", True):
12
+ raise ValueError("the frozen-interface adapter requires its linear interface")
13
+ initial = copy.deepcopy(options)
14
+ initial.pop("checkpoint", None)
15
+ initial.pop("inference_only", None)
16
+ super().__init__(initial)
17
+ self.options = copy.deepcopy(options)
18
+ self.model.requires_grad_(False)
19
+ self.model.encoder.embeddings.interface.requires_grad_(True)
20
+ self.training_scope = "bridge"
21
+ self.identity["training_scope"] = "bridge"
22
+ self.metadata.update(
23
+ **self.identity,
24
+ trainable_parameters=sum(p.numel() for p in self.model.parameters() if p.requires_grad),
25
+ frozen_pretrained_model=True,
26
+ )
27
+ self.initial_frozen_sha256 = self.frozen_state_sha256()
28
+ self.metadata["frozen_state_sha256"] = self.initial_frozen_sha256
29
+ if options.get("checkpoint"):
30
+ self.load_checkpoint(options["checkpoint"])
31
+ if options.get("inference_only", False):
32
+ self.model.requires_grad_(False)
33
+ self.model.eval()
34
+
35
+ def training_mode(self):
36
+ # Autograd still traverses the frozen stack to compute interface gradients.
37
+ self.model.eval()
38
+
39
+ def checkpoint_tensor_names(self):
40
+ return {
41
+ name
42
+ for name in self.model.state_dict()
43
+ if name.startswith("encoder.embeddings.interface.")
44
+ }
45
+
46
+ def frozen_state_sha256(self):
47
+ digest = hashlib.sha256()
48
+ for name, value in sorted(self.model.state_dict().items()):
49
+ if name.startswith("encoder.embeddings.interface."):
50
+ continue
51
+ digest.update(name.encode())
52
+ digest.update(str((value.dtype, tuple(value.shape))).encode())
53
+ digest.update(value.detach().cpu().contiguous().numpy().tobytes())
54
+ return digest.hexdigest()
55
+
56
+ def save_checkpoint(self, directory, training):
57
+ actual = self.frozen_state_sha256()
58
+ if actual != self.initial_frozen_sha256:
59
+ raise RuntimeError("frozen pretrained tensors changed")
60
+ super().save_checkpoint(directory, {**training, "frozen_state_sha256": actual})
61
+
62
+
63
+ def create(options):
64
+ return FrozenInputInterface(options)
ariadne_bench/full_finetune.py ADDED
@@ -0,0 +1,104 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Full task-model fine-tuning of the native Laya specialist."""
2
+
3
+ import copy
4
+ from pathlib import Path
5
+
6
+ import torch
7
+
8
+ from .hybrid import Hybrid
9
+ from .hybrid_control import LayaControl
10
+
11
+
12
+ class FullFineTune(Hybrid):
13
+ prepare = LayaControl.prepare
14
+
15
+ def __init__(self, options):
16
+ import laya
17
+ import transformers
18
+
19
+ self.options = copy.deepcopy(options)
20
+ if options.get("deterministic", False):
21
+ from .reproducibility import configure_determinism
22
+
23
+ configure_determinism(options.get("seed", 42))
24
+ torch.manual_seed(options.get("seed", 42))
25
+ self.device = torch.device(options.get("device", "cpu"))
26
+ if self.device.type == "cuda":
27
+ torch.cuda.set_device(self.device)
28
+ path = Path(options["laya_model"]).resolve(strict=True)
29
+ self.agent = laya.load(str(path), device=str(self.device))
30
+ if self.agent.device != self.device:
31
+ raise RuntimeError("native Laya loaded on an unexpected device")
32
+ self.model, self.tokenizer = self.agent.model, self.agent.tok
33
+ self.dtype = self.agent.dtype
34
+ self.pad_id = self.tokenizer.pad_token_id
35
+ self.max_len = int(options.get("max_len", self.agent.cfg.get("max_len", 512)))
36
+ self.head_max_len = self.agent.cfg.get("head_max_len", 192)
37
+ self.batch_size = int(options.get("batch_size", 16))
38
+ self.training_scope = "full"
39
+ self.native_prediction = True
40
+ self.model.requires_grad_(True)
41
+ # The supervised decision loss has no action/reward target. All parameters
42
+ # on the decision path train; the unused auxiliary action head stays frozen.
43
+ self.model.act_head.requires_grad_(False)
44
+ if options.get("gradient_checkpointing", False):
45
+ self.model.encoder.gradient_checkpointing_enable(
46
+ gradient_checkpointing_kwargs={"use_reentrant": False}
47
+ )
48
+ self.model.head_checkpointing = True
49
+ self.identity = {
50
+ "format_version": 1,
51
+ "laya_model": str(path),
52
+ "training_scope": "full",
53
+ "prompt_format": "original-laya-v1",
54
+ "max_len": self.max_len,
55
+ "head_max_len": self.head_max_len,
56
+ "native_prediction": True,
57
+ }
58
+ self.metadata = {
59
+ **self.identity,
60
+ "device": str(self.device),
61
+ "gpu_name": torch.cuda.get_device_name(self.device)
62
+ if self.device.type == "cuda"
63
+ else None,
64
+ "dtype": str(self.dtype),
65
+ "torch_version": torch.__version__,
66
+ "transformers_version": transformers.__version__,
67
+ "laya_version": laya.__version__,
68
+ "parameters": sum(p.numel() for p in self.model.parameters()),
69
+ "trainable_parameters": sum(
70
+ p.numel() for p in self.model.parameters() if p.requires_grad
71
+ ),
72
+ "frozen_auxiliary_action_head": True,
73
+ "gradient_checkpointing": bool(options.get("gradient_checkpointing", False)),
74
+ "temperature": list(self.agent.temperature),
75
+ "temperature_by_options": dict(self.agent.temperature_by_options),
76
+ }
77
+ if options.get("deterministic", False):
78
+ from .reproducibility import settings
79
+
80
+ self.metadata["reproducibility"] = settings()
81
+ if options.get("checkpoint"):
82
+ self.load_checkpoint(options["checkpoint"])
83
+ if options.get("inference_only", False):
84
+ self.model.requires_grad_(False)
85
+ self.model.eval()
86
+
87
+ def close(self):
88
+ if hasattr(self, "agent"):
89
+ del self.agent
90
+ super().close()
91
+
92
+ def training_mode(self):
93
+ self.model.train()
94
+
95
+ def checkpoint_tensor_names(self):
96
+ return set(self.model.state_dict())
97
+
98
+ def predict(self, state, questions):
99
+ self.model.eval()
100
+ return self.agent.predict(state, questions, max_len=self.max_len)
101
+
102
+
103
+ def create(options):
104
+ return FullFineTune(options)
ariadne_bench/full_input_interface.py ADDED
@@ -0,0 +1,77 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Jointly fine-tune native Laya and an identity-initialized input affine map."""
2
+
3
+ import copy
4
+
5
+ import torch
6
+ from torch import nn
7
+
8
+ from .full_finetune import FullFineTune
9
+ from .interfaces import IdentityAffine
10
+
11
+
12
+ class TrainableInputInterface(nn.Module):
13
+ """Keep gradients through the original embedding, normalization, and map."""
14
+
15
+ def __init__(self, original, width):
16
+ super().__init__()
17
+ self.original = original
18
+ self.interface = IdentityAffine(width)
19
+
20
+ @property
21
+ def tok_embeddings(self):
22
+ return self.original.tok_embeddings
23
+
24
+ def forward(self, input_ids=None, inputs_embeds=None):
25
+ hidden = self.original(input_ids=input_ids, inputs_embeds=inputs_embeds)
26
+ return self.interface(hidden)
27
+
28
+
29
+ class FullInputInterface(FullFineTune):
30
+ def __init__(self, options):
31
+ initial_options = copy.deepcopy(options)
32
+ initial_options.pop("checkpoint", None)
33
+ initial_options.pop("inference_only", None)
34
+ super().__init__(initial_options)
35
+ self.options = copy.deepcopy(options)
36
+ self.head_max_len = int(options.get("head_max_len", self.head_max_len))
37
+ if self.head_max_len < 1:
38
+ raise ValueError("head_max_len must be positive")
39
+ torch.set_float32_matmul_precision("highest")
40
+ add_interface = options.get("add_interface", True)
41
+ if add_interface:
42
+ encoder = self.model.encoder
43
+ encoder.embeddings = TrainableInputInterface(
44
+ encoder.embeddings, encoder.config.hidden_size
45
+ ).to(self.device)
46
+ self.identity.update(
47
+ head_max_len=self.head_max_len,
48
+ interface={"kind": "exact_linear", "position": "after native embeddings"}
49
+ if add_interface
50
+ else None,
51
+ )
52
+ self.metadata.update(
53
+ **self.identity,
54
+ parameters=sum(p.numel() for p in self.model.parameters()),
55
+ trainable_parameters=sum(p.numel() for p in self.model.parameters() if p.requires_grad),
56
+ float32_matmul_precision=torch.get_float32_matmul_precision(),
57
+ interface_parameters=sum(
58
+ p.numel() for p in self.model.encoder.embeddings.interface.parameters()
59
+ )
60
+ if add_interface
61
+ else 0,
62
+ )
63
+ if options.get("checkpoint"):
64
+ self.load_checkpoint(options["checkpoint"])
65
+ if options.get("inference_only", False):
66
+ self.model.requires_grad_(False)
67
+ self.model.eval()
68
+
69
+ def predict(self, state, questions):
70
+ self.model.eval()
71
+ return self.agent.predict(
72
+ state, questions, max_len=self.max_len, head_max_len=self.head_max_len
73
+ )
74
+
75
+
76
+ def create(options):
77
+ return FullInputInterface(options)
ariadne_bench/hybrid.py ADDED
@@ -0,0 +1,342 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Experimental small causal LM -> ModernBERT -> Laya decision model.
2
+
3
+ Optional torch/transformers/laya dependencies are imported only by this module.
4
+ No generation, vocabulary output head, or persistent KV cache is used.
5
+ """
6
+
7
+ import copy
8
+ import hashlib
9
+ import json
10
+ from contextlib import nullcontext
11
+ from pathlib import Path
12
+
13
+ import torch
14
+ from torch import nn
15
+
16
+ from .schema import labels
17
+
18
+
19
+ class ContextualEmbeddings(nn.Module):
20
+ """Replace ModernBERT's entire token lookup / normalization / dropout module."""
21
+
22
+ def forward(self, input_ids=None, inputs_embeds=None):
23
+ if input_ids is not None or inputs_embeds is None:
24
+ raise ValueError("the hybrid encoder accepts contextual embeddings only")
25
+ return inputs_embeds
26
+
27
+
28
+ class SemanticEncoder(nn.Module):
29
+ def __init__(self, llm, bert, layer_indices):
30
+ super().__init__()
31
+ indices = list(layer_indices)
32
+ if not indices or indices != sorted(set(indices)):
33
+ raise ValueError("bert_layers must be a nonempty, increasing list of distinct indices")
34
+ if indices[0] < 0 or indices[-1] >= len(bert.layers):
35
+ raise ValueError("bert_layers contains an out-of-range index")
36
+ self.llm, self.bert = llm, bert
37
+ self.layer_indices = indices
38
+ self.config = bert.config
39
+ self.bert.layers = nn.ModuleList([bert.layers[i] for i in indices])
40
+ self.bert.config.num_hidden_layers = len(indices)
41
+ # Preserve each retained block's original attention type and rotary settings.
42
+ self.bert.config.layer_types = [layer.attention_type for layer in self.bert.layers]
43
+ self.bert.embeddings = ContextualEmbeddings()
44
+ width = llm.config.hidden_size
45
+ self.bridge = nn.Sequential(nn.LayerNorm(width), nn.Linear(width, self.config.hidden_size))
46
+ if width == self.config.hidden_size:
47
+ nn.init.eye_(self.bridge[1].weight)
48
+ nn.init.zeros_(self.bridge[1].bias)
49
+ self.llm.requires_grad_(False).eval()
50
+
51
+ def train(self, mode=True):
52
+ super().train(mode)
53
+ self.llm.eval()
54
+ return self
55
+
56
+ def forward(self, input_ids, attention_mask, **kwargs):
57
+ with torch.no_grad():
58
+ h = self.llm(
59
+ input_ids=input_ids,
60
+ attention_mask=attention_mask,
61
+ use_cache=False,
62
+ output_hidden_states=False,
63
+ return_dict=True,
64
+ ).last_hidden_state
65
+ h = self.bridge(h)
66
+ return self.bert(inputs_embeds=h, attention_mask=attention_mask, return_dict=True)
67
+
68
+
69
+ def build_item(tokenizer, state_ids, question, max_len=1024, min_state_tokens=32):
70
+ """Use the LM tokenizer throughout; gather the final token of each whole option.
71
+
72
+ State comes first so causal option features can attend to it. Only the state
73
+ may be truncated. Reject an overlong rubric instead of silently deleting options.
74
+ """
75
+ from laya.common import QTYPES, render_options
76
+
77
+ keys = labels(question)
78
+ q = {"t": question["type"], "ins": question["instructions"], "crit": question.get("criteria")}
79
+ if "labels" in question:
80
+ q["labels"] = question["labels"]
81
+
82
+ def encode(text):
83
+ return tokenizer.encode(text, add_special_tokens=False)
84
+
85
+ prefix = encode("State:\n")
86
+ heading = encode(f"\nQuestion ({q['t']}): {q['ins']}\nOptions:")
87
+ options = [encode("\nOption: " + text) for text in render_options(q)]
88
+ if len(options) != len(keys) or any(not option for option in options):
89
+ raise ValueError("every option must have a token span")
90
+ room = max_len - len(prefix) - len(heading) - sum(map(len, options))
91
+ if room < min(len(state_ids), min_state_tokens):
92
+ raise ValueError("question and full options exceed the token budget; increase max_len")
93
+ ids = prefix + state_ids[:room] + heading
94
+ markers = []
95
+ for option in options:
96
+ ids.extend(option)
97
+ markers.append(len(ids) - 1)
98
+ return {
99
+ "ids": ids,
100
+ "markers": markers,
101
+ "qtype": QTYPES[question["type"]],
102
+ "keys": keys,
103
+ "state_tokens_original": len(state_ids),
104
+ "state_tokens_used": min(len(state_ids), room),
105
+ }
106
+
107
+
108
+ def collate(items, pad_token_id, device="cpu"):
109
+ if not items:
110
+ raise ValueError("cannot collate an empty decision batch")
111
+ batch, length, options = (
112
+ len(items),
113
+ max(len(i["ids"]) for i in items),
114
+ max(len(i["markers"]) for i in items),
115
+ )
116
+ ids = torch.full((batch, length), pad_token_id, dtype=torch.long, device=device)
117
+ attention = torch.zeros_like(ids)
118
+ positions = torch.zeros((batch, options), dtype=torch.long, device=device)
119
+ marker_mask = torch.zeros_like(positions, dtype=torch.bool)
120
+ for row, item in enumerate(items):
121
+ n, k = len(item["ids"]), len(item["markers"])
122
+ ids[row, :n] = torch.tensor(item["ids"], device=device)
123
+ attention[row, :n] = 1
124
+ positions[row, :k] = torch.tensor(item["markers"], device=device)
125
+ marker_mask[row, :k] = True
126
+ return {
127
+ "input_ids": ids,
128
+ "attention_mask": attention,
129
+ "marker_pos": positions,
130
+ "marker_mask": marker_mask,
131
+ "qtype": torch.tensor([i["qtype"] for i in items], device=device),
132
+ }
133
+
134
+
135
+ class Hybrid:
136
+ """Harness adapter; local source checkpoints remain unchanged."""
137
+
138
+ def __init__(self, options):
139
+ import laya
140
+ import transformers
141
+ from transformers import AutoModel, AutoTokenizer
142
+
143
+ self.options = copy.deepcopy(options)
144
+ torch.manual_seed(options.get("seed", 42))
145
+ self.device = torch.device(options.get("device", "cpu"))
146
+ if self.device.type not in ("cpu", "cuda"):
147
+ raise ValueError("the prototype supports CPU and CUDA")
148
+ self.dtype = torch.bfloat16 if self.device.type == "cuda" else torch.float32
149
+ llm_path = Path(options["llm_model"]).resolve(strict=True)
150
+ laya_path = Path(options["laya_model"]).resolve(strict=True)
151
+ self.tokenizer = AutoTokenizer.from_pretrained(llm_path, local_files_only=True)
152
+ self.pad_id = self.tokenizer.pad_token_id
153
+ if self.pad_id is None:
154
+ self.pad_id = self.tokenizer.eos_token_id
155
+ if self.pad_id is None:
156
+ raise ValueError("the small LM needs a pad or EOS token")
157
+ llm = AutoModel.from_pretrained(
158
+ llm_path, local_files_only=True, dtype=self.dtype, attn_implementation="sdpa"
159
+ )
160
+ if sum(p.numel() for p in llm.parameters()) >= 1_000_000_000:
161
+ raise ValueError("this prototype requires a small LM with fewer than 1B parameters")
162
+ agent = laya.load(str(laya_path), device="cpu")
163
+ self.model = agent.model
164
+ indices = options.get("bert_layers", list(range(28)))
165
+ self.model.encoder = SemanticEncoder(llm, self.model.encoder, indices)
166
+ self.training_scope = options.get("training_scope", "bridge")
167
+ self.model.requires_grad_(False)
168
+ if self.training_scope == "bridge":
169
+ self.model.encoder.bridge.requires_grad_(True)
170
+ elif self.training_scope == "downstream":
171
+ for name, parameter in self.model.named_parameters():
172
+ parameter.requires_grad_(not name.startswith(("encoder.llm.", "act_head.")))
173
+ else:
174
+ raise ValueError("training_scope must be bridge or downstream")
175
+ self.max_len = int(options.get("max_len", 1024))
176
+ self.min_state_tokens = int(options.get("min_state_tokens", 32))
177
+ self.batch_size = int(options.get("batch_size", 16))
178
+ if self.batch_size < 1 or self.max_len < 1 or self.min_state_tokens < 1:
179
+ raise ValueError("batch_size, max_len, and min_state_tokens must be positive")
180
+ if self.max_len > min(
181
+ llm.config.max_position_embeddings, self.model.encoder.config.max_position_embeddings
182
+ ):
183
+ raise ValueError("max_len exceeds a backbone's positional limit")
184
+ self.identity = {
185
+ "format_version": 1,
186
+ "llm_model": str(llm_path),
187
+ "laya_model": str(laya_path),
188
+ "bert_layers": list(indices),
189
+ "max_len": self.max_len,
190
+ "min_state_tokens": self.min_state_tokens,
191
+ "prompt_format": "state-question-whole-options-v1",
192
+ "training_scope": self.training_scope,
193
+ }
194
+ self.metadata = {
195
+ **self.identity,
196
+ "training_status": "untrained hybrid connection",
197
+ "device": str(self.device),
198
+ "gpu_name": torch.cuda.get_device_name(self.device)
199
+ if self.device.type == "cuda"
200
+ else None,
201
+ "dtype": str(self.dtype),
202
+ "torch_version": torch.__version__,
203
+ "transformers_version": transformers.__version__,
204
+ "batch_size": self.batch_size,
205
+ "llm_layers": llm.config.num_hidden_layers,
206
+ "llm_width": llm.config.hidden_size,
207
+ "bert_width": self.model.encoder.config.hidden_size,
208
+ "head_layers": len(self.model.head.layers) if self.model.head else 0,
209
+ "parameters": sum(p.numel() for p in self.model.parameters()),
210
+ "llm_parameters": sum(p.numel() for p in llm.parameters()),
211
+ "trainable_parameters": sum(
212
+ p.numel() for p in self.model.parameters() if p.requires_grad
213
+ ),
214
+ "temperature": 1.0,
215
+ "use_cache": False,
216
+ }
217
+ if options.get("checkpoint"):
218
+ self.load_checkpoint(options["checkpoint"])
219
+ self.model.to(self.device).eval()
220
+
221
+ def training_mode(self):
222
+ # Frozen modules remain deterministic, but autograd still traverses BERT
223
+ # and the scorer to train the connection below them.
224
+ self.model.train(self.training_scope == "downstream")
225
+
226
+ def checkpoint_tensor_names(self):
227
+ if self.training_scope == "bridge":
228
+ return {k for k in self.model.state_dict() if k.startswith("encoder.bridge.")}
229
+ return {k for k in self.model.state_dict() if not k.startswith("encoder.llm.")}
230
+
231
+ def autocast(self):
232
+ return (
233
+ torch.autocast("cuda", dtype=self.dtype)
234
+ if self.device.type == "cuda"
235
+ else nullcontext()
236
+ )
237
+
238
+ def prepare(self, state, questions):
239
+ from laya.common import serialize_state
240
+
241
+ if not questions:
242
+ raise ValueError("at least one question is required")
243
+ state_ids = self.tokenizer.encode(serialize_state(state), add_special_tokens=False)
244
+ return [
245
+ build_item(self.tokenizer, state_ids, q, self.max_len, self.min_state_tokens)
246
+ for q in questions.values()
247
+ ]
248
+
249
+ @torch.inference_mode()
250
+ def predict(self, state, questions):
251
+ self.model.eval()
252
+ items = self.prepare(state, questions)
253
+ probabilities = []
254
+ for start in range(0, len(items), self.batch_size):
255
+ chunk = items[start : start + self.batch_size]
256
+ with self.autocast():
257
+ logits, _ = self.model(**collate(chunk, self.pad_id, self.device))
258
+ p = logits.softmax(-1).cpu().tolist()
259
+ probabilities.extend([row[: len(item["keys"])] for row, item in zip(p, chunk)])
260
+ answers = {}
261
+ for (qid, q), item, p in zip(questions.items(), items, probabilities):
262
+ keys = item["keys"]
263
+ answer = {"type": q["type"], "probabilities": dict(zip(keys, p)), "confidence": max(p)}
264
+ if q["type"] == "choice":
265
+ answer["choice"] = keys[max(range(len(p)), key=p.__getitem__)]
266
+ elif q["type"] == "noul":
267
+ answer["noul"] = p[1]
268
+ else:
269
+ answer["score"] = sum(i * value for i, value in enumerate(p))
270
+ answers[qid] = answer
271
+ return {
272
+ "model": "experimental-lm-laya-hybrid",
273
+ "answers": answers,
274
+ "usage": {
275
+ "llm_forward_calls": (len(items) + self.batch_size - 1) // self.batch_size,
276
+ "generated_tokens": 0,
277
+ "input_tokens": sum(len(i["ids"]) for i in items),
278
+ "truncated_state_questions": sum(
279
+ i["state_tokens_used"] < i["state_tokens_original"] for i in items
280
+ ),
281
+ },
282
+ }
283
+
284
+ def synchronize(self):
285
+ if self.device.type == "cuda":
286
+ torch.cuda.synchronize(self.device)
287
+
288
+ def close(self):
289
+ if hasattr(self, "model"):
290
+ del self.model
291
+ if self.device.type == "cuda":
292
+ torch.cuda.empty_cache()
293
+
294
+ def save_checkpoint(self, directory, training):
295
+ from safetensors.torch import save_file
296
+ from .schema import write_json
297
+
298
+ directory = Path(directory)
299
+ directory.mkdir(parents=True, exist_ok=False)
300
+ # Frozen source weights are referenced; only the selected training scope is saved.
301
+ names = self.checkpoint_tensor_names()
302
+ state = {
303
+ k: v.detach().cpu().contiguous()
304
+ for k, v in self.model.state_dict().items()
305
+ if k in names
306
+ }
307
+ weights = directory / "adapter.safetensors"
308
+ save_file(state, weights)
309
+ write_json(
310
+ directory / "manifest.json",
311
+ {
312
+ "identity": self.identity,
313
+ "training": training,
314
+ "weights_sha256": hashlib.sha256(weights.read_bytes()).hexdigest(),
315
+ },
316
+ )
317
+
318
+ def load_checkpoint(self, directory):
319
+ from safetensors.torch import load_file
320
+
321
+ directory = Path(directory)
322
+ manifest = json.loads((directory / "manifest.json").read_text())
323
+ if manifest["identity"] != self.identity:
324
+ raise ValueError("hybrid checkpoint architecture or tokenizer provenance differs")
325
+ weights = directory / "adapter.safetensors"
326
+ actual_hash = hashlib.sha256(weights.read_bytes()).hexdigest()
327
+ if actual_hash != manifest["weights_sha256"]:
328
+ raise ValueError("hybrid checkpoint hash mismatch")
329
+ state = load_file(weights)
330
+ expected = self.checkpoint_tensor_names()
331
+ if set(state) != expected:
332
+ raise ValueError(
333
+ "hybrid checkpoint must contain exactly the tensors for its training scope"
334
+ )
335
+ self.model.load_state_dict(state, strict=False)
336
+ self.metadata.update(
337
+ {"training_status": manifest["training"], "checkpoint_sha256": actual_hash}
338
+ )
339
+
340
+
341
+ def create(options):
342
+ return Hybrid(options)
ariadne_bench/hybrid_control.py ADDED
@@ -0,0 +1,169 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Matched trainable-interface control using Laya's original token embeddings."""
2
+
3
+ import copy
4
+ from pathlib import Path
5
+
6
+ import torch
7
+ from torch import nn
8
+ from transformers.modeling_outputs import BaseModelOutput
9
+
10
+ from .hybrid import Hybrid, SemanticEncoder
11
+ from .interfaces import make_interface
12
+ from .schema import labels
13
+
14
+
15
+ class FrozenTokenEncoder(nn.Module):
16
+ def __init__(self, embeddings, config):
17
+ super().__init__()
18
+ self.embeddings = embeddings
19
+ self.config = copy.deepcopy(config)
20
+ self.config.num_hidden_layers = 0
21
+
22
+ def forward(self, input_ids, **kwargs):
23
+ return BaseModelOutput(last_hidden_state=self.embeddings(input_ids=input_ids))
24
+
25
+
26
+ class LayaControl(Hybrid):
27
+ def __init__(self, options):
28
+ import laya
29
+ import transformers
30
+
31
+ self.options = copy.deepcopy(options)
32
+ torch.manual_seed(options.get("seed", 42))
33
+ self.device = torch.device(options.get("device", "cpu"))
34
+ if self.device.type not in ("cpu", "cuda"):
35
+ raise ValueError("the prototype supports CPU and CUDA")
36
+ self.dtype = torch.bfloat16 if self.device.type == "cuda" else torch.float32
37
+ self.training_scope = "bridge"
38
+ laya_path = Path(options["laya_model"]).resolve(strict=True)
39
+ self.native_prediction = bool(options.get("native_prediction", False))
40
+ agent = laya.load(
41
+ str(laya_path), device=str(self.device) if self.native_prediction else "cpu"
42
+ )
43
+ if self.native_prediction:
44
+ if agent.device != self.device:
45
+ raise RuntimeError("native Laya loaded on an unexpected device")
46
+ self.agent = agent
47
+ self.dtype = agent.dtype
48
+ self.tokenizer, self.model = agent.tok, agent.model
49
+ self.pad_id = self.tokenizer.pad_token_id
50
+ bert = self.model.encoder
51
+ tokens = FrozenTokenEncoder(bert.embeddings, bert.config)
52
+ indices = options.get("bert_layers", list(range(len(bert.layers))))
53
+ self.model.encoder = SemanticEncoder(tokens, bert, indices)
54
+ if "interface" in options:
55
+ specification = options["interface"]
56
+ if specification.get("kind", "").startswith("exact_linear"):
57
+ torch.set_float32_matmul_precision("highest")
58
+ self.model.encoder.bridge = make_interface(bert.config.hidden_size, specification)
59
+ self.model.requires_grad_(False)
60
+ self.model.encoder.bridge.requires_grad_(True)
61
+ self.max_len = int(options.get("max_len", 512))
62
+ self.head_max_len = agent.cfg.get("head_max_len", 192)
63
+ self.batch_size = int(options.get("batch_size", 16))
64
+ if self.max_len < 1 or self.batch_size < 1:
65
+ raise ValueError("max_len and batch_size must be positive")
66
+ self.identity = {
67
+ "format_version": 1,
68
+ "feature_source": "original Laya token embeddings",
69
+ "laya_model": str(laya_path),
70
+ "bert_layers": list(indices),
71
+ "max_len": self.max_len,
72
+ "head_max_len": self.head_max_len,
73
+ "prompt_format": "original-laya-v1",
74
+ "training_scope": "bridge",
75
+ }
76
+ # Preserve the identity schema of existing saved LayerNorm+Linear checkpoints.
77
+ if "interface" in options:
78
+ self.identity["interface"] = copy.deepcopy(options["interface"])
79
+ if self.native_prediction:
80
+ self.identity["native_prediction"] = True
81
+ self.metadata = {
82
+ **self.identity,
83
+ "training_status": "untrained control interface",
84
+ "device": str(self.device),
85
+ "gpu_name": torch.cuda.get_device_name(self.device)
86
+ if self.device.type == "cuda"
87
+ else None,
88
+ "dtype": str(self.dtype),
89
+ "torch_version": torch.__version__,
90
+ "transformers_version": transformers.__version__,
91
+ "batch_size": self.batch_size,
92
+ "parameters": sum(p.numel() for p in self.model.parameters()),
93
+ "trainable_parameters": sum(
94
+ p.numel() for p in self.model.parameters() if p.requires_grad
95
+ ),
96
+ "temperature": 1.0,
97
+ }
98
+ if "interface" in options:
99
+ self.metadata["float32_matmul_precision"] = torch.get_float32_matmul_precision()
100
+ if self.native_prediction:
101
+ self.metadata.update(
102
+ {
103
+ "prediction_path": "native Laya SDK",
104
+ "temperature": list(agent.temperature),
105
+ "temperature_by_options": dict(agent.temperature_by_options),
106
+ "laya_version": laya.__version__,
107
+ }
108
+ )
109
+ if options.get("checkpoint"):
110
+ self.load_checkpoint(options["checkpoint"])
111
+ self.model.to(self.device).eval()
112
+
113
+ def prepare(self, state, questions):
114
+ from laya.common import QTYPES, build_sequence, serialize_state
115
+
116
+ if not questions:
117
+ raise ValueError("at least one question is required")
118
+ state_ids = self.tokenizer.encode(
119
+ serialize_state(state).replace(self.tokenizer.mask_token, " "), add_special_tokens=False
120
+ )
121
+ result = []
122
+ for question in questions.values():
123
+ q = {
124
+ "t": question["type"],
125
+ "ins": question["instructions"],
126
+ "crit": question.get("criteria"),
127
+ }
128
+ if "labels" in question:
129
+ q["labels"] = question["labels"]
130
+ ids, markers = build_sequence(
131
+ self.tokenizer, state, q, self.max_len, self.head_max_len, state_ids=state_ids
132
+ )
133
+ keys = labels(question)
134
+ if len(keys) != len(markers):
135
+ raise ValueError("the Laya token budget removed an option")
136
+ # build_sequence appends state after its final head SEP and leaves a final SEP.
137
+ last_head_sep = next(
138
+ i for i in range(markers[-1] + 1, len(ids)) if ids[i] == self.tokenizer.sep_token_id
139
+ )
140
+ state_used = min(len(state_ids), max(0, len(ids) - last_head_sep - 2))
141
+ result.append(
142
+ {
143
+ "ids": ids,
144
+ "markers": markers,
145
+ "qtype": QTYPES[q["t"]],
146
+ "keys": keys,
147
+ "state_tokens_original": len(state_ids),
148
+ "state_tokens_used": state_used,
149
+ }
150
+ )
151
+ return result
152
+
153
+ def predict(self, state, questions):
154
+ if self.native_prediction:
155
+ self.model.eval()
156
+ return self.agent.predict(state, questions, max_len=self.max_len)
157
+ response = super().predict(state, questions)
158
+ response["model"] = "laya-interface-control"
159
+ response["usage"]["llm_forward_calls"] = 0
160
+ return response
161
+
162
+ def close(self):
163
+ if hasattr(self, "agent"):
164
+ del self.agent
165
+ super().close()
166
+
167
+
168
+ def create(options):
169
+ return LayaControl(options)
ariadne_bench/interfaces.py ADDED
@@ -0,0 +1,121 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Input transformations with explicit initialization and low-rank factors."""
2
+
3
+ import math
4
+
5
+ import torch
6
+ from torch import nn
7
+ from torch.nn import functional as F
8
+
9
+
10
+ class IdentityAffine(nn.Module):
11
+ """Affine map initialized as identity, preserving input precision under AMP.
12
+
13
+ Autocasting an identity GEMM to BF16 would round FP32 embeddings before
14
+ BERT sees them. Keep this small interface in FP32 and return the input dtype.
15
+ CUDA callers must use highest float32 matmul precision (no TF32 rounding).
16
+ """
17
+
18
+ def __init__(self, width, residual=False):
19
+ super().__init__()
20
+ self.residual = residual
21
+ self.proj = nn.Linear(width, width)
22
+ if residual:
23
+ nn.init.zeros_(self.proj.weight)
24
+ else:
25
+ nn.init.eye_(self.proj.weight)
26
+ nn.init.zeros_(self.proj.bias)
27
+
28
+ def forward(self, x):
29
+ with torch.autocast(device_type=x.device.type, enabled=False):
30
+ value = F.linear(x.float(), self.proj.weight.float(), self.proj.bias.float())
31
+ if self.residual:
32
+ value = x.float() + value
33
+ return value.to(x.dtype)
34
+
35
+
36
+ class LowRankLinear(nn.Module):
37
+ def __init__(self, width, rank, residual=False):
38
+ super().__init__()
39
+ if not 1 <= rank <= width:
40
+ raise ValueError("rank must be between one and the hidden width")
41
+ self.rank, self.residual = rank, residual
42
+ self.a = nn.Parameter(torch.empty(rank, width))
43
+ self.b = nn.Parameter(torch.empty(width, rank))
44
+ self.bias = nn.Parameter(torch.zeros(width))
45
+ if residual:
46
+ nn.init.kaiming_uniform_(self.a, a=math.sqrt(5))
47
+ nn.init.zeros_(self.b)
48
+ else:
49
+ # A pure rank-r map cannot start at identity. Start as an orthogonal
50
+ # rank-r projection instead, and record this in experiment metadata.
51
+ q, _ = torch.linalg.qr(torch.randn(width, rank), mode="reduced")
52
+ with torch.no_grad():
53
+ self.a.copy_(q.T)
54
+ self.b.copy_(q)
55
+
56
+ def forward(self, x):
57
+ value = F.linear(F.linear(x, self.a), self.b, self.bias)
58
+ return x + value if self.residual else value
59
+
60
+ def dense_weight(self):
61
+ weight = self.b @ self.a
62
+ if self.residual:
63
+ weight = weight + torch.eye(weight.shape[0], device=weight.device, dtype=weight.dtype)
64
+ return weight
65
+
66
+ @classmethod
67
+ def from_svd(cls, u, s, vh, bias, rank, residual=False):
68
+ layer = cls(u.shape[0], rank, residual=residual)
69
+ with torch.no_grad():
70
+ root = s[:rank].sqrt()
71
+ layer.a.copy_(root[:, None] * vh[:rank])
72
+ layer.b.copy_(u[:, :rank] * root[None, :])
73
+ layer.bias.copy_(bias)
74
+ return layer
75
+
76
+
77
+ class Diagonal(nn.Module):
78
+ def __init__(self, width):
79
+ super().__init__()
80
+ self.scale = nn.Parameter(torch.ones(width))
81
+ self.bias = nn.Parameter(torch.zeros(width))
82
+
83
+ def forward(self, x):
84
+ return x * self.scale + self.bias
85
+
86
+
87
+ class ResidualInput(nn.Module):
88
+ def __init__(self, width):
89
+ super().__init__()
90
+ self.norm = nn.LayerNorm(width)
91
+ self.linear = nn.Linear(width, width)
92
+ nn.init.zeros_(self.linear.weight)
93
+ nn.init.zeros_(self.linear.bias)
94
+
95
+ def forward(self, x):
96
+ return x + self.linear(self.norm(x))
97
+
98
+
99
+ def make_interface(width, specification):
100
+ kind = specification.get("kind", "norm_linear")
101
+ if kind in ("exact_linear", "exact_linear_residual"):
102
+ return IdentityAffine(width, residual=kind == "exact_linear_residual")
103
+ if kind == "identity":
104
+ return nn.Identity()
105
+ if kind == "norm":
106
+ return nn.LayerNorm(width)
107
+ if kind == "diagonal":
108
+ return Diagonal(width)
109
+ if kind == "residual":
110
+ return ResidualInput(width)
111
+ if kind in ("low_rank", "low_rank_delta"):
112
+ return nn.Sequential(
113
+ nn.LayerNorm(width),
114
+ LowRankLinear(width, specification["rank"], residual=kind == "low_rank_delta"),
115
+ )
116
+ if kind in ("linear", "norm_linear"):
117
+ linear = nn.Linear(width, width)
118
+ nn.init.eye_(linear.weight)
119
+ nn.init.zeros_(linear.bias)
120
+ return linear if kind == "linear" else nn.Sequential(nn.LayerNorm(width), linear)
121
+ raise ValueError(f"unknown interface kind: {kind}")
ariadne_bench/metrics.py ADDED
@@ -0,0 +1,229 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Explicit scoring conventions; no fitted temperatures or test-set thresholds."""
2
+
3
+ import math
4
+ from collections import defaultdict
5
+ from statistics import mean
6
+
7
+ from .schema import distribution, labels
8
+
9
+ METRIC_VERSION = "1:sum-brier:top-label-ece:right-closed:eps-1e-12"
10
+
11
+
12
+ def decision(case, qid, answer):
13
+ q, gold = case["questions"][qid], case["gold"][qid]
14
+ keys = labels(q)
15
+ p = [answer["probabilities"][k] for k in keys]
16
+ target = keys.index(gold["label"])
17
+ row = {
18
+ "case_id": case["id"],
19
+ "suite": case["suite"],
20
+ "question": qid,
21
+ "workflow": case.get("workflow", ""),
22
+ "language": case.get("language", ""),
23
+ "type": q["type"],
24
+ "gold": gold["label"],
25
+ "prediction": answer["label"],
26
+ "correct": float(answer["label"] == gold["label"]),
27
+ "confidence": answer["confidence"],
28
+ "brier_hard": sum((v - float(i == target)) ** 2 for i, v in enumerate(p)),
29
+ "nll_hard": -math.log(max(p[target], 1e-12)),
30
+ "zero_probability_gold": float(p[target] == 0),
31
+ "raw_probability_sum": answer["raw_probability_sum"],
32
+ }
33
+ if "probabilities" in gold:
34
+ g, _ = distribution(gold["probabilities"], keys)
35
+ row.update(
36
+ soft_accuracy=sum(a * b for a, b in zip(p, g)),
37
+ brier_soft=sum((a - b) ** 2 for a, b in zip(p, g)),
38
+ kl_gold_to_prediction=sum(
39
+ b * math.log(b / max(a, 1e-12)) for a, b in zip(p, g) if b > 0
40
+ ),
41
+ total_variation=0.5 * sum(abs(a - b) for a, b in zip(p, g)),
42
+ )
43
+ if q["type"] == "score":
44
+ score = sum(i * v for i, v in enumerate(p))
45
+ error = abs(score - gold.get("score", float(gold["label"])))
46
+ row.update(score_mae=error, within_one_level=float(error <= 1))
47
+ return row
48
+
49
+
50
+ def ece(rows, bins):
51
+ # [0, 1/B], (1/B, 2/B], ..., ((B-1)/B, 1]; includes confidence 0 and 1.
52
+ buckets = defaultdict(list)
53
+ for row in rows:
54
+ index = max(0, min(bins - 1, math.ceil(row["confidence"] * bins) - 1))
55
+ buckets[index].append(row)
56
+ return sum(
57
+ len(rs) * abs(mean(r["confidence"] for r in rs) - mean(r["correct"] for r in rs))
58
+ for rs in buckets.values()
59
+ ) / len(rows)
60
+
61
+
62
+ def summarize(rows, attempted, bins=15):
63
+ result = {
64
+ "attempted": attempted,
65
+ "valid": len(rows),
66
+ "failed": attempted - len(rows),
67
+ "coverage": len(rows) / attempted if attempted else None,
68
+ "accuracy_all": sum(r["correct"] for r in rows) / attempted if attempted else None,
69
+ }
70
+ if not rows:
71
+ result.update(accuracy_valid=None, ece_top_label=None, macro_f1=None)
72
+ return result
73
+ result.update(
74
+ accuracy_valid=mean(r["correct"] for r in rows),
75
+ ece_top_label=ece(rows, bins),
76
+ mean_confidence=mean(r["confidence"] for r in rows),
77
+ )
78
+ for metric in (
79
+ "brier_hard",
80
+ "nll_hard",
81
+ "zero_probability_gold",
82
+ "soft_accuracy",
83
+ "brier_soft",
84
+ "kl_gold_to_prediction",
85
+ "total_variation",
86
+ "score_mae",
87
+ "within_one_level",
88
+ ):
89
+ values = [r[metric] for r in rows if metric in r]
90
+ if values:
91
+ result[metric] = mean(values)
92
+ result[metric + "_n"] = len(values)
93
+ # Semantic label namespaces prevent conflating e.g. ordinal 0 across unrelated tasks.
94
+ confusion = defaultdict(lambda: [0, 0, 0])
95
+ for r in rows:
96
+ ns = (r["suite"], r["workflow"], r["question"])
97
+ gold, pred = ns + (r["gold"],), ns + (r["prediction"],)
98
+ if gold == pred:
99
+ confusion[gold][0] += 1
100
+ else:
101
+ confusion[pred][1] += 1
102
+ confusion[gold][2] += 1
103
+ result["macro_f1"] = mean(2 * tp / (2 * tp + fp + fn) for tp, fp, fn in confusion.values())
104
+ # Include complete confidence ties; these are descriptive curves, not chosen thresholds.
105
+ ranked = sorted(rows, key=lambda r: -r["confidence"])
106
+ curve, errors = [], 0
107
+ for i, row in enumerate(ranked, 1):
108
+ errors += 1 - row["correct"]
109
+ if i == len(ranked) or ranked[i]["confidence"] != row["confidence"]:
110
+ curve.append(
111
+ {
112
+ "threshold": row["confidence"],
113
+ "coverage": i / attempted,
114
+ "risk": errors / i,
115
+ "accepted": i,
116
+ }
117
+ )
118
+ result["risk_coverage"] = curve
119
+ return result
120
+
121
+
122
+ def percentile(values, q):
123
+ if not values:
124
+ return None
125
+ values = sorted(values)
126
+ pos = (len(values) - 1) * q
127
+ lo, hi = math.floor(pos), math.ceil(pos)
128
+ return values[lo] + (values[hi] - values[lo]) * (pos - lo)
129
+
130
+
131
+ def timing(records):
132
+ values = [r["latency_ms"] for r in records if r.get("latency_ms") is not None]
133
+ seconds = sum(values) / 1000
134
+ return {
135
+ "requests": len(records),
136
+ "timed_requests": len(values),
137
+ "p50_ms": percentile(values, 0.5),
138
+ "p95_ms": percentile(values, 0.95),
139
+ "p99_ms": percentile(values, 0.99),
140
+ "total_request_seconds": seconds,
141
+ "attempted_decisions_per_second": sum(r["question_count"] for r in records) / seconds
142
+ if seconds and len(values) == len(records)
143
+ else None,
144
+ }
145
+
146
+
147
+ def report(cases, records, bins=15, tolerance=0.02):
148
+ from .schema import normalize_answer
149
+
150
+ by_id = {r["case_id"]: r for r in records}
151
+ if len(by_id) != len(records) or set(by_id) != {c["id"] for c in cases}:
152
+ raise ValueError("records must contain each case exactly once")
153
+ rows, errors = [], []
154
+ counts = {field: defaultdict(int) for field in ("suite", "workflow", "language", "type")}
155
+ for case in cases:
156
+ record = by_id[case["id"]]
157
+ response = record.get("response") or {}
158
+ answers = response.get("answers", {}) if isinstance(response, dict) else {}
159
+ if not isinstance(answers, dict):
160
+ answers = {}
161
+ extra_keys = set(answers) - set(case["questions"])
162
+ for qid, q in case["questions"].items():
163
+ for field in counts:
164
+ counts[field][q["type"] if field == "type" else case.get(field, "")] += 1
165
+ try:
166
+ if record.get("error"):
167
+ raise ValueError(record["error"])
168
+ if extra_keys:
169
+ raise ValueError("response contains unexpected question keys")
170
+ a = normalize_answer(q, answers.get(qid), tolerance)
171
+ rows.append(decision(case, qid, a))
172
+ except (ValueError, TypeError, KeyError) as exc:
173
+ errors.append({"case_id": case["id"], "question": qid, "error": str(exc)})
174
+ total = sum(len(c["questions"]) for c in cases)
175
+ result = {
176
+ "metric_version": METRIC_VERSION,
177
+ "ece_bins": bins,
178
+ "probability_sum_tolerance": tolerance,
179
+ "overall": summarize(rows, total, bins),
180
+ "latency": timing(records),
181
+ "errors": errors,
182
+ }
183
+ for field, sizes in counts.items():
184
+ result["by_" + field] = {
185
+ key: summarize([r for r in rows if r[field] == key], n, bins)
186
+ for key, n in sizes.items()
187
+ if key
188
+ }
189
+ result["latency_by_question_count"] = {
190
+ str(n): timing([r for r in records if r["question_count"] == n])
191
+ for n in sorted({r["question_count"] for r in records})
192
+ }
193
+ usages = [
194
+ r.get("response", {}).get("usage", {})
195
+ for r in records
196
+ if isinstance(r.get("response"), dict)
197
+ ]
198
+ usages = [
199
+ u
200
+ for u in usages
201
+ if isinstance(u, dict)
202
+ and all(
203
+ isinstance(u.get(k, 0), int) and not isinstance(u.get(k, 0), bool) and u.get(k, 0) >= 0
204
+ for k in ("input_tokens", "output_tokens")
205
+ )
206
+ ]
207
+ result["usage"] = {
208
+ "requests_with_usage": sum(bool(u) for u in usages),
209
+ "input_tokens_known": sum(u.get("input_tokens", 0) or 0 for u in usages),
210
+ "output_tokens_known": sum(u.get("output_tokens", 0) or 0 for u in usages),
211
+ }
212
+ paired = defaultdict(dict)
213
+ case_map = {c["id"]: c for c in cases}
214
+ for row in rows:
215
+ case = case_map[row["case_id"]]
216
+ base = case.get("base_id", case["id"])
217
+ paired[(base, row["question"])][case.get("permutation", 0)] = row["prediction"]
218
+ flips = [
219
+ pred != variants[0]
220
+ for variants in paired.values()
221
+ if 0 in variants
222
+ for variant, pred in variants.items()
223
+ if variant != 0
224
+ ]
225
+ result["option_order"] = {
226
+ "valid_pairs": len(flips),
227
+ "flip_rate": mean(flips) if flips else None,
228
+ }
229
+ return result
ariadne_bench/portable.py ADDED
@@ -0,0 +1,88 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Load a compact affine adapter with pinned, independently verified base files."""
2
+
3
+ import hashlib
4
+ import json
5
+ from pathlib import Path
6
+
7
+
8
+ def sha256(path):
9
+ return hashlib.sha256(Path(path).read_bytes()).hexdigest()
10
+
11
+
12
+ def load_release(
13
+ directory, device="cpu", seed=None, base_path=None, local_files_only=False, deterministic=True
14
+ ):
15
+ from huggingface_hub import snapshot_download
16
+ from safetensors.torch import load_file
17
+ from .frozen_input_interface import FrozenInputInterface
18
+
19
+ root = Path(directory).resolve()
20
+ config = json.loads((root / "adapter_config.json").read_text())
21
+ if config["format_version"] != 1 or config["interface"] != "exact_linear":
22
+ raise ValueError("unsupported adapter format")
23
+ selected = config["candidate_seed"] if seed is None else int(seed)
24
+ entry = config["seeds"][str(selected)]
25
+ weights = (root / entry["weights_file"]).resolve()
26
+ if not weights.is_relative_to(root):
27
+ raise ValueError("adapter weights must be inside the release directory")
28
+ if sha256(weights) != entry["weights_sha256"]:
29
+ raise ValueError("adapter checkpoint checksum mismatch")
30
+ source = (
31
+ Path(base_path)
32
+ if base_path
33
+ else Path(
34
+ snapshot_download(
35
+ config["base_model"],
36
+ revision=config["base_revision"],
37
+ local_files_only=local_files_only,
38
+ allow_patterns=[
39
+ "model.safetensors",
40
+ "rl_agent_config.json",
41
+ "encoder/*",
42
+ "tokenizer/*",
43
+ ],
44
+ )
45
+ )
46
+ )
47
+ for name, expected in config["base_config_sha256"].items():
48
+ if sha256(source / name) != expected:
49
+ raise ValueError(f"base tokenizer/configuration mismatch: {name}")
50
+ adapter = FrozenInputInterface(
51
+ {
52
+ "laya_model": str(source.resolve()),
53
+ "device": device,
54
+ "max_len": config["max_len"],
55
+ "head_max_len": config["head_max_len"],
56
+ "seed": selected,
57
+ "training_scope": "bridge",
58
+ "add_interface": True,
59
+ "deterministic": deterministic,
60
+ }
61
+ )
62
+ try:
63
+ if adapter.initial_frozen_sha256 != config["frozen_state_sha256"]:
64
+ raise ValueError("base model tensors do not match the pinned training source")
65
+ state = load_file(str(weights))
66
+ if set(state) != adapter.checkpoint_tensor_names():
67
+ raise ValueError("checkpoint must contain exactly the affine weight and bias")
68
+ adapter.model.load_state_dict(state, strict=False)
69
+ adapter.model.requires_grad_(False)
70
+ adapter.model.eval()
71
+ adapter.metadata.update(
72
+ trainable_parameters=0,
73
+ checkpoint_sha256=entry["weights_sha256"],
74
+ portable_adapter={
75
+ "base_model": config["base_model"],
76
+ "base_revision": config["base_revision"],
77
+ "seed": selected,
78
+ "selected_epoch": entry["selected_epoch"],
79
+ },
80
+ )
81
+ except Exception:
82
+ adapter.close()
83
+ raise
84
+ return adapter
85
+
86
+
87
+ def create(options):
88
+ return load_release(**options)
ariadne_bench/reproducibility.py ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Opt-in deterministic execution for a fixed GPU, runtime, and data order."""
2
+
3
+ import os
4
+ import random
5
+
6
+ import numpy as np
7
+ import torch
8
+
9
+
10
+ def configure_determinism(seed):
11
+ workspace = os.environ.get("CUBLAS_WORKSPACE_CONFIG")
12
+ if workspace is None:
13
+ if torch.cuda.is_initialized():
14
+ raise RuntimeError("set CUBLAS_WORKSPACE_CONFIG=:4096:8 before CUDA initialization")
15
+ os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8"
16
+ elif workspace not in {":4096:8", ":16:8"}:
17
+ raise ValueError("deterministic execution requires a supported cuBLAS workspace config")
18
+ random.seed(seed)
19
+ np.random.seed(seed)
20
+ torch.manual_seed(seed)
21
+ torch.use_deterministic_algorithms(True, warn_only=False)
22
+ torch.backends.cudnn.benchmark = False
23
+ torch.backends.cudnn.deterministic = True
24
+ torch.backends.cudnn.allow_tf32 = False
25
+ torch.set_float32_matmul_precision("highest")
26
+ # Avoid a backend whose deterministic backward support varies by cuDNN version.
27
+ torch.backends.cuda.enable_cudnn_sdp(False)
28
+ return settings()
29
+
30
+
31
+ def settings():
32
+ return {
33
+ "deterministic_algorithms": torch.are_deterministic_algorithms_enabled(),
34
+ "warn_only": torch.is_deterministic_algorithms_warn_only_enabled(),
35
+ "cublas_workspace_config": os.environ.get("CUBLAS_WORKSPACE_CONFIG"),
36
+ "python_hash_seed": os.environ.get("PYTHONHASHSEED"),
37
+ "cudnn_benchmark": torch.backends.cudnn.benchmark,
38
+ "cudnn_deterministic": torch.backends.cudnn.deterministic,
39
+ "cudnn_allow_tf32": torch.backends.cudnn.allow_tf32,
40
+ "float32_matmul_precision": torch.get_float32_matmul_precision(),
41
+ "cudnn_sdp_enabled": torch.backends.cuda.cudnn_sdp_enabled(),
42
+ "flash_sdp_enabled": torch.backends.cuda.flash_sdp_enabled(),
43
+ "mem_efficient_sdp_enabled": torch.backends.cuda.mem_efficient_sdp_enabled(),
44
+ "math_sdp_enabled": torch.backends.cuda.math_sdp_enabled(),
45
+ "cuda_version": torch.version.cuda,
46
+ "cudnn_version": torch.backends.cudnn.version(),
47
+ "scope": "fixed hardware/runtime; cross-platform bitwise agreement is not promised",
48
+ }
ariadne_bench/runner.py ADDED
@@ -0,0 +1,257 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import copy
2
+ import importlib.metadata
3
+ import json
4
+ import platform
5
+ import time
6
+ from datetime import datetime, timezone
7
+ from pathlib import Path
8
+
9
+ from . import __version__
10
+ from .adapters import create_adapter
11
+ from .metrics import report
12
+ from .schema import digest, dumps, load_bundle, read_jsonl, write_json
13
+
14
+
15
+ def environment():
16
+ versions = {}
17
+ for name in ("laya", "torch", "transformers", "datasets", "huggingface-hub"):
18
+ try:
19
+ versions[name] = importlib.metadata.version(name)
20
+ except importlib.metadata.PackageNotFoundError:
21
+ pass
22
+ return {
23
+ "python": platform.python_version(),
24
+ "platform": platform.platform(),
25
+ "processor": platform.processor(),
26
+ "packages": versions,
27
+ "harness": __version__,
28
+ }
29
+
30
+
31
+ def check_config(config):
32
+ # Config files are copied into artifacts. Credentials must stay in the environment.
33
+ def visit(value):
34
+ if isinstance(value, dict):
35
+ for key, child in value.items():
36
+ if key.lower() in ("api_key", "token", "password", "authorization", "secret"):
37
+ raise ValueError(
38
+ "put credentials in environment variables, not the saved config"
39
+ )
40
+ visit(child)
41
+ elif isinstance(value, list):
42
+ for child in value:
43
+ visit(child)
44
+
45
+ visit(config)
46
+ if config.get("mode") not in ("generalist", "specialist", "reference"):
47
+ raise ValueError("config mode must be generalist, specialist or reference")
48
+ if not config.get("name"):
49
+ raise ValueError("config requires a display name")
50
+ if "endpoint" in config:
51
+ from urllib.parse import urlsplit
52
+
53
+ url = urlsplit(config["endpoint"])
54
+ if url.username or url.password or url.query or url.fragment:
55
+ raise ValueError("endpoint must not contain credentials, a query or a fragment")
56
+
57
+
58
+ def invoke(adapter, case):
59
+ start = time.perf_counter()
60
+ result = {"case_id": case["id"], "question_count": len(case["questions"])}
61
+ try:
62
+ if hasattr(adapter, "synchronize"):
63
+ adapter.synchronize()
64
+ start = time.perf_counter()
65
+ # Copy protects the frozen bundle from model code that mutates its arguments.
66
+ response = adapter.predict(copy.deepcopy(case["state"]), copy.deepcopy(case["questions"]))
67
+ if hasattr(adapter, "synchronize"):
68
+ adapter.synchronize()
69
+ result["latency_ms"] = (time.perf_counter() - start) * 1000
70
+ dumps(response)
71
+ result["response"] = response
72
+ except Exception as exc:
73
+ result["latency_ms"] = (time.perf_counter() - start) * 1000
74
+ # Arbitrary plugin errors can embed secret configuration. Keep their type only.
75
+ result["error"] = type(exc).__name__
76
+ if isinstance(exc, RuntimeError) and str(exc).startswith("HTTP "):
77
+ result["error"] = str(exc)
78
+ return result
79
+
80
+
81
+ def render_report(result):
82
+ meta = result["run"]
83
+ lines = [
84
+ f"# {meta['config']['name']}",
85
+ "",
86
+ f"Mode: **{meta['config']['mode']}**. Profile: `{meta['manifest']['profile']}`.",
87
+ "",
88
+ f"Cases SHA-256: `{meta['manifest']['sha256']}`.",
89
+ "",
90
+ "| Suite | Valid / attempted | Accuracy (all) | Brier (hard) | Brier (soft) | ECE |",
91
+ "|---|---:|---:|---:|---:|---:|",
92
+ ]
93
+ for suite, r in result["by_suite"].items():
94
+
95
+ def fmt(v):
96
+ return "—" if v is None else f"{v:.4f}"
97
+
98
+ lines.append(
99
+ f"| {suite} | {r['valid']} / {r['attempted']} | {fmt(r['accuracy_all'])} | "
100
+ f"{fmt(r.get('brier_hard'))} | {fmt(r.get('brier_soft'))} | {fmt(r.get('ece_top_label'))} |"
101
+ )
102
+ t = result["latency"]
103
+ lines.extend(
104
+ [
105
+ "",
106
+ f"Request p50: {t['p50_ms']} ms; p95: {t['p95_ms']} ms.",
107
+ "",
108
+ "Accuracy (all) counts invalid or missing decisions as incorrect. Proper scores and ECE use valid decisions; check coverage.",
109
+ "ECE uses the predicted option's probability, not the provider's confidence field. Brier sums over classes.",
110
+ "Latency includes the adapter call and local device synchronization; model loading and explicit warmups are separate.",
111
+ "Published reference numbers and local timing runs require matching protocols before comparison.",
112
+ "",
113
+ ]
114
+ )
115
+ return "\n".join(lines)
116
+
117
+
118
+ def run(bundle, config_path, output, bins=15, tolerance=0.02, warmup=0, resume=False):
119
+ cases, manifest = load_bundle(bundle)
120
+ config = json.loads(Path(config_path).read_text())
121
+ check_config(config)
122
+ directory = Path(output)
123
+ identity = digest(
124
+ {
125
+ "manifest": manifest,
126
+ "config": config,
127
+ "bins": bins,
128
+ "tolerance": tolerance,
129
+ "warmup": warmup,
130
+ "harness": __version__,
131
+ }
132
+ )
133
+ if resume:
134
+ saved = json.loads((directory / "run.json").read_text())
135
+ if saved["identity"] != identity:
136
+ raise ValueError("resume requires identical data, config and scoring settings")
137
+ records = (
138
+ read_jsonl(directory / "predictions.jsonl")
139
+ if (directory / "predictions.jsonl").exists()
140
+ else []
141
+ )
142
+ ids = [r["case_id"] for r in records]
143
+ if ids != [c["id"] for c in cases[: len(ids)]]:
144
+ raise ValueError("saved records are not the expected case prefix")
145
+ metadata = saved
146
+ else:
147
+ if directory.exists():
148
+ raise ValueError("run directory exists; choose a new path or --resume")
149
+ records = []
150
+ metadata = {
151
+ "identity": identity,
152
+ "created_utc": datetime.now(timezone.utc).isoformat(),
153
+ "config": config,
154
+ "manifest": manifest,
155
+ "environment": environment(),
156
+ "ece_bins": bins,
157
+ "probability_sum_tolerance": tolerance,
158
+ "warmup_per_question_count": warmup,
159
+ "sessions": [],
160
+ }
161
+ if len(records) < len(cases):
162
+ load_start = time.perf_counter()
163
+ adapter = create_adapter(config)
164
+ session = {
165
+ "model_load_seconds": time.perf_counter() - load_start,
166
+ "adapter_metadata": getattr(adapter, "metadata", {}),
167
+ "start_index": len(records),
168
+ "utc": datetime.now(timezone.utc).isoformat(),
169
+ }
170
+ metadata["sessions"].append(session)
171
+ directory.mkdir(parents=True, exist_ok=True)
172
+ write_json(directory / "run.json", metadata)
173
+ try:
174
+ shapes = {}
175
+ for case in cases:
176
+ shapes.setdefault(len(case["questions"]), case)
177
+ with (directory / "warmup.jsonl").open("a") as f:
178
+ for case in shapes.values():
179
+ for _ in range(warmup):
180
+ record = invoke(adapter, case)
181
+ f.write(dumps(record) + "\n")
182
+ f.flush()
183
+ if record.get("error"):
184
+ raise ValueError(f"warmup failed: {record['error']}; details saved")
185
+ with (directory / "predictions.jsonl").open("a") as f:
186
+ for case in cases[len(records) :]:
187
+ record = invoke(adapter, case)
188
+ f.write(dumps(record) + "\n")
189
+ f.flush()
190
+ records.append(record)
191
+ if len(records) % 50 == 0 or len(records) == len(cases):
192
+ print(f"Completed {len(records)}/{len(cases)} requests", flush=True)
193
+ finally:
194
+ if hasattr(adapter, "close"):
195
+ adapter.close()
196
+ result = report(cases, records, bins, tolerance)
197
+ result["run"] = metadata
198
+ warmups = read_jsonl(directory / "warmup.jsonl")
199
+ result["warmup"] = {
200
+ "requests": len(warmups),
201
+ "usage": [r.get("response", {}).get("usage") for r in warmups],
202
+ }
203
+ result["resolved_models"] = sorted(
204
+ {str(r["response"].get("model")) for r in records if isinstance(r.get("response"), dict)}
205
+ )
206
+ write_json(directory / "report.json", result)
207
+ (directory / "report.md").write_text(render_report(result))
208
+ return result
209
+
210
+
211
+ def score(bundle, predictions, output, bins=15, tolerance=0.02):
212
+ cases, manifest = load_bundle(bundle)
213
+ records = read_jsonl(predictions)
214
+ mapping = {r["case_id"]: r for r in records}
215
+ expected = {c["id"] for c in cases}
216
+ if len(mapping) != len(records) or set(mapping) - expected:
217
+ raise ValueError("duplicate or unknown prediction IDs")
218
+ complete = []
219
+ for case in cases:
220
+ r = dict(mapping.get(case["id"], {"case_id": case["id"], "error": "missing prediction"}))
221
+ r["question_count"] = len(case["questions"])
222
+ complete.append(r)
223
+ result = report(cases, complete, bins, tolerance)
224
+ result["manifest"] = manifest
225
+ write_json(output, result)
226
+ return result
227
+
228
+
229
+ def compare(paths):
230
+ reports = [json.loads(Path(p).read_text()) for p in paths]
231
+ protocols = {
232
+ (
233
+ r["run"]["manifest"]["sha256"],
234
+ r["metric_version"],
235
+ r["ece_bins"],
236
+ r["probability_sum_tolerance"],
237
+ )
238
+ for r in reports
239
+ }
240
+ if len(protocols) != 1:
241
+ raise ValueError("comparison requires identical frozen cases and scoring conventions")
242
+ lines = [
243
+ "| Model | Mode | Valid / attempted | Accuracy (all) | ECE | Request p50 ms |",
244
+ "|---|---|---:|---:|---:|---:|",
245
+ ]
246
+ for r in reports:
247
+ c, s = r["run"]["config"], r["overall"]
248
+ e = "—" if s["ece_top_label"] is None else f"{s['ece_top_label']:.4f}"
249
+ lines.append(
250
+ f"| {c['name']} | {c['mode']} | {s['valid']}/{s['attempted']} | "
251
+ f"{s['accuracy_all']:.4f} | {e} | {r['latency']['p50_ms']} |"
252
+ )
253
+ lines += [
254
+ "",
255
+ "Training modes, runtime versions, hardware and network location remain material differences.",
256
+ ]
257
+ return "\n".join(lines)
ariadne_bench/schema.py ADDED
@@ -0,0 +1,172 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """One ordered question schema for data, native APIs and local model plugins."""
2
+
3
+ import hashlib
4
+ import json
5
+ import math
6
+ from pathlib import Path
7
+
8
+
9
+ def dumps(value):
10
+ # Ordering is intentional: changing option order changes the model's input.
11
+ return json.dumps(value, ensure_ascii=False, separators=(",", ":"), allow_nan=False)
12
+
13
+
14
+ def digest(value):
15
+ return hashlib.sha256(dumps(value).encode()).hexdigest()
16
+
17
+
18
+ def read_jsonl(path):
19
+ return [json.loads(line) for line in Path(path).read_text().splitlines() if line.strip()]
20
+
21
+
22
+ def write_json(path, value):
23
+ Path(path).write_text(json.dumps(value, ensure_ascii=False, indent=2, allow_nan=False) + "\n")
24
+
25
+
26
+ def labels(question):
27
+ kind = question["type"]
28
+ if kind == "noul":
29
+ return ["false", "true"]
30
+ if kind == "choice":
31
+ criteria = question["criteria"]
32
+ if not isinstance(criteria, dict) or not criteria:
33
+ raise ValueError("choice requires a nonempty criteria map")
34
+ return list(criteria)
35
+ if kind == "score":
36
+ criteria = question["criteria"]
37
+ if not isinstance(criteria, list) or len(criteria) < 2:
38
+ raise ValueError("score requires at least two ordered criteria")
39
+ return [str(i) for i in range(len(criteria))]
40
+ raise ValueError(f"unknown question type: {kind}")
41
+
42
+
43
+ def distribution(raw, keys, tolerance=0.02):
44
+ if not isinstance(raw, dict) or set(raw) != set(keys):
45
+ raise ValueError("probability keys must match every option exactly")
46
+ if any(isinstance(raw[k], bool) or not isinstance(raw[k], (int, float)) for k in keys):
47
+ raise ValueError("probabilities must be numbers")
48
+ values = [float(raw[k]) for k in keys]
49
+ if any(not math.isfinite(p) or p < 0 or p > 1 for p in values):
50
+ raise ValueError("probabilities must be finite and in [0, 1]")
51
+ total = sum(values)
52
+ if total <= 0 or abs(total - 1) > tolerance + 1e-12:
53
+ raise ValueError(f"probabilities sum to {total}, outside rounding tolerance {tolerance}")
54
+ return [p / total for p in values], total
55
+
56
+
57
+ def validate_case(case):
58
+ if not isinstance(case.get("id"), str) or not case["id"]:
59
+ raise ValueError("case requires a nonempty string id")
60
+ if not isinstance(case.get("suite"), str) or not case["suite"]:
61
+ raise ValueError("case requires a suite")
62
+ if not isinstance(case.get("state"), (str, dict, list)):
63
+ raise ValueError("state must be text, an object or an array")
64
+ questions, gold = case.get("questions"), case.get("gold")
65
+ if not isinstance(questions, dict) or not questions or set(questions) != set(gold or {}):
66
+ raise ValueError("each question needs exactly one gold entry")
67
+ for qid, question in questions.items():
68
+ if "instructions" not in question:
69
+ raise ValueError(f"missing instructions: {qid}")
70
+ keys = labels(question)
71
+ if gold[qid]["label"] not in keys:
72
+ raise ValueError(f"gold label outside options: {qid}")
73
+ if "probabilities" in gold[qid]:
74
+ distribution(gold[qid]["probabilities"], keys)
75
+ if "score" in gold[qid]:
76
+ score = gold[qid]["score"]
77
+ if not math.isfinite(score) or not 0 <= score <= len(keys) - 1:
78
+ raise ValueError(f"invalid gold score: {qid}")
79
+ dumps(case)
80
+
81
+
82
+ def normalize_answer(question, answer, tolerance=0.02):
83
+ if not isinstance(answer, dict) or answer.get("type") != question["type"]:
84
+ raise ValueError("answer type does not match question")
85
+ keys = labels(question)
86
+ if question["type"] == "noul":
87
+ p = answer.get("noul")
88
+ if isinstance(p, bool) or not isinstance(p, (int, float)):
89
+ raise ValueError("noul must be a probability")
90
+ raw = {"false": 1 - p, "true": p}
91
+ else:
92
+ raw = answer.get("probabilities")
93
+ probs, total = distribution(raw, keys, tolerance)
94
+ idx = max(range(len(keys)), key=probs.__getitem__)
95
+ if question["type"] == "choice":
96
+ chosen = answer.get("choice")
97
+ if chosen not in keys or probs[keys.index(chosen)] < max(probs) - 1e-12:
98
+ raise ValueError("choice is missing or disagrees with the distribution")
99
+ idx = keys.index(chosen) # retain the provider's tie break
100
+ if question["type"] == "score":
101
+ score = answer.get("score")
102
+ if isinstance(score, bool) or not isinstance(score, (int, float)):
103
+ raise ValueError("score is missing or nonnumeric")
104
+ if not math.isfinite(score) or not 0 <= score <= len(keys) - 1:
105
+ raise ValueError("score is outside its rubric")
106
+ expected = sum(i * p for i, p in enumerate(probs))
107
+ if abs(score - expected) > max(0.05, tolerance * (len(keys) - 1)):
108
+ raise ValueError("score disagrees with the distribution's expectation")
109
+ if "confidence" in answer:
110
+ c = answer["confidence"]
111
+ if (
112
+ isinstance(c, bool)
113
+ or not isinstance(c, (int, float))
114
+ or not math.isfinite(c)
115
+ or not 0 <= c <= 1
116
+ ):
117
+ raise ValueError("invalid provider confidence")
118
+ return {
119
+ "label": keys[idx],
120
+ "probabilities": dict(zip(keys, probs)),
121
+ "confidence": probs[idx],
122
+ "raw_probability_sum": total,
123
+ "provider_confidence": answer.get("confidence"),
124
+ }
125
+
126
+
127
+ def save_bundle(directory, cases, metadata):
128
+ directory = Path(directory)
129
+ if directory.exists():
130
+ raise ValueError(f"output already exists: {directory}; use a new directory")
131
+ ids = set()
132
+ for case in cases:
133
+ validate_case(case)
134
+ if case["id"] in ids:
135
+ raise ValueError(f"duplicate case id: {case['id']}")
136
+ ids.add(case["id"])
137
+ if not cases:
138
+ raise ValueError("empty benchmark")
139
+ content = "".join(dumps(c) + "\n" for c in cases).encode()
140
+ directory.mkdir(parents=True)
141
+ (directory / "cases.jsonl").write_bytes(content)
142
+ manifest = {
143
+ "format_version": 1,
144
+ "sha256": hashlib.sha256(content).hexdigest(),
145
+ "cases": len(cases),
146
+ "decisions": sum(len(c["questions"]) for c in cases),
147
+ **metadata,
148
+ }
149
+ write_json(directory / "manifest.json", manifest)
150
+ return manifest
151
+
152
+
153
+ def load_bundle(directory):
154
+ directory = Path(directory)
155
+ manifest = json.loads((directory / "manifest.json").read_text())
156
+ content = (directory / "cases.jsonl").read_bytes()
157
+ if hashlib.sha256(content).hexdigest() != manifest["sha256"]:
158
+ raise ValueError("case file does not match its manifest hash")
159
+ cases = read_jsonl(directory / "cases.jsonl")
160
+ ids = set()
161
+ for case in cases:
162
+ validate_case(case)
163
+ if case["id"] in ids:
164
+ raise ValueError("duplicate case id")
165
+ ids.add(case["id"])
166
+ if (
167
+ not cases
168
+ or len(cases) != manifest["cases"]
169
+ or sum(len(c["questions"]) for c in cases) != manifest["decisions"]
170
+ ):
171
+ raise ValueError("manifest counts disagree with cases")
172
+ return cases, manifest
benchmarks/sources.lock.json ADDED
@@ -0,0 +1,102 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "LocalLLaMA/typed-decisions": {
3
+ "revision": "f7a2487edd7a043a5441a5e9ccc7fe5ddbd9ebe8",
4
+ "license": "apache-2.0",
5
+ "url": "https://huggingface.co/datasets/LocalLLaMA/typed-decisions"
6
+ },
7
+ "SetFit/enron_spam": {
8
+ "revision": "1916f66c89d52221ae33eb57d44498b4f3a5df22",
9
+ "license": "see source dataset card",
10
+ "url": "https://huggingface.co/datasets/SetFit/enron_spam"
11
+ },
12
+ "SetFit/sst5": {
13
+ "revision": "e51bdcd8cd3a30da231967c1a249ba59361279a3",
14
+ "license": "see source dataset card",
15
+ "url": "https://huggingface.co/datasets/SetFit/sst5"
16
+ },
17
+ "Tobi-Bueck/customer-support-tickets": {
18
+ "revision": "ddf1c81a5475992c4fa6752bf1e8b4e31f07bbeb",
19
+ "license": "cc-by-nc-4.0",
20
+ "url": "https://huggingface.co/datasets/Tobi-Bueck/customer-support-tickets"
21
+ },
22
+ "btzsc/btzsc": {
23
+ "revision": "fef2a2ac62b69c58670047dddf045c53d7c3cb5e",
24
+ "license": "see source dataset card",
25
+ "url": "https://huggingface.co/datasets/btzsc/btzsc"
26
+ },
27
+ "dair-ai/emotion": {
28
+ "revision": "cab853a1dbdf4c42c2b3ef2173804746df8825fe",
29
+ "license": [
30
+ "other"
31
+ ],
32
+ "url": "https://huggingface.co/datasets/dair-ai/emotion"
33
+ },
34
+ "deepset/prompt-injections": {
35
+ "revision": "4f61ecb038e9c3fb77e21034b22511b523772cdd",
36
+ "license": "apache-2.0",
37
+ "url": "https://huggingface.co/datasets/deepset/prompt-injections"
38
+ },
39
+ "facebook/xnli": {
40
+ "revision": "b8dd5d7af51114dbda02c0e3f6133f332186418e",
41
+ "license": "see source dataset card",
42
+ "url": "https://huggingface.co/datasets/facebook/xnli"
43
+ },
44
+ "fancyzhx/ag_news": {
45
+ "revision": "eb185aade064a813bc0b7f42de02595523103ca4",
46
+ "license": [
47
+ "unknown"
48
+ ],
49
+ "url": "https://huggingface.co/datasets/fancyzhx/ag_news"
50
+ },
51
+ "google-research-datasets/mbpp": {
52
+ "revision": "4bb6404fdc6cacfda99d4ac4205087b89d32030c",
53
+ "license": [
54
+ "cc-by-4.0"
55
+ ],
56
+ "url": "https://huggingface.co/datasets/google-research-datasets/mbpp"
57
+ },
58
+ "google/boolq": {
59
+ "revision": "35b264d03638db9f4ce671b711558bf7ff0f80d5",
60
+ "license": [
61
+ "cc-by-sa-3.0"
62
+ ],
63
+ "url": "https://huggingface.co/datasets/google/boolq"
64
+ },
65
+ "lmsys/toxic-chat": {
66
+ "revision": "29df8e4dba60e1f4af4b4075c0705c5b313548a8",
67
+ "license": "cc-by-nc-4.0",
68
+ "url": "https://huggingface.co/datasets/lmsys/toxic-chat"
69
+ },
70
+ "microsoft/ms_marco": {
71
+ "revision": "a47ee7aae8d7d466ba15f9f0bfac3b3681087b3a",
72
+ "license": "see source dataset card",
73
+ "url": "https://huggingface.co/datasets/microsoft/ms_marco"
74
+ },
75
+ "mteb/amazon_massive_intent": {
76
+ "revision": "940fd47a81eaa7f2cc7b129674d945d618ac38c2",
77
+ "license": "apache-2.0",
78
+ "url": "https://huggingface.co/datasets/mteb/amazon_massive_intent"
79
+ },
80
+ "mteb/amazon_massive_scenario": {
81
+ "revision": "58871793b91addb7c5f7afff26ccf08737fb6697",
82
+ "license": "apache-2.0",
83
+ "url": "https://huggingface.co/datasets/mteb/amazon_massive_scenario"
84
+ },
85
+ "mteb/banking77": {
86
+ "revision": "18072d2685ea682290f7b8924d94c62acc19c0b2",
87
+ "license": "mit",
88
+ "url": "https://huggingface.co/datasets/mteb/banking77"
89
+ },
90
+ "openai/gsm8k": {
91
+ "revision": "740312add88f781978c0658806c59bc2815b9866",
92
+ "license": [
93
+ "mit"
94
+ ],
95
+ "url": "https://huggingface.co/datasets/openai/gsm8k"
96
+ },
97
+ "zefang-liu/phishing-email-dataset": {
98
+ "revision": "34085a032c123ca237f314a01a67909cdea35e34",
99
+ "license": "lgpl-3.0",
100
+ "url": "https://huggingface.co/datasets/zefang-liu/phishing-email-dataset"
101
+ }
102
+ }
environment.json ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "torch_version": "2.13.0+cu130",
3
+ "transformers_version": "5.17.0",
4
+ "laya_version": "0.3.20",
5
+ "dtype": "torch.bfloat16",
6
+ "gpu_name": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
7
+ "reproducibility": {
8
+ "deterministic_algorithms": true,
9
+ "warn_only": false,
10
+ "cublas_workspace_config": ":4096:8",
11
+ "python_hash_seed": "0",
12
+ "cudnn_benchmark": false,
13
+ "cudnn_deterministic": true,
14
+ "cudnn_allow_tf32": false,
15
+ "float32_matmul_precision": "highest",
16
+ "cudnn_sdp_enabled": false,
17
+ "flash_sdp_enabled": true,
18
+ "mem_efficient_sdp_enabled": true,
19
+ "math_sdp_enabled": true,
20
+ "cuda_version": "13.0",
21
+ "cudnn_version": 92000,
22
+ "scope": "fixed hardware/runtime; cross-platform bitwise agreement is not promised"
23
+ },
24
+ "benchmark_runtime": {
25
+ "python": "3.12.3",
26
+ "platform": "Linux-7.0.0-34-generic-x86_64-with-glibc2.39",
27
+ "processor": "x86_64",
28
+ "packages": {
29
+ "laya": "0.3.20",
30
+ "torch": "2.13.0+cu130",
31
+ "transformers": "5.17.0",
32
+ "datasets": "5.0.1",
33
+ "huggingface-hub": "1.33.0"
34
+ },
35
+ "harness": "0.1.0"
36
+ },
37
+ "process_environment": {
38
+ "PYTHONHASHSEED": "0",
39
+ "CUBLAS_WORKSPACE_CONFIG": ":4096:8",
40
+ "USE_TF": "0",
41
+ "OMP_NUM_THREADS": "4",
42
+ "TOKENIZERS_PARALLELISM": "false"
43
+ }
44
+ }
evidence/adapter-equivalence.json ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "device": "cuda:1",
3
+ "deterministic_algorithms": true,
4
+ "trained_checkpoint_reference": "runs/base-frozen-linear-seeds/seed-0/training/training.json",
5
+ "diagnostic_only_no_optimizer_updates": true,
6
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
7
+ "forward_passed": true,
8
+ "all_frozen_modules_in_eval_mode": true,
9
+ "native_attention_implementation": "sdpa",
10
+ "all_gradients_exact": true,
11
+ "comparisons": [
12
+ {
13
+ "stage": "initial",
14
+ "decisions": 8,
15
+ "loss_difference": 0.0,
16
+ "max_logit_difference": 0.0,
17
+ "max_gradient_difference": 0.0,
18
+ "same_wrapper_repeat_gradient_difference": 0.0,
19
+ "legacy_repeat_gradient_difference": 0.0,
20
+ "max_gradient_magnitude": 0.4774176776409149,
21
+ "cuda_rng_unchanged": true,
22
+ "forward_exact": true,
23
+ "exact_match": true
24
+ },
25
+ {
26
+ "stage": "trained_seed_0",
27
+ "decisions": 8,
28
+ "loss_difference": 0.0,
29
+ "max_logit_difference": 0.0,
30
+ "max_gradient_difference": 0.0,
31
+ "same_wrapper_repeat_gradient_difference": 0.0,
32
+ "legacy_repeat_gradient_difference": 0.0,
33
+ "max_gradient_magnitude": 0.01077677309513092,
34
+ "cuda_rng_unchanged": true,
35
+ "forward_exact": true,
36
+ "exact_match": true
37
+ }
38
+ ]
39
+ }
evidence/audit.json ADDED
@@ -0,0 +1,59 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "frozen_state_sha256": "7560d83b1e1c17cce7eb67b25674fc8e7f0958bac4b47f559c725f8695c4d59b",
3
+ "verified_decisions": 25580,
4
+ "initial_identity_verified": true,
5
+ "runs": [
6
+ {
7
+ "seed": 0,
8
+ "checkpoint_sha256": "bcb37755de54782bc4d1f79615539f48940aedb2b733dadc1033fba2f2304cf8",
9
+ "checkpoint_bytes": 4198616,
10
+ "all_pretrained_tensors_unchanged": true,
11
+ "independent_scoring_matches": true,
12
+ "min_epochs": 10,
13
+ "stop_epoch": 23,
14
+ "minimum_epoch_rule_verified": true
15
+ },
16
+ {
17
+ "seed": 1,
18
+ "checkpoint_sha256": "e5ec5e1d494509f63960b6949af15be8390442271805ba45d796a1911b5c239a",
19
+ "checkpoint_bytes": 4198616,
20
+ "all_pretrained_tensors_unchanged": true,
21
+ "independent_scoring_matches": true,
22
+ "min_epochs": 10,
23
+ "stop_epoch": 17,
24
+ "minimum_epoch_rule_verified": true
25
+ },
26
+ {
27
+ "seed": 2,
28
+ "checkpoint_sha256": "61391c3869c1f4a8449444db6a9e3b660692eefa25e3c149cee0d6af5daf264c",
29
+ "checkpoint_bytes": 4198616,
30
+ "all_pretrained_tensors_unchanged": true,
31
+ "independent_scoring_matches": true,
32
+ "min_epochs": 10,
33
+ "stop_epoch": 18,
34
+ "minimum_epoch_rule_verified": true
35
+ },
36
+ {
37
+ "seed": 3,
38
+ "checkpoint_sha256": "26ed2443b8ea146d3fab98e94ab55356557b9ef10ac5b8ba656b1a37cf4f0f64",
39
+ "checkpoint_bytes": 4198616,
40
+ "all_pretrained_tensors_unchanged": true,
41
+ "independent_scoring_matches": true,
42
+ "min_epochs": 10,
43
+ "stop_epoch": 27,
44
+ "minimum_epoch_rule_verified": true
45
+ },
46
+ {
47
+ "seed": 4,
48
+ "checkpoint_sha256": "603a83ca4996e32b4753250ea6b07856e7cc5196b41e635cdad35ee4270eca38",
49
+ "checkpoint_bytes": 4198616,
50
+ "all_pretrained_tensors_unchanged": true,
51
+ "independent_scoring_matches": true,
52
+ "min_epochs": 10,
53
+ "stop_epoch": 10,
54
+ "minimum_epoch_rule_verified": true
55
+ }
56
+ ],
57
+ "verified_runs": 5,
58
+ "statistics_verified": true
59
+ }
evidence/candidate-uncertainty.json ADDED
@@ -0,0 +1,76 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "candidate_seed": 1,
3
+ "selected_epoch": 14,
4
+ "benchmark_cases_sha256": "10fb671245cdac1ee09e9c4d2b9b09829fbf1abd06e0954fddee0b9ec8519854",
5
+ "resamples": 10000,
6
+ "bootstrap_seed_per_suite": 42,
7
+ "unit": "cases, preserving all decisions in each case",
8
+ "interval": "paired percentile 95%; percentage points",
9
+ "limits": "descriptive resampling of the measured cases for a fixed model; no multiple-comparison correction or training-seed uncertainty; not used for checkpoint selection",
10
+ "suites": {
11
+ "typed-decisions": {
12
+ "cases": 400,
13
+ "accuracy_difference_pp": 40.7,
14
+ "case_bootstrap_95_interval_pp": [
15
+ 37.3,
16
+ 44.101249999999986
17
+ ]
18
+ },
19
+ "ag-news": {
20
+ "cases": 600,
21
+ "accuracy_difference_pp": -1.6666666666666667,
22
+ "case_bootstrap_95_interval_pp": [
23
+ -3.5,
24
+ 0.16666666666666666
25
+ ]
26
+ },
27
+ "boolq": {
28
+ "cases": 600,
29
+ "accuracy_difference_pp": -3.1666666666666665,
30
+ "case_bootstrap_95_interval_pp": [
31
+ -5.833333333333333,
32
+ -0.6666666666666666
33
+ ]
34
+ },
35
+ "emotion": {
36
+ "cases": 600,
37
+ "accuracy_difference_pp": 0.0,
38
+ "case_bootstrap_95_interval_pp": [
39
+ -2.5,
40
+ 2.6666666666666665
41
+ ]
42
+ },
43
+ "prompt-injections": {
44
+ "cases": 116,
45
+ "accuracy_difference_pp": 2.586206896551724,
46
+ "case_bootstrap_95_interval_pp": [
47
+ -5.172413793103448,
48
+ 10.344827586206897
49
+ ]
50
+ },
51
+ "sst5": {
52
+ "cases": 600,
53
+ "accuracy_difference_pp": 5.0,
54
+ "case_bootstrap_95_interval_pp": [
55
+ 1.1666666666666667,
56
+ 9.0
57
+ ]
58
+ },
59
+ "massive-intent.en": {
60
+ "cases": 300,
61
+ "accuracy_difference_pp": -4.333333333333333,
62
+ "case_bootstrap_95_interval_pp": [
63
+ -8.333333333333334,
64
+ -0.3333333333333333
65
+ ]
66
+ },
67
+ "xnli.en": {
68
+ "cases": 300,
69
+ "accuracy_difference_pp": 1.6666666666666667,
70
+ "case_bootstrap_95_interval_pp": [
71
+ -1.0,
72
+ 4.666666666666667
73
+ ]
74
+ }
75
+ }
76
+ }
evidence/identity-check.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "passed": true,
3
+ "cases": 3516,
4
+ "decisions": 5116,
5
+ "mismatched_cases": []
6
+ }
evidence/replay-check.json ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "passed": true,
3
+ "seed": 0,
4
+ "separate_processes": true,
5
+ "epochs_per_process": 2,
6
+ "optimizer_updates_per_process": 338,
7
+ "initial_validation_exact": true,
8
+ "epoch_losses_and_metrics_exact": true,
9
+ "selected_checkpoint_tensors_bitwise_equal": true,
10
+ "checkpoint_sha256": [
11
+ "3a30ce9881420cebc61829e52b41a9e669a114b4dffe0f8f836831572bd70331",
12
+ "3a30ce9881420cebc61829e52b41a9e669a114b4dffe0f8f836831572bd70331"
13
+ ],
14
+ "reproducibility": {
15
+ "deterministic_algorithms": true,
16
+ "warn_only": false,
17
+ "cublas_workspace_config": ":4096:8",
18
+ "python_hash_seed": "0",
19
+ "cudnn_benchmark": false,
20
+ "cudnn_deterministic": true,
21
+ "cudnn_allow_tf32": false,
22
+ "float32_matmul_precision": "highest",
23
+ "cudnn_sdp_enabled": false,
24
+ "flash_sdp_enabled": true,
25
+ "mem_efficient_sdp_enabled": true,
26
+ "math_sdp_enabled": true,
27
+ "cuda_version": "13.0",
28
+ "cudnn_version": 92000,
29
+ "scope": "fixed hardware/runtime; cross-platform bitwise agreement is not promised"
30
+ },
31
+ "limit": "replay verified for this data, device, runtime, and two-epoch prefix"
32
+ }
evidence/seed-0-history.json ADDED
@@ -0,0 +1,416 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "seed": 0,
3
+ "initial_validation": {
4
+ "soft_cross_entropy": 1.544869564374288,
5
+ "accuracy": 0.36833333333333335,
6
+ "brier_soft": 0.43051501592000324,
7
+ "decisions": 600
8
+ },
9
+ "epochs": [
10
+ {
11
+ "epoch": 1,
12
+ "train_soft_cross_entropy": 1.1371864740936843,
13
+ "validation": {
14
+ "soft_cross_entropy": 1.195652896563212,
15
+ "accuracy": 0.3433333333333333,
16
+ "brier_soft": 0.2543176457285881,
17
+ "decisions": 600
18
+ },
19
+ "early_stopping": {
20
+ "patience": 3,
21
+ "min_epochs": 10,
22
+ "epochs_without_improvement": 0,
23
+ "best_validation_loss": 1.195652896563212,
24
+ "improved": true
25
+ }
26
+ },
27
+ {
28
+ "epoch": 2,
29
+ "train_soft_cross_entropy": 1.1867035720966481,
30
+ "validation": {
31
+ "soft_cross_entropy": 1.1503166087468466,
32
+ "accuracy": 0.43666666666666665,
33
+ "brier_soft": 0.2273359453678131,
34
+ "decisions": 600
35
+ },
36
+ "early_stopping": {
37
+ "patience": 3,
38
+ "min_epochs": 10,
39
+ "epochs_without_improvement": 0,
40
+ "best_validation_loss": 1.1503166087468466,
41
+ "improved": true
42
+ }
43
+ },
44
+ {
45
+ "epoch": 3,
46
+ "train_soft_cross_entropy": 1.1479326088340194,
47
+ "validation": {
48
+ "soft_cross_entropy": 1.1359616955121359,
49
+ "accuracy": 0.4583333333333333,
50
+ "brier_soft": 0.21933125893274943,
51
+ "decisions": 600
52
+ },
53
+ "early_stopping": {
54
+ "patience": 3,
55
+ "min_epochs": 10,
56
+ "epochs_without_improvement": 0,
57
+ "best_validation_loss": 1.1359616955121359,
58
+ "improved": true
59
+ }
60
+ },
61
+ {
62
+ "epoch": 4,
63
+ "train_soft_cross_entropy": 1.17192115077266,
64
+ "validation": {
65
+ "soft_cross_entropy": 1.1432366434733072,
66
+ "accuracy": 0.4533333333333333,
67
+ "brier_soft": 0.22181885927915573,
68
+ "decisions": 600
69
+ },
70
+ "early_stopping": {
71
+ "patience": 3,
72
+ "min_epochs": 10,
73
+ "epochs_without_improvement": 1,
74
+ "best_validation_loss": 1.1359616955121359,
75
+ "improved": false
76
+ }
77
+ },
78
+ {
79
+ "epoch": 5,
80
+ "train_soft_cross_entropy": 1.1521123818114951,
81
+ "validation": {
82
+ "soft_cross_entropy": 1.1532048479715984,
83
+ "accuracy": 0.42,
84
+ "brier_soft": 0.22427557816108068,
85
+ "decisions": 600
86
+ },
87
+ "early_stopping": {
88
+ "patience": 3,
89
+ "min_epochs": 10,
90
+ "epochs_without_improvement": 2,
91
+ "best_validation_loss": 1.1359616955121359,
92
+ "improved": false
93
+ }
94
+ },
95
+ {
96
+ "epoch": 6,
97
+ "train_soft_cross_entropy": 1.139600450904281,
98
+ "validation": {
99
+ "soft_cross_entropy": 1.1171702599525453,
100
+ "accuracy": 0.465,
101
+ "brier_soft": 0.21267332146565118,
102
+ "decisions": 600
103
+ },
104
+ "early_stopping": {
105
+ "patience": 3,
106
+ "min_epochs": 10,
107
+ "epochs_without_improvement": 0,
108
+ "best_validation_loss": 1.1171702599525453,
109
+ "improved": true
110
+ }
111
+ },
112
+ {
113
+ "epoch": 7,
114
+ "train_soft_cross_entropy": 1.1084874327094467,
115
+ "validation": {
116
+ "soft_cross_entropy": 1.101010274887085,
117
+ "accuracy": 0.48333333333333334,
118
+ "brier_soft": 0.20341493268807728,
119
+ "decisions": 600
120
+ },
121
+ "early_stopping": {
122
+ "patience": 3,
123
+ "min_epochs": 10,
124
+ "epochs_without_improvement": 0,
125
+ "best_validation_loss": 1.101010274887085,
126
+ "improved": true
127
+ }
128
+ },
129
+ {
130
+ "epoch": 8,
131
+ "train_soft_cross_entropy": 1.0972168815577472,
132
+ "validation": {
133
+ "soft_cross_entropy": 1.0927943936983744,
134
+ "accuracy": 0.4866666666666667,
135
+ "brier_soft": 0.20212342927853266,
136
+ "decisions": 600
137
+ },
138
+ "early_stopping": {
139
+ "patience": 3,
140
+ "min_epochs": 10,
141
+ "epochs_without_improvement": 0,
142
+ "best_validation_loss": 1.0927943936983744,
143
+ "improved": true
144
+ }
145
+ },
146
+ {
147
+ "epoch": 9,
148
+ "train_soft_cross_entropy": 1.0959268622045164,
149
+ "validation": {
150
+ "soft_cross_entropy": 1.0886747964223227,
151
+ "accuracy": 0.485,
152
+ "brier_soft": 0.1981955200433731,
153
+ "decisions": 600
154
+ },
155
+ "early_stopping": {
156
+ "patience": 3,
157
+ "min_epochs": 10,
158
+ "epochs_without_improvement": 0,
159
+ "best_validation_loss": 1.0886747964223227,
160
+ "improved": true
161
+ }
162
+ },
163
+ {
164
+ "epoch": 10,
165
+ "train_soft_cross_entropy": 1.0879168816849036,
166
+ "validation": {
167
+ "soft_cross_entropy": 1.0776302735010783,
168
+ "accuracy": 0.5133333333333333,
169
+ "brier_soft": 0.19176260660092037,
170
+ "decisions": 600
171
+ },
172
+ "early_stopping": {
173
+ "patience": 3,
174
+ "min_epochs": 10,
175
+ "epochs_without_improvement": 0,
176
+ "best_validation_loss": 1.0776302735010783,
177
+ "improved": true
178
+ }
179
+ },
180
+ {
181
+ "epoch": 11,
182
+ "train_soft_cross_entropy": 1.0794475747037817,
183
+ "validation": {
184
+ "soft_cross_entropy": 1.0989113728205362,
185
+ "accuracy": 0.49666666666666665,
186
+ "brier_soft": 0.20668872609734534,
187
+ "decisions": 600
188
+ },
189
+ "early_stopping": {
190
+ "patience": 3,
191
+ "min_epochs": 10,
192
+ "epochs_without_improvement": 1,
193
+ "best_validation_loss": 1.0776302735010783,
194
+ "improved": false
195
+ }
196
+ },
197
+ {
198
+ "epoch": 12,
199
+ "train_soft_cross_entropy": 1.0683085640271506,
200
+ "validation": {
201
+ "soft_cross_entropy": 1.0758668931325277,
202
+ "accuracy": 0.5016666666666667,
203
+ "brier_soft": 0.19317058285077413,
204
+ "decisions": 600
205
+ },
206
+ "early_stopping": {
207
+ "patience": 3,
208
+ "min_epochs": 10,
209
+ "epochs_without_improvement": 0,
210
+ "best_validation_loss": 1.0758668931325277,
211
+ "improved": true
212
+ }
213
+ },
214
+ {
215
+ "epoch": 13,
216
+ "train_soft_cross_entropy": 1.052984880871243,
217
+ "validation": {
218
+ "soft_cross_entropy": 1.0600487383206685,
219
+ "accuracy": 0.5316666666666666,
220
+ "brier_soft": 0.18257314254840215,
221
+ "decisions": 600
222
+ },
223
+ "early_stopping": {
224
+ "patience": 3,
225
+ "min_epochs": 10,
226
+ "epochs_without_improvement": 0,
227
+ "best_validation_loss": 1.0600487383206685,
228
+ "improved": true
229
+ }
230
+ },
231
+ {
232
+ "epoch": 14,
233
+ "train_soft_cross_entropy": 1.0401919462062694,
234
+ "validation": {
235
+ "soft_cross_entropy": 1.0525298221906025,
236
+ "accuracy": 0.5366666666666666,
237
+ "brier_soft": 0.1811353324353695,
238
+ "decisions": 600
239
+ },
240
+ "early_stopping": {
241
+ "patience": 3,
242
+ "min_epochs": 10,
243
+ "epochs_without_improvement": 0,
244
+ "best_validation_loss": 1.0525298221906025,
245
+ "improved": true
246
+ }
247
+ },
248
+ {
249
+ "epoch": 15,
250
+ "train_soft_cross_entropy": 1.0262903751267327,
251
+ "validation": {
252
+ "soft_cross_entropy": 1.0581993921597799,
253
+ "accuracy": 0.5233333333333333,
254
+ "brier_soft": 0.18550985043247542,
255
+ "decisions": 600
256
+ },
257
+ "early_stopping": {
258
+ "patience": 3,
259
+ "min_epochs": 10,
260
+ "epochs_without_improvement": 1,
261
+ "best_validation_loss": 1.0525298221906025,
262
+ "improved": false
263
+ }
264
+ },
265
+ {
266
+ "epoch": 16,
267
+ "train_soft_cross_entropy": 1.0127459985238534,
268
+ "validation": {
269
+ "soft_cross_entropy": 1.0545752588907877,
270
+ "accuracy": 0.54,
271
+ "brier_soft": 0.18113683501879374,
272
+ "decisions": 600
273
+ },
274
+ "early_stopping": {
275
+ "patience": 3,
276
+ "min_epochs": 10,
277
+ "epochs_without_improvement": 2,
278
+ "best_validation_loss": 1.0525298221906025,
279
+ "improved": false
280
+ }
281
+ },
282
+ {
283
+ "epoch": 17,
284
+ "train_soft_cross_entropy": 1.0096984642523306,
285
+ "validation": {
286
+ "soft_cross_entropy": 1.0468252456188203,
287
+ "accuracy": 0.5483333333333333,
288
+ "brier_soft": 0.17509076982736588,
289
+ "decisions": 600
290
+ },
291
+ "early_stopping": {
292
+ "patience": 3,
293
+ "min_epochs": 10,
294
+ "epochs_without_improvement": 0,
295
+ "best_validation_loss": 1.0468252456188203,
296
+ "improved": true
297
+ }
298
+ },
299
+ {
300
+ "epoch": 18,
301
+ "train_soft_cross_entropy": 1.0022709012914588,
302
+ "validation": {
303
+ "soft_cross_entropy": 1.0376164364814757,
304
+ "accuracy": 0.5766666666666667,
305
+ "brier_soft": 0.17226109830041728,
306
+ "decisions": 600
307
+ },
308
+ "early_stopping": {
309
+ "patience": 3,
310
+ "min_epochs": 10,
311
+ "epochs_without_improvement": 0,
312
+ "best_validation_loss": 1.0376164364814757,
313
+ "improved": true
314
+ }
315
+ },
316
+ {
317
+ "epoch": 19,
318
+ "train_soft_cross_entropy": 0.9934269437083492,
319
+ "validation": {
320
+ "soft_cross_entropy": 1.0526774628957112,
321
+ "accuracy": 0.5516666666666666,
322
+ "brier_soft": 0.17908830145994822,
323
+ "decisions": 600
324
+ },
325
+ "early_stopping": {
326
+ "patience": 3,
327
+ "min_epochs": 10,
328
+ "epochs_without_improvement": 1,
329
+ "best_validation_loss": 1.0376164364814757,
330
+ "improved": false
331
+ }
332
+ },
333
+ {
334
+ "epoch": 20,
335
+ "train_soft_cross_entropy": 0.9915926403911025,
336
+ "validation": {
337
+ "soft_cross_entropy": 1.0292588464419048,
338
+ "accuracy": 0.57,
339
+ "brier_soft": 0.16918850486477216,
340
+ "decisions": 600
341
+ },
342
+ "early_stopping": {
343
+ "patience": 3,
344
+ "min_epochs": 10,
345
+ "epochs_without_improvement": 0,
346
+ "best_validation_loss": 1.0292588464419048,
347
+ "improved": true
348
+ }
349
+ },
350
+ {
351
+ "epoch": 21,
352
+ "train_soft_cross_entropy": 0.9833765982698511,
353
+ "validation": {
354
+ "soft_cross_entropy": 1.0493251442909242,
355
+ "accuracy": 0.555,
356
+ "brier_soft": 0.17917159788310527,
357
+ "decisions": 600
358
+ },
359
+ "early_stopping": {
360
+ "patience": 3,
361
+ "min_epochs": 10,
362
+ "epochs_without_improvement": 1,
363
+ "best_validation_loss": 1.0292588464419048,
364
+ "improved": false
365
+ }
366
+ },
367
+ {
368
+ "epoch": 22,
369
+ "train_soft_cross_entropy": 0.9786927858988445,
370
+ "validation": {
371
+ "soft_cross_entropy": 1.0412022503217062,
372
+ "accuracy": 0.555,
373
+ "brier_soft": 0.17517984464764594,
374
+ "decisions": 600
375
+ },
376
+ "early_stopping": {
377
+ "patience": 3,
378
+ "min_epochs": 10,
379
+ "epochs_without_improvement": 2,
380
+ "best_validation_loss": 1.0292588464419048,
381
+ "improved": false
382
+ }
383
+ },
384
+ {
385
+ "epoch": 23,
386
+ "train_soft_cross_entropy": 0.9695542351404826,
387
+ "validation": {
388
+ "soft_cross_entropy": 1.0376375365257262,
389
+ "accuracy": 0.5583333333333333,
390
+ "brier_soft": 0.17371407074232897,
391
+ "decisions": 600
392
+ },
393
+ "early_stopping": {
394
+ "patience": 3,
395
+ "min_epochs": 10,
396
+ "epochs_without_improvement": 3,
397
+ "best_validation_loss": 1.0292588464419048,
398
+ "improved": false
399
+ }
400
+ }
401
+ ],
402
+ "stopping": {
403
+ "reason": "early_stopping",
404
+ "epochs_completed": 23,
405
+ "patience": 3,
406
+ "min_epochs": 10,
407
+ "epochs_without_improvement": 3,
408
+ "metric": "validation.soft_cross_entropy",
409
+ "min_delta": 0.0
410
+ },
411
+ "trainable_names": [
412
+ "encoder.embeddings.interface.proj.weight",
413
+ "encoder.embeddings.interface.proj.bias"
414
+ ],
415
+ "frozen_parameters_unchanged": true
416
+ }
evidence/seed-1-history.json ADDED
@@ -0,0 +1,314 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "seed": 1,
3
+ "initial_validation": {
4
+ "soft_cross_entropy": 1.544869564374288,
5
+ "accuracy": 0.36833333333333335,
6
+ "brier_soft": 0.43051501592000324,
7
+ "decisions": 600
8
+ },
9
+ "epochs": [
10
+ {
11
+ "epoch": 1,
12
+ "train_soft_cross_entropy": 1.0002122403074194,
13
+ "validation": {
14
+ "soft_cross_entropy": 0.9555757478872935,
15
+ "accuracy": 0.665,
16
+ "brier_soft": 0.13395084381103517,
17
+ "decisions": 600
18
+ },
19
+ "early_stopping": {
20
+ "patience": 3,
21
+ "min_epochs": 10,
22
+ "epochs_without_improvement": 0,
23
+ "best_validation_loss": 0.9555757478872935,
24
+ "improved": true
25
+ }
26
+ },
27
+ {
28
+ "epoch": 2,
29
+ "train_soft_cross_entropy": 0.895550852616628,
30
+ "validation": {
31
+ "soft_cross_entropy": 0.8875669745604197,
32
+ "accuracy": 0.7216666666666667,
33
+ "brier_soft": 0.08987226173281669,
34
+ "decisions": 600
35
+ },
36
+ "early_stopping": {
37
+ "patience": 3,
38
+ "min_epochs": 10,
39
+ "epochs_without_improvement": 0,
40
+ "best_validation_loss": 0.8875669745604197,
41
+ "improved": true
42
+ }
43
+ },
44
+ {
45
+ "epoch": 3,
46
+ "train_soft_cross_entropy": 0.8399523215382187,
47
+ "validation": {
48
+ "soft_cross_entropy": 0.8776892550786336,
49
+ "accuracy": 0.725,
50
+ "brier_soft": 0.0829127719004949,
51
+ "decisions": 600
52
+ },
53
+ "early_stopping": {
54
+ "patience": 3,
55
+ "min_epochs": 10,
56
+ "epochs_without_improvement": 0,
57
+ "best_validation_loss": 0.8776892550786336,
58
+ "improved": true
59
+ }
60
+ },
61
+ {
62
+ "epoch": 4,
63
+ "train_soft_cross_entropy": 0.8150597869908368,
64
+ "validation": {
65
+ "soft_cross_entropy": 0.8523939752578735,
66
+ "accuracy": 0.7666666666666667,
67
+ "brier_soft": 0.07098765720923741,
68
+ "decisions": 600
69
+ },
70
+ "early_stopping": {
71
+ "patience": 3,
72
+ "min_epochs": 10,
73
+ "epochs_without_improvement": 0,
74
+ "best_validation_loss": 0.8523939752578735,
75
+ "improved": true
76
+ }
77
+ },
78
+ {
79
+ "epoch": 5,
80
+ "train_soft_cross_entropy": 0.7998251422246297,
81
+ "validation": {
82
+ "soft_cross_entropy": 0.853373521566391,
83
+ "accuracy": 0.745,
84
+ "brier_soft": 0.07141184127579132,
85
+ "decisions": 600
86
+ },
87
+ "early_stopping": {
88
+ "patience": 3,
89
+ "min_epochs": 10,
90
+ "epochs_without_improvement": 1,
91
+ "best_validation_loss": 0.8523939752578735,
92
+ "improved": false
93
+ }
94
+ },
95
+ {
96
+ "epoch": 6,
97
+ "train_soft_cross_entropy": 0.7924317064991704,
98
+ "validation": {
99
+ "soft_cross_entropy": 0.8612222145001094,
100
+ "accuracy": 0.7516666666666667,
101
+ "brier_soft": 0.07590873730679353,
102
+ "decisions": 600
103
+ },
104
+ "early_stopping": {
105
+ "patience": 3,
106
+ "min_epochs": 10,
107
+ "epochs_without_improvement": 2,
108
+ "best_validation_loss": 0.8523939752578735,
109
+ "improved": false
110
+ }
111
+ },
112
+ {
113
+ "epoch": 7,
114
+ "train_soft_cross_entropy": 0.7880261039733887,
115
+ "validation": {
116
+ "soft_cross_entropy": 0.8462035346031189,
117
+ "accuracy": 0.7466666666666667,
118
+ "brier_soft": 0.06817000946650903,
119
+ "decisions": 600
120
+ },
121
+ "early_stopping": {
122
+ "patience": 3,
123
+ "min_epochs": 10,
124
+ "epochs_without_improvement": 0,
125
+ "best_validation_loss": 0.8462035346031189,
126
+ "improved": true
127
+ }
128
+ },
129
+ {
130
+ "epoch": 8,
131
+ "train_soft_cross_entropy": 0.7816576555923179,
132
+ "validation": {
133
+ "soft_cross_entropy": 0.8440792632102966,
134
+ "accuracy": 0.7633333333333333,
135
+ "brier_soft": 0.06713677939027547,
136
+ "decisions": 600
137
+ },
138
+ "early_stopping": {
139
+ "patience": 3,
140
+ "min_epochs": 10,
141
+ "epochs_without_improvement": 0,
142
+ "best_validation_loss": 0.8440792632102966,
143
+ "improved": true
144
+ }
145
+ },
146
+ {
147
+ "epoch": 9,
148
+ "train_soft_cross_entropy": 0.7783506816404837,
149
+ "validation": {
150
+ "soft_cross_entropy": 0.8438087296485901,
151
+ "accuracy": 0.7683333333333333,
152
+ "brier_soft": 0.06672837336858113,
153
+ "decisions": 600
154
+ },
155
+ "early_stopping": {
156
+ "patience": 3,
157
+ "min_epochs": 10,
158
+ "epochs_without_improvement": 0,
159
+ "best_validation_loss": 0.8438087296485901,
160
+ "improved": true
161
+ }
162
+ },
163
+ {
164
+ "epoch": 10,
165
+ "train_soft_cross_entropy": 0.7754799175262451,
166
+ "validation": {
167
+ "soft_cross_entropy": 0.8468067542711893,
168
+ "accuracy": 0.755,
169
+ "brier_soft": 0.06876023932515334,
170
+ "decisions": 600
171
+ },
172
+ "early_stopping": {
173
+ "patience": 3,
174
+ "min_epochs": 10,
175
+ "epochs_without_improvement": 1,
176
+ "best_validation_loss": 0.8438087296485901,
177
+ "improved": false
178
+ }
179
+ },
180
+ {
181
+ "epoch": 11,
182
+ "train_soft_cross_entropy": 0.7734219932114637,
183
+ "validation": {
184
+ "soft_cross_entropy": 0.8521546524763107,
185
+ "accuracy": 0.7783333333333333,
186
+ "brier_soft": 0.07055049358556668,
187
+ "decisions": 600
188
+ },
189
+ "early_stopping": {
190
+ "patience": 3,
191
+ "min_epochs": 10,
192
+ "epochs_without_improvement": 2,
193
+ "best_validation_loss": 0.8438087296485901,
194
+ "improved": false
195
+ }
196
+ },
197
+ {
198
+ "epoch": 12,
199
+ "train_soft_cross_entropy": 0.7719193545094243,
200
+ "validation": {
201
+ "soft_cross_entropy": 0.842797059615453,
202
+ "accuracy": 0.7766666666666666,
203
+ "brier_soft": 0.06609537469533583,
204
+ "decisions": 600
205
+ },
206
+ "early_stopping": {
207
+ "patience": 3,
208
+ "min_epochs": 10,
209
+ "epochs_without_improvement": 0,
210
+ "best_validation_loss": 0.842797059615453,
211
+ "improved": true
212
+ }
213
+ },
214
+ {
215
+ "epoch": 13,
216
+ "train_soft_cross_entropy": 0.7722371352601934,
217
+ "validation": {
218
+ "soft_cross_entropy": 0.848033101161321,
219
+ "accuracy": 0.7633333333333333,
220
+ "brier_soft": 0.06775808438037832,
221
+ "decisions": 600
222
+ },
223
+ "early_stopping": {
224
+ "patience": 3,
225
+ "min_epochs": 10,
226
+ "epochs_without_improvement": 1,
227
+ "best_validation_loss": 0.842797059615453,
228
+ "improved": false
229
+ }
230
+ },
231
+ {
232
+ "epoch": 14,
233
+ "train_soft_cross_entropy": 0.7689926949695305,
234
+ "validation": {
235
+ "soft_cross_entropy": 0.8377540612220764,
236
+ "accuracy": 0.765,
237
+ "brier_soft": 0.06276483290052662,
238
+ "decisions": 600
239
+ },
240
+ "early_stopping": {
241
+ "patience": 3,
242
+ "min_epochs": 10,
243
+ "epochs_without_improvement": 0,
244
+ "best_validation_loss": 0.8377540612220764,
245
+ "improved": true
246
+ }
247
+ },
248
+ {
249
+ "epoch": 15,
250
+ "train_soft_cross_entropy": 0.7658979249000549,
251
+ "validation": {
252
+ "soft_cross_entropy": 0.8519912085930507,
253
+ "accuracy": 0.765,
254
+ "brier_soft": 0.07007166295622785,
255
+ "decisions": 600
256
+ },
257
+ "early_stopping": {
258
+ "patience": 3,
259
+ "min_epochs": 10,
260
+ "epochs_without_improvement": 1,
261
+ "best_validation_loss": 0.8377540612220764,
262
+ "improved": false
263
+ }
264
+ },
265
+ {
266
+ "epoch": 16,
267
+ "train_soft_cross_entropy": 0.7651108237107594,
268
+ "validation": {
269
+ "soft_cross_entropy": 0.8408213845888773,
270
+ "accuracy": 0.7633333333333333,
271
+ "brier_soft": 0.06544813718336324,
272
+ "decisions": 600
273
+ },
274
+ "early_stopping": {
275
+ "patience": 3,
276
+ "min_epochs": 10,
277
+ "epochs_without_improvement": 2,
278
+ "best_validation_loss": 0.8377540612220764,
279
+ "improved": false
280
+ }
281
+ },
282
+ {
283
+ "epoch": 17,
284
+ "train_soft_cross_entropy": 0.7817914198063038,
285
+ "validation": {
286
+ "soft_cross_entropy": 0.8468167595068614,
287
+ "accuracy": 0.7683333333333333,
288
+ "brier_soft": 0.06789324807934463,
289
+ "decisions": 600
290
+ },
291
+ "early_stopping": {
292
+ "patience": 3,
293
+ "min_epochs": 10,
294
+ "epochs_without_improvement": 3,
295
+ "best_validation_loss": 0.8377540612220764,
296
+ "improved": false
297
+ }
298
+ }
299
+ ],
300
+ "stopping": {
301
+ "reason": "early_stopping",
302
+ "epochs_completed": 17,
303
+ "patience": 3,
304
+ "min_epochs": 10,
305
+ "epochs_without_improvement": 3,
306
+ "metric": "validation.soft_cross_entropy",
307
+ "min_delta": 0.0
308
+ },
309
+ "trainable_names": [
310
+ "encoder.embeddings.interface.proj.weight",
311
+ "encoder.embeddings.interface.proj.bias"
312
+ ],
313
+ "frozen_parameters_unchanged": true
314
+ }
evidence/seed-2-history.json ADDED
@@ -0,0 +1,331 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "seed": 2,
3
+ "initial_validation": {
4
+ "soft_cross_entropy": 1.544869564374288,
5
+ "accuracy": 0.36833333333333335,
6
+ "brier_soft": 0.43051501592000324,
7
+ "decisions": 600
8
+ },
9
+ "epochs": [
10
+ {
11
+ "epoch": 1,
12
+ "train_soft_cross_entropy": 1.1849676999339351,
13
+ "validation": {
14
+ "soft_cross_entropy": 1.1535689290364584,
15
+ "accuracy": 0.4266666666666667,
16
+ "brier_soft": 0.2362905572851499,
17
+ "decisions": 600
18
+ },
19
+ "early_stopping": {
20
+ "patience": 3,
21
+ "min_epochs": 10,
22
+ "epochs_without_improvement": 0,
23
+ "best_validation_loss": 1.1535689290364584,
24
+ "improved": true
25
+ }
26
+ },
27
+ {
28
+ "epoch": 2,
29
+ "train_soft_cross_entropy": 1.1660638781830117,
30
+ "validation": {
31
+ "soft_cross_entropy": 1.1240008862813313,
32
+ "accuracy": 0.44666666666666666,
33
+ "brier_soft": 0.2163119477033615,
34
+ "decisions": 600
35
+ },
36
+ "early_stopping": {
37
+ "patience": 3,
38
+ "min_epochs": 10,
39
+ "epochs_without_improvement": 0,
40
+ "best_validation_loss": 1.1240008862813313,
41
+ "improved": true
42
+ }
43
+ },
44
+ {
45
+ "epoch": 3,
46
+ "train_soft_cross_entropy": 1.1152002820262203,
47
+ "validation": {
48
+ "soft_cross_entropy": 1.111793461640676,
49
+ "accuracy": 0.4816666666666667,
50
+ "brier_soft": 0.21397013902664186,
51
+ "decisions": 600
52
+ },
53
+ "early_stopping": {
54
+ "patience": 3,
55
+ "min_epochs": 10,
56
+ "epochs_without_improvement": 0,
57
+ "best_validation_loss": 1.111793461640676,
58
+ "improved": true
59
+ }
60
+ },
61
+ {
62
+ "epoch": 4,
63
+ "train_soft_cross_entropy": 1.0905909954177009,
64
+ "validation": {
65
+ "soft_cross_entropy": 1.0822320612271628,
66
+ "accuracy": 0.49833333333333335,
67
+ "brier_soft": 0.19861485362052916,
68
+ "decisions": 600
69
+ },
70
+ "early_stopping": {
71
+ "patience": 3,
72
+ "min_epochs": 10,
73
+ "epochs_without_improvement": 0,
74
+ "best_validation_loss": 1.0822320612271628,
75
+ "improved": true
76
+ }
77
+ },
78
+ {
79
+ "epoch": 5,
80
+ "train_soft_cross_entropy": 1.0488388820047732,
81
+ "validation": {
82
+ "soft_cross_entropy": 1.0466791025797526,
83
+ "accuracy": 0.5383333333333333,
84
+ "brier_soft": 0.18209601615866025,
85
+ "decisions": 600
86
+ },
87
+ "early_stopping": {
88
+ "patience": 3,
89
+ "min_epochs": 10,
90
+ "epochs_without_improvement": 0,
91
+ "best_validation_loss": 1.0466791025797526,
92
+ "improved": true
93
+ }
94
+ },
95
+ {
96
+ "epoch": 6,
97
+ "train_soft_cross_entropy": 1.0179926924352292,
98
+ "validation": {
99
+ "soft_cross_entropy": 1.0302869494756062,
100
+ "accuracy": 0.5633333333333334,
101
+ "brier_soft": 0.1717192947367827,
102
+ "decisions": 600
103
+ },
104
+ "early_stopping": {
105
+ "patience": 3,
106
+ "min_epochs": 10,
107
+ "epochs_without_improvement": 0,
108
+ "best_validation_loss": 1.0302869494756062,
109
+ "improved": true
110
+ }
111
+ },
112
+ {
113
+ "epoch": 7,
114
+ "train_soft_cross_entropy": 1.0011521806540313,
115
+ "validation": {
116
+ "soft_cross_entropy": 1.0306965279579163,
117
+ "accuracy": 0.5533333333333333,
118
+ "brier_soft": 0.17337830337385338,
119
+ "decisions": 600
120
+ },
121
+ "early_stopping": {
122
+ "patience": 3,
123
+ "min_epochs": 10,
124
+ "epochs_without_improvement": 1,
125
+ "best_validation_loss": 1.0302869494756062,
126
+ "improved": false
127
+ }
128
+ },
129
+ {
130
+ "epoch": 8,
131
+ "train_soft_cross_entropy": 0.9898686430189345,
132
+ "validation": {
133
+ "soft_cross_entropy": 1.0352703229586284,
134
+ "accuracy": 0.5533333333333333,
135
+ "brier_soft": 0.17590213686227799,
136
+ "decisions": 600
137
+ },
138
+ "early_stopping": {
139
+ "patience": 3,
140
+ "min_epochs": 10,
141
+ "epochs_without_improvement": 2,
142
+ "best_validation_loss": 1.0302869494756062,
143
+ "improved": false
144
+ }
145
+ },
146
+ {
147
+ "epoch": 9,
148
+ "train_soft_cross_entropy": 0.9732429999775357,
149
+ "validation": {
150
+ "soft_cross_entropy": 1.0375931040445963,
151
+ "accuracy": 0.56,
152
+ "brier_soft": 0.1798627228786548,
153
+ "decisions": 600
154
+ },
155
+ "early_stopping": {
156
+ "patience": 3,
157
+ "min_epochs": 10,
158
+ "epochs_without_improvement": 3,
159
+ "best_validation_loss": 1.0302869494756062,
160
+ "improved": false
161
+ }
162
+ },
163
+ {
164
+ "epoch": 10,
165
+ "train_soft_cross_entropy": 0.9616305458987201,
166
+ "validation": {
167
+ "soft_cross_entropy": 1.0139233843485513,
168
+ "accuracy": 0.6116666666666667,
169
+ "brier_soft": 0.16307140870640674,
170
+ "decisions": 600
171
+ },
172
+ "early_stopping": {
173
+ "patience": 3,
174
+ "min_epochs": 10,
175
+ "epochs_without_improvement": 0,
176
+ "best_validation_loss": 1.0139233843485513,
177
+ "improved": true
178
+ }
179
+ },
180
+ {
181
+ "epoch": 11,
182
+ "train_soft_cross_entropy": 0.9440543747831274,
183
+ "validation": {
184
+ "soft_cross_entropy": 1.0138032054901123,
185
+ "accuracy": 0.6166666666666667,
186
+ "brier_soft": 0.165894419302543,
187
+ "decisions": 600
188
+ },
189
+ "early_stopping": {
190
+ "patience": 3,
191
+ "min_epochs": 10,
192
+ "epochs_without_improvement": 0,
193
+ "best_validation_loss": 1.0138032054901123,
194
+ "improved": true
195
+ }
196
+ },
197
+ {
198
+ "epoch": 12,
199
+ "train_soft_cross_entropy": 0.9309376364284091,
200
+ "validation": {
201
+ "soft_cross_entropy": 0.9929217569033305,
202
+ "accuracy": 0.6266666666666667,
203
+ "brier_soft": 0.15395031906664372,
204
+ "decisions": 600
205
+ },
206
+ "early_stopping": {
207
+ "patience": 3,
208
+ "min_epochs": 10,
209
+ "epochs_without_improvement": 0,
210
+ "best_validation_loss": 0.9929217569033305,
211
+ "improved": true
212
+ }
213
+ },
214
+ {
215
+ "epoch": 13,
216
+ "train_soft_cross_entropy": 0.9144250082969666,
217
+ "validation": {
218
+ "soft_cross_entropy": 0.9964303588867187,
219
+ "accuracy": 0.6266666666666667,
220
+ "brier_soft": 0.1560174826408426,
221
+ "decisions": 600
222
+ },
223
+ "early_stopping": {
224
+ "patience": 3,
225
+ "min_epochs": 10,
226
+ "epochs_without_improvement": 1,
227
+ "best_validation_loss": 0.9929217569033305,
228
+ "improved": false
229
+ }
230
+ },
231
+ {
232
+ "epoch": 14,
233
+ "train_soft_cross_entropy": 0.9063218560925237,
234
+ "validation": {
235
+ "soft_cross_entropy": 1.005851571559906,
236
+ "accuracy": 0.6133333333333333,
237
+ "brier_soft": 0.1608497215807438,
238
+ "decisions": 600
239
+ },
240
+ "early_stopping": {
241
+ "patience": 3,
242
+ "min_epochs": 10,
243
+ "epochs_without_improvement": 2,
244
+ "best_validation_loss": 0.9929217569033305,
245
+ "improved": false
246
+ }
247
+ },
248
+ {
249
+ "epoch": 15,
250
+ "train_soft_cross_entropy": 0.9014872776137458,
251
+ "validation": {
252
+ "soft_cross_entropy": 0.9867499772707621,
253
+ "accuracy": 0.6316666666666667,
254
+ "brier_soft": 0.14927835414807003,
255
+ "decisions": 600
256
+ },
257
+ "early_stopping": {
258
+ "patience": 3,
259
+ "min_epochs": 10,
260
+ "epochs_without_improvement": 0,
261
+ "best_validation_loss": 0.9867499772707621,
262
+ "improved": true
263
+ }
264
+ },
265
+ {
266
+ "epoch": 16,
267
+ "train_soft_cross_entropy": 0.8920391595805133,
268
+ "validation": {
269
+ "soft_cross_entropy": 0.9926939264933268,
270
+ "accuracy": 0.6283333333333333,
271
+ "brier_soft": 0.15533705055713654,
272
+ "decisions": 600
273
+ },
274
+ "early_stopping": {
275
+ "patience": 3,
276
+ "min_epochs": 10,
277
+ "epochs_without_improvement": 1,
278
+ "best_validation_loss": 0.9867499772707621,
279
+ "improved": false
280
+ }
281
+ },
282
+ {
283
+ "epoch": 17,
284
+ "train_soft_cross_entropy": 0.8795346831833875,
285
+ "validation": {
286
+ "soft_cross_entropy": 1.02633438428243,
287
+ "accuracy": 0.6583333333333333,
288
+ "brier_soft": 0.16615503872434298,
289
+ "decisions": 600
290
+ },
291
+ "early_stopping": {
292
+ "patience": 3,
293
+ "min_epochs": 10,
294
+ "epochs_without_improvement": 2,
295
+ "best_validation_loss": 0.9867499772707621,
296
+ "improved": false
297
+ }
298
+ },
299
+ {
300
+ "epoch": 18,
301
+ "train_soft_cross_entropy": 0.879185232453876,
302
+ "validation": {
303
+ "soft_cross_entropy": 1.0007199597358705,
304
+ "accuracy": 0.65,
305
+ "brier_soft": 0.15312831775595745,
306
+ "decisions": 600
307
+ },
308
+ "early_stopping": {
309
+ "patience": 3,
310
+ "min_epochs": 10,
311
+ "epochs_without_improvement": 3,
312
+ "best_validation_loss": 0.9867499772707621,
313
+ "improved": false
314
+ }
315
+ }
316
+ ],
317
+ "stopping": {
318
+ "reason": "early_stopping",
319
+ "epochs_completed": 18,
320
+ "patience": 3,
321
+ "min_epochs": 10,
322
+ "epochs_without_improvement": 3,
323
+ "metric": "validation.soft_cross_entropy",
324
+ "min_delta": 0.0
325
+ },
326
+ "trainable_names": [
327
+ "encoder.embeddings.interface.proj.weight",
328
+ "encoder.embeddings.interface.proj.bias"
329
+ ],
330
+ "frozen_parameters_unchanged": true
331
+ }
evidence/seed-3-history.json ADDED
@@ -0,0 +1,484 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "seed": 3,
3
+ "initial_validation": {
4
+ "soft_cross_entropy": 1.544869564374288,
5
+ "accuracy": 0.36833333333333335,
6
+ "brier_soft": 0.43051501592000324,
7
+ "decisions": 600
8
+ },
9
+ "epochs": [
10
+ {
11
+ "epoch": 1,
12
+ "train_soft_cross_entropy": 1.1980749452555621,
13
+ "validation": {
14
+ "soft_cross_entropy": 1.16634921391805,
15
+ "accuracy": 0.4266666666666667,
16
+ "brier_soft": 0.2427163557211558,
17
+ "decisions": 600
18
+ },
19
+ "early_stopping": {
20
+ "patience": 3,
21
+ "min_epochs": 10,
22
+ "epochs_without_improvement": 0,
23
+ "best_validation_loss": 1.16634921391805,
24
+ "improved": true
25
+ }
26
+ },
27
+ {
28
+ "epoch": 2,
29
+ "train_soft_cross_entropy": 1.1646321240177862,
30
+ "validation": {
31
+ "soft_cross_entropy": 1.1217502697308859,
32
+ "accuracy": 0.4533333333333333,
33
+ "brier_soft": 0.2186164912581444,
34
+ "decisions": 600
35
+ },
36
+ "early_stopping": {
37
+ "patience": 3,
38
+ "min_epochs": 10,
39
+ "epochs_without_improvement": 0,
40
+ "best_validation_loss": 1.1217502697308859,
41
+ "improved": true
42
+ }
43
+ },
44
+ {
45
+ "epoch": 3,
46
+ "train_soft_cross_entropy": 1.1056555556367944,
47
+ "validation": {
48
+ "soft_cross_entropy": 1.093972578048706,
49
+ "accuracy": 0.505,
50
+ "brier_soft": 0.199787005285422,
51
+ "decisions": 600
52
+ },
53
+ "early_stopping": {
54
+ "patience": 3,
55
+ "min_epochs": 10,
56
+ "epochs_without_improvement": 0,
57
+ "best_validation_loss": 1.093972578048706,
58
+ "improved": true
59
+ }
60
+ },
61
+ {
62
+ "epoch": 4,
63
+ "train_soft_cross_entropy": 1.0671771099832323,
64
+ "validation": {
65
+ "soft_cross_entropy": 1.0549713468551636,
66
+ "accuracy": 0.5516666666666666,
67
+ "brier_soft": 0.18068428387244542,
68
+ "decisions": 600
69
+ },
70
+ "early_stopping": {
71
+ "patience": 3,
72
+ "min_epochs": 10,
73
+ "epochs_without_improvement": 0,
74
+ "best_validation_loss": 1.0549713468551636,
75
+ "improved": true
76
+ }
77
+ },
78
+ {
79
+ "epoch": 5,
80
+ "train_soft_cross_entropy": 1.0494003568755257,
81
+ "validation": {
82
+ "soft_cross_entropy": 1.0567963059743246,
83
+ "accuracy": 0.54,
84
+ "brier_soft": 0.18257476242880027,
85
+ "decisions": 600
86
+ },
87
+ "early_stopping": {
88
+ "patience": 3,
89
+ "min_epochs": 10,
90
+ "epochs_without_improvement": 1,
91
+ "best_validation_loss": 1.0549713468551636,
92
+ "improved": false
93
+ }
94
+ },
95
+ {
96
+ "epoch": 6,
97
+ "train_soft_cross_entropy": 1.038703513851872,
98
+ "validation": {
99
+ "soft_cross_entropy": 1.0705021890004476,
100
+ "accuracy": 0.525,
101
+ "brier_soft": 0.19120527582863966,
102
+ "decisions": 600
103
+ },
104
+ "early_stopping": {
105
+ "patience": 3,
106
+ "min_epochs": 10,
107
+ "epochs_without_improvement": 2,
108
+ "best_validation_loss": 1.0549713468551636,
109
+ "improved": false
110
+ }
111
+ },
112
+ {
113
+ "epoch": 7,
114
+ "train_soft_cross_entropy": 1.0313762316880404,
115
+ "validation": {
116
+ "soft_cross_entropy": 1.0447989773750306,
117
+ "accuracy": 0.5466666666666666,
118
+ "brier_soft": 0.17501757830381393,
119
+ "decisions": 600
120
+ },
121
+ "early_stopping": {
122
+ "patience": 3,
123
+ "min_epochs": 10,
124
+ "epochs_without_improvement": 0,
125
+ "best_validation_loss": 1.0447989773750306,
126
+ "improved": true
127
+ }
128
+ },
129
+ {
130
+ "epoch": 8,
131
+ "train_soft_cross_entropy": 1.017818791601393,
132
+ "validation": {
133
+ "soft_cross_entropy": 1.024161856174469,
134
+ "accuracy": 0.5816666666666667,
135
+ "brier_soft": 0.1630909529576699,
136
+ "decisions": 600
137
+ },
138
+ "early_stopping": {
139
+ "patience": 3,
140
+ "min_epochs": 10,
141
+ "epochs_without_improvement": 0,
142
+ "best_validation_loss": 1.024161856174469,
143
+ "improved": true
144
+ }
145
+ },
146
+ {
147
+ "epoch": 9,
148
+ "train_soft_cross_entropy": 0.9899537578335514,
149
+ "validation": {
150
+ "soft_cross_entropy": 1.0265704361597696,
151
+ "accuracy": 0.6083333333333333,
152
+ "brier_soft": 0.16808087141563496,
153
+ "decisions": 600
154
+ },
155
+ "early_stopping": {
156
+ "patience": 3,
157
+ "min_epochs": 10,
158
+ "epochs_without_improvement": 1,
159
+ "best_validation_loss": 1.024161856174469,
160
+ "improved": false
161
+ }
162
+ },
163
+ {
164
+ "epoch": 10,
165
+ "train_soft_cross_entropy": 0.9632394627288535,
166
+ "validation": {
167
+ "soft_cross_entropy": 1.0183856010437011,
168
+ "accuracy": 0.5983333333333334,
169
+ "brier_soft": 0.1608257148663203,
170
+ "decisions": 600
171
+ },
172
+ "early_stopping": {
173
+ "patience": 3,
174
+ "min_epochs": 10,
175
+ "epochs_without_improvement": 0,
176
+ "best_validation_loss": 1.0183856010437011,
177
+ "improved": true
178
+ }
179
+ },
180
+ {
181
+ "epoch": 11,
182
+ "train_soft_cross_entropy": 0.9342622081438701,
183
+ "validation": {
184
+ "soft_cross_entropy": 1.0024159566561381,
185
+ "accuracy": 0.6333333333333333,
186
+ "brier_soft": 0.14996731283764045,
187
+ "decisions": 600
188
+ },
189
+ "early_stopping": {
190
+ "patience": 3,
191
+ "min_epochs": 10,
192
+ "epochs_without_improvement": 0,
193
+ "best_validation_loss": 1.0024159566561381,
194
+ "improved": true
195
+ }
196
+ },
197
+ {
198
+ "epoch": 12,
199
+ "train_soft_cross_entropy": 0.9137524432606168,
200
+ "validation": {
201
+ "soft_cross_entropy": 0.9873138956228892,
202
+ "accuracy": 0.6516666666666666,
203
+ "brier_soft": 0.14277802929282188,
204
+ "decisions": 600
205
+ },
206
+ "early_stopping": {
207
+ "patience": 3,
208
+ "min_epochs": 10,
209
+ "epochs_without_improvement": 0,
210
+ "best_validation_loss": 0.9873138956228892,
211
+ "improved": true
212
+ }
213
+ },
214
+ {
215
+ "epoch": 13,
216
+ "train_soft_cross_entropy": 0.8881368139496556,
217
+ "validation": {
218
+ "soft_cross_entropy": 0.9626709421475729,
219
+ "accuracy": 0.6716666666666666,
220
+ "brier_soft": 0.12283751085400581,
221
+ "decisions": 600
222
+ },
223
+ "early_stopping": {
224
+ "patience": 3,
225
+ "min_epochs": 10,
226
+ "epochs_without_improvement": 0,
227
+ "best_validation_loss": 0.9626709421475729,
228
+ "improved": true
229
+ }
230
+ },
231
+ {
232
+ "epoch": 14,
233
+ "train_soft_cross_entropy": 0.865500467883216,
234
+ "validation": {
235
+ "soft_cross_entropy": 0.9580995849768321,
236
+ "accuracy": 0.685,
237
+ "brier_soft": 0.12383397127191226,
238
+ "decisions": 600
239
+ },
240
+ "early_stopping": {
241
+ "patience": 3,
242
+ "min_epochs": 10,
243
+ "epochs_without_improvement": 0,
244
+ "best_validation_loss": 0.9580995849768321,
245
+ "improved": true
246
+ }
247
+ },
248
+ {
249
+ "epoch": 15,
250
+ "train_soft_cross_entropy": 0.8549449095461104,
251
+ "validation": {
252
+ "soft_cross_entropy": 0.9462178750832876,
253
+ "accuracy": 0.6916666666666667,
254
+ "brier_soft": 0.1187205430244406,
255
+ "decisions": 600
256
+ },
257
+ "early_stopping": {
258
+ "patience": 3,
259
+ "min_epochs": 10,
260
+ "epochs_without_improvement": 0,
261
+ "best_validation_loss": 0.9462178750832876,
262
+ "improved": true
263
+ }
264
+ },
265
+ {
266
+ "epoch": 16,
267
+ "train_soft_cross_entropy": 0.8394836834183446,
268
+ "validation": {
269
+ "soft_cross_entropy": 0.9463119049866994,
270
+ "accuracy": 0.6933333333333334,
271
+ "brier_soft": 0.11875337022046248,
272
+ "decisions": 600
273
+ },
274
+ "early_stopping": {
275
+ "patience": 3,
276
+ "min_epochs": 10,
277
+ "epochs_without_improvement": 1,
278
+ "best_validation_loss": 0.9462178750832876,
279
+ "improved": false
280
+ }
281
+ },
282
+ {
283
+ "epoch": 17,
284
+ "train_soft_cross_entropy": 0.8286281617040987,
285
+ "validation": {
286
+ "soft_cross_entropy": 0.941375896135966,
287
+ "accuracy": 0.685,
288
+ "brier_soft": 0.11455494280904531,
289
+ "decisions": 600
290
+ },
291
+ "early_stopping": {
292
+ "patience": 3,
293
+ "min_epochs": 10,
294
+ "epochs_without_improvement": 0,
295
+ "best_validation_loss": 0.941375896135966,
296
+ "improved": true
297
+ }
298
+ },
299
+ {
300
+ "epoch": 18,
301
+ "train_soft_cross_entropy": 0.8222652976601212,
302
+ "validation": {
303
+ "soft_cross_entropy": 0.9331588689486185,
304
+ "accuracy": 0.6866666666666666,
305
+ "brier_soft": 0.11237604923546314,
306
+ "decisions": 600
307
+ },
308
+ "early_stopping": {
309
+ "patience": 3,
310
+ "min_epochs": 10,
311
+ "epochs_without_improvement": 0,
312
+ "best_validation_loss": 0.9331588689486185,
313
+ "improved": true
314
+ }
315
+ },
316
+ {
317
+ "epoch": 19,
318
+ "train_soft_cross_entropy": 0.8160834203826056,
319
+ "validation": {
320
+ "soft_cross_entropy": 0.9387085942427317,
321
+ "accuracy": 0.6916666666666667,
322
+ "brier_soft": 0.11558652246991793,
323
+ "decisions": 600
324
+ },
325
+ "early_stopping": {
326
+ "patience": 3,
327
+ "min_epochs": 10,
328
+ "epochs_without_improvement": 1,
329
+ "best_validation_loss": 0.9331588689486185,
330
+ "improved": false
331
+ }
332
+ },
333
+ {
334
+ "epoch": 20,
335
+ "train_soft_cross_entropy": 0.8087538785846146,
336
+ "validation": {
337
+ "soft_cross_entropy": 0.9408066284656524,
338
+ "accuracy": 0.6783333333333333,
339
+ "brier_soft": 0.11930533437679211,
340
+ "decisions": 600
341
+ },
342
+ "early_stopping": {
343
+ "patience": 3,
344
+ "min_epochs": 10,
345
+ "epochs_without_improvement": 2,
346
+ "best_validation_loss": 0.9331588689486185,
347
+ "improved": false
348
+ }
349
+ },
350
+ {
351
+ "epoch": 21,
352
+ "train_soft_cross_entropy": 0.8048860243956248,
353
+ "validation": {
354
+ "soft_cross_entropy": 0.9236933688322703,
355
+ "accuracy": 0.7016666666666667,
356
+ "brier_soft": 0.10792215374608835,
357
+ "decisions": 600
358
+ },
359
+ "early_stopping": {
360
+ "patience": 3,
361
+ "min_epochs": 10,
362
+ "epochs_without_improvement": 0,
363
+ "best_validation_loss": 0.9236933688322703,
364
+ "improved": true
365
+ }
366
+ },
367
+ {
368
+ "epoch": 22,
369
+ "train_soft_cross_entropy": 0.8001213801790167,
370
+ "validation": {
371
+ "soft_cross_entropy": 0.9347231006622314,
372
+ "accuracy": 0.6983333333333334,
373
+ "brier_soft": 0.11313647958139579,
374
+ "decisions": 600
375
+ },
376
+ "early_stopping": {
377
+ "patience": 3,
378
+ "min_epochs": 10,
379
+ "epochs_without_improvement": 1,
380
+ "best_validation_loss": 0.9236933688322703,
381
+ "improved": false
382
+ }
383
+ },
384
+ {
385
+ "epoch": 23,
386
+ "train_soft_cross_entropy": 0.7940637088263476,
387
+ "validation": {
388
+ "soft_cross_entropy": 0.9299098292986552,
389
+ "accuracy": 0.7033333333333334,
390
+ "brier_soft": 0.11475709093113741,
391
+ "decisions": 600
392
+ },
393
+ "early_stopping": {
394
+ "patience": 3,
395
+ "min_epochs": 10,
396
+ "epochs_without_improvement": 2,
397
+ "best_validation_loss": 0.9236933688322703,
398
+ "improved": false
399
+ }
400
+ },
401
+ {
402
+ "epoch": 24,
403
+ "train_soft_cross_entropy": 0.7884197377717054,
404
+ "validation": {
405
+ "soft_cross_entropy": 0.9105557099978129,
406
+ "accuracy": 0.715,
407
+ "brier_soft": 0.1013037375236551,
408
+ "decisions": 600
409
+ },
410
+ "early_stopping": {
411
+ "patience": 3,
412
+ "min_epochs": 10,
413
+ "epochs_without_improvement": 0,
414
+ "best_validation_loss": 0.9105557099978129,
415
+ "improved": true
416
+ }
417
+ },
418
+ {
419
+ "epoch": 25,
420
+ "train_soft_cross_entropy": 0.784954049454795,
421
+ "validation": {
422
+ "soft_cross_entropy": 0.923290346066157,
423
+ "accuracy": 0.71,
424
+ "brier_soft": 0.10530036373684803,
425
+ "decisions": 600
426
+ },
427
+ "early_stopping": {
428
+ "patience": 3,
429
+ "min_epochs": 10,
430
+ "epochs_without_improvement": 1,
431
+ "best_validation_loss": 0.9105557099978129,
432
+ "improved": false
433
+ }
434
+ },
435
+ {
436
+ "epoch": 26,
437
+ "train_soft_cross_entropy": 0.7828840752442677,
438
+ "validation": {
439
+ "soft_cross_entropy": 0.9299345835049947,
440
+ "accuracy": 0.7066666666666667,
441
+ "brier_soft": 0.11173659774164359,
442
+ "decisions": 600
443
+ },
444
+ "early_stopping": {
445
+ "patience": 3,
446
+ "min_epochs": 10,
447
+ "epochs_without_improvement": 2,
448
+ "best_validation_loss": 0.9105557099978129,
449
+ "improved": false
450
+ }
451
+ },
452
+ {
453
+ "epoch": 27,
454
+ "train_soft_cross_entropy": 0.7790651953661883,
455
+ "validation": {
456
+ "soft_cross_entropy": 0.9137676254908244,
457
+ "accuracy": 0.7216666666666667,
458
+ "brier_soft": 0.10172271362195412,
459
+ "decisions": 600
460
+ },
461
+ "early_stopping": {
462
+ "patience": 3,
463
+ "min_epochs": 10,
464
+ "epochs_without_improvement": 3,
465
+ "best_validation_loss": 0.9105557099978129,
466
+ "improved": false
467
+ }
468
+ }
469
+ ],
470
+ "stopping": {
471
+ "reason": "early_stopping",
472
+ "epochs_completed": 27,
473
+ "patience": 3,
474
+ "min_epochs": 10,
475
+ "epochs_without_improvement": 3,
476
+ "metric": "validation.soft_cross_entropy",
477
+ "min_delta": 0.0
478
+ },
479
+ "trainable_names": [
480
+ "encoder.embeddings.interface.proj.weight",
481
+ "encoder.embeddings.interface.proj.bias"
482
+ ],
483
+ "frozen_parameters_unchanged": true
484
+ }
evidence/seed-4-history.json ADDED
@@ -0,0 +1,195 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "seed": 4,
3
+ "initial_validation": {
4
+ "soft_cross_entropy": 1.544869564374288,
5
+ "accuracy": 0.36833333333333335,
6
+ "brier_soft": 0.43051501592000324,
7
+ "decisions": 600
8
+ },
9
+ "epochs": [
10
+ {
11
+ "epoch": 1,
12
+ "train_soft_cross_entropy": 1.002756692568461,
13
+ "validation": {
14
+ "soft_cross_entropy": 0.9602992224693299,
15
+ "accuracy": 0.6783333333333333,
16
+ "brier_soft": 0.1296700432151556,
17
+ "decisions": 600
18
+ },
19
+ "early_stopping": {
20
+ "patience": 3,
21
+ "min_epochs": 10,
22
+ "epochs_without_improvement": 0,
23
+ "best_validation_loss": 0.9602992224693299,
24
+ "improved": true
25
+ }
26
+ },
27
+ {
28
+ "epoch": 2,
29
+ "train_soft_cross_entropy": 0.8854288237624698,
30
+ "validation": {
31
+ "soft_cross_entropy": 0.8732180511951446,
32
+ "accuracy": 0.7316666666666667,
33
+ "brier_soft": 0.08001670623819034,
34
+ "decisions": 600
35
+ },
36
+ "early_stopping": {
37
+ "patience": 3,
38
+ "min_epochs": 10,
39
+ "epochs_without_improvement": 0,
40
+ "best_validation_loss": 0.8732180511951446,
41
+ "improved": true
42
+ }
43
+ },
44
+ {
45
+ "epoch": 3,
46
+ "train_soft_cross_entropy": 0.8375967553809837,
47
+ "validation": {
48
+ "soft_cross_entropy": 0.8618760740756989,
49
+ "accuracy": 0.7566666666666667,
50
+ "brier_soft": 0.07448753335823616,
51
+ "decisions": 600
52
+ },
53
+ "early_stopping": {
54
+ "patience": 3,
55
+ "min_epochs": 10,
56
+ "epochs_without_improvement": 0,
57
+ "best_validation_loss": 0.8618760740756989,
58
+ "improved": true
59
+ }
60
+ },
61
+ {
62
+ "epoch": 4,
63
+ "train_soft_cross_entropy": 0.8120107175244226,
64
+ "validation": {
65
+ "soft_cross_entropy": 0.8502214312553406,
66
+ "accuracy": 0.7683333333333333,
67
+ "brier_soft": 0.06676056392490864,
68
+ "decisions": 600
69
+ },
70
+ "early_stopping": {
71
+ "patience": 3,
72
+ "min_epochs": 10,
73
+ "epochs_without_improvement": 0,
74
+ "best_validation_loss": 0.8502214312553406,
75
+ "improved": true
76
+ }
77
+ },
78
+ {
79
+ "epoch": 5,
80
+ "train_soft_cross_entropy": 0.7984476970301734,
81
+ "validation": {
82
+ "soft_cross_entropy": 0.8527251331011454,
83
+ "accuracy": 0.765,
84
+ "brier_soft": 0.07089023986210426,
85
+ "decisions": 600
86
+ },
87
+ "early_stopping": {
88
+ "patience": 3,
89
+ "min_epochs": 10,
90
+ "epochs_without_improvement": 1,
91
+ "best_validation_loss": 0.8502214312553406,
92
+ "improved": false
93
+ }
94
+ },
95
+ {
96
+ "epoch": 6,
97
+ "train_soft_cross_entropy": 0.7902782445042221,
98
+ "validation": {
99
+ "soft_cross_entropy": 0.8496846203009287,
100
+ "accuracy": 0.765,
101
+ "brier_soft": 0.06887457605761786,
102
+ "decisions": 600
103
+ },
104
+ "early_stopping": {
105
+ "patience": 3,
106
+ "min_epochs": 10,
107
+ "epochs_without_improvement": 0,
108
+ "best_validation_loss": 0.8496846203009287,
109
+ "improved": true
110
+ }
111
+ },
112
+ {
113
+ "epoch": 7,
114
+ "train_soft_cross_entropy": 0.7847388537283297,
115
+ "validation": {
116
+ "soft_cross_entropy": 0.8448993345101674,
117
+ "accuracy": 0.7533333333333333,
118
+ "brier_soft": 0.0668264798882107,
119
+ "decisions": 600
120
+ },
121
+ "early_stopping": {
122
+ "patience": 3,
123
+ "min_epochs": 10,
124
+ "epochs_without_improvement": 0,
125
+ "best_validation_loss": 0.8448993345101674,
126
+ "improved": true
127
+ }
128
+ },
129
+ {
130
+ "epoch": 8,
131
+ "train_soft_cross_entropy": 0.780297756018462,
132
+ "validation": {
133
+ "soft_cross_entropy": 0.8471182523171107,
134
+ "accuracy": 0.7683333333333333,
135
+ "brier_soft": 0.06819427679603299,
136
+ "decisions": 600
137
+ },
138
+ "early_stopping": {
139
+ "patience": 3,
140
+ "min_epochs": 10,
141
+ "epochs_without_improvement": 1,
142
+ "best_validation_loss": 0.8448993345101674,
143
+ "improved": false
144
+ }
145
+ },
146
+ {
147
+ "epoch": 9,
148
+ "train_soft_cross_entropy": 0.7801684511590887,
149
+ "validation": {
150
+ "soft_cross_entropy": 0.8466147363185883,
151
+ "accuracy": 0.775,
152
+ "brier_soft": 0.06677873468647401,
153
+ "decisions": 600
154
+ },
155
+ "early_stopping": {
156
+ "patience": 3,
157
+ "min_epochs": 10,
158
+ "epochs_without_improvement": 2,
159
+ "best_validation_loss": 0.8448993345101674,
160
+ "improved": false
161
+ }
162
+ },
163
+ {
164
+ "epoch": 10,
165
+ "train_soft_cross_entropy": 0.7775638208565888,
166
+ "validation": {
167
+ "soft_cross_entropy": 0.8455785755316416,
168
+ "accuracy": 0.78,
169
+ "brier_soft": 0.06686171248555184,
170
+ "decisions": 600
171
+ },
172
+ "early_stopping": {
173
+ "patience": 3,
174
+ "min_epochs": 10,
175
+ "epochs_without_improvement": 3,
176
+ "best_validation_loss": 0.8448993345101674,
177
+ "improved": false
178
+ }
179
+ }
180
+ ],
181
+ "stopping": {
182
+ "reason": "early_stopping",
183
+ "epochs_completed": 10,
184
+ "patience": 3,
185
+ "min_epochs": 10,
186
+ "epochs_without_improvement": 3,
187
+ "metric": "validation.soft_cross_entropy",
188
+ "min_delta": 0.0
189
+ },
190
+ "trainable_names": [
191
+ "encoder.embeddings.interface.proj.weight",
192
+ "encoder.embeddings.interface.proj.bias"
193
+ ],
194
+ "frozen_parameters_unchanged": true
195
+ }
export-verification.json ADDED
@@ -0,0 +1,42 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "passed": true,
3
+ "candidate_seed": 1,
4
+ "selected_epoch": 14,
5
+ "cases": 3516,
6
+ "decisions": 5116,
7
+ "benchmark_cases_sha256": "10fb671245cdac1ee09e9c4d2b9b09829fbf1abd06e0954fddee0b9ec8519854",
8
+ "checkpoint_sha256": "e5ec5e1d494509f63960b6949af15be8390442271805ba45d796a1911b5c239a",
9
+ "all_answer_objects_exact": true,
10
+ "all_suite_metrics_exact": true,
11
+ "bundled_code_imported_from_outside_workspace": true,
12
+ "local_model_card_metadata_valid": true,
13
+ "device": "cuda:1",
14
+ "model": {
15
+ "gpu_name": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
16
+ "dtype": "torch.bfloat16",
17
+ "torch_version": "2.13.0+cu130",
18
+ "reproducibility": {
19
+ "deterministic_algorithms": true,
20
+ "warn_only": false,
21
+ "cublas_workspace_config": ":4096:8",
22
+ "python_hash_seed": "0",
23
+ "cudnn_benchmark": false,
24
+ "cudnn_deterministic": true,
25
+ "cudnn_allow_tf32": false,
26
+ "float32_matmul_precision": "highest",
27
+ "cudnn_sdp_enabled": false,
28
+ "flash_sdp_enabled": true,
29
+ "mem_efficient_sdp_enabled": true,
30
+ "math_sdp_enabled": true,
31
+ "cuda_version": "13.0",
32
+ "cudnn_version": 92000,
33
+ "scope": "fixed hardware/runtime; cross-platform bitwise agreement is not promised"
34
+ },
35
+ "portable_adapter": {
36
+ "base_model": "convaiinnovations/laya",
37
+ "base_revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851",
38
+ "seed": 1,
39
+ "selected_epoch": 14
40
+ }
41
+ }
42
+ }
licenses/jev-benchmarks-LICENSE ADDED
@@ -0,0 +1,201 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction,
10
+ and distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by
13
+ the copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all
16
+ other entities that control, are controlled by, or are under common
17
+ control with that entity. For the purposes of this definition,
18
+ "control" means (i) the power, direct or indirect, to cause the
19
+ direction or management of such entity, whether by contract or
20
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21
+ outstanding shares, or (iii) beneficial ownership of such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity
24
+ exercising permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation
28
+ source, and configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical
31
+ transformation or translation of a Source form, including but
32
+ not limited to compiled object code, generated documentation,
33
+ and conversions to other media types.
34
+
35
+ "Work" shall mean the work of authorship, whether in Source or
36
+ Object form, made available under the License, as indicated by a
37
+ copyright notice that is included in or attached to the work
38
+ (an example is provided in the Appendix below).
39
+
40
+ "Derivative Works" shall mean any work, whether in Source or Object
41
+ form, that is based on (or derived from) the Work and for which the
42
+ editorial revisions, annotations, elaborations, or other modifications
43
+ represent, as a whole, an original work of authorship. For the purposes
44
+ of this License, Derivative Works shall not include works that remain
45
+ separable from, or merely link (or bind by name) to the interfaces of,
46
+ the Work and Derivative Works thereof.
47
+
48
+ "Contribution" shall mean any work of authorship, including
49
+ the original version of the Work and any modifications or additions
50
+ to that Work or Derivative Works thereof, that is intentionally
51
+ submitted to Licensor for inclusion in the Work by the copyright owner
52
+ or by an individual or Legal Entity authorized to submit on behalf of
53
+ the copyright owner. For the purposes of this definition, "submitted"
54
+ means any form of electronic, verbal, or written communication sent
55
+ to the Licensor or its representatives, including but not limited to
56
+ communication on electronic mailing lists, source code control systems,
57
+ and issue tracking systems that are managed by, or on behalf of, the
58
+ Licensor for the purpose of discussing and improving the Work, but
59
+ excluding communication that is conspicuously marked or otherwise
60
+ designated in writing by the copyright owner as "Not a Contribution."
61
+
62
+ "Contributor" shall mean Licensor and any individual or Legal Entity
63
+ on behalf of whom a Contribution has been received by Licensor and
64
+ subsequently incorporated within the Work.
65
+
66
+ 2. Grant of Copyright License. Subject to the terms and conditions of
67
+ this License, each Contributor hereby grants to You a perpetual,
68
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69
+ copyright license to reproduce, prepare Derivative Works of,
70
+ publicly display, publicly perform, sublicense, and distribute the
71
+ Work and such Derivative Works in Source or Object form.
72
+
73
+ 3. Grant of Patent License. Subject to the terms and conditions of
74
+ this License, each Contributor hereby grants to You a perpetual,
75
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76
+ (except as stated in this section) patent license to make, have made,
77
+ use, offer to sell, sell, import, and otherwise transfer the Work,
78
+ where such license applies only to those patent claims licensable
79
+ by such Contributor that are necessarily infringed by their
80
+ Contribution(s) alone or by combination of their Contribution(s)
81
+ with the Work to which such Contribution(s) was submitted. If You
82
+ institute patent litigation against any entity (including a
83
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84
+ or a Contribution incorporated within the Work constitutes direct
85
+ or contributory patent infringement, then any patent licenses
86
+ granted to You under this License for that Work shall terminate
87
+ as of the date such litigation is filed.
88
+
89
+ 4. Redistribution. You may reproduce and distribute copies of the
90
+ Work or Derivative Works thereof in any medium, with or without
91
+ modifications, and in Source or Object form, provided that You
92
+ meet the following conditions:
93
+
94
+ (a) You must give any other recipients of the Work or
95
+ Derivative Works a copy of this License; and
96
+
97
+ (b) You must cause any modified files to carry prominent notices
98
+ stating that You changed the files; and
99
+
100
+ (c) You must retain, in the Source form of any Derivative Works
101
+ that You distribute, all copyright, patent, trademark, and
102
+ attribution notices from the Source form of the Work,
103
+ excluding those notices that do not pertain to any part of
104
+ the Derivative Works; and
105
+
106
+ (d) If the Work includes a "NOTICE" text file as part of its
107
+ distribution, then any Derivative Works that You distribute must
108
+ include a readable copy of the attribution notices contained
109
+ within such NOTICE file, excluding those notices that do not
110
+ pertain to any part of the Derivative Works, in at least one
111
+ of the following places: within a NOTICE text file distributed
112
+ as part of the Derivative Works; within the Source form or
113
+ documentation, if provided along with the Derivative Works; or,
114
+ within a display generated by the Derivative Works, if and
115
+ wherever such third-party notices normally appear. The contents
116
+ of the NOTICE file are for informational purposes only and
117
+ do not modify the License. You may add Your own attribution
118
+ notices within Derivative Works that You distribute, alongside
119
+ or as an addendum to the NOTICE text from the Work, provided
120
+ that such additional attribution notices cannot be construed
121
+ as modifying the License.
122
+
123
+ You may add Your own copyright statement to Your modifications and
124
+ may provide additional or different license terms and conditions
125
+ for use, reproduction, or distribution of Your modifications, or
126
+ for any such Derivative Works as a whole, provided Your use,
127
+ reproduction, and distribution of the Work otherwise complies with
128
+ the conditions stated in this License.
129
+
130
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131
+ any Contribution intentionally submitted for inclusion in the Work
132
+ by You to the Licensor shall be under the terms and conditions of
133
+ this License, without any additional terms or conditions.
134
+ Notwithstanding the above, nothing herein shall supersede or modify
135
+ the terms of any separate license agreement you may have executed
136
+ with Licensor regarding such Contributions.
137
+
138
+ 6. Trademarks. This License does not grant permission to use the trade
139
+ names, trademarks, service marks, or product names of the Licensor,
140
+ except as required for reasonable and customary use in describing the
141
+ origin of the Work and reproducing the content of the NOTICE file.
142
+
143
+ 7. Disclaimer of Warranty. Unless required by applicable law or
144
+ agreed to in writing, Licensor provides the Work (and each
145
+ Contributor provides its Contributions) on an "AS IS" BASIS,
146
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147
+ implied, including, without limitation, any warranties or conditions
148
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149
+ PARTICULAR PURPOSE. You are solely responsible for determining the
150
+ appropriateness of using or redistributing the Work and assume any
151
+ risks associated with Your exercise of permissions under this License.
152
+
153
+ 8. Limitation of Liability. In no event and under no legal theory,
154
+ whether in tort (including negligence), contract, or otherwise,
155
+ unless required by applicable law (such as deliberate and grossly
156
+ negligent acts) or agreed to in writing, shall any Contributor be
157
+ liable to You for damages, including any direct, indirect, special,
158
+ incidental, or consequential damages of any character arising as a
159
+ result of this License or out of the use or inability to use the
160
+ Work (including but not limited to damages for loss of goodwill,
161
+ work stoppage, computer failure or malfunction, or any and all
162
+ other commercial damages or losses), even if such Contributor
163
+ has been advised of the possibility of such damages.
164
+
165
+ 9. Accepting Warranty or Additional Liability. While redistributing
166
+ the Work or Derivative Works thereof, You may choose to offer,
167
+ and charge a fee for, acceptance of support, warranty, indemnity,
168
+ or other liability obligations and/or rights consistent with this
169
+ License. However, in accepting such obligations, You may act only
170
+ on Your own behalf and on Your sole responsibility, not on behalf
171
+ of any other Contributor, and only if You agree to indemnify,
172
+ defend, and hold each Contributor harmless for any liability
173
+ incurred by, or claims asserted against, such Contributor by reason
174
+ of your accepting any such warranty or additional liability.
175
+
176
+ END OF TERMS AND CONDITIONS
177
+
178
+ APPENDIX: How to apply the Apache License to your work.
179
+
180
+ To apply the Apache License to your work, attach the following
181
+ boilerplate notice, with the fields enclosed by brackets "[]"
182
+ replaced with your own identifying information. (Don't include
183
+ the brackets!) The text should be enclosed in the appropriate
184
+ comment syntax for the file format. We also recommend that a
185
+ file or class name and description of purpose be included on the
186
+ same "printed page" as the copyright notice for easier
187
+ identification within third-party archives.
188
+
189
+ Copyright [yyyy] [name of copyright owner]
190
+
191
+ Licensed under the Apache License, Version 2.0 (the "License");
192
+ you may not use this file except in compliance with the License.
193
+ You may obtain a copy of the License at
194
+
195
+ http://www.apache.org/licenses/LICENSE-2.0
196
+
197
+ Unless required by applicable law or agreed to in writing, software
198
+ distributed under the License is distributed on an "AS IS" BASIS,
199
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
200
+ See the License for the specific language governing permissions and
201
+ limitations under the License.
licenses/laya-LICENSE ADDED
@@ -0,0 +1,176 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction,
10
+ and distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by
13
+ the copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all
16
+ other entities that control, are controlled by, or are under common
17
+ control with that entity. For the purposes of this definition,
18
+ "control" means (i) the power, direct or indirect, to cause the
19
+ direction or management of such entity, whether by contract or
20
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21
+ outstanding shares, or (iii) beneficial ownership of such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity
24
+ exercising permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation
28
+ source, and configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical
31
+ transformation or translation of a Source form, including but
32
+ not limited to compiled object code, generated documentation,
33
+ and conversions to other media types.
34
+
35
+ "Work" shall mean the work of authorship, whether in Source or
36
+ Object form, made available under the License, as indicated by a
37
+ copyright notice that is included in or attached to the work
38
+ (an example is provided in the Appendix below).
39
+
40
+ "Derivative Works" shall mean any work, whether in Source or Object
41
+ form, that is based on (or derived from) the Work and for which the
42
+ editorial revisions, annotations, elaborations, or other modifications
43
+ represent, as a whole, an original work of authorship. For the purposes
44
+ of this License, Derivative Works shall not include works that remain
45
+ separable from, or merely link (or bind by name) to the interfaces of,
46
+ the Work and Derivative Works thereof.
47
+
48
+ "Contribution" shall mean any work of authorship, including
49
+ the original version of the Work and any modifications or additions
50
+ to that Work or Derivative Works thereof, that is intentionally
51
+ submitted to Licensor for inclusion in the Work by the copyright owner
52
+ or by an individual or Legal Entity authorized to submit on behalf of
53
+ the copyright owner. For the purposes of this definition, "submitted"
54
+ means any form of electronic, verbal, or written communication sent
55
+ to the Licensor or its representatives, including but not limited to
56
+ communication on electronic mailing lists, source code control systems,
57
+ and issue tracking systems that are managed by, or on behalf of, the
58
+ Licensor for the purpose of discussing and improving the Work, but
59
+ excluding communication that is conspicuously marked or otherwise
60
+ designated in writing by the copyright owner as "Not a Contribution."
61
+
62
+ "Contributor" shall mean Licensor and any individual or Legal Entity
63
+ on behalf of whom a Contribution has been received by Licensor and
64
+ subsequently incorporated within the Work.
65
+
66
+ 2. Grant of Copyright License. Subject to the terms and conditions of
67
+ this License, each Contributor hereby grants to You a perpetual,
68
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69
+ copyright license to reproduce, prepare Derivative Works of,
70
+ publicly display, publicly perform, sublicense, and distribute the
71
+ Work and such Derivative Works in Source or Object form.
72
+
73
+ 3. Grant of Patent License. Subject to the terms and conditions of
74
+ this License, each Contributor hereby grants to You a perpetual,
75
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76
+ (except as stated in this section) patent license to make, have made,
77
+ use, offer to sell, sell, import, and otherwise transfer the Work,
78
+ where such license applies only to those patent claims licensable
79
+ by such Contributor that are necessarily infringed by their
80
+ Contribution(s) alone or by combination of their Contribution(s)
81
+ with the Work to which such Contribution(s) was submitted. If You
82
+ institute patent litigation against any entity (including a
83
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84
+ or a Contribution incorporated within the Work constitutes direct
85
+ or contributory patent infringement, then any patent licenses
86
+ granted to You under this License for that Work shall terminate
87
+ as of the date such litigation is filed.
88
+
89
+ 4. Redistribution. You may reproduce and distribute copies of the
90
+ Work or Derivative Works thereof in any medium, with or without
91
+ modifications, and in Source or Object form, provided that You
92
+ meet the following conditions:
93
+
94
+ (a) You must give any other recipients of the Work or
95
+ Derivative Works a copy of this License; and
96
+
97
+ (b) You must cause any modified files to carry prominent notices
98
+ stating that You changed the files; and
99
+
100
+ (c) You must retain, in the Source form of any Derivative Works
101
+ that You distribute, all copyright, patent, trademark, and
102
+ attribution notices from the Source form of the Work,
103
+ excluding those notices that do not pertain to any part of
104
+ the Derivative Works; and
105
+
106
+ (d) If the Work includes a "NOTICE" text file as part of its
107
+ distribution, then any Derivative Works that You distribute must
108
+ include a readable copy of the attribution notices contained
109
+ within such NOTICE file, excluding those notices that do not
110
+ pertain to any part of the Derivative Works, in at least one
111
+ of the following places: within a NOTICE text file distributed
112
+ as part of the Derivative Works; within the Source form or
113
+ documentation, if provided along with the Derivative Works; or,
114
+ within a display generated by the Derivative Works, if and
115
+ wherever such third-party notices normally appear. The contents
116
+ of the NOTICE file are for informational purposes only and
117
+ do not modify the License. You may add Your own attribution
118
+ notices within Derivative Works that You distribute, alongside
119
+ or as an addendum to the NOTICE text from the Work, provided
120
+ that such additional attribution notices cannot be construed
121
+ as modifying the License.
122
+
123
+ You may add Your own copyright statement to Your modifications and
124
+ may provide additional or different license terms and conditions
125
+ for use, reproduction, or distribution of Your modifications, or
126
+ for any such Derivative Works as a whole, provided Your use,
127
+ reproduction, and distribution of the Work otherwise complies with
128
+ the conditions stated in this License.
129
+
130
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131
+ any Contribution intentionally submitted for inclusion in the Work
132
+ by You to the Licensor shall be under the terms and conditions of
133
+ this License, without any additional terms or conditions.
134
+ Notwithstanding the above, nothing herein shall supersede or modify
135
+ the terms of any separate license agreement you may have executed
136
+ with Licensor regarding such Contributions.
137
+
138
+ 6. Trademarks. This License does not grant permission to use the trade
139
+ names, trademarks, service marks, or product names of the Licensor,
140
+ except as required for reasonable and customary use in describing the
141
+ origin of the Work and reproducing the content of the NOTICE file.
142
+
143
+ 7. Disclaimer of Warranty. Unless required by applicable law or
144
+ agreed to in writing, Licensor provides the Work (and each
145
+ Contributor provides its Contributions) on an "AS IS" BASIS,
146
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147
+ implied, including, without limitation, any warranties or conditions
148
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149
+ PARTICULAR PURPOSE. You are solely responsible for determining the
150
+ appropriateness of using or redistributing the Work and assume any
151
+ risks associated with Your exercise of permissions under this License.
152
+
153
+ 8. Limitation of Liability. In no event and under no legal theory,
154
+ whether in tort (including negligence), contract, or otherwise,
155
+ unless required by applicable law (such as deliberate and grossly
156
+ negligent acts) or agreed to in writing, shall any Contributor be
157
+ liable to You for damages, including any direct, indirect, special,
158
+ incidental, or consequential damages of any character arising as a
159
+ result of this License or out of the use or inability to use the
160
+ Work (including but not limited to damages for loss of goodwill,
161
+ work stoppage, computer failure or malfunction, or any and all
162
+ other commercial damages or losses), even if such Contributor
163
+ has been advised of the possibility of such damages.
164
+
165
+ 9. Accepting Warranty or Additional Liability. While redistributing
166
+ the Work or Derivative Works thereof, You may choose to offer,
167
+ and charge a fee for, acceptance of support, warranty, indemnity,
168
+ or other liability obligations and/or rights consistent with this
169
+ License. However, in accepting such obligations, You may act only
170
+ on Your own behalf and on Your sole responsibility, not on behalf
171
+ of any other Contributor, and only if You agree to indemnify,
172
+ defend, and hold each Contributor harmless for any liability
173
+ incurred by, or claims asserted against, such Contributor by reason
174
+ of your accepting any such warranty or additional liability.
175
+
176
+ END OF TERMS AND CONDITIONS
load_adapter.py ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Local entry point for the bundled, pinned Laya input adapter."""
2
+
3
+ from pathlib import Path
4
+ import sys
5
+
6
+ PACKAGE = Path(__file__).resolve().parent
7
+ if str(PACKAGE) not in sys.path:
8
+ sys.path.insert(0, str(PACKAGE))
9
+
10
+ from ariadne_bench.portable import load_release # noqa: E402
11
+
12
+
13
+ def load(directory=PACKAGE, **options):
14
+ return load_release(directory, **options)
metrics.json ADDED
@@ -0,0 +1,984 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "baseline": {
3
+ "typed-decisions": 0.3625,
4
+ "ag-news": 0.9483333333333334,
5
+ "boolq": 0.83,
6
+ "emotion": 0.5733333333333334,
7
+ "prompt-injections": 0.6896551724137931,
8
+ "sst5": 0.37,
9
+ "massive-intent.en": 0.7866666666666666,
10
+ "xnli.en": 0.86
11
+ },
12
+ "runs": {
13
+ "0": {
14
+ "seed": 0,
15
+ "selected_epoch": 20,
16
+ "elapsed_s": 1038.9482859019772,
17
+ "initial_validation": {
18
+ "soft_cross_entropy": 1.544869564374288,
19
+ "accuracy": 0.36833333333333335,
20
+ "brier_soft": 0.43051501592000324,
21
+ "decisions": 600
22
+ },
23
+ "selected_validation": {
24
+ "soft_cross_entropy": 1.0292588464419048,
25
+ "accuracy": 0.57,
26
+ "brier_soft": 0.16918850486477216,
27
+ "decisions": 600
28
+ },
29
+ "stopping": {
30
+ "reason": "early_stopping",
31
+ "epochs_completed": 23,
32
+ "patience": 3,
33
+ "min_epochs": 10,
34
+ "epochs_without_improvement": 3,
35
+ "metric": "validation.soft_cross_entropy",
36
+ "min_delta": 0.0
37
+ },
38
+ "frozen_parameters_unchanged": true,
39
+ "suites": {
40
+ "typed-decisions": {
41
+ "attempted": 2000,
42
+ "valid": 2000,
43
+ "failed": 0,
44
+ "coverage": 1.0,
45
+ "accuracy_all": 0.61,
46
+ "accuracy_valid": 0.61,
47
+ "ece_top_label": 0.15303136435282014,
48
+ "mean_confidence": 0.4574502381919959,
49
+ "brier_hard": 0.549798004153015,
50
+ "brier_hard_n": 2000,
51
+ "nll_hard": 0.9521041848093966,
52
+ "nll_hard_n": 2000,
53
+ "zero_probability_gold": 0.0,
54
+ "zero_probability_gold_n": 2000,
55
+ "soft_accuracy": 0.3883819759829014,
56
+ "soft_accuracy_n": 2000,
57
+ "brier_soft": 0.15223049605383382,
58
+ "brier_soft_n": 2000,
59
+ "kl_gold_to_prediction": 0.2712886346630264,
60
+ "kl_gold_to_prediction_n": 2000,
61
+ "total_variation": 0.2797391333751494,
62
+ "total_variation_n": 2000,
63
+ "score_mae": 0.45095113683604626,
64
+ "score_mae_n": 800,
65
+ "within_one_level": 0.915,
66
+ "within_one_level_n": 800,
67
+ "macro_f1": 0.4150680787086972
68
+ },
69
+ "ag-news": {
70
+ "attempted": 600,
71
+ "valid": 600,
72
+ "failed": 0,
73
+ "coverage": 1.0,
74
+ "accuracy_all": 0.31333333333333335,
75
+ "accuracy_valid": 0.31333333333333335,
76
+ "ece_top_label": 0.025934925132124264,
77
+ "mean_confidence": 0.28915707486787573,
78
+ "brier_hard": 0.7441339007069468,
79
+ "brier_hard_n": 600,
80
+ "nll_hard": 1.3750333163467814,
81
+ "nll_hard_n": 600,
82
+ "zero_probability_gold": 0.0,
83
+ "zero_probability_gold_n": 600,
84
+ "macro_f1": 0.29084523228646375
85
+ },
86
+ "emotion": {
87
+ "attempted": 600,
88
+ "valid": 600,
89
+ "failed": 0,
90
+ "coverage": 1.0,
91
+ "accuracy_all": 0.195,
92
+ "accuracy_valid": 0.195,
93
+ "ece_top_label": 0.08537782470783518,
94
+ "mean_confidence": 0.2764666886168944,
95
+ "brier_hard": 0.8415061960524678,
96
+ "brier_hard_n": 600,
97
+ "nll_hard": 1.810555559185162,
98
+ "nll_hard_n": 600,
99
+ "zero_probability_gold": 0.0,
100
+ "zero_probability_gold_n": 600,
101
+ "macro_f1": 0.12883846057517262
102
+ },
103
+ "boolq": {
104
+ "attempted": 600,
105
+ "valid": 600,
106
+ "failed": 0,
107
+ "coverage": 1.0,
108
+ "accuracy_all": 0.5666666666666667,
109
+ "accuracy_valid": 0.5666666666666667,
110
+ "ece_top_label": 0.08635800000000002,
111
+ "mean_confidence": 0.5258943333333334,
112
+ "brier_hard": 0.5020554209333333,
113
+ "brier_hard_n": 600,
114
+ "nll_hard": 0.6953013099942155,
115
+ "nll_hard_n": 600,
116
+ "zero_probability_gold": 0.0,
117
+ "zero_probability_gold_n": 600,
118
+ "macro_f1": 0.5074762578298646
119
+ },
120
+ "sst5": {
121
+ "attempted": 600,
122
+ "valid": 600,
123
+ "failed": 0,
124
+ "coverage": 1.0,
125
+ "accuracy_all": 0.15,
126
+ "accuracy_valid": 0.15,
127
+ "ece_top_label": 0.11939667407180776,
128
+ "mean_confidence": 0.2693966740718078,
129
+ "brier_hard": 0.8207044411796734,
130
+ "brier_hard_n": 600,
131
+ "nll_hard": 1.6589117409852412,
132
+ "nll_hard_n": 600,
133
+ "zero_probability_gold": 0.0,
134
+ "zero_probability_gold_n": 600,
135
+ "score_mae": 1.2247300886774997,
136
+ "score_mae_n": 600,
137
+ "within_one_level": 0.42,
138
+ "within_one_level_n": 600,
139
+ "macro_f1": 0.10629896865332711
140
+ },
141
+ "prompt-injections": {
142
+ "attempted": 116,
143
+ "valid": 116,
144
+ "failed": 0,
145
+ "coverage": 1.0,
146
+ "accuracy_all": 0.5344827586206896,
147
+ "accuracy_valid": 0.5344827586206896,
148
+ "ece_top_label": 0.02089741379310345,
149
+ "mean_confidence": 0.5135853448275862,
150
+ "brier_hard": 0.5003080198275862,
151
+ "brier_hard_n": 116,
152
+ "nll_hard": 0.6934547870034389,
153
+ "nll_hard_n": 116,
154
+ "zero_probability_gold": 0.0,
155
+ "zero_probability_gold_n": 116,
156
+ "macro_f1": 0.513664596273292
157
+ },
158
+ "massive-intent.en": {
159
+ "attempted": 300,
160
+ "valid": 300,
161
+ "failed": 0,
162
+ "coverage": 1.0,
163
+ "accuracy_all": 0.023333333333333334,
164
+ "accuracy_valid": 0.023333333333333334,
165
+ "ece_top_label": 0.09797495336730408,
166
+ "mean_confidence": 0.12130828670063742,
167
+ "brier_hard": 0.9672041702729498,
168
+ "brier_hard_n": 300,
169
+ "nll_hard": 3.1313647377288074,
170
+ "nll_hard_n": 300,
171
+ "zero_probability_gold": 0.0,
172
+ "zero_probability_gold_n": 300,
173
+ "macro_f1": 0.011215849106652133
174
+ },
175
+ "xnli.en": {
176
+ "attempted": 300,
177
+ "valid": 300,
178
+ "failed": 0,
179
+ "coverage": 1.0,
180
+ "accuracy_all": 0.3433333333333333,
181
+ "accuracy_valid": 0.3433333333333333,
182
+ "ece_top_label": 0.05212130996336308,
183
+ "mean_confidence": 0.39545464329669644,
184
+ "brier_hard": 0.6747449551934047,
185
+ "brier_hard_n": 300,
186
+ "nll_hard": 1.1111148411541347,
187
+ "nll_hard_n": 300,
188
+ "zero_probability_gold": 0.0,
189
+ "zero_probability_gold_n": 300,
190
+ "macro_f1": 0.24805362074756226
191
+ }
192
+ }
193
+ },
194
+ "1": {
195
+ "seed": 1,
196
+ "selected_epoch": 14,
197
+ "elapsed_s": 766.9112328969641,
198
+ "initial_validation": {
199
+ "soft_cross_entropy": 1.544869564374288,
200
+ "accuracy": 0.36833333333333335,
201
+ "brier_soft": 0.43051501592000324,
202
+ "decisions": 600
203
+ },
204
+ "selected_validation": {
205
+ "soft_cross_entropy": 0.8377540612220764,
206
+ "accuracy": 0.765,
207
+ "brier_soft": 0.06276483290052662,
208
+ "decisions": 600
209
+ },
210
+ "stopping": {
211
+ "reason": "early_stopping",
212
+ "epochs_completed": 17,
213
+ "patience": 3,
214
+ "min_epochs": 10,
215
+ "epochs_without_improvement": 3,
216
+ "metric": "validation.soft_cross_entropy",
217
+ "min_delta": 0.0
218
+ },
219
+ "frozen_parameters_unchanged": true,
220
+ "suites": {
221
+ "typed-decisions": {
222
+ "attempted": 2000,
223
+ "valid": 2000,
224
+ "failed": 0,
225
+ "coverage": 1.0,
226
+ "accuracy_all": 0.7695,
227
+ "accuracy_valid": 0.7695,
228
+ "ece_top_label": 0.2142237525419539,
229
+ "mean_confidence": 0.555276247458046,
230
+ "brier_hard": 0.3977842163576973,
231
+ "brier_hard_n": 2000,
232
+ "nll_hard": 0.7027635604099817,
233
+ "nll_hard_n": 2000,
234
+ "zero_probability_gold": 0.0,
235
+ "zero_probability_gold_n": 2000,
236
+ "soft_accuracy": 0.47103605907759827,
237
+ "soft_accuracy_n": 2000,
238
+ "brier_soft": 0.06255403710877039,
239
+ "brier_soft_n": 2000,
240
+ "kl_gold_to_prediction": 0.11643450461088396,
241
+ "kl_gold_to_prediction_n": 2000,
242
+ "total_variation": 0.17146599324503523,
243
+ "total_variation_n": 2000,
244
+ "score_mae": 0.2299944493828981,
245
+ "score_mae_n": 800,
246
+ "within_one_level": 0.98875,
247
+ "within_one_level_n": 800,
248
+ "macro_f1": 0.6487623885802034
249
+ },
250
+ "ag-news": {
251
+ "attempted": 600,
252
+ "valid": 600,
253
+ "failed": 0,
254
+ "coverage": 1.0,
255
+ "accuracy_all": 0.9316666666666666,
256
+ "accuracy_valid": 0.9316666666666666,
257
+ "ece_top_label": 0.03387674147756575,
258
+ "mean_confidence": 0.904708562141314,
259
+ "brier_hard": 0.10911433540423408,
260
+ "brier_hard_n": 600,
261
+ "nll_hard": 0.21361121678114303,
262
+ "nll_hard_n": 600,
263
+ "zero_probability_gold": 0.0,
264
+ "zero_probability_gold_n": 600,
265
+ "macro_f1": 0.9283987145646934
266
+ },
267
+ "emotion": {
268
+ "attempted": 600,
269
+ "valid": 600,
270
+ "failed": 0,
271
+ "coverage": 1.0,
272
+ "accuracy_all": 0.5733333333333334,
273
+ "accuracy_valid": 0.5733333333333334,
274
+ "ece_top_label": 0.3169210441851814,
275
+ "mean_confidence": 0.8880587177380197,
276
+ "brier_hard": 0.7264584486728032,
277
+ "brier_hard_n": 600,
278
+ "nll_hard": 2.146154603203373,
279
+ "nll_hard_n": 600,
280
+ "zero_probability_gold": 0.008333333333333333,
281
+ "zero_probability_gold_n": 600,
282
+ "macro_f1": 0.4862286665693279
283
+ },
284
+ "boolq": {
285
+ "attempted": 600,
286
+ "valid": 600,
287
+ "failed": 0,
288
+ "coverage": 1.0,
289
+ "accuracy_all": 0.7983333333333333,
290
+ "accuracy_valid": 0.7983333333333333,
291
+ "ece_top_label": 0.09755433333333334,
292
+ "mean_confidence": 0.8900093333333333,
293
+ "brier_hard": 0.29822958146666667,
294
+ "brier_hard_n": 600,
295
+ "nll_hard": 0.4711253960780479,
296
+ "nll_hard_n": 600,
297
+ "zero_probability_gold": 0.0,
298
+ "zero_probability_gold_n": 600,
299
+ "macro_f1": 0.7813983878883868
300
+ },
301
+ "sst5": {
302
+ "attempted": 600,
303
+ "valid": 600,
304
+ "failed": 0,
305
+ "coverage": 1.0,
306
+ "accuracy_all": 0.42,
307
+ "accuracy_valid": 0.42,
308
+ "ece_top_label": 0.1643168667044709,
309
+ "mean_confidence": 0.5761336745094923,
310
+ "brier_hard": 0.722245716690302,
311
+ "brier_hard_n": 600,
312
+ "nll_hard": 1.412054753482227,
313
+ "nll_hard_n": 600,
314
+ "zero_probability_gold": 0.0,
315
+ "zero_probability_gold_n": 600,
316
+ "score_mae": 0.7360310433394476,
317
+ "score_mae_n": 600,
318
+ "within_one_level": 0.7466666666666667,
319
+ "within_one_level_n": 600,
320
+ "macro_f1": 0.41387920942262574
321
+ },
322
+ "prompt-injections": {
323
+ "attempted": 116,
324
+ "valid": 116,
325
+ "failed": 0,
326
+ "coverage": 1.0,
327
+ "accuracy_all": 0.7155172413793104,
328
+ "accuracy_valid": 0.7155172413793104,
329
+ "ece_top_label": 0.20010517241379305,
330
+ "mean_confidence": 0.9105,
331
+ "brier_hard": 0.43610344172413795,
332
+ "brier_hard_n": 116,
333
+ "nll_hard": 1.2355666268201304,
334
+ "nll_hard_n": 116,
335
+ "zero_probability_gold": 0.008620689655172414,
336
+ "zero_probability_gold_n": 116,
337
+ "macro_f1": 0.7038756091900673
338
+ },
339
+ "massive-intent.en": {
340
+ "attempted": 300,
341
+ "valid": 300,
342
+ "failed": 0,
343
+ "coverage": 1.0,
344
+ "accuracy_all": 0.7433333333333333,
345
+ "accuracy_valid": 0.7433333333333333,
346
+ "ece_top_label": 0.1906540683935104,
347
+ "mean_confidence": 0.9339874017268438,
348
+ "brier_hard": 0.43706654758878133,
349
+ "brier_hard_n": 300,
350
+ "nll_hard": 2.7801978001907903,
351
+ "nll_hard_n": 300,
352
+ "zero_probability_gold": 0.07,
353
+ "zero_probability_gold_n": 300,
354
+ "macro_f1": 0.44346732036749054
355
+ },
356
+ "xnli.en": {
357
+ "attempted": 300,
358
+ "valid": 300,
359
+ "failed": 0,
360
+ "coverage": 1.0,
361
+ "accuracy_all": 0.8766666666666667,
362
+ "accuracy_valid": 0.8766666666666667,
363
+ "ece_top_label": 0.055158283453726184,
364
+ "mean_confidence": 0.913468962679153,
365
+ "brier_hard": 0.19180470102420566,
366
+ "brier_hard_n": 300,
367
+ "nll_hard": 0.3708980570966955,
368
+ "nll_hard_n": 300,
369
+ "zero_probability_gold": 0.0,
370
+ "zero_probability_gold_n": 300,
371
+ "macro_f1": 0.8772028178860477
372
+ }
373
+ }
374
+ },
375
+ "2": {
376
+ "seed": 2,
377
+ "selected_epoch": 15,
378
+ "elapsed_s": 813.5451362769818,
379
+ "initial_validation": {
380
+ "soft_cross_entropy": 1.544869564374288,
381
+ "accuracy": 0.36833333333333335,
382
+ "brier_soft": 0.43051501592000324,
383
+ "decisions": 600
384
+ },
385
+ "selected_validation": {
386
+ "soft_cross_entropy": 0.9867499772707621,
387
+ "accuracy": 0.6316666666666667,
388
+ "brier_soft": 0.14927835414807003,
389
+ "decisions": 600
390
+ },
391
+ "stopping": {
392
+ "reason": "early_stopping",
393
+ "epochs_completed": 18,
394
+ "patience": 3,
395
+ "min_epochs": 10,
396
+ "epochs_without_improvement": 3,
397
+ "metric": "validation.soft_cross_entropy",
398
+ "min_delta": 0.0
399
+ },
400
+ "frozen_parameters_unchanged": true,
401
+ "suites": {
402
+ "typed-decisions": {
403
+ "attempted": 2000,
404
+ "valid": 2000,
405
+ "failed": 0,
406
+ "coverage": 1.0,
407
+ "accuracy_all": 0.661,
408
+ "accuracy_valid": 0.661,
409
+ "ece_top_label": 0.17250655060654457,
410
+ "mean_confidence": 0.48883668313682976,
411
+ "brier_hard": 0.5040496940863629,
412
+ "brier_hard_n": 2000,
413
+ "nll_hard": 0.8694405673021295,
414
+ "nll_hard_n": 2000,
415
+ "zero_probability_gold": 0.0,
416
+ "zero_probability_gold_n": 2000,
417
+ "soft_accuracy": 0.41427322765683283,
418
+ "soft_accuracy_n": 2000,
419
+ "brier_soft": 0.12329331823446005,
420
+ "brier_soft_n": 2000,
421
+ "kl_gold_to_prediction": 0.2167589625120186,
422
+ "kl_gold_to_prediction_n": 2000,
423
+ "total_variation": 0.24557641844905673,
424
+ "total_variation_n": 2000,
425
+ "score_mae": 0.38257377026959916,
426
+ "score_mae_n": 800,
427
+ "within_one_level": 0.92625,
428
+ "within_one_level_n": 800,
429
+ "macro_f1": 0.511489281675723
430
+ },
431
+ "ag-news": {
432
+ "attempted": 600,
433
+ "valid": 600,
434
+ "failed": 0,
435
+ "coverage": 1.0,
436
+ "accuracy_all": 0.2633333333333333,
437
+ "accuracy_valid": 0.2633333333333333,
438
+ "ece_top_label": 0.0415880758215024,
439
+ "mean_confidence": 0.30492140915483573,
440
+ "brier_hard": 0.7533353201524136,
441
+ "brier_hard_n": 600,
442
+ "nll_hard": 1.392051963909145,
443
+ "nll_hard_n": 600,
444
+ "zero_probability_gold": 0.0,
445
+ "zero_probability_gold_n": 600,
446
+ "macro_f1": 0.2196261943166739
447
+ },
448
+ "emotion": {
449
+ "attempted": 600,
450
+ "valid": 600,
451
+ "failed": 0,
452
+ "coverage": 1.0,
453
+ "accuracy_all": 0.04833333333333333,
454
+ "accuracy_valid": 0.04833333333333333,
455
+ "ece_top_label": 0.21796334391120661,
456
+ "mean_confidence": 0.2662966772445399,
457
+ "brier_hard": 0.9083949711964834,
458
+ "brier_hard_n": 600,
459
+ "nll_hard": 2.0132867760168445,
460
+ "nll_hard_n": 600,
461
+ "zero_probability_gold": 0.0,
462
+ "zero_probability_gold_n": 600,
463
+ "macro_f1": 0.03431938431938432
464
+ },
465
+ "boolq": {
466
+ "attempted": 600,
467
+ "valid": 600,
468
+ "failed": 0,
469
+ "coverage": 1.0,
470
+ "accuracy_all": 0.49833333333333335,
471
+ "accuracy_valid": 0.49833333333333335,
472
+ "ece_top_label": 0.03133150000000007,
473
+ "mean_confidence": 0.5290545,
474
+ "brier_hard": 0.5040687543666666,
475
+ "brier_hard_n": 600,
476
+ "nll_hard": 0.6972355680599187,
477
+ "nll_hard_n": 600,
478
+ "zero_probability_gold": 0.0,
479
+ "zero_probability_gold_n": 600,
480
+ "macro_f1": 0.49597983919356775
481
+ },
482
+ "sst5": {
483
+ "attempted": 600,
484
+ "valid": 600,
485
+ "failed": 0,
486
+ "coverage": 1.0,
487
+ "accuracy_all": 0.22666666666666666,
488
+ "accuracy_valid": 0.22666666666666666,
489
+ "ece_top_label": 0.08382145968032653,
490
+ "mean_confidence": 0.30788872429147607,
491
+ "brier_hard": 0.8111226172703578,
492
+ "brier_hard_n": 600,
493
+ "nll_hard": 1.6364255257258482,
494
+ "nll_hard_n": 600,
495
+ "zero_probability_gold": 0.0,
496
+ "zero_probability_gold_n": 600,
497
+ "score_mae": 1.223873285603664,
498
+ "score_mae_n": 600,
499
+ "within_one_level": 0.39,
500
+ "within_one_level_n": 600,
501
+ "macro_f1": 0.12552908285983264
502
+ },
503
+ "prompt-injections": {
504
+ "attempted": 116,
505
+ "valid": 116,
506
+ "failed": 0,
507
+ "coverage": 1.0,
508
+ "accuracy_all": 0.3448275862068966,
509
+ "accuracy_valid": 0.3448275862068966,
510
+ "ece_top_label": 0.1914112068965517,
511
+ "mean_confidence": 0.5362387931034482,
512
+ "brier_hard": 0.5323824856896552,
513
+ "brier_hard_n": 116,
514
+ "nll_hard": 0.7258214322004152,
515
+ "nll_hard_n": 116,
516
+ "zero_probability_gold": 0.0,
517
+ "zero_probability_gold_n": 116,
518
+ "macro_f1": 0.275
519
+ },
520
+ "massive-intent.en": {
521
+ "attempted": 300,
522
+ "valid": 300,
523
+ "failed": 0,
524
+ "coverage": 1.0,
525
+ "accuracy_all": 0.06333333333333334,
526
+ "accuracy_valid": 0.06333333333333334,
527
+ "ece_top_label": 0.03411377590323972,
528
+ "mean_confidence": 0.09744710923657304,
529
+ "brier_hard": 0.9608971275390831,
530
+ "brier_hard_n": 300,
531
+ "nll_hard": 3.135262799945936,
532
+ "nll_hard_n": 300,
533
+ "zero_probability_gold": 0.0,
534
+ "zero_probability_gold_n": 300,
535
+ "macro_f1": 0.02907709642455108
536
+ },
537
+ "xnli.en": {
538
+ "attempted": 300,
539
+ "valid": 300,
540
+ "failed": 0,
541
+ "coverage": 1.0,
542
+ "accuracy_all": 0.3433333333333333,
543
+ "accuracy_valid": 0.3433333333333333,
544
+ "ece_top_label": 0.04294870734165375,
545
+ "mean_confidence": 0.38628204067498706,
546
+ "brier_hard": 0.6761361630010466,
547
+ "brier_hard_n": 300,
548
+ "nll_hard": 1.1128104653939794,
549
+ "nll_hard_n": 300,
550
+ "zero_probability_gold": 0.0,
551
+ "zero_probability_gold_n": 300,
552
+ "macro_f1": 0.3099922839506173
553
+ }
554
+ }
555
+ },
556
+ "3": {
557
+ "seed": 3,
558
+ "selected_epoch": 24,
559
+ "elapsed_s": 1226.4128511130111,
560
+ "initial_validation": {
561
+ "soft_cross_entropy": 1.544869564374288,
562
+ "accuracy": 0.36833333333333335,
563
+ "brier_soft": 0.43051501592000324,
564
+ "decisions": 600
565
+ },
566
+ "selected_validation": {
567
+ "soft_cross_entropy": 0.9105557099978129,
568
+ "accuracy": 0.715,
569
+ "brier_soft": 0.1013037375236551,
570
+ "decisions": 600
571
+ },
572
+ "stopping": {
573
+ "reason": "early_stopping",
574
+ "epochs_completed": 27,
575
+ "patience": 3,
576
+ "min_epochs": 10,
577
+ "epochs_without_improvement": 3,
578
+ "metric": "validation.soft_cross_entropy",
579
+ "min_delta": 0.0
580
+ },
581
+ "frozen_parameters_unchanged": true,
582
+ "suites": {
583
+ "typed-decisions": {
584
+ "attempted": 2000,
585
+ "valid": 2000,
586
+ "failed": 0,
587
+ "coverage": 1.0,
588
+ "accuracy_all": 0.7145,
589
+ "accuracy_valid": 0.7145,
590
+ "ece_top_label": 0.18500231281217855,
591
+ "mean_confidence": 0.5300184130804106,
592
+ "brier_hard": 0.45082257101028994,
593
+ "brier_hard_n": 2000,
594
+ "nll_hard": 0.7891005479512444,
595
+ "nll_hard_n": 2000,
596
+ "zero_probability_gold": 0.0,
597
+ "zero_probability_gold_n": 2000,
598
+ "soft_accuracy": 0.4447570518885741,
599
+ "soft_accuracy_n": 2000,
600
+ "brier_soft": 0.09325671264191153,
601
+ "brier_soft_n": 2000,
602
+ "kl_gold_to_prediction": 0.17157487793178008,
603
+ "kl_gold_to_prediction_n": 2000,
604
+ "total_variation": 0.21097575386758916,
605
+ "total_variation_n": 2000,
606
+ "score_mae": 0.32394138130337413,
607
+ "score_mae_n": 800,
608
+ "within_one_level": 0.96375,
609
+ "within_one_level_n": 800,
610
+ "macro_f1": 0.5630278027844289
611
+ },
612
+ "ag-news": {
613
+ "attempted": 600,
614
+ "valid": 600,
615
+ "failed": 0,
616
+ "coverage": 1.0,
617
+ "accuracy_all": 0.285,
618
+ "accuracy_valid": 0.285,
619
+ "ece_top_label": 0.03904648151058483,
620
+ "mean_confidence": 0.3240464815105848,
621
+ "brier_hard": 0.7625373960165811,
622
+ "brier_hard_n": 600,
623
+ "nll_hard": 1.4140716749662658,
624
+ "nll_hard_n": 600,
625
+ "zero_probability_gold": 0.0,
626
+ "zero_probability_gold_n": 600,
627
+ "macro_f1": 0.25772655273033623
628
+ },
629
+ "emotion": {
630
+ "attempted": 600,
631
+ "valid": 600,
632
+ "failed": 0,
633
+ "coverage": 1.0,
634
+ "accuracy_all": 0.135,
635
+ "accuracy_valid": 0.135,
636
+ "ece_top_label": 0.31863940813484304,
637
+ "mean_confidence": 0.45363940813484305,
638
+ "brier_hard": 0.9777635802768799,
639
+ "brier_hard_n": 600,
640
+ "nll_hard": 2.137352366621794,
641
+ "nll_hard_n": 600,
642
+ "zero_probability_gold": 0.0,
643
+ "zero_probability_gold_n": 600,
644
+ "macro_f1": 0.06590661903472231
645
+ },
646
+ "boolq": {
647
+ "attempted": 600,
648
+ "valid": 600,
649
+ "failed": 0,
650
+ "coverage": 1.0,
651
+ "accuracy_all": 0.48833333333333334,
652
+ "accuracy_valid": 0.48833333333333334,
653
+ "ece_top_label": 0.10787050000000001,
654
+ "mean_confidence": 0.5912141666666667,
655
+ "brier_hard": 0.5250116893,
656
+ "brier_hard_n": 600,
657
+ "nll_hard": 0.719837552287629,
658
+ "nll_hard_n": 600,
659
+ "zero_probability_gold": 0.0,
660
+ "zero_probability_gold_n": 600,
661
+ "macro_f1": 0.48696381173075903
662
+ },
663
+ "sst5": {
664
+ "attempted": 600,
665
+ "valid": 600,
666
+ "failed": 0,
667
+ "coverage": 1.0,
668
+ "accuracy_all": 0.25666666666666665,
669
+ "accuracy_valid": 0.25666666666666665,
670
+ "ece_top_label": 0.05066152500332892,
671
+ "mean_confidence": 0.3073281916699956,
672
+ "brier_hard": 0.8215936069738192,
673
+ "brier_hard_n": 600,
674
+ "nll_hard": 1.6775697808435297,
675
+ "nll_hard_n": 600,
676
+ "zero_probability_gold": 0.0,
677
+ "zero_probability_gold_n": 600,
678
+ "score_mae": 1.245226132804566,
679
+ "score_mae_n": 600,
680
+ "within_one_level": 0.44333333333333336,
681
+ "within_one_level_n": 600,
682
+ "macro_f1": 0.13260195835186606
683
+ },
684
+ "prompt-injections": {
685
+ "attempted": 116,
686
+ "valid": 116,
687
+ "failed": 0,
688
+ "coverage": 1.0,
689
+ "accuracy_all": 0.5689655172413793,
690
+ "accuracy_valid": 0.5689655172413793,
691
+ "ece_top_label": 0.07609396551724136,
692
+ "mean_confidence": 0.6450594827586207,
693
+ "brier_hard": 0.4881634129310345,
694
+ "brier_hard_n": 116,
695
+ "nll_hard": 0.6835747994579267,
696
+ "nll_hard_n": 116,
697
+ "zero_probability_gold": 0.0,
698
+ "zero_probability_gold_n": 116,
699
+ "macro_f1": 0.5461658841940532
700
+ },
701
+ "massive-intent.en": {
702
+ "attempted": 300,
703
+ "valid": 300,
704
+ "failed": 0,
705
+ "coverage": 1.0,
706
+ "accuracy_all": 0.023333333333333334,
707
+ "accuracy_valid": 0.023333333333333334,
708
+ "ece_top_label": 0.14827139054632427,
709
+ "mean_confidence": 0.1716047238796576,
710
+ "brier_hard": 1.012268483556247,
711
+ "brier_hard_n": 300,
712
+ "nll_hard": 3.540188562309456,
713
+ "nll_hard_n": 300,
714
+ "zero_probability_gold": 0.0,
715
+ "zero_probability_gold_n": 300,
716
+ "macro_f1": 0.014601824457593688
717
+ },
718
+ "xnli.en": {
719
+ "attempted": 300,
720
+ "valid": 300,
721
+ "failed": 0,
722
+ "coverage": 1.0,
723
+ "accuracy_all": 0.33666666666666667,
724
+ "accuracy_valid": 0.33666666666666667,
725
+ "ece_top_label": 0.21980800490067007,
726
+ "mean_confidence": 0.5564746715673368,
727
+ "brier_hard": 0.7709342547858443,
728
+ "brier_hard_n": 300,
729
+ "nll_hard": 1.2650663651832452,
730
+ "nll_hard_n": 300,
731
+ "zero_probability_gold": 0.0,
732
+ "zero_probability_gold_n": 300,
733
+ "macro_f1": 0.20238216957239646
734
+ }
735
+ }
736
+ },
737
+ "4": {
738
+ "seed": 4,
739
+ "selected_epoch": 7,
740
+ "elapsed_s": 452.7137592760264,
741
+ "initial_validation": {
742
+ "soft_cross_entropy": 1.544869564374288,
743
+ "accuracy": 0.36833333333333335,
744
+ "brier_soft": 0.43051501592000324,
745
+ "decisions": 600
746
+ },
747
+ "selected_validation": {
748
+ "soft_cross_entropy": 0.8448993345101674,
749
+ "accuracy": 0.7533333333333333,
750
+ "brier_soft": 0.0668264798882107,
751
+ "decisions": 600
752
+ },
753
+ "stopping": {
754
+ "reason": "early_stopping",
755
+ "epochs_completed": 10,
756
+ "patience": 3,
757
+ "min_epochs": 10,
758
+ "epochs_without_improvement": 3,
759
+ "metric": "validation.soft_cross_entropy",
760
+ "min_delta": 0.0
761
+ },
762
+ "frozen_parameters_unchanged": true,
763
+ "suites": {
764
+ "typed-decisions": {
765
+ "attempted": 2000,
766
+ "valid": 2000,
767
+ "failed": 0,
768
+ "coverage": 1.0,
769
+ "accuracy_all": 0.7565,
770
+ "accuracy_valid": 0.7565,
771
+ "ece_top_label": 0.20311433485058888,
772
+ "mean_confidence": 0.5533856651494111,
773
+ "brier_hard": 0.4063906772103004,
774
+ "brier_hard_n": 2000,
775
+ "nll_hard": 0.7152837956058272,
776
+ "nll_hard_n": 2000,
777
+ "zero_probability_gold": 0.0,
778
+ "zero_probability_gold_n": 2000,
779
+ "soft_accuracy": 0.46913207673904656,
780
+ "soft_accuracy_n": 2000,
781
+ "brier_soft": 0.06497074467877281,
782
+ "brier_soft_n": 2000,
783
+ "kl_gold_to_prediction": 0.12038107186141371,
784
+ "kl_gold_to_prediction_n": 2000,
785
+ "total_variation": 0.17619801386776057,
786
+ "total_variation_n": 2000,
787
+ "score_mae": 0.2447599606823941,
788
+ "score_mae_n": 800,
789
+ "within_one_level": 0.99375,
790
+ "within_one_level_n": 800,
791
+ "macro_f1": 0.6349242079343602
792
+ },
793
+ "ag-news": {
794
+ "attempted": 600,
795
+ "valid": 600,
796
+ "failed": 0,
797
+ "coverage": 1.0,
798
+ "accuracy_all": 0.935,
799
+ "accuracy_valid": 0.935,
800
+ "ece_top_label": 0.0277188639855619,
801
+ "mean_confidence": 0.9106097216911906,
802
+ "brier_hard": 0.10223891825148304,
803
+ "brier_hard_n": 600,
804
+ "nll_hard": 0.2046134786862912,
805
+ "nll_hard_n": 600,
806
+ "zero_probability_gold": 0.0,
807
+ "zero_probability_gold_n": 600,
808
+ "macro_f1": 0.9319475717225341
809
+ },
810
+ "emotion": {
811
+ "attempted": 600,
812
+ "valid": 600,
813
+ "failed": 0,
814
+ "coverage": 1.0,
815
+ "accuracy_all": 0.565,
816
+ "accuracy_valid": 0.565,
817
+ "ece_top_label": 0.3127789762695219,
818
+ "mean_confidence": 0.8750361088494466,
819
+ "brier_hard": 0.7278820816911535,
820
+ "brier_hard_n": 600,
821
+ "nll_hard": 2.179428847187102,
822
+ "nll_hard_n": 600,
823
+ "zero_probability_gold": 0.01,
824
+ "zero_probability_gold_n": 600,
825
+ "macro_f1": 0.47063728412574746
826
+ },
827
+ "boolq": {
828
+ "attempted": 600,
829
+ "valid": 600,
830
+ "failed": 0,
831
+ "coverage": 1.0,
832
+ "accuracy_all": 0.8116666666666666,
833
+ "accuracy_valid": 0.8116666666666666,
834
+ "ece_top_label": 0.08128733333333335,
835
+ "mean_confidence": 0.8882866666666667,
836
+ "brier_hard": 0.2800035268666667,
837
+ "brier_hard_n": 600,
838
+ "nll_hard": 0.4526075086973918,
839
+ "nll_hard_n": 600,
840
+ "zero_probability_gold": 0.0,
841
+ "zero_probability_gold_n": 600,
842
+ "macro_f1": 0.7989317880539385
843
+ },
844
+ "sst5": {
845
+ "attempted": 600,
846
+ "valid": 600,
847
+ "failed": 0,
848
+ "coverage": 1.0,
849
+ "accuracy_all": 0.47333333333333333,
850
+ "accuracy_valid": 0.47333333333333333,
851
+ "ece_top_label": 0.11648579870821457,
852
+ "mean_confidence": 0.5717307193165773,
853
+ "brier_hard": 0.6735168048802583,
854
+ "brier_hard_n": 600,
855
+ "nll_hard": 1.2841517570209144,
856
+ "nll_hard_n": 600,
857
+ "zero_probability_gold": 0.0,
858
+ "zero_probability_gold_n": 600,
859
+ "score_mae": 0.6724125988264632,
860
+ "score_mae_n": 600,
861
+ "within_one_level": 0.7883333333333333,
862
+ "within_one_level_n": 600,
863
+ "macro_f1": 0.46002232794452175
864
+ },
865
+ "prompt-injections": {
866
+ "attempted": 116,
867
+ "valid": 116,
868
+ "failed": 0,
869
+ "coverage": 1.0,
870
+ "accuracy_all": 0.7241379310344828,
871
+ "accuracy_valid": 0.7241379310344828,
872
+ "ece_top_label": 0.22484482758620694,
873
+ "mean_confidence": 0.9346275862068966,
874
+ "brier_hard": 0.44889866172413795,
875
+ "brier_hard_n": 116,
876
+ "nll_hard": 1.682957724359292,
877
+ "nll_hard_n": 116,
878
+ "zero_probability_gold": 0.02586206896551724,
879
+ "zero_probability_gold_n": 116,
880
+ "macro_f1": 0.7118012422360249
881
+ },
882
+ "massive-intent.en": {
883
+ "attempted": 300,
884
+ "valid": 300,
885
+ "failed": 0,
886
+ "coverage": 1.0,
887
+ "accuracy_all": 0.7733333333333333,
888
+ "accuracy_valid": 0.7733333333333333,
889
+ "ece_top_label": 0.1762748719939652,
890
+ "mean_confidence": 0.9448706790799233,
891
+ "brier_hard": 0.3914895786578462,
892
+ "brier_hard_n": 300,
893
+ "nll_hard": 2.594218156353015,
894
+ "nll_hard_n": 300,
895
+ "zero_probability_gold": 0.06333333333333334,
896
+ "zero_probability_gold_n": 300,
897
+ "macro_f1": 0.4611733462473961
898
+ },
899
+ "xnli.en": {
900
+ "attempted": 300,
901
+ "valid": 300,
902
+ "failed": 0,
903
+ "coverage": 1.0,
904
+ "accuracy_all": 0.8766666666666667,
905
+ "accuracy_valid": 0.8766666666666667,
906
+ "ece_top_label": 0.059115758529190925,
907
+ "mean_confidence": 0.9151327612640476,
908
+ "brier_hard": 0.1964252077913328,
909
+ "brier_hard_n": 300,
910
+ "nll_hard": 0.36246740512671244,
911
+ "nll_hard_n": 300,
912
+ "zero_probability_gold": 0.0,
913
+ "zero_probability_gold_n": 300,
914
+ "macro_f1": 0.8760253913421515
915
+ }
916
+ }
917
+ }
918
+ },
919
+ "summary": {
920
+ "typed-decisions": {
921
+ "n": 5,
922
+ "mean_percent": 70.23,
923
+ "sample_variance_pp_squared": 44.56824999999997,
924
+ "sample_std_pp": 6.675945625901995
925
+ },
926
+ "ag-news": {
927
+ "n": 5,
928
+ "mean_percent": 54.56666666666666,
929
+ "sample_variance_pp_squared": 1255.536111111111,
930
+ "sample_std_pp": 35.433544997799906
931
+ },
932
+ "boolq": {
933
+ "n": 5,
934
+ "mean_percent": 63.266666666666666,
935
+ "sample_variance_pp_squared": 256.79999999999995,
936
+ "sample_std_pp": 16.0249804992081
937
+ },
938
+ "emotion": {
939
+ "n": 5,
940
+ "mean_percent": 30.333333333333332,
941
+ "sample_variance_pp_squared": 616.1666666666666,
942
+ "sample_std_pp": 24.82270466058577
943
+ },
944
+ "prompt-injections": {
945
+ "n": 5,
946
+ "mean_percent": 57.758620689655174,
947
+ "sample_variance_pp_squared": 241.5279429250891,
948
+ "sample_std_pp": 15.541169290793055
949
+ },
950
+ "sst5": {
951
+ "n": 5,
952
+ "mean_percent": 30.53333333333333,
953
+ "sample_variance_pp_squared": 185.14444444444447,
954
+ "sample_std_pp": 13.606779356057938
955
+ },
956
+ "massive-intent.en": {
957
+ "n": 5,
958
+ "mean_percent": 32.53333333333333,
959
+ "sample_variance_pp_squared": 1566.1999999999998,
960
+ "sample_std_pp": 39.57524478761944
961
+ },
962
+ "xnli.en": {
963
+ "n": 5,
964
+ "mean_percent": 55.53333333333334,
965
+ "sample_variance_pp_squared": 860.5333333333334,
966
+ "sample_std_pp": 29.33484844571953
967
+ }
968
+ },
969
+ "candidate_seed": 1,
970
+ "variance_definition": "sample variance, n-1, percentage points squared",
971
+ "benchmark_manifest": {
972
+ "format_version": 1,
973
+ "sha256": "10fb671245cdac1ee09e9c4d2b9b09829fbf1abd06e0954fddee0b9ec8519854",
974
+ "cases": 3516,
975
+ "decisions": 5116,
976
+ "profile": "heldout-eight",
977
+ "role": "test",
978
+ "parent_bundles": {
979
+ "core": "477c9cbd4f9eaf8844fa4d6870560e751960735e7441b803306d6e011e689435",
980
+ "language_en": "a3d41f5060826770f0f28b16a0aff26b4294691423624349db6b008789ea9096"
981
+ },
982
+ "purpose": "Base vs official typed-decisions specialist vs trained Laya-only input interface; no new training."
983
+ }
984
+ }
requirements.txt ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ torch==2.13.0
2
+ laya==0.3.20
3
+ transformers==5.17.0
4
+ huggingface-hub==1.33.0
5
+ safetensors==0.8.0
6
+ numpy==2.4.4
7
+ tokenizers==0.23.2
8
+ datasets==5.0.1
training_protocol.json ADDED
@@ -0,0 +1,92 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "experiment": "E0i",
3
+ "seeds": [
4
+ 0,
5
+ 1,
6
+ 2,
7
+ 3,
8
+ 4
9
+ ],
10
+ "created_at": "2026-09-27T13:49:50.006241+00:00",
11
+ "data": {
12
+ "train": {
13
+ "format_version": 1,
14
+ "sha256": "5b273d74d4eed80f75763af64cfaf3fabae1d2a31a0b67551a5a753b6972e22e",
15
+ "cases": 1080,
16
+ "decisions": 5400,
17
+ "profile": "hybrid-train",
18
+ "role": "train",
19
+ "source": "LocalLLaMA/typed-decisions",
20
+ "revision": "f7a2487edd7a043a5441a5e9ccc7fe5ddbd9ebe8",
21
+ "source_split": "train",
22
+ "split_seed": 42
23
+ },
24
+ "validation": {
25
+ "format_version": 1,
26
+ "sha256": "95fe23fc9e82f9992c5056e7ab7ff76888aa614f49aa6d48929aba5d4605812a",
27
+ "cases": 120,
28
+ "decisions": 600,
29
+ "profile": "hybrid-validation",
30
+ "role": "validation",
31
+ "source": "LocalLLaMA/typed-decisions",
32
+ "revision": "f7a2487edd7a043a5441a5e9ccc7fe5ddbd9ebe8",
33
+ "source_split": "train",
34
+ "split_seed": 42
35
+ },
36
+ "test": {
37
+ "format_version": 1,
38
+ "sha256": "dd34a1df70013b7835b57aa474e36bf432f6e586a98e960e2b43b80ab7b75d06",
39
+ "cases": 400,
40
+ "decisions": 2000,
41
+ "profile": "typed-decisions-test",
42
+ "role": "test",
43
+ "source_split": "test",
44
+ "parent_sha256": "477c9cbd4f9eaf8844fa4d6870560e751960735e7441b803306d6e011e689435"
45
+ }
46
+ },
47
+ "heldout_manifest": {
48
+ "format_version": 1,
49
+ "sha256": "10fb671245cdac1ee09e9c4d2b9b09829fbf1abd06e0954fddee0b9ec8519854",
50
+ "cases": 3516,
51
+ "decisions": 5116,
52
+ "profile": "heldout-eight",
53
+ "role": "test",
54
+ "parent_bundles": {
55
+ "core": "477c9cbd4f9eaf8844fa4d6870560e751960735e7441b803306d6e011e689435",
56
+ "language_en": "a3d41f5060826770f0f28b16a0aff26b4294691423624349db6b008789ea9096"
57
+ },
58
+ "purpose": "Base vs official typed-decisions specialist vs trained Laya-only input interface; no new training."
59
+ },
60
+ "max_len": 1024,
61
+ "head_max_len": 256,
62
+ "epochs": null,
63
+ "min_epochs": 10,
64
+ "early_stopping_patience": 3,
65
+ "early_stopping_metric": "validation.soft_cross_entropy",
66
+ "early_stopping_min_delta": 0.0,
67
+ "microbatch": 8,
68
+ "accumulation": 4,
69
+ "effective_batch": 32,
70
+ "interface_lr": 0.0003,
71
+ "weight_decay": 0.01,
72
+ "gradient_clip": 1,
73
+ "training_scope": "bridge",
74
+ "training_dropout": false,
75
+ "selection": "minimum validation soft cross-entropy among trained epochs",
76
+ "release_candidate_selection": "lowest selected validation soft cross-entropy across all five seeds; tie goes to the lowest seed; no test-based selection",
77
+ "device": "cuda:1",
78
+ "deterministic": true,
79
+ "native_sdk_calibration": true,
80
+ "publish_policy": "draft locally; upload only after user and assistant agree",
81
+ "baseline_predictions_sha256": "37dff666a00a340cd6e37f775dc60df33a3c6eb78f76bfee7f094443c225b6e8",
82
+ "base_model": "convaiinnovations/laya",
83
+ "base_revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851",
84
+ "training_sources_sha256": {
85
+ "ariadne_bench/reproducibility.py": "5b0c161c596f25277b3326c645df53d5d1f2dc0272da9e77bc4a4c0d56188b6e",
86
+ "ariadne_bench/frozen_input_interface.py": "253f0186dac6e4a7f4b0fcf3c33449c3b3882cf8be538dca6eeca8d9e4ce9b83",
87
+ "ariadne_bench/full_input_interface.py": "a8f440a2185085c64896300a243ac52a9619c4d05a1b4e43c50c2c32aa1da996",
88
+ "ariadne_bench/full_finetune.py": "6067be8cf837d254b919c841c9a9851e1546a413dbf6324ff242e5d2f56c9734",
89
+ "ariadne_bench/interfaces.py": "19700fb8170b2460401900cb96ebdff22bd7bf778ca5132e98d56ba68f344187",
90
+ "ariadne_bench/experiments/align.py": "31e2733dcc223a1e1a6250d41c4158f84091f1cb4067d61388e06b5b01ae34c4"
91
+ }
92
+ }
upload-manifest.json ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "published_by_assistant": false,
3
+ "candidate_seed": 1,
4
+ "selected_epoch": 14,
5
+ "format": "compact affine adapter requiring pinned base Laya",
6
+ "existing_model_card_preserved": true,
7
+ "inference_code_and_weights_match_verified_export": true,
8
+ "removed_only_other_seed_loading_entries": true,
9
+ "files_sha256": {
10
+ "LICENSE": "a6cba85bc92e0cff7a450b1d873c0eaa2e9fc96bf472df0247a26bec77bf3ff9",
11
+ "THIRD_PARTY.md": "3dd2683344d45ce64c15f45c723f4b0db720deee84446365fdf4084a73ebeceb",
12
+ "USAGE.md": "e6d76a3a450fd589e52cab5190a065cbd783082a98d3524071bcf60a7015ebef",
13
+ "adapter.safetensors": "e5ec5e1d494509f63960b6949af15be8390442271805ba45d796a1911b5c239a",
14
+ "adapter_config.json": "8d7450b812f696e1dbeb3d7abfa13fbee0cde95aae0349a94e5f22c9b458ed1f",
15
+ "ariadne_bench/__init__.py": "a73c9fdf21413382cb93b523bfb3925aa3fd354e18faeead4650e86a30645411",
16
+ "ariadne_bench/__main__.py": "935a1c1166b0c1ea35a82256345000bf2c73ded718d77773bc27a71ecce28f7d",
17
+ "ariadne_bench/adapters.py": "3bd4820165be077b3b98fe83ac23e649e1ad69186ad5f2a4f5ee7addf1427cf5",
18
+ "ariadne_bench/cli.py": "de267c0502489eb4bb37e4b8f6faa8da8be37a56ced270f754424a40404d1af6",
19
+ "ariadne_bench/datasets.py": "31c722c64895519a4635b501da88862b446384682e3e8ee884495941d6990912",
20
+ "ariadne_bench/experiments/__init__.py": "95ad0a7f30c4e0a194d0a6e2ae8905abae5d622edbb3d4e5d791b4bc460aee5d",
21
+ "ariadne_bench/experiments/align.py": "31e2733dcc223a1e1a6250d41c4158f84091f1cb4067d61388e06b5b01ae34c4",
22
+ "ariadne_bench/experiments/prepare_alignment.py": "c4f39c10ce46ee13c0d89128d18a85ec4a86729e81e070f5834ed01fc1c405a9",
23
+ "ariadne_bench/experiments/spectrum.py": "22d4801b585c539d0d579a9d2a00076e1002fd337bfe31d1906d8ae3fb8d4320",
24
+ "ariadne_bench/frozen_input_interface.py": "253f0186dac6e4a7f4b0fcf3c33449c3b3882cf8be538dca6eeca8d9e4ce9b83",
25
+ "ariadne_bench/full_finetune.py": "6067be8cf837d254b919c841c9a9851e1546a413dbf6324ff242e5d2f56c9734",
26
+ "ariadne_bench/full_input_interface.py": "a8f440a2185085c64896300a243ac52a9619c4d05a1b4e43c50c2c32aa1da996",
27
+ "ariadne_bench/hybrid.py": "90fbbdab9b8157a32854d27c2d93089be6dc88dfe9320426f102eae4e03d9ffd",
28
+ "ariadne_bench/hybrid_control.py": "1b53678bea73fe7608570b78d7215d1dff2e38d6ad3627615f26c06ae74d8b4c",
29
+ "ariadne_bench/interfaces.py": "19700fb8170b2460401900cb96ebdff22bd7bf778ca5132e98d56ba68f344187",
30
+ "ariadne_bench/metrics.py": "fc6fde20c95d5052de52d05d6fb9b6eb69b780c46966f5bf1e517738b276a4b4",
31
+ "ariadne_bench/portable.py": "259f633a8bb6663d0594d9639ea8d80ef493f5a63d4b11444a4003f2f9a318e8",
32
+ "ariadne_bench/reproducibility.py": "5b0c161c596f25277b3326c645df53d5d1f2dc0272da9e77bc4a4c0d56188b6e",
33
+ "ariadne_bench/runner.py": "b20e2af4ba254e2d48a07563356f95728c4807018c3c989b429e9922674c64cd",
34
+ "ariadne_bench/schema.py": "67794f1172c1ef9d9625b84c8096fd48b8fa7cd10e3f88fc13ba3a98ef7c7a76",
35
+ "benchmarks/sources.lock.json": "6e6eaab2778092c8127257fce202c9fa8d4625dd0d029490dd9f137da91d134c",
36
+ "environment.json": "c5b9c796d97f3dec6d242caf79aa5d2ac69c91717f2be6ed5a51627bf6d58c17",
37
+ "evidence/adapter-equivalence.json": "c77faa691bd273d363dbb196155545b4f72697b9d0f4c80451ffb06b0090d200",
38
+ "evidence/audit.json": "1fb474a47923e22374a47e06f545e747420c0a7bd269bbbeaabb5869e89a8d70",
39
+ "evidence/candidate-uncertainty.json": "39eba4a205eaea5f13384da24ca8563de364c9edb6858c1db38eb4111960ce40",
40
+ "evidence/identity-check.json": "0cc8904dc58e1e0ed49496c7ae01980c52e3a1395d61ef91286cec2c9dcdfa25",
41
+ "evidence/replay-check.json": "4f246eb3d6afc0c0b1d306a20b82060ae564d476ec22451322b065edb946d15e",
42
+ "evidence/seed-0-history.json": "bc3de468970bf350426d9c4281456e258984838feb4d8dbbfe54c8752ebb24b7",
43
+ "evidence/seed-1-history.json": "8ee78b5685138712512b39035ef48a491ed52813c2131a8e9b80715871ab9405",
44
+ "evidence/seed-2-history.json": "d9bf7ab3991b85b6e3877b9ebd4ca2fc113cba1716f56da5aa16f4018ac4b197",
45
+ "evidence/seed-3-history.json": "7c0f19638e31f28e438af879ceaa68bbe06480e6f7fb72e8e749b4aabb304341",
46
+ "evidence/seed-4-history.json": "601c5bd91dcbb004ff8a07621f3f125ba578cc4107a9884e8fe69ac17751a741",
47
+ "export-verification.json": "b9f8667626e2facd03f310c0c67c100ddde9481218c51cbead7972fd048d4afb",
48
+ "licenses/jev-benchmarks-LICENSE": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4",
49
+ "licenses/laya-LICENSE": "a6cba85bc92e0cff7a450b1d873c0eaa2e9fc96bf472df0247a26bec77bf3ff9",
50
+ "load_adapter.py": "2f305ef95fef2b2482e9214836fc0071a59cfd6c6d3aaf42cebdbfb131272081",
51
+ "metrics.json": "dd45292f04dc5b81cc958a39f618aa8748b5198e39b5effee6c95520c92d94b6",
52
+ "requirements.txt": "285a8ae6529abeefd058f2e9679e0e314b19fda38169e685ea16aec59b053bd9",
53
+ "training_protocol.json": "dccf81e3d8a70c756a95a88280cbec139eea43bc391d67fa7b40e384a8824f57"
54
+ }
55
+ }