Release v1.3.1: faster SystemOne batching with exact output parity
Browse files- RUNTIME-RELEASE.json +30 -0
- SERVING_OPTIMIZATION.json +34 -0
- bundle-manifest.json +12 -5
- code/decision_api.py +62 -3
- model-card-example.json +1 -1
- release-manifest.json +24 -10
RUNTIME-RELEASE.json
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"format": "decision-runtime-patch-v1",
|
| 3 |
+
"family": "Nox",
|
| 4 |
+
"release_tag": "v1.3.1",
|
| 5 |
+
"change_kind": "runtime_only",
|
| 6 |
+
"weights_revision": "ad089ad3a5dc9a7a21e6d96db546bb53e2212654",
|
| 7 |
+
"source_bundle_manifest_sha256": "92d7f5be5e1ef21ee682f574b35cc01de4dd8ab16b8aa6edf0dfbdfcfcfabba3",
|
| 8 |
+
"bundle_manifest_sha256": "83876db506b2d98e3e8ce7d34310b21f97bac30d4f5aef3371053798bff08830",
|
| 9 |
+
"changed_inference_files": [
|
| 10 |
+
"code/decision_api.py"
|
| 11 |
+
],
|
| 12 |
+
"weights_tokenizer_prompts_calibration_profile_unchanged": true,
|
| 13 |
+
"default_public_parity": {
|
| 14 |
+
"requests": 2856,
|
| 15 |
+
"answers": 3160,
|
| 16 |
+
"fixtures": 58,
|
| 17 |
+
"raw_logits_probabilities_typed_outputs_exact": true,
|
| 18 |
+
"receipt_sha256": "946dba5c31074b63efa3223ede17857e5f9d1b7b3b44de08cd9add9ceece878c"
|
| 19 |
+
},
|
| 20 |
+
"timing": {
|
| 21 |
+
"receipt_sha256": "501f25efd1813b566d77e98278229440b7a3b4a2358def1462c0171a33e9f3be",
|
| 22 |
+
"shape": "distinct fixed-length questions; Q=32; 499 tokens per question",
|
| 23 |
+
"reference_ms": 419.124,
|
| 24 |
+
"candidate_ms": 414.199,
|
| 25 |
+
"same_measured_api_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152",
|
| 26 |
+
"no_shared_state_neural_cache_claim": true
|
| 27 |
+
},
|
| 28 |
+
"capability_scores_unchanged": true,
|
| 29 |
+
"downloaded_package_offline_proof_required_before_promotion": true
|
| 30 |
+
}
|
SERVING_OPTIMIZATION.json
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"format": "joint-serving-runtime-candidate-v1",
|
| 3 |
+
"family": "Nox",
|
| 4 |
+
"source_publication_revision": "ad089ad3a5dc9a7a21e6d96db546bb53e2212654",
|
| 5 |
+
"source_bundle_manifest_sha256": "92d7f5be5e1ef21ee682f574b35cc01de4dd8ab16b8aa6edf0dfbdfcfcfabba3",
|
| 6 |
+
"source_release_manifest_sha256": "3e11cc6011860ea245c53eb44b9da61a85d5feb5c6caa5da95428d63ede6d712",
|
| 7 |
+
"source_api_sha256": "1b068eccdffd3c3b67bfa52f8f526e6b482d92551c927668ad28a794767f8a40",
|
| 8 |
+
"candidate_api_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152",
|
| 9 |
+
"changed_inference_files": [
|
| 10 |
+
"code/decision_api.py"
|
| 11 |
+
],
|
| 12 |
+
"held_fixed": [
|
| 13 |
+
"weights",
|
| 14 |
+
"tokenizer",
|
| 15 |
+
"prompt",
|
| 16 |
+
"temperature",
|
| 17 |
+
"normalization_profile",
|
| 18 |
+
"BF16_backbone_FP32_head",
|
| 19 |
+
"batch8",
|
| 20 |
+
"input_limit16384",
|
| 21 |
+
"public_wrapper"
|
| 22 |
+
],
|
| 23 |
+
"shared_state_neural_cache": false,
|
| 24 |
+
"cross_request_cache": false,
|
| 25 |
+
"timing_evidence": {
|
| 26 |
+
"analysis/decoder-joint-serving-v1/COMPLETED-PARITY.json": "4187fb76432eb69b263cb6d5ad55f10a09aef384fe405f3ad204acf1edc761b5",
|
| 27 |
+
"analysis/decoder-joint-timing-v1/INDEPENDENTLY-REVIEWED.json": "1b416c8556c2305ffd18bd1e3cd523964820a5e2a3029cf447f20b2829ba82f7",
|
| 28 |
+
"results/joint-serving-timing-v1/summary/SUMMARY.json": "501f25efd1813b566d77e98278229440b7a3b4a2358def1462c0171a33e9f3be",
|
| 29 |
+
"analysis/decoder4b/joint-timing-actual-review-v1/REVIEW.json": "e94fec9d9c6d487200ef7ea1cf82319b33c4836c08d83f0b856bf8590f2fbee6"
|
| 30 |
+
},
|
| 31 |
+
"candidate_default_entrypoint_proof_required": true,
|
| 32 |
+
"published": false,
|
| 33 |
+
"adoption_authorized": false
|
| 34 |
+
}
|
bundle-manifest.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
{
|
| 2 |
"format": "research-pointer-bundle-v1",
|
| 3 |
-
"status": "
|
| 4 |
"files": [
|
| 5 |
{
|
| 6 |
"file": "NORMALIZATION_RUNTIME.md",
|
|
@@ -12,6 +12,11 @@
|
|
| 12 |
"bytes": 2621,
|
| 13 |
"sha256": "fb76d9fc9147e15a678eb91dfc40d7039845a4eaebdecce041a5c9e76c3d3e0f"
|
| 14 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 15 |
{
|
| 16 |
"file": "SOURCE_BUNDLE_MANIFEST.json",
|
| 17 |
"bytes": 5637,
|
|
@@ -49,8 +54,8 @@
|
|
| 49 |
},
|
| 50 |
{
|
| 51 |
"file": "code/decision_api.py",
|
| 52 |
-
"bytes":
|
| 53 |
-
"sha256": "
|
| 54 |
},
|
| 55 |
{
|
| 56 |
"file": "code/decision_model.py",
|
|
@@ -216,7 +221,7 @@
|
|
| 216 |
}
|
| 217 |
],
|
| 218 |
"source_model_code_sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646",
|
| 219 |
-
"source_api_code_sha256": "
|
| 220 |
"dev_sha256": "45b4cd46acc4be53b95c7ed8f333b3533972296a4d0557c7b8c48c9cab6ced61",
|
| 221 |
"production_predictions_sha256": "ba311f5a6bc73182651bdf5c9744f0ed95921110501ddc93dcfd20a1d06bceff",
|
| 222 |
"temperature_sha256": "69d80e5b215e2c1e6872f146fdb7ea5aa95c9fd678ef96208960d50e4e7b635d",
|
|
@@ -227,5 +232,7 @@
|
|
| 227 |
"no_publication_performed": true,
|
| 228 |
"source_bundle_manifest_sha256": "d437bc0149bcb8c9891fbb33c5abc4c2336c3a89ab6cfa0981da5a7f1d19f1f4",
|
| 229 |
"normalization_profile_sha256": "be32858d15233e0a3fbee0e4257fb02be0b3439deee4eb9c3f61151df7b73850",
|
| 230 |
-
"public_wrapper_included": true
|
|
|
|
|
|
|
| 231 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"format": "research-pointer-bundle-v1",
|
| 3 |
+
"status": "joint-serving-candidate-awaiting-default-public-parity",
|
| 4 |
"files": [
|
| 5 |
{
|
| 6 |
"file": "NORMALIZATION_RUNTIME.md",
|
|
|
|
| 12 |
"bytes": 2621,
|
| 13 |
"sha256": "fb76d9fc9147e15a678eb91dfc40d7039845a4eaebdecce041a5c9e76c3d3e0f"
|
| 14 |
},
|
| 15 |
+
{
|
| 16 |
+
"file": "SERVING_OPTIMIZATION.json",
|
| 17 |
+
"bytes": 1548,
|
| 18 |
+
"sha256": "fe2fd8c3e046faec139b8685eed693db0e6aa5235ebe5f860bc9f6dd1888c8ad"
|
| 19 |
+
},
|
| 20 |
{
|
| 21 |
"file": "SOURCE_BUNDLE_MANIFEST.json",
|
| 22 |
"bytes": 5637,
|
|
|
|
| 54 |
},
|
| 55 |
{
|
| 56 |
"file": "code/decision_api.py",
|
| 57 |
+
"bytes": 10952,
|
| 58 |
+
"sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152"
|
| 59 |
},
|
| 60 |
{
|
| 61 |
"file": "code/decision_model.py",
|
|
|
|
| 221 |
}
|
| 222 |
],
|
| 223 |
"source_model_code_sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646",
|
| 224 |
+
"source_api_code_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152",
|
| 225 |
"dev_sha256": "45b4cd46acc4be53b95c7ed8f333b3533972296a4d0557c7b8c48c9cab6ced61",
|
| 226 |
"production_predictions_sha256": "ba311f5a6bc73182651bdf5c9744f0ed95921110501ddc93dcfd20a1d06bceff",
|
| 227 |
"temperature_sha256": "69d80e5b215e2c1e6872f146fdb7ea5aa95c9fd678ef96208960d50e4e7b635d",
|
|
|
|
| 232 |
"no_publication_performed": true,
|
| 233 |
"source_bundle_manifest_sha256": "d437bc0149bcb8c9891fbb33c5abc4c2336c3a89ab6cfa0981da5a7f1d19f1f4",
|
| 234 |
"normalization_profile_sha256": "be32858d15233e0a3fbee0e4257fb02be0b3439deee4eb9c3f61151df7b73850",
|
| 235 |
+
"public_wrapper_included": true,
|
| 236 |
+
"source_published_bundle_manifest_sha256": "92d7f5be5e1ef21ee682f574b35cc01de4dd8ab16b8aa6edf0dfbdfcfcfabba3",
|
| 237 |
+
"serving_optimization_sha256": "fe2fd8c3e046faec139b8685eed693db0e6aa5235ebe5f860bc9f6dd1888c8ad"
|
| 238 |
}
|
code/decision_api.py
CHANGED
|
@@ -102,7 +102,7 @@ class DecisionEngine:
|
|
| 102 |
|
| 103 |
def predict_rows(self, rows):
|
| 104 |
import torch
|
| 105 |
-
encoded=
|
| 106 |
pad=self.tokenizer.pad_token_id if self.tokenizer.pad_token_id is not None else self.tokenizer.eos_token_id
|
| 107 |
records=[]
|
| 108 |
with torch.inference_mode():
|
|
@@ -111,17 +111,28 @@ class DecisionEngine:
|
|
| 111 |
batch={key:value.to(self.device) if torch.is_tensor(value) else value
|
| 112 |
for key,value in self.module.collate(items,pad).items()}
|
| 113 |
with torch.autocast('cuda',dtype=torch.bfloat16):logits=self.model(**batch)
|
|
|
|
|
|
|
|
|
|
| 114 |
for row,item,values in zip(rows[start:start+self.batch_size],items,logits):
|
| 115 |
k=len(row['options']);values=values[:k].float()
|
| 116 |
temperature=self.temperatures.get(row['task_type'],1.)
|
| 117 |
-
probabilities=(values/temperature).softmax(-1)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 118 |
answer=typed_answer(row,probabilities)
|
| 119 |
prediction=max(range(k),key=probabilities.__getitem__)
|
| 120 |
if row['task_type']=='noul':
|
| 121 |
chosen='true' if answer['noul']>=.5 else 'false'
|
| 122 |
prediction=[o['key'] for o in row['options']].index(chosen)
|
| 123 |
rec={'id':row['id'],'status':'ok','prediction':prediction,
|
| 124 |
-
'probabilities':probabilities,'logits':values
|
| 125 |
'native_contract':True,'truncated':False,'input_tokens':len(item['ids']),
|
| 126 |
'prompt_sha256':item['prompt_sha256'],'answer':answer}
|
| 127 |
if row['task_type']=='noul':rec['native_noul']=answer['noul']
|
|
@@ -137,3 +148,51 @@ class DecisionEngine:
|
|
| 137 |
result=self.predict_rows(rows)
|
| 138 |
return {'model':self.model_name,'answers':{r['id']:r['answer'] for r in result},
|
| 139 |
'usage':{'input_tokens':sum(r['input_tokens'] for r in result),'scored_questions':len(result)}}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 102 |
|
| 103 |
def predict_rows(self, rows):
|
| 104 |
import torch
|
| 105 |
+
encoded=encode_request(rows,self.tokenizer,self.module,self.max_length)
|
| 106 |
pad=self.tokenizer.pad_token_id if self.tokenizer.pad_token_id is not None else self.tokenizer.eos_token_id
|
| 107 |
records=[]
|
| 108 |
with torch.inference_mode():
|
|
|
|
| 111 |
batch={key:value.to(self.device) if torch.is_tensor(value) else value
|
| 112 |
for key,value in self.module.collate(items,pad).items()}
|
| 113 |
with torch.autocast('cuda',dtype=torch.bfloat16):logits=self.model(**batch)
|
| 114 |
+
# Preserve each row's original float/temperature/softmax math,
|
| 115 |
+
# but defer host synchronization until the complete batch.
|
| 116 |
+
staged=[];transfers=[]
|
| 117 |
for row,item,values in zip(rows[start:start+self.batch_size],items,logits):
|
| 118 |
k=len(row['options']);values=values[:k].float()
|
| 119 |
temperature=self.temperatures.get(row['task_type'],1.)
|
| 120 |
+
probabilities=(values/temperature).softmax(-1)
|
| 121 |
+
staged.append((row,item,k,temperature))
|
| 122 |
+
transfers.extend((values,probabilities))
|
| 123 |
+
host_values=torch.cat(transfers).tolist()
|
| 124 |
+
offset=0
|
| 125 |
+
for row,item,k,temperature in staged:
|
| 126 |
+
values=host_values[offset:offset+k]
|
| 127 |
+
probabilities=host_values[offset+k:offset+2*k]
|
| 128 |
+
offset+=2*k
|
| 129 |
answer=typed_answer(row,probabilities)
|
| 130 |
prediction=max(range(k),key=probabilities.__getitem__)
|
| 131 |
if row['task_type']=='noul':
|
| 132 |
chosen='true' if answer['noul']>=.5 else 'false'
|
| 133 |
prediction=[o['key'] for o in row['options']].index(chosen)
|
| 134 |
rec={'id':row['id'],'status':'ok','prediction':prediction,
|
| 135 |
+
'probabilities':probabilities,'logits':values,'temperature':temperature,
|
| 136 |
'native_contract':True,'truncated':False,'input_tokens':len(item['ids']),
|
| 137 |
'prompt_sha256':item['prompt_sha256'],'answer':answer}
|
| 138 |
if row['task_type']=='noul':rec['native_noul']=answer['noul']
|
|
|
|
| 148 |
result=self.predict_rows(rows)
|
| 149 |
return {'model':self.model_name,'answers':{r['id']:r['answer'] for r in result},
|
| 150 |
'usage':{'input_tokens':sum(r['input_tokens'] for r in result),'scored_questions':len(result)}}
|
| 151 |
+
|
| 152 |
+
|
| 153 |
+
"""Experimental request-local exact-segment tokenization.
|
| 154 |
+
|
| 155 |
+
Original encode/segments functions remain authoritative. Batch tokenize exact
|
| 156 |
+
whole segments, never split a BPE prefix at a new boundary. No cross-request
|
| 157 |
+
cache, GPU change, prompt change or change to the eight-row inference groups.
|
| 158 |
+
"""
|
| 159 |
+
|
| 160 |
+
|
| 161 |
+
class SegmentLookup:
|
| 162 |
+
def __init__(self, tokenizer, cache):
|
| 163 |
+
self.tokenizer, self.cache = tokenizer, cache
|
| 164 |
+
|
| 165 |
+
def encode(self, text, **kwargs):
|
| 166 |
+
if kwargs == {'add_special_tokens': False} and text in self.cache:
|
| 167 |
+
# Original encode extends its prefix list in place.
|
| 168 |
+
return list(self.cache[text])
|
| 169 |
+
return self.tokenizer.encode(text, **kwargs)
|
| 170 |
+
|
| 171 |
+
|
| 172 |
+
def encode_request(rows, tokenizer, module, max_length=16384,
|
| 173 |
+
max_cached_characters=8_000_000, segment_batch_size=64):
|
| 174 |
+
if max_cached_characters < 0 or segment_batch_size < 1:
|
| 175 |
+
raise ValueError('Invalid tokenizer resource bound')
|
| 176 |
+
unique = {}
|
| 177 |
+
characters = 0
|
| 178 |
+
for row in rows:
|
| 179 |
+
prefix, options, suffix = module.segments(row)
|
| 180 |
+
for segment in (prefix, *options, suffix):
|
| 181 |
+
if segment not in unique:
|
| 182 |
+
unique[segment] = None
|
| 183 |
+
characters += len(segment)
|
| 184 |
+
if characters > max_cached_characters:
|
| 185 |
+
# Preserve the original behavior under the resource cap.
|
| 186 |
+
return [module.encode(r, tokenizer, max_length) for r in rows]
|
| 187 |
+
strings = list(unique)
|
| 188 |
+
for start in range(0, len(strings), segment_batch_size):
|
| 189 |
+
batch = strings[start:start + segment_batch_size]
|
| 190 |
+
result = tokenizer(batch, add_special_tokens=False, padding=False,
|
| 191 |
+
truncation=False, return_attention_mask=False,
|
| 192 |
+
return_token_type_ids=False)['input_ids']
|
| 193 |
+
if len(result) != len(batch):
|
| 194 |
+
raise ValueError('Batch tokenizer output count differs')
|
| 195 |
+
for segment, ids in zip(batch, result):
|
| 196 |
+
unique[segment] = tuple(ids)
|
| 197 |
+
lookup = SegmentLookup(tokenizer, unique)
|
| 198 |
+
return [module.encode(row, lookup, max_length) for row in rows]
|
model-card-example.json
CHANGED
|
@@ -65,7 +65,7 @@
|
|
| 65 |
"direct_engine_exact_response": true,
|
| 66 |
"overflow_rejected": true,
|
| 67 |
"overflow_message": "check: 40083 tokens exceeds max_length=16384; no truncation allowed",
|
| 68 |
-
"bundle_manifest_sha256": "
|
| 69 |
"runtime": {
|
| 70 |
"actual": {
|
| 71 |
"torch": "2.12.0+git6bbd260",
|
|
|
|
| 65 |
"direct_engine_exact_response": true,
|
| 66 |
"overflow_rejected": true,
|
| 67 |
"overflow_message": "check: 40083 tokens exceeds max_length=16384; no truncation allowed",
|
| 68 |
+
"bundle_manifest_sha256": "83876db506b2d98e3e8ce7d34310b21f97bac30d4f5aef3371053798bff08830",
|
| 69 |
"runtime": {
|
| 70 |
"actual": {
|
| 71 |
"torch": "2.12.0+git6bbd260",
|
release-manifest.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
| 1 |
{
|
| 2 |
"format": "decision-public-release-v1",
|
| 3 |
-
"status": "assembled-
|
| 4 |
-
"bundle_manifest_sha256": "
|
| 5 |
-
"readiness_sha256": "
|
| 6 |
"model_card_sha256": "4ec3943c766b36eed886ac2713363dbcbc5b736b942c1e38f13cb1ab5f5758cd",
|
| 7 |
"repo_id": "llm-semantic-router/Decision-1.0-Nox",
|
| 8 |
"assembly_script_sha256": "7831ef5952b8256e585d4efa91cac4ce87738f4eb93c84b1cf1e7250def1f430",
|
| 9 |
-
"original_bundle_manifest_preserved":
|
| 10 |
"files_exclude_this_manifest": true,
|
| 11 |
"files": [
|
| 12 |
{
|
|
@@ -54,6 +54,11 @@
|
|
| 54 |
"bytes": 5001,
|
| 55 |
"sha256": "4ec3943c766b36eed886ac2713363dbcbc5b736b942c1e38f13cb1ab5f5758cd"
|
| 56 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 57 |
{
|
| 58 |
"file": "RUNTIME.md",
|
| 59 |
"bytes": 5205,
|
|
@@ -64,6 +69,11 @@
|
|
| 64 |
"bytes": 2621,
|
| 65 |
"sha256": "fb76d9fc9147e15a678eb91dfc40d7039845a4eaebdecce041a5c9e76c3d3e0f"
|
| 66 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 67 |
{
|
| 68 |
"file": "SOURCE_BUNDLE_MANIFEST.json",
|
| 69 |
"bytes": 5637,
|
|
@@ -261,8 +271,8 @@
|
|
| 261 |
},
|
| 262 |
{
|
| 263 |
"file": "bundle-manifest.json",
|
| 264 |
-
"bytes":
|
| 265 |
-
"sha256": "
|
| 266 |
},
|
| 267 |
{
|
| 268 |
"file": "chat_template.jinja",
|
|
@@ -271,8 +281,8 @@
|
|
| 271 |
},
|
| 272 |
{
|
| 273 |
"file": "code/decision_api.py",
|
| 274 |
-
"bytes":
|
| 275 |
-
"sha256": "
|
| 276 |
},
|
| 277 |
{
|
| 278 |
"file": "code/decision_model.py",
|
|
@@ -342,7 +352,7 @@
|
|
| 342 |
{
|
| 343 |
"file": "model-card-example.json",
|
| 344 |
"bytes": 3775,
|
| 345 |
-
"sha256": "
|
| 346 |
},
|
| 347 |
{
|
| 348 |
"file": "pyproject.toml",
|
|
@@ -492,5 +502,9 @@
|
|
| 492 |
"MATERIALS.json": "copy",
|
| 493 |
"metrics/semantic-consistency.json": "copy"
|
| 494 |
},
|
| 495 |
-
"scope": "Inference weights, original numerical code, calibrated runtime metadata, wrapper, model card, license, attribution and approved assets only."
|
|
|
|
|
|
|
|
|
|
|
|
|
| 496 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"format": "decision-public-release-v1",
|
| 3 |
+
"status": "runtime-patch-assembled-before-staged-download-proof",
|
| 4 |
+
"bundle_manifest_sha256": "83876db506b2d98e3e8ce7d34310b21f97bac30d4f5aef3371053798bff08830",
|
| 5 |
+
"readiness_sha256": "97c0c58b97685f95ad57a65b378f4e4f374ff3f7a7e3f7ef37c7960131981ffe",
|
| 6 |
"model_card_sha256": "4ec3943c766b36eed886ac2713363dbcbc5b736b942c1e38f13cb1ab5f5758cd",
|
| 7 |
"repo_id": "llm-semantic-router/Decision-1.0-Nox",
|
| 8 |
"assembly_script_sha256": "7831ef5952b8256e585d4efa91cac4ce87738f4eb93c84b1cf1e7250def1f430",
|
| 9 |
+
"original_bundle_manifest_preserved": false,
|
| 10 |
"files_exclude_this_manifest": true,
|
| 11 |
"files": [
|
| 12 |
{
|
|
|
|
| 54 |
"bytes": 5001,
|
| 55 |
"sha256": "4ec3943c766b36eed886ac2713363dbcbc5b736b942c1e38f13cb1ab5f5758cd"
|
| 56 |
},
|
| 57 |
+
{
|
| 58 |
+
"file": "RUNTIME-RELEASE.json",
|
| 59 |
+
"bytes": 1264,
|
| 60 |
+
"sha256": "97c0c58b97685f95ad57a65b378f4e4f374ff3f7a7e3f7ef37c7960131981ffe"
|
| 61 |
+
},
|
| 62 |
{
|
| 63 |
"file": "RUNTIME.md",
|
| 64 |
"bytes": 5205,
|
|
|
|
| 69 |
"bytes": 2621,
|
| 70 |
"sha256": "fb76d9fc9147e15a678eb91dfc40d7039845a4eaebdecce041a5c9e76c3d3e0f"
|
| 71 |
},
|
| 72 |
+
{
|
| 73 |
+
"file": "SERVING_OPTIMIZATION.json",
|
| 74 |
+
"bytes": 1548,
|
| 75 |
+
"sha256": "fe2fd8c3e046faec139b8685eed693db0e6aa5235ebe5f860bc9f6dd1888c8ad"
|
| 76 |
+
},
|
| 77 |
{
|
| 78 |
"file": "SOURCE_BUNDLE_MANIFEST.json",
|
| 79 |
"bytes": 5637,
|
|
|
|
| 271 |
},
|
| 272 |
{
|
| 273 |
"file": "bundle-manifest.json",
|
| 274 |
+
"bytes": 7992,
|
| 275 |
+
"sha256": "83876db506b2d98e3e8ce7d34310b21f97bac30d4f5aef3371053798bff08830"
|
| 276 |
},
|
| 277 |
{
|
| 278 |
"file": "chat_template.jinja",
|
|
|
|
| 281 |
},
|
| 282 |
{
|
| 283 |
"file": "code/decision_api.py",
|
| 284 |
+
"bytes": 10952,
|
| 285 |
+
"sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152"
|
| 286 |
},
|
| 287 |
{
|
| 288 |
"file": "code/decision_model.py",
|
|
|
|
| 352 |
{
|
| 353 |
"file": "model-card-example.json",
|
| 354 |
"bytes": 3775,
|
| 355 |
+
"sha256": "6bb34ba80660291be6c53005571a8a1d6546852bca186fff5da915019195d3a9"
|
| 356 |
},
|
| 357 |
{
|
| 358 |
"file": "pyproject.toml",
|
|
|
|
| 502 |
"MATERIALS.json": "copy",
|
| 503 |
"metrics/semantic-consistency.json": "copy"
|
| 504 |
},
|
| 505 |
+
"scope": "Inference weights, original numerical code, calibrated runtime metadata, wrapper, model card, license, attribution and approved assets only.",
|
| 506 |
+
"release_tag": "v1.3.1",
|
| 507 |
+
"change_kind": "runtime_only",
|
| 508 |
+
"previous_main_revision": "3d42ac931bba5726374d9c898e5df70db44ffea1",
|
| 509 |
+
"previous_release_manifest_sha256": "c8a9ce88ed6312a7a822c0ef6d70f7ebceb06fadb323543fdfe494b8bdbba50d"
|
| 510 |
}
|