Vela-1.0-Omni-Nano / benchmarks /inference-equivalence.json
Xunzhuo's picture
Update Hugging Face org references: llm-semantic-router → vllm-sr
f16c74f verified
Raw History Blame Contribute Delete
16.2 kB
{
"schema_version": 1,
"model": "vllm-sr/Vela-1.0-Omni-Nano",
"scope": "Same supported public encode_text, encode_image, and encode_audio computation; no new benchmark run.",
"evaluated_parent": {
"revision": "de02c367bc9aa7a6f7ad3e59fb38962502e021c1",
"native_artifact_sha256": "96d76f97478e114b816df3ab1b3c6947e4c5ef0babfb40ab865d7e3c838d34c4",
"parameters": 144554112
},
"applies_to": {
"revision": "50ce808197fb43a1b8913a66d4a47339ec66ecd10d52cbd9f9e752d02f8c43fa",
"native_artifact_sha256": "50ce808197fb43a1b8913a66d4a47339ec66ecd10d52cbd9f9e752d02f8c43fa",
"parameters": 135383808,
"revision_kind": "native_artifact_fingerprint"
},
"removed_unused_tensors": 95,
"removed_parameters": 9170304,
"removed_percent": 6.3438555106616406,
"retained_tensors": 476,
"retained_values_shapes_and_dtypes_exact": true,
"removed_tensor_keys": [
"audio_encoder.layer_norms.0.bias",
"audio_encoder.layer_norms.0.weight",
"audio_encoder.layer_norms.1.bias",
"audio_encoder.layer_norms.1.weight",
"audio_encoder.layer_norms.2.bias",
"audio_encoder.layer_norms.2.weight",
"audio_encoder.layer_norms.3.bias",
"audio_encoder.layer_norms.3.weight",
"audio_encoder.layer_projections.0.bias",
"audio_encoder.layer_projections.0.weight",
"audio_encoder.layer_projections.1.bias",
"audio_encoder.layer_projections.1.weight",
"audio_encoder.layer_projections.2.bias",
"audio_encoder.layer_projections.2.weight",
"audio_encoder.layer_projections.3.bias",
"audio_encoder.layer_projections.3.weight",
"fusion.fusion_layers.0.ffn.0.bias",
"fusion.fusion_layers.0.ffn.0.weight",
"fusion.fusion_layers.0.ffn.3.bias",
"fusion.fusion_layers.0.ffn.3.weight",
"fusion.fusion_layers.0.norm1.bias",
"fusion.fusion_layers.0.norm1.weight",
"fusion.fusion_layers.0.norm2.bias",
"fusion.fusion_layers.0.norm2.weight",
"fusion.fusion_layers.0.self_attention.in_proj_bias",
"fusion.fusion_layers.0.self_attention.in_proj_weight",
"fusion.fusion_layers.0.self_attention.out_proj.bias",
"fusion.fusion_layers.0.self_attention.out_proj.weight",
"fusion.fusion_layers.1.ffn.0.bias",
"fusion.fusion_layers.1.ffn.0.weight",
"fusion.fusion_layers.1.ffn.3.bias",
"fusion.fusion_layers.1.ffn.3.weight",
"fusion.fusion_layers.1.norm1.bias",
"fusion.fusion_layers.1.norm1.weight",
"fusion.fusion_layers.1.norm2.bias",
"fusion.fusion_layers.1.norm2.weight",
"fusion.fusion_layers.1.self_attention.in_proj_bias",
"fusion.fusion_layers.1.self_attention.in_proj_weight",
"fusion.fusion_layers.1.self_attention.out_proj.bias",
"fusion.fusion_layers.1.self_attention.out_proj.weight",
"fusion.input_projections.audio.projector.0.bias",
"fusion.input_projections.audio.projector.0.weight",
"fusion.input_projections.audio.projector.1.bias",
"fusion.input_projections.audio.projector.1.weight",
"fusion.input_projections.audio.projector.4.bias",
"fusion.input_projections.audio.projector.4.weight",
"fusion.input_projections.image.projector.0.bias",
"fusion.input_projections.image.projector.0.weight",
"fusion.input_projections.image.projector.1.bias",
"fusion.input_projections.image.projector.1.weight",
"fusion.input_projections.image.projector.4.bias",
"fusion.input_projections.image.projector.4.weight",
"fusion.input_projections.text.projector.0.bias",
"fusion.input_projections.text.projector.0.weight",
"fusion.input_projections.text.projector.1.bias",
"fusion.input_projections.text.projector.1.weight",
"fusion.input_projections.text.projector.4.bias",
"fusion.input_projections.text.projector.4.weight",
"fusion.layer_norms.0.bias",
"fusion.layer_norms.0.weight",
"fusion.layer_norms.1.bias",
"fusion.layer_norms.1.weight",
"fusion.layer_projections.0.bias",
"fusion.layer_projections.0.weight",
"fusion.layer_projections.1.bias",
"fusion.layer_projections.1.weight",
"fusion.modality_embeddings.weight",
"fusion.output_projection.bias",
"fusion.output_projection.weight",
"image_encoder.layer_projections.0.bias",
"image_encoder.layer_projections.0.weight",
"image_encoder.layer_projections.1.bias",
"image_encoder.layer_projections.1.weight",
"image_encoder.layer_projections.10.bias",
"image_encoder.layer_projections.10.weight",
"image_encoder.layer_projections.11.bias",
"image_encoder.layer_projections.11.weight",
"image_encoder.layer_projections.2.bias",
"image_encoder.layer_projections.2.weight",
"image_encoder.layer_projections.3.bias",
"image_encoder.layer_projections.3.weight",
"image_encoder.layer_projections.4.bias",
"image_encoder.layer_projections.4.weight",
"image_encoder.layer_projections.5.bias",
"image_encoder.layer_projections.5.weight",
"image_encoder.layer_projections.6.bias",
"image_encoder.layer_projections.6.weight",
"image_encoder.layer_projections.7.bias",
"image_encoder.layer_projections.7.weight",
"image_encoder.layer_projections.8.bias",
"image_encoder.layer_projections.8.weight",
"image_encoder.layer_projections.9.bias",
"image_encoder.layer_projections.9.weight",
"text_encoder.encoder.pooler.dense.bias",
"text_encoder.encoder.pooler.dense.weight"
],
"supported_public_api": [
"encode_text",
"encode_image",
"encode_audio"
],
"omitted_internal_features": [
"fusion",
"intermediate layer exits"
],
"quality_evidence": "Original common14, selected standard11, English41 and audio19 scores retain their measured artifact identities. English41 first applies from its original 299 artifact to the 96d parent through exact text identity; this public-inference bridge then applies those scores to the smaller package.",
"text_context_tokens": 512,
"embedding_dimensions": 384,
"CPU_synthetic_parity": {
"all_retained_tensors_exact": true,
"candidate_fingerprint": "50ce808197fb43a1b8913a66d4a47339ec66ecd10d52cbd9f9e752d02f8c43fa",
"device": "cpu",
"inputs": "deterministic synthetic fixtures, not benchmark quality evaluation",
"old_artifact_new_loader": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
2,
384
]
},
"parameters": 135383808,
"public_boundaries": {
"129": {
"comparison": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
1,
384
]
},
"tokens": 129
},
"512": {
"comparison": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
1,
384
]
},
"tokens": 512
},
"513_rejected": "Text exceeds 512 tokens; shorten or explicitly chunk it",
"audio_bad_rate_rejected": "Resample audio to 16000 Hz before encoding",
"audio_empty_rejected": "Audio must contain finite nonempty mono waveforms",
"audio_over30_rejected": "Audio inputs must be at most 30 seconds",
"text_empty_list_rejected": "Provide at least one text string"
},
"quality_scores_generated": false,
"same_loaded_object_removed_branches": {
"audio5": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
5,
384
]
},
"image4": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
4,
384
]
},
"text8": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
8,
384
]
},
"text_single": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
1,
384
]
}
},
"separate_loaded_native": {
"audio5": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
5,
384
]
},
"image4": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
4,
384
]
},
"text8": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
8,
384
]
},
"text_single": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
1,
384
]
}
},
"source_sha256": "28cb093c40cc9b042a9e9873fe1819721c93fc243da14c5ec1c5b654e79de95d",
"status": "passed",
"supported_public_api": [
"encode_text",
"encode_image",
"encode_audio"
],
"torch": "2.8.0+cpu",
"training_steps": 0,
"unsupported_internal_features": [
"fusion",
"intermediate layer exits"
]
},
"GPU_synthetic_parity": {
"all_retained_tensors_exact": true,
"candidate_fingerprint": "50ce808197fb43a1b8913a66d4a47339ec66ecd10d52cbd9f9e752d02f8c43fa",
"device": "cuda",
"inputs": "deterministic synthetic fixtures, not benchmark quality evaluation",
"old_artifact_new_loader": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
2,
384
]
},
"parameters": 135383808,
"public_boundaries": {
"129": {
"comparison": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
1,
384
]
},
"tokens": 129
},
"512": {
"comparison": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
1,
384
]
},
"tokens": 512
},
"513_rejected": "Text exceeds 512 tokens; shorten or explicitly chunk it",
"audio_bad_rate_rejected": "Resample audio to 16000 Hz before encoding",
"audio_empty_rejected": "Audio must contain finite nonempty mono waveforms",
"audio_over30_rejected": "Audio inputs must be at most 30 seconds",
"text_empty_list_rejected": "Provide at least one text string"
},
"quality_scores_generated": false,
"same_loaded_object_removed_branches": {
"audio5": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
5,
384
]
},
"image4": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
4,
384
]
},
"text8": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
8,
384
]
},
"text_single": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
1,
384
]
}
},
"separate_loaded_native": {
"audio5": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
5,
384
]
},
"image4": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
4,
384
]
},
"text8": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
8,
384
]
},
"text_single": {
"bitwise": true,
"maxabs": 0.0,
"relative_L2": 0.0,
"shape": [
1,
384
]
}
},
"source_sha256": "28cb093c40cc9b042a9e9873fe1819721c93fc243da14c5ec1c5b654e79de95d",
"status": "passed",
"supported_public_api": [
"encode_text",
"encode_image",
"encode_audio"
],
"torch": "2.12.0+git6bbd260",
"training_steps": 0,
"unsupported_internal_features": [
"fusion",
"intermediate layer exits"
]
},
"cost": {
"old_safetensors_bytes": 578289664,
"new_safetensors_bytes": 541598024,
"reduction_bytes": 36691640,
"old_FP32_parameter_bytes": 578216448,
"new_FP32_parameter_bytes": 541535232,
"scope": "Parameter storage only; not measured process peak memory or latency. Vector and index dimensions are unchanged."
},
"native_artifact_files_sha256": {
"LICENSE": "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4",
"NOTICE": "ea26e8a96e7088153afab5ab92d629a12c149d8ecb98d6795d9277b059f6a942",
"components/audio/config.json": "372c430053183035fb9d2d7079482f10053b6ff7d7367525ada9fdeead8adeac",
"components/audio/preprocessor_config.json": "9b5cd03a36fbb8a627c64d98a5b5b126ead95a77720723944487311f0110b666",
"components/image/config.json": "c2b09e8c0a60f3405d3e53897358a8a373bc4286234f30fcc9324d065fc13728",
"components/image/preprocessor_config.json": "5a0e5062d04603dc0e5eff01440f6bebfc66d396f9ca16e0128b31c5730cafc5",
"components/text/1_Pooling/config.json": "d1caf60c96f5fba2157c0c26b76d80818fad6cf0b8eb5e73ec372ff9818eba5c",
"components/text/config.json": "419ef27bf56a60adc670c50610f8099c64b23e4a1912f4ee7e321139518569b2",
"components/text/config_sentence_transformers.json": "940d5f50db195fa6e5e6a4f122c095f77880de259d74b14a65779ed48bdd7c56",
"components/text/modules.json": "84e40c8e006c9b1d6c122e02cba9b02458120b5fb0c87b746c41e0207cf642cf",
"components/text/sentence_bert_config.json": "84e39fda68ccbff05bfa723ae9c0e70e23e2ec373b76e0f8c6e71af72a693cbf",
"components/text/special_tokens_map.json": "5d5b662e421ea9fac075174bb0688ee0d9431699900b90662acd44b2a350503a",
"components/text/tokenizer.json": "d241a60d5e8f04cc1b2b3e9ef7a4921b27bf526d9f6050ab90f9267a1f9e5c66",
"components/text/tokenizer_config.json": "0b29c7bfc889e53b36d9dd3e686dd4300f6525110eaa98c76a5dafceb2029f53",
"components/text/vocab.txt": "07eced375cec144d27c900241f3e339478dec958f92fddbc551f295c992038a3",
"config.json": "3be38bb9a59ba29b719aa4885844fd9f353606b2118885e4171507ca65b72c4c",
"licenses/GIST-MODEL-CARD.md": "e2905b6dc66b39a697cd82f3211eea4a7ce45b17cc720eceae7fdd337b601c3c",
"model.safetensors": "401759e85cec305aca7ff08b31a1a3efa2c410b24b7bb49f9a81de3aabafa52d",
"omni_components/__init__.py": "543f72d073d77a71e7f5ad857f26d53abbecbbba1ed094336f5bf730af0bd641",
"omni_components/audio_encoder.py": "2c1e8e347c3e49a438084cc61f486dda9fde00d9f4b4d7caac828aa9b2f77488",
"omni_components/audio_io.py": "23392eb0784bd92e4b0176004455025faa0d9d47f6216e12da34079c49765546",
"omni_components/embedder.py": "ad709fd80bc1850f609443777ddf3327765f3ea330659f5c154f90fa8ab5e8fd",
"omni_components/fusion.py": "a3f6132f190f1533887a77570f4457a0da4eba698574b90888da477919d5c279",
"omni_components/image_encoder.py": "40e64a4fde0cb1a1416b764a1a8db47105985e3b979bac7b0aa3bc9d789d0f7a",
"omni_components/mini.py": "c62439716bcf06a0051a8d9ec7ea41f917816133f425a7fa004561279daac4fe",
"omni_components/records.py": "acb3a3cda9d260e39007b6aafd74efd002de974700e14a5535f66590b8b801d6",
"omni_components/residual_projection.py": "30028685698c9b4779c44f8ebe28bf80c1689b99594ca4b287db16d133227ad4",
"omni_components/single_modality.py": "a61edcdcac3dbad5c6b84101655e290fa5d74d7292bd8f61105772d9b2135171",
"omni_components/text_backbone.py": "c11a92203f3010ef857d2c809257781150a790acf1877ed83e1a0021449268c3",
"omni_components/text_encoder.py": "578329b6bf656423e0b9543783f30a6455449a7402fd3a1a9f601a8afc9705af",
"vela_omni.py": "a04bd0e9b163b0f440128a3086632d0e6223312e592f247f483c221210f08048"
},
"evidence_sha256": {
"BUILD.json": "d3701e04ca5a0949d4da29994a15ea69847c8fbf44fb7ce9757e1ccc1d051373",
"PARITY_CPU.json": "9d75c24759dc47776a1892b7b9599d2f7a242adb527707ec1054008a1e785540",
"PARITY_CUDA.json": "99f23ead7e05aea2c30e96361e4d617175fe7290563933c0203db45057938729"
},
"training_steps": 0,
"new_quality_scores": false,
"new_latency_claim": false
}