{ "manifest_schema": "0.1.3", "repo": "litert-community/Shieldstral-1.0-3B", "generated": "2026-10-07", "generator": "make_manifest.py", "model": { "display_name": "Shieldstral-1.0-3B", "base_model": "mistralai/Shieldstral-1.0-3B", "architecture": "Mistral3ForConditionalGeneration — pixtral vision tower plus a Ministral3 text decoder; a yes/no safety classifier over text and images", "parameters_b": 3.0, "license": "apache-2.0", "context_length": 4096, "capabilities": { "vision": true, "audio": false, "thinking": { "declared": false, "control": "never" } } }, "variants": [ { "file": "Shieldstral-1.0-3B-vision_int4.litertlm", "sha256": "3f9289463889fe1232da2bee6ab133804b3e5a381808b00dc1b29e9e36b877bc", "size_bytes": 2783331824, "sections": [ { "type": "LlmMetadataProto", "size_bytes": 1175 }, { "type": "HF_Tokenizer_Zlib", "size_bytes": 2806635 }, { "type": "TFLiteModel", "size_bytes": 404228320, "model_type": "tf_lite_embedder" }, { "type": "TFLiteModel", "size_bytes": 1950043824, "model_type": "tf_lite_prefill_decode" }, { "type": "TFLiteModel", "size_bytes": 409057104, "model_type": "tf_lite_vision_encoder" }, { "type": "TFLiteModel", "size_bytes": 17122800, "model_type": "tf_lite_vision_adapter" } ], "quantization": "int4 blockwise-32 (OCTAV) decoder + int8 embedding, int8 pixtral tower, static 560x560", "backends": [ "cpu" ], "measured": [ { "device": "Galaxy S26 (SM-S942Q, Qualcomm SM8850, Adreno)", "backend": "cpu", "runtime": "litert_lm_advanced_main v0.16.0 (XNNPACK CPU)", "prompt_tokens": 231, "decode_tokens": 2, "prefill_tps": "30.7-33.2", "ttft_s": "7.48-8.1", "load_s": 10.1, "peak_memory_mb": 3692, "cache": "no", "runs": 2, "date": "2026-09-05", "source": "S7 Android backfill: same-device CPU control - identical binary, prompt file and flags, only --backend differs; 205-token prompt with --benchmark, 2 cold runs (compile/weight caches deleted between runs, device cooled below 42 C before each run); the GPU path does not run for this file on this handset, so this row has no GPU counterpart; the model stopped after 2 decode token(s) on this generic prompt (task-prompted model), so no decode speed is reported from this run" } ], "default_backend": "cpu", "known_issues": [ "Cannot create a GPU engine: litert-torch 0.9.2 marks the attention softmax as an odml.softmax StableHLO composite and litert-converter 0.3.0 cannot lower it, so the delegate takes 52 of 1187 ops on the first subgraph and the engine is refused. Use Shieldstral-1.0-3B-vision_int4_gpu.litertlm for the same weights on GPU." ], "min_runtime_version": "0.15.0" }, { "file": "Shieldstral-1.0-3B-vision_int4_gpu.litertlm", "sha256": "9dcd46ff64ba528bf57e6153a97c068364efc48c14f36eeb7cada9a56dea7a5b", "size_bytes": 2783086064, "sections": [ { "type": "LlmMetadataProto", "size_bytes": 1175 }, { "type": "HF_Tokenizer_Zlib", "size_bytes": 2806635 }, { "type": "TFLiteModel", "size_bytes": 404228320, "model_type": "tf_lite_embedder" }, { "type": "TFLiteModel", "size_bytes": 1949796112, "model_type": "tf_lite_prefill_decode" }, { "type": "TFLiteModel", "size_bytes": 409057104, "model_type": "tf_lite_vision_encoder" }, { "type": "TFLiteModel", "size_bytes": 17122800, "model_type": "tf_lite_vision_adapter" } ], "quantization": "int4 blockwise-32 (OCTAV) decoder + int8 embedding, int8 pixtral tower, static 560x560", "backends": [ "cpu", "gpu" ], "default_backend": "gpu", "requirements": { "platform_notes": [ "vision_backend must be passed explicitly: without it the engine loads, the conversation is created, and only the first image message fails", "The scoring API is text-only, so image documents get the generated verdict but no continuous score", "Same weights and same pixtral tower as Shieldstral-1.0-3B-vision_int4.litertlm; only the decoder was re-exported" ] }, "measured": [ { "device": "Galaxy S26 (SM-S942Q, Qualcomm SM8850, Adreno)", "backend": "gpu", "runtime": "litert_lm_advanced_main v0.16.0, LiteRT CL delegate", "prompt_tokens": 205, "prefill_tps": 294.2, "decode_tps": 10.8, "ttft_s": 0.88, "runs": 2, "date": "2026-08-28", "source": "S5 Adreno gate, 205-token benchmark; full delegation 15202/15202 across 13 subgraphs, gate peak 1493 MB. No same-device CPU control was taken, so this is a GPU speed and not a reason to prefer GPU over CPU." }, { "device": "Galaxy S26 (SM-S942Q, Qualcomm SM8850, Adreno)", "backend": "cpu", "runtime": "litert_lm_advanced_main v0.16.0 (XNNPACK CPU)", "prompt_tokens": 231, "decode_tokens": 2, "prefill_tps": "20.1-23.2", "ttft_s": "10.71-12.27", "load_s": 13.0, "peak_memory_mb": 3633, "cache": "no", "runs": 2, "date": "2026-09-05", "source": "S7 Android backfill: same-device CPU control - identical binary, prompt file and flags, only --backend differs; 205-token prompt with --benchmark, 2 cold runs (compile/weight caches deleted between runs, device cooled below 42 C before each run); same-device control for an existing GPU row on this handset; the model stopped after 2 decode token(s) on this generic prompt (task-prompted model), so no decode speed is reported from this run" } ], "known_issues": [ "The 100-image letterboxed parity gate was re-run on 2026-08-28: this file returns the same verdict on all 100 images as Shieldstral-1.0-3B-vision_int4.litertlm and as the 2026-08-11 record (accuracy 0.77, F1 0.736, zero flips), both files scored in the same session on the same runtime so the older one is the control. The two device-side margin probes were NOT re-run — the images they used are not archived." ], "min_runtime_version": "0.15.0" }, { "file": "Shieldstral-1.0-3B_int4.litertlm", "sha256": "271ed72950196a064d72947bff68a4080233aa34aa88c24ae2d28ffc83d413dc", "size_bytes": 2358595568, "sections": [ { "type": "LlmMetadataProto", "size_bytes": 357 }, { "type": "HF_Tokenizer_Zlib", "size_bytes": 2806635 }, { "type": "TFLiteModel", "size_bytes": 1949933824, "model_type": "tf_lite_prefill_decode" }, { "type": "TFLiteModel", "size_bytes": 405802992, "model_type": "tf_lite_embedder" } ], "quantization": "int4 blockwise-32 (OCTAV) + int8 embedding, externalised embedder, text only", "backends": [ "cpu", "gpu" ], "default_backend": "gpu", "recommended": [ { "platform": "android", "device_class": "flagship", "backend": "gpu", "reason": "the Android path verified on a Galaxy S26 (S4 gate 2026-08-24): full delegation (15202 ops across 13 subgraphs) and generation; 9.1 tok/s decode, engine init 28.5 s, peak 1401 MB; the repo's other S26-verified file (Shieldstral-1.0-3B_int8.litertlm) decodes 8.2 tok/s on the same handset. No same-device CPU control was taken, so this is the verified choice for the class, not a measured win over CPU" } ], "measured": [ { "device": "Galaxy S26 (SM-S942Q, Qualcomm SM8850, Adreno)", "backend": "gpu", "runtime": "litert_lm_advanced_main v0.16.0, LiteRT CL delegate", "prompt_tokens": 205, "prefill_tps": 298.7, "decode_tps": 9.12, "runs": 1, "date": "2026-08-24", "source": "S4 GPU gate, 205-token benchmark; full delegation 15202/15202 across 13 subgraphs" } ], "min_runtime_version": "0.15.0" }, { "file": "Shieldstral-1.0-3B_int8.litertlm", "sha256": "f950d2d172af44dddc200cdcc23af6646b6ab4e285f406f273f2dee902f7db3a", "size_bytes": 3978727408, "sections": [ { "type": "LlmMetadataProto", "size_bytes": 357 }, { "type": "HF_Tokenizer_Zlib", "size_bytes": 2806635 }, { "type": "TFLiteModel", "size_bytes": 3570066064, "model_type": "tf_lite_prefill_decode" }, { "type": "TFLiteModel", "size_bytes": 405802992, "model_type": "tf_lite_embedder" } ], "quantization": "export-time dynamic int8, externalised embedder, text only", "backends": [ "cpu", "gpu" ], "measured": [ { "device": "Galaxy S26 (SM-S942Q, Qualcomm SM8850, Adreno)", "backend": "gpu", "runtime": "litert_lm_advanced_main v0.16.0, LiteRT CL delegate", "prompt_tokens": 231, "decode_tokens": 2, "prefill_tps": 290.9, "decode_tps": 8.15, "ttft_s": 0.92, "load_s": 6.8, "peak_memory_mb": 1259, "runs": 1, "date": "2026-08-24", "source": "S4 GPU gate, 205-token benchmark, single run; full delegation (15202 ops across 13 subgraphs); generation gate answered the probe correctly. No same-device CPU control was taken, so this is a GPU speed, not a measured win over CPU." }, { "device": "Galaxy S26 (SM-S942Q, Qualcomm SM8850, Adreno)", "backend": "cpu", "runtime": "litert_lm_advanced_main v0.16.0 (XNNPACK CPU)", "prompt_tokens": 231, "decode_tokens": 2, "prefill_tps": "115.2-129.2", "ttft_s": "1.91-2.15", "load_s": 6.8, "peak_memory_mb": 5211, "cache": "no", "runs": 2, "date": "2026-09-05", "source": "S7 Android backfill: same-device CPU control - identical binary, prompt file and flags, only --backend differs; 205-token prompt with --benchmark, 2 cold runs (compile/weight caches deleted between runs, device cooled below 42 C before each run); same-device control for an existing GPU row on this handset; the model stopped after 2 decode token(s) on this generic prompt (task-prompted model), so no decode speed is reported from this run" } ], "default_backend": "cpu", "known_issues": [ "Its 3.33 GiB single section exceeds the practical iOS mmap budget; desktop and high-memory devices only." ], "min_runtime_version": "0.15.0" } ] }