{ "manifest_schema": "0.1.3", "repo": "litert-community/Hy-MT2-1.8B", "generated": "2026-10-07", "generator": "make_manifest.py", "model": { "display_name": "Hy-MT2-1.8B", "base_model": "tencent/Hy-MT2-1.8B", "architecture": "Dense (HunYuanDenseV1ForCausalLM, 32L, 2.04B): GQA attention with QK-norm, hidden 2048, 16 query / 4 KV heads, head_dim 128, intermediate 6144, vocab 120818, tied embeddings. Multilingual translation model (33 languages, instruction-following translation, not open chat)", "parameters_b": 2.04, "license": "apache-2.0", "context_length": 4096, "capabilities": { "vision": false, "audio": false, "thinking": { "declared": false, "control": "never" } } }, "variants": [ { "file": "Hy-MT2-1.8B_int8.litertlm", "sha256": "275ff52bf04c4d577e1bf48733daa77d873d78b0637841574d7a10e1560a011a", "size_bytes": 1827026224, "sections": [ { "type": "LlmMetadataProto", "size_bytes": 1056 }, { "type": "HF_Tokenizer_Zlib", "size_bytes": 2006574 }, { "type": "TFLiteModel", "size_bytes": 1824978224, "model_type": "tf_lite_prefill_decode" } ], "quantization": "export-time dynamic int8 (linears + embedding)", "backends": [ "cpu", "gpu" ], "default_backend": "cpu", "recommended": [ { "platform": "android", "device_class": "flagship", "backend": "gpu", "reason": "measured on one Galaxy S26 (SM-S942Q, Android 16) with litert_lm_advanced_main from a LiteRT-LM main build of 2026-09-18: the OpenCL GPU delegate (LITERT_CL) takes every node of prefill_128 (1518/1518), prefill_16 (1518/1518) and decode (1390/1390), one partition each; 3 of 5 gate prompts were answered correctly. At a 1024-token prompt and 256 decode tokens, gpu decodes 26.29-27.46 tok/s (3 iterations in one process, cold init) against cpu's 13.75-14.99 tok/s (XNNPACK, 4 threads, weight cache, 2 iterations) on the same phone; prefill 576.9-580.2 vs 213.4-260.2 tok/s, peak (VmHWM) 0.83 vs 2.96 GB. Same-device CPU control taken - a measured win for gpu on decode" } ], "requirements": { "platform_notes": [ "Measured with litert-lm 0.17.1 (Mac) and litert_lm_advanced_main from a LiteRT-LM main build of 2026-09-18 (Galaxy S26); the bundle is a plain dense prefill/decode export (generic_model)", "Translation model: prompt with the source card's translation instructions (e.g. 'Translate the following text into . Note that you should **only output the translated result without any additional explanation**:'), not open chat", "Metadata start token deliberately dropped: the chat template renders <|hy_begin_of_sentence|> itself and the engine prepends the metadata start_token unconditionally (proved inside the runtime: [start_token]+prompt and [template-BOS]+prompt generate byte-identical greedy output), so the default export fed BOS twice. This file carries the training stream", "Source rope_scaling {type: dynamic, alpha: 1000} is baked statically at conversion (rope_theta 11158839.925); inv_freq and teacher-forced logits are bitwise-equal to the HF reference, valid to 262144 positions - far past this bundle's 4096 context" ] }, "measured": [ { "device": "Apple M4 Max", "os": "macOS 27.0", "backend": "gpu", "runtime": "litert-lm 0.17.1 (pip CLI benchmark), GPU fp16 activations (WebGPU on Metal)", "prompt_tokens": 256, "decode_tokens": 256, "prefill_tps": 2331.7, "decode_tps": 150.7, "ttft_s": 0.116, "load_s": 1.46, "cache": "no", "runs": 3, "date": "2026-10-05", "source": "GPU-graph re-ship card bench: 3 separate benchmark processes, median; the previous file measured in the same window: prefill 2013.7 tok/s, decode 103.5 tok/s, TTFT 0.137 s" }, { "device": "Apple M4 Max", "os": "macOS 27.0", "backend": "cpu", "runtime": "litert-lm 0.17.1 (pip CLI benchmark) (XNNPACK)", "prompt_tokens": 256, "decode_tokens": 256, "prefill_tps": 207.0, "decode_tps": 31.9, "ttft_s": 1.268, "load_s": 5.32, "cache": "no", "runs": 2, "date": "2026-10-05", "source": "GPU-graph re-ship card bench: 2 separate benchmark processes, median; the previous file measured in the same window: prefill 204.9 tok/s, decode 32.2 tok/s, TTFT 1.281 s" }, { "device": "Apple M4 Max", "os": "macOS 27.0", "backend": "gpu", "runtime": "litert-lm 0.17.1 (pip CLI benchmark), GPU fp16 activations (WebGPU on Metal)", "prompt_tokens": 16, "decode_tokens": 32, "prefill_tps": 683.8, "decode_tps": 137.2, "ttft_s": 0.031, "load_s": 1.44, "cache": "no", "runs": 3, "date": "2026-10-05", "source": "GPU-graph re-ship card bench (short-prompt cell, TTFT is the figure of interest): 3 separate benchmark processes, median; the previous file measured in the same window: prefill 242.9 tok/s, decode 97.5 tok/s, TTFT 0.076 s" }, { "device": "Apple M4 Max", "os": "macOS 27.0", "backend": "cpu", "runtime": "litert-lm 0.17.1 (pip CLI benchmark) (XNNPACK)", "prompt_tokens": 16, "decode_tokens": 32, "prefill_tps": 19.4, "decode_tps": 16.9, "ttft_s": 0.884, "load_s": 5.32, "cache": "no", "runs": 2, "date": "2026-10-05", "source": "GPU-graph re-ship card bench (short-prompt cell, TTFT is the figure of interest): 2 separate benchmark processes, median; the previous file measured in the same window: prefill 15.9 tok/s, decode 17.9 tok/s, TTFT 1.065 s" }, { "device": "Galaxy S26 (SM-S942Q, Qualcomm SM8850, Adreno)", "os": "Android 16", "backend": "gpu", "runtime": "litert_lm_advanced_main (LiteRT-LM main, 2026-09-18 build), LiteRT OpenCL delegate, fp16 activations", "prompt_tokens": 1024, "decode_tokens": 256, "prefill_tps": "576.9-580.2", "decode_tps": "26.29-27.46", "ttft_s": "1.8-1.81", "load_s": 4.94, "max_num_tokens": 1280, "cache": "no", "runs": 3, "date": "2026-10-05", "source": "GPU-graph re-ship device leg (round 20, turn T1): 3 iterations in one process; the handset throttles after the first iteration, so the range spans cold to warm; cold engine init (--disable_cache=true)", "peak_memory_mb": 787.6 }, { "device": "Galaxy S26 (SM-S942Q, Qualcomm SM8850, Adreno)", "os": "Android 16", "backend": "gpu", "runtime": "litert_lm_advanced_main (LiteRT-LM main, 2026-09-18 build), LiteRT OpenCL delegate, fp16 activations", "prompt_tokens": 16, "decode_tokens": 32, "prefill_tps": "317.7-364.2", "decode_tps": "20.1-26.04", "ttft_s": "0.08-0.1", "load_s": 4.91, "max_num_tokens": 1280, "cache": "no", "runs": 3, "date": "2026-10-05", "source": "GPU-graph re-ship device leg (round 20, turn T1): 3 iterations in one process; cold engine init (--disable_cache=true)", "peak_memory_mb": 787.5 }, { "device": "Galaxy S26 (SM-S942Q, Qualcomm SM8850, Adreno)", "os": "Android 16", "backend": "cpu", "runtime": "litert_lm_advanced_main (LiteRT-LM main, 2026-09-18 build), XNNPACK 4 threads, weight cache", "prompt_tokens": 1024, "decode_tokens": 256, "prefill_tps": "213.4-260.2", "decode_tps": "13.75-14.99", "ttft_s": "4.0-4.87", "load_s": 3.72, "max_num_tokens": 1280, "cache": "disk (XNNPACK weight cache)", "runs": 2, "date": "2026-10-05", "source": "GPU-graph re-ship device leg (round 20, turn T1): 2 iterations in one process; init includes writing the weight cache", "peak_memory_mb": 2821.1 } ], "known_issues": [ "8-question sanity gate scores 6/8 on BOTH backends with the same two misses ('Cool' for opposite-of-hot, 'pink' for the rhyme) - a property of this translation-tuned 1.8B, not of a backend. Arithmetic, factual and translation items are all correct; no degeneration", "Translation greedy A/B vs HF bf16 (source README default-translation prompt): byte-identical on 1 of 3 probes, fluent int8-class alternates on the other two (e.g. spectaculaire -> significative)" ] } ] }