"""Build data/mt/: the CPU translation model D-15 requires, converted once and committed. The Space translates Japanese to English with ``Helsinki-NLP/opus-mt-ja-en`` (MarianMT, 76 M parameters, Apache-2.0) converted to **CTranslate2 int8**. 02-RESEARCH.md § Q4 measured why: at two threads the CT2 model loads in 0.25-0.58 s (+103 MB) and translates a sentence in 16-129 ms, against 131-738 ms and +374 MB for transformers + torch - and transformers 5 removed ``pipeline("translation")`` outright. The Space therefore never imports transformers or torch on the translation path; it needs only ``ctranslate2`` and ``sentencepiece``, both pinned in pyproject.toml / requirements.txt (plan 02-01). This script is the reproducibility record for the committed artefact: 1. ``snapshot_download`` the model repo at ONE pinned commit (``OPUS_MT_REVISION``, resolved once with ``HfApi().model_info(MODEL_ID).sha`` on 2026-09-06 and hard-coded - the same discipline as ``KANJI_DATA_COMMIT`` in build_jlpt.py). The 303 MB TensorFlow checkpoint is not fetched. 2. Convert with ``ctranslate2.converters.TransformersConverter`` at ``quantization="int8"`` into ``data/mt/opus-mt-ja-en-ct2-int8/``, copy ``source.spm`` / ``target.spm`` beside it, and normalise the two JSON files the converter writes to LF (it uses the platform newline; the hashes below must be the bytes of every checkout, not of a Windows working copy). 3. **Read back** in the same run: load the converted model with CTranslate2, tokenise the probe sentences with plain ``sentencepiece`` on ``source.spm`` (+ ````) AND with ``transformers.MarianTokenizer`` from the snapshot, and require the two piece lists to be identical. That equality is the measured fact that lets the Space skip transformers. Then translate the probe and require ``station`` in the output. Nothing is recorded otherwise. 4. Fetch the Apache-2.0 text beside the model, write the NOTICE, and write ``data/mt/README.md`` with ``| file | bytes | sha256 |`` rows that ``tests/test_data_assets.py`` re-checks in the quick loop. Re-running is idempotent: CT2 conversion is deterministic for a fixed input and version, so the files are byte-identical and the README is left untouched when every hash still matches (its recorded date, versions and read-back numbers therefore describe the run that produced the committed bytes). The conversion needs transformers + torch, which are deliberately NOT project dependencies. Run it in a throwaway venv that is gitignored (``.venv-mt/``) and never referenced from pyproject.toml:: uv venv .venv-mt --python 3.12 uv pip install --python .venv-mt ctranslate2==4.8.2 sentencepiece==0.2.2 \\ huggingface_hub==1.28.0 transformers==5.16.1 torch \\ --index-url https://download.pytorch.org/whl/cpu --extra-index-url https://pypi.org/simple .venv-mt/Scripts/python.exe scripts/convert_mt.py (If the pytorch CPU index cannot resolve the other packages, install torch first from the CPU index and the rest from PyPI in a second ``uv pip install``.) Alternative not taken (research § Open Questions 2): upload the conversion to an owner model repo (``WolfDavid/opus-mt-ja-en-ct2-int8``) and ``snapshot_download`` it in the translator warm-up. That keeps ~80 MB out of the Space repo but re-downloads it on every container start; LFS in-repo puts the model on disk with the clone, so the first EN reveal never waits on a download. """ from __future__ import annotations import hashlib import shutil import sys import time import urllib.request from datetime import UTC, datetime from pathlib import Path REPO_ROOT = Path(__file__).resolve().parent.parent MT_DIR = REPO_ROOT / "data" / "mt" OUT_DIR = MT_DIR / "opus-mt-ja-en-ct2-int8" README = MT_DIR / "README.md" LICENSE_FILE = MT_DIR / "LICENSE-apache-2.0.txt" NOTICE_FILE = MT_DIR / "NOTICE" MODEL_ID = "Helsinki-NLP/opus-mt-ja-en" # HEAD of the model repo when resolved (2026-09-06). The model itself is the OPUS-MT release # tagged ``opus-2019-12-18`` (Tatoeba BLEU 41.7 / chrF 0.589 per the card). OPUS_MT_REVISION = "0770961a39ba6bd66305b149c3f4110bcafca2e6" OPUS_MT_TAG = "opus-2019-12-18" # Everything except the 303 MB TensorFlow checkpoint; the converter reads the PyTorch weights. SNAPSHOT_PATTERNS = [ "README.md", "config.json", "generation_config.json", "pytorch_model.bin", "source.spm", "target.spm", "tokenizer_config.json", "vocab.json", ] APACHE_URL = "https://www.apache.org/licenses/LICENSE-2.0.txt" APACHE_PHRASES = ("Apache License", "Version 2.0") QUANTIZATION = "int8" INTRA_THREADS = 2 # imitates the 2-vCPU Space container, as research § Q4 measured # The five files under OUT_DIR the README hashes and the tests guard. MODEL_FILES = ("model.bin", "shared_vocabulary.json", "config.json", "source.spm", "target.spm") SPM_FILES = ("source.spm", "target.spm") # Text files the converter writes with the PLATFORM newline (CRLF on Windows). They are # normalised to LF so the bytes hashed here are the bytes of every checkout: the repo # has core.autocrlf=input, so a CRLF working copy would hash differently from the index # and from the Linux Space (found when the first commit warned "CRLF will be replaced"). TEXT_FILES = ("shared_vocabulary.json", "config.json") # Research measured 77,339,435 B and ~0.8 MB; anything below these is a truncated write or an # LFS pointer (~130 B). MIN_MODEL_BYTES = 70_000_000 MIN_SPM_BYTES = 500_000 PROBE_TEXT = "駅はどこですか。" PROBE_KEYWORD = "station" # The plan's six contract sentences plus the known wobble; piece equality is asserted on all. PIECE_CHECK_TEXTS = ( "こんにちは。", PROBE_TEXT, "日本語を勉強しています。", "今日はいい天気ですから、公園を散歩してから、買い物に行きました。", "日本語を練習しましょう。", "昨日何をしましたか。", "はじめまして、よろしくお願いします。", ) VENV_RECIPE = """\ This script needs ctranslate2, sentencepiece, huggingface_hub, transformers AND torch. transformers/torch are deliberately not project dependencies - use a throwaway venv: uv venv .venv-mt --python 3.12 uv pip install --python .venv-mt ctranslate2==4.8.2 sentencepiece==0.2.2 \\ huggingface_hub==1.28.0 transformers==5.16.1 torch \\ --index-url https://download.pytorch.org/whl/cpu --extra-index-url https://pypi.org/simple .venv-mt/Scripts/python.exe scripts/convert_mt.py """ def sha256(data: bytes) -> str: return hashlib.sha256(data).hexdigest() def require_imports() -> dict[str, str]: """Import the conversion stack or exit 2 with the venv recipe. Returns the versions used.""" try: import ctranslate2 import huggingface_hub import sentencepiece import torch import transformers except ImportError as exc: print(f"missing dependency: {exc}\n\n{VENV_RECIPE}", file=sys.stderr) sys.exit(2) return { "ctranslate2": ctranslate2.__version__, "sentencepiece": sentencepiece.__version__, "transformers": transformers.__version__, "torch": torch.__version__, "huggingface_hub": huggingface_hub.__version__, } def download_snapshot() -> Path: from huggingface_hub import snapshot_download print(f"snapshot_download {MODEL_ID}@{OPUS_MT_REVISION[:12]} ...") started = time.perf_counter() path = Path( snapshot_download(MODEL_ID, revision=OPUS_MT_REVISION, allow_patterns=SNAPSHOT_PATTERNS) ) print(f" {path} ({time.perf_counter() - started:.1f} s)") for name in ("pytorch_model.bin", "config.json", "vocab.json", *SPM_FILES): if not (path / name).is_file(): sys.exit(f"snapshot is missing {name}; the pinned revision has changed shape") return path def convert(snapshot: Path) -> float: """Convert the snapshot into OUT_DIR. Returns the seconds taken.""" from ctranslate2.converters import TransformersConverter OUT_DIR.mkdir(parents=True, exist_ok=True) print(f"converting to CTranslate2 {QUANTIZATION} -> {OUT_DIR.relative_to(REPO_ROOT)} ...") started = time.perf_counter() TransformersConverter(str(snapshot)).convert( str(OUT_DIR), quantization=QUANTIZATION, force=True ) seconds = time.perf_counter() - started print(f" converted in {seconds:.1f} s") for name in SPM_FILES: shutil.copyfile(snapshot / name, OUT_DIR / name) print(f" copied {name}") for name in TEXT_FILES: path = OUT_DIR / name data = path.read_bytes() if b"\r\n" in data: path.write_bytes(data.replace(b"\r\n", b"\n")) print(f" normalised {name} to LF") return seconds def sentencepiece_pieces(text: str) -> list[str]: """The Space's tokenisation: plain sentencepiece on source.spm plus the end-of-sentence.""" import sentencepiece as spm processor = spm.SentencePieceProcessor(model_file=str(OUT_DIR / "source.spm")) return processor.encode(text, out_type=str) + [""] def read_back(snapshot: Path) -> tuple[str, float]: """Prove the converted model and the transformers-free tokenisation before recording anything. Returns the probe translation and its milliseconds. """ import ctranslate2 import sentencepiece as spm from transformers import MarianTokenizer reference = MarianTokenizer.from_pretrained(str(snapshot)) for text in PIECE_CHECK_TEXTS: ours = sentencepiece_pieces(text) theirs = reference.convert_ids_to_tokens(reference(text)["input_ids"]) if ours != theirs: sys.exit( f"sentencepiece pieces differ from MarianTokenizer for {text!r}:\n" f" sentencepiece: {ours}\n MarianTokenizer: {theirs}\n" "the Space cannot skip transformers; nothing recorded" ) print(f" sentencepiece == MarianTokenizer on {len(PIECE_CHECK_TEXTS)} sentences") translator = ctranslate2.Translator( str(OUT_DIR), device="cpu", compute_type=QUANTIZATION, inter_threads=1, intra_threads=INTRA_THREADS, ) target = spm.SentencePieceProcessor(model_file=str(OUT_DIR / "target.spm")) started = time.perf_counter() hypothesis = translator.translate_batch( [sentencepiece_pieces(PROBE_TEXT)], beam_size=4, max_decoding_length=128 )[0].hypotheses[0] ms = (time.perf_counter() - started) * 1000.0 translation = target.decode([piece for piece in hypothesis if piece != ""]).strip() print(f" {PROBE_TEXT} -> {translation!r} ({ms:.0f} ms)") if PROBE_KEYWORD not in translation.lower(): sys.exit(f"read-back translation {translation!r} lacks {PROBE_KEYWORD!r}; nothing recorded") return translation, ms def ensure_license() -> None: if LICENSE_FILE.is_file(): text = LICENSE_FILE.read_text(encoding="utf-8") else: print(f"fetching {APACHE_URL} ...") with urllib.request.urlopen(APACHE_URL, timeout=60) as response: # noqa: S310 - https text = response.read().decode("utf-8") LICENSE_FILE.write_text(text, encoding="utf-8", newline="\n") for phrase in APACHE_PHRASES: if phrase not in text: sys.exit(f"{LICENSE_FILE.name} does not contain {phrase!r}") def write_notice() -> None: lines = [ f"{MODEL_ID} (OPUS-MT {OPUS_MT_TAG}, Language Technology Research Group at the " f"University of Helsinki) - Apache License 2.0 - https://huggingface.co/{MODEL_ID}", f"Converted to CTranslate2 {QUANTIZATION} by scripts/convert_mt.py for " f"japanese-learning-avatar from Hub revision {OPUS_MT_REVISION}.", "source.spm and target.spm are the SentencePiece tokenizer models from the same " "repository, copied unmodified.", ] NOTICE_FILE.write_text("\n".join(lines) + "\n", encoding="utf-8", newline="\n") def hash_rows() -> list[dict[str, object]]: rows = [] for name in MODEL_FILES: data = (OUT_DIR / name).read_bytes() rows.append({"name": name, "bytes": len(data), "sha256": sha256(data)}) return rows def check_sizes(rows: list[dict[str, object]]) -> None: sizes = {row["name"]: row["bytes"] for row in rows} if sizes["model.bin"] < MIN_MODEL_BYTES: size = sizes["model.bin"] sys.exit(f"model.bin is {size:,} B (< {MIN_MODEL_BYTES:,}); conversion failed") for name in SPM_FILES: if sizes[name] < MIN_SPM_BYTES: sys.exit(f"{name} is {sizes[name]:,} B (< {MIN_SPM_BYTES:,}); copy failed") def readme_hashes(text: str) -> set[str]: import re return set(re.findall(r"^\| `[^`]+` \| [\d,]+ \| `([0-9a-f]{64})` \|", text, re.M)) def write_readme( rows: list[dict[str, object]], versions: dict[str, str], convert_seconds: float, translation: str, translate_ms: float, ) -> None: date = datetime.now(UTC).strftime("%Y-%m-%d") source = { "model.bin": f"CTranslate2 {QUANTIZATION} conversion of `pytorch_model.bin`", "shared_vocabulary.json": "CTranslate2 conversion of `vocab.json`", "config.json": "written by the CTranslate2 converter", "source.spm": "copied unmodified from the model repo", "target.spm": "copied unmodified from the model repo", } lines = [ "# data/mt - Japanese to English translation model (CTranslate2 int8)", "", f"**Resolved:** {date} ", f"**Model:** `{MODEL_ID}` - Hub revision `{OPUS_MT_REVISION}` " f"(OPUS-MT release `{OPUS_MT_TAG}`) ", f"**Converted with:** ctranslate2 {versions['ctranslate2']}, " f"transformers {versions['transformers']}, torch {versions['torch']}, " f"sentencepiece {versions['sentencepiece']}, huggingface_hub {versions['huggingface_hub']} " f'(`quantization="{QUANTIZATION}"`, {convert_seconds:.0f} s) ', "**Built by:** `scripts/convert_mt.py` (see its docstring for the throwaway venv)", "", "Decision D-15: English on demand comes from a small translation model on the Space's", "CPU, zero GPU quota. 02-RESEARCH.md § Q4 picked OPUS-MT ja-en (Apache-2.0) and measured", "CTranslate2 int8 at 16-129 ms per sentence and +103 MB at two threads, against 131-738 ms", "and +374 MB for transformers + torch - so the model is converted here, once, and the", "Space imports only `ctranslate2` and `sentencepiece`. The conversion script proves that", "plain sentencepiece on `source.spm` (+ ``) yields exactly the pieces", "`transformers.MarianTokenizer` yields, which is what makes transformers unnecessary.", "", "## Files", "", f"All under `{OUT_DIR.relative_to(REPO_ROOT).as_posix()}/`. `model.bin`, `source.spm` and", "`target.spm` are Git LFS objects (`*.bin`, `*.spm` in .gitattributes); the two JSON files", "are plain text.", "", "| File | Bytes | SHA256 | Source |", "|---|---|---|---|", ] for row in rows: lines.append( f"| `{row['name']}` | {row['bytes']:,} | `{row['sha256']}` | {source[row['name']]} |" ) lines += [ "", "## Read-back on the run that produced these bytes", "", f"`{PROBE_TEXT}` -> `{translation}` in {translate_ms:.0f} ms " f'(`ctranslate2.Translator(device="cpu", compute_type="{QUANTIZATION}", ' f"inter_threads=1, intra_threads={INTRA_THREADS})`, beam 4, first call after load).", f"sentencepiece pieces == MarianTokenizer pieces on {len(PIECE_CHECK_TEXTS)} sentences.", "Known wobble (research, not fixed): `はじめまして、よろしくお願いします。` ->", '"Nice to meet you. Nice to meet you."', "", "## Licence", "", f"`{MODEL_ID}` - Apache-2.0 (Language Technology Research Group at the", f"University of Helsinki) - converted to CTranslate2 {QUANTIZATION} by", "`scripts/convert_mt.py`; the Apache-2.0 text is beside the model as", "`LICENSE-apache-2.0.txt` and the `NOTICE` names the source model and the conversion.", "Runtime libraries: CTranslate2 (MIT), SentencePiece (Apache-2.0). On-page credit:", '"Translation: OPUS-MT (Helsinki-NLP, Apache-2.0)".', "", "## Alternative not taken", "", "Research § Open Questions 2 offered publishing the conversion as an owner model repo", "(`WolfDavid/opus-mt-ja-en-ct2-int8`) and `huggingface_hub.snapshot_download`-ing it in", "`warm_translator`. That keeps ~80 MB out of this repo but re-downloads ~80 MB on every", "container start (Space disk is ephemeral) and puts a download in front of the first EN", "reveal. LFS in-repo is one pinned artefact that arrives with the clone; the Hub-repo", "variant stays the fallback if repo size ever becomes a concern.", "", "## Regenerating", "", "```", "uv venv .venv-mt --python 3.12", "uv pip install --python .venv-mt ctranslate2==4.8.2 sentencepiece==0.2.2 \\", " huggingface_hub==1.28.0 transformers==5.16.1 torch \\", " --index-url https://download.pytorch.org/whl/cpu --extra-index-url https://pypi.org/simple", ".venv-mt/Scripts/python.exe scripts/convert_mt.py", "```", "", "The script refuses to record anything unless the piece-equality and `station` read-back", "checks pass, and leaves this file untouched when every hash above still matches.", "", ] README.write_text("\n".join(lines), encoding="utf-8", newline="\n") def main() -> int: versions = require_imports() print("versions:", ", ".join(f"{k} {v}" for k, v in versions.items())) snapshot = download_snapshot() convert_seconds = convert(snapshot) rows = hash_rows() check_sizes(rows) print("read-back ...") translation, translate_ms = read_back(snapshot) ensure_license() write_notice() previous = README.read_text(encoding="utf-8") if README.exists() else "" if readme_hashes(previous) == {row["sha256"] for row in rows}: print(f"{README.relative_to(REPO_ROOT)} already records these hashes; left untouched") else: write_readme(rows, versions, convert_seconds, translation, translate_ms) print(f"wrote {README.relative_to(REPO_ROOT)}") for row in rows: print(f" {row['name']:<24} {row['bytes']:>12,} {row['sha256']}") return 0 if __name__ == "__main__": sys.exit(main())