"""Authenticated Hub preparation and publication (CPU-only Modal functions).""" import os from .config import MODEL_ID, MODEL_REVISION def hub_api(): from huggingface_hub import HfApi names = ( "HF_TOKEN", "HUGGING_FACE_HUB_TOKEN", "HUGGINGFACE_TOKEN", "HUGGINGFACEHUB_API_TOKEN", ) token = next((os.environ[n] for n in names if os.environ.get(n)), None) if not token: raise RuntimeError("huggingface secret must contain HF_TOKEN or a supported HF token key") return HfApi(token=token) def file_digest(path): import hashlib digest = hashlib.sha256() with open(path, "rb") as handle: for chunk in iter(lambda: handle.read(8 * 1024 * 1024), b""): digest.update(chunk) return digest.hexdigest() def assemble_release(source, snapshot, destination, repo_id): """Combine allowlisted artifacts with untouched upstream files and an expanded card.""" import json import shutil from pathlib import Path import yaml source, snapshot, destination = map(Path, (source, snapshot, destination)) manifest = json.loads((source / "ARTIFACTS.json").read_text()) if destination.exists(): raise ValueError("release destination must be new") destination.mkdir(parents=True) for relative, expected in manifest["files"].items(): name = Path(relative) if name.is_absolute() or ".." in name.parts or any(p.startswith(".") for p in name.parts): raise ValueError("unsafe artifact path") path = source / name if path.is_symlink() or not path.is_file() or file_digest(path) != expected: raise ValueError(f"artifact checksum mismatch: {relative}") target = destination / name target.parent.mkdir(parents=True, exist_ok=True) shutil.copyfile(path, target) shutil.copyfile(source / "ARTIFACTS.json", destination / "ARTIFACTS.json") base_files = [] for path in sorted(snapshot.iterdir()): if path.name not in {"LICENSE", "README.md"} and path.suffix not in { ".json", ".safetensors", ".jinja", }: continue target_name = "UPSTREAM_README.md" if path.name == "README.md" else path.name shutil.copyfile(path, destination / target_name) base_files.append( { "upstream_path": path.name, "release_path": target_name, "sha256": file_digest(path), "bytes": path.stat().st_size, } ) original = (snapshot / "README.md").read_text() if not original.startswith("---\n"): raise ValueError("upstream model card must have YAML front matter") _, header, body = original.split("---", 2) metadata = yaml.safe_load(header) metadata["base_model"] = MODEL_ID metadata["model_name"] = "LFM2.5-2.6B-RLCD" metadata["tags"] = list( dict.fromkeys( metadata.get("tags", []) + [ "parallel-constrained-decoding", "structured-generation", "classification", "inference-only", "modal", ] ) ) card = ( "---\n" + yaml.safe_dump(metadata, sort_keys=False, allow_unicode=True) + "---\n\n" + (source / "README-PCD.md").read_text().replace("monotykamary/LFM2.5-2.6B-RLCD", repo_id) + "\n\n## Original LiquidAI model documentation\n\n" "> The following upstream text is preserved verbatim. Its benchmarks describe the original model, " "not our PCD engine. Our measurements and limitations are documented above.\n\n" + body.lstrip("\n") ) (destination / "README.md").write_text(card) (destination / "BASE_MODEL_MANIFEST.json").write_text( json.dumps( { "model_id": MODEL_ID, "revision": MODEL_REVISION, "unchanged_weights": True, "files": base_files, }, indent=2, ) + "\n" ) index = json.loads((destination / "model.safetensors.index.json").read_text()) shards = set(index["weight_map"].values()) if not shards or any(not (destination / name).is_file() for name in shards): raise ValueError("incomplete sharded model") return { str(path.relative_to(destination)): file_digest(path) for path in destination.rglob("*") if path.is_file() } def publish(source="/release-src", public=False): import json import tempfile import time from pathlib import Path from huggingface_hub import HfApi, hf_hub_download, snapshot_download api = hub_api() owner = api.whoami()["name"] repo_id = f"{owner}/LFM2.5-2.6B-RLCD" if api.repo_exists(repo_id): raise RuntimeError( f"{repo_id} already exists; refusing to overwrite or blindly retry a publication" ) snapshot = snapshot_download( MODEL_ID, revision=MODEL_REVISION, local_files_only=True, allow_patterns=["*.json", "*.safetensors", "*.jinja", "README.md", "LICENSE"], ) with tempfile.TemporaryDirectory(prefix="pcd-release-") as work: destination = Path(work) / "bundle" hashes = assemble_release(source, snapshot, destination, repo_id) api.create_repo(repo_id, repo_type="model", private=True, exist_ok=False) commit = api.upload_folder( repo_id=repo_id, folder_path=str(destination), commit_message="feat: add verified inference-only parallel constrained decoding", ) tag = "pcd-preview-" + time.strftime("%Y%m%d-%H%M%S", time.gmtime()) api.create_tag(repo_id, tag=tag, revision=commit.oid) info = api.model_info(repo_id, revision=tag, files_metadata=True) remote = {entry.rfilename: entry for entry in info.siblings} for name, expected in hashes.items(): if name not in remote: raise RuntimeError(f"uploaded file missing: {name}") entry = remote[name] if entry.lfs is not None: observed = entry.lfs.sha256 else: downloaded = hf_hub_download(repo_id, name, revision=tag, token=api.token) observed = file_digest(downloaded) if observed != expected: raise RuntimeError(f"uploaded checksum mismatch: {name}") if public: api.update_repo_settings(repo_id, private=False) info = HfApi(token=False).model_info(repo_id, revision=tag) if info.private: raise RuntimeError("public visibility verification failed") result = { "repo_id": repo_id, "url": f"https://huggingface.co/{repo_id}", "commit": commit.oid, "verified_preview_tag": tag, "public": public, "verified_files": len(hashes), "unchanged_weights": True, } Path("/results").mkdir(exist_ok=True) Path("/results/publication.json").write_text(json.dumps(result, indent=2) + "\n") return result def prepare(): from huggingface_hub import snapshot_download api = hub_api() name = api.whoami()["name"] snapshot = snapshot_download( MODEL_ID, revision=MODEL_REVISION, allow_patterns=["*.json", "*.safetensors", "*.jinja", "README.md", "LICENSE"], ) return { "owner": name, "proposed_repo": f"{name}/LFM2.5-2.6B-RLCD", "model_id": MODEL_ID, "revision": MODEL_REVISION, "snapshot": snapshot, }