"""Fetch every file the Space serves from, on CPU, before Gradio starts. Idempotent by design: each step checks for what it would produce and returns immediately when it is already there, so a warm Space restarts in seconds. Nothing here touches a GPU, so none of it consumes a ZeroGPU allocation. The UMT5-XXL text encoder (11 GB) and its tokenizer are deliberately absent: the exported prompt embedding replaces both, and `sync_vendor.sh` removes them from the vendored auto-download set so inference cannot pull them in later. """ from __future__ import annotations import os import subprocess from pathlib import Path import tyro from fdanyone.assets import ( CHECKPOINT, HF_REPO_ID, HF_REVISION, MHR70_REGRESSOR, WAN_VAE, ) from fdanyone.download import ensure_foreground_model, ensure_models, ensure_turbo_assets SPACE_ASSETS_REPO_ID: str = "pablovela5620/4danyone-space-assets" """Private mirror of SMPL-X, the GVHMR checkpoints, and the prompt embedding.""" PROMPT_EMBEDDING: str = "4danyone/prompt_embedding.safetensors" """Where the exported prompt context lives under the model root.""" GVHMR_REPO_URL: str = "https://github.com/zju3dv/GVHMR.git" """Upstream GVHMR, cloned at boot instead of vendored into the Space repo.""" GVHMR_REVISION: str = "6ec3ca39336c50492c0fae65fba2fb831fc7d866" """The revision pinned as a submodule by the source repository.""" EXAMPLES_REVISION: str = "442816913e7cc75be2ede1a5c93a86d936d032f1" """Revision carrying all twenty Pexels clips; the model pin predates twelve.""" def _snapshot(repo_id: str, revision: str, patterns: list[str], destination: Path) -> None: """Download one immutable revision's matching files straight into place.""" from huggingface_hub import snapshot_download destination.mkdir(parents=True, exist_ok=True) snapshot_download( repo_id=repo_id, revision=revision, allow_patterns=patterns, local_dir=destination, ) def ensure_published_models(model_dir: Path) -> None: """Fetch the DiT checkpoint, the Wan VAE, and the MHR70 regressor.""" wanted: list[str] = [CHECKPOINT, WAN_VAE, MHR70_REGRESSOR] missing: list[str] = [name for name in wanted if not (model_dir / name).is_file()] if missing: _snapshot(HF_REPO_ID, HF_REVISION, missing, model_dir) def ensure_space_assets(model_dir: Path) -> None: """Fetch SMPL-X, the GVHMR checkpoints, and the exported prompt embedding.""" from huggingface_hub import hf_hub_download if not (model_dir / "body_models/smplx/SMPLX_NEUTRAL.npz").is_file(): _snapshot(SPACE_ASSETS_REPO_ID, "main", ["body_models/**"], model_dir) if not (model_dir / "gvhmr/gvhmr_siga24_release.ckpt").is_file(): _snapshot(SPACE_ASSETS_REPO_ID, "main", ["gvhmr/**"], model_dir) embedding: Path = model_dir / PROMPT_EMBEDDING if not embedding.is_file(): # The file sits at the repository root but must land beside the model it # conditions, so download it and move it into the 4danyone directory. staged: Path = Path( hf_hub_download( repo_id=SPACE_ASSETS_REPO_ID, filename="prompt_embedding.safetensors", local_dir=model_dir / ".prompt-embedding", ) ) embedding.parent.mkdir(parents=True, exist_ok=True) staged.replace(embedding) def ensure_example_clips(data_dir: Path) -> None: """Mirror the Pexels example clips (20 files, 209 MB) into the data root. They come from the model repository rather than this repo, under the Pexels license its LICENSE file maps onto ``data/source/pexels``. """ import shutil clips: Path = data_dir / "source" / "pexels" if any(clips.glob("*.mp4")): return # Repository paths carry a leading ``data/`` that the local root already # is, so stage the snapshot and lift the folder into place. staging: Path = data_dir / ".pexels" _snapshot(HF_REPO_ID, EXAMPLES_REVISION, ["data/source/pexels/*.mp4"], staging) clips.parent.mkdir(parents=True, exist_ok=True) (staging / "data" / "source" / "pexels").replace(clips) shutil.rmtree(staging) def ensure_gvhmr_checkout(gvhmr_root: Path) -> None: """Clone GVHMR at its pinned revision into the ephemeral disk.""" if gvhmr_root.is_dir() and (gvhmr_root / ".git").exists(): head: str = subprocess.check_output( ["git", "-C", str(gvhmr_root), "rev-parse", "HEAD"], text=True ).strip() if head == GVHMR_REVISION: return else: gvhmr_root.parent.mkdir(parents=True, exist_ok=True) subprocess.run( ["git", "clone", "--filter=blob:none", GVHMR_REPO_URL, str(gvhmr_root)], check=True, ) subprocess.run(["git", "-C", str(gvhmr_root), "fetch", "origin", GVHMR_REVISION], check=True) subprocess.run(["git", "-C", str(gvhmr_root), "checkout", "--force", GVHMR_REVISION], check=True) def download( model_dir: Path = Path(os.environ.get("FDANYONE_MODEL_DIR", "models")), data_dir: Path = Path(os.environ.get("FDANYONE_DATA_DIR", "data")), ) -> None: """Install every model file and the pinned GVHMR checkout. Args: model_dir: Root holding the 4DAnyone, GVHMR, BiRefNet, and turbo assets. data_dir: Root holding the GVHMR source checkout and per-run scratch. """ models: Path = model_dir.expanduser().resolve() gvhmr_root: Path = (data_dir.expanduser().resolve() / "GVHMR") ensure_gvhmr_checkout(gvhmr_root) ensure_example_clips(data_dir.expanduser().resolve()) ensure_space_assets(models) ensure_published_models(models) ensure_foreground_model(models) ensure_turbo_assets(models) # Verifies every remaining file the inference path resolves and creates the # compatibility links GVHMR reads from its own checkout. Any gap left by the # steps above fails here, at boot, instead of inside a GPU allocation. ensure_models(models, gvhmr_root) print(f"models: {models}\ngvhmr: {gvhmr_root}@{GVHMR_REVISION}") if __name__ == "__main__": tyro.cli(download)