{ "verified_date": "2026-09-10", "system": "Berzelius Hopper", "user_login_endpoint": "berzelius-hopper.nsc.liu.se", "submission_location": "Submit the prepared sbatch jobs from the Hopper login system; Ampere login nodes submit to A100 hardware.", "gpu_nodes": "node101 through node116", "gpu_layout": "8 NVIDIA H200 GPUs, 141 GB per GPU", "operating_system": "Red Hat Enterprise Linux 8.8 according to NSC", "scheduler_constraint": "No -C constraint on Hopper; NSC documents identical Hopper GPU nodes.", "account": "berzelius-2026-196", "account_source": "Existing project eight-GPU evaluation launcher", "build_module": "buildenv-gcccuda/12.9.1-gcc11", "build_module_source": "Existing project Berzelius launcher", "result_storage": "/proj/cvl/users/x_fahkh2/worldmem-baseline-evals", "runtime_verification": "Each full evaluation checks for exactly eight H200 GPUs before generation.", "matrix_flash_attention_portability": "The official FlashAttention 2.8.3 wheel requires GLIBC_2.32. Setup builds the same version from source for SM90 on the target OS, including on setup reruns.", "flash_attention_build_reference": "https://github.com/Dao-AILab/flash-attention/blob/v2.8.3/setup.py", "remote_access_by_agent": "none", "sources": [ "https://www.nsc.liu.se/support/systems/berzelius-getting-started/", "https://www.nsc.liu.se/support/systems/berzelius-gpu/", "https://www.nsc.liu.se/support/batch-jobs/berzelius/", "https://www.nsc.liu.se/support/systems/berzelius-getting-started/berzelius-faq/", "https://www.nsc.liu.se/support/systems/berzelius-software/berzelius-conda-mamba/", "https://www.nsc.liu.se/software/installation-policy/" ], "setup_entrypoint": "deployment/setup_environment_berzelius.sh", "source_reference": "The repository contains the frozen WorldMem metric/VAE source under references/worldmem; no separate DeMemWM checkout is required.", "environment_root": "/proj/cvl/users/x_fahkh2/envs", "environments": [ "decmem-eval", "matrix-game2-eval", "live-eval", "geometry-forcing-eval" ], "environment_setup": "Dedicated CPU sbatch job on berzelius-hopper-cpu: 14 cores, eight hours, no GPUs. Creates all four environment prefixes first, then installs dependencies, builds CUDA extensions, downloads checkpoints, caches metrics and validates existing datasets. No model inference.", "smoke_evaluation": "Four separate per-method sbatch jobs: each uses one H200, 14 CPU cores and a four-hour limit. Each runs generation, scoring and aggregation for one case.", "full_evaluation": "Four separate per-method sbatch submissions: each uses eight H200 GPUs, eight tasks with 14 CPU cores each and a 48-hour limit. Minecraft seeds 42 and 43 are sequential array elements. RE10K uses final100, seed 0. No automatic submission.", "quota_handling": "The user checks current project allocation and storage quota before submission, submits one job at a time and waits for successful completion before dependent work. Remaining quota and remote runtime have not been queried by the agent.", "previous_launcher_comparison": { "main_worldmem_revision": "8f62834150673bff7debe5f41a08a45c7d955226", "minecraft_training_source": "scripts/train_dememwm_sourcefree_ref_8h200_berzelius_200k.sh", "minecraft_verified": "Account at line 9, build module at line 15, dataset and fixed VAE cache at lines 48-49 match the prepared deployment. Setup uses its pinned project-local Miniforge manager; GPU evaluation uses explicit method Python paths without global conda initialization.", "re10k_training_source": "Existing WorldMem-vmem-fair-eval-0b88eda checkout: scripts/train_re10k_faithful_f200b200_dememwm_clipped_8h200_berzelius.sh", "re10k_verified": "Environment parent /proj/cvl/users/x_fahkh2/envs at line 22 and prepared RE10K dataset root at line 25 match.", "intentional_differences": "New method-specific interpreter prefixes; caches inside the new checkout; project-local temporary and Triton cache storage; 14 CPU cores per H200; CPU-only installation; per-GPU inference sharding rather than distributed training. Setup installs an explicit project-local Miniforge manager and restricts conda creation to conda-forge; the documented Miniforge module was unavailable in user job 84924.", "remote_validation": "No cluster connection or runtime inspection by the agent. Setup preflights the actual cluster dataset files." }, "conda_environment_channels": "Environment creation uses --override-channels --channel conda-forge; inherited channels are not used.", "conda_distribution": "Official Miniforge3 24.7.1-2 Linux x86_64 installer", "conda_prefix": "/proj/cvl/users/x_fahkh2/envs/baseline-miniforge3", "conda_installer": "https://github.com/conda-forge/miniforge/releases/download/24.7.1-2/Miniforge3-24.7.1-2-Linux-x86_64.sh", "conda_installation": "CPU batch job downloads the installer under the project cache and invokes bash -b -p at the explicit prefix; no conda init or shell-profile edits. Setup calls the manager by its absolute path.", "temporary_directory": "/proj/cvl/users/x_fahkh2/worldmem-baseline-evals/cache/", "triton_cache_directory": "/proj/cvl/users/x_fahkh2/worldmem-baseline-evals/cache/", "temporary_storage_rule": "Explicit user rule on 2026-09-10: TMPDIR and TRITON_CACHE_DIR must use /proj/cvl/users/x_fahkh2/caches or /proj/cvl/users/x_fahkh2/worldmem-baseline-evals/cache/. All nine jobs select the latter.", "evaluation_interpreters": "All eight GPU scripts call the model and metric Python interpreters by absolute environment paths. They do not run conda info, source a conda shell initializer, or activate a global environment." }