Add production multi-GPU training scripts (B=2 + compile + shards)
Browse files- scripts/launch_4gpu.sh +12 -0
scripts/launch_4gpu.sh
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# Launch Fractus 1B training on 4 GPUs
|
| 3 |
+
set -e
|
| 4 |
+
cd "$(dirname "$0")/.."
|
| 5 |
+
mkdir -p logs checkpoints
|
| 6 |
+
|
| 7 |
+
for i in 0 1 2 3; do
|
| 8 |
+
echo "Starting GPU $i..."
|
| 9 |
+
GPU_ID=$i CUDA_VISIBLE_DEVICES=$i setsid python -u scripts/train_1b_multi_gpu.py > logs/gpu$i.log 2>&1 < /dev/null &
|
| 10 |
+
done
|
| 11 |
+
|
| 12 |
+
echo "All 4 GPUs launched. Monitor with: tail -f logs/gpu*.log"
|