File size: 2,899 Bytes
4294de3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
#!/bin/bash
# Nex-N2.5-mini phase 1 (CPU): verified download -> BF16 GGUF + vision projector -> the three standard 4-bit tiers
# (King: STRIX_LEAN + COHERENT + FAST; no Q8/Q6) with head protection, read back by exact tensor name.
# Runs in the capped scope `nex-conv` (Agnes memory sizing waits for nex-* scopes). Waits until the Agnes BF16
# re-grade (60 GiB on the GPU) is finished so the two large memory users never overlap.
set -uo pipefail
W=/mnt/models/nex-n2.5-mini; T=/opt/llama-rocm/rocmfpx-724; B=$T/build-hipvk/bin; N=Nex-N2.5-mini
A=/mnt/models/agnes-3.0-flash
export LD_LIBRARY_PATH=$B:/opt/rocm-7.2.4/lib TMPDIR=/mnt/models/.tmp PYTHONUNBUFFERED=1
cd $W; mkdir -p gguf out logs
log(){ echo "[$(date -u +%FT%TZ)] $*"; }
until [ -f logs/DOWNLOAD_RC ]; do sleep 20; done
if [ "$(cat logs/DOWNLOAD_RC)" != 0 ] || [ "$(cat logs/DOWNLOAD_VERIFY_RC 2>/dev/null)" != 0 ]; then
  log "download or verify failed -> stop"; log "NEX_PHASE1_FAILED"; exit 1
fi
log "download verified; waiting for the Agnes BF16 re-grade to leave the GPU"
until grep -q "R3 grade" $A/logs/regrade.log 2>/dev/null; do sleep 20; done

log "C1 convert BF16 (the checkpoint has no mtp.* tensors, so no MTP block is emitted)"
python3 $T/convert_hf_to_gguf.py hf --outtype bf16 --model-name "$N" --outfile gguf/$N-BF16.gguf > logs/C1_convert.log 2>&1
rc=$?; log "C1 exit=$rc"; [ $rc -eq 0 ] || { tail -30 logs/C1_convert.log; log "NEX_PHASE1_FAILED"; exit 2; }
log "C2 convert vision projector"
python3 $T/convert_hf_to_gguf.py hf --outtype bf16 --mmproj --model-name "$N" --outfile out/mmproj-$N-BF16.gguf > logs/C2_mmproj.log 2>&1
rc=$?; log "C2 exit=$rc"; [ $rc -eq 0 ] || { tail -30 logs/C2_mmproj.log; log "NEX_PHASE1_FAILED"; exit 3; }
python3 readback.py - - gguf/$N-BF16.gguf | tee logs/C_readback.log
python3 readback.py - - out/mmproj-$N-BF16.gguf | tee -a logs/C_readback.log

BF=gguf/$N-BF16.gguf; Q=$B/llama-quantize
log "Q1 standard tiers"
$Q --output-tensor-type q6_K                             $BF out/$N-Q4_0_ROCMFP4_STRIX_LEAN.gguf Q4_0_ROCMFP4_STRIX_LEAN 16 > logs/Q1_q106.log 2>&1; log "  q106 exit=$?"
$Q --output-tensor-type q6_K --token-embedding-type q6_K $BF out/$N-Q4_0_ROCMFP4_COHERENT.gguf   Q4_0_ROCMFP4_COHERENT   16 > logs/Q1_q102.log 2>&1; log "  q102 exit=$?"
$Q --output-tensor-type q6_K                             $BF out/$N-Q4_0_ROCMFP4_FAST.gguf       Q4_0_ROCMFP4_FAST       16 > logs/Q1_q103.log 2>&1; log "  q103 exit=$?"
python3 readback.py Q6_K Q5_K out/$N-Q4_0_ROCMFP4_STRIX_LEAN.gguf | tee logs/Q_readback.log
python3 readback.py Q6_K Q6_K out/$N-Q4_0_ROCMFP4_COHERENT.gguf   | tee -a logs/Q_readback.log
python3 readback.py Q6_K -    out/$N-Q4_0_ROCMFP4_FAST.gguf       | tee -a logs/Q_readback.log
for l in Q1_q106 Q1_q102 Q1_q103; do printf "%-8s " $l; grep -oE "quant size\s*=\s*[0-9.]+ MiB \([0-9.]+ BPW\)" logs/$l.log; done | tee logs/Q_sizes.log
log "NEX_PHASE1_DONE"