GLM-5.2-MXFP8-NVFP4-NF3-Hybrid / apply-v20-capacity-fixes.sh
madeby561's picture
Publish validated v20 profiles, calibrated KV scales, and release card
2eb778b verified
Raw
History Blame Contribute Delete
979 Bytes
#!/usr/bin/env bash
set -euo pipefail
apply_or_verify() {
local patch_file="$1"
local target_file="$2"
local marker="$3"
if patch --batch --forward --dry-run -p0 -d / -i "${patch_file}" \
>/dev/null 2>&1; then
patch --batch --forward -p0 -d / -i "${patch_file}"
elif ! grep -qF "${marker}" "${target_file}"; then
echo "Patch does not apply and its marker is absent: ${patch_file}" >&2
exit 1
fi
}
apply_or_verify \
/release/patches/v20-mtp-online-quant-inheritance.patch \
/opt/venv/lib/python3.12/site-packages/vllm/config/speculative.py \
"Same-checkpoint MTP/DSpark drafts inherit the target's"
apply_or_verify \
/release/patches/v20-sparkinfer-carry-fold.patch \
/opt/venv/lib/python3.12/site-packages/sparkinfer/attention/nsa_indexer/paged.py \
"SPARKINFER_PAGED_INDEX_TWO_LEVEL_FOLD"
install -m 0755 \
/release/serve-glm52-v16.context-cap.sh \
/usr/local/bin/serve-glm52-v16.sh
exec /usr/local/bin/serve-gilded-gnosis.sh