#!/usr/bin/env bash set -euo pipefail # Reconstructed from recorded working settings; requires the rented CUDA GPU. # Supply a local pinned snapshot path, or download the recorded revision first. : "${TEACHER_MODEL_PATH:?Set TEACHER_MODEL_PATH to the downloaded Qwen snapshot}" exec vllm serve "$TEACHER_MODEL_PATH" \ --served-model-name Qwen/Qwen3.8-27B-FP8 \ --host 127.0.0.1 --port 18000 --tensor-parallel-size 1 \ --language-model-only --max-model-len 8192 --max-num-seqs 32 \ --gpu-memory-utilization 0.85 --reasoning-parser qwen3 \ --compilation-config '{"cudagraph_capture_sizes":[1,2,4,8,16,32]}' \ --speculative-config '{"method":"mtp","num_speculative_tokens":3}'