Download run.sh from episod/hemmingway-1-p300: direct link, hf CLI and curl.
- Browser
- Download file 5.21 kB
-
https://huggingface.co/episod/hemmingway-1-p300/resolve/main/run.sh
- Command line
-
hf download hf://episod/hemmingway-1-p300/run.sh
-
curl -L -o run.sh https://huggingface.co/episod/hemmingway-1-p300/resolve/main/run.sh
5.21 kB
| # Serve this model on TT hardware. Assumes ./install.sh has been run. | |
| set -euo pipefail | |
| HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" | |
| VENV="${VENV:-$HERE/venv}" | |
| PYBIN="$VENV/bin/python" | |
| # Locate ttnn WITHOUT importing it β importing loads _ttnn.so, which is exactly what needs the | |
| # LD_PRELOAD below (chicken-and-egg). find_spec resolves the path without executing the module. | |
| TTNN_DIR="$("$PYBIN" -c 'import importlib.util,os;print(os.path.dirname(importlib.util.find_spec("ttnn").origin))')" | |
| # _ttnncpp.so lives in ttnn.libs/ for an auditwheel-repaired (portable) wheel, or build/lib/ for a | |
| # raw one; preload it to avoid the glibc "static TLS block" error on late dlopen. | |
| # Prefer the auditwheel-vendored copy in *.libs/ (that's the one _ttnn.so actually loads via | |
| # RPATH); fall back to build/lib for a raw (unrepaired) wheel β e.g. the plain `ttnn` PyPI wheel | |
| # a v6 thin bundle installs, which isn't auditwheel-repaired at all. | |
| # `|| true` on each probe: under `set -e -o pipefail`, a `ls <no-match> | head -1` pipe fails | |
| # (pipefail surfaces ls's nonzero exit) and set -e would kill the script on THIS line, before the | |
| # fallback below β or the deliberate `:?` error a few lines down β ever runs. | |
| LD_PRELOAD="$(ls "$TTNN_DIR"/../*.libs/_ttnncpp*.so 2>/dev/null | head -1)" || true | |
| [ -n "$LD_PRELOAD" ] || LD_PRELOAD="$(ls "$TTNN_DIR"/build/lib/_ttnncpp*.so 2>/dev/null | head -1)" || true | |
| export LD_PRELOAD="${LD_PRELOAD:?could not locate _ttnncpp.so in the ttnn install}" | |
| export TT_METAL_HOME="$TTNN_DIR" | |
| # EXTRA_MODELS_DIR is a PARENT of per-model bundle folders; the plugin scans its children for | |
| # each vllm_metadata.json (so the metadata lives in vllm_models/<model>/, not the bundle root). | |
| export EXTRA_MODELS_DIR="$HERE/vllm_models" | |
| export TT_VLLM_BUILTIN_MODELS=0 | |
| # Do NOT set VLLM_PLUGINS: it is an ALLOW-LIST β setting it suppresses the vllm.general_plugins | |
| # group, so the model's tool/reasoning-parser overrides would silently not load. The TT platform | |
| # + model registry load via entry points without it. | |
| export PYTHONPATH="$HERE:${PYTHONPATH:-}" # resolves the adapter/model imports | |
| export MESH_DEVICE="${MESH_DEVICE:-P150x2}" | |
| NCHIPS=2 | |
| if [ -z "${TT_METAL_VISIBLE_DEVICES:-}" ]; then | |
| if [ -n "${TT_VISIBLE_DEVICES:-}" ]; then | |
| IFS=, read -ra _GRANT <<< "$TT_VISIBLE_DEVICES" | |
| if [ "${#_GRANT[@]}" -lt "$NCHIPS" ]; then | |
| echo "run.sh: this model needs $NCHIPS chip(s) but TT_VISIBLE_DEVICES grants ${#_GRANT[@]} ($TT_VISIBLE_DEVICES)" >&2 | |
| exit 1 | |
| fi | |
| TT_VISIBLE_DEVICES="$(IFS=,; echo "${_GRANT[*]:0:$NCHIPS}")" | |
| export TT_VISIBLE_DEVICES | |
| TT_METAL_VISIBLE_DEVICES="0,1" | |
| else | |
| TT_METAL_VISIBLE_DEVICES="0,1" | |
| fi | |
| fi | |
| export TT_METAL_VISIBLE_DEVICES | |
| # HERMETIC RUNTIME: keep every cache/home INSIDE the folder wall, so serving writes and reads | |
| # nothing outside it (the ttnn tensor cache even DEFAULTS to a hard-coded /mnt/... path upstream β | |
| # a classic other-machine leak we must override). Each is overridable if the operator sets it. | |
| export HF_HOME="${HF_HOME:-$HERE/.hf}" # HF weights + hub cache | |
| export TT_CACHE_PATH="${TT_CACHE_PATH:-$HERE/.tt_cache}" # ttnn weight/tensor cache | |
| export TT_CACHE_HOME="${TT_CACHE_HOME:-$HERE/.tt_cache}" # override upstream's /mnt/... default | |
| export XDG_CACHE_HOME="${XDG_CACHE_HOME:-$HERE/.cache}" # generic catch-all (triton, etc.) | |
| export TRITON_CACHE_DIR="${TRITON_CACHE_DIR:-$HERE/.cache/triton}" | |
| export TORCHINDUCTOR_CACHE_DIR="${TORCHINDUCTOR_CACHE_DIR:-$HERE/.cache/inductor}" | |
| export HF_MODEL="$HERE/model-dir" | |
| export MODEL_WEIGHTS_DIR="$HERE/model-dir" | |
| export TT_MODEL_WEIGHTS_REVISION="${TT_MODEL_WEIGHTS_REVISION:-1a5f363a3dd2d1cc456c28b8abbb403b9555efaf}" | |
| export ARCH_NAME="blackhole" | |
| export QWEN36_SKIP_VISION="1" | |
| export QWEN_SDPA_BF8="1" | |
| export TORCHDYNAMO_DISABLE="1" | |
| export TT_QWEN35_TEXT_VER="qwen36_blackhole" | |
| export VLLM_RPC_TIMEOUT="900000" | |
| export VLLM_CONFIGURE_LOGGING="1" | |
| export QWEN36_DRAFTER="dflash2" | |
| export DFLASH_WEIGHTS="incoai/Qwen3.8-27B-DFlash2@dedf8df68adfb1afeaf7b7480c0a0243108177b4" | |
| export QWEN36_DFLASH_TP="1" | |
| export QWEN36_DFLASH_BLOCK="8" | |
| export QWEN36_DFLASH_FOLD_SEED="1" | |
| export QWEN36_DFLASH_SERVE_BLOCK="32" | |
| export QWEN36_PREFILL_BUCKET_TRACE="0" | |
| export TT_METAL_PINNED_MEMORY_CACHE_LIMIT_BYTES="0" | |
| export QWEN36_GDN_SPEC_FUSED="1" | |
| export QWEN36_MAX_TOKENS_ALL_USERS="262144" | |
| CMD=("$PYBIN" -m vllm.entrypoints.openai.api_server --model "$HERE/model-dir" --max_num_seqs 4 --block_size 64 --max_model_len 262144 --additional-config '{"tt": {"l1_small_size": 24576, "fabric_config": "FABRIC_1D", "trace_region_size": 1073741824, "sample_on_device_mode": "decode_only"}}' --max-num-batched-tokens 65536 --enable-auto-tool-choice --tool-call-parser qwen3_coder --reasoning_parser qwen3 --no-async-scheduling "$@") | |
| # TT_MODEL_PRINT=1 (set by `tt-model serve --print`) echoes the fully-resolved command+env | |
| if [ "${TT_MODEL_PRINT:-0}" = "1" ]; then | |
| printf 'LD_PRELOAD=%s TT_METAL_HOME=%s EXTRA_MODELS_DIR=%s MESH_DEVICE=%s HF_MODEL=%s | |
| %s | |
| ' \ | |
| "$LD_PRELOAD" "$TT_METAL_HOME" "$EXTRA_MODELS_DIR" "$MESH_DEVICE" "${HF_MODEL:-}" "${CMD[*]}" | |
| exit 0 | |
| fi | |
| "$PYBIN" "$HERE/prepare_model_dir.py" | |
| exec "${CMD[@]}" | |