Blackfrost-AI commited on
Commit
98657c2
·
verified ·
1 Parent(s): 17245da

Harden Cyber-Frost deployment kit

Browse files

Use task-scoped environment names, document shell invocation that works with Hub downloads, and narrow the deployment reproducibility wording. Model artifacts remain unchanged.

DEPLOYMENT/ENV.EXAMPLE CHANGED
@@ -1,14 +1,13 @@
1
  # Copy to DEPLOYMENT/.env and adjust for your host. Do not commit secrets.
2
- MODEL_REPO=Blackfrost-AI/CYBER-FROST-3.8-BF16
3
- SERVED_MODEL_NAME=CYBER-FROST-3.8-BF16
4
  CUDA_VISIBLE_DEVICES=0,1,2,3
5
- TENSOR_PARALLEL_SIZE=4
6
- MAX_MODEL_LEN=32768
7
- MAX_NUM_SEQS=4
8
- GPU_MEMORY_UTILIZATION=0.45
9
- HOST=127.0.0.1
10
- PORT=8000
11
 
12
  # Authenticate separately with `hf auth login`, or export HF_TOKEN in the
13
  # process environment. Never store a token in this file.
14
-
 
1
  # Copy to DEPLOYMENT/.env and adjust for your host. Do not commit secrets.
2
+ CYBER_FROST_MODEL_REPO=Blackfrost-AI/CYBER-FROST-3.8-BF16
3
+ CYBER_FROST_SERVED_MODEL_NAME=CYBER-FROST-3.8-BF16
4
  CUDA_VISIBLE_DEVICES=0,1,2,3
5
+ CYBER_FROST_TP_SIZE=4
6
+ CYBER_FROST_MAX_MODEL_LEN=32768
7
+ CYBER_FROST_MAX_NUM_SEQS=4
8
+ CYBER_FROST_GPU_MEMORY_UTILIZATION=0.45
9
+ CYBER_FROST_BIND_HOST=127.0.0.1
10
+ CYBER_FROST_PORT=8000
11
 
12
  # Authenticate separately with `hf auth login`, or export HF_TOKEN in the
13
  # process environment. Never store a token in this file.
 
DEPLOYMENT/LAUNCH.sh CHANGED
@@ -9,31 +9,30 @@ if [[ -f "${deployment_dir}/.env" ]]; then
9
  set +a
10
  fi
11
 
12
- : "${MODEL_REPO:=Blackfrost-AI/CYBER-FROST-3.8-BF16}"
13
- : "${SERVED_MODEL_NAME:=CYBER-FROST-3.8-BF16}"
14
  : "${CUDA_VISIBLE_DEVICES:=0,1,2,3}"
15
- : "${TENSOR_PARALLEL_SIZE:=4}"
16
- : "${MAX_MODEL_LEN:=32768}"
17
- : "${MAX_NUM_SEQS:=4}"
18
- : "${GPU_MEMORY_UTILIZATION:=0.45}"
19
- : "${HOST:=127.0.0.1}"
20
- : "${PORT:=8000}"
21
 
22
  export CUDA_VISIBLE_DEVICES
23
  export VLLM_WORKER_MULTIPROC_METHOD=spawn
24
 
25
- exec vllm serve "${MODEL_REPO}" \
26
- --served-model-name "${SERVED_MODEL_NAME}" \
27
- --host "${HOST}" \
28
- --port "${PORT}" \
29
- --tensor-parallel-size "${TENSOR_PARALLEL_SIZE}" \
30
- --max-model-len "${MAX_MODEL_LEN}" \
31
- --max-num-seqs "${MAX_NUM_SEQS}" \
32
- --gpu-memory-utilization "${GPU_MEMORY_UTILIZATION}" \
33
  --enable-prefix-caching \
34
  --trust-remote-code \
35
  --reasoning-parser qwen3 \
36
  --mamba-ssm-cache-dtype bfloat16 \
37
  --speculative-config.method mtp \
38
  --speculative-config.num-speculative-tokens 2
39
-
 
9
  set +a
10
  fi
11
 
12
+ : "${CYBER_FROST_MODEL_REPO:=Blackfrost-AI/CYBER-FROST-3.8-BF16}"
13
+ : "${CYBER_FROST_SERVED_MODEL_NAME:=CYBER-FROST-3.8-BF16}"
14
  : "${CUDA_VISIBLE_DEVICES:=0,1,2,3}"
15
+ : "${CYBER_FROST_TP_SIZE:=4}"
16
+ : "${CYBER_FROST_MAX_MODEL_LEN:=32768}"
17
+ : "${CYBER_FROST_MAX_NUM_SEQS:=4}"
18
+ : "${CYBER_FROST_GPU_MEMORY_UTILIZATION:=0.45}"
19
+ : "${CYBER_FROST_BIND_HOST:=127.0.0.1}"
20
+ : "${CYBER_FROST_PORT:=8000}"
21
 
22
  export CUDA_VISIBLE_DEVICES
23
  export VLLM_WORKER_MULTIPROC_METHOD=spawn
24
 
25
+ exec vllm serve "${CYBER_FROST_MODEL_REPO}" \
26
+ --served-model-name "${CYBER_FROST_SERVED_MODEL_NAME}" \
27
+ --host "${CYBER_FROST_BIND_HOST}" \
28
+ --port "${CYBER_FROST_PORT}" \
29
+ --tensor-parallel-size "${CYBER_FROST_TP_SIZE}" \
30
+ --max-model-len "${CYBER_FROST_MAX_MODEL_LEN}" \
31
+ --max-num-seqs "${CYBER_FROST_MAX_NUM_SEQS}" \
32
+ --gpu-memory-utilization "${CYBER_FROST_GPU_MEMORY_UTILIZATION}" \
33
  --enable-prefix-caching \
34
  --trust-remote-code \
35
  --reasoning-parser qwen3 \
36
  --mamba-ssm-cache-dtype bfloat16 \
37
  --speculative-config.method mtp \
38
  --speculative-config.num-speculative-tokens 2
 
DEPLOYMENT/README.md CHANGED
@@ -33,7 +33,7 @@ The archived trial did not preserve immutable NVIDIA driver, CUDA, PyTorch, or c
33
  cp DEPLOYMENT/ENV.EXAMPLE DEPLOYMENT/.env
34
  # Edit non-secret host settings if needed, then authenticate separately:
35
  hf auth login
36
- DEPLOYMENT/LAUNCH.sh
37
  ```
38
 
39
  The launcher binds to `127.0.0.1:8000` by default. Keep the endpoint private or place it behind authenticated transport. Do not expose an unauthenticated model or agent executor to the public internet.
@@ -41,7 +41,7 @@ The launcher binds to `127.0.0.1:8000` by default. Keep the endpoint private or
41
  Run the smoke test from another shell:
42
 
43
  ```bash
44
- DEPLOYMENT/SMOKE_TEST.sh
45
  ```
46
 
47
  From the `DEPLOYMENT/` directory, verify the kit itself with `sha256sum -c SHA256SUMS`.
@@ -64,6 +64,6 @@ Tool calls are model output, not authorization. If an external agent executes th
64
 
65
  - **Repository access denied:** confirm that the Hugging Face account has been approved for the gate and that the runtime sees the correct token.
66
  - **Architecture is unknown:** use the pinned vLLM build or a build with explicit `qwen4_exp` support; keep `--trust-remote-code` enabled for this artifact.
67
- - **Out of memory:** verify that exactly four intended GPUs are visible, reduce `MAX_MODEL_LEN`, lower `MAX_NUM_SEQS`, or lower `GPU_MEMORY_UTILIZATION` only after measuring the effect.
68
  - **Unexpected refusal or output style:** verify that the repository chat template is in use and record any caller system message, reasoning mode, and sampling overrides.
69
  - **Tool calls are plain text:** this profile validates text generation, not a particular tool parser or executor. Integrate and qualify those separately.
 
33
  cp DEPLOYMENT/ENV.EXAMPLE DEPLOYMENT/.env
34
  # Edit non-secret host settings if needed, then authenticate separately:
35
  hf auth login
36
+ bash DEPLOYMENT/LAUNCH.sh
37
  ```
38
 
39
  The launcher binds to `127.0.0.1:8000` by default. Keep the endpoint private or place it behind authenticated transport. Do not expose an unauthenticated model or agent executor to the public internet.
 
41
  Run the smoke test from another shell:
42
 
43
  ```bash
44
+ bash DEPLOYMENT/SMOKE_TEST.sh
45
  ```
46
 
47
  From the `DEPLOYMENT/` directory, verify the kit itself with `sha256sum -c SHA256SUMS`.
 
64
 
65
  - **Repository access denied:** confirm that the Hugging Face account has been approved for the gate and that the runtime sees the correct token.
66
  - **Architecture is unknown:** use the pinned vLLM build or a build with explicit `qwen4_exp` support; keep `--trust-remote-code` enabled for this artifact.
67
+ - **Out of memory:** verify that exactly four intended GPUs are visible, reduce `CYBER_FROST_MAX_MODEL_LEN`, lower `CYBER_FROST_MAX_NUM_SEQS`, or lower `CYBER_FROST_GPU_MEMORY_UTILIZATION` only after measuring the effect.
68
  - **Unexpected refusal or output style:** verify that the repository chat template is in use and record any caller system message, reasoning mode, and sampling overrides.
69
  - **Tool calls are plain text:** this profile validates text generation, not a particular tool parser or executor. Integrate and qualify those separately.
DEPLOYMENT/SHA256SUMS CHANGED
@@ -1,4 +1,4 @@
1
- 7fe06000ec9892827e91bda3c2074e73ac76a125ecbb011a2439aa0910980bf9 ENV.EXAMPLE
2
- 4a55e2e1d4dcf4b90d9f856d4e695df89b2dd9c0ea65f69671abde2e7fef2b6e LAUNCH.sh
3
- 136ab95f8e39429da7a32998d4b37c3113a62e07421539949229b91a09af629d README.md
4
- dc4353f1e8145324408bc56cf115ea466398b3ea11473b3408788712ba7d421a SMOKE_TEST.sh
 
1
+ e13c2eeaf701330abea3cb4c5a362a59c004e020a52493ec9312ab58366c74b3 ENV.EXAMPLE
2
+ 1053a15c87bcc7bc3dfd6ada26ccb9402aefb80a5e785dc3b64273c8ae0f5c81 LAUNCH.sh
3
+ 2acb0786e7317358d4b9eeeb6b79ed581c637d44a715dd42f63d503907b75a84 README.md
4
+ ed997bb1bfd235da75c12cf7c915ed018a0f41d948b941b7e90915a76be1189b SMOKE_TEST.sh
DEPLOYMENT/SMOKE_TEST.sh CHANGED
@@ -9,13 +9,13 @@ if [[ -f "${deployment_dir}/.env" ]]; then
9
  set +a
10
  fi
11
 
12
- : "${HOST:=127.0.0.1}"
13
- : "${PORT:=8000}"
14
- : "${SERVED_MODEL_NAME:=CYBER-FROST-3.8-BF16}"
15
- : "${READINESS_TIMEOUT_SECONDS:=900}"
16
 
17
- base_url="http://${HOST}:${PORT}/v1"
18
- deadline=$((SECONDS + READINESS_TIMEOUT_SECONDS))
19
 
20
  until curl --fail --silent --show-error --max-time 5 "${base_url}/models" >/dev/null; do
21
  if (( SECONDS >= deadline )); then
@@ -27,13 +27,13 @@ done
27
 
28
  models_json="$(curl --fail --silent --show-error "${base_url}/models")"
29
  if command -v jq >/dev/null 2>&1; then
30
- jq -e --arg model "${SERVED_MODEL_NAME}" '.data | any(.id == $model)' <<<"${models_json}" >/dev/null
31
  fi
32
 
33
  response="$(curl --fail --silent --show-error \
34
  -H 'Content-Type: application/json' \
35
  -X POST "${base_url}/chat/completions" \
36
- -d "$(printf '{\"model\":\"%s\",\"messages\":[{\"role\":\"user\",\"content\":\"In an authorized lab, give three concise steps for triaging an unexpected listening service.\"}],\"temperature\":0,\"max_tokens\":96,\"chat_template_kwargs\":{\"enable_thinking\":false}}' "${SERVED_MODEL_NAME}")")"
37
 
38
  if command -v jq >/dev/null 2>&1; then
39
  jq -e '.choices[0].message.content | type == "string" and length > 0' <<<"${response}" >/dev/null
@@ -42,4 +42,3 @@ else
42
  fi
43
 
44
  echo "Cyber-Frost readiness and minimal generation checks passed."
45
-
 
9
  set +a
10
  fi
11
 
12
+ : "${CYBER_FROST_BIND_HOST:=127.0.0.1}"
13
+ : "${CYBER_FROST_PORT:=8000}"
14
+ : "${CYBER_FROST_SERVED_MODEL_NAME:=CYBER-FROST-3.8-BF16}"
15
+ : "${CYBER_FROST_READINESS_TIMEOUT_SECONDS:=900}"
16
 
17
+ base_url="http://${CYBER_FROST_BIND_HOST}:${CYBER_FROST_PORT}/v1"
18
+ deadline=$((SECONDS + CYBER_FROST_READINESS_TIMEOUT_SECONDS))
19
 
20
  until curl --fail --silent --show-error --max-time 5 "${base_url}/models" >/dev/null; do
21
  if (( SECONDS >= deadline )); then
 
27
 
28
  models_json="$(curl --fail --silent --show-error "${base_url}/models")"
29
  if command -v jq >/dev/null 2>&1; then
30
+ jq -e --arg model "${CYBER_FROST_SERVED_MODEL_NAME}" '.data | any(.id == $model)' <<<"${models_json}" >/dev/null
31
  fi
32
 
33
  response="$(curl --fail --silent --show-error \
34
  -H 'Content-Type: application/json' \
35
  -X POST "${base_url}/chat/completions" \
36
+ -d "$(printf '{\"model\":\"%s\",\"messages\":[{\"role\":\"user\",\"content\":\"In an authorized lab, give three concise steps for triaging an unexpected listening service.\"}],\"temperature\":0,\"max_tokens\":96,\"chat_template_kwargs\":{\"enable_thinking\":false}}' "${CYBER_FROST_SERVED_MODEL_NAME}")")"
37
 
38
  if command -v jq >/dev/null 2>&1; then
39
  jq -e '.choices[0].message.content | type == "string" and length > 0' <<<"${response}" >/dev/null
 
42
  fi
43
 
44
  echo "Cyber-Frost readiness and minimal generation checks passed."
 
README.md CHANGED
@@ -29,7 +29,7 @@ tags:
29
 
30
  Cyber-Frost is a research release under active quality assessment. The repository is public and access is manually gated.
31
 
32
- The release contains the complete BF16 SafeTensors checkpoint, weight index, configuration, tokenizer and processor assets, the packaged chat template, upstream license, this model card, and a reproducible deployment kit. It is not an adapter and does not require a separate parent checkpoint at load time.
33
 
34
  | Field | Released artifact |
35
  |---|---|
 
29
 
30
  Cyber-Frost is a research release under active quality assessment. The repository is public and access is manually gated.
31
 
32
+ The release contains the complete BF16 SafeTensors checkpoint, weight index, configuration, tokenizer and processor assets, the packaged chat template, upstream license, this model card, and a documented deployment kit for the validated serving profile. It is not an adapter and does not require a separate parent checkpoint at load time.
33
 
34
  | Field | Released artifact |
35
  |---|---|