Download start.sh from codebyam/Llama-3.2-3B-Instruct-Q8_0-GGUF-DEMO2: direct link, hf CLI and curl.
- Browser
- Download file 347 Bytes
-
https://huggingface.co/spaces/codebyam/Llama-3.2-3B-Instruct-Q8_0-GGUF-DEMO2/resolve/e0fb7c8fca924c21a1ae1379ddc411bb5f342e72/start.sh
- Command line
-
hf download hf://spaces/codebyam/Llama-3.2-3B-Instruct-Q8_0-GGUF-DEMO2@e0fb7c8fca924c21a1ae1379ddc411bb5f342e72/start.sh
-
curl -L -o start.sh https://huggingface.co/spaces/codebyam/Llama-3.2-3B-Instruct-Q8_0-GGUF-DEMO2/resolve/e0fb7c8fca924c21a1ae1379ddc411bb5f342e72/start.sh
347 Bytes
| # Start llama-server in background | |
| cd /llama.cpp/build | |
| ./bin/llama-server --host 0.0.0.0 --port 8080 --model /models/model.q8_0.gguf --ctx-size 8192 --threads 2 & | |
| # Wait for server to initialize | |
| echo "Waiting for server to start..." | |
| until curl -s "http://localhost:8080/v1/models" >/dev/null; do | |
| sleep 1 | |
| done | |
| cd / | |
| python3 app.py |