Nvidia2 / app.py
maddingface's picture
Create app.py
8e4b2e1 verified
Raw
History Blame Contribute Delete
3.73 kB
import os
# Set these before importing llama_cpp.
# They affect BLAS/OpenMP-style CPU threading.
CPU_COUNT = os.cpu_count() or 2
CPU_THREADS = int(os.getenv("CPU_THREADS", str(CPU_COUNT)))
os.environ.setdefault("OMP_NUM_THREADS", str(CPU_THREADS))
os.environ.setdefault("OPENBLAS_NUM_THREADS", str(CPU_THREADS))
os.environ.setdefault("MKL_NUM_THREADS", str(CPU_THREADS))
os.environ.setdefault("NUMEXPR_NUM_THREADS", str(CPU_THREADS))
from pathlib import Path
import gradio as gr
from huggingface_hub import hf_hub_download
from llama_cpp import Llama
# Official LFM2.5 target model.
# Q4_0 is generally the fastest small CPU option.
MODEL_REPO = os.getenv(
"MODEL_REPO",
"LiquidAI/LFM2.5-2.6B-GGUF",
)
MODEL_FILE = os.getenv(
"MODEL_FILE",
"LFM2.5-2.6B-Q4_0.gguf",
)
# Keep this moderate on shared CPU Spaces.
# 2048 is faster than 4096 and is enough for many API requests.
N_CTX = int(os.getenv("N_CTX", "2048"))
# llama.cpp can use all visible CPUs, but shared Spaces may perform
# better with a slightly lower value. Override with CPU_THREADS.
N_THREADS = int(os.getenv("CPU_THREADS", str(CPU_COUNT)))
MODEL_PATH = hf_hub_download(
repo_id=MODEL_REPO,
filename=MODEL_FILE,
cache_dir="/tmp/huggingface-cache",
)
print(f"Loading model: {MODEL_PATH}")
print(f"CPU threads: {N_THREADS}")
print("GPU layers: 0")
print("Backend: CPU-only")
llm = Llama(
model_path=MODEL_PATH,
# Absolute CPU-only settings.
n_gpu_layers=0,
split_mode=0,
main_gpu=0,
# CPU parallelism.
n_threads=N_THREADS,
n_threads_batch=N_THREADS,
# Prompt-processing batch size.
# Lower this to 256 if memory is limited.
n_batch=512,
# Context size.
n_ctx=N_CTX,
# Memory/performance settings.
use_mmap=True,
use_mlock=False,
# Do not use GPU-oriented KV-cache settings.
offload_kqv=False,
flash_attn=False,
# Prevent noisy native logs after startup.
verbose=False,
)
SYSTEM_PROMPT = os.getenv(
"SYSTEM_PROMPT",
"You are a helpful, concise assistant.",
)
def make_prompt(user_prompt: str) -> str:
"""
LFM2.5 understands the chat-style format stored in the GGUF metadata.
llama-cpp-python's create_chat_completion applies the model template.
"""
return user_prompt.strip()
def generate(user_prompt: str) -> str:
if not user_prompt or not user_prompt.strip():
return "Please enter a message."
response = llm.create_chat_completion(
messages=[
{
"role": "system",
"content": SYSTEM_PROMPT,
},
{
"role": "user",
"content": make_prompt(user_prompt),
},
],
# Generation settings.
max_tokens=int(os.getenv("MAX_TOKENS", "512")),
temperature=float(os.getenv("TEMPERATURE", "0.2")),
top_p=float(os.getenv("TOP_P", "0.9")),
top_k=int(os.getenv("TOP_K", "40")),
repeat_penalty=float(os.getenv("REPEAT_PENALTY", "1.05")),
# Avoid unnecessary response metadata.
stream=False,
)
return response["choices"][0]["message"]["content"].strip()
demo = gr.Interface(
fn=generate,
inputs=gr.Textbox(
label="Prompt",
placeholder="Ask something...",
lines=5,
),
outputs=gr.Textbox(
label="Response",
lines=12,
),
title="LFM2.5 2.6B CPU API",
description="LFM2.5 running through a prebuilt CPU llama.cpp wheel.",
api_name="chat",
)
if __name__ == "__main__":
demo.queue(
max_size=16,
default_concurrency_limit=1,
).launch(
server_name="0.0.0.0",
server_port=7860,
show_api=True,
)