Spaces:
Paused
Paused
Upload 8 files
Browse files- .gitattributes +1 -0
- .gitignore +2 -0
- Dockerfile +36 -0
- README.md +23 -8
- api.py +131 -0
- index.html +120 -0
- llm_manager.py +38 -0
- requirements.txt +7 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
demo_meting.mp3 filter=lfs diff=lfs merge=lfs -text
|
.gitignore
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
.env
|
| 2 |
+
/venv
|
Dockerfile
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
FROM python:3.10-slim
|
| 2 |
+
|
| 3 |
+
# Install system dependencies required for building llama-cpp-python
|
| 4 |
+
RUN apt-get update && apt-get install -y \
|
| 5 |
+
build-essential \
|
| 6 |
+
curl \
|
| 7 |
+
software-properties-common \
|
| 8 |
+
git \
|
| 9 |
+
&& rm -rf /var/lib/apt/lists/*
|
| 10 |
+
|
| 11 |
+
# Set user configuration for HF Spaces
|
| 12 |
+
RUN useradd -m -u 1000 user
|
| 13 |
+
USER user
|
| 14 |
+
ENV HOME=/home/user \
|
| 15 |
+
PATH=/home/user/.local/bin:$PATH \
|
| 16 |
+
PYTHONUNBUFFERED=1 \
|
| 17 |
+
HF_HOME=/home/user/.cache/huggingface
|
| 18 |
+
|
| 19 |
+
WORKDIR $HOME/app
|
| 20 |
+
|
| 21 |
+
# Pre-create and chown the cache directory to prevent permission issues
|
| 22 |
+
RUN mkdir -p $HF_HOME && chown -R user:user $HF_HOME
|
| 23 |
+
|
| 24 |
+
# Install Python requirements
|
| 25 |
+
COPY --chown=user requirements.txt .
|
| 26 |
+
# Install standard dependencies first
|
| 27 |
+
RUN pip install --no-cache-dir -r requirements.txt
|
| 28 |
+
|
| 29 |
+
# Copy application files
|
| 30 |
+
COPY --chown=user . .
|
| 31 |
+
|
| 32 |
+
# Expose the port HF Spaces expects
|
| 33 |
+
EXPOSE 7860
|
| 34 |
+
|
| 35 |
+
# Run the API server
|
| 36 |
+
CMD ["uvicorn", "api:app", "--host", "0.0.0.0", "--port", "7860"]
|
README.md
CHANGED
|
@@ -1,12 +1,27 @@
|
|
| 1 |
---
|
| 2 |
-
title:
|
| 3 |
-
emoji:
|
| 4 |
-
colorFrom:
|
| 5 |
-
colorTo:
|
| 6 |
-
sdk:
|
| 7 |
-
sdk_version: 6.9.0
|
| 8 |
-
app_file: app.py
|
| 9 |
pinned: false
|
|
|
|
| 10 |
---
|
| 11 |
|
| 12 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
---
|
| 2 |
+
title: Custom Ollama API
|
| 3 |
+
emoji: 🦙
|
| 4 |
+
colorFrom: blue
|
| 5 |
+
colorTo: purple
|
| 6 |
+
sdk: docker
|
|
|
|
|
|
|
| 7 |
pinned: false
|
| 8 |
+
app_port: 7860
|
| 9 |
---
|
| 10 |
|
| 11 |
+
# Custom Ollama-Compatible LLM API
|
| 12 |
+
|
| 13 |
+
This repository hosts a custom, Ollama-compatible API for the `TeichAI/Qwen3-14B-Claude-4.5-Opus-High-Reasoning-Distill-GGUF` model, powered by `llama-cpp-python`.
|
| 14 |
+
|
| 15 |
+
It is designed to run as a Docker Space on Hugging Face.
|
| 16 |
+
|
| 17 |
+
## Endpoints Supported
|
| 18 |
+
- `GET /api/tags`: Lists the available model (`qwen3`).
|
| 19 |
+
- `POST /api/generate`: Generate completion responses, supporting streaming.
|
| 20 |
+
- `POST /api/chat`: Chat interaction endpoint, supporting streaming.
|
| 21 |
+
|
| 22 |
+
## Usage in Editors (Cursor, Zed, VS Code)
|
| 23 |
+
Set this space's URL as your Custom Ollama API endpoint.
|
| 24 |
+
|
| 25 |
+
Example: `https://[your-hf-username]-[your-space-name].hf.space/api`
|
| 26 |
+
Model Name: `qwen3`
|
| 27 |
+
|
api.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from fastapi import FastAPI, Request
|
| 2 |
+
from fastapi.responses import StreamingResponse, JSONResponse
|
| 3 |
+
from llm_manager import LLMManager
|
| 4 |
+
import json
|
| 5 |
+
import time
|
| 6 |
+
import asyncio
|
| 7 |
+
from typing import Optional
|
| 8 |
+
|
| 9 |
+
app = FastAPI()
|
| 10 |
+
llm = None
|
| 11 |
+
|
| 12 |
+
@app.on_event("startup")
|
| 13 |
+
async def startup_event():
|
| 14 |
+
global llm
|
| 15 |
+
llm = LLMManager()
|
| 16 |
+
|
| 17 |
+
@app.get("/")
|
| 18 |
+
async def root():
|
| 19 |
+
return {"status": "running", "model": llm.model_file if llm else "loading"}
|
| 20 |
+
|
| 21 |
+
@app.post("/api/generate")
|
| 22 |
+
async def generate(request: Request):
|
| 23 |
+
data = await request.json()
|
| 24 |
+
prompt = data.get("prompt")
|
| 25 |
+
stream = data.get("stream", True)
|
| 26 |
+
model_name = data.get("model", "qwen3")
|
| 27 |
+
|
| 28 |
+
if not prompt:
|
| 29 |
+
return JSONResponse({"error": "Prompt is required"}, status_code=400)
|
| 30 |
+
|
| 31 |
+
def stream_response():
|
| 32 |
+
response = llm.generate(prompt, stream=True)
|
| 33 |
+
for chunk in response:
|
| 34 |
+
yield json.dumps({
|
| 35 |
+
"model": model_name,
|
| 36 |
+
"created_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
|
| 37 |
+
"response": chunk["choices"][0]["text"],
|
| 38 |
+
"done": False
|
| 39 |
+
}) + "\n"
|
| 40 |
+
|
| 41 |
+
yield json.dumps({
|
| 42 |
+
"model": model_name,
|
| 43 |
+
"created_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
|
| 44 |
+
"done": True,
|
| 45 |
+
"context": [], # Placeholder
|
| 46 |
+
"total_duration": 0,
|
| 47 |
+
"load_duration": 0,
|
| 48 |
+
"prompt_eval_count": 0,
|
| 49 |
+
"prompt_eval_duration": 0,
|
| 50 |
+
"eval_count": 0,
|
| 51 |
+
"eval_duration": 0
|
| 52 |
+
}) + "\n"
|
| 53 |
+
|
| 54 |
+
if stream:
|
| 55 |
+
return StreamingResponse(stream_response(), media_type="application/x-ndjson")
|
| 56 |
+
else:
|
| 57 |
+
response = llm.generate(prompt, stream=False)
|
| 58 |
+
return {
|
| 59 |
+
"model": model_name,
|
| 60 |
+
"created_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
|
| 61 |
+
"response": response["choices"][0]["text"],
|
| 62 |
+
"done": True,
|
| 63 |
+
"context": [],
|
| 64 |
+
"total_duration": 0,
|
| 65 |
+
"load_duration": 0,
|
| 66 |
+
"prompt_eval_count": 0,
|
| 67 |
+
"prompt_eval_duration": 0,
|
| 68 |
+
"eval_count": 0,
|
| 69 |
+
"eval_duration": 0
|
| 70 |
+
}
|
| 71 |
+
|
| 72 |
+
@app.post("/api/chat")
|
| 73 |
+
async def chat(request: Request):
|
| 74 |
+
data = await request.json()
|
| 75 |
+
messages = data.get("messages", [])
|
| 76 |
+
stream = data.get("stream", True)
|
| 77 |
+
model_name = data.get("model", "qwen3")
|
| 78 |
+
|
| 79 |
+
def stream_chat():
|
| 80 |
+
response = llm.chat_completion(messages, stream=True)
|
| 81 |
+
for chunk in response:
|
| 82 |
+
if "choices" in chunk and len(chunk["choices"]) > 0:
|
| 83 |
+
delta = chunk["choices"][0].get("delta", {})
|
| 84 |
+
content = delta.get("content", "")
|
| 85 |
+
yield json.dumps({
|
| 86 |
+
"model": model_name,
|
| 87 |
+
"created_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
|
| 88 |
+
"message": {"role": "assistant", "content": content},
|
| 89 |
+
"done": False
|
| 90 |
+
}) + "\n"
|
| 91 |
+
|
| 92 |
+
yield json.dumps({
|
| 93 |
+
"model": model_name,
|
| 94 |
+
"created_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
|
| 95 |
+
"done": True
|
| 96 |
+
}) + "\n"
|
| 97 |
+
|
| 98 |
+
if stream:
|
| 99 |
+
return StreamingResponse(stream_chat(), media_type="application/x-ndjson")
|
| 100 |
+
else:
|
| 101 |
+
response = llm.chat_completion(messages, stream=False)
|
| 102 |
+
return {
|
| 103 |
+
"model": model_name,
|
| 104 |
+
"created_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
|
| 105 |
+
"message": response["choices"][0]["message"],
|
| 106 |
+
"done": True
|
| 107 |
+
}
|
| 108 |
+
|
| 109 |
+
@app.get("/api/tags")
|
| 110 |
+
async def tags():
|
| 111 |
+
return {
|
| 112 |
+
"models": [
|
| 113 |
+
{
|
| 114 |
+
"name": "qwen3",
|
| 115 |
+
"modified_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
|
| 116 |
+
"size": 0, # TBD
|
| 117 |
+
"digest": "qwen3-digest",
|
| 118 |
+
"details": {
|
| 119 |
+
"format": "gguf",
|
| 120 |
+
"family": "qwen",
|
| 121 |
+
"families": ["qwen"],
|
| 122 |
+
"parameter_size": "14B",
|
| 123 |
+
"quantization_level": "Q4_K_M"
|
| 124 |
+
}
|
| 125 |
+
}
|
| 126 |
+
]
|
| 127 |
+
}
|
| 128 |
+
|
| 129 |
+
if __name__ == "__main__":
|
| 130 |
+
import uvicorn
|
| 131 |
+
uvicorn.run(app, host="0.0.0.0", port=7860)
|
index.html
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
<!DOCTYPE html>
|
| 2 |
+
<html lang="en">
|
| 3 |
+
<head>
|
| 4 |
+
<meta charset="UTF-8">
|
| 5 |
+
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
| 6 |
+
<title>Qwen3 API Server</title>
|
| 7 |
+
<style>
|
| 8 |
+
@import url('https://fonts.googleapis.com/css2?family=Inter:wght@400;500;600&display=swap');
|
| 9 |
+
body {
|
| 10 |
+
font-family: 'Inter', sans-serif;
|
| 11 |
+
background-color: #0f172a;
|
| 12 |
+
color: #f8fafc;
|
| 13 |
+
display: flex;
|
| 14 |
+
justify-content: center;
|
| 15 |
+
align-items: center;
|
| 16 |
+
min-height: 100vh;
|
| 17 |
+
margin: 0;
|
| 18 |
+
padding: 20px;
|
| 19 |
+
}
|
| 20 |
+
.container {
|
| 21 |
+
background-color: #1e293b;
|
| 22 |
+
padding: 40px;
|
| 23 |
+
border-radius: 12px;
|
| 24 |
+
box-shadow: 0 10px 15px -3px rgba(0, 0, 0, 0.5);
|
| 25 |
+
max-width: 600px;
|
| 26 |
+
width: 100%;
|
| 27 |
+
}
|
| 28 |
+
h1 {
|
| 29 |
+
color: #38bdf8;
|
| 30 |
+
margin-top: 0;
|
| 31 |
+
margin-bottom: 24px;
|
| 32 |
+
}
|
| 33 |
+
.status {
|
| 34 |
+
display: inline-block;
|
| 35 |
+
padding: 6px 12px;
|
| 36 |
+
border-radius: 9999px;
|
| 37 |
+
font-size: 14px;
|
| 38 |
+
font-weight: 500;
|
| 39 |
+
background-color: #064e3b;
|
| 40 |
+
color: #34d399;
|
| 41 |
+
margin-bottom: 24px;
|
| 42 |
+
}
|
| 43 |
+
code {
|
| 44 |
+
background-color: #0f172a;
|
| 45 |
+
padding: 2px 6px;
|
| 46 |
+
border-radius: 4px;
|
| 47 |
+
color: #fca5a5;
|
| 48 |
+
font-family: monospace;
|
| 49 |
+
}
|
| 50 |
+
pre {
|
| 51 |
+
background-color: #0f172a;
|
| 52 |
+
padding: 16px;
|
| 53 |
+
border-radius: 8px;
|
| 54 |
+
overflow-x: auto;
|
| 55 |
+
color: #94a3b8;
|
| 56 |
+
border: 1px solid #334155;
|
| 57 |
+
}
|
| 58 |
+
.section {
|
| 59 |
+
margin-bottom: 24px;
|
| 60 |
+
}
|
| 61 |
+
h2 {
|
| 62 |
+
font-size: 1.2rem;
|
| 63 |
+
color: #cbd5e1;
|
| 64 |
+
margin-bottom: 12px;
|
| 65 |
+
}
|
| 66 |
+
</style>
|
| 67 |
+
</head>
|
| 68 |
+
<body>
|
| 69 |
+
<div class="container">
|
| 70 |
+
<h1>Ollama-Compatible API</h1>
|
| 71 |
+
<div class="status" id="status-badge">Checking status...</div>
|
| 72 |
+
|
| 73 |
+
<div class="section">
|
| 74 |
+
<h2>Model Info</h2>
|
| 75 |
+
<p>Running model: <code id="model-name">Loading...</code></p>
|
| 76 |
+
</div>
|
| 77 |
+
|
| 78 |
+
<div class="section">
|
| 79 |
+
<h2>Editor Integration</h2>
|
| 80 |
+
<p>To use this in Zed or Cursor, configure your custom Ollama settings:</p>
|
| 81 |
+
<pre>
|
| 82 |
+
{
|
| 83 |
+
"api_url": "https://[your-space].hf.space",
|
| 84 |
+
"model": "qwen3"
|
| 85 |
+
}</pre>
|
| 86 |
+
</div>
|
| 87 |
+
|
| 88 |
+
<div class="section">
|
| 89 |
+
<h2>Available Endpoints</h2>
|
| 90 |
+
<ul>
|
| 91 |
+
<li><code>GET /api/tags</code></li>
|
| 92 |
+
<li><code>POST /api/generate</code></li>
|
| 93 |
+
<li><code>POST /api/chat</code></li>
|
| 94 |
+
</ul>
|
| 95 |
+
</div>
|
| 96 |
+
</div>
|
| 97 |
+
|
| 98 |
+
<script>
|
| 99 |
+
// Check API status
|
| 100 |
+
fetch('/api/tags')
|
| 101 |
+
.then(response => response.json())
|
| 102 |
+
.then(data => {
|
| 103 |
+
const badge = document.getElementById('status-badge');
|
| 104 |
+
badge.textContent = 'API is Online';
|
| 105 |
+
badge.style.backgroundColor = '#064e3b';
|
| 106 |
+
badge.style.color = '#34d399';
|
| 107 |
+
|
| 108 |
+
if (data.models && data.models.length > 0) {
|
| 109 |
+
document.getElementById('model-name').textContent = data.models[0].name;
|
| 110 |
+
}
|
| 111 |
+
})
|
| 112 |
+
.catch(error => {
|
| 113 |
+
const badge = document.getElementById('status-badge');
|
| 114 |
+
badge.textContent = 'API is Offline';
|
| 115 |
+
badge.style.backgroundColor = '#7f1d1d';
|
| 116 |
+
badge.style.color = '#fca5a5';
|
| 117 |
+
});
|
| 118 |
+
</script>
|
| 119 |
+
</body>
|
| 120 |
+
</html>
|
llm_manager.py
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import os
|
| 2 |
+
from llama_cpp import Llama
|
| 3 |
+
from huggingface_hub import hf_hub_download
|
| 4 |
+
|
| 5 |
+
class LLMManager:
|
| 6 |
+
def __init__(self, model_repo="TeichAI/Qwen3-14B-Claude-4.5-Opus-High-Reasoning-Distill-GGUF",
|
| 7 |
+
model_file="qwen3-14b-claude-4.5-opus-high-reasoning-distill-q4_k_m.gguf"):
|
| 8 |
+
self.model_repo = model_repo
|
| 9 |
+
self.model_file = model_file
|
| 10 |
+
self.model = None
|
| 11 |
+
self._load_model()
|
| 12 |
+
|
| 13 |
+
def _load_model(self):
|
| 14 |
+
print(f"Downloading model {self.model_file} from {self.model_repo}...")
|
| 15 |
+
model_path = hf_hub_download(repo_id=self.model_repo, filename=self.model_file)
|
| 16 |
+
|
| 17 |
+
print("Initializing Llama model...")
|
| 18 |
+
# n_gpu_layers=-1 will use all available GPU layers if compiled with GPU support
|
| 19 |
+
self.model = Llama(
|
| 20 |
+
model_path=model_path,
|
| 21 |
+
n_ctx=4096,
|
| 22 |
+
n_threads=os.cpu_count(),
|
| 23 |
+
n_gpu_layers=-1 if os.path.exists("/dev/nvidia0") else 0
|
| 24 |
+
)
|
| 25 |
+
print("Model loaded successfully.")
|
| 26 |
+
|
| 27 |
+
def generate(self, prompt, stream=True, **kwargs):
|
| 28 |
+
if stream:
|
| 29 |
+
return self.model(prompt, stream=True, **kwargs)
|
| 30 |
+
else:
|
| 31 |
+
return self.model(prompt, stream=False, **kwargs)
|
| 32 |
+
|
| 33 |
+
def chat_completion(self, messages, stream=True, **kwargs):
|
| 34 |
+
return self.model.create_chat_completion(
|
| 35 |
+
messages=messages,
|
| 36 |
+
stream=stream,
|
| 37 |
+
**kwargs
|
| 38 |
+
)
|
requirements.txt
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
fastapi
|
| 2 |
+
uvicorn
|
| 3 |
+
python-multipart
|
| 4 |
+
python-dotenv
|
| 5 |
+
llama-cpp-python
|
| 6 |
+
huggingface_hub
|
| 7 |
+
sse-starlette
|