bankrlebhai commited on
Commit
5803f54
·
verified ·
1 Parent(s): c92ed6f

Upload 8 files

Browse files
Files changed (8) hide show
  1. .gitattributes +1 -0
  2. .gitignore +2 -0
  3. Dockerfile +36 -0
  4. README.md +23 -8
  5. api.py +131 -0
  6. index.html +120 -0
  7. llm_manager.py +38 -0
  8. requirements.txt +7 -0
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ demo_meting.mp3 filter=lfs diff=lfs merge=lfs -text
.gitignore ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ .env
2
+ /venv
Dockerfile ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ FROM python:3.10-slim
2
+
3
+ # Install system dependencies required for building llama-cpp-python
4
+ RUN apt-get update && apt-get install -y \
5
+ build-essential \
6
+ curl \
7
+ software-properties-common \
8
+ git \
9
+ && rm -rf /var/lib/apt/lists/*
10
+
11
+ # Set user configuration for HF Spaces
12
+ RUN useradd -m -u 1000 user
13
+ USER user
14
+ ENV HOME=/home/user \
15
+ PATH=/home/user/.local/bin:$PATH \
16
+ PYTHONUNBUFFERED=1 \
17
+ HF_HOME=/home/user/.cache/huggingface
18
+
19
+ WORKDIR $HOME/app
20
+
21
+ # Pre-create and chown the cache directory to prevent permission issues
22
+ RUN mkdir -p $HF_HOME && chown -R user:user $HF_HOME
23
+
24
+ # Install Python requirements
25
+ COPY --chown=user requirements.txt .
26
+ # Install standard dependencies first
27
+ RUN pip install --no-cache-dir -r requirements.txt
28
+
29
+ # Copy application files
30
+ COPY --chown=user . .
31
+
32
+ # Expose the port HF Spaces expects
33
+ EXPOSE 7860
34
+
35
+ # Run the API server
36
+ CMD ["uvicorn", "api:app", "--host", "0.0.0.0", "--port", "7860"]
README.md CHANGED
@@ -1,12 +1,27 @@
1
  ---
2
- title: Freellm
3
- emoji: 👁
4
- colorFrom: gray
5
- colorTo: yellow
6
- sdk: gradio
7
- sdk_version: 6.9.0
8
- app_file: app.py
9
  pinned: false
 
10
  ---
11
 
12
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  ---
2
+ title: Custom Ollama API
3
+ emoji: 🦙
4
+ colorFrom: blue
5
+ colorTo: purple
6
+ sdk: docker
 
 
7
  pinned: false
8
+ app_port: 7860
9
  ---
10
 
11
+ # Custom Ollama-Compatible LLM API
12
+
13
+ This repository hosts a custom, Ollama-compatible API for the `TeichAI/Qwen3-14B-Claude-4.5-Opus-High-Reasoning-Distill-GGUF` model, powered by `llama-cpp-python`.
14
+
15
+ It is designed to run as a Docker Space on Hugging Face.
16
+
17
+ ## Endpoints Supported
18
+ - `GET /api/tags`: Lists the available model (`qwen3`).
19
+ - `POST /api/generate`: Generate completion responses, supporting streaming.
20
+ - `POST /api/chat`: Chat interaction endpoint, supporting streaming.
21
+
22
+ ## Usage in Editors (Cursor, Zed, VS Code)
23
+ Set this space's URL as your Custom Ollama API endpoint.
24
+
25
+ Example: `https://[your-hf-username]-[your-space-name].hf.space/api`
26
+ Model Name: `qwen3`
27
+
api.py ADDED
@@ -0,0 +1,131 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from fastapi import FastAPI, Request
2
+ from fastapi.responses import StreamingResponse, JSONResponse
3
+ from llm_manager import LLMManager
4
+ import json
5
+ import time
6
+ import asyncio
7
+ from typing import Optional
8
+
9
+ app = FastAPI()
10
+ llm = None
11
+
12
+ @app.on_event("startup")
13
+ async def startup_event():
14
+ global llm
15
+ llm = LLMManager()
16
+
17
+ @app.get("/")
18
+ async def root():
19
+ return {"status": "running", "model": llm.model_file if llm else "loading"}
20
+
21
+ @app.post("/api/generate")
22
+ async def generate(request: Request):
23
+ data = await request.json()
24
+ prompt = data.get("prompt")
25
+ stream = data.get("stream", True)
26
+ model_name = data.get("model", "qwen3")
27
+
28
+ if not prompt:
29
+ return JSONResponse({"error": "Prompt is required"}, status_code=400)
30
+
31
+ def stream_response():
32
+ response = llm.generate(prompt, stream=True)
33
+ for chunk in response:
34
+ yield json.dumps({
35
+ "model": model_name,
36
+ "created_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
37
+ "response": chunk["choices"][0]["text"],
38
+ "done": False
39
+ }) + "\n"
40
+
41
+ yield json.dumps({
42
+ "model": model_name,
43
+ "created_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
44
+ "done": True,
45
+ "context": [], # Placeholder
46
+ "total_duration": 0,
47
+ "load_duration": 0,
48
+ "prompt_eval_count": 0,
49
+ "prompt_eval_duration": 0,
50
+ "eval_count": 0,
51
+ "eval_duration": 0
52
+ }) + "\n"
53
+
54
+ if stream:
55
+ return StreamingResponse(stream_response(), media_type="application/x-ndjson")
56
+ else:
57
+ response = llm.generate(prompt, stream=False)
58
+ return {
59
+ "model": model_name,
60
+ "created_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
61
+ "response": response["choices"][0]["text"],
62
+ "done": True,
63
+ "context": [],
64
+ "total_duration": 0,
65
+ "load_duration": 0,
66
+ "prompt_eval_count": 0,
67
+ "prompt_eval_duration": 0,
68
+ "eval_count": 0,
69
+ "eval_duration": 0
70
+ }
71
+
72
+ @app.post("/api/chat")
73
+ async def chat(request: Request):
74
+ data = await request.json()
75
+ messages = data.get("messages", [])
76
+ stream = data.get("stream", True)
77
+ model_name = data.get("model", "qwen3")
78
+
79
+ def stream_chat():
80
+ response = llm.chat_completion(messages, stream=True)
81
+ for chunk in response:
82
+ if "choices" in chunk and len(chunk["choices"]) > 0:
83
+ delta = chunk["choices"][0].get("delta", {})
84
+ content = delta.get("content", "")
85
+ yield json.dumps({
86
+ "model": model_name,
87
+ "created_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
88
+ "message": {"role": "assistant", "content": content},
89
+ "done": False
90
+ }) + "\n"
91
+
92
+ yield json.dumps({
93
+ "model": model_name,
94
+ "created_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
95
+ "done": True
96
+ }) + "\n"
97
+
98
+ if stream:
99
+ return StreamingResponse(stream_chat(), media_type="application/x-ndjson")
100
+ else:
101
+ response = llm.chat_completion(messages, stream=False)
102
+ return {
103
+ "model": model_name,
104
+ "created_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
105
+ "message": response["choices"][0]["message"],
106
+ "done": True
107
+ }
108
+
109
+ @app.get("/api/tags")
110
+ async def tags():
111
+ return {
112
+ "models": [
113
+ {
114
+ "name": "qwen3",
115
+ "modified_at": time.strftime("%Y-%m-%dT%H:%M:%S.000Z", time.gmtime()),
116
+ "size": 0, # TBD
117
+ "digest": "qwen3-digest",
118
+ "details": {
119
+ "format": "gguf",
120
+ "family": "qwen",
121
+ "families": ["qwen"],
122
+ "parameter_size": "14B",
123
+ "quantization_level": "Q4_K_M"
124
+ }
125
+ }
126
+ ]
127
+ }
128
+
129
+ if __name__ == "__main__":
130
+ import uvicorn
131
+ uvicorn.run(app, host="0.0.0.0", port=7860)
index.html ADDED
@@ -0,0 +1,120 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ <!DOCTYPE html>
2
+ <html lang="en">
3
+ <head>
4
+ <meta charset="UTF-8">
5
+ <meta name="viewport" content="width=device-width, initial-scale=1.0">
6
+ <title>Qwen3 API Server</title>
7
+ <style>
8
+ @import url('https://fonts.googleapis.com/css2?family=Inter:wght@400;500;600&display=swap');
9
+ body {
10
+ font-family: 'Inter', sans-serif;
11
+ background-color: #0f172a;
12
+ color: #f8fafc;
13
+ display: flex;
14
+ justify-content: center;
15
+ align-items: center;
16
+ min-height: 100vh;
17
+ margin: 0;
18
+ padding: 20px;
19
+ }
20
+ .container {
21
+ background-color: #1e293b;
22
+ padding: 40px;
23
+ border-radius: 12px;
24
+ box-shadow: 0 10px 15px -3px rgba(0, 0, 0, 0.5);
25
+ max-width: 600px;
26
+ width: 100%;
27
+ }
28
+ h1 {
29
+ color: #38bdf8;
30
+ margin-top: 0;
31
+ margin-bottom: 24px;
32
+ }
33
+ .status {
34
+ display: inline-block;
35
+ padding: 6px 12px;
36
+ border-radius: 9999px;
37
+ font-size: 14px;
38
+ font-weight: 500;
39
+ background-color: #064e3b;
40
+ color: #34d399;
41
+ margin-bottom: 24px;
42
+ }
43
+ code {
44
+ background-color: #0f172a;
45
+ padding: 2px 6px;
46
+ border-radius: 4px;
47
+ color: #fca5a5;
48
+ font-family: monospace;
49
+ }
50
+ pre {
51
+ background-color: #0f172a;
52
+ padding: 16px;
53
+ border-radius: 8px;
54
+ overflow-x: auto;
55
+ color: #94a3b8;
56
+ border: 1px solid #334155;
57
+ }
58
+ .section {
59
+ margin-bottom: 24px;
60
+ }
61
+ h2 {
62
+ font-size: 1.2rem;
63
+ color: #cbd5e1;
64
+ margin-bottom: 12px;
65
+ }
66
+ </style>
67
+ </head>
68
+ <body>
69
+ <div class="container">
70
+ <h1>Ollama-Compatible API</h1>
71
+ <div class="status" id="status-badge">Checking status...</div>
72
+
73
+ <div class="section">
74
+ <h2>Model Info</h2>
75
+ <p>Running model: <code id="model-name">Loading...</code></p>
76
+ </div>
77
+
78
+ <div class="section">
79
+ <h2>Editor Integration</h2>
80
+ <p>To use this in Zed or Cursor, configure your custom Ollama settings:</p>
81
+ <pre>
82
+ {
83
+ "api_url": "https://[your-space].hf.space",
84
+ "model": "qwen3"
85
+ }</pre>
86
+ </div>
87
+
88
+ <div class="section">
89
+ <h2>Available Endpoints</h2>
90
+ <ul>
91
+ <li><code>GET /api/tags</code></li>
92
+ <li><code>POST /api/generate</code></li>
93
+ <li><code>POST /api/chat</code></li>
94
+ </ul>
95
+ </div>
96
+ </div>
97
+
98
+ <script>
99
+ // Check API status
100
+ fetch('/api/tags')
101
+ .then(response => response.json())
102
+ .then(data => {
103
+ const badge = document.getElementById('status-badge');
104
+ badge.textContent = 'API is Online';
105
+ badge.style.backgroundColor = '#064e3b';
106
+ badge.style.color = '#34d399';
107
+
108
+ if (data.models && data.models.length > 0) {
109
+ document.getElementById('model-name').textContent = data.models[0].name;
110
+ }
111
+ })
112
+ .catch(error => {
113
+ const badge = document.getElementById('status-badge');
114
+ badge.textContent = 'API is Offline';
115
+ badge.style.backgroundColor = '#7f1d1d';
116
+ badge.style.color = '#fca5a5';
117
+ });
118
+ </script>
119
+ </body>
120
+ </html>
llm_manager.py ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ from llama_cpp import Llama
3
+ from huggingface_hub import hf_hub_download
4
+
5
+ class LLMManager:
6
+ def __init__(self, model_repo="TeichAI/Qwen3-14B-Claude-4.5-Opus-High-Reasoning-Distill-GGUF",
7
+ model_file="qwen3-14b-claude-4.5-opus-high-reasoning-distill-q4_k_m.gguf"):
8
+ self.model_repo = model_repo
9
+ self.model_file = model_file
10
+ self.model = None
11
+ self._load_model()
12
+
13
+ def _load_model(self):
14
+ print(f"Downloading model {self.model_file} from {self.model_repo}...")
15
+ model_path = hf_hub_download(repo_id=self.model_repo, filename=self.model_file)
16
+
17
+ print("Initializing Llama model...")
18
+ # n_gpu_layers=-1 will use all available GPU layers if compiled with GPU support
19
+ self.model = Llama(
20
+ model_path=model_path,
21
+ n_ctx=4096,
22
+ n_threads=os.cpu_count(),
23
+ n_gpu_layers=-1 if os.path.exists("/dev/nvidia0") else 0
24
+ )
25
+ print("Model loaded successfully.")
26
+
27
+ def generate(self, prompt, stream=True, **kwargs):
28
+ if stream:
29
+ return self.model(prompt, stream=True, **kwargs)
30
+ else:
31
+ return self.model(prompt, stream=False, **kwargs)
32
+
33
+ def chat_completion(self, messages, stream=True, **kwargs):
34
+ return self.model.create_chat_completion(
35
+ messages=messages,
36
+ stream=stream,
37
+ **kwargs
38
+ )
requirements.txt ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ fastapi
2
+ uvicorn
3
+ python-multipart
4
+ python-dotenv
5
+ llama-cpp-python
6
+ huggingface_hub
7
+ sse-starlette