from fastapi import FastAPI, BackgroundTasks from pydantic import BaseModel from llama_cpp import Llama import uuid app = FastAPI() # Загружаем Llama-3-8B (сжатую до Q4_K_M — это ~5ГБ) # Она влезет в 16ГБ ОЗУ и будет на порядок умнее GPT-2 и TinyLlama llm = Llama.from_pretrained( repo_id="MaziyarPanahi/Llama-3-8B-Instruct-v0.1-GGUF", filename="*Q4_K_M.gguf", n_ctx=2048, # Контекст (память) n_threads=2 # Используем оба ядра твоего бесплатного CPU ) tasks = {} class GenerateRequest(BaseModel): prompt: str system_prompt: str = "Ты — опытный ИИ-помощник. Отвечай четко и только на русском языке." custom_id: str = None max_tokens: int = 256 temperature: float = 0.7 def run_ai(task_id, prompt, system_prompt, max_tokens, temp): try: # Форматируем запрос под стандарт Llama 3 full_prompt = f"<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\n{system_prompt}<|eot_id|><|start_header_id|>user<|end_header_id|>\n\n{prompt}<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n\n" output = llm( full_prompt, max_tokens=max_tokens, temperature=temp, stop=["<|eot_id|>", "<|end_of_text|>"], echo=False ) tasks[task_id]["result"] = output["choices"][0]["text"].strip() tasks[task_id]["status"] = "ready" except Exception as e: tasks[task_id] = {"status": "error", "result": str(e)} @app.post("/generate") async def generate(req: GenerateRequest, background_tasks: BackgroundTasks): task_id = req.custom_id if req.custom_id else str(uuid.uuid4()) tasks[task_id] = {"status": "pending", "result": ""} background_tasks.add_task( run_ai, task_id, req.prompt, req.system_prompt, req.max_tokens, req.temperature ) return {"task_id": task_id} @app.get("/check/{task_id}") async def check(task_id: str): data = tasks.get(task_id) if not data: return {"task_id": task_id, "status": "not_found", "result": None} return {"task_id": task_id, "status": data["status"], "result": data["result"]} @app.get("/") async def root(): return {"status": "running", "model": "Llama-3-8B-Instruct-GGUF"}