Spaces:
Sleeping
Sleeping
Download src/uploader.py from thaidinhz1/rag-vietnamese: direct link, hf CLI and curl.
- Browser
- Download file 2.99 kB
-
https://huggingface.co/spaces/thaidinhz1/rag-vietnamese/resolve/main/src/uploader.py
- Command line
-
hf download hf://spaces/thaidinhz1/rag-vietnamese/src/uploader.py
-
curl -L -o uploader.py https://huggingface.co/spaces/thaidinhz1/rag-vietnamese/resolve/main/src/uploader.py
2.99 kB
| """ | |
| Ephemeral PDF upload — parse, chunk, embed, lưu in-memory. | |
| Mất khi server restart, đủ để demo. | |
| """ | |
| import io | |
| import time | |
| from pypdf import PdfReader | |
| from src.embedder import embed_texts | |
| CHUNK_SIZE = 500 | |
| CHUNK_OVERLAP = 50 | |
| # In-memory store: list of {"text", "metadata", "embedding"} | |
| _uploaded_chunks: list[dict] = [] | |
| _uploaded_files: list[dict] = [] # {"name", "pages", "chunks"} | |
| def get_uploaded_files() -> list[dict]: | |
| return _uploaded_files | |
| def get_uploaded_chunks() -> list[dict]: | |
| return _uploaded_chunks | |
| def clear_uploads(): | |
| _uploaded_chunks.clear() | |
| _uploaded_files.clear() | |
| def _chunk_text(text: str, source: str, page: int) -> list[dict]: | |
| words = text.split() | |
| chunks = [] | |
| i = 0 | |
| while i < len(words): | |
| chunk_words = words[i:i + CHUNK_SIZE] | |
| chunk_text = " ".join(chunk_words).strip() | |
| if len(chunk_text) > 50: | |
| chunks.append({ | |
| "text": chunk_text, | |
| "metadata": {"source": source, "page": page}, | |
| }) | |
| i += CHUNK_SIZE - CHUNK_OVERLAP | |
| return chunks | |
| def ingest_pdf(filename: str, file_bytes: bytes) -> dict: | |
| """Parse PDF, chunk, embed và lưu vào in-memory store.""" | |
| reader = PdfReader(io.BytesIO(file_bytes)) | |
| all_chunks = [] | |
| for page_num, page in enumerate(reader.pages, 1): | |
| text = page.extract_text() or "" | |
| if text.strip(): | |
| chunks = _chunk_text(text, filename, page_num) | |
| all_chunks.extend(chunks) | |
| if not all_chunks: | |
| return {"filename": filename, "pages": len(reader.pages), "chunks": 0} | |
| # Embed theo batch | |
| BATCH = 20 | |
| texts = [c["text"] for c in all_chunks] | |
| embeddings = [] | |
| for i in range(0, len(texts), BATCH): | |
| batch_embs = embed_texts(texts[i:i + BATCH]) | |
| embeddings.extend(batch_embs) | |
| if i + BATCH < len(texts): | |
| time.sleep(BATCH * 1.5) | |
| for chunk, emb in zip(all_chunks, embeddings): | |
| chunk["embedding"] = emb | |
| _uploaded_chunks.append(chunk) | |
| file_info = { | |
| "name": filename, | |
| "pages": len(reader.pages), | |
| "chunks": len(all_chunks), | |
| } | |
| _uploaded_files.append(file_info) | |
| return file_info | |
| def search_uploaded(query_embedding: list[float], query_text: str, top_k: int = 10) -> list[dict]: | |
| """Cosine similarity search trên uploaded chunks.""" | |
| if not _uploaded_chunks: | |
| return [] | |
| import math | |
| def cosine(a, b): | |
| dot = sum(x * y for x, y in zip(a, b)) | |
| na = math.sqrt(sum(x * x for x in a)) | |
| nb = math.sqrt(sum(x * x for x in b)) | |
| return dot / (na * nb + 1e-9) | |
| scored = [] | |
| for chunk in _uploaded_chunks: | |
| score = cosine(query_embedding, chunk["embedding"]) | |
| scored.append({ | |
| "text": chunk["text"], | |
| "metadata": chunk["metadata"], | |
| "score": round(score, 4), | |
| }) | |
| scored.sort(key=lambda x: x["score"], reverse=True) | |
| return scored[:top_k] | |