File size: 2,986 Bytes
f7f0189
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
"""
Ephemeral PDF upload — parse, chunk, embed, lưu in-memory.
Mất khi server restart, đủ để demo.
"""
import io
import time
from pypdf import PdfReader
from src.embedder import embed_texts

CHUNK_SIZE = 500
CHUNK_OVERLAP = 50

# In-memory store: list of {"text", "metadata", "embedding"}
_uploaded_chunks: list[dict] = []
_uploaded_files: list[dict] = []  # {"name", "pages", "chunks"}


def get_uploaded_files() -> list[dict]:
    return _uploaded_files


def get_uploaded_chunks() -> list[dict]:
    return _uploaded_chunks


def clear_uploads():
    _uploaded_chunks.clear()
    _uploaded_files.clear()


def _chunk_text(text: str, source: str, page: int) -> list[dict]:
    words = text.split()
    chunks = []
    i = 0
    while i < len(words):
        chunk_words = words[i:i + CHUNK_SIZE]
        chunk_text = " ".join(chunk_words).strip()
        if len(chunk_text) > 50:
            chunks.append({
                "text": chunk_text,
                "metadata": {"source": source, "page": page},
            })
        i += CHUNK_SIZE - CHUNK_OVERLAP
    return chunks


def ingest_pdf(filename: str, file_bytes: bytes) -> dict:
    """Parse PDF, chunk, embed và lưu vào in-memory store."""
    reader = PdfReader(io.BytesIO(file_bytes))
    all_chunks = []

    for page_num, page in enumerate(reader.pages, 1):
        text = page.extract_text() or ""
        if text.strip():
            chunks = _chunk_text(text, filename, page_num)
            all_chunks.extend(chunks)

    if not all_chunks:
        return {"filename": filename, "pages": len(reader.pages), "chunks": 0}

    # Embed theo batch
    BATCH = 20
    texts = [c["text"] for c in all_chunks]
    embeddings = []
    for i in range(0, len(texts), BATCH):
        batch_embs = embed_texts(texts[i:i + BATCH])
        embeddings.extend(batch_embs)
        if i + BATCH < len(texts):
            time.sleep(BATCH * 1.5)

    for chunk, emb in zip(all_chunks, embeddings):
        chunk["embedding"] = emb
        _uploaded_chunks.append(chunk)

    file_info = {
        "name": filename,
        "pages": len(reader.pages),
        "chunks": len(all_chunks),
    }
    _uploaded_files.append(file_info)
    return file_info


def search_uploaded(query_embedding: list[float], query_text: str, top_k: int = 10) -> list[dict]:
    """Cosine similarity search trên uploaded chunks."""
    if not _uploaded_chunks:
        return []

    import math

    def cosine(a, b):
        dot = sum(x * y for x, y in zip(a, b))
        na = math.sqrt(sum(x * x for x in a))
        nb = math.sqrt(sum(x * x for x in b))
        return dot / (na * nb + 1e-9)

    scored = []
    for chunk in _uploaded_chunks:
        score = cosine(query_embedding, chunk["embedding"])
        scored.append({
            "text": chunk["text"],
            "metadata": chunk["metadata"],
            "score": round(score, 4),
        })

    scored.sort(key=lambda x: x["score"], reverse=True)
    return scored[:top_k]