File size: 2,197 Bytes
63590e6
02ce710
 
63590e6
02ce710
 
af92b4d
 
 
02ce710
d287ce2
 
af92b4d
 
 
d287ce2
 
63590e6
 
 
02ce710
af92b4d
 
63590e6
 
 
 
af92b4d
 
 
 
 
 
 
63590e6
 
 
 
d287ce2
63590e6
 
 
 
af92b4d
 
63590e6
 
 
 
af92b4d
 
 
 
 
 
 
63590e6
 
af92b4d
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
import sys

PDF_FOLDER = "data/pdf_text"
IMAGE_FOLDER = "data/images"

if __name__ == "__main__":
    args = sys.argv[1:]
    mode = args[0] if args else "all"
    contextual = "--contextual" in args

    if mode == "all":
        import subprocess
        extra = ["--contextual"] if contextual else []
        subprocess.run([sys.executable, __file__, "pdf"] + extra, check=True)
        subprocess.run([sys.executable, __file__, "images"] + extra, check=True)

    elif mode == "pdf":
        from src.parsers.router import parse_folder as parse_folder_pdf
        from src.chunker import chunk_pages
        from src.vector_store import add_chunks

        label = "CONTEXTUAL" if contextual else "STANDARD"
        print(f"=== INGEST PDF ({label}) ===")
        pdf_pages = parse_folder_pdf(PDF_FOLDER, extensions={".pdf"})
        print(f"Tổng: {len(pdf_pages)} đoạn từ PDF")
        chunks = chunk_pages(pdf_pages)
        print(f"Tổng: {len(chunks)} chunks")

        if contextual and chunks:
            from src.contextualizer import add_context_to_chunks
            print("Đang sinh context cho từng chunk...")
            chunks = add_context_to_chunks(chunks, pdf_pages)
            print("Context xong!")

        if chunks:
            add_chunks(chunks)
        print("PDF xong!\n")

    elif mode == "images":
        from src.parsers.router import parse_folder as parse_folder_img
        from src.chunker import chunk_pages
        from src.vector_store import add_chunks

        label = "CONTEXTUAL" if contextual else "STANDARD"
        print(f"=== INGEST ẢNH ({label}) ===")
        image_pages = parse_folder_img(IMAGE_FOLDER, extensions={".jpg", ".jpeg", ".png"})
        print(f"Tổng: {len(image_pages)} đoạn từ ảnh")
        chunks = chunk_pages(image_pages)
        print(f"Tổng: {len(chunks)} chunks")

        if contextual and chunks:
            from src.contextualizer import add_context_to_chunks
            print("Đang sinh context cho từng chunk...")
            chunks = add_context_to_chunks(chunks, image_pages)
            print("Context xong!")

        if chunks:
            add_chunks(chunks)
        print("Ảnh xong!")