Spaces:
Sleeping
Sleeping
Download ingest.py from thaidinhz1/rag-vietnamese: direct link, hf CLI and curl.
- Browser
- Download file 2.2 kB
-
https://huggingface.co/spaces/thaidinhz1/rag-vietnamese/resolve/main/ingest.py
- Command line
-
hf download hf://spaces/thaidinhz1/rag-vietnamese/ingest.py
-
curl -L -o ingest.py https://huggingface.co/spaces/thaidinhz1/rag-vietnamese/resolve/main/ingest.py
2.2 kB
| import sys | |
| PDF_FOLDER = "data/pdf_text" | |
| IMAGE_FOLDER = "data/images" | |
| if __name__ == "__main__": | |
| args = sys.argv[1:] | |
| mode = args[0] if args else "all" | |
| contextual = "--contextual" in args | |
| if mode == "all": | |
| import subprocess | |
| extra = ["--contextual"] if contextual else [] | |
| subprocess.run([sys.executable, __file__, "pdf"] + extra, check=True) | |
| subprocess.run([sys.executable, __file__, "images"] + extra, check=True) | |
| elif mode == "pdf": | |
| from src.parsers.router import parse_folder as parse_folder_pdf | |
| from src.chunker import chunk_pages | |
| from src.vector_store import add_chunks | |
| label = "CONTEXTUAL" if contextual else "STANDARD" | |
| print(f"=== INGEST PDF ({label}) ===") | |
| pdf_pages = parse_folder_pdf(PDF_FOLDER, extensions={".pdf"}) | |
| print(f"Tổng: {len(pdf_pages)} đoạn từ PDF") | |
| chunks = chunk_pages(pdf_pages) | |
| print(f"Tổng: {len(chunks)} chunks") | |
| if contextual and chunks: | |
| from src.contextualizer import add_context_to_chunks | |
| print("Đang sinh context cho từng chunk...") | |
| chunks = add_context_to_chunks(chunks, pdf_pages) | |
| print("Context xong!") | |
| if chunks: | |
| add_chunks(chunks) | |
| print("PDF xong!\n") | |
| elif mode == "images": | |
| from src.parsers.router import parse_folder as parse_folder_img | |
| from src.chunker import chunk_pages | |
| from src.vector_store import add_chunks | |
| label = "CONTEXTUAL" if contextual else "STANDARD" | |
| print(f"=== INGEST ẢNH ({label}) ===") | |
| image_pages = parse_folder_img(IMAGE_FOLDER, extensions={".jpg", ".jpeg", ".png"}) | |
| print(f"Tổng: {len(image_pages)} đoạn từ ảnh") | |
| chunks = chunk_pages(image_pages) | |
| print(f"Tổng: {len(chunks)} chunks") | |
| if contextual and chunks: | |
| from src.contextualizer import add_context_to_chunks | |
| print("Đang sinh context cho từng chunk...") | |
| chunks = add_context_to_chunks(chunks, image_pages) | |
| print("Context xong!") | |
| if chunks: | |
| add_chunks(chunks) | |
| print("Ảnh xong!") | |