Spaces:
Sleeping
Sleeping
Download compare_retrieval.py from thaidinhz1/rag-vietnamese: direct link, hf CLI and curl.
- Browser
- Download file 2.12 kB
-
https://huggingface.co/spaces/thaidinhz1/rag-vietnamese/resolve/9e7ded3dd31be325a258cd2b19fbc3441d5fe4f5/compare_retrieval.py
- Command line
-
hf download hf://spaces/thaidinhz1/rag-vietnamese@9e7ded3dd31be325a258cd2b19fbc3441d5fe4f5/compare_retrieval.py
-
curl -L -o compare_retrieval.py https://huggingface.co/spaces/thaidinhz1/rag-vietnamese/resolve/9e7ded3dd31be325a258cd2b19fbc3441d5fe4f5/compare_retrieval.py
2.12 kB
| """ | |
| So sánh ColPali (visual) vs Text RAG (pypdf + Gemini embed) trên cùng câu hỏi. | |
| Usage: python compare_retrieval.py | |
| """ | |
| from dotenv import load_dotenv | |
| load_dotenv() | |
| TEST_QUESTIONS = [ | |
| "Doanh thu thuần của công ty trong quý 1 năm 2026 là bao nhiêu?", | |
| "Lợi nhuận sau thuế quý 1 năm 2026?", | |
| "Chi phí tài chính của FPT trong nửa đầu năm 2016?", | |
| "Tổng tài sản của TIG tại cuối quý 1 năm 2026?", | |
| "FPT hoạt động trong những lĩnh vực kinh doanh nào?", | |
| ] | |
| def run_text_retrieval(question: str, top_k: int = 3) -> list[dict]: | |
| from src.rag import retrieve | |
| from src.embedder import embed_query | |
| contexts, sources = retrieve(question, top_k=top_k) | |
| return sources | |
| def run_colpali_retrieval(question: str, top_k: int = 3) -> list[dict]: | |
| from src.colpali_retriever import query | |
| return query(question, top_k=top_k) | |
| def main(): | |
| print("=" * 60) | |
| print("RETRIEVAL COMPARISON: ColPali vs Text RAG") | |
| print("=" * 60) | |
| results = [] | |
| for q in TEST_QUESTIONS: | |
| print(f"\nQ: {q}") | |
| print("-" * 50) | |
| print("[Text RAG]") | |
| try: | |
| text_hits = run_text_retrieval(q, top_k=3) | |
| for h in text_hits: | |
| print(f" {h['source']} p.{h['page']} | rerank={h['rerank_score']:.3f}") | |
| except Exception as e: | |
| print(f" ERROR: {e}") | |
| text_hits = [] | |
| print("[ColPali]") | |
| try: | |
| colpali_hits = run_colpali_retrieval(q, top_k=3) | |
| for h in colpali_hits: | |
| print(f" {h['source']} p.{h['page']} | score={h['score']:.3f}") | |
| except Exception as e: | |
| print(f" ERROR: {e}") | |
| colpali_hits = [] | |
| results.append({ | |
| "question": q, | |
| "text_rag": text_hits, | |
| "colpali": colpali_hits, | |
| }) | |
| import json | |
| with open("comparison_results.json", "w", encoding="utf-8") as f: | |
| json.dump(results, f, ensure_ascii=False, indent=2) | |
| print("\n\nResults saved to comparison_results.json") | |
| if __name__ == "__main__": | |
| main() | |