rag-vietnamese / compare_retrieval.py
thaidinhz1's picture
feat: add ColPali visual retrieval + contextual retrieval + PDF eval
af92b4d
Raw History Blame
2.12 kB
"""
So sánh ColPali (visual) vs Text RAG (pypdf + Gemini embed) trên cùng câu hỏi.
Usage: python compare_retrieval.py
"""
from dotenv import load_dotenv
load_dotenv()
TEST_QUESTIONS = [
"Doanh thu thuần của công ty trong quý 1 năm 2026 là bao nhiêu?",
"Lợi nhuận sau thuế quý 1 năm 2026?",
"Chi phí tài chính của FPT trong nửa đầu năm 2016?",
"Tổng tài sản của TIG tại cuối quý 1 năm 2026?",
"FPT hoạt động trong những lĩnh vực kinh doanh nào?",
]
def run_text_retrieval(question: str, top_k: int = 3) -> list[dict]:
from src.rag import retrieve
from src.embedder import embed_query
contexts, sources = retrieve(question, top_k=top_k)
return sources
def run_colpali_retrieval(question: str, top_k: int = 3) -> list[dict]:
from src.colpali_retriever import query
return query(question, top_k=top_k)
def main():
print("=" * 60)
print("RETRIEVAL COMPARISON: ColPali vs Text RAG")
print("=" * 60)
results = []
for q in TEST_QUESTIONS:
print(f"\nQ: {q}")
print("-" * 50)
print("[Text RAG]")
try:
text_hits = run_text_retrieval(q, top_k=3)
for h in text_hits:
print(f" {h['source']} p.{h['page']} | rerank={h['rerank_score']:.3f}")
except Exception as e:
print(f" ERROR: {e}")
text_hits = []
print("[ColPali]")
try:
colpali_hits = run_colpali_retrieval(q, top_k=3)
for h in colpali_hits:
print(f" {h['source']} p.{h['page']} | score={h['score']:.3f}")
except Exception as e:
print(f" ERROR: {e}")
colpali_hits = []
results.append({
"question": q,
"text_rag": text_hits,
"colpali": colpali_hits,
})
import json
with open("comparison_results.json", "w", encoding="utf-8") as f:
json.dump(results, f, ensure_ascii=False, indent=2)
print("\n\nResults saved to comparison_results.json")
if __name__ == "__main__":
main()