Spaces:
Sleeping
Sleeping
File size: 2,568 Bytes
d0d3eac c420a6b d0d3eac ab96168 c420a6b d0d3eac ab96168 d0d3eac ab96168 c420a6b ab96168 c420a6b ab96168 c420a6b d0d3eac ab96168 d0d3eac c420a6b d0d3eac c420a6b ab96168 d0d3eac c420a6b d0d3eac ab96168 3a96c2e ab96168 c420a6b d0d3eac | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 | import gradio as gr
from transformers import AutoTokenizer, AutoModelForQuestionAnswering
import torch
from pypdf import PdfReader
MODEL_NAME = "harishforaiandml/my-pretrained-qa-model"
tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME)
model = AutoModelForQuestionAnswering.from_pretrained(MODEL_NAME)
model.eval()
# ----------------------------
# PDF TEXT EXTRACTION
# ----------------------------
def extract_text_from_pdf(pdf_file):
reader = PdfReader(pdf_file)
text = ""
for page in reader.pages:
text += page.extract_text() + "\n"
return text
# ----------------------------
# SIMPLE CHUNKING (mini RAG)
# ----------------------------
def chunk_text(text, chunk_size=500):
words = text.split()
chunks = []
for i in range(0, len(words), chunk_size):
chunk = " ".join(words[i:i + chunk_size])
chunks.append(chunk)
return chunks
# ----------------------------
# FIND BEST CHUNK (retrieval step)
# ----------------------------
def get_best_chunk(chunks, question):
question_words = set(question.lower().split())
best_chunk = ""
best_score = 0
for chunk in chunks:
chunk_words = set(chunk.lower().split())
score = len(question_words.intersection(chunk_words))
if score > best_score:
best_score = score
best_chunk = chunk
return best_chunk
# ----------------------------
# QA FUNCTION
# ----------------------------
def ask_pdf(pdf_file, question):
if pdf_file is None:
return "Please upload a PDF"
text = extract_text_from_pdf(pdf_file)
chunks = chunk_text(text)
context = get_best_chunk(chunks, question)
inputs = tokenizer(
question,
context,
return_tensors="pt",
truncation=True
)
with torch.no_grad():
outputs = model(**inputs)
start = torch.argmax(outputs.start_logits)
end = torch.argmax(outputs.end_logits) + 1
answer_tokens = inputs["input_ids"][0][start:end]
answer = tokenizer.decode(answer_tokens, skip_special_tokens=True)
if answer.strip() == "":
return "Not found in document"
return answer
# ----------------------------
# GRADIO UI
# ----------------------------
demo = gr.Interface(
fn=ask_pdf,
inputs=[
gr.File(label="Upload PDF"),
gr.Textbox(label="Question")
],
outputs="text",
title="📄 PDF QA (Mini RAG System)",
description="Upload a PDF and ask questions. System finds best chunk and extracts answer using DistilBERT."
)
demo.launch() |