File size: 2,568 Bytes
d0d3eac
c420a6b
 
 
d0d3eac
ab96168
 
 
 
 
 
 
c420a6b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d0d3eac
ab96168
 
 
 
 
d0d3eac
 
ab96168
 
 
c420a6b
 
ab96168
c420a6b
ab96168
 
 
c420a6b
d0d3eac
ab96168
 
d0d3eac
c420a6b
 
 
d0d3eac
c420a6b
ab96168
d0d3eac
c420a6b
d0d3eac
 
ab96168
3a96c2e
ab96168
c420a6b
 
d0d3eac
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
import gradio as gr
from transformers import AutoTokenizer, AutoModelForQuestionAnswering
import torch
from pypdf import PdfReader

MODEL_NAME = "harishforaiandml/my-pretrained-qa-model"

tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME)
model = AutoModelForQuestionAnswering.from_pretrained(MODEL_NAME)
model.eval()


# ----------------------------
# PDF TEXT EXTRACTION
# ----------------------------
def extract_text_from_pdf(pdf_file):
    reader = PdfReader(pdf_file)
    text = ""

    for page in reader.pages:
        text += page.extract_text() + "\n"

    return text


# ----------------------------
# SIMPLE CHUNKING (mini RAG)
# ----------------------------
def chunk_text(text, chunk_size=500):
    words = text.split()
    chunks = []

    for i in range(0, len(words), chunk_size):
        chunk = " ".join(words[i:i + chunk_size])
        chunks.append(chunk)

    return chunks


# ----------------------------
# FIND BEST CHUNK (retrieval step)
# ----------------------------
def get_best_chunk(chunks, question):
    question_words = set(question.lower().split())

    best_chunk = ""
    best_score = 0

    for chunk in chunks:
        chunk_words = set(chunk.lower().split())
        score = len(question_words.intersection(chunk_words))

        if score > best_score:
            best_score = score
            best_chunk = chunk

    return best_chunk


# ----------------------------
# QA FUNCTION
# ----------------------------
def ask_pdf(pdf_file, question):

    if pdf_file is None:
        return "Please upload a PDF"

    text = extract_text_from_pdf(pdf_file)
    chunks = chunk_text(text)

    context = get_best_chunk(chunks, question)

    inputs = tokenizer(
        question,
        context,
        return_tensors="pt",
        truncation=True
    )

    with torch.no_grad():
        outputs = model(**inputs)

    start = torch.argmax(outputs.start_logits)
    end = torch.argmax(outputs.end_logits) + 1

    answer_tokens = inputs["input_ids"][0][start:end]
    answer = tokenizer.decode(answer_tokens, skip_special_tokens=True)

    if answer.strip() == "":
        return "Not found in document"

    return answer


# ----------------------------
# GRADIO UI
# ----------------------------
demo = gr.Interface(
    fn=ask_pdf,

    inputs=[
        gr.File(label="Upload PDF"),
        gr.Textbox(label="Question")
    ],

    outputs="text",

    title="📄 PDF QA (Mini RAG System)",
    description="Upload a PDF and ask questions. System finds best chunk and extracts answer using DistilBERT."
)

demo.launch()