Spaces:
Sleeping
Sleeping
Download app.py from harishforaiandml/qa-app-pdf: direct link, hf CLI and curl.
- Browser
- Download file 2.57 kB
-
https://huggingface.co/spaces/harishforaiandml/qa-app-pdf/resolve/main/app.py
- Command line
-
hf download hf://spaces/harishforaiandml/qa-app-pdf/app.py
-
curl -L -o app.py https://huggingface.co/spaces/harishforaiandml/qa-app-pdf/resolve/main/app.py
2.57 kB
| import gradio as gr | |
| from transformers import AutoTokenizer, AutoModelForQuestionAnswering | |
| import torch | |
| from pypdf import PdfReader | |
| MODEL_NAME = "harishforaiandml/my-pretrained-qa-model" | |
| tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME) | |
| model = AutoModelForQuestionAnswering.from_pretrained(MODEL_NAME) | |
| model.eval() | |
| # ---------------------------- | |
| # PDF TEXT EXTRACTION | |
| # ---------------------------- | |
| def extract_text_from_pdf(pdf_file): | |
| reader = PdfReader(pdf_file) | |
| text = "" | |
| for page in reader.pages: | |
| text += page.extract_text() + "\n" | |
| return text | |
| # ---------------------------- | |
| # SIMPLE CHUNKING (mini RAG) | |
| # ---------------------------- | |
| def chunk_text(text, chunk_size=500): | |
| words = text.split() | |
| chunks = [] | |
| for i in range(0, len(words), chunk_size): | |
| chunk = " ".join(words[i:i + chunk_size]) | |
| chunks.append(chunk) | |
| return chunks | |
| # ---------------------------- | |
| # FIND BEST CHUNK (retrieval step) | |
| # ---------------------------- | |
| def get_best_chunk(chunks, question): | |
| question_words = set(question.lower().split()) | |
| best_chunk = "" | |
| best_score = 0 | |
| for chunk in chunks: | |
| chunk_words = set(chunk.lower().split()) | |
| score = len(question_words.intersection(chunk_words)) | |
| if score > best_score: | |
| best_score = score | |
| best_chunk = chunk | |
| return best_chunk | |
| # ---------------------------- | |
| # QA FUNCTION | |
| # ---------------------------- | |
| def ask_pdf(pdf_file, question): | |
| if pdf_file is None: | |
| return "Please upload a PDF" | |
| text = extract_text_from_pdf(pdf_file) | |
| chunks = chunk_text(text) | |
| context = get_best_chunk(chunks, question) | |
| inputs = tokenizer( | |
| question, | |
| context, | |
| return_tensors="pt", | |
| truncation=True | |
| ) | |
| with torch.no_grad(): | |
| outputs = model(**inputs) | |
| start = torch.argmax(outputs.start_logits) | |
| end = torch.argmax(outputs.end_logits) + 1 | |
| answer_tokens = inputs["input_ids"][0][start:end] | |
| answer = tokenizer.decode(answer_tokens, skip_special_tokens=True) | |
| if answer.strip() == "": | |
| return "Not found in document" | |
| return answer | |
| # ---------------------------- | |
| # GRADIO UI | |
| # ---------------------------- | |
| demo = gr.Interface( | |
| fn=ask_pdf, | |
| inputs=[ | |
| gr.File(label="Upload PDF"), | |
| gr.Textbox(label="Question") | |
| ], | |
| outputs="text", | |
| title="📄 PDF QA (Mini RAG System)", | |
| description="Upload a PDF and ask questions. System finds best chunk and extracts answer using DistilBERT." | |
| ) | |
| demo.launch() |