Spaces:
Sleeping
Sleeping
Download app.py from dhanush1601/Pdf_chat_Assistant: direct link, hf CLI and curl.
- Browser
- Download file 2.88 kB
-
https://huggingface.co/spaces/dhanush1601/Pdf_chat_Assistant/resolve/main/app.py
- Command line
-
hf download hf://spaces/dhanush1601/Pdf_chat_Assistant/app.py
-
curl -L -o app.py https://huggingface.co/spaces/dhanush1601/Pdf_chat_Assistant/resolve/main/app.py
2.88 kB
| # rag_upload_app_local.py | |
| import os | |
| from PyPDF2 import PdfReader | |
| import gradio as gr | |
| # LangChain imports | |
| from langchain.text_splitter import RecursiveCharacterTextSplitter | |
| from langchain.embeddings import HuggingFaceEmbeddings | |
| from langchain.vectorstores import FAISS | |
| from langchain.llms import HuggingFacePipeline | |
| from langchain.chains import RetrievalQA | |
| # Transformers imports | |
| from transformers import AutoTokenizer, AutoModelForSeq2SeqLM, pipeline | |
| # --- CONFIG --- | |
| EMBEDDING_MODEL = "sentence-transformers/all-MiniLM-L6-v2" | |
| LOCAL_MODEL = "google/flan-t5-base" # lightweight model | |
| # Load local HuggingFace model | |
| tokenizer = AutoTokenizer.from_pretrained(LOCAL_MODEL) | |
| model = AutoModelForSeq2SeqLM.from_pretrained(LOCAL_MODEL) | |
| pipe = pipeline("text2text-generation", model=model, tokenizer=tokenizer, max_length=512) | |
| llm = HuggingFacePipeline(pipeline=pipe) | |
| def process_document(file): | |
| try: | |
| if file is None: | |
| return None, "โ ๏ธ Please upload a document." | |
| # Extract text from PDF | |
| text = "" | |
| reader = PdfReader(file) | |
| for page in reader.pages: | |
| page_text = page.extract_text() | |
| if page_text: | |
| text += page_text + "\n" | |
| if not text.strip(): | |
| return None, "โ ๏ธ No text could be extracted. Try another PDF." | |
| # Split text into chunks | |
| splitter = RecursiveCharacterTextSplitter(chunk_size=500, chunk_overlap=50) | |
| chunks = splitter.split_text(text) | |
| # Create embeddings + FAISS index | |
| embedder = HuggingFaceEmbeddings(model_name=EMBEDDING_MODEL) | |
| db = FAISS.from_texts(chunks, embedder) | |
| # Create retriever + QA chain | |
| retriever = db.as_retriever(search_kwargs={"k": 4}) | |
| qa = RetrievalQA.from_chain_type(llm=llm, chain_type="stuff", retriever=retriever) | |
| return qa, f"โ Document processed successfully with {len(chunks)} chunks!" | |
| except Exception as e: | |
| return None, f"โ Error: {str(e)}" | |
| def answer_question(qa, question): | |
| if qa is None: | |
| return "Please upload and process a document first." | |
| return qa.run(question) | |
| with gr.Blocks() as demo: | |
| gr.Markdown("## ๐ PDF CHAT ASSISSTANT") | |
| with gr.Row(): | |
| file_input = gr.File(label="Upload PDF Document", type="filepath") | |
| status = gr.Textbox(label="Status", interactive=False, lines=6) # ๐น bigger | |
| process_btn = gr.Button("Process Document") | |
| with gr.Row(): | |
| question = gr.Textbox(label="Ask a Question", lines=3, placeholder="Type your question here...") # ๐น taller | |
| answer = gr.Textbox(label="Answer", lines=8) # ๐น taller answer box | |
| qa_state = gr.State() | |
| process_btn.click(fn=process_document, inputs=file_input, outputs=[qa_state, status]) | |
| question.submit(fn=answer_question, inputs=[qa_state, question], outputs=answer) | |
| demo.launch() | |