JAA-ATS-Tool / src /resume_parser.py
saitejatirunagari's picture
Initial commit: JAA ATS Tool — PM job search + AI assessment + ATS resume gen
7ff6662
Raw
History Blame
1.58 kB
import os
import re
import pdfplumber
class ResumeParser:
def __init__(self, pdf_path: str):
self.pdf_path = pdf_path
def parse(self) -> str:
if not os.path.exists(self.pdf_path):
raise FileNotFoundError(
f"Resume not found at '{self.pdf_path}'.\n"
"Please place your PDF resume at: data/resume/resume.pdf"
)
text = ""
with pdfplumber.open(self.pdf_path) as pdf:
for page in pdf.pages:
page_text = page.extract_text()
if page_text:
text += page_text + "\n"
if not text.strip():
raise ValueError("Could not extract text from resume PDF. Make sure it's not scanned/image-only.")
return self._clean_text(text)
def _clean_text(self, text: str) -> str:
# Remove excessive whitespace
text = re.sub(r"\n{3,}", "\n\n", text)
text = re.sub(r"[ \t]{2,}", " ", text)
return text.strip()
def get_contact_info(self, text: str) -> dict:
info = {}
email_match = re.search(r"[\w.+-]+@[\w-]+\.[a-zA-Z]{2,}", text)
if email_match:
info["email"] = email_match.group()
phone_match = re.search(r"[\+]?[0-9]{1,4}[\s.-]?[0-9]{4,5}[\s.-]?[0-9]{4,5}", text)
if phone_match:
info["phone"] = phone_match.group()
linkedin_match = re.search(r"linkedin\.com/in/[\w-]+", text, re.I)
if linkedin_match:
info["linkedin"] = "https://" + linkedin_match.group()
return info