JAA-ATS-Tool / tests /test_pdf_validate.py
saitejatirunagari's picture
fix: evidence-gated ATS pipeline — stop scraped-page contamination & fabrication
c425de3
Raw
History Blame
3.51 kB
"""PDF parse-validation tests (success + failure), no LaTeX engine required.
Builds controlled PDFs with reportlab so both the pass path (clean, ordered
sections) and the fail path (missing section / leaked injection marker /
unparseable) are exercised deterministically.
Run: python -m pytest tests/test_pdf_validate.py -x -q
"""
from __future__ import annotations
import os
import sys
import tempfile
import pytest
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from src.pdf_validate import validate_pdf, _extract_pdf_text
reportlab = pytest.importorskip("reportlab")
from reportlab.pdfgen import canvas # noqa: E402
from reportlab.lib.pagesizes import letter # noqa: E402
def _make_pdf(lines, path):
c = canvas.Canvas(path, pagesize=letter)
y = 750
for ln in lines:
c.drawString(72, y, ln)
y -= 18
if y < 72:
c.showPage()
y = 750
c.save()
def test_pdf_valid_when_sections_present_and_ordered():
with tempfile.TemporaryDirectory() as d:
p = os.path.join(d, "good.pdf")
_make_pdf([
"SAITEJA TIRUNAGARI", "Product Manager | email@example.com",
"EXPERIENCE",
"Owned roadmap and stakeholder management. Ran A/B testing.",
"EDUCATION", "IIM Rohtak - Product Management",
"SKILLS", "SQL, Product Analytics, Agile",
], p)
r = validate_pdf(p, expected_sections=["experience", "education", "skills"],
expected_contact=["email@example.com"])
assert r["parser_recovered_text"]
assert not r["sections_missing"], r
assert r["sections_in_order"] is True
assert r["contact_present"] is True
assert not r["forbidden_markers_found"]
assert r["ok"] is True
def test_pdf_flags_missing_section():
with tempfile.TemporaryDirectory() as d:
p = os.path.join(d, "nomissing.pdf")
_make_pdf(["EXPERIENCE", "Did things.", "SKILLS", "SQL"], p)
r = validate_pdf(p, expected_sections=["experience", "education", "skills"])
assert "education" in r["sections_missing"]
assert r["ok"] is False
def test_pdf_flags_injected_marker():
with tempfile.TemporaryDirectory() as d:
p = os.path.join(d, "marker.pdf")
_make_pdf([
"EXPERIENCE", "Real bullet.",
"Core focus areas include sql, agile, python.", # injected-style block
"EDUCATION", "Deg", "SKILLS", "SQL",
], p)
r = validate_pdf(p, expected_sections=["experience", "education", "skills"])
assert r["forbidden_markers_found"], "should flag injected summary block"
assert r["ok"] is False
def test_pdf_unparseable_returns_not_ok():
with tempfile.TemporaryDirectory() as d:
p = os.path.join(d, "empty.pdf")
# An image-only / empty PDF: create a blank page with no text.
c = canvas.Canvas(p, pagesize=letter)
c.showPage()
c.save()
r = validate_pdf(p, expected_sections=["experience"])
assert r["ok"] is False
assert r["parser_recovered_text"] is False
def test_existing_sample_pdf_parses():
sample = os.path.join(os.path.dirname(os.path.abspath(__file__)), "v2_output.pdf")
if not os.path.exists(sample):
pytest.skip("no sample PDF")
text = _extract_pdf_text(sample)
assert text and len(text) > 200
if __name__ == "__main__":
sys.exit(pytest.main([__file__, "-x", "-q"]))