Spaces:
Sleeping
Sleeping
| """PDF parse-validation tests (success + failure), no LaTeX engine required. | |
| Builds controlled PDFs with reportlab so both the pass path (clean, ordered | |
| sections) and the fail path (missing section / leaked injection marker / | |
| unparseable) are exercised deterministically. | |
| Run: python -m pytest tests/test_pdf_validate.py -x -q | |
| """ | |
| from __future__ import annotations | |
| import os | |
| import sys | |
| import tempfile | |
| import pytest | |
| sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) | |
| from src.pdf_validate import validate_pdf, _extract_pdf_text | |
| reportlab = pytest.importorskip("reportlab") | |
| from reportlab.pdfgen import canvas # noqa: E402 | |
| from reportlab.lib.pagesizes import letter # noqa: E402 | |
| def _make_pdf(lines, path): | |
| c = canvas.Canvas(path, pagesize=letter) | |
| y = 750 | |
| for ln in lines: | |
| c.drawString(72, y, ln) | |
| y -= 18 | |
| if y < 72: | |
| c.showPage() | |
| y = 750 | |
| c.save() | |
| def test_pdf_valid_when_sections_present_and_ordered(): | |
| with tempfile.TemporaryDirectory() as d: | |
| p = os.path.join(d, "good.pdf") | |
| _make_pdf([ | |
| "SAITEJA TIRUNAGARI", "Product Manager | email@example.com", | |
| "EXPERIENCE", | |
| "Owned roadmap and stakeholder management. Ran A/B testing.", | |
| "EDUCATION", "IIM Rohtak - Product Management", | |
| "SKILLS", "SQL, Product Analytics, Agile", | |
| ], p) | |
| r = validate_pdf(p, expected_sections=["experience", "education", "skills"], | |
| expected_contact=["email@example.com"]) | |
| assert r["parser_recovered_text"] | |
| assert not r["sections_missing"], r | |
| assert r["sections_in_order"] is True | |
| assert r["contact_present"] is True | |
| assert not r["forbidden_markers_found"] | |
| assert r["ok"] is True | |
| def test_pdf_flags_missing_section(): | |
| with tempfile.TemporaryDirectory() as d: | |
| p = os.path.join(d, "nomissing.pdf") | |
| _make_pdf(["EXPERIENCE", "Did things.", "SKILLS", "SQL"], p) | |
| r = validate_pdf(p, expected_sections=["experience", "education", "skills"]) | |
| assert "education" in r["sections_missing"] | |
| assert r["ok"] is False | |
| def test_pdf_flags_injected_marker(): | |
| with tempfile.TemporaryDirectory() as d: | |
| p = os.path.join(d, "marker.pdf") | |
| _make_pdf([ | |
| "EXPERIENCE", "Real bullet.", | |
| "Core focus areas include sql, agile, python.", # injected-style block | |
| "EDUCATION", "Deg", "SKILLS", "SQL", | |
| ], p) | |
| r = validate_pdf(p, expected_sections=["experience", "education", "skills"]) | |
| assert r["forbidden_markers_found"], "should flag injected summary block" | |
| assert r["ok"] is False | |
| def test_pdf_unparseable_returns_not_ok(): | |
| with tempfile.TemporaryDirectory() as d: | |
| p = os.path.join(d, "empty.pdf") | |
| # An image-only / empty PDF: create a blank page with no text. | |
| c = canvas.Canvas(p, pagesize=letter) | |
| c.showPage() | |
| c.save() | |
| r = validate_pdf(p, expected_sections=["experience"]) | |
| assert r["ok"] is False | |
| assert r["parser_recovered_text"] is False | |
| def test_existing_sample_pdf_parses(): | |
| sample = os.path.join(os.path.dirname(os.path.abspath(__file__)), "v2_output.pdf") | |
| if not os.path.exists(sample): | |
| pytest.skip("no sample PDF") | |
| text = _extract_pdf_text(sample) | |
| assert text and len(text) > 200 | |
| if __name__ == "__main__": | |
| sys.exit(pytest.main([__file__, "-x", "-q"])) | |