File size: 3,508 Bytes
c425de3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
"""PDF parse-validation tests (success + failure), no LaTeX engine required.

Builds controlled PDFs with reportlab so both the pass path (clean, ordered
sections) and the fail path (missing section / leaked injection marker /
unparseable) are exercised deterministically.

Run:  python -m pytest tests/test_pdf_validate.py -x -q
"""
from __future__ import annotations

import os
import sys
import tempfile

import pytest

sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))

from src.pdf_validate import validate_pdf, _extract_pdf_text

reportlab = pytest.importorskip("reportlab")
from reportlab.pdfgen import canvas  # noqa: E402
from reportlab.lib.pagesizes import letter  # noqa: E402


def _make_pdf(lines, path):
    c = canvas.Canvas(path, pagesize=letter)
    y = 750
    for ln in lines:
        c.drawString(72, y, ln)
        y -= 18
        if y < 72:
            c.showPage()
            y = 750
    c.save()


def test_pdf_valid_when_sections_present_and_ordered():
    with tempfile.TemporaryDirectory() as d:
        p = os.path.join(d, "good.pdf")
        _make_pdf([
            "SAITEJA TIRUNAGARI", "Product Manager | email@example.com",
            "EXPERIENCE",
            "Owned roadmap and stakeholder management. Ran A/B testing.",
            "EDUCATION", "IIM Rohtak - Product Management",
            "SKILLS", "SQL, Product Analytics, Agile",
        ], p)
        r = validate_pdf(p, expected_sections=["experience", "education", "skills"],
                         expected_contact=["email@example.com"])
        assert r["parser_recovered_text"]
        assert not r["sections_missing"], r
        assert r["sections_in_order"] is True
        assert r["contact_present"] is True
        assert not r["forbidden_markers_found"]
        assert r["ok"] is True


def test_pdf_flags_missing_section():
    with tempfile.TemporaryDirectory() as d:
        p = os.path.join(d, "nomissing.pdf")
        _make_pdf(["EXPERIENCE", "Did things.", "SKILLS", "SQL"], p)
        r = validate_pdf(p, expected_sections=["experience", "education", "skills"])
        assert "education" in r["sections_missing"]
        assert r["ok"] is False


def test_pdf_flags_injected_marker():
    with tempfile.TemporaryDirectory() as d:
        p = os.path.join(d, "marker.pdf")
        _make_pdf([
            "EXPERIENCE", "Real bullet.",
            "Core focus areas include sql, agile, python.",   # injected-style block
            "EDUCATION", "Deg", "SKILLS", "SQL",
        ], p)
        r = validate_pdf(p, expected_sections=["experience", "education", "skills"])
        assert r["forbidden_markers_found"], "should flag injected summary block"
        assert r["ok"] is False


def test_pdf_unparseable_returns_not_ok():
    with tempfile.TemporaryDirectory() as d:
        p = os.path.join(d, "empty.pdf")
        # An image-only / empty PDF: create a blank page with no text.
        c = canvas.Canvas(p, pagesize=letter)
        c.showPage()
        c.save()
        r = validate_pdf(p, expected_sections=["experience"])
        assert r["ok"] is False
        assert r["parser_recovered_text"] is False


def test_existing_sample_pdf_parses():
    sample = os.path.join(os.path.dirname(os.path.abspath(__file__)), "v2_output.pdf")
    if not os.path.exists(sample):
        pytest.skip("no sample PDF")
    text = _extract_pdf_text(sample)
    assert text and len(text) > 200


if __name__ == "__main__":
    sys.exit(pytest.main([__file__, "-x", "-q"]))