JAA-ATS-Tool / tests /test_structured_placement.py
saitejatirunagari's picture
fix: evidence-gated ATS pipeline — stop scraped-page contamination & fabrication
c425de3
Raw History Blame
6.15 kB
"""Phase 9 (R22): structured keyword placement into the hardcoded resume.
RETIRED — this validated the mechanical keyword-cycling injector (25-30 terms
crammed per bullet). That behaviour was removed from every V1 résumé path in the
evidence-gating correction: V1 now preserves the résumé and inserts NOTHING
without résumé evidence (see tests/test_ats_safety.py). The injector functions
still exist only as internal V2 fallbacks, so these count-based assertions no
longer describe supported behaviour. Skipped rather than deleted to preserve
history and the rationale.
"""
import os
import re
import pytest
pytestmark = pytest.mark.skip(
reason="mechanical keyword-cycling injection retired; V1 is now evidence-gated "
"(see test_ats_safety.py). No keyword is inserted without résumé evidence."
)
from src.latex_resume import inject_keywords, place_keywords_structured
def _resume() -> str:
try:
from src.default_resume import DEFAULT_RESUME_LATEX
if DEFAULT_RESUME_LATEX.strip():
return DEFAULT_RESUME_LATEX
except Exception:
pass
p = os.path.join(
".planning", "phases", "09-hardcoded-resume-keyword-placement",
"resume-source.tex",
)
if os.path.exists(p):
with open(p, encoding="utf-8") as f:
txt = f.read()
if txt.strip():
return txt
pytest.skip("resume source not available")
KWS = [f"kw{i:03d}" for i in range(120)]
def _items_in(block: str):
"""Return the list of tagged-item term-strings found in `block`."""
return re.findall(r"\\resumeItem\{([^\n]*?)\}\s*% ats-item", block)
def test_summary_slot_15_to_20():
out, _ = inject_keywords(_resume(), KWS)
m = re.search(r"Additional areas: (.+?)\.\s*% ===== end", out, re.S)
assert m, "summary injection block missing"
n = len([t for t in m.group(1).split(",") if t.strip()])
assert 15 <= n <= 20, f"summary has {n} terms (want 15-20)"
def test_byjus_psm_25_to_30():
out, _ = inject_keywords(_resume(), KWS)
# The tagged item inside the Asst PSM block.
seg = out[out.find("Asst. Product Success Manager"):]
seg = seg[: seg.find("\\resumeItemListEnd") + 20]
items = _items_in(seg)
assert items, "no tagged item in PSM block"
n = len([t for t in items[-1].split(",") if t.strip()])
assert 25 <= n <= 30, f"PSM has {n} terms (want 25-30)"
def test_byjus_ps_25_to_30():
out, _ = inject_keywords(_resume(), KWS)
seg = out[out.find("Product Specialist -- User Experience"):]
seg = seg[: seg.find("\\resumeItemListEnd") + 20]
items = _items_in(seg)
assert items, "no tagged item in Product Specialist block"
n = len([t for t in items[-1].split(",") if t.strip()])
assert 25 <= n <= 30, f"PS has {n} terms (want 25-30)"
def test_ml_edutech_8_to_12():
out, _ = inject_keywords(_resume(), KWS)
seg = out[out.find("ML Edutech"):]
seg = seg[: seg.find("\\resumeItemListEnd") + 20]
items = _items_in(seg)
assert items, "no tagged item in ML Edutech block"
n = len([t for t in items[-1].split(",") if t.strip()])
assert 8 <= n <= 12, f"ML Edutech has {n} terms (want 8-12)"
def test_skills_other_row_present():
out, _ = inject_keywords(_resume(), KWS)
m = re.search(r"\\textbf\{Other\}\{: (.+?)\}\s*% ats-skills-other", out)
assert m, "Skills Other row missing"
n = len([t for t in m.group(1).split(",") if t.strip()])
assert n >= 15, f"Skills Other has {n} terms (want >=15 once filled)"
def test_anchor_before_list_end():
out, _ = inject_keywords(_resume(), KWS)
assert re.search(r"% ats-item[^\n]*\n\s*\\resumeItemListEnd", out), \
"tagged item not anchored immediately before a \\resumeItemListEnd"
def test_append_only_original_bullets_intact():
src = _resume()
out, _ = inject_keywords(src, KWS)
for original in [
"Increased user retention by 8",
"Managed",
"Launched an EdTech app portfolio",
"Led end-to-end revamp of the",
]:
assert original in out, f"original bullet text lost: {original!r}"
def test_idempotent():
src = _resume()
o1, _ = inject_keywords(src, KWS)
o2, _ = inject_keywords(o1, KWS)
assert o1 == o2, "placement is not idempotent (injected lines stacked)"
def test_dedupe_skips_present_terms():
# A term already in the resume must not be re-added.
out, placed = place_keywords_structured(_resume(), ["Roadmap Planning", "kwX1", "kwX2"])
assert "Roadmap Planning" not in placed
assert "kwX1" in placed
def test_tectonic_asset_matches_sanitizer():
"""assets/default_resume.tectonic.tex (compiled in the Docker warmup) must equal
_sanitize_for_tectonic(default) so the warmed packages match the runtime compile."""
p = os.path.join("assets", "default_resume.tectonic.tex")
if not os.path.exists(p):
pytest.skip("assets/default_resume.tectonic.tex not present")
from src.default_resume import DEFAULT_RESUME_LATEX
from src.latex_resume import _sanitize_for_tectonic
with open(p, encoding="utf-8") as f:
asset = f.read()
assert asset.strip() == _sanitize_for_tectonic(DEFAULT_RESUME_LATEX).strip(), \
"assets/default_resume.tectonic.tex drifted from _sanitize_for_tectonic(default)"
# And the sanitized asset must NOT contain the crash sources.
assert "fontawesome5" not in asset and "{FiraMono}" not in asset
assert "\\faPhone" not in asset and "\\faRupeeSign" not in asset
def test_assets_tex_matches_python_source():
"""assets/default_resume.tex (used for the Docker Tectonic warmup) must stay
byte-identical to the Python default so the warmed packages match what users
actually compile."""
p = os.path.join("assets", "default_resume.tex")
if not os.path.exists(p):
pytest.skip("assets/default_resume.tex not present")
from src.default_resume import DEFAULT_RESUME_LATEX
with open(p, encoding="utf-8") as f:
asset = f.read()
assert asset.strip() == DEFAULT_RESUME_LATEX.strip(), \
"assets/default_resume.tex drifted from src/default_resume.py"