JAA-ATS-Tool / tests /test_v1_generalization.py
saitejatirunagari's picture
feat: V1 NIM fallback + 90% gate + PDF-scored alignment + correction pass
c687f2b
Raw
History Blame
11.4 kB
"""Generalization + missing-keyword + stuffing tests (Steps 12 & 15).
Proves V1 is NOT hard-coded to one role: six different JDs produce different
critical keywords; mandatory vs preferred are distinguished; supported terms are
integrated and unsupported stay gaps; stuffed résumés score lower; and the score
never credits a phrase absent from the final résumé text.
Deterministic (mock LLM). Run: python -m pytest tests/test_v1_generalization.py -q
"""
from __future__ import annotations
import inspect
import os
import re
import sys
import pytest
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from src.ats_safe import generate_alignment_safe
from src.ats_score import compute_coverage, score_alignment
def _crit(p, cat, req, var=None, imp="high"):
return {"exact_phrase": p, "normalized_concept": p.lower(), "category": cat,
"requirement_type": req, "importance": imp, "source_text": p,
"semantic_variants": var or [], "confidence": 0.9,
"requires_resume_evidence": True}
class RoleLLM:
def __init__(self, criteria):
self._c = criteria
def extract_keywords_structured(self, clean_jd):
return [dict(c) for c in self._c]
# Six roles, each with DISTINCT role-specific criteria.
ROLES = {
"product_management": [
_crit("stakeholder management", "soft_skill", "required", ["stakeholder communication"]),
_crit("roadmap prioritization", "responsibility", "required", ["roadmap planning"]),
_crit("A/B testing", "hard_skill", "preferred", ["experiments"]),
_crit("SQL", "tool", "required"),
],
"software_engineering": [
_crit("distributed systems", "hard_skill", "required", ["distributed backends"]),
_crit("Kubernetes", "tool", "required", ["k8s"]),
_crit("microservices", "hard_skill", "required", ["services architecture"]),
_crit("CI/CD", "tool", "preferred"),
],
"data_analysis": [
_crit("data visualization", "hard_skill", "required", ["dashboards"]),
_crit("statistical modeling", "hard_skill", "required", ["statistics"]),
_crit("Python", "tool", "required"),
_crit("ETL pipelines", "hard_skill", "preferred", ["data pipelines"]),
],
"marketing": [
_crit("demand generation", "responsibility", "required", ["lead gen"]),
_crit("SEO", "hard_skill", "required", ["search optimization"]),
_crit("marketing automation", "tool", "preferred"),
_crit("campaign management", "responsibility", "required", ["campaigns"]),
],
"project_management": [
_crit("stakeholder management", "soft_skill", "required", ["stakeholder communication"]),
_crit("risk management", "responsibility", "required", ["risk mitigation"]),
_crit("Agile", "hard_skill", "required", ["scrum"]),
_crit("resource planning", "responsibility", "preferred"),
],
"operations": [
_crit("supply chain", "domain", "required", ["logistics"]),
_crit("process optimization", "responsibility", "required", ["process improvement"]),
_crit("inventory management", "hard_skill", "required", ["stock management"]),
_crit("vendor management", "responsibility", "preferred"),
],
}
def make_jd(criteria):
"""Build a valid, isolatable JD that contains each criterion's exact phrase in
the appropriate section (so the traceability gate — correctly — accepts them)."""
req = [c["exact_phrase"] for c in criteria if c["requirement_type"] == "required"]
pref = [c["exact_phrase"] for c in criteria if c["requirement_type"] != "required"]
lines = ["About the Role",
"In this role you will own key initiatives and deliver measurable "
"outcomes with our team.", "",
"Responsibilities"]
for c in criteria:
lines.append(f"- You will drive {c['exact_phrase']} to support team outcomes.")
lines += ["", "Requirements",
"- You must have 5+ years of relevant professional experience in the field."]
for p in req:
lines.append(f"- You must have strong {p} for this role.")
lines += ["", "Preferred Qualifications"]
for p in (pref or ["cross-functional collaboration"]):
lines.append(f"- Experience with {p} is nice to have.")
return "\n".join(lines)
RESUME = r"""\section{EXPERIENCE}
\resumeItem{Owned stakeholder communication and roadmap planning; ran experiments and built SQL dashboards for 1M+ users.}
\resumeItem{Managed campaigns and risk mitigation across Agile teams; improved process improvement and logistics.}
\section{SKILLS}
\resumeItem{SQL, Python, Agile.}
"""
def _run(criteria, resume=RESUME):
return generate_alignment_safe(resume, make_jd(criteria), company="X", job_title="Role",
llm_client=RoleLLM(criteria), rewrite_fn=None,
compile_pdf=False)
def test_each_role_produces_distinct_critical_keywords():
crit_sets = {}
for role, crits in ROLES.items():
safe = _run(crits)
crit_sets[role] = frozenset(c["concept"] for c in safe["calibration"])
# No two roles share the same critical-keyword set → not a fixed list.
seen = list(crit_sets.values())
assert len(set(seen)) == len(seen), "roles reuse the same critical keyword set"
# PM and SWE must differ substantially.
assert crit_sets["product_management"] != crit_sets["software_engineering"]
def test_mandatory_vs_preferred_distinguished():
safe = _run(ROLES["product_management"])
reqs = {c["exact_phrase"] for c in safe["extraction"]["valid"]
if c["requirement_type"] == "required"}
prefs = {c["exact_phrase"] for c in safe["extraction"]["valid"]
if c["requirement_type"] == "preferred"}
assert "SQL" in reqs and "A/B testing" in prefs
assert reqs and prefs and not (reqs & prefs)
def test_unsupported_terms_stay_gaps():
# SWE criteria vs a PM résumé → distributed systems / kubernetes unsupported.
safe = _run(ROLES["software_engineering"])
gaps = {g["keyword"] for g in safe["evidence"]["gaps"]}
tex = safe["tex"].lower()
assert "kubernetes" in gaps and "kubernetes" not in tex
assert "distributed systems" in gaps
def test_no_fixed_keyword_list_reused_across_roles():
all_terms = []
for crits in ROLES.values():
safe = _run(crits)
all_terms.append(tuple(sorted(c["concept"] for c in safe["calibration"])))
assert len(set(all_terms)) >= 5, "critical keywords barely vary across roles"
def test_score_never_credits_absent_phrase():
# A criterion whose phrase is NOT in the résumé must not count as covered.
crits = [_crit("blockchain", "hard_skill", "required")]
ev = {"covered": [], "partial": [], "gaps": [{"keyword": "blockchain",
"requirement_type": "required"}]}
cov = compute_coverage(crits, ev, "I build web apps with SQL.")
assert cov["critical_family_coverage"] in (0.0, None)
def test_keyword_stuffed_resume_scores_lower():
crits = ROLES["software_engineering"]
natural = r"\resumeItem{Built microservices and distributed systems on Kubernetes with CI/CD.}"
stuffed = (r"\resumeItem{Built microservices and distributed systems on Kubernetes with CI/CD.}"
r"\resumeItem{Skills: Kubernetes, Docker, Go, Rust, Java, C++, Scala, Kafka, Redis, gRPC, Terraform.}")
s_nat = generate_alignment_safe(natural, make_jd(crits), llm_client=RoleLLM(crits),
rewrite_fn=None, compile_pdf=False)
s_stf = generate_alignment_safe(stuffed, make_jd(crits), llm_client=RoleLLM(crits),
rewrite_fn=None, compile_pdf=False)
assert s_stf["internal_alignment_estimate"]["penalties"], "stuffing not penalized"
assert (s_stf["internal_alignment_estimate"]["after"]
<= s_nat["internal_alignment_estimate"]["after"] + 0.01)
def test_missing_supported_mandatory_flagged():
# Résumé supports 'stakeholder management' (via communication) but never the
# exact phrase, and no rewriter runs → it must show as missing in coverage.
crits = [_crit("stakeholder management", "soft_skill", "required", ["stakeholder communication"])]
safe = generate_alignment_safe(
r"\resumeItem{Owned stakeholder communication for the team.}",
make_jd(crits), llm_client=RoleLLM(crits), rewrite_fn=None, compile_pdf=False)
cov = safe["internal_alignment_estimate"]["coverage"]
# supported but exact phrase absent → critical exact-phrase coverage < 1
assert (cov["critical_exact_phrase_coverage"] or 0) < 1.0
assert not safe["internal_alignment_estimate"]["gate_90_passed"]
def test_reaches_90_when_genuinely_supported():
"""When the candidate GENUINELY supports every mandatory/critical criterion,
truthful rewriting lifts the alignment to >=90 and the 90% gate passes — with
zero unsupported insertions. (No fabrication; the fixture really supports it.)"""
resume = (r"\section{EXPERIENCE}"
r"\resumeItem{Owned stakeholder communication and product roadmap planning "
r"for a B2B SaaS platform serving 1M+ users, lifting activation 18\%.}"
r"\resumeItem{Ran experiments with cross-functional teams and built SQL "
r"dashboards to guide decisions.}"
r"\section{SKILLS}\resumeItem{SQL, Product Analytics.}")
crits = [
_crit("stakeholder management", "soft_skill", "required", ["stakeholder communication"], "critical"),
_crit("roadmap prioritization", "responsibility", "required", ["product roadmap planning"], "critical"),
_crit("product experimentation", "hard_skill", "required", ["experiments"], "critical"),
_crit("cross-functional collaboration", "responsibility", "required", ["cross-functional teams"], "critical"),
_crit("SQL", "tool", "required"),
]
rwmap = {
"stakeholder management": ("stakeholder communication", "stakeholder management"),
"roadmap prioritization": ("product roadmap planning", "roadmap prioritization"),
"product experimentation": ("Ran experiments", "Ran product experimentation"),
"cross-functional collaboration": ("with cross-functional teams",
"through cross-functional collaboration with teams"),
}
def rw(o, t, c, cat):
m = rwmap.get(t.lower())
return re.sub(re.escape(m[0]), m[1], o, count=1, flags=re.IGNORECASE) if m else o
safe = generate_alignment_safe(resume, make_jd(crits), llm_client=RoleLLM(crits),
rewrite_fn=rw, compile_pdf=False)
e = safe["internal_alignment_estimate"]
assert e["before"] < 90 <= e["after"], (e["before"], e["after"])
assert e["gate_90_passed"] is True
assert e["unsupported_insertions"] == 0
assert e["coverage"]["mandatory_coverage"] == 1.0
assert (e["coverage"]["critical_exact_phrase_coverage"] or 0) >= 0.85
def test_routes_share_pipeline():
import api_server
src = inspect.getsource(api_server)
assert src.count("generate_alignment_safe") >= 2
assert src.count("build_llm") >= 2 # SSE + blocking both use NIM fallback
if __name__ == "__main__":
sys.exit(pytest.main([__file__, "-x", "-q"]))