Spaces:
Sleeping
Sleeping
| """Generalization + missing-keyword + stuffing tests (Steps 12 & 15). | |
| Proves V1 is NOT hard-coded to one role: six different JDs produce different | |
| critical keywords; mandatory vs preferred are distinguished; supported terms are | |
| integrated and unsupported stay gaps; stuffed résumés score lower; and the score | |
| never credits a phrase absent from the final résumé text. | |
| Deterministic (mock LLM). Run: python -m pytest tests/test_v1_generalization.py -q | |
| """ | |
| from __future__ import annotations | |
| import inspect | |
| import os | |
| import re | |
| import sys | |
| import pytest | |
| sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) | |
| from src.ats_safe import generate_alignment_safe | |
| from src.ats_score import compute_coverage, score_alignment | |
| def _crit(p, cat, req, var=None, imp="high"): | |
| return {"exact_phrase": p, "normalized_concept": p.lower(), "category": cat, | |
| "requirement_type": req, "importance": imp, "source_text": p, | |
| "semantic_variants": var or [], "confidence": 0.9, | |
| "requires_resume_evidence": True} | |
| class RoleLLM: | |
| def __init__(self, criteria): | |
| self._c = criteria | |
| def extract_keywords_structured(self, clean_jd): | |
| return [dict(c) for c in self._c] | |
| # Six roles, each with DISTINCT role-specific criteria. | |
| ROLES = { | |
| "product_management": [ | |
| _crit("stakeholder management", "soft_skill", "required", ["stakeholder communication"]), | |
| _crit("roadmap prioritization", "responsibility", "required", ["roadmap planning"]), | |
| _crit("A/B testing", "hard_skill", "preferred", ["experiments"]), | |
| _crit("SQL", "tool", "required"), | |
| ], | |
| "software_engineering": [ | |
| _crit("distributed systems", "hard_skill", "required", ["distributed backends"]), | |
| _crit("Kubernetes", "tool", "required", ["k8s"]), | |
| _crit("microservices", "hard_skill", "required", ["services architecture"]), | |
| _crit("CI/CD", "tool", "preferred"), | |
| ], | |
| "data_analysis": [ | |
| _crit("data visualization", "hard_skill", "required", ["dashboards"]), | |
| _crit("statistical modeling", "hard_skill", "required", ["statistics"]), | |
| _crit("Python", "tool", "required"), | |
| _crit("ETL pipelines", "hard_skill", "preferred", ["data pipelines"]), | |
| ], | |
| "marketing": [ | |
| _crit("demand generation", "responsibility", "required", ["lead gen"]), | |
| _crit("SEO", "hard_skill", "required", ["search optimization"]), | |
| _crit("marketing automation", "tool", "preferred"), | |
| _crit("campaign management", "responsibility", "required", ["campaigns"]), | |
| ], | |
| "project_management": [ | |
| _crit("stakeholder management", "soft_skill", "required", ["stakeholder communication"]), | |
| _crit("risk management", "responsibility", "required", ["risk mitigation"]), | |
| _crit("Agile", "hard_skill", "required", ["scrum"]), | |
| _crit("resource planning", "responsibility", "preferred"), | |
| ], | |
| "operations": [ | |
| _crit("supply chain", "domain", "required", ["logistics"]), | |
| _crit("process optimization", "responsibility", "required", ["process improvement"]), | |
| _crit("inventory management", "hard_skill", "required", ["stock management"]), | |
| _crit("vendor management", "responsibility", "preferred"), | |
| ], | |
| } | |
| def make_jd(criteria): | |
| """Build a valid, isolatable JD that contains each criterion's exact phrase in | |
| the appropriate section (so the traceability gate — correctly — accepts them).""" | |
| req = [c["exact_phrase"] for c in criteria if c["requirement_type"] == "required"] | |
| pref = [c["exact_phrase"] for c in criteria if c["requirement_type"] != "required"] | |
| lines = ["About the Role", | |
| "In this role you will own key initiatives and deliver measurable " | |
| "outcomes with our team.", "", | |
| "Responsibilities"] | |
| for c in criteria: | |
| lines.append(f"- You will drive {c['exact_phrase']} to support team outcomes.") | |
| lines += ["", "Requirements", | |
| "- You must have 5+ years of relevant professional experience in the field."] | |
| for p in req: | |
| lines.append(f"- You must have strong {p} for this role.") | |
| lines += ["", "Preferred Qualifications"] | |
| for p in (pref or ["cross-functional collaboration"]): | |
| lines.append(f"- Experience with {p} is nice to have.") | |
| return "\n".join(lines) | |
| RESUME = r"""\section{EXPERIENCE} | |
| \resumeItem{Owned stakeholder communication and roadmap planning; ran experiments and built SQL dashboards for 1M+ users.} | |
| \resumeItem{Managed campaigns and risk mitigation across Agile teams; improved process improvement and logistics.} | |
| \section{SKILLS} | |
| \resumeItem{SQL, Python, Agile.} | |
| """ | |
| def _run(criteria, resume=RESUME): | |
| return generate_alignment_safe(resume, make_jd(criteria), company="X", job_title="Role", | |
| llm_client=RoleLLM(criteria), rewrite_fn=None, | |
| compile_pdf=False) | |
| def test_each_role_produces_distinct_critical_keywords(): | |
| crit_sets = {} | |
| for role, crits in ROLES.items(): | |
| safe = _run(crits) | |
| crit_sets[role] = frozenset(c["concept"] for c in safe["calibration"]) | |
| # No two roles share the same critical-keyword set → not a fixed list. | |
| seen = list(crit_sets.values()) | |
| assert len(set(seen)) == len(seen), "roles reuse the same critical keyword set" | |
| # PM and SWE must differ substantially. | |
| assert crit_sets["product_management"] != crit_sets["software_engineering"] | |
| def test_mandatory_vs_preferred_distinguished(): | |
| safe = _run(ROLES["product_management"]) | |
| reqs = {c["exact_phrase"] for c in safe["extraction"]["valid"] | |
| if c["requirement_type"] == "required"} | |
| prefs = {c["exact_phrase"] for c in safe["extraction"]["valid"] | |
| if c["requirement_type"] == "preferred"} | |
| assert "SQL" in reqs and "A/B testing" in prefs | |
| assert reqs and prefs and not (reqs & prefs) | |
| def test_unsupported_terms_stay_gaps(): | |
| # SWE criteria vs a PM résumé → distributed systems / kubernetes unsupported. | |
| safe = _run(ROLES["software_engineering"]) | |
| gaps = {g["keyword"] for g in safe["evidence"]["gaps"]} | |
| tex = safe["tex"].lower() | |
| assert "kubernetes" in gaps and "kubernetes" not in tex | |
| assert "distributed systems" in gaps | |
| def test_no_fixed_keyword_list_reused_across_roles(): | |
| all_terms = [] | |
| for crits in ROLES.values(): | |
| safe = _run(crits) | |
| all_terms.append(tuple(sorted(c["concept"] for c in safe["calibration"]))) | |
| assert len(set(all_terms)) >= 5, "critical keywords barely vary across roles" | |
| def test_score_never_credits_absent_phrase(): | |
| # A criterion whose phrase is NOT in the résumé must not count as covered. | |
| crits = [_crit("blockchain", "hard_skill", "required")] | |
| ev = {"covered": [], "partial": [], "gaps": [{"keyword": "blockchain", | |
| "requirement_type": "required"}]} | |
| cov = compute_coverage(crits, ev, "I build web apps with SQL.") | |
| assert cov["critical_family_coverage"] in (0.0, None) | |
| def test_keyword_stuffed_resume_scores_lower(): | |
| crits = ROLES["software_engineering"] | |
| natural = r"\resumeItem{Built microservices and distributed systems on Kubernetes with CI/CD.}" | |
| stuffed = (r"\resumeItem{Built microservices and distributed systems on Kubernetes with CI/CD.}" | |
| r"\resumeItem{Skills: Kubernetes, Docker, Go, Rust, Java, C++, Scala, Kafka, Redis, gRPC, Terraform.}") | |
| s_nat = generate_alignment_safe(natural, make_jd(crits), llm_client=RoleLLM(crits), | |
| rewrite_fn=None, compile_pdf=False) | |
| s_stf = generate_alignment_safe(stuffed, make_jd(crits), llm_client=RoleLLM(crits), | |
| rewrite_fn=None, compile_pdf=False) | |
| assert s_stf["internal_alignment_estimate"]["penalties"], "stuffing not penalized" | |
| assert (s_stf["internal_alignment_estimate"]["after"] | |
| <= s_nat["internal_alignment_estimate"]["after"] + 0.01) | |
| def test_missing_supported_mandatory_flagged(): | |
| # Résumé supports 'stakeholder management' (via communication) but never the | |
| # exact phrase, and no rewriter runs → it must show as missing in coverage. | |
| crits = [_crit("stakeholder management", "soft_skill", "required", ["stakeholder communication"])] | |
| safe = generate_alignment_safe( | |
| r"\resumeItem{Owned stakeholder communication for the team.}", | |
| make_jd(crits), llm_client=RoleLLM(crits), rewrite_fn=None, compile_pdf=False) | |
| cov = safe["internal_alignment_estimate"]["coverage"] | |
| # supported but exact phrase absent → critical exact-phrase coverage < 1 | |
| assert (cov["critical_exact_phrase_coverage"] or 0) < 1.0 | |
| assert not safe["internal_alignment_estimate"]["gate_90_passed"] | |
| def test_reaches_90_when_genuinely_supported(): | |
| """When the candidate GENUINELY supports every mandatory/critical criterion, | |
| truthful rewriting lifts the alignment to >=90 and the 90% gate passes — with | |
| zero unsupported insertions. (No fabrication; the fixture really supports it.)""" | |
| resume = (r"\section{EXPERIENCE}" | |
| r"\resumeItem{Owned stakeholder communication and product roadmap planning " | |
| r"for a B2B SaaS platform serving 1M+ users, lifting activation 18\%.}" | |
| r"\resumeItem{Ran experiments with cross-functional teams and built SQL " | |
| r"dashboards to guide decisions.}" | |
| r"\section{SKILLS}\resumeItem{SQL, Product Analytics.}") | |
| crits = [ | |
| _crit("stakeholder management", "soft_skill", "required", ["stakeholder communication"], "critical"), | |
| _crit("roadmap prioritization", "responsibility", "required", ["product roadmap planning"], "critical"), | |
| _crit("product experimentation", "hard_skill", "required", ["experiments"], "critical"), | |
| _crit("cross-functional collaboration", "responsibility", "required", ["cross-functional teams"], "critical"), | |
| _crit("SQL", "tool", "required"), | |
| ] | |
| rwmap = { | |
| "stakeholder management": ("stakeholder communication", "stakeholder management"), | |
| "roadmap prioritization": ("product roadmap planning", "roadmap prioritization"), | |
| "product experimentation": ("Ran experiments", "Ran product experimentation"), | |
| "cross-functional collaboration": ("with cross-functional teams", | |
| "through cross-functional collaboration with teams"), | |
| } | |
| def rw(o, t, c, cat): | |
| m = rwmap.get(t.lower()) | |
| return re.sub(re.escape(m[0]), m[1], o, count=1, flags=re.IGNORECASE) if m else o | |
| safe = generate_alignment_safe(resume, make_jd(crits), llm_client=RoleLLM(crits), | |
| rewrite_fn=rw, compile_pdf=False) | |
| e = safe["internal_alignment_estimate"] | |
| assert e["before"] < 90 <= e["after"], (e["before"], e["after"]) | |
| assert e["gate_90_passed"] is True | |
| assert e["unsupported_insertions"] == 0 | |
| assert e["coverage"]["mandatory_coverage"] == 1.0 | |
| assert (e["coverage"]["critical_exact_phrase_coverage"] or 0) >= 0.85 | |
| def test_routes_share_pipeline(): | |
| import api_server | |
| src = inspect.getsource(api_server) | |
| assert src.count("generate_alignment_safe") >= 2 | |
| assert src.count("build_llm") >= 2 # SSE + blocking both use NIM fallback | |
| if __name__ == "__main__": | |
| sys.exit(pytest.main([__file__, "-x", "-q"])) | |