JAA-ATS-Tool / scripts /verify_scoring_v2.py
saitejatirunagari's picture
feat(ats): v2 pipeline β€” structured JD, evidence matching, weighted 2-part score
2bcd6c2
Raw
History Blame
3.59 kB
"""
Regression for the v2 ATS pipeline (jd_analyzer + evidence_matcher +
ats_scoring_v2 + ats_report). Deterministic only (no LLM) β€” proves the floor.
Asserts:
1. Structured JD analysis returns categorized requirements.
2. Evidence matcher marks out-of-domain skills (SIEM/SOAR/threat intel) as
GAPS against the real PM resume (no faking), and PM-core as supported/transferable.
3. Weighted JD match + ATS readability produce sane numbers.
4. The full report assembles with all explanation fields.
5. Missing-keyword feedback loop decides correctly.
"""
import os, sys, io, shutil
sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8")
sys.path.insert(0, os.path.abspath("."))
from src.resume_parser_v2 import parse_resume_pdf
from src.jd_analyzer import analyze_jd
from src.evidence_matcher import classify_all, gap_terms
from src.ats_scoring_v2 import score_ats_readability, score_jd_match
from src.ats_report import build_ats_report, reconcile_missing_keywords
src_pdf = r"C:\Users\Nxtwave\Desktop\resume\Saiteja_Tirunagari_Resume A 26 - Copy.pdf"
dst = "data/resume/resume.pdf"
os.makedirs("data/resume", exist_ok=True)
if not os.path.exists(dst) and os.path.exists(src_pdf):
shutil.copyfile(src_pdf, dst)
base = parse_resume_pdf(dst).to_flat_text()
ok = True
def check(name, cond, extra=""):
global ok
ok = ok and cond
print(f" [{'PASS' if cond else 'FAIL'}] {name} {extra}")
# 1. Structured analysis
jd_sec = open("tests/fixtures/jds/sumo_logic_pm.txt", encoding="utf-8").read()
req = analyze_jd(jd_sec)
print("1. Structured JD analysis (Sumo Logic / security PM):")
check("has target title", bool(req.target_role_titles), req.target_role_titles)
check("has hard skills", len(req.required_hard_skills + req.preferred_hard_skills) > 0)
check("has domain terms", len(req.domain_terms) > 0, [d.term for d in req.domain_terms])
# 2. Evidence honesty: security domain must be gaps; PM-core not faked
verdicts = classify_all(req.all_requirements(), base)
gaps = {v.keyword.lower() for v in gap_terms(verdicts)}
print("2. Evidence honesty:")
check("SIEM is a gap (not faked)", "siem" in gaps)
check("threat intelligence is a gap", "threat intelligence" in gaps)
supported = {v.keyword.lower() for v in verdicts if v.evidence_status in ("supported", "transferable")}
check("product management supported/transferable", "product management" in supported)
# 3 + 4. Full report on an IN-DOMAIN PM JD (use base as stand-in final text)
jd_pm = open("tests/fixtures/jds/generic_pm_3_7yrs.txt", encoding="utf-8").read()
rep = build_ats_report(base, base, jd_pm)
print("3+4. Report (generic PM JD):", rep["estimated_scores"])
check("ats_readability is int 0-100", 0 <= rep["estimated_scores"]["ats_readability"] <= 100)
check("jd_match is int 0-100", 0 <= rep["estimated_scores"]["jd_match"] <= 100)
check("has combined_range", "-" in rep["estimated_scores"]["combined_range"])
for f in ("strong_matches", "weak_matches", "unsupported_missing_keywords",
"formatting_checks", "recommendations", "jd_match_breakdown"):
check(f"report has {f}", f in rep)
# 5. Feedback loop
rows = reconcile_missing_keywords(["Kubernetes", "stakeholder management", "blockchain"],
jd_pm, base)
print("5. Feedback loop:")
dmap = {r["keyword"]: r["decision"] for r in rows}
check("Kubernetes -> ignore (not in JD)", dmap.get("Kubernetes") == "ignore")
check("blockchain -> ignore (not in JD)", dmap.get("blockchain") == "ignore")
print("\n" + ("βœ“ ALL PASS" if ok else "βœ— SOME FAILED"))
sys.exit(0 if ok else 1)