""" Verify the new systemic keyword extraction works on ALL 7 JDs (4 tuned + 3 new) without per-JD noise tuning. The 3 "new" JDs (Navi, zenda, generic_pm) have NEVER been used to tune the noise filter. If they score well, the systemic approach works. """ import os, sys, io, shutil sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding='utf-8') src_pdf = r'C:\Users\Nxtwave\Desktop\resume\Saiteja_Tirunagari_Resume A 26 - Copy.pdf' dst_pdf = 'data/resume/resume.pdf' os.makedirs(os.path.dirname(dst_pdf), exist_ok=True) if not os.path.exists(dst_pdf): shutil.copyfile(src_pdf, dst_pdf) # Use the same backfill / pitch / weave flow as production from src.resume_customizer import ResumeCustomizer, _read_docx_text from src.llm_client import LLMClient from src.ats_scorer import score_resume, extract_jd_keywords def make_weak_response(resume_dict, job_title, company): """Worst-case production LLM: 2 roles, no pitch, 3 bullets each.""" return { "name": resume_dict["name"], "contact": resume_dict["contact"], "summary": ( "Product Manager with 5+ years of experience driving product development " "in fast-paced startups. Built funnel optimization and AI-powered features " "that scaled to 140,000+ users with measurable business impact." ), "roles": [ { "title": resume_dict["roles"][0]["title"], "company": resume_dict["roles"][0]["company"], "location": resume_dict["roles"][0]["location"], "dates": resume_dict["roles"][0]["dates"], "bullets": [ "Led NIAT Application Portal revamp scaling to 141,269 verified leads", "Drove payment conversion lift of +35.87 percentage points", "Built AI chatbot generating 6,776 leads", ], }, { "title": resume_dict["roles"][1]["title"], "company": resume_dict["roles"][1]["company"], "location": resume_dict["roles"][1]["location"], "dates": resume_dict["roles"][1]["dates"], "bullets": [ "Managed 20 customer-success specialists covering 40,000 customers", "Maintained refund rate below 5% and satisfaction above 95%", ], }, ], "achievements": [ "Scaled to 141,269 verified leads", "Lifted payment conversion +35.87 pp", ], "education": resume_dict["education"], } _original_v4 = LLMClient.tailor_resume_v4 def _mocked_v4(self, cfg, resume_dict, jd_text, job_title, company, assessment): return make_weak_response(resume_dict, job_title, company) LLMClient.tailor_resume_v4 = _mocked_v4 rc = ResumeCustomizer.__new__(ResumeCustomizer) rc.llm = LLMClient.__new__(LLMClient) rc.resume_text = '' rc.output_dir = 'data/output/resumes/_systemic_test' os.makedirs(rc.output_dir, exist_ok=True) rc.fast_model_cfg = {"model": "fake", "api_key": "fake", "base_url": "https://fake"} rc._pending_summary_inject = [] print('━' * 86) print(f'{"JD (tuned)":<32} {"ATS":>5} {"JD-match":>10} {"Words":>6} {"JD-kw":>6}') print('━' * 86) # Group 1: JDs I've tuned the noise filter for tuned = [ ('airtel_pm', 'Product Manager', 'Airtel'), ('sumo_logic_pm', 'Product Manager', 'Sumo Logic'), ('edgeverve_pm', 'Product Manager', 'EdgeVerve'), ('aditya_birla_apm', 'Associate Product Manager', 'Aditya Birla Capital'), ] # Group 2: NEW JDs — never used to tune anything new_jds = [ ('navi_pm', 'Product Manager', 'Navi'), ('zenda_apm', 'Associate Product Manager', 'zenda'), ('generic_pm_3_7yrs', 'Product Manager', 'Generic'), ] for jd_file, role, company in tuned: with open(f'tests/fixtures/jds/{jd_file}.txt', encoding='utf-8') as f: jd = f.read() job = {"title": role, "company": company, "description": jd, "_raw_assessment": {}} filepath = os.path.join(rc.output_dir, f'{company.replace(" ", "_")}.docx') result = rc._generate_resume_v4(job, cfg=rc.fast_model_cfg, filepath=filepath) if not result: print(f'{(company):<32} FAILED') continue text = _read_docx_text(result) r = score_resume(text, jd) print(f'{company:<32} {r["ats_score"]:>4} {r["matched_count"]}/{r["total_jd_kw"]:<3} {r["word_count"]:>5} {r["total_jd_kw"]:>4}') print() print(f'{"JD (NEW — never seen)":<32} {"ATS":>5} {"JD-match":>10} {"Words":>6} {"JD-kw":>6}') print('-' * 86) for jd_file, role, company in new_jds: with open(f'tests/fixtures/jds/{jd_file}.txt', encoding='utf-8') as f: jd = f.read() job = {"title": role, "company": company, "description": jd, "_raw_assessment": {}} filepath = os.path.join(rc.output_dir, f'{company.replace(" ", "_")}_{jd_file}.docx') result = rc._generate_resume_v4(job, cfg=rc.fast_model_cfg, filepath=filepath) if not result: print(f'{(company):<32} FAILED') continue text = _read_docx_text(result) r = score_resume(text, jd) print(f'{company:<32} {r["ats_score"]:>4} {r["matched_count"]}/{r["total_jd_kw"]:<3} {r["word_count"]:>5} {r["total_jd_kw"]:>4}') print('━' * 86) print('\nThe NEW JDs simulate production new-job behavior. If they score ~the') print('same as the tuned JDs, the systemic extraction works without per-JD') print('blocklist tuning.')