Spaces:
Sleeping
Sleeping
File size: 6,999 Bytes
9bf4a3d | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 | """
Per-job validation package (spec #9).
For every generated resume we can export a self-contained folder that bundles
everything needed for MANUAL Jobalytics / Simplify verification and for auditing
the model-independent pipeline:
final resume DOCX + PDF
parsed_resume.txt (the EXPORTED file re-parsed back to text)
jd.txt (the job description we tailored against)
manifest.json (scores, keyword lists, risk terms, provider info)
README.md (human-readable summary)
It never generates or scores anything — it only collects what the pipeline
already produced on the job dict (`_v2_report`). This is what the user opens to
confirm our internal numbers against a real external checker.
"""
from __future__ import annotations
import os
import re
import json
import shutil
from datetime import datetime
def _safe(s: str, n: int = 60) -> str:
return re.sub(r'[\\/*?:"<>|]', "", str(s or ""))[:n].strip() or "item"
def build_validation_package(job: dict, base_resume_text: str = None,
out_root: str = "data/output/validation") -> str:
"""Write a validation package folder for one job and return its path."""
report = job.get("_v2_report", {}) or {}
est = report.get("estimated_scores", {}) or {}
date_str = datetime.now().strftime("%Y-%m-%d")
folder = os.path.join(
out_root, date_str,
f"{_safe(job.get('company', 'Company'), 40)}_{_safe(job.get('title', 'Role'), 40)}",
)
os.makedirs(folder, exist_ok=True)
# 1. Copy the resume files
docx_path = job.get("resume_path", "")
pdf_path = job.get("resume_pdf_path", "")
parsed_text = ""
if docx_path and os.path.exists(docx_path):
try:
shutil.copy2(docx_path, os.path.join(folder, "resume" + os.path.splitext(docx_path)[1]))
except Exception:
pass
try:
from .resume_customizer import _read_docx_text
parsed_text = _read_docx_text(docx_path)
except Exception:
parsed_text = ""
if pdf_path and os.path.exists(pdf_path):
try:
shutil.copy2(pdf_path, os.path.join(folder, "resume.pdf"))
except Exception:
pass
# 2. Parsed text + JD
jd_text = job.get("description", "") or ""
with open(os.path.join(folder, "parsed_resume.txt"), "w", encoding="utf-8") as f:
f.write(parsed_text)
with open(os.path.join(folder, "jd.txt"), "w", encoding="utf-8") as f:
f.write(jd_text)
# 3. Manifest — every field spec #9 asks for
risky = report.get("risky_review_table", []) or []
medium_terms = report.get("review_terms_for_user_review", []) or []
high_terms = report.get("high_risk_terms_for_confirmation", []) or []
blocked_terms = [d.get("keyword") for d in report.get("unsupported_missing_keywords", []) or []]
manifest = {
"job": {
"title": job.get("title", ""),
"company": job.get("company", ""),
"location": job.get("location", ""),
"platform": job.get("platform", job.get("source", "")),
"url": job.get("url", ""),
},
"status": job.get("status", report.get("status", "")),
"download_allowed": bool(report.get("download_allowed", job.get("download_allowed"))),
"quality_flag": report.get("quality_flag", job.get("quality_flag", "")),
"scores": {
"internal_jd_match": est.get("jd_match", job.get("ats_score_after", 0)),
"independent_jd_match": report.get("independent_jd_match",
job.get("independent_jd_match", 0)),
"ats_readability": est.get("ats_readability", 0),
"combined_range": est.get("combined_range", ""),
},
"provider": {
"provider_used": report.get("provider_used", job.get("provider_used", "")),
"provider_response_quality": report.get("provider_response_quality", ""),
"provider_attempts": report.get("provider_attempts", []),
},
"repair_attempts": report.get("repair_attempts", []),
"keywords": {
"jd_covered_terms": report.get("covered_terms", []),
"jd_missing_terms": report.get("missing_terms", []),
"keywords_added_evidenced": report.get("evidenced_terms", []),
"keywords_added_skills_only": report.get("skills_only_terms", []),
"medium_risk_terms": medium_terms,
"high_risk_terms": high_terms,
"blocked_terms": blocked_terms,
},
"risky_review_table": risky,
"files": {
"docx": os.path.basename(docx_path) if docx_path else "",
"pdf": "resume.pdf" if (pdf_path and os.path.exists(os.path.join(folder, "resume.pdf"))) else "",
"parsed_resume": "parsed_resume.txt",
"jd": "jd.txt",
},
"generated_at": datetime.now().isoformat(timespec="seconds"),
}
with open(os.path.join(folder, "manifest.json"), "w", encoding="utf-8") as f:
json.dump(manifest, f, ensure_ascii=False, indent=2)
# 4. Human-readable summary
s = manifest["scores"]
lines = [
f"# Validation Package — {job.get('title','')} at {job.get('company','')}",
"",
f"- Status: **{manifest['status']}** | Download allowed: {manifest['download_allowed']}",
f"- Provider used: **{manifest['provider']['provider_used']}** "
f"(schema: {manifest['provider']['provider_response_quality']})",
f"- Internal JD match: **{s['internal_jd_match']}**",
f"- Independent JD match: **{s['independent_jd_match']}**",
f"- ATS readability: **{s['ats_readability']}**",
f"- Repair attempts: {len(manifest['repair_attempts'])}",
"",
"## How to verify",
"1. Open `resume.pdf` (or the DOCX) and upload it to Jobalytics / Simplify.",
"2. Paste `jd.txt` as the job description.",
"3. Compare the external score to the independent score above.",
"4. Review `manifest.json` -> keywords.medium_risk_terms / high_risk_terms / blocked_terms.",
"",
"## Provider attempts",
]
for a in manifest["provider"]["provider_attempts"]:
lines.append(f"- {a}")
with open(os.path.join(folder, "README.md"), "w", encoding="utf-8") as f:
f.write("\n".join(lines))
return folder
def build_packages_for_jobs(jobs: list, base_resume_text: str = None,
out_root: str = "data/output/validation") -> list:
"""Build packages for a list of jobs; returns list of folder paths."""
paths = []
for job in jobs:
if not job.get("resume_path"):
continue
try:
paths.append(build_validation_package(job, base_resume_text, out_root))
except Exception as e:
print(f"[validation-package] {job.get('company','?')}: {e}")
return paths
|