JAA-ATS-Tool / src /candidate_vault.py
saitejatirunagari's picture
feat(ats): Maximum ATS Mode (User-Confirmed Skill Expansion) + coverage report
b3704a6
Raw
History Blame
6.27 kB
"""
Candidate Experience Expansion Vault (spec item #6).
A persistent store of everything we've learned about the candidate's usable
experience across jobs — so the system gets stronger over time and stays
consistent (a term confirmed safe on one job stays safe on the next).
Persisted to data/candidate_vault.json. Each entry:
{ term, category, source, confidence, usage_guidance, example_bullet }
source ∈ resume | inferred_plausible | user_confirmed | jd_expansion
confidence ∈ high | medium | low
usage_guidance ∈ safe_to_use | use_carefully | ask_user | blocked
The vault FEEDS the fit classifier:
- user_confirmed terms → always safe (treated as explicit)
- blocked terms (user said no)→ always blocked
"""
from __future__ import annotations
import json
import os
from typing import Dict, List
_VAULT_PATH = "data/candidate_vault.json"
# Per spec #8: only resume_original + user_confirmed are FULLY safe. Everything
# inferred/expanded stays REVIEWABLE — the vault must never silently promote a
# risky/plausible JD term to "safe" just because it appeared in past resumes.
_FIT_TO_GUIDANCE = {
"explicit": "safe_to_use", # actually present in the original resume
"plausible": "use_carefully", # reviewable until user confirms
"adjacent": "use_carefully",
"risky": "ask_user",
"blocked": "blocked",
}
_FIT_TO_SOURCE = {
"explicit": "resume_original",
"plausible": "inferred_plausible",
"adjacent": "inferred_plausible",
"risky": "risky_review",
"blocked": "blocked",
}
_FIT_TO_CONF = {"explicit": "high", "plausible": "medium", "adjacent": "medium",
"risky": "low", "blocked": "low"}
def load_vault(path: str = _VAULT_PATH) -> Dict[str, dict]:
"""Return {term_lower: entry}. Empty dict if none yet."""
try:
with open(path, encoding="utf-8") as f:
data = json.load(f)
return {e["term"].lower(): e for e in data.get("entries", []) if e.get("term")}
except Exception:
return {}
def save_vault(vault: Dict[str, dict], path: str = _VAULT_PATH) -> None:
try:
os.makedirs(os.path.dirname(path), exist_ok=True)
with open(path, "w", encoding="utf-8") as f:
json.dump({"entries": list(vault.values())}, f, ensure_ascii=False, indent=2)
except Exception as e:
print(f"[vault] save skipped: {e}")
def update_from_fit(verdicts, path: str = _VAULT_PATH) -> Dict[str, dict]:
"""Merge a job's FitVerdicts into the vault. user_confirmed entries are never
downgraded by automated runs."""
vault = load_vault(path)
for v in verdicts:
key = v.keyword.lower().strip()
if not key:
continue
existing = vault.get(key)
# Never overwrite a user_confirmed / user-blocked decision automatically.
if existing and existing.get("source") == "user_confirmed":
continue
vault[key] = {
"term": v.keyword,
"category": v.category,
"source": _FIT_TO_SOURCE.get(v.fit_status, "jd_expansion"),
"confidence": _FIT_TO_CONF.get(v.fit_status, "low"),
"usage_guidance": _FIT_TO_GUIDANCE.get(v.fit_status, "use_carefully"),
"example_bullet": (existing or {}).get("example_bullet", ""),
}
save_vault(vault, path)
return vault
def confirm_terms(terms: List[str], confirmed: bool = True,
path: str = _VAULT_PATH) -> None:
"""User explicitly confirms (or rejects) terms — the strongest signal."""
vault = load_vault(path)
for t in terms:
key = t.lower().strip()
if not key:
continue
e = vault.get(key, {"term": t, "category": "user", "example_bullet": ""})
e["source"] = "user_confirmed"
e["confidence"] = "high"
e["usage_guidance"] = "safe_to_use" if confirmed else "blocked"
vault[key] = e
save_vault(vault, path)
def user_confirmed_terms(path: str = _VAULT_PATH) -> set:
v = load_vault(path)
return {k for k, e in v.items()
if e.get("source") == "user_confirmed" and e.get("usage_guidance") == "safe_to_use"}
def user_blocked_terms(path: str = _VAULT_PATH) -> set:
v = load_vault(path)
return {k for k, e in v.items() if e.get("usage_guidance") == "blocked"}
def confirm_expansion_terms(terms: List[str], category: str = "max_ats_expansion",
path: str = _VAULT_PATH) -> List[str]:
"""Persist Maximum-ATS / user-confirmed expansion terms as user_confirmed so
future resumes treat them as safe. Returns the list of newly-stored terms.
This is the explicit-confirmation path: the user confirmed (via the extension
"Maximum ATS Mode" / "Confirm these terms") that these are interview-supportable.
Truly high-risk terms are still only stored when the user confirms them here —
nothing is silently promoted by an automated run.
"""
terms = [t for t in (terms or []) if t and t.strip()]
if not terms:
return []
vault = load_vault(path)
blocked = {k for k, e in vault.items() if e.get("usage_guidance") == "blocked"}
stored: List[str] = []
for t in terms:
key = t.lower().strip()
if not key or key in blocked:
continue # never override a user block
e = vault.get(key, {"term": t, "example_bullet": ""})
e["term"] = e.get("term") or t
e["category"] = e.get("category") or category
e["source"] = "user_confirmed"
e["confidence"] = "high"
e["usage_guidance"] = "safe_to_use"
vault[key] = e
stored.append(t)
save_vault(vault, path)
return stored
def vault_summary(path: str = _VAULT_PATH) -> dict:
"""Reporting helper: counts + the user-confirmed term list, so the UI can
show what's been added to the vault."""
v = load_vault(path)
by_source: Dict[str, int] = {}
for e in v.values():
by_source[e.get("source", "unknown")] = by_source.get(e.get("source", "unknown"), 0) + 1
return {
"total": len(v),
"by_source": by_source,
"user_confirmed": sorted(user_confirmed_terms(path)),
"blocked": sorted(user_blocked_terms(path)),
}