File size: 16,523 Bytes
dd47e55
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6c45f52
 
 
 
 
 
 
 
 
 
dd47e55
 
6d3a6a8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
dd47e55
 
 
 
 
 
 
 
 
 
 
 
 
b3704a6
 
 
 
 
 
 
 
 
 
 
 
 
 
dd47e55
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6d3a6a8
 
 
 
 
 
 
dd47e55
6d3a6a8
 
 
 
 
 
dd47e55
 
 
 
 
 
 
6d3a6a8
 
 
dd47e55
 
 
 
 
 
db0cec8
6d3a6a8
b3704a6
 
dd47e55
 
 
 
db0cec8
 
 
 
 
 
 
 
 
 
b3704a6
 
 
6d3a6a8
dd47e55
 
 
 
 
 
 
b3704a6
 
 
 
 
 
 
 
 
dd47e55
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
db0cec8
b3704a6
 
 
dd47e55
6d3a6a8
db0cec8
 
 
 
 
 
 
 
b3704a6
 
 
 
 
 
db0cec8
6d3a6a8
b3704a6
 
dd47e55
 
 
7b41b0f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
dd47e55
 
 
 
 
 
 
7b41b0f
 
 
 
 
 
 
 
 
 
dd47e55
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
"""
Candidate Fit Expansion (spec: aggressive_plausible_match mode).

Philosophy shift: the uploaded resume is a BASE PROFILE, not the candidate's
full experience record. "Not in the resume" != "false". For a strong ATS resume
we INCLUDE every JD requirement that is plausible for the candidate's role,
seniority, and domain β€” the candidate validates in the interview.

We still refuse to fabricate the things that are checkable and damaging to lie
about: regulated certifications/licenses, degrees not present, unrelated deep
technical specialties, and seniority jumps. Fake employers/dates/achievements
are handled elsewhere (the tailoring contract never invents those).

Per-requirement output:
  fit_status ∈ explicit | plausible | adjacent | risky | blocked
  action     ∈ include  | include_carefully | ask_user | block
"""

from __future__ import annotations

import re
from dataclasses import dataclass, field, asdict
from typing import List, Dict

from .ats_scorer import _kw_in_text
from .jd_analyzer import Requirement, JDRequirements
from .evidence_matcher import _RELATED, _PM_CORE_TRANSFERABLE

# ── BLOCK rules β€” the only hard "no". Everything else is includable. ──────────

# Regulated / credential terms: claiming these without holding them is harmful
# and checkable. Surfaced as ask_user (degrees) or block (licenses/clearances).
_REGULATED_CRED = re.compile(
    r"\b(cpa|cfa|series\s?\d+|bar exam|licensed|license\b|pmp\b|csm\b|cissp|"
    r"ccsp|cisa|cism|comptia|security\+|aws certified|gcp certified|"
    r"azure certified|certified scrum|six sigma (black|green) belt|"
    r"md\b|m\.?d\.?|rn\b|registered nurse|security clearance|clearance\b|"
    r"professional engineer|pe license)\b", re.I)

# Deep technical specialties that don't fit a Product Manager profile β€” claiming
# hands-on here is an unrealistic specialty jump. (Tool-level familiarity may be
# fine; hands-on engineering is risky/blocked.)
_DEEP_TECH_SPECIALTY = {
    "kernel", "firmware", "embedded systems", "fpga", "vlsi", "rtl", "verilog",
    "assembly", "device drivers", "bioinformatics", "actuarial", "soldering",
    "circuit design", "cryptography research", "penetration testing",
    "reverse engineering", "malware analysis", "exploit development",
    # Robotics / hardware / control hands-on engineering β€” a PM may COORDINATE
    # these but must never CLAIM having built them. (General business domains like
    # real estate, HRMS, fintech, healthcare are intentionally NOT here β€” those are
    # transferable for a PM and stay weavable.)
    "robotics", "robotic", "autonomous systems", "autonomous vehicles",
    "self-driving", "lidar", "slam", "sensor fusion", "motion planning",
    "path planning", "perception systems", "control systems", "mechatronics",
    "mechanical engineering", "electrical engineering", "hardware engineering",
    "hardware design", "pcb design", "pcb", "rtos", "microcontroller", "actuators",
    "servo", "plc programming", "scada", "asic", "semiconductor", "embedded firmware",
}

# Hands-on software-engineering hard skills. A PM may COORDINATE these but
# claiming them as personal hard skills on a PM resume is a specialty mismatch β€”
# treat as RISKY (review), and BLOCK the most clearly engineer-only ones. This
# is what keeps a backend-engineer JD from becoming a CLEAN_90 PM resume.
_ENGINEERING_HARDCORE = {
    "java", "c++", "c#", "golang", "go programming", "rust", "scala", "kotlin",
    "spring boot", "spring framework", ".net", "microservices architecture",
    "distributed systems", "system design", "data structures", "algorithms",
    "multithreading", "concurrency", "jvm", "garbage collection", "compilers",
    "operating systems", "computer networks", "tcp/ip", "grpc", "graphql api design",
    "database internals", "query optimization", "kafka", "rabbitmq", "redis",
    "elasticsearch internals", "hadoop", "spark", "mapreduce", "cuda",
    "tensorflow", "pytorch", "deep learning models", "model training",
    "backend development", "frontend development", "full stack development",
    "react.js", "angular", "node.js", "android development", "ios development",
}

# Seniority signals above the candidate's level (set after inferring level).
_SENIOR_TERMS = {"vp", "vice president", "director", "head of", "chief",
                 "cto", "cpo", "ceo", "svp", "evp", "20+ years", "15+ years"}

# Specialized DOMAIN tooling/knowledge β€” includable but flagged for review when
# it's outside the candidate's evidenced domains (use_carefully).
_SPECIALIZED_DOMAIN = {
    "siem", "soar", "xdr", "edr", "threat intelligence", "threat detection",
    "security operations", "secops", "incident response", "vulnerability management",
    "hl7", "fhir", "hipaa", "clinical", "pharmacovigilance", "underwriting",
    "actuarial", "basel", "kyc", "aml", "pci dss", "swift", "fix protocol",
}

# Maximum ATS Mode: normal PM/Product/AI/SaaS/B2B/agile vocabulary treated as
# user-confirmed / interview-supportable (loaded from config; safe fallback).
try:
    from config import MAXIMUM_ATS_SAFE_TERMS as _MAX_ATS_SAFE
except Exception:  # pragma: no cover - config import safety
    _MAX_ATS_SAFE = set()


def _is_max_ats_safe(term: str) -> bool:
    """True when `term` is normal target-role PM/AI/SaaS vocabulary that
    Maximum ATS Mode may treat as user-confirmed. Never matches credentials,
    seniority, employers, or specialized hands-on engineering terms."""
    return term.lower().strip() in _MAX_ATS_SAFE


@dataclass
class FitVerdict:
    keyword: str
    category: str
    fit_status: str       # explicit|plausible|adjacent|risky|blocked
    action: str           # include|include_carefully|ask_user|block
    reason: str = ""
    importance: str = "preferred"
    recommended_placement: List[str] = field(default_factory=list)

    def to_dict(self):
        return asdict(self)


def infer_seniority(base_resume_text: str) -> str:
    """Rough candidate level from the base resume."""
    low = base_resume_text.lower()
    if any(t in low for t in ("vp ", "vice president", "director", "head of", "chief")):
        return "senior_plus"
    m = re.search(r"(\d{1,2})\+?\s*years", low)
    yrs = int(m.group(1)) if m else 0
    if yrs >= 8 or "senior" in low or "lead" in low or "principal" in low:
        return "senior"
    return "mid"


def _candidate_years(base_resume_text: str) -> int:
    yrs = [int(m) for m in re.findall(r"(\d{1,2})\+?\s*years?", base_resume_text.lower())]
    return max(yrs) if yrs else 0


def _is_blocked(term: str, category: str, seniority: str,
                candidate_years: int = 0) -> tuple:
    t = term.lower().strip()
    # Seniority overreach: a JD year requirement above the candidate's actual
    # experience must NOT be claimed (would be a fabricated tenure).
    if category == "seniority":
        m = re.search(r"(\d{1,2})", t)
        if m and candidate_years and int(m.group(1)) > candidate_years + 1:
            return ("block", f"requires {m.group(1)}y; candidate has ~{candidate_years}y")
    if _REGULATED_CRED.search(t):
        # Degrees/credentials β†’ ask the user rather than hard-block (they may hold it)
        if category in ("certification", "education"):
            return ("ask_user", "credential β€” confirm you hold it")
        return ("block", "regulated credential/license not in profile")
    if t in _DEEP_TECH_SPECIALTY:
        return ("block", "deep technical specialty outside a PM profile")
    if t in _ENGINEERING_HARDCORE:
        # Engineer-only hard skill on a PM resume β†’ review, don't auto-include.
        return ("ask_user", "hands-on engineering skill β€” confirm before claiming")
    if t in _SENIOR_TERMS and seniority != "senior_plus":
        return ("block", "seniority above candidate level")
    return ("", "")


def classify_fit(req: Requirement, base_resume_text: str,
                 candidate_domains: set = None, seniority: str = "mid",
                 confirmed: set = None, blocked: set = None,
                 candidate_years: int = 0,
                 maximum_ats_mode: bool = False) -> FitVerdict:
    term = req.term
    tl = term.lower().strip()
    low = base_resume_text.lower()

    # 0a. Vault overrides (strongest signal β€” the user's own decisions)
    if blocked and tl in blocked:
        return FitVerdict(term, req.category, "blocked", "block",
                          "user-marked do-not-use", req.importance,
                          req.recommended_placement)
    if confirmed and tl in confirmed:
        return FitVerdict(term, req.category, "explicit", "include",
                          "user-confirmed", req.importance,
                          req.recommended_placement)

    # 0. Hard blocks first β€” Maximum ATS Mode does NOT relax these (degrees,
    # certs/licenses, regulated credentials, seniority jumps, deep specialties,
    # hands-on engineering). The honesty boundary is absolute.
    act, why = _is_blocked(term, req.category, seniority, candidate_years)
    if act == "block":
        return FitVerdict(term, req.category, "blocked", "block", why,
                          req.importance, req.recommended_placement)
    if act == "ask_user":
        return FitVerdict(term, req.category, "risky", "ask_user", why,
                          req.importance, req.recommended_placement)

    # 0b. Maximum ATS Mode: normal PM/Product/AI/SaaS/B2B/agile vocabulary is
    # treated as user-confirmed (the base resume is incomplete; these are
    # interview-supportable craft terms). Runs AFTER hard blocks so credentials/
    # seniority/engineering still win, and only for the curated safe set.
    if maximum_ats_mode and _is_max_ats_safe(term):
        return FitVerdict(term, req.category, "explicit", "include",
                          "Maximum ATS Mode: user-confirmed PM/AI craft term",
                          req.importance, req.recommended_placement)

    # 1. Explicit β€” already in the resume
    if _kw_in_text(term, low) or any(_kw_in_text(a, low) for a in (req.aliases or [])):
        return FitVerdict(term, req.category, "explicit", "include",
                          "present in resume", req.importance, req.recommended_placement)

    # 2. Adjacent β€” related evidence present in the resume
    related = _RELATED.get(tl, [])
    if any(_kw_in_text(r, low) for r in related):
        return FitVerdict(term, req.category, "adjacent", "include",
                          "related experience present", req.importance,
                          req.recommended_placement)

    # 3. Specialized domain knowledge outside evidenced domains β†’ risky (include
    #    carefully, flag for review). NOT in the candidate's domain β†’ review.
    if tl in _SPECIALIZED_DOMAIN:
        in_domain = candidate_domains and tl in candidate_domains
        if in_domain:
            return FitVerdict(term, req.category, "adjacent", "include",
                              "within candidate domain", req.importance,
                              req.recommended_placement)
        return FitVerdict(term, req.category, "risky", "include_carefully",
                          "specialized domain term β€” review before applying",
                          req.importance, req.recommended_placement)

    # 4. PM-universal craft / soft skills / standard tools+methods β†’ PLAUSIBLE.
    #    The aggressive-expansion default: reasonable for the candidate's role.
    if (req.category in ("hard_skill", "tool", "responsibility", "soft_skill", "domain")
            or tl in _PM_CORE_TRANSFERABLE):
        return FitVerdict(term, req.category, "plausible", "include",
                          "plausible for candidate role/seniority", req.importance,
                          req.recommended_placement)

    # 5. Seniority signals / titles that match level β†’ plausible; else handled above
    if req.category in ("seniority",):
        return FitVerdict(term, req.category, "plausible", "include",
                          "matches candidate level", req.importance,
                          req.recommended_placement)

    # Fallback: treat as plausible (aggressive mode) unless it was blocked above.
    return FitVerdict(term, req.category, "plausible", "include",
                      "default-include (aggressive plausible match)",
                      req.importance, req.recommended_placement)


def classify_all_fit(req: JDRequirements, base_resume_text: str,
                     candidate_domains: set = None,
                     use_vault: bool = True,
                     maximum_ats_mode: bool = False,
                     extra_confirmed: set = None) -> List[FitVerdict]:
    seniority = infer_seniority(base_resume_text)
    cand_years = _candidate_years(base_resume_text)
    confirmed = blocked = None
    if use_vault:
        try:
            from .candidate_vault import user_confirmed_terms, user_blocked_terms
            confirmed = user_confirmed_terms()
            blocked = user_blocked_terms()
        except Exception:
            pass
    # Per-request user confirmations (e.g. extension "confirm these terms") merge
    # on top of the persisted vault, but never override a vault block.
    if extra_confirmed:
        confirmed = set(confirmed or set()) | {t.lower().strip() for t in extra_confirmed}
        if blocked:
            confirmed -= blocked
    return [classify_fit(r, base_resume_text, candidate_domains, seniority,
                         confirmed=confirmed, blocked=blocked,
                         candidate_years=cand_years,
                         maximum_ats_mode=maximum_ats_mode)
            for r in req.all_requirements()]


# ── Risk severity (AUTO_AGGRESSIVE mode) ─────────────────────────────────────

LOW = "LOW_RISK_AUTO_INCLUDED"
MEDIUM = "MEDIUM_RISK_REVIEW_RECOMMENDED"
HIGH = "HIGH_RISK_NEEDS_CONFIRMATION"
BLOCKED = "BLOCKED_DO_NOT_INCLUDE"


def severity(v: FitVerdict) -> str:
    """Map a fit verdict to a risk severity level.

    LOW    β€” common PM/product/business/analytics/agile terms, normal tools,
             normal responsibilities, soft skills, anything already in the resume.
    MEDIUM β€” domain/industry terms & plausible tools not in the base resume
             (auto-include but flag for review).
    HIGH   β€” specialized platforms (SIEM/SOAR), compliance/regulatory terms,
             engineering hard skills, seniority-sensitive, credential-required
             (only used with explicit user confirmation).
    BLOCKEDβ€” degrees/certs/licenses/fakes/seniority jumps/deep specialties.
    """
    if v.action == "block":
        return BLOCKED
    if v.action in ("ask_user", "include_carefully"):
        # specialized-domain / engineering / credential / regulatory β†’ confirm
        return HIGH
    # action == include
    if v.fit_status == "explicit":
        return LOW
    if v.category == "domain":
        return MEDIUM
    if v.fit_status == "adjacent":
        return MEDIUM
    return LOW   # plausible common PM craft / tools / methods / soft skills


def by_severity(verdicts: List[FitVerdict], level: str) -> List[FitVerdict]:
    return [v for v in verdicts if severity(v) == level]


# ── Convenience splits for the tailoring engine ──────────────────────────────

def includable(verdicts: List[FitVerdict]) -> List[FitVerdict]:
    """explicit + plausible + adjacent β†’ safe to include in the resume."""
    return [v for v in verdicts if v.action == "include"]


def auto_terms(verdicts: List[FitVerdict]) -> List[FitVerdict]:
    """AUTO_AGGRESSIVE: LOW + MEDIUM are auto-included (MEDIUM flags review)."""
    return [v for v in verdicts if severity(v) in (LOW, MEDIUM)]


def high_risk_terms(verdicts: List[FitVerdict]) -> List[FitVerdict]:
    """HIGH β€” only included with user confirmation; gate the job if needed for 90."""
    return [v for v in verdicts if severity(v) == HIGH]


def review_terms(verdicts: List[FitVerdict]) -> List[FitVerdict]:
    """risky β†’ include only when needed for 90%, flagged for user review."""
    return [v for v in verdicts if v.action == "include_carefully"]


def ask_user_terms(verdicts: List[FitVerdict]) -> List[FitVerdict]:
    return [v for v in verdicts if v.action == "ask_user"]


def blocked_terms(verdicts: List[FitVerdict]) -> List[FitVerdict]:
    return [v for v in verdicts if v.action == "block"]