File size: 3,646 Bytes
9bf4a3d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
"""
Provider evaluation harness (spec #6).

Runs the SAME jobs through each configured provider (Kimi / NVIDIA / ... / Stub)
with the same base resume, JD set, and automation mode, then prints a per-job
table and a per-provider summary so you can decide which model performs best.

Usage:
    PYTHONPATH=. python scripts/evaluate_model_providers.py            # all JD fixtures
    PYTHONPATH=. python scripts/evaluate_model_providers.py --jobs 4   # first 4 JDs

This is the script to run with REAL provider keys (Kimi 2.6 / NVIDIA) — stub-only
runs validate plumbing, NOT production quality.
"""

import os
import sys
import io
import json
import argparse

sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..")))
if hasattr(sys.stdout, "buffer"):
    sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8", errors="replace")

from src.provider_eval import (
    load_base_resume, load_jd_fixtures, run_provider_matrix, summarize,
)
from src.providers import build_provider_chain
from src.llm_client import LLMClient


def _fmt_row(r):
    return (f"{r['provider'][:16]:<16} {r['job'][:18]:<18} "
            f"int={r['internal']:>3} ind={r['independent']:>3} "
            f"read={r['readability']:>3} {r['status'][:22]:<22} "
            f"rep={r['repair_attempts']} risk={r['risk_terms']} "
            f"sch_err={r['schema_errors']} dl={'Y' if r['download_allowed'] else 'n'} "
            f"{r['runtime_s']}s")


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--jobs", type=int, default=None, help="limit number of JD fixtures")
    args = ap.parse_args()

    base = load_base_resume()
    if base is None:
        print("ERROR: no base resume. Add data/resume/resume.pdf or data/resume/_parsed.json.")
        sys.exit(1)

    jobs = load_jd_fixtures(limit=args.jobs)
    if not jobs:
        print("ERROR: no JD fixtures in tests/fixtures/jds/.")
        sys.exit(1)

    llm = LLMClient.__new__(LLMClient)
    providers = build_provider_chain(llm)
    real = [p for p in providers if getattr(p, "cfg", {}).get("api_key")]
    print(f"Providers under test: {[p.name for p in providers]}")
    if not real:
        print("\n*** NO PROVIDER API KEYS DETECTED — running StubProvider only. ***")
        print("*** This validates plumbing ONLY. Run again with Kimi/NVIDIA keys")
        print("*** in .env for a real production-quality comparison. ***\n")

    out_root = os.path.join("data", "output", "eval")
    rows = run_provider_matrix(providers, base, jobs, out_root, llm)

    print("\n=== PER-JOB RESULTS ===")
    print("Provider         Job                int ind read Status                 rep risk sch dl  time")
    for r in rows:
        print(_fmt_row(r))

    print("\n=== PER-PROVIDER SUMMARY ===")
    summ = summarize(rows)
    print(f"{'Provider':<16} {'Ready%':>6} {'AvgInd':>7} {'AvgInt':>7} "
          f"{'AvgRep':>7} {'Sch%':>5} {'Err%':>5} {'Risk':>5} {'Runtime':>8}  Recommendation")
    for s in summ:
        print(f"{s['provider'][:16]:<16} {s['ready_rate']:>6} {s['avg_independent']:>7} "
              f"{s['avg_internal']:>7} {s['avg_repairs']:>7} {s['schema_error_rate']:>5} "
              f"{s['provider_error_rate']:>5} {s['risk_overuse']:>5} "
              f"{s['avg_runtime_s']:>7}s  {s['recommendation']}")

    os.makedirs(out_root, exist_ok=True)
    out_json = os.path.join(out_root, "provider_eval.json")
    with open(out_json, "w", encoding="utf-8") as f:
        json.dump({"rows": rows, "summary": summ}, f, ensure_ascii=False, indent=2)
    print(f"\nSaved full results -> {out_json}")


if __name__ == "__main__":
    main()