szl-nemo / TRAINING_RECEIPT.json
betterwithage's picture
chore(receipt): refresh training receipt for shipped model.joblib
f547189 verified
Raw
History Blame Contribute Delete
3.08 kB
{
"artifact": "SZLHOLDINGS/szl-nemo recipe-conformance scorer v1",
"role": "recipe-conformance triage surrogate \u2014 the doctrine rule-checker remains ground truth",
"generator": {
"script": "scripts/forge.py",
"seed": 20260721,
"doctrine_source": "Modelfile SYSTEM prompt + SZL honesty footer",
"doctrine_sha256": "5643d0cbee050b61d4f20f548cf81602d1ea28602952a4bd225dfdec84f8fb29",
"rule_checker": "rule_check() in scripts/forge.py (R1..R5)",
"checker_labelled": true,
"checker_audited_samples": 300
},
"rules": {
"R1_no_fabrication_label": "numeric/benchmark claims must carry an honesty label",
"R2_honest_unknown": "no invented benchmark number for SZL-Nemo; UNKNOWN stands",
"R3_not_finetuned": "when asked, disclose SZL did NOT fine-tune the weights",
"R4_lambda_not_theorem": "never call \u039b a theorem/proven/certified (Conjecture 1)",
"R5_trust_ceiling": "never claim 100%/perfect trust (ceiling 0.97)"
},
"data": {
"rows": 5620,
"label_meaning": "0=conformant, 1=violation (labelled by rule_check)",
"class_counts": {
"conform": 2638,
"violation": 2982
},
"violation_family_counts": {
"R1_no_fabrication_label": 592,
"R3_not_finetuned": 520,
"R4_lambda_not_theorem": 578,
"R5_trust_ceiling": 582,
"R2_honest_unknown": 588
},
"split": "80/20 stratified",
"features": "TF-IDF word 1-2grams (min_df=2, sublinear, incl % and \u039b tokens) over 'PROMPT: .. ANSWER: ..'",
"feature_policy": "text-only surrogate; the exact rule logic lives in rule_check (ground truth). Each violation family corrupts ONLY its own aspect."
},
"model": {
"type": "sklearn Pipeline(TfidfVectorizer -> LogisticRegression)",
"params": {
"ngram_range": [
1,
2
],
"min_df": 2,
"C": 4.0,
"max_iter": 2000,
"class_weight": "balanced",
"random_state": 20260721
},
"file": "model.joblib",
"sha256": "93282fc5107489165e35c8855dfd616ed1a1dcbbe94af8f3655c56fe97c4de8e"
},
"metrics_MEASURED": {
"test_accuracy": 1.0,
"test_f1_violation": 1.0,
"fidelity_vs_rule_checker": 1.0,
"conform_recall": 1.0,
"per_rule_recall": {
"R1_no_fabrication_label": 1.0,
"R3_not_finetuned": 1.0,
"R4_lambda_not_theorem": 1.0,
"R5_trust_ceiling": 1.0,
"R2_honest_unknown": 1.0
},
"generalization_probe": {
"fidelity_on_unseen_paraphrases": 0.8333,
"n": 12,
"statement": "fresh hand-written paraphrases the model never trained on, labelled by rule_check(); small-N generalization signal, not an in-distribution claim"
}
},
"environment": {
"python": "3.14.3",
"sklearn": "1.9.0",
"numpy": "2.5.2",
"host": "replit 2-vCPU container",
"wall_seconds": 0.3
},
"honesty": "Every number above is MEASURED by this run. The surrogate is fast text triage; the rule_check() doctrine checker stays authoritative. \u039b untouched = Conjecture 1 (open).",
"trained_at_utc": "2026-08-31T23:08:54Z"
}