File size: 2,379 Bytes
54bb6a3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
"""Baseline fixed-horizon risk classifier."""

from __future__ import annotations

from dataclasses import dataclass

import numpy as np
import pandas as pd

try:
    from sklearn.impute import SimpleImputer
    from sklearn.linear_model import LogisticRegression
    from sklearn.pipeline import Pipeline
    from sklearn.preprocessing import StandardScaler
except ImportError:  # pragma: no cover
    SimpleImputer = LogisticRegression = Pipeline = StandardScaler = None


@dataclass
class RiskClassifier:
    feature_columns: list[str]

    def __post_init__(self) -> None:
        self.pipeline = None
        if Pipeline is not None:
            self.pipeline = Pipeline(
                steps=[
                    ("imputer", SimpleImputer(strategy="median")),
                    ("scaler", StandardScaler()),
                    ("model", LogisticRegression(max_iter=1000, class_weight="balanced")),
                ]
            )

    def fit(self, frame: pd.DataFrame, target_col: str = "event") -> "RiskClassifier":
        if self.pipeline is not None:
            self.pipeline.fit(frame[self.feature_columns], frame[target_col])
            return self

        x = frame[self.feature_columns].astype(float)
        y = frame[target_col].astype(int)
        self.medians_ = x.median()
        x = x.fillna(self.medians_)
        self.means_ = x.mean()
        self.stds_ = x.std(ddof=0).replace(0, 1.0)
        z = (x - self.means_) / self.stds_
        pos = z[y == 1].mean()
        neg = z[y == 0].mean()
        self.direction_ = (pos - neg).fillna(0.0)
        self.intercept_ = -float((pos + neg).fillna(0.0).dot(self.direction_) / 2.0)
        return self

    def predict_risk(self, frame: pd.DataFrame) -> pd.Series:
        if self.pipeline is not None:
            probabilities = self.pipeline.predict_proba(frame[self.feature_columns])[:, 1]
            return pd.Series(probabilities, index=frame.index, name="risk")

        if not hasattr(self, "direction_"):
            raise RuntimeError("RiskClassifier must be fitted before predict_risk")
        x = frame[self.feature_columns].astype(float).fillna(self.medians_)
        z = (x - self.means_) / self.stds_
        logits = z.dot(self.direction_) + self.intercept_
        probabilities = 1.0 / (1.0 + np.exp(-logits))
        return pd.Series(probabilities, index=frame.index, name="risk")