Spaces:
Sleeping
Sleeping
File size: 4,383 Bytes
b0fa475 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 | """Review filter helpers for the Scrub replacement workflow.
The filters are intentionally presentation-oriented. They help the user focus on
specific rows, but they should not silently remove hidden rows from the export
workflow unless the UI explicitly implements safe merge-back behaviour.
"""
from __future__ import annotations
from typing import Any, Iterable, Mapping
FILTER_SHOW_ALL = "Toon alles"
FILTER_NEEDS_REVIEW = "Alleen controle nodig"
FILTER_LEGAL_REFERENCES = "Alleen juridische referenties"
FILTER_NAMES_ADDRESSES = "Alleen namen/adressen"
FILTER_LOW_CONFIDENCE = "Alleen lage zekerheid"
REVIEW_FILTER_OPTIONS = [
FILTER_SHOW_ALL,
FILTER_NEEDS_REVIEW,
FILTER_LEGAL_REFERENCES,
FILTER_NAMES_ADDRESSES,
FILTER_LOW_CONFIDENCE,
]
LEGAL_REFERENCE_ENTITY_TYPES = {
"NL_LEGAL_CASE_NUMBER",
"NL_ROLNUMMER",
"NL_REKESTNUMMER",
"NL_PARKETNUMMER",
"NL_DOSSIER_NUMBER",
"NL_CLIENT_NUMBER",
"NL_CJIB_NUMBER",
"NL_POLICE_REPORT_NUMBER",
"NL_INSURANCE_CLAIM_NUMBER",
"NL_INCIDENT_NUMBER",
"NL_CLAIM_NUMBER",
"NL_OTHER_REFERENCE",
"NL_CLIENT_REFERENCE",
"NL_CASE_REFERENCE",
"NL_INTERNAL_REFERENCE",
"NL_CONTEXTUAL_REFERENCE",
"NL_INVOICE_NUMBER",
"NL_ORDER_OR_CONTRACT_NUMBER",
"NL_SCHOOL_REFERENCE",
"NL_CHILD_PROTECTION_REFERENCE",
"NL_EMPLOYMENT_REFERENCE",
"NL_INSURANCE_REFERENCE",
"NL_HEALTHCARE_REFERENCE",
"NL_POLICE_REFERENCE",
"NL_IMMIGRATION_REFERENCE",
"NL_MUNICIPAL_REFERENCE",
"NL_REAL_ESTATE_REFERENCE",
"NL_VEHICLE_REFERENCE",
"NL_OBJECT_REFERENCE",
"NL_SUSPICIOUS_REFERENCE_CANDIDATE",
"NL_POSSIBLE_LICENSE_PLATE",
"NL_ECLI",
"NL_KVK_NUMBER",
"NL_VAT_NUMBER",
"NL_BIG_NUMBER",
}
NAME_ADDRESS_ENTITY_TYPES = {
"PERSON",
"NL_LEGAL_PARTY_NAME",
"LOCATION",
"NL_ADDRESS",
"NL_POSTCODE",
"ORGANIZATION",
"NL_COURT_OR_AUTHORITY",
}
def _cell(row: Mapping[str, Any], key: str, default: Any = "") -> Any:
value = row.get(key, default)
return default if value is None else value
def _normalised_entity_type(row: Mapping[str, Any]) -> str:
return str(_cell(row, "entity_type", "")).strip().upper()
def _normalised_status(row: Mapping[str, Any]) -> str:
status = str(_cell(row, "review_status", "")).strip().lower()
label = str(_cell(row, "review_status_label", "")).strip().lower()
if status:
return status
if label == "controle nodig":
return "needs_review"
if label == "automatisch vervangen":
return "auto_detected"
if label == "handmatig toegevoegd":
return "manual"
if label == "onthouden vervanging":
return "remembered"
return ""
def _score(row: Mapping[str, Any]) -> float | None:
raw = _cell(row, "score", None)
if raw in (None, ""):
return None
try:
return float(raw)
except Exception:
return None
def row_matches_review_filter(row: Mapping[str, Any], filter_label: str) -> bool:
if filter_label == FILTER_SHOW_ALL:
return True
if filter_label == FILTER_NEEDS_REVIEW:
return _normalised_status(row) == "needs_review"
entity_type = _normalised_entity_type(row)
if filter_label == FILTER_LEGAL_REFERENCES:
return entity_type in LEGAL_REFERENCE_ENTITY_TYPES
if filter_label == FILTER_NAMES_ADDRESSES:
return entity_type in NAME_ADDRESS_ENTITY_TYPES
if filter_label == FILTER_LOW_CONFIDENCE:
confidence = str(_cell(row, "confidence", "")).strip().lower()
score = _score(row)
return confidence == "laag" or (score is not None and score < 0.60)
return True
def filter_review_records(records: Iterable[Mapping[str, Any]], filter_label: str) -> list[Mapping[str, Any]]:
return [row for row in records if row_matches_review_filter(row, filter_label)]
def filter_review_dataframe(df, filter_label: str):
"""Filter a pandas-like DataFrame without importing pandas at module import.
Tests use the record-level function. The Streamlit app can use this helper
with a pandas DataFrame.
"""
if filter_label == FILTER_SHOW_ALL:
return df
if df is None or len(df) == 0:
return df
mask = [row_matches_review_filter(row, filter_label) for _, row in df.iterrows()]
return df.loc[mask].reset_index(drop=True)
|