Readiness verdicts now ride additively on /api/upload and /api/confirm-visit: - core/worklist_readiness.py: ReadinessIndex — row-to-line membership resolved through the dedup group key (patient_id, device_type, DOS), never re-derived from a row's own plan_type (kills a reviewer-reproduced false green) - RecordOut gains plan_type / readiness_status / readiness_items (additive, default None); UploadResponse gains readiness_stats (reconciles with rows) - Minimal plan_type CSV mapping (client-supplied, lowercased+trimmed, never guessed) per the umbrella reactivation plan; full enforcement lands in P5 - Fixes pre-existing record_lookup miss for order-numbered CSVs; fallback fields now come from the dedup MergedShipment (latest-non-null, collected order numbers) so doc_state and the verdict grade from the same values - confirm-visit normalizes plan_type identically to the CSV path Verification: 132 tests green (108 baseline + 24 wiring/regression); adversarial 3-lens review + refutation pass (7 findings confirmed, all fixed, regression-locked); independent code-auditor verdict SHIP; Pi E2E 33/33. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
346 lines
13 KiB
Python
346 lines
13 KiB
Python
"""
|
||
CSV header normalization for Signal.
|
||
|
||
Maps messy supplier CSV exports to canonical ShipmentRecord fields.
|
||
Tolerates header drift, alternative column names, and common date formats.
|
||
"""
|
||
|
||
import csv
|
||
import io
|
||
import re
|
||
from datetime import date, datetime
|
||
from typing import Optional
|
||
|
||
import sys
|
||
from pathlib import Path
|
||
sys.path.insert(0, str(Path(__file__).parent.parent))
|
||
from core.coverage_calculator import ShipmentRecord
|
||
|
||
HEADER_MAP: dict[str, list[str]] = {
|
||
"patient_id": [
|
||
"patient_id", "patientid", "patient id", "pt_id", "pt id",
|
||
"mrn", "account_number", "account number", "account_no",
|
||
"patient_account", "acct_no", "acct no", "acct #", "acct#",
|
||
"id", "patient", "member_id", "member id",
|
||
"external patient ref", "external_patient_ref", "external ref",
|
||
],
|
||
"device_type": [
|
||
"device_type", "device type", "device", "devicetype",
|
||
"product_type", "product type", "product", "item",
|
||
"item_description", "item description", "hcpcs_description",
|
||
"hcpcs description", "description", "product_name",
|
||
"dme", "dme description", "dme_description", "dme desc",
|
||
"equipment", "equipment description",
|
||
],
|
||
"shipment_date": [
|
||
"shipment_date", "shipment date", "ship_date", "ship date",
|
||
"dispense_date", "dispense date", "dispensedate",
|
||
"service_date", "service date",
|
||
"order_date", "order date", "date_of_service", "dos",
|
||
"fill_date", "fill date", "last_ship_date", "last ship date",
|
||
],
|
||
"quantity": [
|
||
"quantity", "qty", "units", "count", "qty_dispensed",
|
||
"units_dispensed", "quantity_dispensed", "qty_shipped",
|
||
],
|
||
"payer": [
|
||
"payer", "insurance", "insurance_name", "insurance name",
|
||
"plan", "plan_name", "plan name", "payer_name", "payer name",
|
||
"primary_payer", "primary payer", "ins_name", "carrier",
|
||
],
|
||
# Client-mapped plan type — grading-critical, NEVER derived from the payer
|
||
# name. Bare "plan"/"plan_name" stay payer aliases (a "Plan" column in the
|
||
# wild is a payer name); stealing them would silently re-map existing CSVs.
|
||
"plan_type": [
|
||
"plan_type", "plan type", "plan category", "plan_category",
|
||
"coverage type", "coverage_type", "insurance_type", "insurance type",
|
||
],
|
||
"component": [
|
||
"component", "item_type", "component_type", "type", "supply_type",
|
||
],
|
||
"csv_visit_date": [
|
||
"visit_date", "visit date", "qualifying_visit_date", "qualifying visit date",
|
||
"last_visit_date", "last visit date", "face_to_face_date", "f2f_date",
|
||
"encounter_date", "encounter date", "physician_visit_date",
|
||
],
|
||
"csv_swo_status": [
|
||
"swo_status", "swo status", "swo", "standing_written_order",
|
||
"standing written order", "order_status", "order status",
|
||
],
|
||
"csv_pecos_verified": [
|
||
"pecos_verified", "pecos verified", "pecos", "pecos_status",
|
||
"enrollment_verified", "enrollment verified",
|
||
],
|
||
"csv_pa_status": [
|
||
"pa_status", "pa status", "prior_auth_status", "prior auth status",
|
||
"prior_authorization_status", "auth_status", "pa", "authorization",
|
||
],
|
||
"csv_diagnosis_on_file": [
|
||
"diagnosis_on_file", "diagnosis on file", "diagnosis", "dx_on_file",
|
||
"dx on file", "icd_on_file", "icd on file",
|
||
],
|
||
"csv_transfer_from": [
|
||
"transfer_from", "transfer from", "previous_supplier", "previous supplier",
|
||
"prior_supplier", "prior supplier", "transfer_status", "transferred_from",
|
||
],
|
||
"order_number": [
|
||
"order_number", "order number", "order_no", "order no", "order#",
|
||
"claim_number", "claim number", "brightree_order", "order_id",
|
||
"rx_number", "rx number", "rx#", "dispense_number",
|
||
],
|
||
"hcpcs": [
|
||
"hcpcs", "hcpcs_code", "hcpcs code", "procedure_code", "procedure code",
|
||
"billing_code", "billing code", "item_code", "item code", "hcpc",
|
||
],
|
||
}
|
||
|
||
DEVICE_MAP: dict[str, str] = {
|
||
# Dexcom G7 variants
|
||
"dexcom g7": "dexcom_g7",
|
||
"dexcom_g7": "dexcom_g7",
|
||
"dexcomg7": "dexcom_g7",
|
||
"dexcom g-7": "dexcom_g7",
|
||
"g7": "dexcom_g7",
|
||
"g7 sensor": "dexcom_g7",
|
||
"dexcom g7 sensor": "dexcom_g7",
|
||
"dexcom g7 cgm": "dexcom_g7",
|
||
# Dexcom G6 variants
|
||
"dexcom g6": "dexcom_g6",
|
||
"dexcom_g6": "dexcom_g6",
|
||
"dexcomg6": "dexcom_g6",
|
||
"dexcom g-6": "dexcom_g6",
|
||
"g6": "dexcom_g6",
|
||
"g6 sensor": "dexcom_g6",
|
||
"dexcom g6 sensor": "dexcom_g6",
|
||
"dexcom g6 cgm": "dexcom_g6",
|
||
# FreeStyle Libre 2 variants
|
||
"freestyle libre 2": "freestyle_libre_2",
|
||
"freestyle_libre_2": "freestyle_libre_2",
|
||
"freestylelibre2": "freestyle_libre_2",
|
||
"libre 2": "freestyle_libre_2",
|
||
"libre2": "freestyle_libre_2",
|
||
"fsl2": "freestyle_libre_2",
|
||
"fs libre 2": "freestyle_libre_2",
|
||
"freestyle libre 2 sensor": "freestyle_libre_2",
|
||
# FreeStyle Libre 3 variants
|
||
"freestyle libre 3": "freestyle_libre_3",
|
||
"freestyle_libre_3": "freestyle_libre_3",
|
||
"freestylelibre3": "freestyle_libre_3",
|
||
"libre 3": "freestyle_libre_3",
|
||
"libre3": "freestyle_libre_3",
|
||
"fsl3": "freestyle_libre_3",
|
||
"fs libre 3": "freestyle_libre_3",
|
||
"freestyle libre 3 sensor": "freestyle_libre_3",
|
||
# FreeStyle Libre (no version — default to current)
|
||
"freestyle libre": "freestyle_libre_3",
|
||
"freestylelibre": "freestyle_libre_3",
|
||
"libre": "freestyle_libre_3",
|
||
"fsl": "freestyle_libre_3",
|
||
# Omnipod variants
|
||
"omnipod 5": "omnipod_5",
|
||
"omnipod_5": "omnipod_5",
|
||
"omnipod5": "omnipod_5",
|
||
"omnipod": "omnipod_5",
|
||
"op5": "omnipod_5",
|
||
# HCPCS codes used as device descriptions in some billing exports
|
||
"e2103": "dexcom_g7",
|
||
"a4239": "dexcom_g7",
|
||
"a4253": "freestyle_libre_3",
|
||
# Generic CGM descriptions
|
||
"cgm": "dexcom_g7",
|
||
"continuous glucose monitor": "dexcom_g7",
|
||
"continuous glucose monitoring": "dexcom_g7",
|
||
"glucose monitor": "dexcom_g7",
|
||
"glucose monitoring": "dexcom_g7",
|
||
"cgm sensor": "dexcom_g7",
|
||
"cgm supply": "dexcom_g7",
|
||
"cgm supplies": "dexcom_g7",
|
||
"dme cgm": "dexcom_g7",
|
||
}
|
||
|
||
DATE_FORMATS = [
|
||
"%Y-%m-%d",
|
||
"%m/%d/%Y",
|
||
"%m-%d-%Y",
|
||
"%d/%m/%Y",
|
||
"%m/%d/%y",
|
||
"%Y%m%d",
|
||
"%d-%b-%Y",
|
||
"%b %d, %Y",
|
||
"%B %d, %Y",
|
||
"%m/%d/%Y %H:%M:%S",
|
||
"%Y-%m-%dT%H:%M:%S",
|
||
]
|
||
|
||
|
||
def _normalize_key(s: str) -> str:
|
||
return s.strip().lower().replace("-", " ").replace("_", " ")
|
||
|
||
|
||
def _map_header(raw: str) -> Optional[str]:
|
||
key = _normalize_key(raw)
|
||
for canonical, aliases in HEADER_MAP.items():
|
||
if key in [_normalize_key(a) for a in aliases]:
|
||
return canonical
|
||
return None
|
||
|
||
|
||
def _map_header_with_confidence(raw: str) -> tuple[Optional[str], str]:
|
||
"""Return (canonical_field, confidence) where confidence is 'high' or 'inferred'."""
|
||
key = _normalize_key(raw)
|
||
for canonical, aliases in HEADER_MAP.items():
|
||
if key == _normalize_key(canonical):
|
||
return canonical, "high"
|
||
if key in [_normalize_key(a) for a in aliases]:
|
||
return canonical, "inferred"
|
||
return None, "unmapped"
|
||
|
||
|
||
def _parse_date(value: str) -> Optional[date]:
|
||
value = value.strip()
|
||
for fmt in DATE_FORMATS:
|
||
try:
|
||
return datetime.strptime(value, fmt).date()
|
||
except ValueError:
|
||
continue
|
||
return None
|
||
|
||
|
||
def _normalize_device(value: str) -> Optional[str]:
|
||
if not value or not value.strip():
|
||
return None
|
||
key = _normalize_key(value)
|
||
key_compact = re.sub(r"\s+", "", key)
|
||
# Exact / alias match
|
||
for alias, canonical in DEVICE_MAP.items():
|
||
alias_compact = re.sub(r"\s+", "", alias)
|
||
if key == alias or key_compact == alias_compact:
|
||
return canonical
|
||
# Fuzzy fallback — detect CGM family by keyword so real-world naming
|
||
# variants from billing exports don't silently drop patients.
|
||
if "dexcom" in key:
|
||
return "dexcom_g7" if "g7" in key_compact else "dexcom_g6"
|
||
if "libre" in key or "freestyle" in key:
|
||
return "freestyle_libre_3" if "3" in key else "freestyle_libre_2"
|
||
if "omnipod" in key:
|
||
return "omnipod_5"
|
||
if any(k in key for k in ("cgm", "continuous glucose", "glucose monitor", "e2103", "a4239", "a4253")):
|
||
return "dexcom_g7"
|
||
return None
|
||
|
||
|
||
def normalize_csv(text: str) -> tuple[list[ShipmentRecord], list[str], dict]:
|
||
"""
|
||
Parse raw CSV text and return (records, skipped_reasons, mapping_summary).
|
||
Tolerates header drift and normalizes device/payer/date values.
|
||
|
||
mapping_summary format:
|
||
{
|
||
"mapped": {canonical_field: {"raw_header": str, "confidence": "high"|"inferred"}},
|
||
"unmapped_columns": [str],
|
||
"required_missing": [str],
|
||
}
|
||
"""
|
||
reader = csv.DictReader(io.StringIO(text.strip().lstrip("")))
|
||
if not reader.fieldnames:
|
||
return [], ["No headers found in file"], {}
|
||
|
||
column_map: dict[str, str] = {}
|
||
mapping_detail: dict[str, dict] = {}
|
||
unmapped_columns: list[str] = []
|
||
|
||
for raw_header in reader.fieldnames:
|
||
canonical, confidence = _map_header_with_confidence(raw_header)
|
||
if canonical:
|
||
column_map[raw_header] = canonical
|
||
mapping_detail[canonical] = {"raw_header": raw_header, "confidence": confidence}
|
||
else:
|
||
unmapped_columns.append(raw_header)
|
||
|
||
required_fields = {"patient_id", "device_type", "shipment_date"}
|
||
required_missing = [f for f in required_fields if f not in mapping_detail]
|
||
|
||
mapping_summary = {
|
||
"mapped": mapping_detail,
|
||
"unmapped_columns": unmapped_columns,
|
||
"required_missing": required_missing,
|
||
}
|
||
|
||
records: list[ShipmentRecord] = []
|
||
skipped: list[str] = []
|
||
|
||
for i, row in enumerate(reader, start=2): # noqa: B007
|
||
mapped: dict[str, str] = {}
|
||
for raw_h, canonical in column_map.items():
|
||
mapped[canonical] = (row.get(raw_h) or "").strip()
|
||
|
||
patient_id = mapped.get("patient_id", "").strip()
|
||
if not patient_id:
|
||
skipped.append(f"Row {i}: missing patient_id")
|
||
continue
|
||
|
||
raw_device = mapped.get("device_type", "")
|
||
device_type = _normalize_device(raw_device)
|
||
if not device_type:
|
||
skipped.append(f"Row {i} ({patient_id}): unrecognized device '{raw_device}'")
|
||
continue
|
||
|
||
raw_date = mapped.get("shipment_date", "")
|
||
shipment_date = _parse_date(raw_date)
|
||
if not shipment_date:
|
||
skipped.append(f"Row {i} ({patient_id}): unparseable date '{raw_date}'")
|
||
continue
|
||
|
||
raw_qty = mapped.get("quantity", "1")
|
||
try:
|
||
quantity = max(1, int(float(raw_qty)))
|
||
except (ValueError, TypeError):
|
||
quantity = 1
|
||
|
||
# Raw payer string passes through untouched. coverage_calculator._normalize_payer
|
||
# is the single normalization point; the raw value survives for display.
|
||
payer = mapped.get("payer", "").strip()
|
||
component = (mapped.get("component", "sensor") or "sensor").lower().strip()
|
||
if component not in ("sensor", "transmitter", "pod"):
|
||
component = "sensor"
|
||
|
||
# Optional doc fields — only populated if the CSV contains these columns
|
||
csv_visit_date: Optional[date] = None
|
||
raw_visit = mapped.get("csv_visit_date", "")
|
||
if raw_visit:
|
||
csv_visit_date = _parse_date(raw_visit) # None if unparseable (non-error)
|
||
|
||
csv_swo_status = mapped.get("csv_swo_status") or None
|
||
csv_pecos_verified = mapped.get("csv_pecos_verified") or None
|
||
csv_pa_status = mapped.get("csv_pa_status") or None
|
||
csv_diagnosis_on_file = mapped.get("csv_diagnosis_on_file") or None
|
||
transfer_raw = mapped.get("csv_transfer_from", "").strip()
|
||
csv_transfer_from = transfer_raw if transfer_raw else None
|
||
order_number = mapped.get("order_number") or None
|
||
hcpcs = mapped.get("hcpcs") or None
|
||
|
||
# Client-supplied plan type, carried verbatim (lowercased and trimmed
|
||
# only — canonicalization of synonyms is a later, separate step).
|
||
# Anything the readiness engine does not recognize grades as
|
||
# "Plan Type Needed"; it is never guessed from the payer name.
|
||
plan_type = (mapped.get("plan_type") or "").strip().lower() or None
|
||
|
||
records.append(ShipmentRecord(
|
||
patient_id=patient_id,
|
||
device_type=device_type,
|
||
shipment_date=shipment_date,
|
||
quantity=quantity,
|
||
payer=payer,
|
||
component=component,
|
||
csv_visit_date=csv_visit_date,
|
||
csv_swo_status=csv_swo_status,
|
||
csv_pecos_verified=csv_pecos_verified,
|
||
csv_pa_status=csv_pa_status,
|
||
csv_diagnosis_on_file=csv_diagnosis_on_file,
|
||
csv_transfer_from=csv_transfer_from,
|
||
order_number=order_number,
|
||
hcpcs=hcpcs,
|
||
plan_type=plan_type,
|
||
))
|
||
|
||
return records, skipped, mapping_summary
|