Report AUCs in Delphi2M format
This commit is contained in:
183
delphi2m_auc_report.py
Normal file
183
delphi2m_auc_report.py
Normal file
@@ -0,0 +1,183 @@
|
||||
"""Build Delphi2M-style sex-specific AUC reports.
|
||||
|
||||
The Delphi2M evaluation code uses 0.1 years for the no-gap evaluation. The
|
||||
published report displays that point as 0 months, while retaining the actual
|
||||
0.1-year evaluation period in this project's report output.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Dict, Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
|
||||
DEFAULT_DELPHI2M_PERIODS_YEARS = (0.1, 1.0, 5.0, 10.0)
|
||||
|
||||
_CHAPTER_SHORT_NAMES = {
|
||||
"I": "I. Infectious Diseases",
|
||||
"II": "II. Neoplasms",
|
||||
"III": "III. Blood & Immune Disorders",
|
||||
"IV": "IV. Metabolic Diseases",
|
||||
"V": "V. Mental Disorders",
|
||||
"VI": "VI. Nervous System Diseases",
|
||||
"VII": "VII. Eye Diseases",
|
||||
"VIII": "VIII. Ear Diseases",
|
||||
"IX": "IX. Circulatory Diseases",
|
||||
"X": "X. Respiratory Diseases",
|
||||
"XI": "XI. Digestive Diseases",
|
||||
"XII": "XII. Skin Diseases",
|
||||
"XIII": "XIII. Musculoskeletal Diseases",
|
||||
"XIV": "XIV. Genitourinary Diseases",
|
||||
"XV": "XV. Pregnancy & Childbirth",
|
||||
"XVI": "XVI. Perinatal Conditions",
|
||||
"XVII": "XVII. Congenital Abnormalities",
|
||||
"XVIII": "XVIII. Symptoms & Signs",
|
||||
"XIX": "XIX. Injury & Poisoning",
|
||||
"XX": "XX. External Causes",
|
||||
"XXI": "XXI. Health Services",
|
||||
"XXII": "XXII. Special Purposes",
|
||||
"Death": "Death",
|
||||
"Unmapped": "Unmapped",
|
||||
}
|
||||
|
||||
|
||||
def _is_no_gap(period_years: float) -> bool:
|
||||
return bool(np.isclose(float(period_years), 0.1, rtol=0.0, atol=1e-8))
|
||||
|
||||
|
||||
def _canonical_period_years(period_years: float) -> float:
|
||||
value = float(period_years)
|
||||
for canonical in DEFAULT_DELPHI2M_PERIODS_YEARS:
|
||||
if np.isclose(value, canonical, rtol=0.0, atol=1e-6):
|
||||
return float(canonical)
|
||||
return value
|
||||
|
||||
|
||||
def _gap_months(period_years: float) -> int:
|
||||
if _is_no_gap(period_years):
|
||||
return 0
|
||||
return int(round(float(period_years) * 12.0))
|
||||
|
||||
|
||||
def _gap_label(period_years: float) -> str:
|
||||
if _is_no_gap(period_years):
|
||||
return "No gap"
|
||||
value = float(period_years)
|
||||
value_text = f"{value:g}"
|
||||
unit = "year" if np.isclose(value, 1.0) else "years"
|
||||
return f"{value_text} {unit}"
|
||||
|
||||
|
||||
def _load_chapter_by_code(
|
||||
chapter_mapping_path: Optional[str | Path] = None,
|
||||
) -> Dict[str, str]:
|
||||
if chapter_mapping_path is None:
|
||||
chapter_mapping_path = Path(__file__).with_name(
|
||||
"icd10_chapter_organ_mapping.csv"
|
||||
)
|
||||
path = Path(chapter_mapping_path)
|
||||
if not path.exists():
|
||||
return {}
|
||||
|
||||
mapping = pd.read_csv(
|
||||
path,
|
||||
usecols=["code", "icd10_chapter"],
|
||||
dtype={"code": str, "icd10_chapter": str},
|
||||
)
|
||||
mapping["code"] = mapping["code"].str.strip()
|
||||
mapping["chapter"] = (
|
||||
mapping["icd10_chapter"]
|
||||
.str.strip()
|
||||
.map(_CHAPTER_SHORT_NAMES)
|
||||
.fillna("Unmapped")
|
||||
)
|
||||
return dict(zip(mapping["code"], mapping["chapter"]))
|
||||
|
||||
|
||||
def build_delphi2m_auc_report(
|
||||
df_unpooled: pd.DataFrame,
|
||||
*,
|
||||
period_col: str,
|
||||
chapter_mapping_path: Optional[str | Path] = None,
|
||||
) -> pd.DataFrame:
|
||||
"""Aggregate age strata by sex and return a Delphi2M-style AUC report.
|
||||
|
||||
Required input columns are ``token``, ``label_code``, ``sex``,
|
||||
``auc_delong``, and the supplied ``period_col`` (``offset`` or
|
||||
``horizon``). The output begins with the five columns used by Delphi2M
|
||||
Fig. 2e and then records the actual evaluation period and ICD-10 code.
|
||||
"""
|
||||
required = {"token", "label_code", "sex", "auc_delong", period_col}
|
||||
missing = sorted(required - set(df_unpooled.columns))
|
||||
if missing:
|
||||
raise ValueError(
|
||||
"Cannot build Delphi2M AUC report; missing columns: "
|
||||
+ ", ".join(missing)
|
||||
)
|
||||
|
||||
source = df_unpooled.loc[
|
||||
:,
|
||||
["token", "label_code", "sex", "auc_delong", period_col],
|
||||
].copy()
|
||||
source["sex"] = source["sex"].astype(str).str.strip().str.lower()
|
||||
source = source[source["sex"].isin(["female", "male"])]
|
||||
source["auc_delong"] = pd.to_numeric(
|
||||
source["auc_delong"], errors="coerce"
|
||||
)
|
||||
source[period_col] = pd.to_numeric(source[period_col], errors="coerce")
|
||||
source = source.dropna(subset=[period_col, "auc_delong"])
|
||||
source[period_col] = source[period_col].map(_canonical_period_years)
|
||||
|
||||
if source.empty:
|
||||
raise ValueError("Cannot build Delphi2M AUC report from empty AUC data.")
|
||||
|
||||
grouped = (
|
||||
source.groupby(
|
||||
["token", "label_code", period_col, "sex"],
|
||||
dropna=False,
|
||||
as_index=False,
|
||||
)
|
||||
.agg(auc=("auc_delong", "mean"))
|
||||
)
|
||||
report = (
|
||||
grouped.pivot(
|
||||
index=["token", "label_code", period_col],
|
||||
columns="sex",
|
||||
values="auc",
|
||||
)
|
||||
.reset_index()
|
||||
.rename_axis(columns=None)
|
||||
.rename(columns={"female": "Female", "male": "Male"})
|
||||
)
|
||||
for col in ["Female", "Male"]:
|
||||
if col not in report.columns:
|
||||
report[col] = np.nan
|
||||
|
||||
chapter_by_code = _load_chapter_by_code(chapter_mapping_path)
|
||||
report["chapter"] = (
|
||||
report["label_code"].astype(str).map(chapter_by_code).fillna("Unmapped")
|
||||
)
|
||||
report["Gap, months"] = report[period_col].map(_gap_months).astype("Int64")
|
||||
report["Gap label"] = report[period_col].map(_gap_label)
|
||||
report["icd10"] = pd.to_numeric(report["token"], errors="coerce").astype(
|
||||
"Int64"
|
||||
)
|
||||
|
||||
report = report.sort_values(
|
||||
["icd10", period_col], kind="stable", ignore_index=True
|
||||
)
|
||||
return report.loc[
|
||||
:,
|
||||
[
|
||||
"Gap, months",
|
||||
"chapter",
|
||||
"icd10",
|
||||
"Female",
|
||||
"Male",
|
||||
period_col,
|
||||
"Gap label",
|
||||
"label_code",
|
||||
],
|
||||
]
|
||||
Reference in New Issue
Block a user