load.py

38 lines · 1,597 bytes · view raw

"""Load raw CMS CSVs with explicit string dtypes, then coerce numerics.

Returns both frames in a consistent shape ready for `clean.py`.
"""
import pandas as pd
from .config import DATA_RAW, INPATIENT_NUMERIC, OUTPATIENT_NUMERIC, COLUMN_RENAME


def load_inpatient() -> pd.DataFrame:
    """Read the raw inpatient CSV. All columns as str first."""
    df = pd.read_csv(DATA_RAW / "inpatient_2023.csv", dtype=str)
    df = df.rename(columns=COLUMN_RENAME)
    # Resolve the duplicate rename key for medicare_payment (inpatient only has
    # Avg_Mdcr_Pymt_Amt). Both datasets rename to avg_medicare_payment, which
    # is intentional.
    for col in INPATIENT_NUMERIC:
        if col in df.columns:
            df[col] = pd.to_numeric(df[col], errors="coerce")
    df = df.dropna(subset=["total_discharges", "avg_total_payment", "avg_medicare_payment"])
    return df.reset_index(drop=True)


def load_outpatient() -> pd.DataFrame:
    """Read the raw outpatient CSV. All columns as str first, numerics coerced."""
    df = pd.read_csv(DATA_RAW / "outpatient_2023.csv", dtype=str)
    df = df.rename(columns=COLUMN_RENAME)
    for col in OUTPATIENT_NUMERIC:
        if col in df.columns:
            df[col] = pd.to_numeric(df[col], errors="coerce")
    # Note: we DO NOT drop suppressed rows here. `clean.py` decides per-analysis.
    return df.reset_index(drop=True)


if __name__ == "__main__":  # pragma: no cover
    inp = load_inpatient()
    out = load_outpatient()
    print(f"inpatient: {inp.shape} | dtypes:\n{inp.dtypes}")
    print(f"\noutpatient: {out.shape} | dtypes:\n{out.dtypes}")