{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":112899,"databundleVersionId":13449579,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-11T19:23:28.629621Z","iopub.execute_input":"2025-09-11T19:23:28.630032Z","iopub.status.idle":"2025-09-11T19:23:28.986136Z","shell.execute_reply.started":"2025-09-11T19:23:28.63Z","shell.execute_reply":"2025-09-11T19:23:28.985195Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Chest X-Ray Competition — EDA\n\n**Includes:**\n- Full EDA (sanity checks, imbalance, co-occurrence, correlations)\n- ViewPosition & ViewCategory analysis\n- Patient/Study counts (leakage risk)\n- \"No Finding\" exclusivity\n- Optional image size probe + sample grids\n- **Class imbalance tables + suggested class weights** (for BCE / Focal Loss)\n","metadata":{}},{"cell_type":"markdown","source":"## Parameters","metadata":{}},{"cell_type":"code","source":"# === Kaggle dataset handles ===\nDATASET_DIR = \"/kaggle/input/grand-xray-slam-division-a\"   \nTRAIN_CSV   = f\"{DATASET_DIR}/train1.csv\"         \nIMG_DIR     = f\"{DATASET_DIR}/train1\"     \n\n# Outputs\nOUT_DIR   = \"/kaggle/working/eda_outputs\"  # persisted between cells\nSAMPLE_PER_LABEL = 12\nSIZE_PROBE_N = 1000\nDISPLAY_TOP_N = 30","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T20:02:38.203344Z","iopub.execute_input":"2025-09-11T20:02:38.203806Z","iopub.status.idle":"2025-09-11T20:02:38.209067Z","shell.execute_reply.started":"2025-09-11T20:02:38.203779Z","shell.execute_reply":"2025-09-11T20:02:38.208234Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import os\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\npd.set_option(\"display.max_columns\", 200)\nplt.rcParams[\"figure.figsize\"] = (8, 5)\n\n# Ensure output dirs\nout_dir = Path(OUT_DIR)\n(out_dir / \"plots\").mkdir(parents=True, exist_ok=True)\n(out_dir / \"tables\").mkdir(exist_ok=True)\n(out_dir / \"samples\").mkdir(exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T19:23:28.994017Z","iopub.execute_input":"2025-09-11T19:23:28.994346Z","iopub.status.idle":"2025-09-11T19:23:29.016133Z","shell.execute_reply.started":"2025-09-11T19:23:28.994315Z","shell.execute_reply":"2025-09-11T19:23:29.014914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Configuration: label and metadata columns","metadata":{}},{"cell_type":"code","source":"PATHOLOGY_COLS = [\n    'Atelectasis','Cardiomegaly','Consolidation','Edema',\n    'Enlarged Cardiomediastinum','Fracture','Lung Lesion','Lung Opacity',\n    'No Finding','Pleural Effusion','Pleural Other','Pneumonia',\n    'Pneumothorax','Support Devices'\n]\n\nMETA_COLS = [\n    'Image_name','Patient_ID','Study','Sex','Age','ViewCategory','ViewPosition'\n]\n\nEXPECTED_COLS = META_COLS + PATHOLOGY_COLS","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T19:23:29.018096Z","iopub.execute_input":"2025-09-11T19:23:29.018439Z","iopub.status.idle":"2025-09-11T19:23:29.039595Z","shell.execute_reply.started":"2025-09-11T19:23:29.018405Z","shell.execute_reply":"2025-09-11T19:23:29.038546Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(TRAIN_CSV)\nprint(f\"Loaded {len(df)} rows from {TRAIN_CSV}\")\n\n# Build filepath if not present and IMG_DIR provided\nif \"filepath\" not in df.columns and \"Image_name\" in df.columns and IMG_DIR:\n    df[\"filepath\"] = df[\"Image_name\"].apply(lambda x: str(Path(IMG_DIR) / x))\n\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T19:23:49.338927Z","iopub.execute_input":"2025-09-11T19:23:49.339234Z","iopub.status.idle":"2025-09-11T19:23:50.564496Z","shell.execute_reply.started":"2025-09-11T19:23:49.33921Z","shell.execute_reply":"2025-09-11T19:23:50.563591Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Helpers","metadata":{}},{"cell_type":"code","source":"def save_table(df, name):\n    p = Path(OUT_DIR) / \"tables\" / f\"{name}.csv\"\n    df.to_csv(p, index=False)\n    print(f\"[saved] {p}\")\n    return p\n\ndef barplot_series(s, title, xlabel, ylabel, fname, rotate=45):\n    ax = s.plot(kind='bar')\n    ax.set_title(title)\n    ax.set_xlabel(xlabel)\n    ax.set_ylabel(ylabel)\n    for container in ax.containers:\n        ax.bar_label(container, fmt='{:,.0f}', label_type='edge', padding=1)\n    plt.xticks(rotation=rotate, ha='right')\n    plt.tight_layout()\n    out = Path(OUT_DIR) / \"plots\" / f\"{fname}.png\"\n    plt.savefig(out, dpi=150)\n    plt.show()\n    plt.close()\n    print(f\"[saved] {out}\")\n    return out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T19:57:22.174137Z","iopub.execute_input":"2025-09-11T19:57:22.174435Z","iopub.status.idle":"2025-09-11T19:57:22.181901Z","shell.execute_reply.started":"2025-09-11T19:57:22.174411Z","shell.execute_reply":"2025-09-11T19:57:22.180829Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataframe info & nulls","metadata":{}},{"cell_type":"code","source":"missing_cols = [c for c in EXPECTED_COLS if c not in df.columns]\nif missing_cols:\n    print(\"[warn] Missing expected columns:\", missing_cols)\n\ninfo = pd.DataFrame({\n    \"column\": df.columns,\n    \"dtype\": [str(t) for t in df.dtypes.values],\n    \"non_null\": df.notnull().sum().values,\n    \"nulls\": df.isnull().sum().values,\n    \"per_of_nulls\":100*df.isnull().sum().values/df.shape[0],\n    \"unique\": [df[c].nunique() for c in df.columns]\n})\nsave_table(info, \"00_dataframe_info\")\ninfo","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T20:47:06.408267Z","iopub.execute_input":"2025-09-11T20:47:06.408604Z","iopub.status.idle":"2025-09-11T20:47:06.629111Z","shell.execute_reply.started":"2025-09-11T20:47:06.408581Z","shell.execute_reply":"2025-09-11T20:47:06.628008Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Age distribution","metadata":{}},{"cell_type":"code","source":"df.Age.plot(kind = 'hist', bins = 30, title = \"Age distribution\")\nplt.xlabel(\"Age (years)\")\nplt.ylabel(\"Count\")\nplt.tight_layout()\nplt.savefig(Path(OUT_DIR) / \"plots\" / \"01_age_hist.png\", dpi=150)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T20:46:41.469093Z","iopub.execute_input":"2025-09-11T20:46:41.469376Z","iopub.status.idle":"2025-09-11T20:46:41.83397Z","shell.execute_reply.started":"2025-09-11T20:46:41.469356Z","shell.execute_reply":"2025-09-11T20:46:41.832792Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sex, ViewPosition, ViewCategory","metadata":{}},{"cell_type":"code","source":"if \"Sex\" in df.columns:\n    sex_counts = df[\"Sex\"].value_counts(dropna=False)\n    save_table(sex_counts, \"02_sex_counts\")\n    barplot_series(sex_counts, \"Sex distribution\", \"Sex\", \"Count\", \"02_sex_counts\")\n\nif \"ViewPosition\" in df.columns:\n    vp_counts = df[\"ViewPosition\"].value_counts(dropna=False)\n    save_table(vp_counts, \"03_viewposition_counts\")\n    barplot_series(vp_counts, \"ViewPosition distribution\", \"ViewPosition\", \"Count\", \"03_viewposition_counts\")\n\nif \"ViewCategory\" in df.columns:\n    vc_counts = df[\"ViewCategory\"].value_counts(dropna=False)\n    save_table(vc_counts, \"04_viewcategory_counts\")\n    barplot_series(vc_counts, \"ViewCategory distribution\", \"ViewCategory\", \"Count\", \"04_viewcategory_counts\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T20:45:58.651073Z","iopub.execute_input":"2025-09-11T20:45:58.65136Z","iopub.status.idle":"2025-09-11T20:45:59.631516Z","shell.execute_reply.started":"2025-09-11T20:45:58.651338Z","shell.execute_reply":"2025-09-11T20:45:59.630622Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Images per Patient / Study","metadata":{}},{"cell_type":"code","source":"if \"Patient_ID\" in df.columns and \"Image_name\" in df.columns:\n    imgs_per_patient = df.groupby(\"Patient_ID\")[\"Image_name\"].count().sort_values(ascending=False)\n    save_table(imgs_per_patient.reset_index(name=\"images\"), \"05_images_per_patient\")\n    barplot_series(imgs_per_patient.head(DISPLAY_TOP_N), f\"Images per patient (top {DISPLAY_TOP_N})\", \"Patient_ID\", \"Images\", \"05_images_per_patient_top\")\n\nif \"Study\" in df.columns and \"Image_name\" in df.columns:\n    imgs_per_study = df.groupby([\"Study\", \"Patient_ID\"])[\"Image_name\"].count().sort_values(ascending=False)\n    save_table(imgs_per_study.reset_index(name=\"images\"), \"06_images_per_study\")\n    barplot_series(imgs_per_study.head(DISPLAY_TOP_N), f\"Images per study (top {DISPLAY_TOP_N})\", \"Study\", \"Images\", \"06_images_per_study_top\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T20:10:50.550209Z","iopub.execute_input":"2025-09-11T20:10:50.55055Z","iopub.status.idle":"2025-09-11T20:10:52.483125Z","shell.execute_reply.started":"2025-09-11T20:10:50.550524Z","shell.execute_reply":"2025-09-11T20:10:52.482053Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Label imbalance (positives per condition)","metadata":{}},{"cell_type":"markdown","source":"## \"No Finding\" exclusivity checks","metadata":{}},{"cell_type":"code","source":"present_cols = [c for c in PATHOLOGY_COLS if c in df.columns]\nlabel_sums = df[present_cols].fillna(0).sum().sort_values(ascending=False)\nsave_table(label_sums, \"07_label_positives\")\nbarplot_series(label_sums, \"Label positives (class imbalance)\", \"Label\", \"Positive count\", \"07_label_positives\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T20:13:32.656779Z","iopub.execute_input":"2025-09-11T20:13:32.657065Z","iopub.status.idle":"2025-09-11T20:13:33.278282Z","shell.execute_reply.started":"2025-09-11T20:13:32.657044Z","shell.execute_reply":"2025-09-11T20:13:33.277195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nif \"No Finding\" in present_cols:\n    nf = df[\"No Finding\"].fillna(0).astype(int)\n    co_counts = {}\n    for c in present_cols:\n        if c == \"No Finding\":\n            continue\n        both = ((df[c].fillna(0).astype(int) == 1) & (nf == 1)).sum()\n        co_counts[c] = both\n    co_df = pd.Series(co_counts).sort_values(ascending=False).to_frame(\"co_occurrence_with_NoFinding\")\n    save_table(co_df.reset_index(names=[\"label\",\"count\"]), \"08_no_finding_cooccurrence\")\n    barplot_series(co_df[\"co_occurrence_with_NoFinding\"], '\"No Finding\" co-occurrence (should be near 0)', \"Label\", \"Count\", \"08_no_finding_cooccurrence\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T20:15:42.390336Z","iopub.execute_input":"2025-09-11T20:15:42.390983Z","iopub.status.idle":"2025-09-11T20:15:42.908321Z","shell.execute_reply.started":"2025-09-11T20:15:42.390956Z","shell.execute_reply":"2025-09-11T20:15:42.907318Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Co-occurrence matrix (counts)","metadata":{}},{"cell_type":"code","source":"def heatmap(df, title, fname, annotate=False):\n    plt.figure(figsize=(10, 8))\n    plt.imshow(df.values, aspect='auto', interpolation='nearest')\n    plt.xticks(range(df.shape[1]), df.columns, rotation=90)\n    plt.yticks(range(df.shape[0]), df.index)\n    plt.title(title)\n    plt.colorbar()\n    if annotate:\n        for i in range(df.shape[0]):\n            for j in range(df.shape[1]):\n                plt.text(j, i, f\"{df.iat[i,j]:.2f}\", ha='center', va='center', fontsize=7)\n    plt.tight_layout()\n    out = Path(OUT_DIR) / \"plots\" / f\"{fname}.png\"\n    plt.savefig(out, dpi=150)\n    plt.show()\n    plt.close()\n    print(f\"[saved] {out}\")\n    return out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T20:49:41.750132Z","iopub.execute_input":"2025-09-11T20:49:41.75081Z","iopub.status.idle":"2025-09-11T20:49:41.757875Z","shell.execute_reply.started":"2025-09-11T20:49:41.75078Z","shell.execute_reply":"2025-09-11T20:49:41.756825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bin_df = df[present_cols].fillna(0).astype(int)\nco_mat = pd.DataFrame(0, index=present_cols, columns=present_cols, dtype=int)\nfor a in present_cols:\n    for b in present_cols:\n        co_mat.loc[a, b] = int(((bin_df[a]==1) & (bin_df[b]==1)).sum())\nsave_table(co_mat.reset_index().rename(columns={\"index\":\"label\"}), \"09_cooccurrence_counts\")\nheatmap(co_mat, \"Label co-occurrence (counts)\", \"09_cooccurrence_counts\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T20:49:46.470223Z","iopub.execute_input":"2025-09-11T20:49:46.470519Z","iopub.status.idle":"2025-09-11T20:49:47.466331Z","shell.execute_reply.started":"2025-09-11T20:49:46.470491Z","shell.execute_reply":"2025-09-11T20:49:47.465539Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Label rates by ViewPosition","metadata":{}},{"cell_type":"code","source":"if \"ViewPosition\" in df.columns:\n    rates = (df.groupby(\"ViewPosition\")[present_cols]\n               .mean()\n               .sort_index())\n    save_table(rates.reset_index(), \"11_label_rates_by_viewposition\")\n    heatmap(rates, \"Label rates by ViewPosition\", \"11_label_rates_by_viewposition\", annotate=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T20:55:39.143956Z","iopub.execute_input":"2025-09-11T20:55:39.144242Z","iopub.status.idle":"2025-09-11T20:55:40.077182Z","shell.execute_reply.started":"2025-09-11T20:55:39.144222Z","shell.execute_reply":"2025-09-11T20:55:40.076053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def image_sizes(img_paths, max_n=1000):\n    sizes = []\n    for i, p in enumerate(img_paths):\n        if i >= max_n:\n            break\n        try:\n            with Image.open(p) as im:\n                sizes.append(im.size)\n        except Exception:\n            continue\n    return sizes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T20:58:08.693288Z","iopub.execute_input":"2025-09-11T20:58:08.694122Z","iopub.status.idle":"2025-09-11T20:58:08.699367Z","shell.execute_reply.started":"2025-09-11T20:58:08.694091Z","shell.execute_reply":"2025-09-11T20:58:08.698369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if \"filepath\" in df.columns and IMG_DIR:\n    sizes = image_sizes(df[\"filepath\"].values, max_n=SIZE_PROBE_N)\n    if sizes:\n        wh = pd.DataFrame(sizes, columns=[\"width\",\"height\"])\n        save_table(wh.describe().reset_index(), \"12_image_size_summary\")\n        plt.scatter(wh[\"width\"], wh[\"height\"], s=8, alpha=0.6)\n        plt.title(\"Image sizes (sample)\")\n        plt.xlabel(\"Width\"); plt.ylabel(\"Height\"); plt.tight_layout()\n        plt.savefig(Path(OUT_DIR) / \"plots\" / \"12_image_sizes_scatter.png\", dpi=150)\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T20:58:10.224292Z","iopub.execute_input":"2025-09-11T20:58:10.224603Z","iopub.status.idle":"2025-09-11T20:58:22.451918Z","shell.execute_reply.started":"2025-09-11T20:58:10.22458Z","shell.execute_reply":"2025-09-11T20:58:22.450953Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Class imbalance: suggested weights (BCE & Focal)","metadata":{}},{"cell_type":"code","source":"label_sums","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T21:02:09.895019Z","iopub.execute_input":"2025-09-11T21:02:09.895337Z","iopub.status.idle":"2025-09-11T21:02:09.902238Z","shell.execute_reply.started":"2025-09-11T21:02:09.895314Z","shell.execute_reply":"2025-09-11T21:02:09.901384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"N = len(df)\npos = label_sums.reindex(present_cols).astype(float)  # align order\nneg = N - pos\n\n# Avoid div by zero\neps = 1e-9\npos_rate = (pos / (N + eps)).rename(\"pos_rate\")\nneg_rate = (neg / (N + eps)).rename(\"neg_rate\")\n\n# BCE class weights per label (common choice): w_pos = N/pos, w_neg = N/neg (or normalized variant)\nw_pos = (N / (pos + eps)).rename(\"w_pos\")\nw_neg = (N / (neg + eps)).rename(\"w_neg\")\n\n# Focal loss alpha suggestion: alpha ~ neg_rate (penalize positives more when rare)\nalpha = neg_rate.rename(\"alpha_focal_suggestion\")\ngamma = pd.Series(2.0, index=present_cols, name=\"gamma_default\")  # typical\n\n# \"Effective number of samples\" (Cui et al.) weights\nbeta = 0.999\neff_num = (1 - beta**pos) / (1 - beta)  # effective samples per class (positives)\nw_effective = (1.0 / (eff_num + eps))\nw_effective = (w_effective / w_effective.max()).rename(\"w_effective_norm\")\n\nweights_df = pd.concat([pos.astype(int).rename(\"positives\"),\n                        neg.astype(int).rename(\"negatives\"),\n                        pos_rate, neg_rate, w_pos, w_neg, alpha, gamma, w_effective], axis=1)\nsave_table(weights_df.reset_index().rename(columns={\"index\": \"label\"}), \"13_class_weights\")\n\ndisplay(weights_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T21:01:33.480383Z","iopub.execute_input":"2025-09-11T21:01:33.480856Z","iopub.status.idle":"2025-09-11T21:01:33.688775Z","shell.execute_reply.started":"2025-09-11T21:01:33.480826Z","shell.execute_reply":"2025-09-11T21:01:33.686374Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Summary","metadata":{}},{"cell_type":"code","source":"print(\"EDA complete.\")\nprint(\"Outputs:\")\nprint(\" - Tables:\", Path(OUT_DIR) / \"tables\")\nprint(\" - Plots: \", Path(OUT_DIR) / \"plots\")\nprint(\" - Samples:\", Path(OUT_DIR) / \"samples\")\nprint(\" - Weights:\", Path(OUT_DIR) / \"tables\" / \"13_class_weights.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-11T21:03:22.798572Z","iopub.execute_input":"2025-09-11T21:03:22.798867Z","iopub.status.idle":"2025-09-11T21:03:22.805303Z","shell.execute_reply.started":"2025-09-11T21:03:22.798843Z","shell.execute_reply":"2025-09-11T21:03:22.804187Z"}},"outputs":[],"execution_count":null}]}