{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nBASE = \"/kaggle/input\"\n\nfor root, dirs, files in os.walk(BASE):\n    level = root.replace(BASE, \"\").count(os.sep)\n\n    if level <= 2:\n        print(root)\n        for file in files[:10]:\n            print(\"   \", file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:13:35.601261Z","iopub.execute_input":"2026-08-23T02:13:35.601595Z","iopub.status.idle":"2026-08-23T02:14:02.069573Z","shell.execute_reply.started":"2026-08-23T02:13:35.601564Z","shell.execute_reply":"2026-08-23T02:14:02.068755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"INPUT DIRECTORIES:\")\nfor item in os.listdir(\"/kaggle/input\"):\n    print(\" -\", item)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:14:19.539739Z","iopub.execute_input":"2026-08-23T02:14:19.540082Z","iopub.status.idle":"2026-08-23T02:14:19.545889Z","shell.execute_reply.started":"2026-08-23T02:14:19.540052Z","shell.execute_reply":"2026-08-23T02:14:19.544958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nBASE = \"/kaggle/input/competitions\"\n\nfor root, dirs, files in os.walk(BASE):\n    level = root.replace(BASE, \"\").count(os.sep)\n\n    if level <= 3:\n        print(root)\n        for file in files[:15]:\n            print(\"   \", file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:14:33.158368Z","iopub.execute_input":"2026-08-23T02:14:33.158719Z","iopub.status.idle":"2026-08-23T02:14:38.884022Z","shell.execute_reply.started":"2026-08-23T02:14:33.158664Z","shell.execute_reply":"2026-08-23T02:14:38.883157Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nDATA_DIR = \"/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection\"\n\ntrain_csv = f\"{DATA_DIR}/train.csv\"\n\ndf = pd.read_csv(train_csv)\n\nprint(\"Dataset shape:\", df.shape)\n\nprint(\"\\nColumns:\")\nprint(df.columns.tolist())\n\nprint(\"\\nFirst 10 rows:\")\ndisplay(df.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:14:49.314442Z","iopub.execute_input":"2026-08-23T02:14:49.314878Z","iopub.status.idle":"2026-08-23T02:14:49.840454Z","shell.execute_reply.started":"2026-08-23T02:14:49.314846Z","shell.execute_reply":"2026-08-23T02:14:49.839757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Unique classes:\")\nprint(df[\"class_name\"].unique())\n\nprint(\"\\nClass counts:\")\ndisplay(\n    df[\"class_name\"]\n    .value_counts()\n    .rename_axis(\"class_name\")\n    .reset_index(name=\"count\")\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:15:00.421206Z","iopub.execute_input":"2026-08-23T02:15:00.421549Z","iopub.status.idle":"2026-08-23T02:15:00.443123Z","shell.execute_reply.started":"2026-08-23T02:15:00.421519Z","shell.execute_reply":"2026-08-23T02:15:00.442283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Unique classes:\")\nprint(df[\"class_name\"].unique())\n\nprint(\"\\nClass counts:\")\nclass_counts = (\n    df[\"class_name\"]\n    .value_counts()\n    .rename_axis(\"class_name\")\n    .reset_index(name=\"count\")\n)\n\ndisplay(class_counts)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:15:16.659353Z","iopub.execute_input":"2026-08-23T02:15:16.660018Z","iopub.status.idle":"2026-08-23T02:15:16.680591Z","shell.execute_reply.started":"2026-08-23T02:15:16.659986Z","shell.execute_reply":"2026-08-23T02:15:16.679768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Missing values:\")\ndisplay(df.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:15:28.442133Z","iopub.execute_input":"2026-08-23T02:15:28.442476Z","iopub.status.idle":"2026-08-23T02:15:28.462143Z","shell.execute_reply.started":"2026-08-23T02:15:28.442446Z","shell.execute_reply":"2026-08-23T02:15:28.461236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Number of unique images:\", df[\"image_id\"].nunique())\nprint(\"Number of annotation rows:\", len(df))\n\nprint(\"\\nAnnotations per image:\")\nannotations_per_image = df.groupby(\"image_id\").size()\n\ndisplay(annotations_per_image.describe())\n\nprint(\"\\nRadiologists:\")\nprint(\"Unique radiologists:\", df[\"rad_id\"].nunique())\nprint(df[\"rad_id\"].value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:15:58.844803Z","iopub.execute_input":"2026-08-23T02:15:58.845183Z","iopub.status.idle":"2026-08-23T02:15:58.907191Z","shell.execute_reply.started":"2026-08-23T02:15:58.845153Z","shell.execute_reply":"2026-08-23T02:15:58.906269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Radiologists per image:\")\n\nradiologists_per_image = df.groupby(\"image_id\")[\"rad_id\"].nunique()\n\ndisplay(radiologists_per_image.describe())\n\nprint(\"\\nDistribution:\")\ndisplay(\n    radiologists_per_image\n    .value_counts()\n    .sort_index()\n    .rename_axis(\"number_of_radiologists\")\n    .reset_index(name=\"number_of_images\")\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:16:13.068109Z","iopub.execute_input":"2026-08-23T02:16:13.068453Z","iopub.status.idle":"2026-08-23T02:16:13.115387Z","shell.execute_reply.started":"2026-08-23T02:16:13.068423Z","shell.execute_reply":"2026-08-23T02:16:13.114504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Radiologists per image:\")\n\nradiologists_per_image = df.groupby(\"image_id\")[\"rad_id\"].nunique()\n\ndisplay(radiologists_per_image.describe())\n\nprint(\"\\nDistribution:\")\ndisplay(\n    radiologists_per_image\n    .value_counts()\n    .sort_index()\n    .rename_axis(\"number_of_radiologists\")\n    .reset_index(name=\"number_of_images\")\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:16:33.837846Z","iopub.execute_input":"2026-08-23T02:16:33.838156Z","iopub.status.idle":"2026-08-23T02:16:33.884051Z","shell.execute_reply.started":"2026-08-23T02:16:33.838129Z","shell.execute_reply":"2026-08-23T02:16:33.883165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# How many different radiologists labeled each class?\nagreement = (\n    df.groupby([\"image_id\", \"class_name\"])[\"rad_id\"]\n      .nunique()\n      .reset_index(name=\"radiologist_count\")\n)\n\nprint(\"Radiologist agreement for Nodule/Mass:\")\ndisplay(\n    agreement[agreement[\"class_name\"] == \"Nodule/Mass\"]\n    .describe()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:16:45.562263Z","iopub.execute_input":"2026-08-23T02:16:45.562574Z","iopub.status.idle":"2026-08-23T02:16:45.623092Z","shell.execute_reply.started":"2026-08-23T02:16:45.562546Z","shell.execute_reply":"2026-08-23T02:16:45.622201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Nodule/Mass examples with multiple radiologists:\")\n\nnodule_agreement = agreement[\n    agreement[\"class_name\"] == \"Nodule/Mass\"\n].sort_values(\"radiologist_count\", ascending=False)\n\ndisplay(nodule_agreement.head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:16:54.792917Z","iopub.execute_input":"2026-08-23T02:16:54.793237Z","iopub.status.idle":"2026-08-23T02:16:54.80955Z","shell.execute_reply.started":"2026-08-23T02:16:54.793202Z","shell.execute_reply":"2026-08-23T02:16:54.808572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Nodule/Mass agreement across the 3 radiologists\n\nnodule = df[df[\"class_name\"] == \"Nodule/Mass\"].copy()\n\nnodule_by_image = (\n    nodule.groupby(\"image_id\")[\"rad_id\"]\n    .nunique()\n    .value_counts()\n    .sort_index()\n)\n\nprint(\"Number of radiologists identifying Nodule/Mass:\")\ndisplay(\n    nodule_by_image\n    .rename_axis(\"radiologists_identifying_nodule_mass\")\n    .reset_index(name=\"number_of_images\")\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:17:21.833942Z","iopub.execute_input":"2026-08-23T02:17:21.834261Z","iopub.status.idle":"2026-08-23T02:17:21.854711Z","shell.execute_reply.started":"2026-08-23T02:17:21.83423Z","shell.execute_reply":"2026-08-23T02:17:21.853762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Show examples where multiple radiologists identified Nodule/Mass\n\nnodule_agreement = (\n    nodule.groupby(\"image_id\")\n    .agg(\n        radiologist_count=(\"rad_id\", \"nunique\"),\n        radiologists=(\"rad_id\", lambda x: \", \".join(sorted(x.unique())))\n    )\n    .reset_index()\n    .sort_values(\"radiologist_count\", ascending=False)\n)\n\nprint(\"Examples of Nodule/Mass agreement:\")\ndisplay(nodule_agreement.head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:17:33.423128Z","iopub.execute_input":"2026-08-23T02:17:33.423431Z","iopub.status.idle":"2026-08-23T02:17:33.486408Z","shell.execute_reply.started":"2026-08-23T02:17:33.423406Z","shell.execute_reply":"2026-08-23T02:17:33.485579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Inspect all annotations for a few images with Nodule/Mass\n\nexample_ids = nodule_agreement.head(5)[\"image_id\"].tolist()\n\ndisplay(\n    df[df[\"image_id\"].isin(example_ids)]\n    .sort_values([\"image_id\", \"rad_id\"])\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:17:48.003499Z","iopub.execute_input":"2026-08-23T02:17:48.004423Z","iopub.status.idle":"2026-08-23T02:17:48.042136Z","shell.execute_reply.started":"2026-08-23T02:17:48.004388Z","shell.execute_reply":"2026-08-23T02:17:48.041105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nodule = df[df[\"class_name\"] == \"Nodule/Mass\"].copy()\n\nprint(\"Nodule/Mass annotation rows:\", len(nodule))\nprint(\"Unique Nodule/Mass images:\", nodule[\"image_id\"].nunique())\n\ndisplay(\n    nodule[\n        [\n            \"image_id\",\n            \"rad_id\",\n            \"x_min\",\n            \"y_min\",\n            \"x_max\",\n            \"y_max\"\n        ]\n    ].head(20)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:18:06.326165Z","iopub.execute_input":"2026-08-23T02:18:06.327358Z","iopub.status.idle":"2026-08-23T02:18:06.355024Z","shell.execute_reply.started":"2026-08-23T02:18:06.327318Z","shell.execute_reply":"2026-08-23T02:18:06.354274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nodule[\"box_width\"] = nodule[\"x_max\"] - nodule[\"x_min\"]\nnodule[\"box_height\"] = nodule[\"y_max\"] - nodule[\"y_min\"]\nnodule[\"box_area\"] = nodule[\"box_width\"] * nodule[\"box_height\"]\n\ndisplay(\n    nodule[\n        [\n            \"box_width\",\n            \"box_height\",\n            \"box_area\"\n        ]\n    ].describe()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:18:18.307937Z","iopub.execute_input":"2026-08-23T02:18:18.308239Z","iopub.status.idle":"2026-08-23T02:18:18.329666Z","shell.execute_reply.started":"2026-08-23T02:18:18.308212Z","shell.execute_reply":"2026-08-23T02:18:18.328781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nnodule = df[df[\"class_name\"] == \"Nodule/Mass\"].copy()\n\ndef calculate_iou(box1, box2):\n    x1_min, y1_min, x1_max, y1_max = box1\n    x2_min, y2_min, x2_max, y2_max = box2\n\n    inter_xmin = max(x1_min, x2_min)\n    inter_ymin = max(y1_min, y2_min)\n    inter_xmax = min(x1_max, x2_max)\n    inter_ymax = min(y1_max, y2_max)\n\n    inter_w = max(0, inter_xmax - inter_xmin)\n    inter_h = max(0, inter_ymax - inter_ymin)\n    intersection = inter_w * inter_h\n\n    area1 = max(0, x1_max - x1_min) * max(0, y1_max - y1_min)\n    area2 = max(0, x2_max - x2_min) * max(0, y2_max - y2_min)\n\n    union = area1 + area2 - intersection\n\n    if union == 0:\n        return 0.0\n\n    return intersection / union\n\n\n# Calculate pairwise IoU within each image\nious = []\n\nfor image_id, group in nodule.groupby(\"image_id\"):\n\n    records = group[\n        [\"rad_id\", \"x_min\", \"y_min\", \"x_max\", \"y_max\"]\n    ].to_dict(\"records\")\n\n    for i in range(len(records)):\n        for j in range(i + 1, len(records)):\n\n            box1 = (\n                records[i][\"x_min\"],\n                records[i][\"y_min\"],\n                records[i][\"x_max\"],\n                records[i][\"y_max\"]\n            )\n\n            box2 = (\n                records[j][\"x_min\"],\n                records[j][\"y_min\"],\n                records[j][\"x_max\"],\n                records[j][\"y_max\"]\n            )\n\n            iou = calculate_iou(box1, box2)\n\n            ious.append({\n                \"image_id\": image_id,\n                \"rad1\": records[i][\"rad_id\"],\n                \"rad2\": records[j][\"rad_id\"],\n                \"iou\": iou\n            })\n\niou_df = pd.DataFrame(ious)\n\nprint(\"Number of Nodule/Mass annotation pairs:\", len(iou_df))\n\nprint(\"\\nIoU statistics:\")\ndisplay(iou_df[\"iou\"].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:18:33.355182Z","iopub.execute_input":"2026-08-23T02:18:33.356111Z","iopub.status.idle":"2026-08-23T02:18:34.253438Z","shell.execute_reply.started":"2026-08-23T02:18:33.356073Z","shell.execute_reply":"2026-08-23T02:18:34.25276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\nIoU categories:\")\n\niou_categories = pd.cut(\n    iou_df[\"iou\"],\n    bins=[-0.001, 0, 0.25, 0.50, 0.75, 1.0],\n    labels=[\n        \"No overlap\",\n        \"Low overlap\",\n        \"Moderate overlap\",\n        \"High overlap\",\n        \"Very high overlap\"\n    ]\n)\n\ndisplay(\n    iou_categories\n    .value_counts()\n    .sort_index()\n    .rename_axis(\"IoU category\")\n    .reset_index(name=\"annotation_pairs\")\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:18:49.331841Z","iopub.execute_input":"2026-08-23T02:18:49.332185Z","iopub.status.idle":"2026-08-23T02:18:49.349412Z","shell.execute_reply.started":"2026-08-23T02:18:49.332155Z","shell.execute_reply":"2026-08-23T02:18:49.348718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pydicom\n\nTRAIN_DIR = (\n    \"/kaggle/input/competitions/\"\n    \"vinbigdata-chest-xray-abnormalities-detection/train\"\n)\n\nsample_files = [\n    f for f in os.listdir(TRAIN_DIR)\n    if f.endswith(\".dicom\")\n][:10]\n\nprint(\"Sample DICOM files:\", len(sample_files))\n\nfor filename in sample_files:\n    path = os.path.join(TRAIN_DIR, filename)\n\n    ds = pydicom.dcmread(path, stop_before_pixels=False)\n\n    print(\"\\nFile:\", filename)\n    print(\"Rows:\", getattr(ds, \"Rows\", None))\n    print(\"Columns:\", getattr(ds, \"Columns\", None))\n    print(\"Photometric Interpretation:\",\n          getattr(ds, \"PhotometricInterpretation\", None))\n    print(\"Pixel Spacing:\",\n          getattr(ds, \"PixelSpacing\", None))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:19:08.155426Z","iopub.execute_input":"2026-08-23T02:19:08.155869Z","iopub.status.idle":"2026-08-23T02:19:10.28371Z","shell.execute_reply.started":"2026-08-23T02:19:08.155826Z","shell.execute_reply":"2026-08-23T02:19:10.28275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --------------------------------------------------\n# RESTART-SAFE: Load annotations + image dimensions\n# --------------------------------------------------\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport pydicom\n\nDATA_DIR = (\n    \"/kaggle/input/competitions/\"\n    \"vinbigdata-chest-xray-abnormalities-detection\"\n)\n\nTRAIN_DIR = os.path.join(DATA_DIR, \"train\")\n\n# Always reload the original annotation CSV\ndf = pd.read_csv(\n    os.path.join(DATA_DIR, \"train.csv\")\n)\n\nprint(\"Original annotation table:\", df.shape)\n\n# --------------------------------------------------\n# Read image dimensions\n# --------------------------------------------------\n\nimage_dimensions = {}\n\ndicom_files = [\n    f for f in os.listdir(TRAIN_DIR)\n    if f.endswith(\".dicom\")\n]\n\nprint(\"DICOM files found:\", len(dicom_files))\n\nfor filename in dicom_files:\n\n    image_id = filename.replace(\".dicom\", \"\")\n    path = os.path.join(TRAIN_DIR, filename)\n\n    try:\n        ds = pydicom.dcmread(\n            path,\n            stop_before_pixels=True\n        )\n\n        image_dimensions[image_id] = {\n            \"height\": int(ds.Rows),\n            \"width\": int(ds.Columns),\n            \"photometric\": getattr(\n                ds,\n                \"PhotometricInterpretation\",\n                None\n            )\n        }\n\n    except Exception as e:\n        print(\"Failed:\", filename, e)\n\nprint(\n    \"Images with dimensions:\",\n    len(image_dimensions)\n)\n\n# --------------------------------------------------\n# Convert dimensions to DataFrame\n# --------------------------------------------------\n\ndims = (\n    pd.DataFrame.from_dict(\n        image_dimensions,\n        orient=\"index\"\n    )\n    .rename_axis(\"image_id\")\n    .reset_index()\n)\n\nprint(\"\\nDimension table:\")\ndisplay(dims.head())\n\n# --------------------------------------------------\n# Merge\n# --------------------------------------------------\n\ndf = df.merge(\n    dims,\n    on=\"image_id\",\n    how=\"left\",\n    validate=\"many_to_one\"\n)\n\nprint(\"\\nMerged dataset:\", df.shape)\n\nprint(\"\\nMissing image dimensions:\")\nprint(\n    df[\n        [\"height\", \"width\", \"photometric\"]\n    ].isnull().sum()\n)\n\ndisplay(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:21:50.383648Z","iopub.execute_input":"2026-08-23T02:21:50.384109Z","iopub.status.idle":"2026-08-23T02:22:46.214869Z","shell.execute_reply.started":"2026-08-23T02:21:50.384078Z","shell.execute_reply":"2026-08-23T02:22:46.213868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --------------------------------------------------\n# CREATE NORMALIZED BOUNDING-BOX FEATURES\n# --------------------------------------------------\n\nbbox_mask = (\n    df[\"x_min\"].notna()\n    & df[\"y_min\"].notna()\n    & df[\"x_max\"].notna()\n    & df[\"y_max\"].notna()\n)\n\ndf[\"box_width\"] = np.where(\n    bbox_mask,\n    df[\"x_max\"] - df[\"x_min\"],\n    np.nan\n)\n\ndf[\"box_height\"] = np.where(\n    bbox_mask,\n    df[\"y_max\"] - df[\"y_min\"],\n    np.nan\n)\n\ndf[\"box_area\"] = (\n    df[\"box_width\"] *\n    df[\"box_height\"]\n)\n\n# Normalized dimensions\ndf[\"box_width_norm\"] = (\n    df[\"box_width\"] / df[\"width\"]\n)\n\ndf[\"box_height_norm\"] = (\n    df[\"box_height\"] / df[\"height\"]\n)\n\ndf[\"box_area_norm\"] = (\n    df[\"box_area\"] /\n    (df[\"width\"] * df[\"height\"])\n)\n\n# Normalized center coordinates\ndf[\"center_x_norm\"] = (\n    ((df[\"x_min\"] + df[\"x_max\"]) / 2)\n    / df[\"width\"]\n)\n\ndf[\"center_y_norm\"] = (\n    ((df[\"y_min\"] + df[\"y_max\"]) / 2)\n    / df[\"height\"]\n)\n\nprint(\"Bounding-box features created.\")\n\ndisplay(\n    df[\n        [\n            \"image_id\",\n            \"class_name\",\n            \"box_width_norm\",\n            \"box_height_norm\",\n            \"box_area_norm\",\n            \"center_x_norm\",\n            \"center_y_norm\"\n        ]\n    ].head(20)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:22:46.216482Z","iopub.execute_input":"2026-08-23T02:22:46.216872Z","iopub.status.idle":"2026-08-23T02:22:46.256822Z","shell.execute_reply.started":"2026-08-23T02:22:46.216829Z","shell.execute_reply":"2026-08-23T02:22:46.255643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CONAN IMAGING — STEP 2\n# Load VinBigData annotations + DICOM image dimensions\n# ============================================================\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport pydicom\n\n# ------------------------------------------------------------\n# 1. Paths\n# ------------------------------------------------------------\n\nDATA_DIR = (\n    \"/kaggle/input/competitions/\"\n    \"vinbigdata-chest-xray-abnormalities-detection\"\n)\n\nTRAIN_DIR = os.path.join(DATA_DIR, \"train\")\nCSV_PATH = os.path.join(DATA_DIR, \"train.csv\")\n\nprint(\"DATA DIRECTORY:\")\nprint(DATA_DIR)\n\nprint(\"\\nTRAIN DIRECTORY:\")\nprint(TRAIN_DIR)\n\n# ------------------------------------------------------------\n# 2. Load original VinBigData annotations\n# ------------------------------------------------------------\n\ndf = pd.read_csv(CSV_PATH)\n\nprint(\"\\nOriginal annotation dataset:\")\nprint(\"Rows:\", len(df))\nprint(\"Columns:\", len(df.columns))\nprint(df.columns.tolist())\n\n# ------------------------------------------------------------\n# 3. Find all DICOM files\n# ------------------------------------------------------------\n\ndicom_files = [\n    f for f in os.listdir(TRAIN_DIR)\n    if f.lower().endswith(\".dicom\")\n]\n\nprint(\"\\nDICOM files found:\", len(dicom_files))\n\n# ------------------------------------------------------------\n# 4. Read image dimensions\n# ------------------------------------------------------------\n\nimage_dimensions = {}\n\nfailed_files = []\n\nfor i, filename in enumerate(dicom_files):\n\n    image_id = filename.rsplit(\".\", 1)[0]\n\n    path = os.path.join(TRAIN_DIR, filename)\n\n    try:\n\n        ds = pydicom.dcmread(\n            path,\n            stop_before_pixels=True\n        )\n\n        image_dimensions[image_id] = {\n            \"height\": int(ds.Rows),\n            \"width\": int(ds.Columns),\n            \"photometric\": getattr(\n                ds,\n                \"PhotometricInterpretation\",\n                None\n            )\n        }\n\n    except Exception as e:\n\n        failed_files.append(\n            (filename, str(e))\n        )\n\n# ------------------------------------------------------------\n# 5. Create image-dimension table\n# ------------------------------------------------------------\n\ndims = pd.DataFrame(\n    [\n        {\n            \"image_id\": image_id,\n            \"height\": values[\"height\"],\n            \"width\": values[\"width\"],\n            \"photometric\": values[\"photometric\"]\n        }\n        for image_id, values\n        in image_dimensions.items()\n    ]\n)\n\nprint(\"\\nImage dimension table:\")\nprint(\"Rows:\", len(dims))\n\ndisplay(dims.head())\n\n# ------------------------------------------------------------\n# 6. Check dimension reading\n# ------------------------------------------------------------\n\nprint(\"\\nFailed DICOM files:\", len(failed_files))\n\nif failed_files:\n    display(\n        pd.DataFrame(\n            failed_files,\n            columns=[\"filename\", \"error\"]\n        ).head()\n    )\n\n# ------------------------------------------------------------\n# 7. Check duplicate image IDs\n# ------------------------------------------------------------\n\nprint(\n    \"\\nDuplicate image IDs in dimension table:\",\n    dims[\"image_id\"].duplicated().sum()\n)\n\n# ------------------------------------------------------------\n# 8. Merge annotations with dimensions\n# ------------------------------------------------------------\n\ndf = df.merge(\n    dims,\n    on=\"image_id\",\n    how=\"left\",\n    validate=\"many_to_one\"\n)\n\nprint(\"\\nMerged dataset:\")\nprint(\"Rows:\", len(df))\nprint(\"Columns:\", len(df.columns))\n\nprint(\"\\nColumns:\")\nprint(df.columns.tolist())\n\n# ------------------------------------------------------------\n# 9. Verify missing dimensions\n# ------------------------------------------------------------\n\nprint(\"\\nMissing dimensions:\")\n\ndisplay(\n    df[\n        [\"height\", \"width\", \"photometric\"]\n    ].isnull().sum()\n)\n\n# ------------------------------------------------------------\n# 10. Verify image coverage\n# ------------------------------------------------------------\n\nannotation_images = set(\n    df[\"image_id\"].unique()\n)\n\ndimension_images = set(\n    dims[\"image_id\"].unique()\n)\n\nprint(\"\\nUnique annotation images:\",\n      len(annotation_images))\n\nprint(\"Unique dimension images:\",\n      len(dimension_images))\n\nprint(\n    \"Annotation images without dimensions:\",\n    len(annotation_images - dimension_images)\n)\n\n# ------------------------------------------------------------\n# 11. Final preview\n# ------------------------------------------------------------\n\nprint(\"\\nFinal preview:\")\n\ndisplay(\n    df[\n        [\n            \"image_id\",\n            \"class_name\",\n            \"rad_id\",\n            \"x_min\",\n            \"y_min\",\n            \"x_max\",\n            \"y_max\",\n            \"height\",\n            \"width\",\n            \"photometric\"\n        ]\n    ].head(10)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:23:03.819672Z","iopub.execute_input":"2026-08-23T02:23:03.820741Z","iopub.status.idle":"2026-08-23T02:23:23.000948Z","shell.execute_reply.started":"2026-08-23T02:23:03.820675Z","shell.execute_reply":"2026-08-23T02:23:22.999931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CONAN IMAGING — STEP 3\n# Create normalized bounding-box features\n# ============================================================\n\nimport numpy as np\nimport pandas as pd\n\n# ------------------------------------------------------------\n# 1. Identify rows that actually contain bounding boxes\n# ------------------------------------------------------------\n\nbbox_mask = (\n    df[\"x_min\"].notna()\n    & df[\"y_min\"].notna()\n    & df[\"x_max\"].notna()\n    & df[\"y_max\"].notna()\n)\n\nprint(\"Total annotation rows:\", len(df))\nprint(\"Rows with bounding boxes:\", bbox_mask.sum())\nprint(\"Rows without bounding boxes:\", (~bbox_mask).sum())\n\n# ------------------------------------------------------------\n# 2. Raw bounding-box dimensions\n# ------------------------------------------------------------\n\ndf[\"box_width\"] = np.where(\n    bbox_mask,\n    df[\"x_max\"] - df[\"x_min\"],\n    np.nan\n)\n\ndf[\"box_height\"] = np.where(\n    bbox_mask,\n    df[\"y_max\"] - df[\"y_min\"],\n    np.nan\n)\n\ndf[\"box_area\"] = (\n    df[\"box_width\"] *\n    df[\"box_height\"]\n)\n\n# ------------------------------------------------------------\n# 3. Normalized bounding-box dimensions\n# ------------------------------------------------------------\n\ndf[\"box_width_norm\"] = (\n    df[\"box_width\"] / df[\"width\"]\n)\n\ndf[\"box_height_norm\"] = (\n    df[\"box_height\"] / df[\"height\"]\n)\n\ndf[\"box_area_norm\"] = (\n    df[\"box_area\"] /\n    (df[\"width\"] * df[\"height\"])\n)\n\n# ------------------------------------------------------------\n# 4. Bounding-box center\n# ------------------------------------------------------------\n\ndf[\"center_x\"] = (\n    df[\"x_min\"] + df[\"x_max\"]\n) / 2\n\ndf[\"center_y\"] = (\n    df[\"y_min\"] + df[\"y_max\"]\n) / 2\n\n# ------------------------------------------------------------\n# 5. Normalize center to 0–1\n# ------------------------------------------------------------\n\ndf[\"center_x_norm\"] = (\n    df[\"center_x\"] / df[\"width\"]\n)\n\ndf[\"center_y_norm\"] = (\n    df[\"center_y\"] / df[\"height\"]\n)\n\n# ------------------------------------------------------------\n# 6. Basic validity checks\n# ------------------------------------------------------------\n\nprint(\"\\nBounding-box feature ranges:\")\n\ndisplay(\n    df.loc[bbox_mask, [\n        \"box_width_norm\",\n        \"box_height_norm\",\n        \"box_area_norm\",\n        \"center_x_norm\",\n        \"center_y_norm\"\n    ]].describe()\n)\n\n# ------------------------------------------------------------\n# 7. Check for invalid normalized values\n# ------------------------------------------------------------\n\nfeature_cols = [\n    \"box_width_norm\",\n    \"box_height_norm\",\n    \"box_area_norm\",\n    \"center_x_norm\",\n    \"center_y_norm\"\n]\n\nprint(\"\\nInvalid values:\")\n\nfor col in feature_cols:\n    invalid = (\n        (df[col] < 0) |\n        (df[col] > 1)\n    ).sum()\n\n    print(f\"{col}: {invalid}\")\n\n# ------------------------------------------------------------\n# 8. Preview\n# ------------------------------------------------------------\n\nprint(\"\\nFeature preview:\")\n\ndisplay(\n    df.loc[\n        bbox_mask,\n        [\n            \"image_id\",\n            \"class_name\",\n            \"rad_id\",\n            \"box_width_norm\",\n            \"box_height_norm\",\n            \"box_area_norm\",\n            \"center_x_norm\",\n            \"center_y_norm\"\n        ]\n    ].head(20)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:23:30.797467Z","iopub.execute_input":"2026-08-23T02:23:30.797973Z","iopub.status.idle":"2026-08-23T02:23:30.884534Z","shell.execute_reply.started":"2026-08-23T02:23:30.797912Z","shell.execute_reply":"2026-08-23T02:23:30.88339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CONAN IMAGING — STEP 4\n# Build image-level Nodule/Mass features\n# ============================================================\n\n# ------------------------------------------------------------\n# 1. Select Nodule/Mass annotations\n# ------------------------------------------------------------\n\nnm = df[\n    df[\"class_name\"] == \"Nodule/Mass\"\n].copy()\n\nprint(\"Nodule/Mass annotation rows:\", len(nm))\nprint(\"Unique Nodule/Mass images:\", nm[\"image_id\"].nunique())\n\n# ------------------------------------------------------------\n# 2. Number of unique radiologists per image\n# ------------------------------------------------------------\n\nnm_radiologists = (\n    nm.groupby(\"image_id\")[\"rad_id\"]\n    .nunique()\n    .rename(\"nodule_mass_radiologists\")\n)\n\n# ------------------------------------------------------------\n# 3. Annotation count per image\n# ------------------------------------------------------------\n\nnm_annotations = (\n    nm.groupby(\"image_id\")\n    .size()\n    .rename(\"nodule_mass_annotation_count\")\n)\n\n# ------------------------------------------------------------\n# 4. Radiologist agreement\n# ------------------------------------------------------------\n\nnm_agreement = (\n    nm_radiologists / 3.0\n).rename(\"nodule_mass_agreement\")\n\n# ------------------------------------------------------------\n# 5. Aggregate normalized box measurements\n# ------------------------------------------------------------\n\nnm_size_location = (\n    nm.groupby(\"image_id\")\n    .agg(\n        nodule_mass_max_width=(\n            \"box_width_norm\",\n            \"max\"\n        ),\n\n        nodule_mass_mean_width=(\n            \"box_width_norm\",\n            \"mean\"\n        ),\n\n        nodule_mass_max_height=(\n            \"box_height_norm\",\n            \"max\"\n        ),\n\n        nodule_mass_mean_height=(\n            \"box_height_norm\",\n            \"mean\"\n        ),\n\n        nodule_mass_max_area=(\n            \"box_area_norm\",\n            \"max\"\n        ),\n\n        nodule_mass_mean_area=(\n            \"box_area_norm\",\n            \"mean\"\n        ),\n\n        nodule_mass_mean_x=(\n            \"center_x_norm\",\n            \"mean\"\n        ),\n\n        nodule_mass_mean_y=(\n            \"center_y_norm\",\n            \"mean\"\n        ),\n\n        nodule_mass_min_x=(\n            \"center_x_norm\",\n            \"min\"\n        ),\n\n        nodule_mass_max_x=(\n            \"center_x_norm\",\n            \"max\"\n        ),\n\n        nodule_mass_min_y=(\n            \"center_y_norm\",\n            \"min\"\n        ),\n\n        nodule_mass_max_y=(\n            \"center_y_norm\",\n            \"max\"\n        )\n    )\n)\n\n# ------------------------------------------------------------\n# 6. Combine Nodule/Mass features\n# ------------------------------------------------------------\n\nnodule_features = pd.concat(\n    [\n        nm_radiologists,\n        nm_annotations,\n        nm_agreement,\n        nm_size_location\n    ],\n    axis=1\n).reset_index()\n\n# ------------------------------------------------------------\n# 7. Explicit presence flag\n# ------------------------------------------------------------\n\nnodule_features[\"nodule_mass_present\"] = 1\n\n# ------------------------------------------------------------\n# 8. Display result\n# ------------------------------------------------------------\n\nprint(\"\\nNodule/Mass image-level dataset:\")\nprint(\"Rows:\", len(nodule_features))\nprint(\"Columns:\", len(nodule_features.columns))\n\ndisplay(nodule_features.head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:23:51.630189Z","iopub.execute_input":"2026-08-23T02:23:51.630607Z","iopub.status.idle":"2026-08-23T02:23:51.694747Z","shell.execute_reply.started":"2026-08-23T02:23:51.630574Z","shell.execute_reply":"2026-08-23T02:23:51.693905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# VERIFY NODULE/MASS AGREEMENT\n# ============================================================\n\nprint(\"Radiologists identifying Nodule/Mass:\")\n\ndisplay(\n    nodule_features[\n        \"nodule_mass_radiologists\"\n    ]\n    .value_counts()\n    .sort_index()\n    .rename_axis(\"radiologists\")\n    .reset_index(name=\"images\")\n)\n\nprint(\"\\nAgreement distribution:\")\n\ndisplay(\n    nodule_features[\n        \"nodule_mass_agreement\"\n    ]\n    .value_counts()\n    .sort_index()\n    .rename_axis(\"agreement\")\n    .reset_index(name=\"images\")\n)\n\nprint(\"\\nAnnotation count distribution:\")\n\ndisplay(\n    nodule_features[\n        \"nodule_mass_annotation_count\"\n    ].describe()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:24:09.936418Z","iopub.execute_input":"2026-08-23T02:24:09.936824Z","iopub.status.idle":"2026-08-23T02:24:09.963232Z","shell.execute_reply.started":"2026-08-23T02:24:09.936791Z","shell.execute_reply":"2026-08-23T02:24:09.962149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CONAN IMAGING — COMPLETE NODULE/MASS FEATURE TABLE\n# ============================================================\n\n# Get one row per image from the original dataset\nimage_base = (\n    df[\n        [\n            \"image_id\",\n            \"height\",\n            \"width\",\n            \"photometric\"\n        ]\n    ]\n    .drop_duplicates(\"image_id\")\n    .copy()\n)\n\nprint(\"Unique images:\", len(image_base))\n\n# Merge Nodule/Mass features\nimaging_nm = image_base.merge(\n    nodule_features,\n    on=\"image_id\",\n    how=\"left\"\n)\n\n# ------------------------------------------------------------\n# Fill values for images without Nodule/Mass\n# ------------------------------------------------------------\n\nzero_columns = [\n    \"nodule_mass_radiologists\",\n    \"nodule_mass_annotation_count\",\n    \"nodule_mass_agreement\",\n    \"nodule_mass_present\"\n]\n\nfor col in zero_columns:\n    imaging_nm[col] = (\n        imaging_nm[col]\n        .fillna(0)\n    )\n\n# Size/location features are undefined when\n# no Nodule/Mass exists.\n# For a risk model, zero provides a clear\n# \"no lesion evidence\" representation.\n\nfeature_zero_columns = [\n    \"nodule_mass_max_width\",\n    \"nodule_mass_mean_width\",\n    \"nodule_mass_max_height\",\n    \"nodule_mass_mean_height\",\n    \"nodule_mass_max_area\",\n    \"nodule_mass_mean_area\",\n    \"nodule_mass_mean_x\",\n    \"nodule_mass_mean_y\",\n    \"nodule_mass_min_x\",\n    \"nodule_mass_max_x\",\n    \"nodule_mass_min_y\",\n    \"nodule_mass_max_y\"\n]\n\nfor col in feature_zero_columns:\n    imaging_nm[col] = (\n        imaging_nm[col]\n        .fillna(0)\n    )\n\n# ------------------------------------------------------------\n# Final verification\n# ------------------------------------------------------------\n\nprint(\"\\nFINAL NODULE/MASS TABLE\")\nprint(\"-----------------------\")\nprint(\"Rows:\", len(imaging_nm))\nprint(\"Columns:\", len(imaging_nm.columns))\n\nprint(\"\\nNodule/Mass presence:\")\ndisplay(\n    imaging_nm[\n        \"nodule_mass_present\"\n    ]\n    .value_counts()\n    .sort_index()\n    .rename_axis(\"present\")\n    .reset_index(name=\"images\")\n)\n\nprint(\"\\nMissing values:\")\ndisplay(\n    imaging_nm.isnull().sum()\n)\n\nprint(\"\\nPreview:\")\ndisplay(\n    imaging_nm.head(10)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:24:25.893644Z","iopub.execute_input":"2026-08-23T02:24:25.894306Z","iopub.status.idle":"2026-08-23T02:24:25.964023Z","shell.execute_reply.started":"2026-08-23T02:24:25.894271Z","shell.execute_reply":"2026-08-23T02:24:25.962831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CONAN IMAGING — STEP 4 VERIFICATION\n# ============================================================\n\nprint(\"Nodule/Mass radiologist agreement\")\nprint(\"=\" * 50)\n\nagreement_table = (\n    nodule_features[\n        \"nodule_mass_radiologists\"\n    ]\n    .value_counts()\n    .sort_index()\n    .rename_axis(\"radiologists_identifying_nodule_mass\")\n    .reset_index(name=\"number_of_images\")\n)\n\ndisplay(agreement_table)\n\n\nprint(\"\\nAgreement proportions\")\nprint(\"=\" * 50)\n\nagreement_proportion = (\n    nodule_features[\n        \"nodule_mass_agreement\"\n    ]\n    .value_counts()\n    .sort_index()\n    .rename_axis(\"agreement\")\n    .reset_index(name=\"number_of_images\")\n)\n\nagreement_proportion[\"percentage\"] = (\n    agreement_proportion[\"number_of_images\"]\n    / len(nodule_features)\n    * 100\n)\n\ndisplay(agreement_proportion)\n\n\nprint(\"\\nAnnotation count statistics\")\nprint(\"=\" * 50)\n\ndisplay(\n    nodule_features[\n        \"nodule_mass_annotation_count\"\n    ].describe()\n)\n\n\nprint(\"\\nCheck total Nodule/Mass images:\")\nprint(\n    \"Total:\",\n    len(nodule_features)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:24:42.20256Z","iopub.execute_input":"2026-08-23T02:24:42.203505Z","iopub.status.idle":"2026-08-23T02:24:42.23163Z","shell.execute_reply.started":"2026-08-23T02:24:42.20347Z","shell.execute_reply":"2026-08-23T02:24:42.230524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CONAN IMAGING — STEP 5\n# Build complete image-level radiographic feature matrix\n# ============================================================\n\nimport numpy as np\nimport pandas as pd\n\n# ------------------------------------------------------------\n# 1. Define actual abnormality classes\n# ------------------------------------------------------------\n\nabnormalities = [\n    \"Aortic enlargement\",\n    \"Cardiomegaly\",\n    \"Pleural thickening\",\n    \"ILD\",\n    \"Nodule/Mass\",\n    \"Pulmonary fibrosis\",\n    \"Lung Opacity\",\n    \"Atelectasis\",\n    \"Other lesion\",\n    \"Infiltration\",\n    \"Pleural effusion\",\n    \"Calcification\",\n    \"Consolidation\",\n    \"Pneumothorax\"\n]\n\nprint(\"Number of abnormality classes:\", len(abnormalities))\nprint(abnormalities)\n\n# ------------------------------------------------------------\n# 2. Start with exactly one row per X-ray\n# ------------------------------------------------------------\n\nimage_features = (\n    df[\n        [\n            \"image_id\",\n            \"height\",\n            \"width\",\n            \"photometric\"\n        ]\n    ]\n    .drop_duplicates(\"image_id\")\n    .copy()\n)\n\nprint(\"\\nInitial image table:\")\nprint(\"Rows:\", len(image_features))\n\n# ------------------------------------------------------------\n# 3. Process each abnormality\n# ------------------------------------------------------------\n\nfor abnormality in abnormalities:\n\n    # Create a safe column prefix\n    prefix = (\n        abnormality\n        .lower()\n        .replace(\"/\", \"_\")\n        .replace(\" \", \"_\")\n    )\n\n    # Select annotations for this abnormality\n    subset = df[\n        df[\"class_name\"] == abnormality\n    ].copy()\n\n    if len(subset) == 0:\n        print(f\"WARNING: No annotations for {abnormality}\")\n        continue\n\n    # --------------------------------------------------------\n    # Radiologist count\n    # --------------------------------------------------------\n\n    radiologist_count = (\n        subset.groupby(\"image_id\")[\"rad_id\"]\n        .nunique()\n        .rename(\n            f\"{prefix}_radiologists\"\n        )\n    )\n\n    # --------------------------------------------------------\n    # Annotation count\n    # --------------------------------------------------------\n\n    annotation_count = (\n        subset.groupby(\"image_id\")\n        .size()\n        .rename(\n            f\"{prefix}_annotation_count\"\n        )\n    )\n\n    # --------------------------------------------------------\n    # Agreement\n    #\n    # Each VinBigData image was evaluated by 3 radiologists.\n    # --------------------------------------------------------\n\n    agreement = (\n        radiologist_count / 3.0\n    ).rename(\n        f\"{prefix}_agreement\"\n    )\n\n    # --------------------------------------------------------\n    # Spatial and size features\n    # --------------------------------------------------------\n\n    measurements = (\n        subset.groupby(\"image_id\")\n        .agg(\n\n            max_width=(\n                \"box_width_norm\",\n                \"max\"\n            ),\n\n            mean_width=(\n                \"box_width_norm\",\n                \"mean\"\n            ),\n\n            max_height=(\n                \"box_height_norm\",\n                \"max\"\n            ),\n\n            mean_height=(\n                \"box_height_norm\",\n                \"mean\"\n            ),\n\n            max_area=(\n                \"box_area_norm\",\n                \"max\"\n            ),\n\n            mean_area=(\n                \"box_area_norm\",\n                \"mean\"\n            ),\n\n            mean_x=(\n                \"center_x_norm\",\n                \"mean\"\n            ),\n\n            mean_y=(\n                \"center_y_norm\",\n                \"mean\"\n            )\n        )\n    )\n\n    # Rename measurement columns\n    measurements = measurements.rename(\n        columns={\n            \"max_width\":\n                f\"{prefix}_max_width\",\n\n            \"mean_width\":\n                f\"{prefix}_mean_width\",\n\n            \"max_height\":\n                f\"{prefix}_max_height\",\n\n            \"mean_height\":\n                f\"{prefix}_mean_height\",\n\n            \"max_area\":\n                f\"{prefix}_max_area\",\n\n            \"mean_area\":\n                f\"{prefix}_mean_area\",\n\n            \"mean_x\":\n                f\"{prefix}_mean_x\",\n\n            \"mean_y\":\n                f\"{prefix}_mean_y\"\n        }\n    )\n\n    # --------------------------------------------------------\n    # Combine this abnormality's features\n    # --------------------------------------------------------\n\n    abnormality_features = pd.concat(\n        [\n            radiologist_count,\n            annotation_count,\n            agreement,\n            measurements\n        ],\n        axis=1\n    ).reset_index()\n\n    # --------------------------------------------------------\n    # Presence indicator\n    # --------------------------------------------------------\n\n    abnormality_features[\n        f\"{prefix}_present\"\n    ] = 1\n\n    # --------------------------------------------------------\n    # Merge into master image table\n    # --------------------------------------------------------\n\n    image_features = image_features.merge(\n        abnormality_features,\n        on=\"image_id\",\n        how=\"left\",\n        validate=\"one_to_one\"\n    )\n\n    # --------------------------------------------------------\n    # Fill absence with zero\n    # --------------------------------------------------------\n\n    new_columns = [\n        c for c in abnormality_features.columns\n        if c != \"image_id\"\n    ]\n\n    image_features[new_columns] = (\n        image_features[new_columns]\n        .fillna(0)\n    )\n\n    print(\n        f\"{abnormality:25s} → \"\n        f\"{len(subset):5d} annotations | \"\n        f\"{subset['image_id'].nunique():5d} images\"\n    )\n\n# ------------------------------------------------------------\n# 4. Final dimensions\n# ------------------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"FINAL RADIOGRAPHIC FEATURE MATRIX\")\nprint(\"=\" * 70)\n\nprint(\"Images:\", len(image_features))\nprint(\"Features:\", len(image_features.columns))\n\nprint(\"\\nShape:\")\nprint(image_features.shape)\n\n# ------------------------------------------------------------\n# 5. Verify no duplicate image IDs\n# ------------------------------------------------------------\n\nprint(\n    \"\\nDuplicate image IDs:\",\n    image_features[\"image_id\"].duplicated().sum()\n)\n\n# ------------------------------------------------------------\n# 6. Missing-value check\n# ------------------------------------------------------------\n\nmissing = (\n    image_features\n    .isnull()\n    .sum()\n)\n\nprint(\"\\nColumns with missing values:\")\ndisplay(\n    missing[missing > 0]\n)\n\n# ------------------------------------------------------------\n# 7. Display presence counts\n# ------------------------------------------------------------\n\npresence_columns = [\n    c for c in image_features.columns\n    if c.endswith(\"_present\")\n]\n\npresence_summary = []\n\nfor col in presence_columns:\n\n    presence_summary.append({\n        \"feature\": col,\n        \"images_present\":\n            int(image_features[col].sum()),\n        \"percentage\":\n            float(\n                image_features[col].mean() * 100\n            )\n    })\n\npresence_summary = pd.DataFrame(\n    presence_summary\n).sort_values(\n    \"images_present\",\n    ascending=False\n)\n\nprint(\"\\nAbnormality presence:\")\ndisplay(presence_summary)\n\n# ------------------------------------------------------------\n# 8. Preview\n# ------------------------------------------------------------\n\nprint(\"\\nFeature matrix preview:\")\ndisplay(\n    image_features.head()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:25:02.030712Z","iopub.execute_input":"2026-08-23T02:25:02.031143Z","iopub.status.idle":"2026-08-23T02:25:02.994624Z","shell.execute_reply.started":"2026-08-23T02:25:02.031112Z","shell.execute_reply":"2026-08-23T02:25:02.993859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CONAN IMAGING — STEP 6\n# Create interpretable candidate imaging predictors\n# ============================================================\n\n# ------------------------------------------------------------\n# Abnormality prefixes\n# ------------------------------------------------------------\n\nabnormality_prefixes = [\n    \"aortic_enlargement\",\n    \"cardiomegaly\",\n    \"pleural_thickening\",\n    \"ild\",\n    \"nodule_mass\",\n    \"pulmonary_fibrosis\",\n    \"lung_opacity\",\n    \"atelectasis\",\n    \"other_lesion\",\n    \"infiltration\",\n    \"pleural_effusion\",\n    \"calcification\",\n    \"consolidation\",\n    \"pneumothorax\"\n]\n\n# ------------------------------------------------------------\n# Candidate features\n#\n# We intentionally DO NOT include:\n# - radiologists\n# - annotation_count\n#\n# because these are highly related to agreement and can\n# introduce redundant information.\n# ------------------------------------------------------------\n\ncandidate_columns = [\n    \"image_id\",\n    \"height\",\n    \"width\",\n    \"photometric\"\n]\n\nfor prefix in abnormality_prefixes:\n\n    candidate_columns.extend([\n        f\"{prefix}_present\",\n        f\"{prefix}_agreement\",\n        f\"{prefix}_mean_area\",\n        f\"{prefix}_mean_width\",\n        f\"{prefix}_mean_height\",\n        f\"{prefix}_mean_x\",\n        f\"{prefix}_mean_y\"\n    ])\n\ncandidate_features = image_features[\n    candidate_columns\n].copy()\n\n# ------------------------------------------------------------\n# Verify\n# ------------------------------------------------------------\n\nprint(\"=\" * 70)\nprint(\"CANDIDATE IMAGING FEATURE MATRIX\")\nprint(\"=\" * 70)\n\nprint(\"Rows:\", candidate_features.shape[0])\nprint(\"Columns:\", candidate_features.shape[1])\n\nprint(\"\\nShape:\")\nprint(candidate_features.shape)\n\nprint(\"\\nMissing values:\")\nprint(\n    candidate_features.isnull().sum().sum()\n)\n\nprint(\"\\nDuplicate image IDs:\")\nprint(\n    candidate_features[\"image_id\"].duplicated().sum()\n)\n\n# ------------------------------------------------------------\n# Count candidate predictors\n# ------------------------------------------------------------\n\npredictor_columns = [\n    c for c in candidate_features.columns\n    if c not in [\n        \"image_id\",\n        \"height\",\n        \"width\",\n        \"photometric\"\n    ]\n]\n\nprint(\"\\nNumber of candidate predictors:\")\nprint(len(predictor_columns))\n\n# ------------------------------------------------------------\n# Show columns\n# ------------------------------------------------------------\n\nprint(\"\\nCandidate predictors:\")\n\nfor i, col in enumerate(predictor_columns, 1):\n    print(f\"{i:02d}. {col}\")\n\n# ------------------------------------------------------------\n# Save for next step\n# ------------------------------------------------------------\n\ncandidate_features.to_csv(\n    \"/kaggle/working/conan_imaging_candidate_features.csv\",\n    index=False\n)\n\nprint(\n    \"\\nSaved:\"\n    \"\\n/kaggle/working/conan_imaging_candidate_features.csv\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:25:24.473796Z","iopub.execute_input":"2026-08-23T02:25:24.474585Z","iopub.status.idle":"2026-08-23T02:25:25.454889Z","shell.execute_reply.started":"2026-08-23T02:25:24.474543Z","shell.execute_reply":"2026-08-23T02:25:25.453795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"Searching for JSRT metadata...\")\nprint(\"=\" * 70)\n\nmatches = []\n\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    for file in files:\n        if \"jsrt\" in file.lower() or \"metadata\" in file.lower():\n            matches.append(os.path.join(root, file))\n\nfor root, dirs, files in os.walk(\"/kaggle/working\"):\n    for file in files:\n        if \"jsrt\" in file.lower() or \"metadata\" in file.lower():\n            matches.append(os.path.join(root, file))\n\nif matches:\n    print(\"FOUND:\")\n    for path in matches:\n        print(path)\nelse:\n    print(\"NO JSRT FILE FOUND.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:26:11.150063Z","iopub.execute_input":"2026-08-23T02:26:11.150387Z","iopub.status.idle":"2026-08-23T02:26:17.414837Z","shell.execute_reply.started":"2026-08-23T02:26:11.15036Z","shell.execute_reply":"2026-08-23T02:26:17.413863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nFEATURE_PATH = \"/kaggle/working/conan_imaging_candidate_features.csv\"\n\ndf = pd.read_csv(FEATURE_PATH)\n\nprint(\"=\" * 70)\nprint(\"CONAN IMAGING FEATURE MATRIX\")\nprint(\"=\" * 70)\n\nprint(\"Shape:\", df.shape)\nprint(\"\\nColumns:\", len(df.columns))\nprint(\"\\nMissing values:\", df.isnull().sum().sum())\nprint(\"Duplicate image IDs:\", df[\"image_id\"].duplicated().sum())\n\nprint(\"\\nFirst 5 rows:\")\ndisplay(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:29:29.415927Z","iopub.execute_input":"2026-08-23T02:29:29.416278Z","iopub.status.idle":"2026-08-23T02:29:29.656649Z","shell.execute_reply.started":"2026-08-23T02:29:29.416242Z","shell.execute_reply":"2026-08-23T02:29:29.655867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nFEATURE_PATH = \"/kaggle/working/conan_imaging_candidate_features.csv\"\n\ndf = pd.read_csv(FEATURE_PATH)\n\n# Metadata — NOT model predictors\nmetadata_cols = [\n    \"image_id\",\n    \"height\",\n    \"width\",\n    \"photometric\"\n]\n\n# Candidate imaging predictors\npredictor_cols = [\n    c for c in df.columns\n    if c not in metadata_cols\n]\n\nX = df[predictor_cols].copy()\n\nprint(\"=\" * 70)\nprint(\"CONAN — IMAGING PREDICTOR SET\")\nprint(\"=\" * 70)\n\nprint(\"Total images:\", len(df))\nprint(\"Metadata columns:\", len(metadata_cols))\nprint(\"Candidate predictors:\", len(predictor_cols))\nprint(\"Predictor matrix:\", X.shape)\n\nprint(\"\\nMetadata:\")\nprint(metadata_cols)\n\nprint(\"\\nPredictor types:\")\nprint(X.dtypes.value_counts())\n\nprint(\"\\nConstant predictors:\")\nconstant_cols = [\n    c for c in predictor_cols\n    if X[c].nunique(dropna=False) <= 1\n]\n\nprint(\"Count:\", len(constant_cols))\n\nif constant_cols:\n    print(constant_cols)\n\nprint(\"\\nPredictors with non-zero values:\")\nfor c in predictor_cols:\n    nonzero = (X[c] != 0).sum()\n    if nonzero > 0:\n        print(f\"{c:45s} {nonzero:6d} ({nonzero/len(X)*100:6.2f}%)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:29:41.840588Z","iopub.execute_input":"2026-08-23T02:29:41.840976Z","iopub.status.idle":"2026-08-23T02:29:42.076281Z","shell.execute_reply.started":"2026-08-23T02:29:41.840927Z","shell.execute_reply":"2026-08-23T02:29:42.075139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Load candidate matrix\nFEATURE_PATH = \"/kaggle/working/conan_imaging_candidate_features.csv\"\ndf = pd.read_csv(FEATURE_PATH)\n\nmetadata_cols = [\n    \"image_id\",\n    \"height\",\n    \"width\",\n    \"photometric\"\n]\n\npredictor_cols = [\n    c for c in df.columns\n    if c not in metadata_cols\n]\n\nX = df[predictor_cols].copy()\n\nprint(\"=\" * 70)\nprint(\"CONAN — IMAGING FEATURE REDUNDANCY ANALYSIS\")\nprint(\"=\" * 70)\n\n# ------------------------------------------------------------\n# 1. Correlation matrix\n# ------------------------------------------------------------\n\ncorr = X.corr().abs()\n\n# Ignore self-correlation\nnp.fill_diagonal(corr.values, 0)\n\n# Find highly correlated pairs\nthreshold = 0.90\n\nhigh_corr_pairs = []\n\nfor i in range(len(corr.columns)):\n    for j in range(i + 1, len(corr.columns)):\n\n        value = corr.iloc[i, j]\n\n        if value >= threshold:\n            high_corr_pairs.append({\n                \"feature_1\": corr.columns[i],\n                \"feature_2\": corr.columns[j],\n                \"abs_correlation\": value\n            })\n\nhigh_corr = pd.DataFrame(high_corr_pairs)\n\nif len(high_corr) > 0:\n    high_corr = high_corr.sort_values(\n        \"abs_correlation\",\n        ascending=False\n    )\n\nprint(\"\\nHighly correlated pairs (|r| >= 0.90):\")\nprint(\"Count:\", len(high_corr))\n\nif len(high_corr) > 0:\n    display(high_corr.head(50))\nelse:\n    print(\"No highly correlated pairs found.\")\n\n# ------------------------------------------------------------\n# 2. Feature variance\n# ------------------------------------------------------------\n\nvariance = X.var().sort_values()\n\nprint(\"\\nLowest-variance predictors:\")\ndisplay(\n    variance.head(20).to_frame(\"variance\")\n)\n\n# ------------------------------------------------------------\n# 3. Number of unique values\n# ------------------------------------------------------------\n\nunique_counts = X.nunique().sort_values()\n\nprint(\"\\nPredictors with the fewest unique values:\")\ndisplay(\n    unique_counts.head(30).to_frame(\"unique_values\")\n)\n\n# ------------------------------------------------------------\n# 4. Summary\n# ------------------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"SUMMARY\")\nprint(\"=\" * 70)\n\nprint(\"Total predictors:\", len(predictor_cols))\nprint(\"Highly correlated pairs:\", len(high_corr))\nprint(\"Minimum variance:\", variance.min())\nprint(\"Maximum variance:\", variance.max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:29:58.180903Z","iopub.execute_input":"2026-08-23T02:29:58.185947Z","iopub.status.idle":"2026-08-23T02:29:59.531256Z","shell.execute_reply.started":"2026-08-23T02:29:58.185556Z","shell.execute_reply":"2026-08-23T02:29:59.530234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nprint(\"=\" * 70)\nprint(\"CONAN — REDUCED IMAGING FEATURE SET\")\nprint(\"=\" * 70)\n\n# ============================================================\n# 1. LOCATE THE EXISTING CANDIDATE FEATURE MATRIX\n# ============================================================\n\ncandidate_path = \"/kaggle/working/conan_imaging_candidate_features.csv\"\n\ndf = pd.read_csv(candidate_path)\n\nprint(\"\\nLoaded:\")\nprint(candidate_path)\n\nprint(\"\\nOriginal dataset shape:\")\nprint(df.shape)\n\n\n# ============================================================\n# 2. DEFINE THE 14 RADIOGRAPHIC ABNORMALITIES\n# ============================================================\n\nabnormalities = [\n    \"aortic_enlargement\",\n    \"cardiomegaly\",\n    \"pleural_thickening\",\n    \"ild\",\n    \"nodule_mass\",\n    \"pulmonary_fibrosis\",\n    \"lung_opacity\",\n    \"atelectasis\",\n    \"other_lesion\",\n    \"infiltration\",\n    \"pleural_effusion\",\n    \"calcification\",\n    \"consolidation\",\n    \"pneumothorax\"\n]\n\n\n# ============================================================\n# 3. SELECT REDUCED PREDICTORS\n# ============================================================\n\nselected_features = []\n\nfor abnormality in abnormalities:\n\n    selected_features.extend([\n        f\"{abnormality}_present\",\n        f\"{abnormality}_agreement\",\n        f\"{abnormality}_mean_area\"\n    ])\n\n\n# ============================================================\n# 4. METADATA\n# ============================================================\n\nmetadata = [\n    \"image_id\",\n    \"height\",\n    \"width\",\n    \"photometric\"\n]\n\n\n# ============================================================\n# 5. VERIFY REQUIRED COLUMNS\n# ============================================================\n\nrequired_columns = metadata + selected_features\n\nmissing_columns = [\n    col for col in required_columns\n    if col not in df.columns\n]\n\nif missing_columns:\n\n    print(\"\\nERROR — Missing columns:\")\n    \n    for col in missing_columns:\n        print(\" -\", col)\n\n    raise ValueError(\n        f\"{len(missing_columns)} required columns are missing.\"\n    )\n\n\n# ============================================================\n# 6. CREATE REDUCED MATRIX\n# ============================================================\n\nreduced_df = df[required_columns].copy()\n\n\n# ============================================================\n# 7. BASIC VALIDATION\n# ============================================================\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"REDUCED FEATURE MATRIX\")\nprint(\"=\" * 70)\n\nprint(\"\\nRows:\")\nprint(len(reduced_df))\n\nprint(\"\\nMetadata columns:\")\nprint(len(metadata))\n\nprint(\"\\nPredictor columns:\")\nprint(len(selected_features))\n\nprint(\"\\nTotal columns:\")\nprint(len(reduced_df.columns))\n\nprint(\"\\nShape:\")\nprint(reduced_df.shape)\n\nprint(\"\\nMissing values:\")\nprint(\n    reduced_df[selected_features]\n    .isnull()\n    .sum()\n    .sum()\n)\n\nprint(\"\\nDuplicate image IDs:\")\nprint(\n    reduced_df[\"image_id\"]\n    .duplicated()\n    .sum()\n)\n\n\n# ============================================================\n# 8. CHECK THAT EACH IMAGE OCCURS ONCE\n# ============================================================\n\nunique_images = reduced_df[\"image_id\"].nunique()\n\nprint(\"\\nUnique images:\")\nprint(unique_images)\n\nif unique_images != len(reduced_df):\n\n    print(\"\\nWARNING:\")\n    print(\"Some image IDs occur more than once.\")\n\nelse:\n\n    print(\"\\n✓ One row per image confirmed.\")\n\n\n# ============================================================\n# 9. SHOW SELECTED PREDICTORS\n# ============================================================\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"SELECTED IMAGING PREDICTORS\")\nprint(\"=\" * 70)\n\nfor i, feature in enumerate(selected_features, 1):\n    print(f\"{i:02d}. {feature}\")\n\n\n# ============================================================\n# 10. CHECK PREDICTOR TYPES\n# ============================================================\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"PREDICTOR TYPES\")\nprint(\"=\" * 70)\n\nprint(\n    reduced_df[selected_features]\n    .dtypes\n    .value_counts()\n)\n\n\n# ============================================================\n# 11. SAVE\n# ============================================================\n\noutput_path = \"/kaggle/working/conan_imaging_reduced_features.csv\"\n\nreduced_df.to_csv(\n    output_path,\n    index=False\n)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"SAVED\")\nprint(\"=\" * 70)\n\nprint(output_path)\n\n\n# ============================================================\n# 12. PREVIEW\n# ============================================================\n\nprint(\"\\nFirst 5 rows:\")\n\ndisplay(\n    reduced_df.head()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:30:49.304809Z","iopub.execute_input":"2026-08-23T02:30:49.305201Z","iopub.status.idle":"2026-08-23T02:30:49.949557Z","shell.execute_reply.started":"2026-08-23T02:30:49.30517Z","shell.execute_reply":"2026-08-23T02:30:49.948051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# CONAN — IMAGING TARGET PREPARATION\n# ================================================================\n\nimport os\nimport pandas as pd\nimport numpy as np\n\nprint(\"=\" * 70)\nprint(\"CONAN — IMAGING TARGET PREPARATION\")\nprint(\"=\" * 70)\n\n# ------------------------------------------------\n# 1. Load reduced VinBigData feature matrix\n# ------------------------------------------------\n\nFEATURE_PATH = \"/kaggle/working/conan_imaging_reduced_features.csv\"\n\nif not os.path.exists(FEATURE_PATH):\n    raise FileNotFoundError(\n        f\"Feature matrix not found:\\n{FEATURE_PATH}\"\n    )\n\ndf = pd.read_csv(FEATURE_PATH)\n\nprint(\"\\nLoaded:\")\nprint(FEATURE_PATH)\n\nprint(\"\\nDataset shape:\")\nprint(df.shape)\n\nprint(\"\\nColumns:\")\nprint(df.columns.tolist())\n\n# ------------------------------------------------\n# 2. Basic validation\n# ------------------------------------------------\n\nrequired_metadata = [\n    \"image_id\",\n    \"height\",\n    \"width\",\n    \"photometric\"\n]\n\nmissing_metadata = [\n    c for c in required_metadata\n    if c not in df.columns\n]\n\nif missing_metadata:\n    raise ValueError(\n        f\"Missing metadata columns: {missing_metadata}\"\n    )\n\nprint(\"\\nMetadata validation:\")\nprint(\"✓ image_id\")\nprint(\"✓ height\")\nprint(\"✓ width\")\nprint(\"✓ photometric\")\n\n# ------------------------------------------------\n# 3. Check image IDs\n# ------------------------------------------------\n\nprint(\"\\nImage ID validation\")\n\nprint(\"Total rows:\", len(df))\nprint(\"Unique image IDs:\", df[\"image_id\"].nunique())\nprint(\"Duplicate IDs:\", df[\"image_id\"].duplicated().sum())\n\nassert df[\"image_id\"].nunique() == len(df)\n\nprint(\"✓ One row per image\")\n\n# ------------------------------------------------\n# 4. Check target-like abnormalities\n# ------------------------------------------------\n\nprint(\"\\nNodule/Mass radiographic evidence\")\n\nif \"nodule_mass_present\" in df.columns:\n\n    nodule_count = int(\n        df[\"nodule_mass_present\"].sum()\n    )\n\n    total = len(df)\n\n    print(\n        f\"Nodule/Mass positive: {nodule_count:,} \"\n        f\"({100*nodule_count/total:.2f}%)\"\n    )\n\n    print(\n        f\"Nodule/Mass negative: {total-nodule_count:,} \"\n        f\"({100*(total-nodule_count)/total:.2f}%)\"\n    )\n\nprint(\"\\nIMPORTANT:\")\nprint(\"nodule_mass_present is NOT being used as a lung-cancer label.\")\nprint(\"It is treated as radiographic evidence only.\")\n\n# ------------------------------------------------\n# 5. Check candidate target columns\n# ------------------------------------------------\n\npossible_targets = [\n    \"lung_cancer\",\n    \"LUNG_CANCER\",\n    \"cancer\",\n    \"Cancer\",\n    \"diagnosis\",\n    \"Diagnosis\",\n    \"risk_label\"\n]\n\nfound_targets = [\n    c for c in possible_targets\n    if c in df.columns\n]\n\nprint(\"\\nPossible cancer target columns:\")\nprint(found_targets)\n\nif not found_targets:\n\n    print(\"\\n✓ No cancer target is embedded in VinBigData.\")\n    print(\"\\nThis is expected.\")\n    print(\"A separate cancer-labeled dataset is required.\")\n\n# ------------------------------------------------\n# 6. Save validated feature matrix\n# ------------------------------------------------\n\nOUTPUT = \"/kaggle/working/conan_imaging_features_validated.csv\"\n\ndf.to_csv(\n    OUTPUT,\n    index=False\n)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"VALIDATED FEATURE MATRIX SAVED\")\nprint(\"=\" * 70)\n\nprint(OUTPUT)\nprint(\"Shape:\", df.shape)\n\n# ------------------------------------------------\n# 7. Final architecture reminder\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"CONAN IMAGING PIPELINE\")\nprint(\"=\" * 70)\n\nprint(\"\"\"\nVinBigData\n    ↓\n14 radiographic abnormalities\n    ↓\n98 candidate predictors\n    ↓\nredundancy / variance analysis\n    ↓\n42 reduced predictors\n    ↓\n15,000-image radiographic feature matrix\n    ↓\n        ┌─────────────────────────────┐\n        │                              │\n        │  Separate cancer-labeled    │\n        │  dataset (JSRT / cancer)    │\n        │                              │\n        └──────────────┬──────────────┘\n                       ↓\n             Lung-cancer target\n                       ↓\n              Imaging risk model\n                       ↓\n          P(lung cancer | X-ray)\n\"\"\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:31:12.742813Z","iopub.execute_input":"2026-08-23T02:31:12.74319Z","iopub.status.idle":"2026-08-23T02:31:13.292806Z","shell.execute_reply.started":"2026-08-23T02:31:12.74316Z","shell.execute_reply":"2026-08-23T02:31:13.291358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# CONAN — SEARCH FOR CANCER-LABELED IMAGING DATA\n# ================================================================\n\nimport os\nimport glob\nimport pandas as pd\n\nprint(\"=\" * 70)\nprint(\"CONAN — SEARCHING FOR CANCER-LABELED IMAGING DATA\")\nprint(\"=\" * 70)\n\nsearch_roots = [\n    \"/kaggle/input\",\n    \"/kaggle/working\"\n]\n\n# ------------------------------------------------\n# 1. Search filenames\n# ------------------------------------------------\n\nkeywords = [\n    \"jsrt\",\n    \"cancer\",\n    \"lung\",\n    \"diagnosis\",\n    \"metadata\",\n    \"nodule\"\n]\n\nmatches = []\n\nfor root in search_roots:\n    for dirpath, dirnames, filenames in os.walk(root):\n\n        for filename in filenames:\n\n            lower = filename.lower()\n\n            if any(k in lower for k in keywords):\n\n                matches.append(\n                    os.path.join(dirpath, filename)\n                )\n\nprint(\"\\nPotentially relevant files:\")\nprint(\"-\" * 70)\n\nif matches:\n    for path in sorted(matches):\n        print(path)\nelse:\n    print(\"NONE FOUND\")\n\n# ------------------------------------------------\n# 2. Search directories\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"RELEVANT DIRECTORIES\")\nprint(\"=\" * 70)\n\ndirectory_matches = []\n\nfor root in search_roots:\n    for dirpath, dirnames, filenames in os.walk(root):\n\n        lower_path = dirpath.lower()\n\n        if any(k in lower_path for k in keywords):\n\n            directory_matches.append(dirpath)\n\nfor path in sorted(set(directory_matches)):\n    print(path)\n\n# ------------------------------------------------\n# 3. Look specifically for CSV files\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"CSV FILES WITH POSSIBLE LABEL INFORMATION\")\nprint(\"=\" * 70)\n\ncsv_files = []\n\nfor root in search_roots:\n    csv_files.extend(\n        glob.glob(\n            os.path.join(root, \"**\", \"*.csv\"),\n            recursive=True\n        )\n    )\n\nfor path in sorted(set(csv_files)):\n\n    filename = os.path.basename(path).lower()\n\n    if any(\n        k in filename\n        for k in [\n            \"jsrt\",\n            \"cancer\",\n            \"lung\",\n            \"diagnosis\",\n            \"metadata\",\n            \"label\"\n        ]\n    ):\n\n        print(path)\n\n# ------------------------------------------------\n# 4. Search for known JSRT metadata columns\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"SEARCHING CSV CONTENT\")\nprint(\"=\" * 70)\n\nknown_jsrt_columns = {\n    \"study_id\",\n    \"subtlety\",\n    \"size\",\n    \"age\",\n    \"gender\",\n    \"x\",\n    \"y\",\n    \"state\",\n    \"position\",\n    \"diagnosis\"\n}\n\nfor path in sorted(set(csv_files)):\n\n    try:\n        sample = pd.read_csv(\n            path,\n            nrows=5\n        )\n\n        cols = set(sample.columns)\n\n        overlap = cols.intersection(\n            known_jsrt_columns\n        )\n\n        if len(overlap) >= 3:\n\n            print(\"\\nPOSSIBLE JSRT FILE:\")\n            print(path)\n\n            print(\"Columns:\")\n            print(sample.columns.tolist())\n\n            print(\n                \"Matching JSRT columns:\",\n                sorted(overlap)\n            )\n\n    except Exception:\n        pass\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"SEARCH COMPLETE\")\nprint(\"=\" * 70)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:31:28.674002Z","iopub.execute_input":"2026-08-23T02:31:28.67437Z","iopub.status.idle":"2026-08-23T02:31:35.819178Z","shell.execute_reply.started":"2026-08-23T02:31:28.674329Z","shell.execute_reply":"2026-08-23T02:31:35.818322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"=\" * 70)\nprint(\"CONAN — CHECKING AVAILABLE KAGGLE DATASETS\")\nprint(\"=\" * 70)\n\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n\n    # Only show top-level dataset folders\n    depth = root.replace(\"/kaggle/input\", \"\").count(os.sep)\n\n    if depth <= 1:\n        print(root)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"DATASET FOLDERS\")\nprint(\"=\" * 70)\n\nfor item in sorted(os.listdir(\"/kaggle/input\")):\n    path = os.path.join(\"/kaggle/input\", item)\n\n    if os.path.isdir(path):\n        print(\"📁\", item)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:31:46.479964Z","iopub.execute_input":"2026-08-23T02:31:46.480281Z","iopub.status.idle":"2026-08-23T02:31:46.556351Z","shell.execute_reply.started":"2026-08-23T02:31:46.480254Z","shell.execute_reply":"2026-08-23T02:31:46.553302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# CONAN — JSRT LUNG CANCER TARGET CONSTRUCTION\n# ================================================================\n\nimport pandas as pd\nimport numpy as np\n\nprint(\"=\" * 70)\nprint(\"CONAN — JSRT LUNG CANCER TARGET CONSTRUCTION\")\nprint(\"=\" * 70)\n\n# ------------------------------------------------\n# Load JSRT metadata\n# ------------------------------------------------\n\nJSRT_PATH = \"/kaggle/input/jsrt-metadata/jsrt_metadata.csv\"\n\n# If your Kaggle path is different, automatically locate it\nimport os\nimport glob\n\nif not os.path.exists(JSRT_PATH):\n\n    candidates = glob.glob(\n        \"/kaggle/input/**/*.csv\",\n        recursive=True\n    )\n\n    jsrt_candidates = []\n\n    for path in candidates:\n\n        try:\n            temp = pd.read_csv(path, nrows=5)\n\n            required = {\n                \"study_id\",\n                \"subtlety\",\n                \"size\",\n                \"age\",\n                \"gender\",\n                \"x\",\n                \"y\",\n                \"state\",\n                \"position\",\n                \"diagnosis\"\n            }\n\n            if required.issubset(set(temp.columns)):\n                jsrt_candidates.append(path)\n\n        except Exception:\n            pass\n\n    if len(jsrt_candidates) == 0:\n        raise FileNotFoundError(\n            \"Could not locate JSRT metadata.\"\n        )\n\n    JSRT_PATH = jsrt_candidates[0]\n\nprint(\"\\nUsing:\")\nprint(JSRT_PATH)\n\njsrt = pd.read_csv(JSRT_PATH)\n\nprint(\"\\nOriginal shape:\")\nprint(jsrt.shape)\n\n# ------------------------------------------------\n# Validate columns\n# ------------------------------------------------\n\nrequired_columns = [\n    \"study_id\",\n    \"subtlety\",\n    \"size\",\n    \"age\",\n    \"gender\",\n    \"x\",\n    \"y\",\n    \"state\",\n    \"position\",\n    \"diagnosis\"\n]\n\nmissing = [\n    c for c in required_columns\n    if c not in jsrt.columns\n]\n\nif missing:\n    raise ValueError(\n        f\"Missing columns: {missing}\"\n    )\n\nprint(\"\\n✓ Required JSRT columns confirmed\")\n\n# ------------------------------------------------\n# Inspect state\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"JSRT STATE DISTRIBUTION\")\nprint(\"=\" * 70)\n\nprint(\n    jsrt[\"state\"]\n    .value_counts(dropna=False)\n)\n\nprint(\"\\nPercentages:\")\nprint(\n    (jsrt[\"state\"].value_counts(normalize=True) * 100)\n    .round(2)\n)\n\n# ------------------------------------------------\n# Inspect diagnosis\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"JSRT DIAGNOSIS DISTRIBUTION\")\nprint(\"=\" * 70)\n\nprint(\n    jsrt[\"diagnosis\"]\n    .value_counts(dropna=False)\n)\n\n# ------------------------------------------------\n# Construct cancer target\n# ------------------------------------------------\n\njsrt[\"lung_cancer\"] = (\n    jsrt[\"state\"]\n    .astype(str)\n    .str.strip()\n    .str.lower()\n    .eq(\"malignant\")\n    .astype(int)\n)\n\n# ------------------------------------------------\n# Validate target\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"LUNG CANCER TARGET\")\nprint(\"=\" * 70)\n\nprint(\n    jsrt[\"lung_cancer\"]\n    .value_counts()\n    .sort_index()\n)\n\nprint(\"\\nTarget percentages:\")\n\ntarget_pct = (\n    jsrt[\"lung_cancer\"]\n    .value_counts(normalize=True)\n    .sort_index() * 100\n)\n\nprint(target_pct.round(2))\n\n# ------------------------------------------------\n# Check state ↔ target consistency\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"TARGET VALIDATION\")\nprint(\"=\" * 70)\n\nvalidation = pd.crosstab(\n    jsrt[\"state\"],\n    jsrt[\"lung_cancer\"]\n)\n\nprint(validation)\n\n# ------------------------------------------------\n# Missing values\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"MISSING VALUES\")\nprint(\"=\" * 70)\n\nprint(\n    jsrt[required_columns + [\"lung_cancer\"]]\n    .isna()\n    .sum()\n)\n\n# ------------------------------------------------\n# Duplicate studies\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"IMAGE ID VALIDATION\")\nprint(\"=\" * 70)\n\nprint(\"Rows:\", len(jsrt))\nprint(\n    \"Unique study IDs:\",\n    jsrt[\"study_id\"].nunique()\n)\n\nprint(\n    \"Duplicate study IDs:\",\n    jsrt[\"study_id\"].duplicated().sum()\n)\n\n# ------------------------------------------------\n# Save target dataset\n# ------------------------------------------------\n\nOUTPUT = \"/kaggle/working/conan_jsrt_cancer_target.csv\"\n\njsrt.to_csv(\n    OUTPUT,\n    index=False\n)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"SAVED\")\nprint(\"=\" * 70)\n\nprint(OUTPUT)\nprint(\"Shape:\", jsrt.shape)\n\nprint(\"\\nFirst 10 cases:\")\nprint(\n    jsrt[\n        [\n            \"study_id\",\n            \"age\",\n            \"gender\",\n            \"size\",\n            \"state\",\n            \"diagnosis\",\n            \"lung_cancer\"\n        ]\n    ].head(10).to_string(index=False)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:31:59.787849Z","iopub.execute_input":"2026-08-23T02:31:59.789049Z","iopub.status.idle":"2026-08-23T02:32:01.113675Z","shell.execute_reply.started":"2026-08-23T02:31:59.789006Z","shell.execute_reply":"2026-08-23T02:32:01.11264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"=\" * 70)\nprint(\"SEARCHING KAGGLE INPUT FOR JSRT METADATA\")\nprint(\"=\" * 70)\n\nmatches = []\n\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    for file in files:\n        if file.lower() == \"jsrt_metadata.csv\":\n            matches.append(os.path.join(root, file))\n\nprint(\"\\nMatches found:\")\nfor path in matches:\n    print(path)\n\nif not matches:\n    print(\"\\n❌ jsrt_metadata.csv is NOT mounted in this notebook.\")\n    print(\"\\nAvailable datasets:\")\n    for item in os.listdir(\"/kaggle/input\"):\n        print(\" -\", item)\nelse:\n    print(\"\\n✓ JSRT metadata found!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:32:47.048386Z","iopub.execute_input":"2026-08-23T02:32:47.048763Z","iopub.status.idle":"2026-08-23T02:32:52.356589Z","shell.execute_reply.started":"2026-08-23T02:32:47.048727Z","shell.execute_reply":"2026-08-23T02:32:52.355553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"=\" * 70)\nprint(\"SEARCHING FOR JSRT METADATA\")\nprint(\"=\" * 70)\n\nmatches = []\n\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    for file in files:\n        if file.lower() == \"jsrt_metadata.csv\":\n            matches.append(os.path.join(root, file))\n\nprint(\"\\nFound:\", len(matches))\n\nfor path in matches:\n    print(path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:33:05.051914Z","iopub.execute_input":"2026-08-23T02:33:05.052288Z","iopub.status.idle":"2026-08-23T02:33:06.304563Z","shell.execute_reply.started":"2026-08-23T02:33:05.052258Z","shell.execute_reply":"2026-08-23T02:33:06.30351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\nprint(\"=\" * 70)\nprint(\"CONAN — JSRT METADATA INSPECTION\")\nprint(\"=\" * 70)\n\n# Find jsrt_metadata.csv automatically\njsrt_candidates = []\n\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    for file in files:\n        if file.lower() == \"jsrt_metadata.csv\":\n            jsrt_candidates.append(os.path.join(root, file))\n\nif len(jsrt_candidates) == 0:\n    raise FileNotFoundError(\n        \"jsrt_metadata.csv was not found. \"\n        \"Please add the raddar JSRT dataset to Kaggle Inputs.\"\n    )\n\njsrt_path = jsrt_candidates[0]\n\nprint(\"\\nJSRT metadata found:\")\nprint(jsrt_path)\n\n# Load\njsrt = pd.read_csv(jsrt_path)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"JSRT DATASET\")\nprint(\"=\" * 70)\n\nprint(\"Shape:\", jsrt.shape)\n\nprint(\"\\nColumns:\")\nfor i, col in enumerate(jsrt.columns, 1):\n    print(f\"{i:02d}. {col}\")\n\nprint(\"\\nFirst 5 rows:\")\ndisplay(jsrt.head())\n\nprint(\"\\nData types:\")\nprint(jsrt.dtypes)\n\nprint(\"\\nMissing values:\")\nprint(jsrt.isna().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:33:26.398036Z","iopub.execute_input":"2026-08-23T02:33:26.398459Z","iopub.status.idle":"2026-08-23T02:33:26.485898Z","shell.execute_reply.started":"2026-08-23T02:33:26.398365Z","shell.execute_reply":"2026-08-23T02:33:26.484951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nprint(\"=\" * 70)\nprint(\"CONAN — JSRT CANCER TARGET ANALYSIS\")\nprint(\"=\" * 70)\n\n# Load JSRT metadata\njsrt = pd.read_csv(\n    \"/kaggle/input/datasets/raddar/nodules-in-chest-xrays-jsrt/jsrt_metadata.csv\"\n)\n\nprint(\"\\nOriginal JSRT records:\", len(jsrt))\n\n# ------------------------------------------------------------\n# 1. Examine state\n# ------------------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"STATE DISTRIBUTION\")\nprint(\"=\" * 70)\n\nprint(jsrt[\"state\"].value_counts(dropna=False))\n\n# ------------------------------------------------------------\n# 2. Examine diagnosis\n# ------------------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"DIAGNOSIS DISTRIBUTION\")\nprint(\"=\" * 70)\n\nprint(jsrt[\"diagnosis\"].value_counts(dropna=False))\n\n# ------------------------------------------------------------\n# 3. Identify records usable for cancer target\n# ------------------------------------------------------------\n\ntarget_mask = jsrt[\"state\"].notna()\n\njsrt_target = jsrt[target_mask].copy()\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"USABLE CANCER-LABELED RECORDS\")\nprint(\"=\" * 70)\n\nprint(\"Total records:\", len(jsrt))\nprint(\"Usable labeled records:\", len(jsrt_target))\nprint(\"Excluded unlabeled records:\", len(jsrt) - len(jsrt_target))\n\n# ------------------------------------------------------------\n# 4. Create binary cancer target\n# ------------------------------------------------------------\n\njsrt_target[\"lung_cancer\"] = (\n    jsrt_target[\"state\"]\n    .astype(str)\n    .str.strip()\n    .str.lower()\n    .map({\n        \"malignant\": 1,\n        \"benign\": 0\n    })\n)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"LUNG CANCER TARGET\")\nprint(\"=\" * 70)\n\nprint(jsrt_target[\"lung_cancer\"].value_counts(dropna=False))\n\n# ------------------------------------------------------------\n# 5. Check whether any state values failed mapping\n# ------------------------------------------------------------\n\nunmapped = jsrt_target[\n    jsrt_target[\"lung_cancer\"].isna()\n]\n\nprint(\"\\nUnmapped state values:\", len(unmapped))\n\nif len(unmapped) > 0:\n    print(unmapped[\"state\"].value_counts(dropna=False))\n\n# ------------------------------------------------------------\n# 6. Cancer target percentages\n# ------------------------------------------------------------\n\nvalid_target = jsrt_target[\"lung_cancer\"].notna()\n\ntarget_counts = (\n    jsrt_target.loc[valid_target, \"lung_cancer\"]\n    .value_counts()\n    .sort_index()\n)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"TARGET DISTRIBUTION\")\nprint(\"=\" * 70)\n\nfor label, count in target_counts.items():\n    name = \"Non-cancer / Benign\" if label == 0 else \"Lung cancer / Malignant\"\n    percentage = count / valid_target.sum() * 100\n    print(f\"{name:25s}: {count:4d} ({percentage:6.2f}%)\")\n\n# ------------------------------------------------------------\n# 7. Check diagnosis consistency\n# ------------------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"STATE × DIAGNOSIS\")\nprint(\"=\" * 70)\n\nprint(\n    pd.crosstab(\n        jsrt_target[\"state\"],\n        jsrt_target[\"diagnosis\"],\n        dropna=False\n    )\n)\n\n# ------------------------------------------------------------\n# 8. Save preliminary target dataset\n# ------------------------------------------------------------\n\noutput_path = \"/kaggle/working/conan_jsrt_target_prepared.csv\"\n\njsrt_target.to_csv(output_path, index=False)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"SAVED\")\nprint(\"=\" * 70)\n\nprint(output_path)\nprint(\"Shape:\", jsrt_target.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:33:33.835852Z","iopub.execute_input":"2026-08-23T02:33:33.836196Z","iopub.status.idle":"2026-08-23T02:33:33.883988Z","shell.execute_reply.started":"2026-08-23T02:33:33.836166Z","shell.execute_reply":"2026-08-23T02:33:33.882982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# CONAN — JSRT CANCER DATASET CLEANING + IMAGE VERIFICATION\n# ================================================================\n\nimport os\nimport glob\nimport pandas as pd\nimport numpy as np\n\nprint(\"=\" * 70)\nprint(\"CONAN — JSRT CANCER DATASET PREPARATION\")\nprint(\"=\" * 70)\n\n# ------------------------------------------------\n# 1. Load JSRT metadata\n# ------------------------------------------------\n\nmetadata_path = \"/kaggle/input/datasets/raddar/nodules-in-chest-xrays-jsrt/jsrt_metadata.csv\"\n\nif not os.path.exists(metadata_path):\n    raise FileNotFoundError(\n        f\"JSRT metadata not found:\\n{metadata_path}\"\n    )\n\njsrt = pd.read_csv(metadata_path)\n\nprint(\"\\nLoaded:\")\nprint(metadata_path)\n\nprint(\"\\nOriginal shape:\")\nprint(jsrt.shape)\n\nprint(\"\\nColumns:\")\nprint(jsrt.columns.tolist())\n\n\n# ------------------------------------------------\n# 2. Construct proper cancer target\n# ------------------------------------------------\n\n# Only malignant and benign cases are suitable for\n# supervised lung-cancer classification.\n#\n# malignant -> 1\n# benign    -> 0\n# non-nodule -> excluded\n\njsrt_cancer = jsrt[\n    jsrt[\"state\"].isin([\"malignant\", \"benign\"])\n].copy()\n\njsrt_cancer[\"lung_cancer\"] = (\n    jsrt_cancer[\"state\"]\n    .map({\n        \"malignant\": 1,\n        \"benign\": 0\n    })\n    .astype(int)\n)\n\n\n# ------------------------------------------------\n# 3. Remove records without a valid diagnosis\n# ------------------------------------------------\n\njsrt_cancer = jsrt_cancer[\n    jsrt_cancer[\"diagnosis\"].notna()\n].copy()\n\n\n# ------------------------------------------------\n# 4. Validate target\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"CLEAN CANCER-LABELED JSRT DATASET\")\nprint(\"=\" * 70)\n\nprint(\"\\nRecords:\")\nprint(len(jsrt_cancer))\n\nprint(\"\\nTarget distribution:\")\n\ntarget_counts = jsrt_cancer[\"lung_cancer\"].value_counts().sort_index()\n\nfor target, count in target_counts.items():\n\n    label = \"Non-cancer / Benign\" if target == 0 else \"Lung cancer / Malignant\"\n\n    percentage = count / len(jsrt_cancer) * 100\n\n    print(\n        f\"{label:<25}: {count:4d} \"\n        f\"({percentage:6.2f}%)\"\n    )\n\n\n# ------------------------------------------------\n# 5. Verify no non-nodule cases remain\n# ------------------------------------------------\n\nprint(\"\\nState distribution:\")\nprint(jsrt_cancer[\"state\"].value_counts())\n\nassert \"non-nodule\" not in jsrt_cancer[\"state\"].values\n\nprint(\"\\n✓ Non-nodule records excluded\")\n\n\n# ------------------------------------------------\n# 6. Check missing values\n# ------------------------------------------------\n\nprint(\"\\nMissing values:\")\n\nmissing = jsrt_cancer.isna().sum()\n\nprint(missing[missing > 0])\n\n# Some variables such as subtlety, size, x, y may still\n# be missing in other JSRT records, so report them rather\n# than silently filling them.\n\nprint(\"\\n\")\n\n\n# ------------------------------------------------\n# 7. Search for actual JSRT image files\n# ------------------------------------------------\n\nprint(\"=\" * 70)\nprint(\"SEARCHING FOR JSRT IMAGE FILES\")\nprint(\"=\" * 70)\n\nsearch_roots = [\n    \"/kaggle/input/datasets/raddar/nodules-in-chest-xrays-jsrt\",\n    \"/kaggle/input\",\n    \"/kaggle/working\"\n]\n\nimage_extensions = [\n    \"*.png\",\n    \"*.jpg\",\n    \"*.jpeg\",\n    \"*.bmp\",\n    \"*.tif\",\n    \"*.tiff\"\n]\n\nimage_files = []\n\nfor root in search_roots:\n\n    if not os.path.exists(root):\n        continue\n\n    for ext in image_extensions:\n\n        image_files.extend(\n            glob.glob(\n                os.path.join(root, \"**\", ext),\n                recursive=True\n            )\n        )\n\n# Remove duplicates\nimage_files = sorted(set(image_files))\n\nprint(\"\\nTotal image files discovered:\")\nprint(len(image_files))\n\n\n# ------------------------------------------------\n# 8. Identify JSRT images by study ID\n# ------------------------------------------------\n\nimage_map = {}\n\nfor path in image_files:\n\n    filename = os.path.basename(path)\n\n    # JSRT study IDs look like:\n    # JPCLN001.png\n    # JPCLN002.png\n    # etc.\n\n    if filename.startswith(\"JPCLN\"):\n\n        image_map[filename] = path\n\n\nprint(\"\\nJSRT images identified:\")\nprint(len(image_map))\n\n\n# ------------------------------------------------\n# 9. Match metadata records to image files\n# ------------------------------------------------\n\njsrt_cancer[\"image_path\"] = jsrt_cancer[\"study_id\"].map(image_map)\n\nmatched = jsrt_cancer[\"image_path\"].notna().sum()\nunmatched = jsrt_cancer[\"image_path\"].isna().sum()\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"JSRT IMAGE-METADATA MATCHING\")\nprint(\"=\" * 70)\n\nprint(f\"\\nCancer-labeled records : {len(jsrt_cancer)}\")\nprint(f\"Images matched         : {matched}\")\nprint(f\"Images not found       : {unmatched}\")\n\n\n# ------------------------------------------------\n# 10. Display unmatched records if any\n# ------------------------------------------------\n\nif unmatched > 0:\n\n    print(\"\\nUnmatched study IDs:\")\n\n    print(\n        jsrt_cancer.loc[\n            jsrt_cancer[\"image_path\"].isna(),\n            [\"study_id\", \"state\", \"diagnosis\"]\n        ].to_string(index=False)\n    )\n\nelse:\n\n    print(\"\\n✓ Every cancer-labeled JSRT record has an image\")\n\n\n# ------------------------------------------------\n# 11. Keep only records with actual images\n# ------------------------------------------------\n\njsrt_cancer_images = jsrt_cancer[\n    jsrt_cancer[\"image_path\"].notna()\n].copy()\n\nprint(\"\\nFinal image-based cancer dataset:\")\nprint(jsrt_cancer_images.shape)\n\n\n# ------------------------------------------------\n# 12. Final validation\n# ------------------------------------------------\n\nassert jsrt_cancer_images[\"study_id\"].is_unique\nassert jsrt_cancer_images[\"lung_cancer\"].isin([0, 1]).all()\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"FINAL VALIDATION\")\nprint(\"=\" * 70)\n\nprint(\"\\nUnique study IDs:\")\nprint(jsrt_cancer_images[\"study_id\"].nunique())\n\nprint(\"\\nFinal target distribution:\")\nprint(\n    jsrt_cancer_images[\"lung_cancer\"]\n    .value_counts()\n    .sort_index()\n)\n\nprint(\"\\nFinal states:\")\nprint(\n    jsrt_cancer_images[\"state\"]\n    .value_counts()\n)\n\nprint(\"\\n✓ One row per JSRT image\")\nprint(\"✓ Only benign and malignant cases retained\")\nprint(\"✓ Non-nodule cases excluded\")\nprint(\"✓ Cancer target = malignant vs benign\")\nprint(\"✓ Image paths verified\")\n\n\n# ------------------------------------------------\n# 13. Save clean dataset\n# ------------------------------------------------\n\noutput_path = \"/kaggle/working/conan_jsrt_cancer_images.csv\"\n\njsrt_cancer_images.to_csv(\n    output_path,\n    index=False\n)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"SAVED\")\nprint(\"=\" * 70)\n\nprint(output_path)\nprint(f\"Shape: {jsrt_cancer_images.shape}\")\n\nprint(\"\\nFirst 5 records:\")\n\ndisplay(\n    jsrt_cancer_images[\n        [\n            \"study_id\",\n            \"state\",\n            \"diagnosis\",\n            \"lung_cancer\",\n            \"image_path\"\n        ]\n    ].head()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:33:54.022325Z","iopub.execute_input":"2026-08-23T02:33:54.022835Z","iopub.status.idle":"2026-08-23T02:33:59.996274Z","shell.execute_reply.started":"2026-08-23T02:33:54.022793Z","shell.execute_reply":"2026-08-23T02:33:59.99529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# CONAN — JSRT STRATIFIED 5-FOLD CROSS-VALIDATION SETUP\n# ================================================================\n\nimport pandas as pd\nimport numpy as np\n\nfrom sklearn.model_selection import StratifiedKFold\n\nprint(\"=\" * 70)\nprint(\"CONAN — JSRT 5-FOLD CROSS-VALIDATION\")\nprint(\"=\" * 70)\n\n# ------------------------------------------------\n# 1. Load cleaned JSRT dataset\n# ------------------------------------------------\n\ninput_path = \"/kaggle/working/conan_jsrt_cancer_images.csv\"\n\ndf = pd.read_csv(input_path)\n\nprint(\"\\nLoaded:\")\nprint(input_path)\n\nprint(\"\\nDataset shape:\")\nprint(df.shape)\n\n\n# ------------------------------------------------\n# 2. Validate target\n# ------------------------------------------------\n\nassert \"lung_cancer\" in df.columns\nassert \"image_path\" in df.columns\n\nprint(\"\\nTarget distribution:\")\nprint(df[\"lung_cancer\"].value_counts().sort_index())\n\n\n# ------------------------------------------------\n# 3. Create stratified 5-fold split\n# ------------------------------------------------\n\nskf = StratifiedKFold(\n    n_splits=5,\n    shuffle=True,\n    random_state=42\n)\n\ndf[\"fold\"] = -1\n\nfor fold, (train_idx, val_idx) in enumerate(\n    skf.split(df, df[\"lung_cancer\"])\n):\n\n    df.loc[val_idx, \"fold\"] = fold\n\n\n# ------------------------------------------------\n# 4. Validate folds\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"FOLD DISTRIBUTION\")\nprint(\"=\" * 70)\n\nfor fold in range(5):\n\n    fold_data = df[df[\"fold\"] == fold]\n\n    malignant = int(\n        (fold_data[\"lung_cancer\"] == 1).sum()\n    )\n\n    benign = int(\n        (fold_data[\"lung_cancer\"] == 0).sum()\n    )\n\n    print(\n        f\"Fold {fold}: \"\n        f\"{len(fold_data):3d} images | \"\n        f\"Benign = {benign:2d} | \"\n        f\"Malignant = {malignant:2d}\"\n    )\n\n\n# ------------------------------------------------\n# 5. Check every image belongs to exactly one fold\n# ------------------------------------------------\n\nassert df[\"fold\"].isin(range(5)).all()\nassert df[\"study_id\"].is_unique\nassert df[\"image_path\"].notna().all()\n\nprint(\"\\n✓ Every image has exactly one fold\")\nprint(\"✓ No duplicate study IDs\")\nprint(\"✓ All image paths exist in the dataset\")\n\n\n# ------------------------------------------------\n# 6. Check class balance\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"CLASS DISTRIBUTION BY FOLD\")\nprint(\"=\" * 70)\n\nfold_table = pd.crosstab(\n    df[\"fold\"],\n    df[\"lung_cancer\"]\n)\n\nfold_table.columns = [\n    \"Benign\" if c == 0 else \"Malignant\"\n    for c in fold_table.columns\n]\n\nprint(fold_table)\n\n\n# ------------------------------------------------\n# 7. Save\n# ------------------------------------------------\n\noutput_path = \"/kaggle/working/conan_jsrt_5fold.csv\"\n\ndf.to_csv(\n    output_path,\n    index=False\n)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"SAVED\")\nprint(\"=\" * 70)\n\nprint(output_path)\nprint(\"Shape:\", df.shape)\n\nprint(\"\\nColumns:\")\nprint(df.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:34:21.470196Z","iopub.execute_input":"2026-08-23T02:34:21.470568Z","iopub.status.idle":"2026-08-23T02:34:22.840656Z","shell.execute_reply.started":"2026-08-23T02:34:21.470538Z","shell.execute_reply":"2026-08-23T02:34:22.839631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# CONAN — JSRT IMAGE PREPROCESSING + RESNET-50 FEATURE EXTRACTION\n# ================================================================\n\nimport os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\n\nimport torch\nimport torchvision\nfrom torchvision import models, transforms\n\nprint(\"=\" * 70)\nprint(\"CONAN — JSRT CNN FEATURE EXTRACTION\")\nprint(\"=\" * 70)\n\n# ------------------------------------------------\n# 1. Load the 5-fold dataset\n# ------------------------------------------------\n\ninput_path = \"/kaggle/working/conan_jsrt_5fold.csv\"\n\ndf = pd.read_csv(input_path)\n\nprint(\"\\nLoaded:\")\nprint(input_path)\n\nprint(\"\\nDataset:\")\nprint(df.shape)\n\n\n# ------------------------------------------------\n# 2. Select device\n# ------------------------------------------------\n\ndevice = torch.device(\n    \"cuda\" if torch.cuda.is_available() else \"cpu\"\n)\n\nprint(\"\\nDevice:\")\nprint(device)\n\nif torch.cuda.is_available():\n    print(\"GPU:\", torch.cuda.get_device_name(0))\n\n\n# ------------------------------------------------\n# 3. Load pretrained ResNet-50\n# ------------------------------------------------\n\nprint(\"\\nLoading pretrained ResNet-50...\")\n\nweights = models.ResNet50_Weights.DEFAULT\n\nresnet = models.resnet50(\n    weights=weights\n)\n\n# Remove final classification layer.\n# Output becomes the 2048-dimensional image embedding.\n\nfeature_extractor = torch.nn.Sequential(\n    *list(resnet.children())[:-1]\n)\n\nfeature_extractor = feature_extractor.to(device)\nfeature_extractor.eval()\n\nprint(\"✓ ResNet-50 loaded\")\nprint(\"✓ Classification layer removed\")\nprint(\"Embedding size: 2048\")\n\n\n# ------------------------------------------------\n# 4. Image preprocessing\n# ------------------------------------------------\n\npreprocess = transforms.Compose([\n    transforms.Resize((224, 224)),\n\n    transforms.Grayscale(num_output_channels=3),\n\n    transforms.ToTensor(),\n\n    transforms.Normalize(\n        mean=[0.485, 0.456, 0.406],\n        std=[0.229, 0.224, 0.225]\n    )\n])\n\n\n# ------------------------------------------------\n# 5. Verify image paths\n# ------------------------------------------------\n\nmissing_images = []\n\nfor path in df[\"image_path\"]:\n\n    if not os.path.exists(path):\n        missing_images.append(path)\n\nprint(\"\\nImage verification:\")\n\nprint(\"Total images:\", len(df))\nprint(\"Missing images:\", len(missing_images))\n\nif missing_images:\n\n    print(\"\\nFirst missing files:\")\n    print(missing_images[:10])\n\n    raise FileNotFoundError(\n        \"Some JSRT images cannot be found.\"\n    )\n\nprint(\"✓ All JSRT images found\")\n\n\n# ------------------------------------------------\n# 6. Extract CNN embeddings\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"EXTRACTING CNN EMBEDDINGS\")\nprint(\"=\" * 70)\n\nembeddings = []\n\nwith torch.no_grad():\n\n    for i, path in enumerate(df[\"image_path\"]):\n\n        try:\n\n            image = Image.open(path).convert(\"L\")\n\n            image_tensor = preprocess(\n                image\n            ).unsqueeze(0).to(device)\n\n            embedding = feature_extractor(\n                image_tensor\n            )\n\n            embedding = embedding.squeeze()\n\n            embedding = embedding.cpu().numpy()\n\n            embeddings.append(embedding)\n\n        except Exception as e:\n\n            print(\n                f\"\\nError processing image {i}: \"\n                f\"{path}\"\n            )\n\n            print(e)\n\n            raise\n\n        if (i + 1) % 10 == 0:\n\n            print(\n                f\"Processed \"\n                f\"{i + 1}/{len(df)}\"\n            )\n\n\n# ------------------------------------------------\n# 7. Convert to matrix\n# ------------------------------------------------\n\nX_cnn = np.vstack(embeddings)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"CNN EMBEDDING MATRIX\")\nprint(\"=\" * 70)\n\nprint(\"\\nShape:\")\nprint(X_cnn.shape)\n\nprint(\"\\nExpected:\")\nprint(\"(154, 2048)\")\n\nassert X_cnn.shape == (len(df), 2048)\n\nprint(\"\\n✓ Correct embedding dimensions\")\n\n\n# ------------------------------------------------\n# 8. Save embeddings\n# ------------------------------------------------\n\nembedding_columns = [\n    f\"cnn_{i:04d}\"\n    for i in range(X_cnn.shape[1])\n]\n\nembedding_df = pd.DataFrame(\n    X_cnn,\n    columns=embedding_columns\n)\n\nembedding_df.insert(\n    0,\n    \"study_id\",\n    df[\"study_id\"].values\n)\n\nembedding_df[\"lung_cancer\"] = (\n    df[\"lung_cancer\"].values\n)\n\nembedding_df[\"fold\"] = (\n    df[\"fold\"].values\n)\n\n\noutput_path = (\n    \"/kaggle/working/\"\n    \"conan_jsrt_resnet50_embeddings.csv\"\n)\n\nembedding_df.to_csv(\n    output_path,\n    index=False\n)\n\n\n# ------------------------------------------------\n# 9. Final output\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"SAVED\")\nprint(\"=\" * 70)\n\nprint(output_path)\n\nprint(\"\\nShape:\")\nprint(embedding_df.shape)\n\nprint(\"\\nColumns:\")\nprint(\n    f\"study_id + \"\n    f\"{len(embedding_columns)} CNN features + \"\n    f\"lung_cancer + fold\"\n)\n\nprint(\"\\n✓ CNN feature extraction complete\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:34:39.735416Z","iopub.execute_input":"2026-08-23T02:34:39.735843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# CONAN — CNN EMBEDDING 5-FOLD OOF CANCER RISK MODEL\n# ================================================================\n\nimport os\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import (\n    roc_auc_score,\n    accuracy_score,\n    balanced_accuracy_score,\n    precision_score,\n    recall_score,\n    f1_score,\n    confusion_matrix\n)\n\nprint(\"=\" * 70)\nprint(\"CONAN — CNN EMBEDDING 5-FOLD OOF CANCER RISK MODEL\")\nprint(\"=\" * 70)\n\n# ------------------------------------------------\n# 1. LOAD CNN EMBEDDINGS\n# ------------------------------------------------\n\npath = \"/kaggle/working/conan_jsrt_resnet50_embeddings.csv\"\n\nif not os.path.exists(path):\n    raise FileNotFoundError(\n        f\"Could not find:\\n{path}\"\n    )\n\ndf = pd.read_csv(path)\n\nprint(\"\\nLoaded:\")\nprint(path)\n\nprint(\"\\nDataset shape:\")\nprint(df.shape)\n\n# ------------------------------------------------\n# 2. IDENTIFY CNN FEATURES\n# ------------------------------------------------\n\nexcluded = [\n    \"study_id\",\n    \"lung_cancer\",\n    \"fold\"\n]\n\nfeature_cols = [\n    c for c in df.columns\n    if c not in excluded\n]\n\nprint(\"\\nCNN feature count:\")\nprint(len(feature_cols))\n\nif len(feature_cols) != 2048:\n    raise ValueError(\n        f\"Expected 2048 CNN features, found {len(feature_cols)}\"\n    )\n\nX = df[feature_cols].astype(np.float32).values\ny = df[\"lung_cancer\"].astype(int).values\nfolds = df[\"fold\"].astype(int).values\n\nprint(\"\\nFeature matrix:\")\nprint(X.shape)\n\nprint(\"\\nTarget distribution:\")\nprint(pd.Series(y).value_counts().sort_index())\n\nprint(\"\\nFold distribution:\")\nprint(\n    pd.DataFrame({\n        \"fold\": folds,\n        \"lung_cancer\": y\n    })\n    .groupby([\"fold\", \"lung_cancer\"])\n    .size()\n    .unstack(fill_value=0)\n)\n\n# ------------------------------------------------\n# 3. OOF PREDICTIONS\n# ------------------------------------------------\n\noof_pred = np.zeros(len(df), dtype=float)\noof_class = np.zeros(len(df), dtype=int)\n\nfold_results = []\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"TRAINING 5 FOLDS\")\nprint(\"=\" * 70)\n\nfor fold in sorted(np.unique(folds)):\n\n    train_idx = folds != fold\n    test_idx = folds == fold\n\n    X_train = X[train_idx]\n    X_test = X[test_idx]\n\n    y_train = y[train_idx]\n    y_test = y[test_idx]\n\n    print(f\"\\nFold {fold}\")\n    print(\"-\" * 50)\n    print(f\"Training images : {len(X_train)}\")\n    print(f\"Validation images: {len(X_test)}\")\n    print(\n        f\"Train cancer    : {y_train.sum()} / {len(y_train)}\"\n    )\n    print(\n        f\"Valid cancer    : {y_test.sum()} / {len(y_test)}\"\n    )\n\n    # ------------------------------------------------\n    # IMPORTANT:\n    # Scaling is fitted ONLY on training data.\n    # ------------------------------------------------\n\n    model = Pipeline([\n        (\n            \"scaler\",\n            StandardScaler()\n        ),\n        (\n            \"classifier\",\n            LogisticRegression(\n                C=0.03,\n                max_iter=3000,\n                class_weight=\"balanced\",\n                solver=\"liblinear\",\n                random_state=42\n            )\n        )\n    ])\n\n    model.fit(X_train, y_train)\n\n    prob = model.predict_proba(X_test)[:, 1]\n    pred = (prob >= 0.5).astype(int)\n\n    oof_pred[test_idx] = prob\n    oof_class[test_idx] = pred\n\n    fold_auc = roc_auc_score(y_test, prob)\n\n    fold_results.append({\n        \"fold\": fold,\n        \"n\": len(y_test),\n        \"benign\": int((y_test == 0).sum()),\n        \"malignant\": int((y_test == 1).sum()),\n        \"roc_auc\": fold_auc\n    })\n\n    print(f\"Fold ROC-AUC: {fold_auc:.4f}\")\n\n# ------------------------------------------------\n# 4. OVERALL OOF PERFORMANCE\n# ------------------------------------------------\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"OVERALL 5-FOLD OOF PERFORMANCE\")\nprint(\"=\" * 70)\n\nauc = roc_auc_score(y, oof_pred)\naccuracy = accuracy_score(y, oof_class)\nbalanced_acc = balanced_accuracy_score(y, oof_class)\nprecision = precision_score(y, oof_class, zero_division=0)\nrecall = recall_score(y, oof_class, zero_division=0)\nf1 = f1_score(y, oof_class, zero_division=0)\n\nprint(f\"\\nROC-AUC            : {auc:.4f}\")\nprint(f\"Accuracy           : {accuracy:.4f}\")\nprint(f\"Balanced Accuracy  : {balanced_acc:.4f}\")\nprint(f\"Precision          : {precision:.4f}\")\nprint(f\"Recall             : {recall:.4f}\")\nprint(f\"F1 Score           : {f1:.4f}\")\n\n# ------------------------------------------------\n# 5. CONFUSION MATRIX\n# ------------------------------------------------\n\ncm = confusion_matrix(y, oof_class)\n\nprint(\"\\nConfusion Matrix:\")\nprint(cm)\n\n# ------------------------------------------------\n# 6. FOLD SUMMARY\n# ------------------------------------------------\n\nfold_df = pd.DataFrame(fold_results)\n\nprint(\"\\nFold Results:\")\nprint(fold_df.to_string(index=False))\n\nprint(\"\\nFold ROC-AUC mean:\")\nprint(f\"{fold_df['roc_auc'].mean():.4f}\")\n\nprint(\"\\nFold ROC-AUC standard deviation:\")\nprint(f\"{fold_df['roc_auc'].std():.4f}\")\n\n# ------------------------------------------------\n# 7. SAVE OOF PREDICTIONS\n# ------------------------------------------------\n\noof_df = df[\n    [\"study_id\", \"lung_cancer\", \"fold\"]\n].copy()\n\noof_df[\"cnn_risk_probability\"] = oof_pred\noof_df[\"cnn_prediction\"] = oof_class\n\noutput_path = \"/kaggle/working/conan_jsrt_cnn_oof_predictions.csv\"\n\noof_df.to_csv(output_path, index=False)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"SAVED\")\nprint(\"=\" * 70)\n\nprint(output_path)\nprint(\"Shape:\", oof_df.shape)\n\nprint(\"\\n✓ Every image has an out-of-fold CNN prediction\")\nprint(\"✓ No validation image was used to train its fold model\")\nprint(\"✓ CNN risk probability generated\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:38:30.122509Z","iopub.execute_input":"2026-08-23T02:38:30.123129Z","iopub.status.idle":"2026-08-23T02:38:30.768247Z","shell.execute_reply.started":"2026-08-23T02:38:30.123073Z","shell.execute_reply":"2026-08-23T02:38:30.767061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================================================\n# CONAN — JSRT RESNET-50 FINE-TUNING\n# 5-FOLD OUT-OF-FOLD LUNG CANCER RISK MODEL\n# ================================================================\n\nimport os\nimport copy\nimport random\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models, transforms\n\nfrom sklearn.metrics import (\n    roc_auc_score,\n    accuracy_score,\n    balanced_accuracy_score,\n    precision_score,\n    recall_score,\n    f1_score,\n    confusion_matrix\n)\n\n# ================================================================\n# CONFIGURATION\n# ================================================================\n\nSEED = 42\n\nBATCH_SIZE = 8\nNUM_EPOCHS = 15\nLEARNING_RATE = 1e-4\nWEIGHT_DECAY = 1e-4\n\nPATIENCE = 4\n\nIMAGE_SIZE = 224\n\nDEVICE = torch.device(\n    \"cuda\" if torch.cuda.is_available() else \"cpu\"\n)\n\nprint(\"=\" * 70)\nprint(\"CONAN — JSRT RESNET-50 FINE-TUNING\")\nprint(\"=\" * 70)\n\nprint(\"\\nDevice:\")\nprint(DEVICE)\n\n# ------------------------------------------------\n# Reproducibility\n# ------------------------------------------------\n\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\n\nif torch.cuda.is_available():\n    torch.cuda.manual_seed_all(SEED)\n\n# ================================================================\n# 1. LOAD JSRT DATA\n# ================================================================\n\nDATA_PATH = \"/kaggle/working/conan_jsrt_5fold.csv\"\n\nif not os.path.exists(DATA_PATH):\n    raise FileNotFoundError(\n        f\"Dataset not found:\\n{DATA_PATH}\"\n    )\n\ndf = pd.read_csv(DATA_PATH)\n\nprint(\"\\nLoaded:\")\nprint(DATA_PATH)\n\nprint(\"\\nDataset shape:\")\nprint(df.shape)\n\nrequired_columns = [\n    \"study_id\",\n    \"lung_cancer\",\n    \"image_path\",\n    \"fold\"\n]\n\nfor col in required_columns:\n    if col not in df.columns:\n        raise ValueError(\n            f\"Required column missing: {col}\"\n        )\n\n# ================================================================\n# 2. VALIDATE DATA\n# ================================================================\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"DATA VALIDATION\")\nprint(\"=\" * 70)\n\nprint(\"\\nTarget distribution:\")\nprint(df[\"lung_cancer\"].value_counts().sort_index())\n\nprint(\"\\nFold distribution:\")\nprint(\n    df.groupby([\"fold\", \"lung_cancer\"])\n      .size()\n      .unstack(fill_value=0)\n)\n\n# Check duplicate studies\nduplicates = df[\"study_id\"].duplicated().sum()\n\nprint(\"\\nDuplicate study IDs:\")\nprint(duplicates)\n\nif duplicates != 0:\n    raise ValueError(\"Duplicate study IDs detected.\")\n\n# Check image paths\nmissing_images = [\n    p for p in df[\"image_path\"]\n    if not os.path.exists(p)\n]\n\nprint(\"\\nMissing images:\")\nprint(len(missing_images))\n\nif len(missing_images) > 0:\n    print(missing_images[:10])\n    raise FileNotFoundError(\n        \"Some JSRT images could not be found.\"\n    )\n\nprint(\"\\n✓ Dataset validated\")\nprint(\"✓ One row per image\")\nprint(\"✓ All images exist\")\nprint(\"✓ No duplicate study IDs\")\n\n# ================================================================\n# 3. DATASET CLASS\n# ================================================================\n\nclass JSRTDataset(Dataset):\n\n    def __init__(self, dataframe, transform=None):\n\n        self.df = dataframe.reset_index(drop=True)\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n\n        row = self.df.iloc[idx]\n\n        image_path = row[\"image_path\"]\n        label = int(row[\"lung_cancer\"])\n\n        # ------------------------------------------------\n        # JSRT X-ray\n        # ------------------------------------------------\n\n        image = Image.open(image_path).convert(\"RGB\")\n\n        if self.transform is not None:\n            image = self.transform(image)\n\n        return image, torch.tensor(\n            label,\n            dtype=torch.float32\n        )\n\n# ================================================================\n# 4. IMAGE TRANSFORMS\n# ================================================================\n\n# Training augmentation\ntrain_transform = transforms.Compose([\n\n    transforms.Resize(\n        (IMAGE_SIZE, IMAGE_SIZE)\n    ),\n\n    transforms.RandomHorizontalFlip(\n        p=0.5\n    ),\n\n    transforms.RandomRotation(\n        degrees=5\n    ),\n\n    transforms.RandomAffine(\n        degrees=0,\n        translate=(0.03, 0.03),\n        scale=(0.97, 1.03)\n    ),\n\n    transforms.ToTensor(),\n\n    transforms.Normalize(\n        mean=[0.485, 0.456, 0.406],\n        std=[0.229, 0.224, 0.225]\n    )\n])\n\n# Validation transform\nval_transform = transforms.Compose([\n\n    transforms.Resize(\n        (IMAGE_SIZE, IMAGE_SIZE)\n    ),\n\n    transforms.ToTensor(),\n\n    transforms.Normalize(\n        mean=[0.485, 0.456, 0.406],\n        std=[0.229, 0.224, 0.225]\n    )\n])\n\nprint(\"\\n✓ Training augmentation configured\")\nprint(\"✓ Validation transformation configured\")\n\n# ================================================================\n# 5. RESNET-50 MODEL\n# ================================================================\n\ndef create_model():\n\n    print(\"\\nLoading pretrained ResNet-50...\")\n\n    model = models.resnet50(\n        weights=models.ResNet50_Weights.DEFAULT\n    )\n\n    # ------------------------------------------------\n    # Freeze entire network initially\n    # ------------------------------------------------\n\n    for param in model.parameters():\n        param.requires_grad = False\n\n    # ------------------------------------------------\n    # Fine-tune final convolutional block\n    # ------------------------------------------------\n\n    for param in model.layer4.parameters():\n        param.requires_grad = True\n\n    # ------------------------------------------------\n    # Replace classifier\n    # ------------------------------------------------\n\n    num_features = model.fc.in_features\n\n    model.fc = nn.Sequential(\n\n        nn.Dropout(\n            p=0.40\n        ),\n\n        nn.Linear(\n            num_features,\n            1\n        )\n    )\n\n    # New classifier must train\n    for param in model.fc.parameters():\n        param.requires_grad = True\n\n    return model.to(DEVICE)\n\n# ================================================================\n# 6. TRAINING FUNCTION\n# ================================================================\n\ndef train_one_fold(\n    train_df,\n    val_df,\n    fold_number\n):\n\n    print(\"\\n\" + \"=\" * 70)\n    print(f\"TRAINING FOLD {fold_number}\")\n    print(\"=\" * 70)\n\n    print(\n        f\"Training images   : {len(train_df)}\"\n    )\n\n    print(\n        f\"Validation images : {len(val_df)}\"\n    )\n\n    print(\n        f\"Training cancer   : {train_df['lung_cancer'].sum()} / {len(train_df)}\"\n    )\n\n    print(\n        f\"Validation cancer : {val_df['lung_cancer'].sum()} / {len(val_df)}\"\n    )\n\n    # ------------------------------------------------\n    # Datasets\n    # ------------------------------------------------\n\n    train_dataset = JSRTDataset(\n        train_df,\n        transform=train_transform\n    )\n\n    val_dataset = JSRTDataset(\n        val_df,\n        transform=val_transform\n    )\n\n    train_loader = DataLoader(\n        train_dataset,\n        batch_size=BATCH_SIZE,\n        shuffle=True,\n        num_workers=2,\n        pin_memory=torch.cuda.is_available()\n    )\n\n    val_loader = DataLoader(\n        val_dataset,\n        batch_size=BATCH_SIZE,\n        shuffle=False,\n        num_workers=2,\n        pin_memory=torch.cuda.is_available()\n    )\n\n    # ------------------------------------------------\n    # Model\n    # ------------------------------------------------\n\n    model = create_model()\n\n    # ------------------------------------------------\n    # Class weighting\n    # ------------------------------------------------\n\n    train_labels = train_df[\"lung_cancer\"].values\n\n    n_negative = np.sum(train_labels == 0)\n    n_positive = np.sum(train_labels == 1)\n\n    pos_weight = (\n        n_negative / n_positive\n    )\n\n    pos_weight_tensor = torch.tensor(\n        [pos_weight],\n        dtype=torch.float32,\n        device=DEVICE\n    )\n\n    criterion = nn.BCEWithLogitsLoss(\n        pos_weight=pos_weight_tensor\n    )\n\n    # ------------------------------------------------\n    # Optimizer\n    # ------------------------------------------------\n\n    trainable_params = [\n        p for p in model.parameters()\n        if p.requires_grad\n    ]\n\n    optimizer = torch.optim.AdamW(\n        trainable_params,\n        lr=LEARNING_RATE,\n        weight_decay=WEIGHT_DECAY\n    )\n\n    scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(\n        optimizer,\n        mode=\"max\",\n        factor=0.5,\n        patience=2\n    )\n\n    # ------------------------------------------------\n    # Early stopping\n    # ------------------------------------------------\n\n    best_auc = -np.inf\n    best_state = None\n    patience_counter = 0\n\n    # ====================================================\n    # EPOCH LOOP\n    # ====================================================\n\n    for epoch in range(NUM_EPOCHS):\n\n        # ------------------------------------------------\n        # TRAIN\n        # ------------------------------------------------\n\n        model.train()\n\n        train_loss = 0.0\n\n        for images, labels in train_loader:\n\n            images = images.to(\n                DEVICE,\n                non_blocking=True\n            )\n\n            labels = labels.to(\n                DEVICE,\n                non_blocking=True\n            ).unsqueeze(1)\n\n            optimizer.zero_grad()\n\n            logits = model(images)\n\n            loss = criterion(\n                logits,\n                labels\n            )\n\n            loss.backward()\n\n            optimizer.step()\n\n            train_loss += (\n                loss.item() * images.size(0)\n            )\n\n        train_loss /= len(train_dataset)\n\n        # ------------------------------------------------\n        # VALIDATION\n        # ------------------------------------------------\n\n        model.eval()\n\n        val_probs = []\n        val_labels = []\n\n        with torch.no_grad():\n\n            for images, labels in val_loader:\n\n                images = images.to(\n                    DEVICE,\n                    non_blocking=True\n                )\n\n                logits = model(images)\n\n                probs = torch.sigmoid(\n                    logits\n                ).flatten()\n\n                val_probs.extend(\n                    probs.cpu().numpy()\n                )\n\n                val_labels.extend(\n                    labels.numpy()\n                )\n\n        val_probs = np.asarray(\n            val_probs\n        )\n\n        val_labels = np.asarray(\n            val_labels\n        ).astype(int)\n\n        val_auc = roc_auc_score(\n            val_labels,\n            val_probs\n        )\n\n        scheduler.step(val_auc)\n\n        print(\n            f\"Epoch {epoch + 1:02d}/{NUM_EPOCHS} | \"\n            f\"Loss: {train_loss:.4f} | \"\n            f\"Val AUC: {val_auc:.4f}\"\n        )\n\n        # ------------------------------------------------\n        # Save best model\n        # ------------------------------------------------\n\n        if val_auc > best_auc:\n\n            best_auc = val_auc\n\n            best_state = copy.deepcopy(\n                model.state_dict()\n            )\n\n            patience_counter = 0\n\n        else:\n\n            patience_counter += 1\n\n        if patience_counter >= PATIENCE:\n\n            print(\n                f\"Early stopping at epoch \"\n                f\"{epoch + 1}\"\n            )\n\n            break\n\n    # ====================================================\n    # RESTORE BEST MODEL\n    # ====================================================\n\n    if best_state is not None:\n\n        model.load_state_dict(\n            best_state\n        )\n\n    print(\n        f\"\\nBest validation ROC-AUC: \"\n        f\"{best_auc:.4f}\"\n    )\n\n    # ====================================================\n    # FINAL VALIDATION PREDICTIONS\n    # ====================================================\n\n    model.eval()\n\n    final_probs = []\n\n    with torch.no_grad():\n\n        for images, labels in val_loader:\n\n            images = images.to(\n                DEVICE,\n                non_blocking=True\n            )\n\n            logits = model(images)\n\n            probs = torch.sigmoid(\n                logits\n            ).flatten()\n\n            final_probs.extend(\n                probs.cpu().numpy()\n            )\n\n    final_probs = np.asarray(\n        final_probs\n    )\n\n    return final_probs, best_auc\n\n# ================================================================\n# 7. 5-FOLD OUT-OF-FOLD TRAINING\n# ================================================================\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"5-FOLD OUT-OF-FOLD FINE-TUNING\")\nprint(\"=\" * 70)\n\noof_predictions = np.zeros(\n    len(df),\n    dtype=float\n)\n\nfold_results = []\n\nunique_folds = sorted(\n    df[\"fold\"].unique()\n)\n\nfor fold in unique_folds:\n\n    train_df = df[\n        df[\"fold\"] != fold\n    ].copy()\n\n    val_df = df[\n        df[\"fold\"] == fold\n    ].copy()\n\n    val_probs, best_auc = train_one_fold(\n        train_df,\n        val_df,\n        fold\n    )\n\n    # Put predictions back into original rows\n    val_indices = val_df.index.values\n\n    oof_predictions[\n        val_indices\n    ] = val_probs\n\n    fold_results.append({\n\n        \"fold\": int(fold),\n\n        \"n\": len(val_df),\n\n        \"benign\": int(\n            (val_df[\"lung_cancer\"] == 0).sum()\n        ),\n\n        \"malignant\": int(\n            (val_df[\"lung_cancer\"] == 1).sum()\n        ),\n\n        \"roc_auc\": best_auc\n    })\n\n# ================================================================\n# 8. OVERALL OOF PERFORMANCE\n# ================================================================\n\ny_true = df[\n    \"lung_cancer\"\n].astype(int).values\n\noof_classes = (\n    oof_predictions >= 0.5\n).astype(int)\n\noverall_auc = roc_auc_score(\n    y_true,\n    oof_predictions\n)\n\naccuracy = accuracy_score(\n    y_true,\n    oof_classes\n)\n\nbalanced_accuracy = balanced_accuracy_score(\n    y_true,\n    oof_classes\n)\n\nprecision = precision_score(\n    y_true,\n    oof_classes,\n    zero_division=0\n)\n\nrecall = recall_score(\n    y_true,\n    oof_classes,\n    zero_division=0\n)\n\nf1 = f1_score(\n    y_true,\n    oof_classes,\n    zero_division=0\n)\n\ncm = confusion_matrix(\n    y_true,\n    oof_classes\n)\n\n# ================================================================\n# 9. RESULTS\n# ================================================================\n\nfold_df = pd.DataFrame(\n    fold_results\n)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"FINAL 5-FOLD OOF RESULTS\")\nprint(\"=\" * 70)\n\nprint(\n    f\"\\nOverall OOF ROC-AUC   : \"\n    f\"{overall_auc:.4f}\"\n)\n\nprint(\n    f\"Accuracy              : \"\n    f\"{accuracy:.4f}\"\n)\n\nprint(\n    f\"Balanced Accuracy     : \"\n    f\"{balanced_accuracy:.4f}\"\n)\n\nprint(\n    f\"Precision             : \"\n    f\"{precision:.4f}\"\n)\n\nprint(\n    f\"Recall                : \"\n    f\"{recall:.4f}\"\n)\n\nprint(\n    f\"F1 Score              : \"\n    f\"{f1:.4f}\"\n)\n\nprint(\"\\nConfusion Matrix:\")\nprint(cm)\n\nprint(\"\\nFold Results:\")\nprint(\n    fold_df.to_string(\n        index=False\n    )\n)\n\nprint(\n    \"\\nMean Fold ROC-AUC: \"\n    f\"{fold_df['roc_auc'].mean():.4f}\"\n)\n\nprint(\n    \"Fold ROC-AUC SD: \"\n    f\"{fold_df['roc_auc'].std():.4f}\"\n)\n\n# ================================================================\n# 10. SAVE OOF PREDICTIONS\n# ================================================================\n\noutput = df[\n    [\n        \"study_id\",\n        \"lung_cancer\",\n        \"fold\"\n    ]\n].copy()\n\noutput[\n    \"cnn_finetuned_risk_probability\"\n] = oof_predictions\n\noutput[\n    \"cnn_finetuned_prediction\"\n] = oof_classes\n\noutput_path = (\n    \"/kaggle/working/\"\n    \"conan_jsrt_resnet50_finetuned_oof.csv\"\n)\n\noutput.to_csv(\n    output_path,\n    index=False\n)\n\nprint(\"\\n\" + \"=\" * 70)\nprint(\"SAVED\")\nprint(\"=\" * 70)\n\nprint(output_path)\nprint(\"Shape:\", output.shape)\n\nprint(\"\\n✓ 5-fold OOF predictions generated\")\nprint(\"✓ Validation images were not used for training\")\nprint(\"✓ ResNet-50 layer4 was fine-tuned\")\nprint(\"✓ Cancer probabilities generated\")\nprint(\"✓ Results are ready for comparison\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T02:40:24.580383Z","iopub.execute_input":"2026-08-23T02:40:24.580929Z","iopub.status.idle":"2026-08-23T03:08:42.278065Z","shell.execute_reply.started":"2026-08-23T02:40:24.580886Z","shell.execute_reply":"2026-08-23T03:08:42.27674Z"}},"outputs":[],"execution_count":null}]}