{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":36363,"sourceType":"competition"},{"sourceId":5718655,"sourceType":"datasetVersion"},{"sourceId":6587064,"sourceType":"datasetVersion"},{"sourceId":6733646,"sourceType":"datasetVersion"},{"sourceId":11502660,"sourceType":"datasetVersion"},{"sourceId":213440416,"sourceType":"kernelVersion"},{"sourceId":349660,"sourceType":"modelInstanceVersion"}],"isGpuEnabled":false,"isInternetEnabled":true,"language":"python","sourceType":"notebook"},"papermill":{"default_parameters":{},"duration":1158.094961,"end_time":"2025-04-22T06:00:06.283427","environment_variables":{},"exception":true,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-04-22T05:40:48.188466","version":"2.4.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"print(\"VERSION_CHECK_23AUG2026\")","metadata":{"papermill":{"duration":9.833909,"end_time":"2025-04-22T05:41:01.139504","exception":false,"start_time":"2025-04-22T05:40:51.305595","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T10:25:59.163305Z","iopub.execute_input":"2026-08-23T10:25:59.164083Z","iopub.status.idle":"2026-08-23T10:26:04.904115Z","shell.execute_reply.started":"2026-08-23T10:25:59.16405Z","shell.execute_reply":"2026-08-23T10:26:04.903128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install the necessary image decompressors directly from PyPI\n!pip install -q python-gdcm pylibjpeg pylibjpeg-libjpeg pydicom","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T14:47:57.322892Z","iopub.execute_input":"2026-08-23T14:47:57.323581Z","iopub.status.idle":"2026-08-23T14:48:04.435995Z","shell.execute_reply.started":"2026-08-23T14:47:57.323542Z","shell.execute_reply":"2026-08-23T14:48:04.434647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import pydicom for CT scans, nibabel for 3D segmentation files, and TensorFlow/Keras to build the neural network\nimport pandas as pd\nimport numpy as np\nimport pydicom as dicom\nimport glob\nimport nibabel as nib\nimport os\nimport re\nimport cv2\nimport random\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, callbacks\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.models import Model\nfrom sklearn.model_selection import StratifiedKFold\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.applications import InceptionV3, DenseNet121, InceptionResNetV2\n\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten\nfrom keras.layers import Conv2D, MaxPooling2D\nfrom keras.utils import to_categorical, plot_model\nfrom keras.preprocessing import image\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import precision_score, recall_score, f1_score, multilabel_confusion_matrix\nfrom sklearn.metrics import roc_auc_score, roc_curve\npd.set_option('display.max_columns', None)","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:48:04.438113Z","iopub.execute_input":"2026-08-23T14:48:04.438593Z","iopub.status.idle":"2026-08-23T14:48:26.676339Z","shell.execute_reply.started":"2026-08-23T14:48:04.438546Z","shell.execute_reply":"2026-08-23T14:48:26.675334Z"},"papermill":{"duration":10.469796,"end_time":"2025-04-22T05:41:11.61493","exception":false,"start_time":"2025-04-22T05:41:01.145134","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data Loading\nbase_dir = r'/kaggle/input/competitions/rsna-2022-cervical-spine-fracture-detection'\ntrain_images = os.path.join(base_dir,'train_images')\ntest_images = os.path.join(base_dir,'test_images')\nsegmentation_data = r'/kaggle/input/datasets/saikiranvarma/rsna-cervical-fracture-segmentations-npy/npy_segmentations'\ntrain_data = pd.read_csv(os.path.join(base_dir,'train.csv'))\nsegmentation_meta_data = pd.read_csv(r'/kaggle/input/datasets/saikiranvarma/rsna-cervical-fracture-segmentation-metadata/meta_segmentation.csv')","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:48:32.776322Z","iopub.execute_input":"2026-08-23T14:48:32.776914Z","iopub.status.idle":"2026-08-23T14:48:32.998613Z","shell.execute_reply.started":"2026-08-23T14:48:32.776883Z","shell.execute_reply":"2026-08-23T14:48:32.997753Z"},"papermill":{"duration":0.202945,"end_time":"2025-04-22T05:41:11.824118","exception":false,"start_time":"2025-04-22T05:41:11.621173","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"segmentation_meta_data.shape","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:48:41.464702Z","iopub.execute_input":"2026-08-23T14:48:41.465056Z","iopub.status.idle":"2026-08-23T14:48:41.472498Z","shell.execute_reply.started":"2026-08-23T14:48:41.465019Z","shell.execute_reply":"2026-08-23T14:48:41.471621Z"},"papermill":{"duration":0.013812,"end_time":"2025-04-22T05:41:11.843573","exception":false,"start_time":"2025-04-22T05:41:11.829761","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"segmentation_meta_data.columns","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:48:45.427754Z","iopub.execute_input":"2026-08-23T14:48:45.428715Z","iopub.status.idle":"2026-08-23T14:48:45.436906Z","shell.execute_reply.started":"2026-08-23T14:48:45.428675Z","shell.execute_reply":"2026-08-23T14:48:45.435902Z"},"papermill":{"duration":0.012811,"end_time":"2025-04-22T05:41:11.861644","exception":false,"start_time":"2025-04-22T05:41:11.848833","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"segmentation_meta_data['PhotometricInterpretation'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:49:13.550687Z","iopub.execute_input":"2026-08-23T14:49:13.551038Z","iopub.status.idle":"2026-08-23T14:49:13.561593Z","shell.execute_reply.started":"2026-08-23T14:49:13.551007Z","shell.execute_reply":"2026-08-23T14:49:13.560507Z"},"papermill":{"duration":0.022562,"end_time":"2025-04-22T05:41:11.889603","exception":false,"start_time":"2025-04-22T05:41:11.867041","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns = ['StudyInstanceUID','SOPInstanceUID','C1','C2','C3','C4','C5','C6','C7']","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:49:15.923304Z","iopub.execute_input":"2026-08-23T14:49:15.923643Z","iopub.status.idle":"2026-08-23T14:49:15.928337Z","shell.execute_reply.started":"2026-08-23T14:49:15.923616Z","shell.execute_reply":"2026-08-23T14:49:15.927299Z"},"papermill":{"duration":0.011194,"end_time":"2025-04-22T05:41:11.906139","exception":false,"start_time":"2025-04-22T05:41:11.894945","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seg_labels = segmentation_meta_data[columns].copy()","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:49:18.836223Z","iopub.execute_input":"2026-08-23T14:49:18.836602Z","iopub.status.idle":"2026-08-23T14:49:18.851286Z","shell.execute_reply.started":"2026-08-23T14:49:18.836571Z","shell.execute_reply":"2026-08-23T14:49:18.850332Z"},"papermill":{"duration":0.018497,"end_time":"2025-04-22T05:41:11.930096","exception":false,"start_time":"2025-04-22T05:41:11.911599","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seg_labels.head(2)","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:49:25.650324Z","iopub.execute_input":"2026-08-23T14:49:25.650659Z","iopub.status.idle":"2026-08-23T14:49:25.677272Z","shell.execute_reply.started":"2026-08-23T14:49:25.650629Z","shell.execute_reply":"2026-08-23T14:49:25.676278Z"},"papermill":{"duration":0.020625,"end_time":"2025-04-22T05:41:11.956652","exception":false,"start_time":"2025-04-22T05:41:11.936027","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Get Slice instance number\nseg_labels.loc[:,'slice'] = seg_labels['SOPInstanceUID'].apply(lambda x:x.split('.')[-1:][0])","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:49:28.685567Z","iopub.execute_input":"2026-08-23T14:49:28.685911Z","iopub.status.idle":"2026-08-23T14:49:28.709515Z","shell.execute_reply.started":"2026-08-23T14:49:28.68588Z","shell.execute_reply":"2026-08-23T14:49:28.708614Z"},"papermill":{"duration":0.026506,"end_time":"2025-04-22T05:41:11.988674","exception":false,"start_time":"2025-04-22T05:41:11.962168","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seg_labels","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:49:36.666456Z","iopub.execute_input":"2026-08-23T14:49:36.666783Z","iopub.status.idle":"2026-08-23T14:49:36.679712Z","shell.execute_reply.started":"2026-08-23T14:49:36.666755Z","shell.execute_reply":"2026-08-23T14:49:36.67863Z"},"papermill":{"duration":0.020167,"end_time":"2025-04-22T05:41:12.014387","exception":false,"start_time":"2025-04-22T05:41:11.99422","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to load DICOM images\ndef load_scan(dcm_paths):  \n    patient_scan = [dicom.dcmread(paths) for paths in dcm_paths]\n    return patient_scan\n    \n# Converts raw image arrays into Hounsfield Units (HU)\ndef get_pixels_hu(img):\n    image = cv2.resize(img.pixel_array,(128, 128),interpolation = cv2.INTER_NEAREST)\n    image = image.astype(np.int16)\n    # Set outside-of-scan pixels to 0, the intercept is usually -1024, so air is approximately 0\n    image[image <= -1000] = 0\n    # Convert to Hounsfield units (HU)    \n    intercept = np.array(img.RescaleIntercept)\n    slope = np.array(img.RescaleSlope)\n    image= (slope * image.astype(\"float64\")) + intercept\n#     plt.imshow(image.astype(\"int16\"), cmap='bone') \n    return image.astype(\"int16\")","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:51:18.985413Z","iopub.execute_input":"2026-08-23T14:51:18.986008Z","iopub.status.idle":"2026-08-23T14:51:18.993106Z","shell.execute_reply.started":"2026-08-23T14:51:18.985975Z","shell.execute_reply":"2026-08-23T14:51:18.992222Z"},"papermill":{"duration":0.013477,"end_time":"2025-04-22T05:41:12.034055","exception":false,"start_time":"2025-04-22T05:41:12.020578","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list(seg_labels['StudyInstanceUID'].unique())","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:50:01.006609Z","iopub.execute_input":"2026-08-23T14:50:01.006945Z","iopub.status.idle":"2026-08-23T14:50:01.016026Z","shell.execute_reply.started":"2026-08-23T14:50:01.006914Z","shell.execute_reply":"2026-08-23T14:50:01.015062Z"},"papermill":{"duration":0.017853,"end_time":"2025-04-22T05:41:12.058043","exception":false,"start_time":"2025-04-22T05:41:12.04019","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Collects and sorts a patient's entire 3D slice sequence\ndef get_image(study_instance):\n    path = '/kaggle/input/competitions/rsna-2022-cervical-spine-fracture-detection/train_images'\n    patient_slices = []\n    org_images = []\n    # 1. Filter rows and sort slices numerically to preserve spatial stack order\n    patient_df = seg_labels[seg_labels['StudyInstanceUID'] == study_instance].copy()\n    patient_df['slice_int'] = patient_df['slice'].astype(int)\n    patient_df = patient_df.sort_values('slice_int')\n    \n    # 2. Re-extract correctly ordered string labels for the file system lookups\n    slices = list(patient_df['slice'].astype(str))\n    \n    # 3. Secure full file system paths using standard path join utilities\n    dcm_paths = [os.path.join(path, study_instance, f\"{slic}.dcm\") for slic in slices]\n    \n    # 4. Load the structural scan matrices\n    image = load_scan(dcm_paths)\n    org_images.append(image)\n    \n    # 5. Extract bone density Hounsfield Units via the decompression backends\n    slices_p = [dicom.dcmread(dcm_path) for dcm_path in dcm_paths]\n    patient_slice = [get_pixels_hu(slic) for slic in slices_p]\n    patient_slices.append(patient_slice)\n    \n    return org_images, patient_slices, slices","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T14:50:55.272606Z","iopub.execute_input":"2026-08-23T14:50:55.273023Z","iopub.status.idle":"2026-08-23T14:50:55.2803Z","shell.execute_reply.started":"2026-08-23T14:50:55.27299Z","shell.execute_reply":"2026-08-23T14:50:55.279435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Get a list of patient folders that actually exist on disk\navailable_folders = os.listdir('/kaggle/input/competitions/rsna-2022-cervical-spine-fracture-detection/train_images')\n\n# 2. Get the list of unique patients from metadata\nmeta_patients = seg_labels['StudyInstanceUID'].unique()\n\n# 3. Find the overlapping intersection (patients present in both places)\nvalid_patients = [p for p in meta_patients if p in available_folders]\n\nprint(f\"Total matching patients ready for processing: {len(valid_patients)}\")\nif len(valid_patients) > 0:\n    print(f\"👉 Use this valid Patient ID instead: '{valid_patients[0]}'\")\nelse:\n    print(\"❌ No matching folders found. Please double-check your dataset mount paths.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T14:50:14.577843Z","iopub.execute_input":"2026-08-23T14:50:14.57816Z","iopub.status.idle":"2026-08-23T14:50:14.644553Z","shell.execute_reply.started":"2026-08-23T14:50:14.578133Z","shell.execute_reply":"2026-08-23T14:50:14.643268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study_instance = '1.2.826.0.1.3680043.1363'\norg_images, patient_slices, slices = get_image(study_instance = study_instance)\nprint(f\"✅ Loaded {len(slices)} frames sequentially for patient {study_instance}\")\nprint(f\"First frame index: {slices[0]} | Last frame index: {slices[-1]}\")","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:52:00.457103Z","iopub.execute_input":"2026-08-23T14:52:00.458013Z","iopub.status.idle":"2026-08-23T14:52:06.040316Z","shell.execute_reply.started":"2026-08-23T14:52:00.457956Z","shell.execute_reply":"2026-08-23T14:52:06.039135Z"},"papermill":{"duration":11.211047,"end_time":"2025-04-22T05:41:23.294334","exception":false,"start_time":"2025-04-22T05:41:12.083287","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dat  = seg_labels[seg_labels['StudyInstanceUID']==study_instance][['SOPInstanceUID', 'C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']]\ndat","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:52:17.376365Z","iopub.execute_input":"2026-08-23T14:52:17.37669Z","iopub.status.idle":"2026-08-23T14:52:17.394372Z","shell.execute_reply.started":"2026-08-23T14:52:17.376661Z","shell.execute_reply":"2026-08-23T14:52:17.393493Z"},"papermill":{"duration":0.022912,"end_time":"2025-04-22T05:41:23.323775","exception":false,"start_time":"2025-04-22T05:41:23.300863","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.rc('xtick',labelsize=8)\nplt.rc('ytick',labelsize=8)\n\nstart = 0\nimg = 67\nlabel = dat[dat['SOPInstanceUID']==study_instance+'.1.'+slices[start]][['C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']].to_string(index=False).split('\\n')\nplt.figure(figsize=(11, 8))\n# Ploting pixel array\nplt.subplot(2, 2, 1)\nplt.imshow(org_images[start][img].pixel_array,cmap='bone', aspect='auto')\nplt.title('Original image')\nplt.axis(\"off\")\n\n# Ploting pixel array distribution\nplt.subplot(2, 2, 2)\nplt.hist(org_images[start][img].pixel_array.flatten(),color=\"b\",bins=50)\n# plt.title('Pixel array distribution')\nplt.xlabel(\"Pixel Values\")\nplt.ylabel(\"Fequency\")\n\n#Ploting HU array\nplt.subplot(2, 2, 3)\nplt.imshow(get_pixels_hu(org_images[start][img]),cmap='bone', aspect='auto')\nplt.title('Processed image')\nplt.axis(\"off\")\n\n# Ploting HU distribution\nplt.subplot(2, 2, 4)\nplt.hist(patient_slices[start][img].flatten(),color=\"b\",bins=50)\n# plt.title('HU distribution')\nplt.xlabel(\"HU Values\")\nplt.ylabel(\"Fequency\")\nplt.suptitle(f\"{label[0]} \\n {label[1].strip()}\", y=0.98, fontsize=12)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:52:25.36942Z","iopub.execute_input":"2026-08-23T14:52:25.369749Z","iopub.status.idle":"2026-08-23T14:52:26.053369Z","shell.execute_reply.started":"2026-08-23T14:52:25.369719Z","shell.execute_reply":"2026-08-23T14:52:26.052287Z"},"papermill":{"duration":0.673766,"end_time":"2025-04-22T05:41:24.00363","exception":false,"start_time":"2025-04-22T05:41:23.329864","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label, label[1].split()","metadata":{"execution":{"iopub.status.busy":"2026-08-23T14:52:34.298715Z","iopub.execute_input":"2026-08-23T14:52:34.299789Z","iopub.status.idle":"2026-08-23T14:52:34.305291Z","shell.execute_reply.started":"2026-08-23T14:52:34.299749Z","shell.execute_reply":"2026-08-23T14:52:34.304401Z"},"papermill":{"duration":0.017503,"end_time":"2025-04-22T05:41:24.031252","exception":false,"start_time":"2025-04-22T05:41:24.013749","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dicom(path):\n    # Safety Check: Skip if the file was one of the missing patient slices\n    if not os.path.exists(path):\n        return None\n        \n    img = dicom.dcmread(path)\n    \n    # ─── REMOVED THE BREAKING OVERRIDE LINE ───\n    \n    # Extract the correct structural Hounsfield Unit array naturally\n    data = get_pixels_hu(img)\n    \n    # Normalize the pixel distribution\n    data = data - np.min(data)\n    if np.max(data) != 0:\n        data = data / np.max(data)\n        \n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T14:52:37.082575Z","iopub.execute_input":"2026-08-23T14:52:37.082928Z","iopub.status.idle":"2026-08-23T14:52:37.088693Z","shell.execute_reply.started":"2026-08-23T14:52:37.082887Z","shell.execute_reply":"2026-08-23T14:52:37.087788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Packages image files into structured NumPy arrays for training\ndef ImgDataGenerator(train_df,base_path):\n    '''\n    Function to read dicom image path and store the images as numpy arrays.\n\n    Parameters:\n    train_df: Pandas dataframe.\n    base_path: Python list containing image filepaths.\n\n    Returns:\n    [Train image dataset, Train image labels]\n\n    '''\n    trainset = []\n    trainlabel = []\n    for i in tqdm(range(len(train_df))):\n        study_id = train_df.loc[i,'StudyInstanceUID']\n        slice_id = train_df.loc[i,'slice']+'.dcm'\n        study_path = study_id+'/'+slice_id\n\n        path = os.path.join(base_path, study_path)\n\n        img = load_dicom(path)\n        # If the file didn't exist locally, skip to the next profile seamlessly\n        if img is None:\n            continue\n        img = cv2.resize(img, (128 , 128))\n        image = img_to_array(img)\n        image = image / 255.0\n        trainset += [image]\n        cur_label = [train_df.loc[i,f'C{j}'] for j in range(1,8)]\n        trainlabel += [cur_label]\n\n    return np.array(trainset), np.array(trainlabel)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T14:52:43.717135Z","iopub.execute_input":"2026-08-23T14:52:43.717511Z","iopub.status.idle":"2026-08-23T14:52:43.725795Z","shell.execute_reply.started":"2026-08-23T14:52:43.717481Z","shell.execute_reply":"2026-08-23T14:52:43.72494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def metrics(y_test, y_pred_binary):\n    '''\n    Function to display accuracy, precision, recall and f1-score for the classification task.\n    \n    Parameters:\n    y_test: True labels.\n    y_pred_binary: Predicted binary labels.\n\n    Returns:\n    Pandas dataframe containing class-wise Sensitivity, Specificity, and F1-score.\n    \n    '''\n    classes = np.array(seg_labels.columns[2:-1])\n    df_res = []\n    precision_per_class = precision_score(y_test, y_pred_binary, average=None)\n    recall_per_class = recall_score(y_test, y_pred_binary, average=None)\n    f1_per_class = f1_score(y_test, y_pred_binary, average=None)\n\n    for i in range(len(classes)):\n        df_res.append([classes[i], recall_per_class[i], precision_per_class[i], f1_per_class[i]])\n    df_res = pd.DataFrame(df_res, columns = ['Class','Sensitivity','Specificity', 'F1-score'])\n    return df_res","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T14:52:50.049642Z","iopub.execute_input":"2026-08-23T14:52:50.04997Z","iopub.status.idle":"2026-08-23T14:52:50.056363Z","shell.execute_reply.started":"2026-08-23T14:52:50.04994Z","shell.execute_reply":"2026-08-23T14:52:50.055469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_history(history):\n    '''\n    Function to plot the train and validation accuracy and loss.\n    \n    Parameters:\n    history: model train history\n\n    Returns:\n    None.\n    \n    '''\n    hist = history.history\n    plt.figure(figsize=(8, 4));\n    plt.suptitle(f\"Performance Metrics\", fontsize=12)\n\n    # Actual and validation losses\n    plt.subplot(1, 2, 1);\n    plt.plot(hist['loss'], label='train')\n    plt.plot(hist['val_loss'], label='validation')\n    plt.title('Train and val loss curve', fontsize=8)\n    plt.legend()\n\n    # Actual and validation accuracy\n    plt.subplot(1, 2, 2);\n    plt.plot(hist['binary_accuracy'], label='train')\n    plt.plot(hist['val_binary_accuracy'], label='validation')\n    plt.title('Train and val accuracy curve', fontsize=8)\n    plt.legend();","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T14:53:03.585214Z","iopub.execute_input":"2026-08-23T14:53:03.585596Z","iopub.status.idle":"2026-08-23T14:53:03.592278Z","shell.execute_reply.started":"2026-08-23T14:53:03.585566Z","shell.execute_reply":"2026-08-23T14:53:03.591421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def callback(model_name, patience=5): \n#     '''\n#     Function to define callback for model training.\n    \n#     Parameters:\n#     model_name(string): Name for the saved model with `.h5` extension.\n#     patience: Patience for early stopping. Usually, the value lies between 5-11.\n\n#     Returns:\n#     [Early Stopping Callback, Model Checkpoint Callback]\n    \n#     '''\n#     early_stopping = callbacks.EarlyStopping(patience=patience, restore_best_weights=True)\n#     model_checkpoint = callbacks.ModelCheckpoint(model_name, save_best_only=True)\n#     learning_rate_reduction = callbacks.ReduceLROnPlateau(monitor='val_acc', \n#                                                         patience=2, \n#                                                         verbose=1, \n#                                                         factor=0.5, \n#                                                         min_lr=0.00001)\n#     return [early_stopping, model_checkpoint, learning_rate_reduction]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T10:32:51.985189Z","iopub.execute_input":"2026-08-23T10:32:51.98559Z","iopub.status.idle":"2026-08-23T10:32:51.990935Z","shell.execute_reply.started":"2026-08-23T10:32:51.985562Z","shell.execute_reply":"2026-08-23T10:32:51.990192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Convert train images of segmented studyids to array\n# X_seg, y_seg = ImgDataGenerator(seg_labels,train_images)\n# X_seg.shape,y_seg.shape","metadata":{"execution":{"iopub.status.busy":"2026-08-23T10:33:00.158912Z","iopub.execute_input":"2026-08-23T10:33:00.159547Z","iopub.status.idle":"2026-08-23T10:41:50.382222Z","shell.execute_reply.started":"2026-08-23T10:33:00.159519Z","shell.execute_reply":"2026-08-23T10:41:50.38153Z"},"papermill":{"duration":519.64089,"end_time":"2025-04-22T05:50:03.716764","exception":false,"start_time":"2025-04-22T05:41:24.075874","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import GroupShuffleSplit\n\n# ─── PATIENT-LEVEL SPLIT (prevents data leakage) ───\n# Split by StudyInstanceUID so all slices from one patient stay together\n\n# 1. Get the patient ID for every row in seg_labels\ngroups = seg_labels['StudyInstanceUID'].values\n\n# 2. Create a patient-grouped 90/10 split\ngss = GroupShuffleSplit(n_splits=1, test_size=0.10, random_state=42)\ntrain_idx, test_idx = next(gss.split(seg_labels, groups=groups))\n\n# 3. Build separate label dataframes for each split\ntrain_df = seg_labels.iloc[train_idx].reset_index(drop=True)\ntest_df  = seg_labels.iloc[test_idx].reset_index(drop=True)\n\n# 4. Verify NO patient overlap (critical check)\ntrain_patients = set(train_df['StudyInstanceUID'])\ntest_patients  = set(test_df['StudyInstanceUID'])\noverlap = train_patients & test_patients\n\nprint(f\"Train patients: {len(train_patients)}\")\nprint(f\"Test  patients: {len(test_patients)}\")\nprint(f\"Patient overlap: {len(overlap)}  <-- MUST be 0\")\nprint(f\"Train slices: {len(train_df)} | Test slices: {len(test_df)}\")\n\nassert len(overlap) == 0, \"❌ LEAKAGE: patients appear in both sets!\"\nprint(\"✅ Patient-level split verified - no leakage\")\n\n# 5. Generate images SEPARATELY for each patient group\nprint(\"\\n📦 Loading training slices...\")\nX_train, y_train = ImgDataGenerator(train_df, train_images)\n\nprint(\"📦 Loading test slices...\")\nX_test, y_test = ImgDataGenerator(test_df, train_images)\n\n# 6. Cast labels to float32 (as before)\ny_train, y_test = y_train.astype('float32'), y_test.astype('float32')\n\nprint(f\"\\nX_train: {X_train.shape} | y_train: {y_train.shape}\")\nprint(f\"X_test:  {X_test.shape} | y_test:  {y_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T14:53:12.724746Z","iopub.execute_input":"2026-08-23T14:53:12.725145Z","iopub.status.idle":"2026-08-23T14:55:19.890227Z","shell.execute_reply.started":"2026-08-23T14:53:12.725104Z","shell.execute_reply":"2026-08-23T14:55:19.888929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Divide train and test data\n# X_train, X_test, y_train, y_test = train_test_split(X_seg, y_seg, random_state=42, test_size=0.1)\n# y_train, y_test = y_train.astype('float32'), y_test.astype('float32')\n# X_train.shape, y_train.shape, X_test.shape, y_test.shape","metadata":{"execution":{"iopub.status.busy":"2026-08-22T08:33:29.963526Z","iopub.execute_input":"2026-08-22T08:33:29.964449Z","iopub.status.idle":"2026-08-22T08:33:30.518576Z","shell.execute_reply.started":"2026-08-22T08:33:29.964416Z","shell.execute_reply":"2026-08-22T08:33:30.517958Z"},"papermill":{"duration":6.470738,"end_time":"2025-04-22T05:50:10.380948","exception":false,"start_time":"2025-04-22T05:50:03.91021","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = ['C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']","metadata":{"execution":{"iopub.status.busy":"2026-08-22T08:33:34.530438Z","iopub.execute_input":"2026-08-22T08:33:34.531184Z","iopub.status.idle":"2026-08-22T08:33:34.534937Z","shell.execute_reply.started":"2026-08-22T08:33:34.531154Z","shell.execute_reply":"2026-08-22T08:33:34.534051Z"},"papermill":{"duration":0.248936,"end_time":"2025-04-22T05:50:10.82503","exception":false,"start_time":"2025-04-22T05:50:10.576094","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare data batches for training images\nX_train_tensor = tf.data.Dataset.from_tensor_slices(X_train)\ny_train_tensor = tf.data.Dataset.from_tensor_slices(y_train)\ntrain_dataset = tf.data.Dataset.zip((X_train_tensor, y_train_tensor)).batch(16).prefetch(tf.data.AUTOTUNE)\n# Prepare data batches for validation images\nX_test_tensor = tf.data.Dataset.from_tensor_slices(X_test)\ny_test_tensor = tf.data.Dataset.from_tensor_slices(y_test)\nval_dataset = tf.data.Dataset.zip((X_test_tensor, y_test_tensor)).batch(16).prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2026-08-22T08:33:38.201606Z","iopub.execute_input":"2026-08-22T08:33:38.202055Z","iopub.status.idle":"2026-08-22T08:33:42.475477Z","shell.execute_reply.started":"2026-08-22T08:33:38.202025Z","shell.execute_reply":"2026-08-22T08:33:42.47453Z"},"papermill":{"duration":13.224773,"end_time":"2025-04-22T05:50:24.242685","exception":false,"start_time":"2025-04-22T05:50:11.017912","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Modelling\nWe'll opt for three different modelling approaches:\n1. **Custom CNN** - Here we will implement our own custom CNN model from scratch involving blocks of Convolution, Pooling and DropOut layers.\n2. **Transfer Learning models** - In this section, we will employ various pre-trained deep learning models to further improve on the results of the Custom CNN model.\n3. **Encoder decoder Architecture** - Finally, we will implement an encoder-decoder model where we will use the U-Net model for encoder and the best performing transfer learning model as a decoder. ","metadata":{"papermill":{"duration":0.197724,"end_time":"2025-04-22T05:50:24.633924","exception":false,"start_time":"2025-04-22T05:50:24.4362","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Network Architecture (MobileNetV2 U-Net)\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, Model\nfrom tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\n\n# ─── Standard Convolution Block ───\ndef conv_block(inputs, num_filters):\n    x = layers.Conv2D(num_filters, 3, padding=\"same\")(inputs)\n    x = layers.BatchNormalization()(x)\n    x = layers.Activation(\"relu\")(x)\n    x = layers.Conv2D(num_filters, 3, padding=\"same\")(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Activation(\"relu\")(x)\n    return x\n\n# ─── Standard Decoder Block ───\ndef decoder_block(inputs, skip_features, num_filters):\n    x = layers.Conv2DTranspose(num_filters, (2, 2), strides=2, padding=\"same\")(inputs)\n    x = layers.Concatenate()([x, skip_features])\n    x = conv_block(x, num_filters)\n    return x\n\n# ─── Dynamic Index-Based U-Net Builder ───\ndef build_mobilenet_unet(input_shape=(128, 128, 1)):\n    inputs = layers.Input(input_shape)\n    \n    # Force single-channel grayscale up to 3 channels for MobileNetV2\n    x_rgb = layers.Lambda(lambda x: tf.repeat(x, repeats=3, axis=-1))(inputs)\n    \n    encoder = MobileNetV2(include_top=False, weights=\"imagenet\", input_tensor=x_rgb)\n    encoder.trainable = True \n    \n    s1 = encoder.layers[0].output                               \n    s2 = encoder.layers[11].output                              \n    s3 = encoder.layers[29].output                              \n    s4 = encoder.layers[56].output                              \n    bridge = encoder.layers[117].output                         \n\n    d1 = decoder_block(bridge, s4, 256)                          \n    d2 = decoder_block(d1, s3, 128)                              \n    d3 = decoder_block(d2, s2, 64)                               \n    d4 = decoder_block(d3, s1, 32)                               \n    \n    gap = layers.GlobalAveragePooling2D()(d4)\n    dropout = layers.Dropout(0.3)(gap)\n    outputs = layers.Dense(7, activation=\"sigmoid\")(dropout)     \n    \n    return Model(inputs, outputs, name=\"Fixed-Grayscale-MobileNetV2-UNet\")\n\n# ─── INITIALIZE FRESH MODEL ───\nmodel_0 = build_mobilenet_unet(input_shape=(128, 128, 1))\n\n# ─── model summary layer by layer ───\nmodel_0.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T08:34:06.432646Z","iopub.execute_input":"2026-08-22T08:34:06.433456Z","iopub.status.idle":"2026-08-22T08:47:54.158106Z","shell.execute_reply.started":"2026-08-22T08:34:06.433424Z","shell.execute_reply":"2026-08-22T08:47:54.157049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # ─── Configures the loss function, evaluation metric, and optimizer ───\n# model_0.compile(\n#     loss=\"binary_crossentropy\",\n#     optimizer=keras.optimizers.SGD(learning_rate=0.01, momentum=0.9, nesterov=True),\n#     metrics=[tf.keras.metrics.BinaryAccuracy()]\n# )\n\n# # ─── Run the full training loop using the custom callback configuration ───\n# early_stopping = EarlyStopping(monitor='val_loss', patience=15, restore_best_weights=True)\n# model_checkpoint = ModelCheckpoint(\"model_0_Cervical_UNet.keras\", save_best_only=True, monitor='val_loss', mode='min')\n\n# print(\"🚀 Starting a clean training run on your 1-channel tensors...\")\n# history = model_0.fit(\n#     train_dataset,\n#     epochs=2,\n#     validation_data=val_dataset,\n#     callbacks=[early_stopping, model_checkpoint]\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T09:08:05.149814Z","iopub.execute_input":"2026-08-22T09:08:05.150454Z","iopub.status.idle":"2026-08-22T09:14:51.684007Z","shell.execute_reply.started":"2026-08-22T09:08:05.150422Z","shell.execute_reply":"2026-08-22T09:14:51.682923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─── Weighted loss to fix majority-class collapse ───\ndef weighted_bce(pos_weight=5.0):\n    \"\"\"\n    Penalizes missed fractures ~pos_weight× harder than false alarms.\n    Forces the model to actually predict positives instead of\n    collapsing to 'always Healthy'.\n    \"\"\"\n    def loss(y_true, y_pred):\n        y_true = tf.cast(y_true, tf.float32)\n        y_pred = tf.clip_by_value(y_pred, 1e-7, 1 - 1e-7)\n        bce = -(pos_weight * y_true * tf.math.log(y_pred) +\n                (1 - y_true) * tf.math.log(1 - y_pred))\n        return tf.reduce_mean(bce)\n    return loss\n\n# ─── Configures the loss function, evaluation metrics, and optimizer ───\nmodel_0.compile(\n    loss=weighted_bce(pos_weight=3.0),\n    optimizer=keras.optimizers.SGD(learning_rate=0.001, momentum=0.9, nesterov=True),\n    metrics=[\n        tf.keras.metrics.BinaryAccuracy(name=\"binary_accuracy\"),\n        tf.keras.metrics.Recall(name=\"recall\"),\n        tf.keras.metrics.AUC(name=\"auc\")\n    ]\n)\n\n# ─── Callbacks: monitor recall/AUC, NOT accuracy ───\nearly_stopping = EarlyStopping(\n    monitor='val_auc',\n    mode='max',\n    patience=15,\n    restore_best_weights=True\n)\n\nmodel_checkpoint = ModelCheckpoint(\n    \"model_0_Cervical_UNet.keras\",\n    save_best_only=True,\n    monitor='val_auc',\n    mode='max'\n)\n\nprint(\"🚀 Starting final training run on 100 epochs on 1 channel...\")\nhistory = model_0.fit(\n    train_dataset,\n    epochs=100,\n    validation_data=val_dataset,\n    callbacks=[early_stopping, model_checkpoint]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T12:14:06.436755Z","iopub.execute_input":"2026-08-23T12:14:06.437021Z","iopub.status.idle":"2026-08-23T12:14:06.459294Z","shell.execute_reply.started":"2026-08-23T12:14:06.43699Z","shell.execute_reply":"2026-08-23T12:14:06.458189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom as dicom\nimport matplotlib.pyplot as plt\n\n# 1. Load the exact same DICOM image\nimage_path = '/kaggle/input/competitions/rsna-2022-cervical-spine-fracture-detection/train_images/1.2.826.0.1.3680043.10001/150.dcm'\nmedical_data = dicom.dcmread(image_path)\nimg = medical_data.pixel_array\n\n# 2. Plot the image using Matplotlib\nplt.figure(figsize=(6, 6))\nplt.imshow(img, cmap='bone')  # 'bone' applies the official grayscale radiology look\nplt.title(f\"Patient Scan Slice: {image_path.split('/')[-1]}\\nID: 1.2.826.0.1.3680043.10001\", fontsize=12, pad=10)\nplt.axis('off')  # Hides graph grid lines to keep the medical view clean\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T09:30:14.825577Z","iopub.execute_input":"2026-08-22T09:30:14.8264Z","iopub.status.idle":"2026-08-22T09:30:15.022064Z","shell.execute_reply.started":"2026-08-22T09:30:14.826367Z","shell.execute_reply":"2026-08-22T09:30:15.021244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom as dicom\nimport cv2\nimport numpy as np\n\n# 1. Point to a real DICOM file inside the attached RSNA competition folder\nimage_path = '/kaggle/input/competitions/rsna-2022-cervical-spine-fracture-detection/train_images/1.2.826.0.1.3680043.10001/150.dcm'\n\n# 2. Read the medical scan file safely using pydicom\ntry:\n    medical_data = dicom.dcmread(image_path)\n    img = medical_data.pixel_array  # Extract raw 2D pixel matrix\n    print(\"✅ Real medical DICOM image loaded successfully!\")\nexcept Exception as e:\n    raise FileNotFoundError(f\"Could not read the DICOM file. Please ensure the dataset is attached. Error: {e}\")\n\n# 3. Resize the matrix to match your model's target dimensions (128x128)\nimg = cv2.resize(img, (128, 128))  \n\n# 4. Standardize and normalize intensity distribution to a [0, 1] range\nimg = img - np.min(img)\nif np.max(img) != 0:\n    img = img / np.max(img)\n\n# 5. Add Batch dimension (axis 0) and Single Grayscale Channel dimension (axis -1)\n# This transforms your matrix from (128, 128) -> safely into (1, 128, 128, 1)\nimg_array = np.expand_dims(img, axis=(0, -1))  \nimg_array = img_array.astype(np.float32)\n\nprint(\"Array shape heading to your model:\", img_array.shape)\n\n# 6. Predict using your active trained model variable (model_0)\nprediction = model_0.predict(img_array)\nprint(\"Prediction Matrix (C1-C7 Probabilities):\", prediction)","metadata":{"execution":{"iopub.status.busy":"2026-08-22T09:32:25.018924Z","iopub.execute_input":"2026-08-22T09:32:25.01939Z","iopub.status.idle":"2026-08-22T09:32:25.124913Z","shell.execute_reply.started":"2026-08-22T09:32:25.019356Z","shell.execute_reply":"2026-08-22T09:32:25.124003Z"},"papermill":{"duration":12.911578,"end_time":"2025-04-22T05:59:29.919808","exception":false,"start_time":"2025-04-22T05:59:17.00823","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Computes the overall patient fracture probability\n\nimport numpy as np\n\n# 1. Structural map matching your model's 7 outputs\nVERTEBRAE_MAP = {\n    0: \"Cervical Vertebra C1\",\n    1: \"Cervical Vertebra C2\",\n    2: \"Cervical Vertebra C3\",\n    3: \"Cervical Vertebra C4\",\n    4: \"Cervical Vertebra C5\",\n    5: \"Cervical Vertebra C6\",\n    6: \"Cervical Vertebra C7\"\n}\n\ndef classify_spine_prediction(prediction, threshold=0.5):\n    \"\"\"\n    Translates C1-C7 multi-label probabilities and computes the patient_overall prediction.\n    \"\"\"\n    # Extract the first row of predictions from the batch array\n    prob_vector = prediction[0]\n    \n    # ─── CALCULATE THE OMITTED OVERALL PATIENT TARGET ───\n    # Probability that no fractures exist across any vertebrae\n    prob_no_fractures = np.prod(1.0 - prob_vector)\n    # The overall patient fracture probability is the complement\n    patient_overall_prob = 1.0 - prob_no_fractures\n    \n    # Identify which individual vertebrae cross our classification threshold\n    fractured_indices = np.where(prob_vector >= threshold)[0]\n    \n    # Format individual vertebrae results\n    if len(fractured_indices) == 0:\n        vertebrae_text = \"no individual vertebra fractures crossing threshold parameters\"\n    else:\n        detected_locations = [VERTEBRAE_MAP[idx] for idx in fractured_indices]\n        vertebrae_text = f\"cervical spine fracture localized at {', '.join(detected_locations)}\"\n        \n    # Build complete diagnostic summary including the overall column metrics\n    diagnostic_summary = (\n        f\"Overall Patient Fracture Probability: {patient_overall_prob:.2%}. \"\n        f\"Anatomical breakdown indicates a {vertebrae_text}.\"\n    )\n    \n    return diagnostic_summary\n\n# 2. Run prediction vector through the updated overall-aware translator\nresult = classify_spine_prediction(prediction)\nprint(result)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T09:32:42.749809Z","iopub.execute_input":"2026-08-22T09:32:42.75064Z","iopub.status.idle":"2026-08-22T09:32:42.757833Z","shell.execute_reply.started":"2026-08-22T09:32:42.750609Z","shell.execute_reply":"2026-08-22T09:32:42.756998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Check if history exists in the workspace\nif 'history' in locals():\n    hist = history.history\n    epochs_range = range(1, len(hist['loss']) + 1)\n    \n    plt.figure(figsize=(12, 5))\n    \n    # Plot 1: Binary Crossentropy Loss Curve\n    plt.subplot(1, 2, 1)\n    plt.plot(epochs_range, hist['loss'], label='Training Loss', color='royalblue', lw=2)\n    plt.plot(epochs_range, hist['val_loss'], label='Validation Loss', color='darkorange', lw=2)\n    plt.title('Multi-Label Optimization: Loss Curve', fontsize=12, pad=10)\n    plt.xlabel('Epochs', fontsize=10)\n    plt.ylabel('Loss Value', fontsize=10)\n    plt.legend(fontsize=10)\n    plt.grid(True, linestyle='--', alpha=0.5)\n    \n    # Plot 2: Binary Accuracy Curve\n    plt.subplot(1, 2, 2)\n    # Note: Keras 3 keys might vary slightly, using a fallback check\n    train_acc_key = 'binary_accuracy' if 'binary_accuracy' in hist else 'accuracy'\n    val_acc_key = 'val_binary_accuracy' if 'val_binary_accuracy' in hist else 'val_accuracy'\n    \n    plt.plot(epochs_range, hist[train_acc_key], label='Training Accuracy', color='royalblue', lw=2)\n    plt.plot(epochs_range, hist[val_acc_key], label='Validation Accuracy', color='darkorange', lw=2)\n    plt.title('Vertebrae Classification: Binary Accuracy', fontsize=12, pad=10)\n    plt.xlabel('Epochs', fontsize=10)\n    plt.ylabel('Accuracy Score', fontsize=10)\n    plt.legend(fontsize=10)\n    plt.grid(True, linestyle='--', alpha=0.5)\n    \n    plt.tight_layout()\n    plt.savefig('cervical_spine_training_metrics.png', dpi=300)\n    plt.show()\n    \n    print(f\"📊 Final Metrics for Presentation:\")\n    print(f\"🥇 Best Training Accuracy:   {max(hist[train_acc_key]):.2%}\")\n    print(f\"🎯 Best Validation Accuracy: {max(hist[val_acc_key]):.2%}\")\nelse:\n    print(\"⚠️ 'history' object not found. Run this block after your background training completes.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T14:18:53.566978Z","iopub.execute_input":"2026-08-22T14:18:53.567159Z","iopub.status.idle":"2026-08-22T14:18:53.58015Z","shell.execute_reply.started":"2026-08-22T14:18:53.56714Z","shell.execute_reply":"2026-08-22T14:18:53.579359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pydicom as dicom\nimport cv2\nimport numpy as np\nfrom tqdm import tqdm\n\n# 1. Base directory path matching your Kaggle environment layout\nBASE_DIR = '/kaggle/input/competitions/rsna-2022-cervical-spine-fracture-detection'\ntest_csv_path = os.path.join(BASE_DIR, 'test.csv')\n\n# Use test.csv to get the exact active test patient folder list visible on your machine\nif os.path.exists(test_csv_path):\n    test_df = pd.read_csv(test_csv_path)\n    test_patients = test_df['StudyInstanceUID'].unique()\n    test_predictions = []\n    \n    print(f\"🔬 Found {len(test_patients)} sample test folders on disk. Running inference preview...\")\n    \n    # 2. Loop through every active test patient folder\n    for patient_id in tqdm(test_patients):\n        patient_dir = os.path.join(BASE_DIR, 'test_images', patient_id)\n        \n        if os.path.exists(patient_dir):\n            # Grab all slice files and sort them to calculate the exact middle index\n            slices = sorted(os.listdir(patient_dir))\n            middle_slice = slices[len(slices) // 2]\n            slice_path = os.path.join(patient_dir, middle_slice)\n            \n            # Read and process the test slice matrix\n            medical_data = dicom.dcmread(slice_path)\n            img = cv2.resize(medical_data.pixel_array, (128, 128))\n            img = (img - np.min(img)) / (np.max(img) - np.min(img) if np.max(img) != np.min(img) else 1)\n            img_array = np.expand_dims(img, axis=(0, -1)).astype(np.float32)\n            \n            # Run inference using your active model_0 instance weights\n            # Added a flat squeeze to map the prediction array seamlessly\n            preds = np.squeeze(model_0.predict(img_array, verbose=0)) \n            \n            # Compute the overall patient target probability programmatically\n            prob_no_fractures = np.prod(1.0 - preds)\n            patient_overall_prob = 1.0 - prob_no_fractures\n            \n            # Append predictions matching Kaggle's row_id constraints: C1-C7\n            for i in range(1, 8):\n                test_predictions.append({\"row_id\": f\"{patient_id}_C{i}\", \"fracture\": preds[i-1]})\n            # Append overall patient target score\n            test_predictions.append({\"row_id\": f\"{patient_id}_patient_overall\", \"fracture\": patient_overall_prob})\n        else:\n            # Fallback if a directory pointer error occurs\n            for i in range(1, 8):\n                test_predictions.append({\"row_id\": f\"{patient_id}_C{i}\", \"fracture\": 0.5})\n            test_predictions.append({\"row_id\": f\"{patient_id}_patient_overall\", \"fracture\": 0.5})\n\n    # 3. Create the final dataframe and export it to disk\n    submission_df = pd.DataFrame(test_predictions)\n    submission_df.to_csv('submission.csv', index=False)\n    print(\"🎉 Success! 'submission.csv' generated and saved to your output directory.\")\n    \n    # ─── VISUAL DATA INSPECTION ───\n    print(\"\\n\" + \"=\"*25 + \" SUBMISSION PREVIEW \" + \"=\"*25 + \"\\n\")\n    # Display the first 16 rows to show full classification distribution maps for C1-C7 and overall\n    display(submission_df.head(16))\n    \nelse:\n    print(f\"❌ Core test configuration mapping file not found at: {test_csv_path}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T12:57:02.096093Z","iopub.execute_input":"2026-08-22T12:57:02.096768Z","iopub.status.idle":"2026-08-22T12:57:03.401886Z","shell.execute_reply.started":"2026-08-22T12:57:02.096724Z","shell.execute_reply":"2026-08-22T12:57:03.401269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.metrics import classification_report\n\nprint(\"🧠 Using trained model for validation evaluation...\")\n\n# Use the already-trained model in memory\nevaluated_model = model_0\n\n# Extract true labels and predictions from validation dataset\nprint(\"📦 Unpacking validation dataset batches...\")\n\ny_true_list = []\ny_pred_list = []\n\nfor images, labels in val_dataset:\n    preds = evaluated_model.predict(images, verbose=0)\n\n    y_true_list.append(labels.numpy())\n    y_pred_list.append(preds)\n\n# Concatenate all batches\ny_true = np.vstack(y_true_list)\ny_pred_probs = np.vstack(y_pred_list)\n\n# Convert probabilities to binary predictions\ny_pred = (y_pred_probs >= 0.5).astype(int)\n\n# Generate classification report\nvertebrae_names = [f\"Vertebra C{i}\" for i in range(1, 8)]\n\nreport = classification_report(\n    y_true,\n    y_pred,\n    target_names=vertebrae_names,\n    digits=4\n)\n\nprint(\"\\n\" + \"=\"*20 + \" 📊 MULTI-LABEL CLASSIFICATION REPORT (VALIDATION SET) \" + \"=\"*20 + \"\\n\")\nprint(report)\nprint(\"=\"*95)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T14:22:12.966539Z","iopub.execute_input":"2026-08-22T14:22:12.967375Z","iopub.status.idle":"2026-08-22T14:22:26.834061Z","shell.execute_reply.started":"2026-08-22T14:22:12.967344Z","shell.execute_reply":"2026-08-22T14:22:26.833313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\nprint(\"🧠 Using trained model already in memory...\")\n\nevaluated_model = model_0\n\nif 'val_dataset' in locals():\n    print(\"📦 Processing validation data streams...\")\n\n    y_true_list, y_pred_list = [], []\n\n    for images, labels in val_dataset:\n        preds = evaluated_model.predict(images, verbose=0)\n\n        y_true_list.append(labels.numpy())\n        y_pred_list.append(preds)\n\n    y_true = np.vstack(y_true_list)\n    y_pred = (np.vstack(y_pred_list) >= 0.5).astype(int)\n\n    fig, axes = plt.subplots(2, 4, figsize=(20, 10))\n    axes = axes.flatten()\n\n    vertebrae_labels = [f\"Vertebra C{i}\" for i in range(1, 8)]\n\n    print(\"📊 Rendering confusion matrix heatmaps...\")\n\n    for i in range(7):\n        cm = confusion_matrix(y_true[:, i], y_pred[:, i], labels=[0, 1])\n\n        disp = ConfusionMatrixDisplay(\n            confusion_matrix=cm,\n            display_labels=[\"Healthy\", \"Fractured\"]\n        )\n\n        disp.plot(\n            ax=axes[i],\n            cmap=\"Blues\",\n            colorbar=False,\n            values_format=\"d\"\n        )\n\n        axes[i].set_title(\n            f\"Confusion Matrix: {vertebrae_labels[i]}\",\n            fontsize=12,\n            pad=8\n        )\n\n    fig.delaxes(axes[7])\n\n    plt.tight_layout()\n    plt.savefig(\n        \"vertebrae_confusion_matrices.png\",\n        dpi=300,\n        bbox_inches=\"tight\"\n    )\n    plt.show()\n\n    print(\"🎉 Success! 'vertebrae_confusion_matrices.png' generated.\")\n\nelse:\n    print(\"⚠️ val_dataset is missing.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-22T13:46:45.828946Z","iopub.execute_input":"2026-08-22T13:46:45.829528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import numpy as np\n# import os\n\n# seg_path = '/kaggle/input/datasets/saikiranvarma/rsna-cervical-fracture-segmentations-npy/npy_segmentations'\n\n# files = os.listdir(seg_path)\n# print(f\"Total mask files: {len(files)}\")\n# print(f\"Sample filenames: {files[:3]}\")\n\n# sample = np.load(os.path.join(seg_path, files[0]))\n# print(f\"\\nSample shape: {sample.shape}\")\n# print(f\"Data type: {sample.dtype}\")\n# print(f\"Unique values: {np.unique(sample)}\")\n# print(f\"Min: {sample.min()}, Max: {sample.max()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-23T14:39:56.044738Z","iopub.execute_input":"2026-08-23T14:39:56.045028Z","iopub.status.idle":"2026-08-23T14:39:58.777968Z","shell.execute_reply.started":"2026-08-23T14:39:56.044999Z","shell.execute_reply":"2026-08-23T14:39:58.776939Z"}},"outputs":[],"execution_count":null}]}