{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":36363,"databundleVersionId":4050810,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Install essential DICOM handlers\n!pip install pylibjpeg pylibjpeg-libjpeg gdcm --quiet\n\n# Core imports\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport matplotlib.pyplot as plt\nimport os\nimport cv2\nfrom tqdm import tqdm\n\n# TF/Keras imports\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, callbacks\n\n# Verify TF version and GPU\nprint(\"TF Version:\", tf.__version__)\nprint(\"GPU Available:\", tf.config.list_physical_devices('GPU'))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:11:45.799603Z","iopub.execute_input":"2025-03-27T17:11:45.799956Z","iopub.status.idle":"2025-03-27T17:11:48.984378Z","shell.execute_reply.started":"2025-03-27T17:11:45.799932Z","shell.execute_reply":"2025-03-27T17:11:48.983477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# First, fix numpy version compatibility\n!pip install --force-reinstall numpy==1.24.3\n%reset -f  # Restart Python session\n\n# Reinstall core packages with version locking\n!pip install --no-cache-dir \\\n    pydicom==2.4.3 \\\n    tensorflow==2.17.1 \\\n    matplotlib==3.7.5 \\\n    pandas==2.0.3\n\n# Verify installation\nimport numpy as np\nprint(\"NumPy Version:\", np.__version__)  # Should be 1.24.3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:11:48.985741Z","iopub.execute_input":"2025-03-27T17:11:48.985991Z","iopub.status.idle":"2025-03-27T17:12:07.379571Z","shell.execute_reply.started":"2025-03-27T17:11:48.985968Z","shell.execute_reply":"2025-03-27T17:12:07.378494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Corrected analyze_study function\ndef analyze_study(study_uid):\n    path = f\"/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images/{study_uid}\"\n    if not os.path.exists(path):\n        return None\n    \n    try:\n        slices = sorted(os.listdir(path), key=lambda x: int(x.split('.')[0]))\n        sample_slice = pydicom.dcmread(f\"{path}/{slices[0]}\")  # Fixed missing }\n        \n        return {\n            'num_slices': len(slices),\n            'slice_thickness': sample_slice.get('SliceThickness', 'Missing'),\n            'pixel_spacing': sample_slice.get('PixelSpacing', [1.0, 1.0]),\n            'modality': sample_slice.get('Modality', 'CT')\n        }\n    except Exception as e:\n        print(f\"Error analyzing {study_uid}: {str(e)}\")\n        return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:12:07.381799Z","iopub.execute_input":"2025-03-27T17:12:07.382171Z","iopub.status.idle":"2025-03-27T17:12:07.387677Z","shell.execute_reply.started":"2025-03-27T17:12:07.382132Z","shell.execute_reply":"2025-03-27T17:12:07.386831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Completely clean the environment\n!pip install --force-reinstall --ignore-installed \\\n    numpy==1.23.5 \\\n    tensorflow==2.12.0 \\\n    pandas==1.5.3 \\\n    matplotlib==3.7.1 \\\n    pydicom==2.3.1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:12:07.389172Z","iopub.execute_input":"2025-03-27T17:12:07.389484Z","iopub.status.idle":"2025-03-27T17:13:03.578698Z","shell.execute_reply.started":"2025-03-27T17:12:07.389455Z","shell.execute_reply":"2025-03-27T17:13:03.577434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\n\nprint(\"NumPy:\", np.__version__)\nprint(\"TF:\", tf.__version__)\nprint(\"GPU Available:\", tf.config.list_physical_devices('GPU'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:13:03.579732Z","iopub.execute_input":"2025-03-27T17:13:03.580102Z","iopub.status.idle":"2025-03-27T17:13:03.587155Z","shell.execute_reply.started":"2025-03-27T17:13:03.580067Z","shell.execute_reply":"2025-03-27T17:13:03.586235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# First, clean existing installations\n!pip uninstall -y tensorflow numpy\n\n# Install compatible versions\n!pip install --no-cache-dir \\\n    tensorflow==2.13.0 \\\n    numpy==1.24.3 \\\n    pydicom==2.3.1\n\n# Restart kernel after this","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:13:03.588022Z","iopub.execute_input":"2025-03-27T17:13:03.58829Z","iopub.status.idle":"2025-03-27T17:14:02.1676Z","shell.execute_reply.started":"2025-03-27T17:13:03.588256Z","shell.execute_reply":"2025-03-27T17:14:02.16675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\n\nprint(\"TF Version:\", tf.__version__)\nprint(\"NumPy Version:\", np.__version__)\nprint(\"GPU Available:\", tf.config.list_physical_devices('GPU'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:02.168724Z","iopub.execute_input":"2025-03-27T17:14:02.169053Z","iopub.status.idle":"2025-03-27T17:14:02.175551Z","shell.execute_reply.started":"2025-03-27T17:14:02.169019Z","shell.execute_reply":"2025-03-27T17:14:02.174626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clean existing installations\n!pip uninstall -y tensorflow numpy keras\n\n# Install specific compatible versions\n!pip install --user --no-cache-dir \\\n    tensorflow==2.12.0 \\\n    numpy==1.24.3 \\\n    protobuf==3.20.3 \\\n    absl-py==1.4.0 \\\n    pydicom==2.3.1\n\n# Restart kernel after installation","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:02.179285Z","iopub.execute_input":"2025-03-27T17:14:02.179711Z","iopub.status.idle":"2025-03-27T17:14:06.701519Z","shell.execute_reply.started":"2025-03-27T17:14:02.179529Z","shell.execute_reply":"2025-03-27T17:14:06.700668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nprint(\"TF Version:\", tf.__version__)\nprint(\"Initialization successful:\", tf.constant(1) + tf.constant(1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:06.703332Z","iopub.execute_input":"2025-03-27T17:14:06.703555Z","iopub.status.idle":"2025-03-27T17:14:06.710043Z","shell.execute_reply.started":"2025-03-27T17:14:06.703534Z","shell.execute_reply":"2025-03-27T17:14:06.709052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ndef load_dicom_v2(path):\n    try:\n        dicom = pydicom.dcmread(path)\n        img = dicom.pixel_array.astype(float)\n        img = (img - img.min()) / (img.max() - img.min() + 1e-8)\n        return img.astype(np.float32)\n    except Exception as e:\n        print(f\"Error loading {path}: {str(e)}\")\n        return None\n\n# Test loading\nsample_path = \"/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images/1.2.826.0.1.3680043.10001/1.dcm\"\ntest_img = load_dicom_v2(sample_path)\n\nif test_img is not None:\n    plt.figure(figsize=(6,6))\n    plt.imshow(test_img, cmap='bone')\n    plt.title(\"Sample Cervical Spine Image\")\n    plt.axis('off')\n    plt.show()\n    print(f\"Image shape: {test_img.shape}, Value range: [{test_img.min():.2f}, {test_img.max():.2f}]\")\nelse:\n    print(\"Failed to load sample image\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:06.710993Z","iopub.execute_input":"2025-03-27T17:14:06.711206Z","iopub.status.idle":"2025-03-27T17:14:06.958606Z","shell.execute_reply.started":"2025-03-27T17:14:06.711186Z","shell.execute_reply":"2025-03-27T17:14:06.957758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Load Data\nimport pandas as pd\n\ntry:\n    train_df = pd.read_csv('/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv')\n    print(\"Training data loaded successfully!\")\n    print(\"Columns:\", train_df.columns.tolist())\n    print(\"\\nFirst 3 rows:\")\n    display(train_df.head(3))\nexcept FileNotFoundError:\n    print(\"Error: Training CSV file not found. Check dataset path!\")\nexcept Exception as e:\n    print(f\"Unexpected error: {str(e)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:20:26.034124Z","iopub.execute_input":"2025-03-27T17:20:26.034419Z","iopub.status.idle":"2025-03-27T17:20:26.043662Z","shell.execute_reply.started":"2025-03-27T17:20:26.034395Z","shell.execute_reply":"2025-03-27T17:20:26.042924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ncsv_path = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv'\nprint(f\"File exists: {os.path.exists(csv_path)}\")  # Should output \"True\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:32:49.840602Z","iopub.execute_input":"2025-03-27T17:32:49.840938Z","iopub.status.idle":"2025-03-27T17:32:49.850484Z","shell.execute_reply.started":"2025-03-27T17:32:49.840913Z","shell.execute_reply":"2025-03-27T17:32:49.849656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(csv_path, 'r') as f:\n    lines = [next(f) for _ in range(5)]\nprint(\"Raw CSV lines:\")\nfor line in lines:\n    print(repr(line))  # Show hidden characters like quotes or commas","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:33:08.703923Z","iopub.execute_input":"2025-03-27T17:33:08.704204Z","iopub.status.idle":"2025-03-27T17:33:08.710379Z","shell.execute_reply.started":"2025-03-27T17:33:08.704182Z","shell.execute_reply":"2025-03-27T17:33:08.709503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ntry:\n    train_df = pd.read_csv(\n        csv_path,\n        dtype=str,  # Load everything as strings\n        engine='python',  # Use Python parser for better error handling\n        quotechar='\"',  # Handle quoted fields properly\n        on_bad_lines='warn'  # Flag problematic rows\n    )\n    \n    print(\"Temporary columns:\", train_df.columns.tolist())\n    \n    # Clean all columns of unexpected characters\n    for col in train_df.columns:\n        train_df[col] = train_df[col].str.replace(r'[^a-zA-Z0-9\\.]', '', regex=True)\n    \n    # Convert numeric columns\n    numeric_cols = ['patient_overall', 'C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']\n    train_df[numeric_cols] = train_df[numeric_cols].apply(pd.to_numeric, errors='coerce')\n    \n    print(\"\\nData types after cleanup:\")\n    print(train_df.dtypes)\n    print(\"\\nSample data:\")\n    print(train_df.head(3).to_string())\n    \nexcept pd.errors.ParserError as e:\n    print(f\"CSV Parsing Error: {str(e)}\")\nexcept Exception as e:\n    print(f\"Unexpected error: {str(e)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:33:27.758051Z","iopub.execute_input":"2025-03-27T17:33:27.758368Z","iopub.status.idle":"2025-03-27T17:33:27.772632Z","shell.execute_reply.started":"2025-03-27T17:33:27.758342Z","shell.execute_reply":"2025-03-27T17:33:27.771736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check available columns\nprint(\"Training Data Columns:\")\nprint(train_df.columns.tolist())\n\n# Check study-slice relationship\nprint(\"\\nUnique Studies:\", train_df['StudyInstanceUID'].nunique())\nprint(\"Total Rows:\", len(train_df))\n\n# Verify one study\nsample_study = train_df.iloc[0]['StudyInstanceUID']# Check available columns\nprint(\"Training Data Columns:\")\nprint(train_df.columns.tolist())\n\n# Check study-slice relationship\nprint(\"\\nUnique Studies:\", train_df['StudyInstanceUID'].nunique())\nprint(\"Total Rows:\", len(train_df))\n\n# Verify one study\nsample_study = train_df.iloc[0]['StudyInstanceUID']\nstudy_slices = train_df[train_df['StudyInstanceUID'] == sample_study]\nprint(f\"\\nSample Study ({sample_study}) has {len(study_slices)} slices\")\nstudy_slices = train_df[train_df['StudyInstanceUID'] == sample_study]\nprint(f\"\\nSample Study ({sample_study}) has {len(study_slices)} slices\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:06.970472Z","iopub.execute_input":"2025-03-27T17:14:06.970801Z","iopub.status.idle":"2025-03-27T17:14:07.012852Z","shell.execute_reply.started":"2025-03-27T17:14:06.970772Z","shell.execute_reply":"2025-03-27T17:14:07.011302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\nplt.figure(figsize=(10,5))\nplt.subplot(1,2,1)\nsns.countplot(x='patient_overall', data=train_df)\nplt.title('Overall Fracture Distribution')\n\nplt.subplot(1,2,2)\nvertebrae_counts = train_df[['C1','C2','C3','C4','C5','C6','C7']].sum()\nsns.barplot(x=vertebrae_counts.index, y=vertebrae_counts.values)\nplt.title('Vertebrae-wise Fracture Counts')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.013265Z","iopub.status.idle":"2025-03-27T17:14:07.013506Z","shell.execute_reply":"2025-03-27T17:14:07.013408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create study-level dataframe\nstudy_df = train_df.groupby('StudyInstanceUID').agg({\n    'patient_overall': 'max',\n    'C1': 'max',\n    'C2': 'max',\n    'C3': 'max',\n    'C4': 'max',\n    'C5': 'max',\n    'C6': 'max',\n    'C7': 'max'\n}).reset_index()\n\nprint(\"\\nStudy-level Statistics:\")\nprint(f\"Total studies: {len(study_df)}\")\nprint(f\"Fracture prevalence: {study_df['patient_overall'].mean():.2%}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.014192Z","iopub.status.idle":"2025-03-27T17:14:07.014429Z","shell.execute_reply":"2025-03-27T17:14:07.014331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nfrom sklearn.model_selection import train_test_split\n\nclass SpineDataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, study_list, batch_size=8, dim=(512,512), n_channels=1):\n        self.study_list = study_list\n        self.batch_size = batch_size\n        self.dim = dim\n        self.n_channels = n_channels\n        self.on_epoch_end()\n        \n    def __len__(self):\n        return int(np.ceil(len(self.study_list) / self.batch_size))\n    \n    def __getitem__(self, index):\n        batch_studies = self.study_list[index*self.batch_size:(index+1)*self.batch_size]\n        \n        X = np.empty((len(batch_studies), *self.dim, self.n_channels))\n        y_patient = np.empty(len(batch_studies), dtype=np.float32)\n        y_vertebrae = np.empty((len(batch_studies), 7), dtype=np.float32)\n        \n        for i, study_id in enumerate(batch_studies):\n            study_path = f\"/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images/{study_id}\"\n            slices = sorted(os.listdir(study_path), key=lambda x: int(x.split('.')[0]))\n            \n            # Load middle slice\n            mid_idx = len(slices) // 2\n            img = load_dicom_v2(f\"{study_path}/{slices[mid_idx]}\")\n            \n            # Preprocessing\n            img = cv2.resize(img, self.dim)\n            if img.ndim == 2: \n                img = np.expand_dims(img, axis=-1)\n                \n            X[i,] = img\n            study_data = study_df[study_df['StudyInstanceUID'] == study_id].iloc[0]\n            y_patient[i] = study_data['patient_overall']\n            y_vertebrae[i] = study_data[['C1','C2','C3','C4','C5','C6','C7']].values\n            \n        return X, {'patient_output': y_patient, 'vertebrae_output': y_vertebrae}\n    \n    def on_epoch_end(self):\n        self.indices = np.arange(len(self.study_list))\n        np.random.shuffle(self.indices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.015147Z","iopub.status.idle":"2025-03-27T17:14:07.01539Z","shell.execute_reply":"2025-03-27T17:14:07.01529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_multi_task_model(input_shape=(512,512,1)):\n    # Shared backbone\n    inputs = tf.keras.Input(shape=input_shape)\n    x = tf.keras.layers.Conv2D(32, (3,3), activation='relu')(inputs)\n    x = tf.keras.layers.MaxPooling2D((2,2))(x)\n    x = tf.keras.layers.Conv2D(64, (3,3), activation='relu')(x)\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    \n    # Patient-level prediction\n    patient_out = tf.keras.layers.Dense(1, activation='sigmoid', name='patient_output')(x)\n    \n    # Vertebrae-level predictions\n    vertebrae_out = tf.keras.layers.Dense(7, activation='sigmoid', name='vertebrae_output')(x)\n    \n    model = tf.keras.Model(inputs=inputs, outputs=[patient_out, vertebrae_out])\n    \n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(1e-4),\n        loss={\n            'patient_output': 'binary_crossentropy',\n            'vertebrae_output': 'binary_crossentropy'\n        },\n        metrics={\n            'patient_output': ['accuracy', tf.keras.metrics.AUC(name='auc')],\n            'vertebrae_output': tf.keras.metrics.AUC(name='vertebrae_auc', multi_label=True)\n        },\n        loss_weights=[0.7, 0.3]  # Weight patient-level prediction higher\n    )\n    return model\n\nmodel = build_multi_task_model()\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.016317Z","iopub.status.idle":"2025-03-27T17:14:07.016574Z","shell.execute_reply":"2025-03-27T17:14:07.016475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split studies into train/validation\ntrain_studies, val_studies = train_test_split(\n    study_df['StudyInstanceUID'].values,\n    test_size=0.2,\n    stratify=study_df['patient_overall'],\n    random_state=42\n)\n\nprint(f\"Training studies: {len(train_studies)}\")\nprint(f\"Validation studies: {len(val_studies)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.017479Z","iopub.status.idle":"2025-03-27T17:14:07.017855Z","shell.execute_reply":"2025-03-27T17:14:07.017719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Update ModelCheckpoint callback to use .keras extension\ncallbacks = [\n    tf.keras.callbacks.ModelCheckpoint('best_model.keras',  # Changed to .keras format\n                                      save_best_only=True,\n                                      monitor='val_patient_output_auc',\n                                      mode='max'),\n    tf.keras.callbacks.EarlyStopping(patience=5,\n                                   restore_best_weights=True,\n                                   monitor='val_patient_output_auc',\n                                   mode='max')\n]\n\n# Add class weights for imbalance handling (from Phase 5.2)\npatient_weights = {0: 1, 1: study_df['patient_overall'].value_counts()[0]/study_df['patient_overall'].value_counts()[1]}\n\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=30,\n    callbacks=callbacks,\n    class_weight={'patient_output': patient_weights},\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.018505Z","iopub.status.idle":"2025-03-27T17:14:07.018823Z","shell.execute_reply":"2025-03-27T17:14:07.018668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add missing import at the top of your notebook\nimport os\n\n# Revised SpineDataGenerator with error handling\nclass SpineDataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, study_list, batch_size=8, dim=(512,512), n_channels=1):\n        self.study_list = study_list\n        self.batch_size = batch_size\n        self.dim = dim\n        self.n_channels = n_channels\n        self.on_epoch_end()\n        \n    def __len__(self):\n        return int(np.ceil(len(self.study_list) / self.batch_size))\n    \n    def __getitem__(self, index):\n        batch_studies = self.study_list[index*self.batch_size:(index+1)*self.batch_size]\n        \n        X = np.empty((len(batch_studies), *self.dim, self.n_channels))\n        y_patient = np.empty(len(batch_studies), dtype=np.float32)\n        y_vertebrae = np.empty((len(batch_studies), 7), dtype=np.float32)\n        \n        for i, study_id in enumerate(batch_studies):\n            try:\n                study_path = f\"/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images/{study_id}\"\n                if not os.path.exists(study_path):\n                    raise FileNotFoundError(f\"Study directory not found: {study_path}\")\n                    \n                slices = sorted(os.listdir(study_path), key=lambda x: int(x.split('.')[0]))\n                mid_idx = len(slices) // 2\n                \n                img = load_dicom_v2(f\"{study_path}/{slices[mid_idx]}\")\n                img = cv2.resize(img, self.dim)\n                if img.ndim == 2: \n                    img = np.expand_dims(img, axis=-1)\n                    \n                X[i,] = img\n                study_data = study_df[study_df['StudyInstanceUID'] == study_id].iloc[0]\n                y_patient[i] = study_data['patient_overall']\n                y_vertebrae[i] = study_data[['C1','C2','C3','C4','C5','C6','C7']].values\n                \n            except Exception as e:\n                print(f\"Error processing {study_id}: {str(e)}\")\n                # Handle failed samples by returning zeros\n                X[i,] = np.zeros((*self.dim, self.n_channels))\n                y_patient[i] = 0\n                y_vertebrae[i] = np.zeros(7)\n                \n        return X, {'patient_output': y_patient, 'vertebrae_output': y_vertebrae}\n    \n    def on_epoch_end(self):\n        self.indices = np.arange(len(self.study_list))\n        np.random.shuffle(self.indices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.019905Z","iopub.status.idle":"2025-03-27T17:14:07.020287Z","shell.execute_reply":"2025-03-27T17:14:07.020102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Test the generator with first 5 studies\ntest_gen = SpineDataGenerator(study_df['StudyInstanceUID'].values[:5], batch_size=2)\nX_batch, y_batch = test_gen[0]\n\nprint(\"Batch shape:\", X_batch.shape)\nprint(\"Patient labels:\", y_batch['patient_output'])\nprint(\"Vertebrae labels:\", y_batch['vertebrae_output'][0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.02115Z","iopub.status.idle":"2025-03-27T17:14:07.021471Z","shell.execute_reply":"2025-03-27T17:14:07.021356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add these at the top\nimport numpy as np\nimport cv2  # For image resizing","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.022581Z","iopub.status.idle":"2025-03-27T17:14:07.022936Z","shell.execute_reply":"2025-03-27T17:14:07.022782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure study_df is properly created from train_df\nstudy_df = train_df.groupby('StudyInstanceUID').agg({\n    'patient_overall': 'max',\n    'C1': 'max',\n    'C2': 'max',\n    'C3': 'max',\n    'C4': 'max',\n    'C5': 'max',\n    'C6': 'max',\n    'C7': 'max'\n}).reset_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.023824Z","iopub.status.idle":"2025-03-27T17:14:07.024127Z","shell.execute_reply":"2025-03-27T17:14:07.024021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Modify load_dicom_v2 to ensure proper normalization\ndef load_dicom_v2(path):\n    try:\n        dicom = pydicom.dcmread(path)\n        img = dicom.pixel_array.astype(float)\n        img = (img - img.min()) / (img.max() - img.min() + 1e-8)\n        return img.astype(np.float32)\n    except Exception as e:\n        print(f\"Error loading {path}: {str(e)}\")\n        return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.024927Z","iopub.status.idle":"2025-03-27T17:14:07.025163Z","shell.execute_reply":"2025-03-27T17:14:07.025067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport cv2\nimport pydicom\nimport tensorflow as tf\n\n# --- Data Loading Functions ---\ndef load_dicom_v2(path):\n    \"\"\"Improved DICOM loader with error handling\"\"\"\n    try:\n        dicom = pydicom.dcmread(path)\n        img = dicom.pixel_array.astype(float)\n        img = (img - img.min()) / (img.max() - img.min() + 1e-8)\n        return img.astype(np.float32)\n    except Exception as e:\n        print(f\"Error loading {path}: {str(e)}\")\n        return None\n\n# --- Data Generator Class ---\nclass SpineDataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, study_list, batch_size=8, dim=(512,512), n_channels=1):\n        self.study_list = study_list\n        self.batch_size = batch_size\n        self.dim = dim\n        self.n_channels = n_channels\n        self.on_epoch_end()\n        \n    def __len__(self):\n        return int(np.ceil(len(self.study_list) / self.batch_size))\n    \n    def __getitem__(self, index):\n        batch_studies = self.study_list[index*self.batch_size:(index+1)*self.batch_size]\n        \n        X = np.empty((len(batch_studies), *self.dim, self.n_channels))\n        y_patient = np.empty(len(batch_studies), dtype=np.float32)\n        y_vertebrae = np.empty((len(batch_studies), 7), dtype=np.float32)\n        \n        for i, study_id in enumerate(batch_studies):\n            try:\n                study_path = f\"/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images/{study_id}\"\n                \n                if not os.path.exists(study_path):\n                    raise FileNotFoundError(f\"Missing study: {study_id}\")\n                    \n                slices = sorted(os.listdir(study_path), \n                              key=lambda x: int(x.split('.')[0]))\n                mid_idx = len(slices) // 2\n                \n                img = load_dicom_v2(f\"{study_path}/{slices[mid_idx]}\")\n                if img is None:\n                    raise ValueError(\"Empty image\")\n                \n                # Resize and normalize\n                img = cv2.resize(img, self.dim)\n                if img.ndim == 2:\n                    img = np.expand_dims(img, axis=-1)\n                \n                X[i,] = img\n                \n                # Get labels\n                study_data = study_df[study_df['StudyInstanceUID'] == study_id].iloc[0]\n                y_patient[i] = study_data['patient_overall']\n                y_vertebrae[i] = study_data[['C1','C2','C3','C4','C5','C6','C7']].values\n                \n            except Exception as e:\n                print(f\"Error processing {study_id}: {str(e)}\")\n                # Fallback to empty data\n                X[i,] = np.zeros((*self.dim, self.n_channels))\n                y_patient[i] = 0\n                y_vertebrae[i] = np.zeros(7)\n                \n        return X, {'patient_output': y_patient, 'vertebrae_output': y_vertebrae}\n    \n    def on_epoch_end(self):\n        self.indices = np.arange(len(self.study_list))\n        np.random.shuffle(self.indices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.025855Z","iopub.status.idle":"2025-03-27T17:14:07.026126Z","shell.execute_reply":"2025-03-27T17:14:07.026019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Test with known valid study\ntest_study = study_df.iloc[0]['StudyInstanceUID']\ntest_gen = SpineDataGenerator([test_study], batch_size=1)\nX, y = test_gen[0]\n\nprint(\"Image shape:\", X.shape)\nprint(\"Patient label:\", y['patient_output'][0])\nprint(\"Vertebrae labels:\", y['vertebrae_output'][0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.026884Z","iopub.status.idle":"2025-03-27T17:14:07.027162Z","shell.execute_reply":"2025-03-27T17:14:07.027053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add to model architecture\ndata_aug = tf.keras.Sequential([\n    tf.keras.layers.RandomFlip(\"horizontal_and_vertical\"),\n    tf.keras.layers.RandomRotation(0.1),\n    tf.keras.layers.RandomContrast(0.1)\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.028128Z","iopub.status.idle":"2025-03-27T17:14:07.028465Z","shell.execute_reply":"2025-03-27T17:14:07.028312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate weights for imbalance\nclass_counts = study_df['patient_overall'].value_counts()\npatient_weights = {0: 1/class_counts[0], 1: 1/class_counts[1]}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.029166Z","iopub.status.idle":"2025-03-27T17:14:07.029503Z","shell.execute_reply":"2025-03-27T17:14:07.029368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=30,\n    class_weight={'patient_output': patient_weights},\n    callbacks=[\n        tf.keras.callbacks.ModelCheckpoint(\n            'best_model.keras',\n            monitor='val_patient_output_auc',\n            save_best_only=True,\n            mode='max'\n        ),\n        tf.keras.callbacks.EarlyStopping(\n            monitor='val_patient_output_auc',\n            patience=5,\n            mode='max'\n        )\n    ]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.030135Z","iopub.status.idle":"2025-03-27T17:14:07.030455Z","shell.execute_reply":"2025-03-27T17:14:07.030325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install required DICOM handlers\n!pip install pylibjpeg pylibjpeg-libjpeg gdcm --force-reinstall\n\n# Restart kernel after installation (Session → Restart Session)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.031139Z","iopub.status.idle":"2025-03-27T17:14:07.031482Z","shell.execute_reply":"2025-03-27T17:14:07.031319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport cv2\nimport pydicom\nimport tensorflow as tf\n\n# ----- Data Loading Function -----\ndef load_dicom_safe(path):\n    try:\n        dicom = pydicom.dcmread(path)\n        img = dicom.pixel_array.astype(float)\n        img = (img - img.min()) / (img.max() - img.min() + 1e-8)\n        return img.astype(np.float32)\n    except Exception as e:\n        print(f\"Error loading {path}: {str(e)}\")\n        return None\n\n# ----- Fixed Data Generator -----\nclass SpineDataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, study_list, batch_size=8, dim=(512,512)):\n        super().__init__()  # Important fix!\n        self.study_list = study_list\n        self.batch_size = batch_size\n        self.dim = dim\n        self.on_epoch_end()\n        \n    def __len__(self):\n        return int(np.ceil(len(self.study_list) / self.batch_size))\n    \n    def __getitem__(self, index):\n        batch_studies = self.study_list[index*self.batch_size:(index+1)*self.batch_size]\n        \n        X = []\n        y_patient = []\n        y_vertebrae = []\n        \n        for study_id in batch_studies:\n            try:\n                # 1. Check study folder exists\n                study_path = f\"/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images/{study_id}\"\n                if not os.path.exists(study_path):\n                    continue\n                    \n                # 2. Get middle slice\n                slices = sorted(os.listdir(study_path), key=lambda x: int(x.split('.')[0]))\n                mid_idx = len(slices) // 2\n                img_path = f\"{study_path}/{slices[mid_idx]}\"\n                \n                # 3. Load and validate image\n                img = load_dicom_safe(img_path)\n                if img is None or img.size == 0:\n                    continue\n                \n                # 4. Resize and format\n                img = cv2.resize(img, self.dim)\n                if img.ndim == 2:\n                    img = np.expand_dims(img, axis=-1)\n                    \n                X.append(img)\n                \n                # 5. Get labels\n                study_data = train_df[train_df['StudyInstanceUID'] == study_id].iloc[0]\n                y_patient.append(study_data['patient_overall'])\n                y_vertebrae.append(study_data[['C1','C2','C3','C4','C5','C6','C7']].values)\n                \n            except Exception as e:\n                print(f\"Skipping {study_id}: {str(e)}\")\n                continue\n                \n        # Handle empty batches\n        if len(X) == 0:\n            return np.zeros((self.batch_size, *self.dim, 1)), \\\n                   {'patient_output': np.zeros(self.batch_size), \n                    'vertebrae_output': np.zeros((self.batch_size, 7))}\n        \n        return np.array(X), {\n            'patient_output': np.array(y_patient),\n            'vertebrae_output': np.array(y_vertebrae)\n        }\n    \n    def on_epoch_end(self):\n        self.indices = np.arange(len(self.study_list))\n        np.random.shuffle(self.indices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.032459Z","iopub.status.idle":"2025-03-27T17:14:07.032756Z","shell.execute_reply":"2025-03-27T17:14:07.032621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the training data\ntrain_df = pd.read_csv('/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv')\n\n# Create test generator\ntest_gen = SpineDataGenerator(\n    study_list=train_df['StudyInstanceUID'].values[:5],  # First 5 studies\n    batch_size=2,\n    dim=(512,512)\n)\n\n# Get one batch\nX_batch, y_batch = test_gen[0]\nprint(\"Batch shape:\", X_batch.shape)\nprint(\"Patient labels:\", y_batch['patient_output'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.033616Z","iopub.status.idle":"2025-03-27T17:14:07.0339Z","shell.execute_reply":"2025-03-27T17:14:07.033792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run this FIRST in a new cell\n!pip uninstall -y pydicom pylibjpeg pylibjpeg-libjpeg gdcm\n!pip install --user pydicom==2.4.3 pylibjpeg pylibjpeg-libjpeg gdcm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.034849Z","iopub.status.idle":"2025-03-27T17:14:07.035234Z","shell.execute_reply":"2025-03-27T17:14:07.035072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef load_dicom_final(path):\n    try:\n        # Force GDCM decompression\n        dicom = pydicom.dcmread(path, force=True)\n        dicom.file_meta.TransferSyntaxUID = pydicom.uid.ImplicitVRLittleEndian\n        \n        if 'PixelData' not in dicom:\n            return None\n            \n        img = apply_voi_lut(dicom.pixel_array, dicom)\n        img = (img - img.min()) / (img.max() - img.min() + 1e-8)\n        return img.astype(np.float32)\n    except Exception as e:\n        print(f\"Skipped {path.split('/')[-2]}: {str(e)}\")\n        return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.03592Z","iopub.status.idle":"2025-03-27T17:14:07.036286Z","shell.execute_reply":"2025-03-27T17:14:07.036125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class FinalSpineGenerator(tf.keras.utils.Sequence):\n    def __init__(self, study_list, batch_size=8, dim=(512,512)):\n        super().__init__()\n        self.study_list = study_list\n        self.batch_size = batch_size\n        self.dim = dim\n        self.on_epoch_end()\n        \n    def __len__(self):\n        return int(np.ceil(len(self.study_list) / self.batch_size))\n    \n    def __getitem__(self, index):\n        batch_studies = self.study_list[index*self.batch_size:(index+1)*self.batch_size]\n        \n        X = []\n        y_patient = []\n        y_vertebrae = []\n        \n        for study_id in batch_studies:\n            try:\n                study_path = f\"/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images/{study_id}\"\n                if not os.path.exists(study_path):\n                    continue\n                    \n                # Load ALL slices and pick clearest middle one\n                slices = sorted([f for f in os.listdir(study_path) if f.endswith('.dcm')], \n                              key=lambda x: int(x.split('.')[0]))\n                if not slices:\n                    continue\n                    \n                # Try multiple slices for valid image\n                for slice_idx in [len(slices)//2, 0, -1]:  # Middle, first, last\n                    img = load_dicom_final(f\"{study_path}/{slices[slice_idx]}\")\n                    if img is not None:\n                        break\n                \n                if img is None:\n                    continue\n                \n                # Resize and format\n                img = cv2.resize(img, self.dim)\n                img = np.expand_dims(img, axis=-1)  # Add channel dimension\n                \n                X.append(img)\n                study_data = train_df[train_df['StudyInstanceUID'] == study_id].iloc[0]\n                y_patient.append(study_data['patient_overall'])\n                y_vertebrae.append(study_data[['C1','C2','C3','C4','C5','C6','C7']].values)\n                \n            except Exception as e:\n                continue\n                \n        # Fallback for empty batches\n        if len(X) == 0:\n            return np.zeros((self.batch_size, *self.dim, 1)), \\\n                   {'patient_output': np.zeros(self.batch_size), \n                    'vertebrae_output': np.zeros((self.batch_size, 7))}\n        \n        return np.array(X), {\n            'patient_output': np.array(y_patient),\n            'vertebrae_output': np.array(y_vertebrae)\n        }\n    \n    def on_epoch_end(self):\n        np.random.shuffle(self.study_list)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.037033Z","iopub.status.idle":"2025-03-27T17:14:07.037305Z","shell.execute_reply":"2025-03-27T17:14:07.037195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_gen = FinalSpineGenerator(\n    study_list=['1.2.826.0.1.3680043.9443'],  # Previously failing study\n    batch_size=1\n)\n\nX, y = test_gen[0]\nprint(\"Final test - Batch shape:\", X.shape)\nprint(\"Sample values:\", X[0, 250:255, 250:255, 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.03815Z","iopub.status.idle":"2025-03-27T17:14:07.038503Z","shell.execute_reply":"2025-03-27T17:14:07.038324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Split data\ntrain_studies, val_studies = train_test_split(\n    train_df['StudyInstanceUID'].unique(),\n    test_size=0.2,\n    stratify=train_df.groupby('StudyInstanceUID')['patient_overall'].first(),\n    random_state=42\n)\n\n# 2. Create generators\ntrain_gen = FinalSpineGenerator(train_studies, batch_size=16)\nval_gen = FinalSpineGenerator(val_studies, batch_size=16)\n\n# 3. Train model\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=30,\n    callbacks=[\n        tf.keras.callbacks.ModelCheckpoint(\n            'best_model.keras',\n            monitor='val_patient_output_auc',\n            save_best_only=True,\n            mode='max'\n        )\n    ],\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T17:14:07.039617Z","iopub.status.idle":"2025-03-27T17:14:07.039944Z","shell.execute_reply":"2025-03-27T17:14:07.039812Z"}},"outputs":[],"execution_count":null}]}