{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":36363,"databundleVersionId":4050810,"sourceType":"competition"},{"sourceId":5718655,"sourceType":"datasetVersion","datasetId":2373279}],"dockerImageVersionId":30213,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Some required installation for `PYDICOM` data**. As stated in the [data section](https://www.kaggle.com/competitions/rsna-2022-cervical-spine-fracture-detection/data):\n\n> The **DICOM** image files are ≤ 1 mm slice thickness, axial orientation, and bone kernel. Note that some of the DICOM files are **JPEG compressed**. **You may require additional resources to read the pixel array of these files, such as GDCM and pylibjpeg.**","metadata":{}},{"cell_type":"code","source":"import os\nimport pydicom\nimport SimpleITK as sitk\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom sklearn.metrics import classification_report, accuracy_score\nfrom multiprocessing import Pool","metadata":{"execution":{"iopub.status.busy":"2024-07-18T15:54:25.531864Z","iopub.execute_input":"2024-07-18T15:54:25.532349Z","iopub.status.idle":"2024-07-18T15:54:25.539808Z","shell.execute_reply.started":"2024-07-18T15:54:25.532307Z","shell.execute_reply":"2024-07-18T15:54:25.538536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dicom_image(file_path):\n    dicom = sitk.ReadImage(file_path)\n    img = sitk.GetArrayFromImage(dicom)[0]\n    \n    # Check if the image is a 2D array\n    if img.ndim != 2:\n        raise ValueError(f\"Expected 2D image, got {img.ndim}D image\")\n\n    # Convert to uint8 if necessary\n    if img.dtype != np.uint8:\n        img = cv2.normalize(img, None, 0, 255, cv2.NORM_MINMAX).astype(np.uint8)\n\n    # Resize the image to 128x128\n    img = cv2.resize(img, (128, 128))\n    img = img / 255.0  # Normalize pixel values to [0, 1]\n    return img","metadata":{"execution":{"iopub.status.busy":"2024-07-18T15:54:34.589418Z","iopub.execute_input":"2024-07-18T15:54:34.589915Z","iopub.status.idle":"2024-07-18T15:54:34.599023Z","shell.execute_reply.started":"2024-07-18T15:54:34.589872Z","shell.execute_reply":"2024-07-18T15:54:34.597735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_images_from_subfolder(subfolder, limit_per_subfolder):\n    file_paths = []\n    count = 0\n    with os.scandir(subfolder) as it:\n        for entry in it:\n            if entry.is_file() and entry.name.endswith('.dcm'):\n                file_paths.append(entry.path)\n                count += 1\n                if count >= limit_per_subfolder:\n                    break\n    return file_paths\n\ndef extract_images_into_dataframe(directory, limit_per_subfolder=15):\n    all_files = []\n    subfolders = [f.path for f in os.scandir(directory) if f.is_dir()]\n    \n    with Pool(processes=os.cpu_count()) as pool:\n        results = pool.starmap(extract_images_from_subfolder, [(subfolder, limit_per_subfolder) for subfolder in subfolders])\n    \n    for result in results:\n        all_files.extend(result)\n    \n    df = pd.DataFrame({'FilePath': all_files})\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-07-18T15:54:38.951143Z","iopub.execute_input":"2024-07-18T15:54:38.952321Z","iopub.status.idle":"2024-07-18T15:54:38.961885Z","shell.execute_reply.started":"2024-07-18T15:54:38.952269Z","shell.execute_reply":"2024-07-18T15:54:38.960672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define paths and parameters\nmain_folder_path = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images'\nlimit_per_subfolder = 15\n\n# Extract 15 DICOM file paths per subfolder into a DataFrame\ndf = extract_images_into_dataframe(main_folder_path, limit_per_subfolder)\n\n# Display the dataframe (optional)\nprint(df.shape)\nprint(df.head())","metadata":{"execution":{"iopub.status.busy":"2024-07-18T15:54:45.471479Z","iopub.execute_input":"2024-07-18T15:54:45.471929Z","iopub.status.idle":"2024-07-18T15:56:17.580593Z","shell.execute_reply.started":"2024-07-18T15:54:45.471889Z","shell.execute_reply":"2024-07-18T15:56:17.579148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load images and create arrays\nimages = []\nfor file_path in df['FilePath']:\n    try:\n        img = load_dicom_image(file_path)\n        images.append(img)\n    except Exception as e:\n        print(f\"Error processing file {file_path}: {e}\")\n\nimages = np.array(images)\nprint(images.shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-18T15:56:50.725745Z","iopub.execute_input":"2024-07-18T15:56:50.726262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract StudyInstanceUID from file paths\ndf['StudyInstanceUID'] = df['FilePath'].apply(lambda x: os.path.basename(os.path.dirname(x)))\n\n# Merge with train.csv to get labels\ntrain_df = pd.read_csv('/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv')\ndf = df.merge(train_df[['StudyInstanceUID', 'patient_overall']], on='StudyInstanceUID', how='left')\n\n# Extract labels\nlabels = df['patient_overall'].values\nprint(labels.shape)\n\n# Split Data into Training and Validation Sets\nX_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.2, random_state=42)\n\nprint(X_train.shape, y_train.shape)\nprint(X_val.shape, y_val.shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-18T15:53:56.870797Z","iopub.status.idle":"2024-07-18T15:53:56.87124Z","shell.execute_reply.started":"2024-07-18T15:53:56.871031Z","shell.execute_reply":"2024-07-18T15:53:56.871051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define and Train the Model\ndef create_model():\n    inputs = Input(shape=(128, 128, 1))\n    x = Conv2D(32, (3, 3), activation='relu')(inputs)\n    x = MaxPooling2D((2, 2))(x)\n    x = Conv2D(64, (3, 3), activation='relu')(x)\n    x = MaxPooling2D((2, 2))(x)\n    x = Flatten()(x)\n    x = Dense(128, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    outputs = Dense(1, activation='sigmoid')(x)\n    model = Model(inputs, outputs)\n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n    return model\n\nmodel = create_model()\nmodel.summary()\n\n# Train the Model\nhistory = model.fit(X_train, y_train, epochs=10, batch_size=32, validation_data=(X_val, y_val))","metadata":{"execution":{"iopub.status.busy":"2024-07-18T15:53:56.873266Z","iopub.status.idle":"2024-07-18T15:53:56.87379Z","shell.execute_reply.started":"2024-07-18T15:53:56.873545Z","shell.execute_reply":"2024-07-18T15:53:56.873569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the Model\ny_val_pred = model.predict(X_val)\ny_val_pred = (y_val_pred > 0.5).astype(int)\nprint(classification_report(y_val, y_val_pred))\nprint('Accuracy:', accuracy_score(y_val, y_val_pred))\n\n# Visualize Training History\nplt.plot(history.history['accuracy'], label='train_accuracy')\nplt.plot(history.history['val_accuracy'], label='val_accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()\n\nplt.plot(history.history['loss'], label='train_loss')\nplt.plot(history.history['val_loss'], label='val_loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-18T15:53:56.876021Z","iopub.status.idle":"2024-07-18T15:53:56.876525Z","shell.execute_reply.started":"2024-07-18T15:53:56.87627Z","shell.execute_reply":"2024-07-18T15:53:56.876293Z"},"trusted":true},"execution_count":null,"outputs":[]}]}