{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Setting up :","metadata":{}},{"cell_type":"markdown","source":"## Importing librairies :","metadata":{}},{"cell_type":"code","source":"# Importing librairies unable to download directly as the internet is disabled :\n\nimport pkgutil\ncheck_module = True if pkgutil.find_loader(\"hydra\") else False\nif not check_module:\n    import subprocess    \n    subprocess.run('python -m pip install --no-index --find-links=/kaggle/input/notebook-to-download-packages python-gdcm'.split())\n    subprocess.run('python -m pip install --no-index --find-links=/kaggle/input/notebook-to-download-packages pylibjpeg'.split())\n    subprocess.run('python -m pip install --no-index --find-links=/kaggle/input/notebook-to-download-packages pylibjpeg-libjpeg'.split())\n    subprocess.run('python -m pip install --no-index --find-links=/kaggle/input/notebook-to-download-packages pydicom'.split())    \n    subprocess.run('python -m pip install --no-index --find-links=/kaggle/input/notebook-to-download-packages livelossplot'.split())    \nelse:\n    print(\"Environment is already setup\")\n\ndel check_module","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:05:14.310057Z","iopub.execute_input":"2022-10-23T17:05:14.310954Z","iopub.status.idle":"2022-10-23T17:06:00.685718Z","shell.execute_reply.started":"2022-10-23T17:05:14.310851Z","shell.execute_reply":"2022-10-23T17:06:00.68377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport glob\nimport gc\n\nimport numpy as np\nimport pandas as pd\nfrom scipy import ndimage\n\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\n\nimport pydicom as dicom\nimport gdcm\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom livelossplot import PlotLossesKeras","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:06:00.68845Z","iopub.execute_input":"2022-10-23T17:06:00.68886Z","iopub.status.idle":"2022-10-23T17:06:05.602683Z","shell.execute_reply.started":"2022-10-23T17:06:00.688818Z","shell.execute_reply":"2022-10-23T17:06:05.601703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Defining useful features and pathes :","metadata":{}},{"cell_type":"code","source":"## Defining each useful path :\n\n# Where the original images are :\nTRAIN_IMAGES_PATH = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images/'\nTEST_IMAGES_PATH = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/test_images/'\n\n# Pathes of the dataframes :\nTRAIN_CSV_PATH = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv'\nTEST_CSV_PATH = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/test.csv'\n\n# Where we are going to save the preprocessed data :\nTEST_OUTPUT_PATH = './test_arrays/'\nif not os.path.exists(TEST_OUTPUT_PATH): os.mkdir(TEST_OUTPUT_PATH)  \n    \n# Where the trained model is :\nMODEL_PATH = \"/kaggle/input/3d-conv-training-phase\"","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:06:05.604043Z","iopub.execute_input":"2022-10-23T17:06:05.604723Z","iopub.status.idle":"2022-10-23T17:06:05.612098Z","shell.execute_reply.started":"2022-10-23T17:06:05.604684Z","shell.execute_reply":"2022-10-23T17:06:05.610971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Defining useful features :\n\n# Defining desired width and depth of the output arrays :\n\ndesired_width = 64\ndesired_height = 64\ndesired_depth = 64\n\n# Putting images paths into lists : \n\ntrain_images = os.listdir(TRAIN_IMAGES_PATH)\ntest_images = os.listdir(TEST_IMAGES_PATH)\n\n# Reading dataframes :\n\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\ntest_df = pd.read_csv(TEST_CSV_PATH)\n","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:06:05.614928Z","iopub.execute_input":"2022-10-23T17:06:05.615472Z","iopub.status.idle":"2022-10-23T17:06:05.724311Z","shell.execute_reply.started":"2022-10-23T17:06:05.615418Z","shell.execute_reply":"2022-10-23T17:06:05.723322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading and transforming test images into 3D volumes :","metadata":{}},{"cell_type":"code","source":"### Creating a function to load dicom files, even if compressed :\n\ndef load_dicom(path: str):\n    \"\"\"Load a dicom file (.dcm) even if it is compressed.\"\"\"\n    try:\n        file_as_array = dicom.dcmread(path).pixel_array\n    except:\n        decompressed_file = gdcm.ImageReader().SetFileName(path).Read()\n        file_as_array = decompressed_file.pixel_array\n    return(file_as_array)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:06:05.72593Z","iopub.execute_input":"2022-10-23T17:06:05.726292Z","iopub.status.idle":"2022-10-23T17:06:05.733504Z","shell.execute_reply.started":"2022-10-23T17:06:05.726255Z","shell.execute_reply":"2022-10-23T17:06:05.731584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating the preprocessing function including loading dicom normalization and resize :\n\ndef preprocessing_slice(slice_path: str):\n    \"\"\"Load dicom of a slice, normalize and resize it\"\"\"\n    # Loading dicom :\n    slice_array = load_dicom(slice_path)\n    slice_array = slice_array.astype(np.uint8)\n    \n    # Normalization :\n    slice_array = slice_array - np.min(slice_array)\n    if np.max(slice_array) != 0:\n        slice_array = slice_array / np.max(slice_array)\n    slice_array = (slice_array * 255).astype(np.uint8)\n        \n    # Resize (2D) :  \n    width_factor = desired_width / slice_array.shape[0]\n    height_factor = desired_height / slice_array.shape[1]\n    \n    slice_array = ndimage.zoom(slice_array, (width_factor, height_factor), order=3) # resize with spline interpolation of order 3\n    \n    return(slice_array)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:06:05.735675Z","iopub.execute_input":"2022-10-23T17:06:05.736418Z","iopub.status.idle":"2022-10-23T17:06:05.744885Z","shell.execute_reply.started":"2022-10-23T17:06:05.73638Z","shell.execute_reply":"2022-10-23T17:06:05.743478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize_depth(numpy_volume: np.array, desired_depth=desired_depth):\n    \"\"\"Resize across z-axis\"\"\"\n    ## Get current depth\n    current_depth = numpy_volume.shape[0]\n    ## Compute depth factor\n    depth_factor = desired_depth / current_depth\n    \n    ## Resize across z-axis\n    # Rotate\n    numpy_volume = ndimage.rotate(numpy_volume, 90, reshape=False)\n    # Resize\n    volume = ndimage.zoom(numpy_volume, (depth_factor, 1, 1), order=1) # resize with spline interpolation of order 1\n    return volume","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:06:05.746551Z","iopub.execute_input":"2022-10-23T17:06:05.747025Z","iopub.status.idle":"2022-10-23T17:06:05.755265Z","shell.execute_reply.started":"2022-10-23T17:06:05.746991Z","shell.execute_reply":"2022-10-23T17:06:05.754279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a function to parallelize most of the charge of loading the dicom files :\n\ndef load_and_stack_dicom_parallel(scan_path: str):\n    \"\"\"Load all dicom files from a scan and stack them all in a numpy array\"\"\"\n    # Defining slice paths :\n    slice_paths = sorted(glob.glob(os.path.join(scan_path, \"*\")),\n                         key=lambda x: int(x.split('/')[-1].split(\".\")[0]))\n    \n    # Preprocessing slices :\n    images = Parallel(n_jobs=-1)(delayed(preprocessing_slice)(filename) for filename in slice_paths)\n    \n    # Returning stacked slices as a resized on depth (3rd dimension) volume :\n    return(tf.expand_dims(resize_depth(np.array(images)), axis=3))","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:06:05.756988Z","iopub.execute_input":"2022-10-23T17:06:05.75771Z","iopub.status.idle":"2022-10-23T17:06:05.768406Z","shell.execute_reply.started":"2022-10-23T17:06:05.757675Z","shell.execute_reply":"2022-10-23T17:06:05.767493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to create and save the 3D volumes corresponding to a scan :\n\ndef save_3D_arrays(scan_path: str, output_path: str):\n    \"\"\"Create and save the 3D arrays corresponding to a scan\"\"\"\n    \n    # Preprocessing and creation :\n    volume = load_and_stack_dicom_parallel(scan_path=scan_path)\n    \n    # Saving the numpy array :\n    volume_file_name = output_path + scan_path.split('/')[-1] + '.npy'\n    np.save(volume_file_name, volume)\n    \n    # Deleting in memory :\n    del volume\n    \n    return None","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:06:05.77055Z","iopub.execute_input":"2022-10-23T17:06:05.770966Z","iopub.status.idle":"2022-10-23T17:06:05.778757Z","shell.execute_reply.started":"2022-10-23T17:06:05.770933Z","shell.execute_reply":"2022-10-23T17:06:05.777812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creation of the preprocessed array volumes :\n\nfor i in tqdm(range(len(test_images))):\n    case_path = TEST_IMAGES_PATH + test_images[i]\n    save_3D_arrays(case_path, TEST_OUTPUT_PATH)\n\n\n# Free up memory :\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:06:05.782638Z","iopub.execute_input":"2022-10-23T17:06:05.782965Z","iopub.status.idle":"2022-10-23T17:06:35.157738Z","shell.execute_reply.started":"2022-10-23T17:06:05.78294Z","shell.execute_reply":"2022-10-23T17:06:35.156696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inferences : Loading the trained model and making predictions :","metadata":{}},{"cell_type":"code","source":"# Loading the trained model :\n\nmodel = keras.models.load_model(os.path.join(MODEL_PATH,\"InceptionV3-b-64x64x64\"))","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:07:03.819533Z","iopub.execute_input":"2022-10-23T17:07:03.819903Z","iopub.status.idle":"2022-10-23T17:07:20.922644Z","shell.execute_reply.started":"2022-10-23T17:07:03.819873Z","shell.execute_reply":"2022-10-23T17:07:20.921494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Testing its predictions using a bacth sample :\n\n#sample_indexes = np.random.randint(0, len(val_IDs)-1, 40) # getting random indexes to take samples from the validation set\nbatch_x = np.empty((len(test_images), 64, 64, 64, 1))\nfor i_slice in range(0, len(test_images)):\n    batch_x[i_slice,] = np.load(os.path.join(TEST_OUTPUT_PATH, os.listdir(TEST_OUTPUT_PATH)[i_slice]))\n\npredictions = model.predict(batch_x)\nprint(predictions)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:07:20.927217Z","iopub.execute_input":"2022-10-23T17:07:20.927548Z","iopub.status.idle":"2022-10-23T17:07:29.169329Z","shell.execute_reply.started":"2022-10-23T17:07:20.927518Z","shell.execute_reply":"2022-10-23T17:07:29.168285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating submission :","metadata":{}},{"cell_type":"code","source":"# Putting predictions outputs in the right order for the submission file ('patient_overall' at the end of each sequence) :\n\nsubmission = predictions.copy()\n\nfor sequence_index in range(submission.shape[0]):\n    submission[sequence_index] = np.concatenate((submission[sequence_index][1:], [submission[sequence_index][0]]))\n    \nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:07:29.170999Z","iopub.execute_input":"2022-10-23T17:07:29.171344Z","iopub.status.idle":"2022-10-23T17:07:29.180356Z","shell.execute_reply.started":"2022-10-23T17:07:29.171316Z","shell.execute_reply":"2022-10-23T17:07:29.179063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Creating dataframe :\n\n# Putting every predictions output as a single vector (corresponding to the column 'fractured' in the dataframe) :\n\nfractured_col_values = np.concatenate(submission.copy())\n    \n# Creating the 'row_id' column values :\n\nrow_id_values = []\nfor ID in test_images:\n    row_id_values.append(ID+'_C1')\n    row_id_values.append(ID+'_C2')\n    row_id_values.append(ID+'_C3')\n    row_id_values.append(ID+'_C4')\n    row_id_values.append(ID+'_C5')\n    row_id_values.append(ID+'_C6')\n    row_id_values.append(ID+'_C7')\n    row_id_values.append(ID+'_patient_overall')\n    \n# Creating the dataframe :\n\nsubmissions_df = pd.DataFrame({\"row_id\": row_id_values, \"fractured\": fractured_col_values})\n\nsubmissions_df","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:07:29.183463Z","iopub.execute_input":"2022-10-23T17:07:29.184241Z","iopub.status.idle":"2022-10-23T17:07:29.205312Z","shell.execute_reply.started":"2022-10-23T17:07:29.184177Z","shell.execute_reply":"2022-10-23T17:07:29.204469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Uploading submission csv file :\n\nsubmissions_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T17:08:06.598139Z","iopub.execute_input":"2022-10-23T17:08:06.598559Z","iopub.status.idle":"2022-10-23T17:08:06.605627Z","shell.execute_reply.started":"2022-10-23T17:08:06.598525Z","shell.execute_reply":"2022-10-23T17:08:06.604211Z"},"trusted":true},"execution_count":null,"outputs":[]}]}