{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Quick EDA and Xception fine-tuning on Keras 🍒","metadata":{}},{"cell_type":"markdown","source":"In this notebook you can find code to do the following:\n    \n-  Install the required libraries to display DICOM images\n-  Use helper functions to display DICOM images\n-  Fine-tune a keras model on the training data\n-  Use the resulting model to predict on new images\n-  Submit your results to kaggle\n    \n    \n    \n**Do not forget to vote up!**","metadata":{}},{"cell_type":"markdown","source":"## Start-Up","metadata":{}},{"cell_type":"markdown","source":"### Import libraries","metadata":{}},{"cell_type":"code","source":"# Install to visualize dicom images\n!pip install -qU python-gdcm pydicom pylibjpeg","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:45:51.943594Z","iopub.execute_input":"2023-01-13T15:45:51.943971Z","iopub.status.idle":"2023-01-13T15:46:03.894509Z","shell.execute_reply.started":"2023-01-13T15:45:51.943896Z","shell.execute_reply":"2023-01-13T15:46:03.89381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport glob\nimport cv2\nimport seaborn as sns\n\n# To work with DICOM images\nimport gdcm\nimport pydicom\n\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-13T15:46:18.722011Z","iopub.execute_input":"2023-01-13T15:46:18.722354Z","iopub.status.idle":"2023-01-13T15:46:19.535892Z","shell.execute_reply.started":"2023-01-13T15:46:18.722318Z","shell.execute_reply":"2023-01-13T15:46:19.53496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Helper functions","metadata":{}},{"cell_type":"code","source":"def rescale_img_to_hu(dcm_ds):\n    \"\"\"\n    Rescales the image to Hounsfield unit.\n    Thank you https://www.kaggle.com/code/allunia/rsna-csf-cervical-spine-fracture-eda/notebook\n    \"\"\"\n    data = dcm_ds.pixel_array\n    if dcm_ds.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    return data * dcm_ds.RescaleSlope + dcm_ds.RescaleIntercept\n\ndef show_images_for_patient(patient_id):\n    \"\"\"\n    Thank you\n    https://www.kaggle.com/code/radek1/eda-training-a-fast-ai-model-submission\n    although with some changes.\n    \"\"\"\n    patient_dir = os.path.join('../input/rsna-breast-cancer-detection/train_images', str(patient_id))\n    print(patient_dir)\n    num_images = len([name for name in os.listdir(patient_dir)])\n    print(f\"Number of images for patient: {num_images}\")\n    fig, axs = plt.subplots(2, 2, figsize=(24,15))\n    axs = axs.flatten()\n    for i, img_file in enumerate(os.listdir(patient_dir)):\n        img_path = os.path.join(patient_dir, img_file)\n        ds = pydicom.dcmread(img_path)\n        axs[i].imshow(rescale_img_to_hu(ds), cmap=\"bone\")\n        # Break if there are more than 4 images for visualization purposes\n        if i==3: break\n            ","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:46:21.280288Z","iopub.execute_input":"2023-01-13T15:46:21.280618Z","iopub.status.idle":"2023-01-13T15:46:21.289791Z","shell.execute_reply.started":"2023-01-13T15:46:21.280587Z","shell.execute_reply":"2023-01-13T15:46:21.288812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  EDA","metadata":{}},{"cell_type":"markdown","source":"### Inspect training file","metadata":{}},{"cell_type":"code","source":"train_pd = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_pd.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:46:22.274685Z","iopub.execute_input":"2023-01-13T15:46:22.275035Z","iopub.status.idle":"2023-01-13T15:46:22.395136Z","shell.execute_reply.started":"2023-01-13T15:46:22.275011Z","shell.execute_reply":"2023-01-13T15:46:22.394279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Do some preprocessing","metadata":{}},{"cell_type":"code","source":"# Add path column for further data generator\ntrain_pd['path'] = '/kaggle/input/rsna-breast-cancer-256-pngs/' + train_pd['patient_id'].astype(str) + '_' + train_pd['image_id'].astype(str) + '.png'\n\n# Convert target to string\ntrain_pd['cancer'] = train_pd['cancer'].astype(str)\n\n# Convert target to onehot\nfrom sklearn.preprocessing import OneHotEncoder\nohe = OneHotEncoder()\ntrain_pd['cancer_one_hot'] = ohe.fit_transform(train_pd[['cancer']]).toarray().tolist()","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:46:27.20032Z","iopub.execute_input":"2023-01-13T15:46:27.201121Z","iopub.status.idle":"2023-01-13T15:46:27.493031Z","shell.execute_reply.started":"2023-01-13T15:46:27.201092Z","shell.execute_reply":"2023-01-13T15:46:27.492158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Quick'n'dirty EDA","metadata":{}},{"cell_type":"code","source":"# Imbalanced target\ntrain_pd['cancer'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:46:29.349907Z","iopub.execute_input":"2023-01-13T15:46:29.350222Z","iopub.status.idle":"2023-01-13T15:46:29.364536Z","shell.execute_reply.started":"2023-01-13T15:46:29.350197Z","shell.execute_reply":"2023-01-13T15:46:29.363587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Relationship cancer with implant\n# train_pd['implant'].value_counts()\ntrain_pd.groupby('implant')['cancer'].mean()","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:46:29.557875Z","iopub.execute_input":"2023-01-13T15:46:29.55818Z","iopub.status.idle":"2023-01-13T15:46:29.61133Z","shell.execute_reply.started":"2023-01-13T15:46:29.558157Z","shell.execute_reply":"2023-01-13T15:46:29.610359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Target wrt difficult case --> More than 10% were difficult to diagnose negative\ntrain_pd[['cancer', 'difficult_negative_case']].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:46:30.08753Z","iopub.execute_input":"2023-01-13T15:46:30.087876Z","iopub.status.idle":"2023-01-13T15:46:30.114096Z","shell.execute_reply.started":"2023-01-13T15:46:30.087852Z","shell.execute_reply":"2023-01-13T15:46:30.113233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show me one image with cancer\npatient_id_with_cancer = train_pd[train_pd.cancer=='1']['patient_id'].iloc[0]\nshow_images_for_patient(10011)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:46:35.075378Z","iopub.execute_input":"2023-01-13T15:46:35.075729Z","iopub.status.idle":"2023-01-13T15:46:38.683725Z","shell.execute_reply.started":"2023-01-13T15:46:35.075699Z","shell.execute_reply":"2023-01-13T15:46:38.682746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Should we care about implants?\npatient_id_with_implants = train_pd[train_pd.implant==1]['patient_id'].iloc[0]\nshow_images_for_patient(patient_id_with_implants)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:46:42.447653Z","iopub.execute_input":"2023-01-13T15:46:42.448008Z","iopub.status.idle":"2023-01-13T15:46:46.880608Z","shell.execute_reply.started":"2023-01-13T15:46:42.447981Z","shell.execute_reply":"2023-01-13T15:46:46.879562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"markdown","source":"Since files are too big I used the dataset rsna-breast-cancer-256-pngs. You can import it by going to \"+Add Data\" and selecting the dataset.","metadata":{}},{"cell_type":"markdown","source":"### Load libraries","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:46:46.882484Z","iopub.execute_input":"2023-01-13T15:46:46.882835Z","iopub.status.idle":"2023-01-13T15:46:50.663726Z","shell.execute_reply.started":"2023-01-13T15:46:46.882797Z","shell.execute_reply":"2023-01-13T15:46:50.663022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Download base model","metadata":{}},{"cell_type":"code","source":"base_model = keras.applications.Xception(\n    weights='imagenet',  # Load weights pre-trained on ImageNet.\n    input_shape=(256, 256, 3),\n    include_top=False\n)\n# Freeze the layers of the base model\nbase_model.trainable = False","metadata":{"execution":{"iopub.status.busy":"2023-01-13T14:34:35.958457Z","iopub.execute_input":"2023-01-13T14:34:35.959915Z","iopub.status.idle":"2023-01-13T14:34:42.487917Z","shell.execute_reply.started":"2023-01-13T14:34:35.959873Z","shell.execute_reply":"2023-01-13T14:34:42.486925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Wrap base model with final model\n\n# Input layer\ninputs = keras.Input(shape=(256, 256, 3))\n\n# Base model layer\nx = base_model(inputs, training=False)\n\n# Pooling to reduce the number of dimensions\nx = keras.layers.GlobalAveragePooling2D()(x)\n\n# Dense layer to learn new stuff\nx = keras.layers.Dense(16, activation='relu')(x)\n\n# Output layer for loss\noutputs = keras.layers.Dense(2, activation='softmax')(x)\n\n# All togeteher now... all togeeeeether\nmodel = keras.Model(inputs, outputs)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T14:35:18.251312Z","iopub.execute_input":"2023-01-13T14:35:18.251667Z","iopub.status.idle":"2023-01-13T14:35:18.592307Z","shell.execute_reply.started":"2023-01-13T14:35:18.251638Z","shell.execute_reply":"2023-01-13T14:35:18.591288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print a summary of the model we just created\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-01-13T14:35:19.254129Z","iopub.execute_input":"2023-01-13T14:35:19.255093Z","iopub.status.idle":"2023-01-13T14:35:19.270509Z","shell.execute_reply.started":"2023-01-13T14:35:19.255059Z","shell.execute_reply":"2023-01-13T14:35:19.269372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define loss and metrics to display\nmodel.compile(\n    optimizer=keras.optimizers.Adam(),\n    loss='categorical_crossentropy',\n    metrics=[keras.metrics.BinaryAccuracy()]\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T14:35:24.639567Z","iopub.execute_input":"2023-01-13T14:35:24.641363Z","iopub.status.idle":"2023-01-13T14:35:24.676244Z","shell.execute_reply.started":"2023-01-13T14:35:24.64128Z","shell.execute_reply":"2023-01-13T14:35:24.675032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create data generator\nfrom keras.preprocessing.image import ImageDataGenerator\n\ndatagen = ImageDataGenerator(\n    horizontal_flip=False,\n    vertical_flip=False\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T14:35:25.769432Z","iopub.execute_input":"2023-01-13T14:35:25.769808Z","iopub.status.idle":"2023-01-13T14:35:25.775527Z","shell.execute_reply.started":"2023-01-13T14:35:25.769778Z","shell.execute_reply":"2023-01-13T14:35:25.774461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split dataset\nfrom sklearn.model_selection import train_test_split\ndf_train, df_val = train_test_split(train_pd, test_size=0.1, stratify=train_pd['cancer'])\n\n# Sample for testing\n# df_train = df_train.head(100)\n# df_val = df_val.head(50)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T14:35:57.318315Z","iopub.execute_input":"2023-01-13T14:35:57.318805Z","iopub.status.idle":"2023-01-13T14:35:57.428077Z","shell.execute_reply.started":"2023-01-13T14:35:57.318763Z","shell.execute_reply":"2023-01-13T14:35:57.426855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create training flow\ntrain_flow = datagen.flow_from_dataframe(\n    df_train,\n    x_col='path',\n    y_col='cancer_one_hot',\n    target_size=(256, 256),\n    color_mode='rgb',\n    class_mode='categorical',\n    batch_size=32,\n    shuffle=True\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T14:35:59.911507Z","iopub.execute_input":"2023-01-13T14:35:59.911882Z","iopub.status.idle":"2023-01-13T14:38:51.871872Z","shell.execute_reply.started":"2023-01-13T14:35:59.911851Z","shell.execute_reply":"2023-01-13T14:38:51.87067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create validation flow\nval_flow = datagen.flow_from_dataframe(\n    df_val,\n    x_col='path',\n    y_col='cancer_one_hot',\n    target_size=(256, 256),\n    color_mode='rgb',\n    class_mode='categorical',\n    batch_size=32,\n    shuffle=True\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T14:38:51.874077Z","iopub.execute_input":"2023-01-13T14:38:51.874615Z","iopub.status.idle":"2023-01-13T14:39:10.994802Z","shell.execute_reply.started":"2023-01-13T14:38:51.874577Z","shell.execute_reply":"2023-01-13T14:39:10.993728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train","metadata":{}},{"cell_type":"code","source":"# Fit the model\nmodel.fit(\n    x=train_flow,\n    epochs=1,\n    validation_data=val_flow,\n    class_weight={0: 1, 1:10}    # Since the dataset is very imbalanced I gave some weights to hopefully help the NN a bit\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T14:57:55.175742Z","iopub.execute_input":"2023-01-13T14:57:55.176111Z","iopub.status.idle":"2023-01-13T15:03:05.16325Z","shell.execute_reply.started":"2023-01-13T14:57:55.17608Z","shell.execute_reply":"2023-01-13T15:03:05.162281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predict (not finished yet)","metadata":{}},{"cell_type":"code","source":"from keras.preprocessing import image\n\ndef get_image_from_path(img_path):\n    img = image.load_img(img_path, target_size = (256, 256)) \n    img = image.img_to_array(img)\n    img = np.expand_dims(img, axis = 0)\n    return img\n\npath_to_image = '/kaggle/input/rsna-breast-cancer-256-pngs/10006_1459541791.png'\nimg = get_image_from_path(path_to_image)\n# Predict the result\nmodel.predict(img)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:27:47.859616Z","iopub.execute_input":"2023-01-13T15:27:47.860162Z","iopub.status.idle":"2023-01-13T15:27:47.932241Z","shell.execute_reply.started":"2023-01-13T15:27:47.860116Z","shell.execute_reply":"2023-01-13T15:27:47.930939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Convert test images to png and predict","metadata":{}},{"cell_type":"code","source":"def save_dicom_to_png(f, size=512, save_folder=\"\", extension=\"png\"):\n    patient = f.split('/')[-2]\n    image = f.split('/')[-1][:-4]\n\n    dicom = pydicom.dcmread(f)\n    img = dicom.pixel_array\n\n    img = (img - img.min()) / (img.max() - img.min())\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n\n    img = cv2.resize(img, (size, size))\n    \n    save_to = save_folder + f\"{patient}_{image}.{extension}\"\n\n    cv2.imwrite(save_to, (img * 255).astype(np.uint8))\n    return\n\ndef convert_dicom_to_png(f, size=512, save_folder=\"\", extension=\"png\"):\n    patient = f.split('/')[-2]\n    image = f.split('/')[-1][:-4]\n\n    dicom = pydicom.dcmread(f)\n    img = dicom.pixel_array\n\n    img = (img - img.min()) / (img.max() - img.min())\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n\n    img = cv2.resize(img, (size, size))\n    \n    save_to = save_folder + f\"{patient}_{image}.{extension}\"\n\n#     cv2.imwrite(save_to, (img * 255).astype(np.uint8))\n    return (img * 255).astype(np.uint8)","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:35:51.106265Z","iopub.execute_input":"2023-01-13T15:35:51.106629Z","iopub.status.idle":"2023-01-13T15:35:51.116474Z","shell.execute_reply.started":"2023-01-13T15:35:51.1066Z","shell.execute_reply":"2023-01-13T15:35:51.115264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_pd = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:17:59.512625Z","iopub.execute_input":"2023-01-13T15:17:59.513044Z","iopub.status.idle":"2023-01-13T15:17:59.524814Z","shell.execute_reply.started":"2023-01-13T15:17:59.513012Z","shell.execute_reply":"2023-01-13T15:17:59.52359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_pd['path'] = '/kaggle/input/rsna-breast-cancer-detection/test_images/' + test_pd['patient_id'].astype(str) + '/' + test_pd['image_id'].astype(str) + '.dcm'\n# test_pd['path'] = '/kaggle/working/output/' + test_pd['patient_id'].astype(str) + '_' + test_pd['image_id'].astype(str) + '.png'","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:37:34.783128Z","iopub.execute_input":"2023-01-13T15:37:34.783523Z","iopub.status.idle":"2023-01-13T15:37:34.793332Z","shell.execute_reply.started":"2023-01-13T15:37:34.783489Z","shell.execute_reply":"2023-01-13T15:37:34.792114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert all test files into png format\n# _ = Parallel(n_jobs=4)(\n#     delayed(save_dicom_to_png)(uid, size=256, save_folder='output/', extension='png')\n#     for uid in tqdm(test_pd.path.tolist())\n# )","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:35:58.21472Z","iopub.execute_input":"2023-01-13T15:35:58.2151Z","iopub.status.idle":"2023-01-13T15:36:00.391212Z","shell.execute_reply.started":"2023-01-13T15:35:58.215065Z","shell.execute_reply":"2023-01-13T15:36:00.390155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prediction_ids = []\n# predictions = []\n# for idx, row in test_pd.iterrows():\n#     # Get variables\n#     path_to_png = row['path']\n#     prediction_ids.append(row['prediction_id'])\n#     # Convert to png and save\n# #     img = save_dicom_to_png(path_to_dcm, size=256, save_folder='kaggle/working/output/', extension='png')\n#     # Read image\n#     img = get_image_from_path(path_to_png)\n#     prediction = model.predict(img)\n#     predictions.append(prediction)\n#     break","metadata":{"execution":{"iopub.status.busy":"2023-01-13T15:37:49.699609Z","iopub.execute_input":"2023-01-13T15:37:49.700015Z","iopub.status.idle":"2023-01-13T15:37:49.750349Z","shell.execute_reply.started":"2023-01-13T15:37:49.699979Z","shell.execute_reply":"2023-01-13T15:37:49.748837Z"},"trusted":true},"execution_count":null,"outputs":[]}]}