{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\n\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom\n\nfrom keras import layers\nfrom keras.applications.densenet import DenseNet121\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.initializers import Constant\nfrom keras.models import Sequential\n\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.python.ops import array_ops\nfrom tqdm import tqdm\n\nfrom keras import backend as K\nimport tensorflow as tf","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-05T16:33:40.437343Z","iopub.execute_input":"2022-12-05T16:33:40.438082Z","iopub.status.idle":"2022-12-05T16:33:47.45676Z","shell.execute_reply.started":"2022-12-05T16:33:40.437969Z","shell.execute_reply":"2022-12-05T16:33:47.455702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train.csv')\nsub_df = pd.read_csv('/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_sample_submission.csv')\n\ntrain_df['filename'] = train_df['ID'].apply(lambda st: \"ID_\" + st.split('_')[1] + \".png\")\ntrain_df['type'] = train_df['ID'].apply(lambda st: st.split('_')[2])\nsub_df['filename'] = sub_df['ID'].apply(lambda st: \"ID_\" + st.split('_')[1] + \".png\")\nsub_df['type'] = sub_df['ID'].apply(lambda st: st.split('_')[2])\n\nprint(train_df.shape)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-05T16:33:47.458451Z","iopub.execute_input":"2022-12-05T16:33:47.459104Z","iopub.status.idle":"2022-12-05T16:33:57.974637Z","shell.execute_reply.started":"2022-12-05T16:33:47.459069Z","shell.execute_reply":"2022-12-05T16:33:57.97366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.DataFrame(sub_df.filename.unique(), columns=['filename'])\nprint(test_df.shape)\ntest_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:03:59.470261Z","iopub.execute_input":"2022-12-05T17:03:59.471497Z","iopub.status.idle":"2022-12-05T17:03:59.567835Z","shell.execute_reply.started":"2022-12-05T17:03:59.471448Z","shell.execute_reply":"2022-12-05T17:03:59.566849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(1749)\nsample_files = np.random.choice(os.listdir('/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train/'), 4000)\nsample_df = train_df[train_df.filename.apply(lambda x: x.replace('.png', '.dcm')).isin(sample_files)]","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:04:54.180834Z","iopub.execute_input":"2022-12-05T17:04:54.181782Z","iopub.status.idle":"2022-12-05T17:04:57.471337Z","shell.execute_reply.started":"2022-12-05T17:04:54.181743Z","shell.execute_reply":"2022-12-05T17:04:57.47023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pivot_df = sample_df[['Label', 'filename', 'type']].drop_duplicates().pivot(\n    index='filename', columns='type', values='Label').reset_index()\nprint(pivot_df.shape)\npivot_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:04:57.910514Z","iopub.execute_input":"2022-12-05T17:04:57.911219Z","iopub.status.idle":"2022-12-05T17:04:57.96815Z","shell.execute_reply.started":"2022-12-05T17:04:57.911183Z","shell.execute_reply":"2022-12-05T17:04:57.967007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Rescale, Resize and Convert to PNG**","metadata":{}},{"cell_type":"code","source":"# Ref: https://www.kaggle.com/omission/eda-view-dicom-images-with-correct-windowing\ndef window_image(img, window_center,window_width, intercept, slope, rescale=True):\n\n    img = (img*slope +intercept)\n    img_min = window_center - window_width//2\n    img_max = window_center + window_width//2\n    img[img<img_min] = img_min\n    img[img>img_max] = img_max\n    \n    if rescale:\n        # Extra rescaling to 0-1, not in the original notebook\n        img = (img - img_min) / (img_max - img_min)\n    \n    return img\n    \ndef get_first_of_dicom_field_as_int(x):\n    #get x[0] as in int is x is a 'pydicom.multival.MultiValue', otherwise get int(x)\n    if type(x) == pydicom.multival.MultiValue:\n        return int(x[0])\n    else:\n        return int(x)\n\ndef get_windowing(data):\n    dicom_fields = [data[('0028','1050')].value, #window center\n                    data[('0028','1051')].value, #window width\n                    data[('0028','1052')].value, #intercept\n                    data[('0028','1053')].value] #slope\n    return [get_first_of_dicom_field_as_int(x) for x in dicom_fields]\n","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:05:56.607191Z","iopub.execute_input":"2022-12-05T17:05:56.607598Z","iopub.status.idle":"2022-12-05T17:05:56.618741Z","shell.execute_reply.started":"2022-12-05T17:05:56.607555Z","shell.execute_reply":"2022-12-05T17:05:56.617655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_and_resize(filenames, load_dir):    \n    save_dir = '/kaggle/tmp/'\n    if not os.path.exists(save_dir):\n        os.makedirs(save_dir)\n\n    for filename in tqdm(filenames):\n        path = load_dir + filename\n        new_path = save_dir + filename.replace('.dcm', '.png')\n        \n        dcm = pydicom.dcmread(path)\n        window_center , window_width, intercept, slope = get_windowing(dcm)\n        img = dcm.pixel_array\n        img = window_image(img, window_center, window_width, intercept, slope)\n        \n        resized = cv2.resize(img, (224, 224))\n        res = cv2.imwrite(new_path, resized)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:06:11.9381Z","iopub.execute_input":"2022-12-05T17:06:11.938475Z","iopub.status.idle":"2022-12-05T17:06:11.945206Z","shell.execute_reply.started":"2022-12-05T17:06:11.938443Z","shell.execute_reply":"2022-12-05T17:06:11.94418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_and_resize(filenames=sample_files, load_dir='/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_train/')\nsave_and_resize(filenames=os.listdir('/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_test/'), load_dir='/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/stage_2_test/')","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:06:24.343934Z","iopub.execute_input":"2022-12-05T17:06:24.344318Z","iopub.status.idle":"2022-12-05T17:43:34.42546Z","shell.execute_reply.started":"2022-12-05T17:06:24.344269Z","shell.execute_reply":"2022-12-05T17:43:34.424409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Generator**","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 64\n\ndef create_datagen():\n    return ImageDataGenerator(validation_split=0.15)\n\ndef create_test_gen():\n    return ImageDataGenerator().flow_from_dataframe(\n        test_df,\n        directory='/kaggle/tmp/',\n        x_col='filename',\n        class_mode=None,\n        target_size=(224, 224),\n        batch_size=BATCH_SIZE,\n        shuffle=False\n    )\n\ndef create_flow(datagen, subset):\n    return datagen.flow_from_dataframe(\n        pivot_df, \n        directory='/kaggle/tmp/',\n        x_col='filename', \n        y_col=['any', 'epidural', 'intraparenchymal', \n               'intraventricular', 'subarachnoid', 'subdural'],\n        class_mode='multi_output',\n        target_size=(224, 224),\n        batch_size=BATCH_SIZE,\n        subset=subset\n    )\n\n# Using original generator\ndata_generator = create_datagen()\ntrain_gen = create_flow(data_generator, 'training')\nval_gen = create_flow(data_generator, 'validation')\ntest_gen = create_test_gen()","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:43:34.42734Z","iopub.execute_input":"2022-12-05T17:43:34.428213Z","iopub.status.idle":"2022-12-05T17:43:35.476749Z","shell.execute_reply.started":"2022-12-05T17:43:34.428171Z","shell.execute_reply":"2022-12-05T17:43:35.475754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train Model**\n\nLoss function\n\nFocal loss is good for unbalanced datasets, like this one.","metadata":{}},{"cell_type":"code","source":"def focal_loss(prediction_tensor, target_tensor, weights=None, alpha=0.25, gamma=2):\n    r\"\"\"Compute focal loss for predictions.\n        Multi-labels Focal loss formula:\n            FL = -alpha * (z-p)^gamma * log(p) -(1-alpha) * p^gamma * log(1-p)\n                 ,which alpha = 0.25, gamma = 2, p = sigmoid(x), z = target_tensor.\n\n    \"\"\"\n    sigmoid_p = tf.nn.sigmoid(prediction_tensor)\n    zeros = array_ops.zeros_like(sigmoid_p, dtype=sigmoid_p.dtype)\n    \n    # For poitive prediction, only need consider front part loss, back part is 0;\n    # target_tensor > zeros <=> z=1, so poitive coefficient = z - p.\n    pos_p_sub = array_ops.where(target_tensor > zeros, target_tensor - sigmoid_p, zeros)\n    \n    # For negative prediction, only need consider back part loss, front part is 0;\n    # target_tensor > zeros <=> z=1, so negative coefficient = 0.\n    neg_p_sub = array_ops.where(target_tensor > zeros, zeros, sigmoid_p)\n    per_entry_cross_ent = - alpha * (pos_p_sub ** gamma) * tf.log(tf.clip_by_value(sigmoid_p, 1e-8, 1.0)) \\\n                          - (1 - alpha) * (neg_p_sub ** gamma) * tf.log(tf.clip_by_value(1.0 - sigmoid_p, 1e-8, 1.0))\n    return tf.reduce_sum(per_entry_cross_ent)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:46:26.537574Z","iopub.execute_input":"2022-12-05T17:46:26.537982Z","iopub.status.idle":"2022-12-05T17:46:26.545396Z","shell.execute_reply.started":"2022-12-05T17:46:26.537947Z","shell.execute_reply":"2022-12-05T17:46:26.544398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"densenet = DenseNet121( \n    weights='/kaggle/input/densenet-keras/DenseNet-BC-121-32-no-top.h5',\n    include_top=False,\n    input_shape=(224,224,3)\n)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:47:21.01725Z","iopub.execute_input":"2022-12-05T17:47:21.017625Z","iopub.status.idle":"2022-12-05T17:47:24.188294Z","shell.execute_reply.started":"2022-12-05T17:47:21.017592Z","shell.execute_reply":"2022-12-05T17:47:24.187115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model():\n    model = Sequential()\n    model.add(densenet)\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.5))\n#     model.add(layers.Dense(6, activation='sigmoid', \n#                            bias_initializer=Constant(value=-5.5)))\n    model.add(layers.Dense(6, activation='sigmoid'))\n    \n    model.compile(\n#         loss=focal_loss,\n        loss='categorical_crossentropy',\n        optimizer=Adam(lr=0.001),\n        metrics=['accuracy']\n    )\n    \n    return model\n","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:47:40.139123Z","iopub.execute_input":"2022-12-05T17:47:40.139505Z","iopub.status.idle":"2022-12-05T17:47:40.150858Z","shell.execute_reply.started":"2022-12-05T17:47:40.13947Z","shell.execute_reply":"2022-12-05T17:47:40.149806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = build_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:47:40.154219Z","iopub.execute_input":"2022-12-05T17:47:40.15526Z","iopub.status.idle":"2022-12-05T17:47:41.003991Z","shell.execute_reply.started":"2022-12-05T17:47:40.15522Z","shell.execute_reply":"2022-12-05T17:47:41.002757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint = ModelCheckpoint(\n    'model.h5', \n    monitor='val_loss', \n    verbose=0, \n    save_best_only=True, \n    save_weights_only=False,\n    mode='auto'\n)\n\ntotal_steps = sample_files.shape[0] / BATCH_SIZE\n\nhistory = model.fit_generator(\n    train_gen,\n    steps_per_epoch=2000,\n    validation_data=val_gen,\n    validation_steps=total_steps * 0.15,\n    callbacks=[checkpoint],\n    epochs=5\n)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:48:29.081565Z","iopub.execute_input":"2022-12-05T17:48:29.082279Z","iopub.status.idle":"2022-12-05T17:48:30.743194Z","shell.execute_reply.started":"2022-12-05T17:48:29.082243Z","shell.execute_reply":"2022-12-05T17:48:30.74165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('history.json', 'w') as f:\n    json.dump(history.history, f)\n\nhistory_df = pd.DataFrame(history.history)\nhistory_df[['loss', 'val_loss']].plot()\nhistory_df[['acc', 'val_acc']].plot()\n","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:47:42.393085Z","iopub.status.idle":"2022-12-05T17:47:42.393655Z","shell.execute_reply.started":"2022-12-05T17:47:42.393366Z","shell.execute_reply":"2022-12-05T17:47:42.393396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Submission:**","metadata":{}},{"cell_type":"code","source":"model.load_weights('model.h5')\ny_test = model.predict_generator(\n    test_gen,\n    steps=len(test_gen),\n    verbose=1\n)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:47:42.395163Z","iopub.status.idle":"2022-12-05T17:47:42.396234Z","shell.execute_reply.started":"2022-12-05T17:47:42.395938Z","shell.execute_reply":"2022-12-05T17:47:42.395971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Append the output predicts in the wide format to the y_test\ntest_df = test_df.join(pd.DataFrame(y_test, columns=[\n    'any', 'epidural', 'intraparenchymal', 'intraventricular', 'subarachnoid', 'subdural'\n]))\n\n# Unpivot table, i.e. wide (N x 6) to long format (6N x 1)\ntest_df = test_df.melt(id_vars=['filename'])\n\n# Combine the filename column with the variable column\ntest_df['ID'] = test_df.filename.apply(lambda x: x.replace('.png', '')) + '_' + test_df.variable\ntest_df['Label'] = test_df['value']\n\ntest_df[['ID', 'Label']].to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T17:47:42.398154Z","iopub.status.idle":"2022-12-05T17:47:42.398697Z","shell.execute_reply.started":"2022-12-05T17:47:42.398402Z","shell.execute_reply":"2022-12-05T17:47:42.398437Z"},"trusted":true},"execution_count":null,"outputs":[]}]}