{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! pip install -q git+https://github.com/keras-team/keras-cv","metadata":{"_kg_hide-output":true,"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-08-28T06:41:27.370514Z","iopub.execute_input":"2023-08-28T06:41:27.371068Z","iopub.status.idle":"2023-08-28T06:41:58.809516Z","shell.execute_reply.started":"2023-08-28T06:41:27.371017Z","shell.execute_reply":"2023-08-28T06:41:58.808198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n# You can use `tensorflow`, `pytorch`, `jax` here\n# KerasCore makes the notebook backend agnostic :)\nos.environ[\"KERAS_BACKEND\"] = \"tensorflow\"\n\nimport keras_cv\n# from keras_cv.layers import Augmenter\nimport keras_core as keras\nfrom keras_core import layers\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom matplotlib import pyplot as plt\nfrom sklearn.model_selection import train_test_split\n\nimport gc\nimport cv2\nimport pydicom\nfrom joblib import Parallel, delayed\n\nfrom tqdm.notebook import tqdm\nfrom glob import glob","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:41:58.812136Z","iopub.execute_input":"2023-08-28T06:41:58.812894Z","iopub.status.idle":"2023-08-28T06:41:59.412952Z","shell.execute_reply.started":"2023-08-28T06:41:58.812853Z","shell.execute_reply":"2023-08-28T06:41:59.41196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config","metadata":{}},{"cell_type":"code","source":"class Config:\n    SEED = 42\n    IMAGE_SIZE = [256, 256]\n    BATCH_SIZE = 16\n    EPOCHS = 10\n    TARGET_COLS  = [\n        \"bowel_injury\", \"extravasation_injury\",\n        \"kidney_healthy\", \"kidney_low\", \"kidney_high\",\n        \"liver_healthy\", \"liver_low\", \"liver_high\",\n        \"spleen_healthy\", \"spleen_low\", \"spleen_high\",\n    ]\n    AUTOTUNE = tf.data.AUTOTUNE\n\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:41:59.414289Z","iopub.execute_input":"2023-08-28T06:41:59.414737Z","iopub.status.idle":"2023-08-28T06:41:59.422075Z","shell.execute_reply.started":"2023-08-28T06:41:59.414704Z","shell.execute_reply":"2023-08-28T06:41:59.420402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reproducibility","metadata":{}},{"cell_type":"code","source":"keras.utils.set_random_seed(seed=config.SEED)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:41:59.424746Z","iopub.execute_input":"2023-08-28T06:41:59.42568Z","iopub.status.idle":"2023-08-28T06:41:59.433377Z","shell.execute_reply.started":"2023-08-28T06:41:59.425648Z","shell.execute_reply":"2023-08-28T06:41:59.432273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset\n\nThe dataset provided in the competition consists of DICOM images. We will not be training on the DICOM images, rather would work on PNG image which are extracted from the DICOM format.\n\n[A helpful resource on the conversion of DICOM to PNG](https://www.kaggle.com/code/radek1/how-to-process-dicom-images-to-pngs)","metadata":{}},{"cell_type":"code","source":"BASE_PATH = f\"/kaggle/input/rsna-atd-512x512-png-v2-dataset\"","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-08-28T06:41:59.434662Z","iopub.execute_input":"2023-08-28T06:41:59.435071Z","iopub.status.idle":"2023-08-28T06:41:59.444648Z","shell.execute_reply.started":"2023-08-28T06:41:59.435038Z","shell.execute_reply":"2023-08-28T06:41:59.443592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Meta Data\n\nThe `train.csv` file contains the following meta information:\n\n- `patient_id`: A unique ID code for each patient.\n- `series_id`: A unique ID code for each scan.\n- `instance_number`: The image number within the scan. The lowest instance number for many series is above zero as the original scans were cropped to the abdomen.\n- `[bowel/extravasation]_[healthy/injury]`: The two injury types with binary targets.\n- `[kidney/liver/spleen]_[healthy/low/high]`: The three injury types with three target levels.\n- `any_injury`: Whether the patient had any injury at all.\n","metadata":{}},{"cell_type":"code","source":"# train\ndataframe = pd.read_csv(f\"{BASE_PATH}/train.csv\")\ndataframe[\"image_path\"] = f\"{BASE_PATH}/train_images\"\\\n                    + \"/\" + dataframe.patient_id.astype(str)\\\n                    + \"/\" + dataframe.series_id.astype(str)\\\n                    + \"/\" + dataframe.instance_number.astype(str) +\".png\"\ndataframe = dataframe.drop_duplicates()\n\ndataframe.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:41:59.445911Z","iopub.execute_input":"2023-08-28T06:41:59.446906Z","iopub.status.idle":"2023-08-28T06:41:59.60696Z","shell.execute_reply.started":"2023-08-28T06:41:59.446872Z","shell.execute_reply":"2023-08-28T06:41:59.606023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We split the training dataset into train and validation. This is a common practise in the Machine Learning pipelines. We not only want to train our model, but also want to validate it's training.\n\nA small catch here is that the training and validation data should have an aligned data distribution. Here we handle that by grouping the lables and then splitting the dataset. This ensures an aligned data distribution between the training and the validation splits.","metadata":{}},{"cell_type":"code","source":"# Function to handle the split for each group\ndef split_group(group, test_size=0.2):\n    if len(group) == 1:\n        return (group, pd.DataFrame()) if np.random.rand() < test_size else (pd.DataFrame(), group)\n    else:\n        return train_test_split(group, test_size=test_size, random_state=42)\n\n# Initialize the train and validation datasets\ntrain_data = pd.DataFrame()\nval_data = pd.DataFrame()\n\n# Iterate through the groups and split them, handling single-sample groups\nfor _, group in dataframe.groupby(config.TARGET_COLS):\n    train_group, val_group = split_group(group)\n    train_data = pd.concat([train_data, train_group], ignore_index=True)\n    val_data = pd.concat([val_data, val_group], ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:41:59.608271Z","iopub.execute_input":"2023-08-28T06:41:59.609594Z","iopub.status.idle":"2023-08-28T06:41:59.703756Z","shell.execute_reply.started":"2023-08-28T06:41:59.609559Z","shell.execute_reply":"2023-08-28T06:41:59.702701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape, val_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:41:59.705235Z","iopub.execute_input":"2023-08-28T06:41:59.705838Z","iopub.status.idle":"2023-08-28T06:41:59.712885Z","shell.execute_reply.started":"2023-08-28T06:41:59.705802Z","shell.execute_reply":"2023-08-28T06:41:59.711927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Pipeline /w tf.data\n\nHere we build the data pipeline using `tf.data`. Using `tf.data` we can map out data to an augmentation pipeline simple by using the ` map` API.\n\nAdding augmentations to the data pipeline is as simple as adding a layer into the list of layers that the `Augmenter` processes.\n\nReference: https://keras.io/api/keras_cv/layers/augmentation/","metadata":{}},{"cell_type":"code","source":"def decode_image_and_label(image_path, label):\n    file_bytes = tf.io.read_file(image_path)\n    image = tf.io.decode_png(file_bytes, channels=3, dtype=tf.uint8)\n    image = tf.image.resize(image, config.IMAGE_SIZE, method=\"bilinear\")\n    image = tf.cast(image, tf.float32) / 255.0\n    \n    label = tf.cast(label, tf.float32)\n    #         bowel       fluid       kidney      liver       spleen\n    labels = (label[0:1], label[1:2], label[2:5], label[5:8], label[8:11])\n    \n    return (image, labels)\n\n\ndef apply_augmentation(images, labels):\n    augmenter = keras_cv.layers.Augmenter(\n        [\n            keras_cv.layers.RandomFlip(mode=\"horizontal_and_vertical\"),\n            keras_cv.layers.RandomCutout(height_factor=0.2, width_factor=0.2),\n            \n        ]\n    )\n    return (augmenter(images), labels)\n\n\ndef build_dataset(image_paths, labels):\n    ds = (\n        tf.data.Dataset.from_tensor_slices((image_paths, labels))\n        .map(decode_image_and_label, num_parallel_calls=config.AUTOTUNE)\n        .shuffle(config.BATCH_SIZE * 10)\n        .batch(config.BATCH_SIZE)\n        .map(apply_augmentation, num_parallel_calls=config.AUTOTUNE)\n        .prefetch(config.AUTOTUNE)\n    )\n    return ds","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:41:59.714823Z","iopub.execute_input":"2023-08-28T06:41:59.71554Z","iopub.status.idle":"2023-08-28T06:41:59.727118Z","shell.execute_reply.started":"2023-08-28T06:41:59.715503Z","shell.execute_reply":"2023-08-28T06:41:59.726364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths  = train_data.image_path.tolist()\nlabels = train_data[config.TARGET_COLS].values\n\nds = build_dataset(image_paths=paths, labels=labels)\nimages, labels = next(iter(ds))\nimages.shape, [label.shape for label in labels]","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:41:59.730428Z","iopub.execute_input":"2023-08-28T06:41:59.731087Z","iopub.status.idle":"2023-08-28T06:42:04.177154Z","shell.execute_reply.started":"2023-08-28T06:41:59.73103Z","shell.execute_reply":"2023-08-28T06:42:04.175309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# No more customizing your plots by hand, KerasCV has your back ;)\nkeras_cv.visualization.plot_image_gallery(\n    images=images,\n    value_range=(0, 1),\n    rows=2,\n    cols=2,\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:42:04.178367Z","iopub.status.idle":"2023-08-28T06:42:04.179539Z","shell.execute_reply.started":"2023-08-28T06:42:04.179269Z","shell.execute_reply":"2023-08-28T06:42:04.179292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build Model\n\nWe are going to load a pretrained model from the [list of avaiable backbones in KerasCV](https://keras.io/api/keras_cv/models/backbones/). We are using the `ResNetBackbone` as our backbone. The practise of using a pretrained model and finetuning it to a specific dataset is prevalent in the DL community.\n\nWe use the [Functional API](https://keras.io/guides/functional_api/) of Keras to build the model. The design of the model would be such that we input a single image and we get different heads for the various predictions we need (kidney, spleen...).\n\nWe have also added a Learning Rate scheduler for you to work with. When an athlete trains, the first step is always to warm up. We take a similar approach to training our models. We warm up with model where the learning rate increses from the initial LR to a higher LR. After the warmup stage we provide a decay algorithm (cosine here). A list of all the learning rate scheduler can be found [here](https://keras.io/api/optimizers/learning_rate_schedules/).","metadata":{}},{"cell_type":"code","source":"def build_model(warmup_steps, decay_steps):\n    # Define Input\n    inputs = keras.Input(shape=config.IMAGE_SIZE + [3,], batch_size=config.BATCH_SIZE)\n    \n    # Define Backbone\n    backbone = keras_cv.models.ResNetBackbone.from_preset(\"resnet50_imagenet\")\n    backbone.include_rescaling = False\n    x = backbone(inputs)\n    \n    # GAP to get the activation maps\n    gap = keras.layers.GlobalAveragePooling2D()\n    x = gap(x)\n\n    # Define 'necks' for each head\n    x_bowel = keras.layers.Dense(32, activation='silu')(x)\n    x_extra = keras.layers.Dense(32, activation='silu')(x)\n    x_liver = keras.layers.Dense(32, activation='silu')(x)\n    x_kidney = keras.layers.Dense(32, activation='silu')(x)\n    x_spleen = keras.layers.Dense(32, activation='silu')(x)\n\n    # Define heads\n    out_bowel = keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel) # use sigmoid to convert predictions to [0-1]\n    out_extra = keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra) # use sigmoid to convert predictions to [0-1]\n    out_liver = keras.layers.Dense(3, name='liver', activation='softmax')(x_liver) # use softmax for the liver head\n    out_kidney = keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney) # use softmax for the kidney head\n    out_spleen = keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen) # use softmax for the spleen head\n    \n    # Concatenate the outputs\n    outputs = [out_bowel, out_extra, out_liver, out_kidney, out_spleen]\n\n    # Create model\n    print(\"[INFO] Building the model...\")\n    model = keras.Model(inputs=inputs, outputs=outputs)\n    \n    # Cosine Decay\n    cosine_decay = keras.optimizers.schedules.CosineDecay(\n        initial_learning_rate=1e-4,\n        decay_steps=decay_steps,\n        alpha=0.0,\n        warmup_target=1e-3,\n        warmup_steps=warmup_steps,\n    )\n\n    # Compile the model\n    optimizer = keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = {\n        \"bowel\":keras.losses.BinaryCrossentropy(),\n        \"extra\":keras.losses.BinaryCrossentropy(),\n        \"liver\":keras.losses.CategoricalCrossentropy(),\n        \"kidney\":keras.losses.CategoricalCrossentropy(),\n        \"spleen\":keras.losses.CategoricalCrossentropy(),\n    }\n    metrics = {\n        \"bowel\":[\"accuracy\"],\n        \"extra\":[\"accuracy\"],\n        \"liver\":[\"accuracy\"],\n        \"kidney\":[\"accuracy\"],\n        \"spleen\":[\"accuracy\"],\n    }\n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n      loss=loss,\n      metrics=metrics\n    )\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.608324Z","iopub.status.idle":"2023-08-28T06:39:19.609043Z","shell.execute_reply.started":"2023-08-28T06:39:19.608794Z","shell.execute_reply":"2023-08-28T06:39:19.608818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train the model with \"model.fit\"","metadata":{}},{"cell_type":"code","source":"# get image_paths and labels\nprint(\"[INFO] Building the dataset...\")\ntrain_paths = train_data.image_path.values; train_labels = train_data[config.TARGET_COLS].values.astype(np.float32)\nvalid_paths = val_data.image_path.values; valid_labels = val_data[config.TARGET_COLS].values.astype(np.float32)\n\n# train and valid dataset\ntrain_ds = build_dataset(image_paths=train_paths, labels=train_labels)\nval_ds = build_dataset(image_paths=valid_paths, labels=valid_labels)\n\ntotal_train_steps = train_ds.cardinality().numpy() * config.BATCH_SIZE * config.EPOCHS\nwarmup_steps = int(total_train_steps * 0.10)\ndecay_steps = total_train_steps - warmup_steps\n\nprint(f\"{total_train_steps=}\")\nprint(f\"{warmup_steps=}\")\nprint(f\"{decay_steps=}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.610587Z","iopub.status.idle":"2023-08-28T06:39:19.611296Z","shell.execute_reply.started":"2023-08-28T06:39:19.611046Z","shell.execute_reply":"2023-08-28T06:39:19.61107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build the model\nprint(\"[INFO] Building the model...\")\nmodel = build_model(warmup_steps, decay_steps)\n\n# train\nprint(\"[INFO] Training...\")\nhistory = model.fit(\n    train_ds,\n    epochs=Config.EPOCHS,\n    validation_data=val_ds,\n)","metadata":{"_kg_hide-input":true,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2023-08-28T06:39:19.612568Z","iopub.status.idle":"2023-08-28T06:39:19.613284Z","shell.execute_reply.started":"2023-08-28T06:39:19.613035Z","shell.execute_reply":"2023-08-28T06:39:19.613058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize the training plots","metadata":{}},{"cell_type":"code","source":"# Create a 3x2 grid for the subplots\nfig, axes = plt.subplots(5, 1, figsize=(5, 15))\n\n# Flatten axes to iterate through them\naxes = axes.flatten()\n\n# Iterate through the metrics and plot them\nfor i, name in enumerate([\"bowel\", \"extra\", \"kidney\", \"liver\", \"spleen\"]):\n    # Plot training accuracy\n    axes[i].plot(history.history[name + '_accuracy'], label='Training ' + name)\n    # Plot validation accuracy\n    axes[i].plot(history.history['val_' + name + '_accuracy'], label='Validation ' + name)\n    axes[i].set_title(name)\n    axes[i].set_xlabel('Epoch')\n    axes[i].set_ylabel('Accuracy')\n    axes[i].legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.614781Z","iopub.status.idle":"2023-08-28T06:39:19.615489Z","shell.execute_reply.started":"2023-08-28T06:39:19.615235Z","shell.execute_reply":"2023-08-28T06:39:19.615258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history[\"loss\"], label=\"loss\")\nplt.plot(history.history[\"val_loss\"], label=\"val loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.616737Z","iopub.status.idle":"2023-08-28T06:39:19.617442Z","shell.execute_reply.started":"2023-08-28T06:39:19.6172Z","shell.execute_reply":"2023-08-28T06:39:19.617224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# store best results\nbest_epoch = np.argmin(history.history['val_loss'])\nbest_loss = history.history['val_loss'][best_epoch]\nbest_acc_bowel = history.history['val_bowel_accuracy'][best_epoch]\nbest_acc_extra = history.history['val_extra_accuracy'][best_epoch]\nbest_acc_liver = history.history['val_liver_accuracy'][best_epoch]\nbest_acc_kidney = history.history['val_kidney_accuracy'][best_epoch]\nbest_acc_spleen = history.history['val_spleen_accuracy'][best_epoch]\n\n# Find mean accuracy\nbest_acc = np.mean(\n    [best_acc_bowel,\n     best_acc_extra,\n     best_acc_liver,\n     best_acc_kidney,\n     best_acc_spleen\n])\n\n\nprint(f'>>>> BEST Loss  : {best_loss:.3f}\\n>>>> BEST Acc   : {best_acc:.3f}\\n>>>> BEST Epoch : {best_epoch}\\n')\nprint('ORGAN Acc:')\nprint(f'  >>>> {\"Bowel\".ljust(15)} : {best_acc_bowel:.3f}')\nprint(f'  >>>> {\"Extravasation\".ljust(15)} : {best_acc_extra:.3f}')\nprint(f'  >>>> {\"Liver\".ljust(15)} : {best_acc_liver:.3f}')\nprint(f'  >>>> {\"Kidney\".ljust(15)} : {best_acc_kidney:.3f}')\nprint(f'  >>>> {\"Spleen\".ljust(15)} : {best_acc_spleen:.3f}')","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.618753Z","iopub.status.idle":"2023-08-28T06:39:19.619445Z","shell.execute_reply.started":"2023-08-28T06:39:19.619201Z","shell.execute_reply":"2023-08-28T06:39:19.619223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Store the model for inference","metadata":{}},{"cell_type":"code","source":"# Save the model\nmodel.save(\"rsna-atd_m01.keras\")","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.620718Z","iopub.status.idle":"2023-08-28T06:39:19.621417Z","shell.execute_reply.started":"2023-08-28T06:39:19.621177Z","shell.execute_reply":"2023-08-28T06:39:19.621199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/rsna-2023-abdominal-trauma-detection\"\nIMAGE_DIR = \"/tmp/dataset/rsna-atd\"\nINPUT_MODEL_PATH = \"/kaggle/working/rsna-atd_m01.keras\"\nMODEL_PATH = \"/kaggle/working/rsna-atd_m01.keras\"\nSTRIDE = 10","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.622677Z","iopub.status.idle":"2023-08-28T06:39:19.623381Z","shell.execute_reply.started":"2023-08-28T06:39:19.623142Z","shell.execute_reply":"2023-08-28T06:39:19.623165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    IMAGE_SIZE = [256, 256]\n    RESIZE_DIM = 256\n    BATCH_SIZE = 32\n    AUTOTUNE = tf.data.AUTOTUNE\n    TARGET_COLS  = [\"bowel_healthy\", \"bowel_injury\", \"extravasation_healthy\",\n                   \"extravasation_injury\", \"kidney_healthy\", \"kidney_low\",\n                   \"kidney_high\", \"liver_healthy\", \"liver_low\", \"liver_high\",\n                   \"spleen_healthy\", \"spleen_low\", \"spleen_high\"]\n\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.624646Z","iopub.status.idle":"2023-08-28T06:39:19.625355Z","shell.execute_reply.started":"2023-08-28T06:39:19.625107Z","shell.execute_reply":"2023-08-28T06:39:19.62513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize the Trained Model\n\n# ! cp {INPUT_MODEL_PATH} ./\n\nmodel = keras.models.load_model(MODEL_PATH)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.62661Z","iopub.status.idle":"2023-08-28T06:39:19.627315Z","shell.execute_reply.started":"2023-08-28T06:39:19.62707Z","shell.execute_reply":"2023-08-28T06:39:19.627092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data pipeline\n\nmeta_df = pd.read_csv(f\"{BASE_PATH}/test_series_meta.csv\")\n\n# Checking if patients are repeated by finding the number of unique patient IDs\nnum_rows = meta_df.shape[0]\nunique_patients = meta_df[\"patient_id\"].nunique()\n\nprint(f\"{num_rows=}\")\nprint(f\"{unique_patients=}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.628577Z","iopub.status.idle":"2023-08-28T06:39:19.629326Z","shell.execute_reply.started":"2023-08-28T06:39:19.629031Z","shell.execute_reply":"2023-08-28T06:39:19.629054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df[\"dicom_folder\"] = BASE_PATH + \"/\" + \"test_images\"\\\n                                    + \"/\" + meta_df.patient_id.astype(str)\\\n                                    + \"/\" + meta_df.series_id.astype(str)\n\ntest_folders = meta_df.dicom_folder.tolist()\ntest_paths = []\nfor folder in tqdm(test_folders):\n    test_paths += sorted(glob(os.path.join(folder, \"*dcm\")))[::STRIDE]","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.630793Z","iopub.status.idle":"2023-08-28T06:39:19.631552Z","shell.execute_reply.started":"2023-08-28T06:39:19.631266Z","shell.execute_reply":"2023-08-28T06:39:19.63129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.DataFrame(test_paths, columns=[\"dicom_path\"])\ntest_df[\"patient_id\"] = test_df.dicom_path.map(lambda x: x.split(\"/\")[-3]).astype(int)\ntest_df[\"series_id\"] = test_df.dicom_path.map(lambda x: x.split(\"/\")[-2]).astype(int)\ntest_df[\"instance_number\"] = test_df.dicom_path.map(lambda x: x.split(\"/\")[-1].replace(\".dcm\",\"\")).astype(int)\n\ntest_df[\"image_path\"] = f\"{IMAGE_DIR}/test_images\"\\\n                    + \"/\" + test_df.patient_id.astype(str)\\\n                    + \"/\" + test_df.series_id.astype(str)\\\n                    + \"/\" + test_df.instance_number.astype(str) +\".png\"\n\ntest_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.632823Z","iopub.status.idle":"2023-08-28T06:39:19.633579Z","shell.execute_reply.started":"2023-08-28T06:39:19.633315Z","shell.execute_reply":"2023-08-28T06:39:19.633339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking if patients are repeated by finding the number of unique patient IDs\nnum_rows = test_df.shape[0]\nunique_patients = test_df[\"patient_id\"].nunique()\n\nprint(f\"{num_rows=}\")\nprint(f\"{unique_patients=}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.634875Z","iopub.status.idle":"2023-08-28T06:39:19.635591Z","shell.execute_reply.started":"2023-08-28T06:39:19.635326Z","shell.execute_reply":"2023-08-28T06:39:19.635349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DICOM to PNG pipeline\n\n!rm -r {IMAGE_DIR}\nos.makedirs(f\"{IMAGE_DIR}/train_images\", exist_ok=True)\nos.makedirs(f\"{IMAGE_DIR}/test_images\", exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.63686Z","iopub.status.idle":"2023-08-28T06:39:19.63761Z","shell.execute_reply.started":"2023-08-28T06:39:19.637308Z","shell.execute_reply":"2023-08-28T06:39:19.637331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def standardize_pixel_array(dcm):\n    # Correct DICOM pixel_array if PixelRepresentation == 1.\n    pixel_array = dcm.pixel_array\n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n    return pixel_array\n\ndef read_xray(path, fix_monochrome=True):\n    dicom = pydicom.dcmread(path)\n    data = standardize_pixel_array(dicom)\n    data = data - np.min(data)\n    data = data / (np.max(data) + 1e-5)\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = 1.0 - data\n    return data\n\ndef resize_and_save(file_path):\n    img = read_xray(file_path)\n    h, w = img.shape[:2]  # orig hw\n    img = cv2.resize(img, (config.RESIZE_DIM, config.RESIZE_DIM), cv2.INTER_LINEAR)\n    img = (img * 255).astype(np.uint8)\n    \n    sub_path = file_path.split(\"/\",4)[-1].split(\".dcm\")[0] + \".png\"\n    infos = sub_path.split(\"/\")\n    sub_path = file_path.split(\"/\",4)[-1].split(\".dcm\")[0] + \".png\"\n    infos = sub_path.split(\"/\")\n    pid = infos[-3]\n    sid = infos[-2]\n    iid = infos[-1]; iid = iid.replace(\".png\",\"\")\n    new_path = os.path.join(IMAGE_DIR, sub_path)\n    os.makedirs(new_path.rsplit(\"/\",1)[0], exist_ok=True)\n    cv2.imwrite(new_path, img)\n    return","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.6389Z","iopub.status.idle":"2023-08-28T06:39:19.639616Z","shell.execute_reply.started":"2023-08-28T06:39:19.639346Z","shell.execute_reply":"2023-08-28T06:39:19.63937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfile_paths = test_df.dicom_path.tolist()\n_ = Parallel(n_jobs=2, backend=\"threading\")(\n    delayed(resize_and_save)(file_path) for file_path in tqdm(file_paths, leave=True, position=0)\n)\n\ndel _; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.640881Z","iopub.status.idle":"2023-08-28T06:39:19.641588Z","shell.execute_reply.started":"2023-08-28T06:39:19.641325Z","shell.execute_reply":"2023-08-28T06:39:19.641348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Building tf.data pipeline\n\ndef decode_image(image_path):\n    file_bytes = tf.io.read_file(image_path)\n    image = tf.io.decode_png(file_bytes, channels=3, dtype=tf.uint8)\n    image = tf.image.resize(image, config.IMAGE_SIZE, method=\"bilinear\")\n    image = tf.cast(image, tf.float32) / 255.0\n    return image\n\ndef build_dataset(image_paths):\n    ds = (\n        tf.data.Dataset.from_tensor_slices(image_paths)\n        .map(decode_image, num_parallel_calls=config.AUTOTUNE)\n        .shuffle(config.BATCH_SIZE * 10)\n        .batch(config.BATCH_SIZE)\n        .prefetch(config.AUTOTUNE)\n    )\n    return ds","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.642903Z","iopub.status.idle":"2023-08-28T06:39:19.643666Z","shell.execute_reply.started":"2023-08-28T06:39:19.643372Z","shell.execute_reply":"2023-08-28T06:39:19.643396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths  = test_df.image_path.tolist()\n\nds = build_dataset(paths)\nimages = next(iter(ds))\n\nimages.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.645139Z","iopub.status.idle":"2023-08-28T06:39:19.645864Z","shell.execute_reply.started":"2023-08-28T06:39:19.645605Z","shell.execute_reply":"2023-08-28T06:39:19.645629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras_cv.visualization.plot_image_gallery(\n    images=images,\n    value_range=(0, 1),\n    rows=1,\n    cols=3,\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.647108Z","iopub.status.idle":"2023-08-28T06:39:19.647857Z","shell.execute_reply.started":"2023-08-28T06:39:19.647603Z","shell.execute_reply":"2023-08-28T06:39:19.647626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Inference\n\ndef post_proc(pred):\n    proc_pred = np.empty((pred.shape[0], 2*2 + 3*3), dtype=\"float32\")\n\n    # bowel, extravasation\n    proc_pred[:, 0] = pred[:, 0]\n    proc_pred[:, 1] = 1 - proc_pred[:, 0]\n    proc_pred[:, 2] = pred[:, 1]\n    proc_pred[:, 3] = 1 - proc_pred[:, 2]\n    \n    # liver, kidney, sneel\n    proc_pred[:, 4:7] = pred[:, 2:5]\n    proc_pred[:, 7:10] = pred[:, 5:8]\n    proc_pred[:, 10:13] = pred[:, 8:11]\n\n    return proc_pred","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.649394Z","iopub.status.idle":"2023-08-28T06:39:19.650229Z","shell.execute_reply.started":"2023-08-28T06:39:19.649942Z","shell.execute_reply":"2023-08-28T06:39:19.64997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting unique patient IDs from test dataset\npatient_ids = test_df[\"patient_id\"].unique()\n\n# Initializing array to store predictions\npatient_preds = np.zeros(\n    shape=(len(patient_ids), 2*2 + 3*3),\n    dtype=\"float32\"\n)\n\n# Iterating over each patient\nfor pidx, patient_id in tqdm(enumerate(patient_ids), total=len(patient_ids), desc=\"Patients \"):\n    print(f\"Patient ID: {patient_id}\")\n    \n    # Query the dataframe for a particular patient\n    patient_df = test_df.query(\"patient_id == @patient_id\")\n    \n    # Getting image paths for a patient\n    patient_paths = patient_df.image_path.tolist()\n\n    # Building dataset for prediction\n    dtest = build_dataset(patient_paths)\n    \n    # Predicting with the model\n    pred = model.predict(dtest)\n    pred = np.concatenate(pred, axis=-1).astype(\"float32\")\n    pred = pred[:len(patient_paths), :]\n    pred = np.mean(pred.reshape(1, len(patient_paths), 11), axis=0)\n    pred = np.max(pred, axis=0, keepdims=True)\n    \n    patient_preds[pidx, :] += post_proc(pred)[0]\n    \n\n    # Deleting variables to free up memory \n    del patient_df, patient_paths, dtest, pred; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.651588Z","iopub.status.idle":"2023-08-28T06:39:19.652352Z","shell.execute_reply.started":"2023-08-28T06:39:19.65209Z","shell.execute_reply":"2023-08-28T06:39:19.652114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submission\n\n!rm -rf {MODEL_PATH}","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.653615Z","iopub.status.idle":"2023-08-28T06:39:19.654324Z","shell.execute_reply.started":"2023-08-28T06:39:19.654075Z","shell.execute_reply":"2023-08-28T06:39:19.654099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Submission\npred_df = pd.DataFrame({\"patient_id\":patient_ids,})\npred_df[config.TARGET_COLS] = patient_preds.astype(\"float32\")\n\n# Align with sample submission\nsub_df = pd.read_csv(f\"{BASE_PATH}/sample_submission.csv\")\nsub_df = sub_df[[\"patient_id\"]]\nsub_df = sub_df.merge(pred_df, on=\"patient_id\", how=\"left\")\n\nsub_df[config.TARGET_COLS] = sub_df[config.TARGET_COLS].astype(\"float32\")","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.655704Z","iopub.status.idle":"2023-08-28T06:39:19.656467Z","shell.execute_reply.started":"2023-08-28T06:39:19.656213Z","shell.execute_reply":"2023-08-28T06:39:19.656238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Store submission\nsub_df.to_csv(\"submission.csv\",index=False)\nsub_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T06:39:19.657804Z","iopub.status.idle":"2023-08-28T06:39:19.658518Z","shell.execute_reply.started":"2023-08-28T06:39:19.658253Z","shell.execute_reply":"2023-08-28T06:39:19.658276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}