{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**The first version of the notebook is the basic version which only includes training. This notebook fixes two main errors observed in original notebook by Keras:**\nhttps://www.kaggle.com/code/aritrag/kerascv-starter-notebook-train\n\n**The following errors were reported/observed in the aforementioned notebook and are fixed in this notebook:**\n\nAttributeError: module 'keras_cv.layers' has no attribute 'Augmenter'\n\nTensor error observed when training.\n\nMatplotlib error in subplots\n\n**This notebook is work in progress and expect EDA, Advanced training modules and Inference in later versions of this notebook. Thank you for your support.**","metadata":{}},{"cell_type":"code","source":"! pip install -q git+https://github.com/keras-team/keras-cv","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:29:06.759878Z","iopub.execute_input":"2023-09-02T04:29:06.76037Z","iopub.status.idle":"2023-09-02T04:29:36.994466Z","shell.execute_reply.started":"2023-09-02T04:29:06.760329Z","shell.execute_reply":"2023-09-02T04:29:36.99295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install -q keras-core","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:29:36.997699Z","iopub.execute_input":"2023-09-02T04:29:36.998928Z","iopub.status.idle":"2023-09-02T04:29:48.374715Z","shell.execute_reply.started":"2023-09-02T04:29:36.998891Z","shell.execute_reply":"2023-09-02T04:29:48.373447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"tensorflow\"\nimport tensorflow as tf\nimport keras_cv\nimport keras_core as keras\nfrom keras_core import layers\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib as plt","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:39:35.552898Z","iopub.execute_input":"2023-09-02T04:39:35.554053Z","iopub.status.idle":"2023-09-02T04:39:41.486852Z","shell.execute_reply.started":"2023-09-02T04:39:35.554007Z","shell.execute_reply":"2023-09-02T04:39:41.485776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(keras_cv.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:39:41.490404Z","iopub.execute_input":"2023-09-02T04:39:41.490942Z","iopub.status.idle":"2023-09-02T04:39:41.496877Z","shell.execute_reply.started":"2023-09-02T04:39:41.490913Z","shell.execute_reply":"2023-09-02T04:39:41.495795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras_cv import layers","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:39:42.2386Z","iopub.execute_input":"2023-09-02T04:39:42.238991Z","iopub.status.idle":"2023-09-02T04:39:42.244405Z","shell.execute_reply.started":"2023-09-02T04:39:42.238945Z","shell.execute_reply":"2023-09-02T04:39:42.243287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    IMAGE_SIZE = [256, 256]\n    BATCH_SIZE = 6\n    EPOCHS = 20\n    TARGET_COLS  = [\n        \"bowel_injury\", \"bowel_healthy\", \"extravasation_healthy\", \"extravasation_injury\",\n        \"kidney_healthy\", \"kidney_low\", \"kidney_high\",\n        \"liver_healthy\", \"liver_low\", \"liver_high\",\n        \"spleen_healthy\", \"spleen_low\", \"spleen_high\",\n    ]\n    AUTOTUNE = tf.data.AUTOTUNE\n\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:39:43.016927Z","iopub.execute_input":"2023-09-02T04:39:43.017661Z","iopub.status.idle":"2023-09-02T04:39:43.023903Z","shell.execute_reply.started":"2023-09-02T04:39:43.017626Z","shell.execute_reply":"2023-09-02T04:39:43.022789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = f\"/kaggle/input/rsna-atd-512x512-png-v2-dataset\"","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:39:44.22326Z","iopub.execute_input":"2023-09-02T04:39:44.223619Z","iopub.status.idle":"2023-09-02T04:39:44.228661Z","shell.execute_reply.started":"2023-09-02T04:39:44.22359Z","shell.execute_reply":"2023-09-02T04:39:44.227665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train\ndataframe = pd.read_csv(f\"{BASE_PATH}/train.csv\")\ndataframe[\"image_path\"] = f\"{BASE_PATH}/train_images\"\\\n                    + \"/\" + dataframe.patient_id.astype(str)\\\n                    + \"/\" + dataframe.series_id.astype(str)\\\n                    + \"/\" + dataframe.instance_number.astype(str) +\".png\"\ndataframe = dataframe.drop_duplicates()\n\ndataframe.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:39:45.41707Z","iopub.execute_input":"2023-09-02T04:39:45.41748Z","iopub.status.idle":"2023-09-02T04:39:45.549424Z","shell.execute_reply.started":"2023-09-02T04:39:45.417447Z","shell.execute_reply":"2023-09-02T04:39:45.548354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to handle the split for each group\ndef split_group(group, test_size=0.2):\n    if len(group) == 1:\n        return (group, pd.DataFrame()) if np.random.rand() < test_size else (pd.DataFrame(), group)\n    else:\n        return train_test_split(group, test_size=test_size, random_state=42)\n\n# Initialize the train and validation datasets\ntrain_data = pd.DataFrame()\nval_data = pd.DataFrame()\n\n# Iterate through the groups and split them, handling single-sample groups\nfor _, group in dataframe.groupby(config.TARGET_COLS):\n    train_group, val_group = split_group(group)\n    train_data = pd.concat([train_data, train_group], ignore_index=True)\n    val_data = pd.concat([val_data, val_group], ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:39:45.593401Z","iopub.execute_input":"2023-09-02T04:39:45.594149Z","iopub.status.idle":"2023-09-02T04:39:45.692403Z","shell.execute_reply.started":"2023-09-02T04:39:45.594114Z","shell.execute_reply":"2023-09-02T04:39:45.691279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape, val_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:39:45.832927Z","iopub.execute_input":"2023-09-02T04:39:45.83333Z","iopub.status.idle":"2023-09-02T04:39:45.840626Z","shell.execute_reply.started":"2023-09-02T04:39:45.833301Z","shell.execute_reply":"2023-09-02T04:39:45.839537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image_and_label(image_path, label):\n    file_bytes = tf.io.read_file(image_path)\n    image = tf.io.decode_png(file_bytes, channels=3, dtype=tf.uint8)\n    image = tf.image.resize(image, config.IMAGE_SIZE, method=\"bilinear\")\n    image = tf.cast(image, tf.float32) / 255.0\n    \n    label = tf.cast(label, tf.float32)\n    #         bowel       fluid       kidney      liver       spleen\n    labels = (label[0:1], label[1:2], label[2:5], label[5:8], label[8:11])\n    \n    return (image, labels)\n\n\ndef apply_augmentation(images, labels):\n    augmenter = keras_cv.layers.Augmenter(\n        [\n            keras_cv.layers.RandomFlip(mode=\"horizontal_and_vertical\"),\n            keras_cv.layers.RandomCutout(height_factor=0.2, width_factor=0.2),\n            \n        ]\n    )\n    return (augmenter(images), labels)\n\n\ndef build_dataset(image_paths, labels):\n    ds = (\n        tf.data.Dataset.from_tensor_slices((image_paths, labels))\n        .map(decode_image_and_label, num_parallel_calls=config.AUTOTUNE)\n        .shuffle(config.BATCH_SIZE * 10)\n        .batch(config.BATCH_SIZE)\n        .map(apply_augmentation, num_parallel_calls=config.AUTOTUNE)\n        .prefetch(config.AUTOTUNE)\n    )\n    return ds","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:39:46.545145Z","iopub.execute_input":"2023-09-02T04:39:46.545873Z","iopub.status.idle":"2023-09-02T04:39:46.555859Z","shell.execute_reply.started":"2023-09-02T04:39:46.545836Z","shell.execute_reply":"2023-09-02T04:39:46.554634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths  = train_data.image_path.tolist()\nlabels = train_data[config.TARGET_COLS].values\n\nds = build_dataset(image_paths=paths, labels=labels)\nimages, labels = next(iter(ds))\nimages.shape, [label.shape for label in labels]","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:39:47.129081Z","iopub.execute_input":"2023-09-02T04:39:47.129458Z","iopub.status.idle":"2023-09-02T04:39:59.053466Z","shell.execute_reply.started":"2023-09-02T04:39:47.129427Z","shell.execute_reply":"2023-09-02T04:39:59.05251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# No more customizing your plots by hand, KerasCV has your back ;)\nkeras_cv.visualization.plot_image_gallery(\n    images=images,\n    value_range=(0, 1),\n    rows=2,\n    cols=2,\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:40:00.6588Z","iopub.execute_input":"2023-09-02T04:40:00.659175Z","iopub.status.idle":"2023-09-02T04:40:01.552552Z","shell.execute_reply.started":"2023-09-02T04:40:00.659142Z","shell.execute_reply":"2023-09-02T04:40:01.551544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fixed the tensor input error by modifying the #Define Backbone Section","metadata":{}},{"cell_type":"code","source":"def build_model(warmup_steps, decay_steps):\n    # Define Input\n    inputs = keras.Input(shape=config.IMAGE_SIZE + [3,], batch_size=config.BATCH_SIZE)\n    \n    # Define Backbone\n    backbone = keras_cv.models.ResNetBackbone.from_preset(\"resnet50_imagenet\")\n    include_rescaling = False\n    x = inputs\n    \n    # GAP to get the activation maps\n    gap = keras.layers.GlobalAveragePooling2D()\n    x = gap(x)\n\n    # Define 'necks' for each head\n    x_bowel = keras.layers.Dense(200, activation='silu')(x)\n    keras.layers.Dropout(0.1)\n    keras.layers.Dense(100, activation='silu')\n    keras.layers.Dropout(0.1)\n    keras.layers.Dense(50, activation='silu')\n    x_extra = keras.layers.Dense(200, activation='silu')(x)\n    keras.layers.Dropout(0.1)\n    keras.layers.Dense(100, activation='silu')\n    keras.layers.Dropout(0.1)\n    keras.layers.Dense(50, activation='silu')\n    x_liver = keras.layers.Dense(200, activation='silu')(x)\n    keras.layers.Dropout(0.1)\n    keras.layers.Dense(100, activation='silu')\n    keras.layers.Dropout(0.1)\n    keras.layers.Dense(50, activation='silu')\n    x_kidney = keras.layers.Dense(200, activation='silu')(x)\n    keras.layers.Dropout(0.1)\n    keras.layers.Dense(100, activation='silu')\n    keras.layers.Dropout(0.1)\n    keras.layers.Dense(50, activation='silu')\n    x_spleen = keras.layers.Dense(200, activation='silu')(x)\n    keras.layers.Dropout(0.1)\n    keras.layers.Dense(100, activation='silu')\n    keras.layers.Dropout(0.1)\n    keras.layers.Dense(50, activation='silu')\n\n    # Define heads\n    out_bowel = keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel) # use sigmoid to convert predictions to [0-1]\n    out_extra = keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra) # use sigmoid to convert predictions to [0-1]\n    out_liver = keras.layers.Dense(3, name='liver', activation='softmax')(x_liver) # use softmax for the liver head\n    out_kidney = keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney) # use softmax for the kidney head\n    out_spleen = keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen) # use softmax for the spleen head\n    \n    # Concatenate the outputs\n    outputs = [out_bowel, out_extra, out_liver, out_kidney, out_spleen]\n\n    # Create model\n    print(\"[INFO] Building the model...\")\n    model = keras.Model(inputs=inputs, outputs=outputs)\n    \n    # Cosine Decay\n    cosine_decay = keras.optimizers.schedules.CosineDecay(\n        initial_learning_rate=1e-4,\n        decay_steps=decay_steps,\n        alpha=0.0,\n        warmup_target=1e-3,\n        warmup_steps=warmup_steps,\n    )\n\n    # Compile the model\n    optimizer = keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = {\n        \"bowel\":keras.losses.BinaryCrossentropy(),\n        \"extra\":keras.losses.BinaryCrossentropy(),\n        \"liver\":keras.losses.CategoricalCrossentropy(),\n        \"kidney\":keras.losses.CategoricalCrossentropy(),\n        \"spleen\":keras.losses.CategoricalCrossentropy(),\n    }\n    metrics = {\n        \"bowel\":[\"accuracy\"],\n        \"extra\":[\"accuracy\"],\n        \"liver\":[\"accuracy\"],\n        \"kidney\":[\"accuracy\"],\n        \"spleen\":[\"accuracy\"],\n    }\n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n      loss=loss,\n      metrics=metrics\n    )\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:43:01.471448Z","iopub.execute_input":"2023-09-02T04:43:01.471833Z","iopub.status.idle":"2023-09-02T04:43:01.492885Z","shell.execute_reply.started":"2023-09-02T04:43:01.471802Z","shell.execute_reply":"2023-09-02T04:43:01.491564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get image_paths and labels\nprint(\"[INFO] Building the dataset...\")\ntrain_paths = train_data.image_path.values; train_labels = train_data[config.TARGET_COLS].values.astype(np.float32)\nvalid_paths = val_data.image_path.values; valid_labels = val_data[config.TARGET_COLS].values.astype(np.float32)\n\n# train and valid dataset\ntrain_ds = build_dataset(image_paths=train_paths, labels=train_labels)\nval_ds = build_dataset(image_paths=valid_paths, labels=valid_labels)\n\ntotal_train_steps = train_ds.cardinality().numpy() * config.BATCH_SIZE * config.EPOCHS\nwarmup_steps = int(total_train_steps * 0.20)\ndecay_steps = total_train_steps - warmup_steps\n\nprint(f\"{total_train_steps=}\")\nprint(f\"{warmup_steps=}\")\nprint(f\"{decay_steps=}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:43:02.19164Z","iopub.execute_input":"2023-09-02T04:43:02.192063Z","iopub.status.idle":"2023-09-02T04:43:03.998171Z","shell.execute_reply.started":"2023-09-02T04:43:02.192029Z","shell.execute_reply":"2023-09-02T04:43:03.997146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build the model\nprint(\"[INFO] Building the model...\")\nmodel = build_model(warmup_steps, decay_steps)\n\n# train\nprint(\"[INFO] Training...\")\nhistory = model.fit(\n    train_ds,\n    epochs=config.EPOCHS,\n    validation_data=val_ds,\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-02T04:43:03.999949Z","iopub.execute_input":"2023-09-02T04:43:04.000263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Fix the error in the referenced notebook : subplots doesn't exist in matplotlib.\nChanging the code to the below will fix this error.**","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\n\n# Create a 3x2 grid for the subplots\nfig = plt.figure(figsize=(5, 15))\ngs = gridspec.GridSpec(3, 2)\n\n# Iterate through the metrics and plot them\nfor i, name in enumerate([\"bowel\", \"extra\", \"kidney\", \"liver\", \"spleen\"]):\n    # Get the current subplot\n    ax = fig.add_subplot(gs[i])\n\n    # Plot training accuracy\n    ax.plot(history.history[name + '_accuracy'], label='Training ' + name)\n\n    # Plot validation accuracy\n    ax.plot(history.history['val_' + name + '_accuracy'], label='Validation ' + name)\n\n    # Set the title, xlabel, and ylabel\n    ax.set_title(name)\n    ax.set_xlabel('Epoch')\n    ax.set_ylabel('Accuracy')\n    ax.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T16:38:29.18624Z","iopub.execute_input":"2023-09-01T16:38:29.186616Z","iopub.status.idle":"2023-09-01T16:38:30.381661Z","shell.execute_reply.started":"2023-09-01T16:38:29.186584Z","shell.execute_reply":"2023-09-01T16:38:30.380628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history[\"loss\"], label=\"loss\")\nplt.plot(history.history[\"val_loss\"], label=\"val loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T16:38:41.906905Z","iopub.execute_input":"2023-09-01T16:38:41.907312Z","iopub.status.idle":"2023-09-01T16:38:42.156012Z","shell.execute_reply.started":"2023-09-01T16:38:41.907279Z","shell.execute_reply":"2023-09-01T16:38:42.154966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# store best results\nbest_epoch = np.argmin(history.history['val_loss'])\nbest_loss = history.history['val_loss'][best_epoch]\nbest_acc_bowel = history.history['val_bowel_accuracy'][best_epoch]\nbest_acc_extra = history.history['val_extra_accuracy'][best_epoch]\nbest_acc_liver = history.history['val_liver_accuracy'][best_epoch]\nbest_acc_kidney = history.history['val_kidney_accuracy'][best_epoch]\nbest_acc_spleen = history.history['val_spleen_accuracy'][best_epoch]\n\n# Find mean accuracy\nbest_acc = np.mean(\n    [best_acc_bowel,\n     best_acc_extra,\n     best_acc_liver,\n     best_acc_kidney,\n     best_acc_spleen\n])\n\n\nprint(f'>>>> BEST Loss  : {best_loss:.3f}\\n>>>> BEST Acc   : {best_acc:.3f}\\n>>>> BEST Epoch : {best_epoch}\\n')\nprint('ORGAN Acc:')\nprint(f'  >>>> {\"Bowel\".ljust(15)} : {best_acc_bowel:.3f}')\nprint(f'  >>>> {\"Extravasation\".ljust(15)} : {best_acc_extra:.3f}')\nprint(f'  >>>> {\"Liver\".ljust(15)} : {best_acc_liver:.3f}')\nprint(f'  >>>> {\"Kidney\".ljust(15)} : {best_acc_kidney:.3f}')\nprint(f'  >>>> {\"Spleen\".ljust(15)} : {best_acc_spleen:.3f}')","metadata":{"execution":{"iopub.status.busy":"2023-09-01T16:38:55.525068Z","iopub.execute_input":"2023-09-01T16:38:55.525431Z","iopub.status.idle":"2023-09-01T16:38:55.535113Z","shell.execute_reply.started":"2023-09-01T16:38:55.525402Z","shell.execute_reply":"2023-09-01T16:38:55.534024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the model\nmodel.save(\"rsna-v1.keras\")","metadata":{"execution":{"iopub.status.busy":"2023-09-01T16:39:11.145888Z","iopub.execute_input":"2023-09-01T16:39:11.146288Z","iopub.status.idle":"2023-09-01T16:39:11.204979Z","shell.execute_reply.started":"2023-09-01T16:39:11.146257Z","shell.execute_reply":"2023-09-01T16:39:11.203979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **INFERENCE**","metadata":{}},{"cell_type":"code","source":"! cp {/kaggle/working/rsna-v1.keras} ./\n\nmodel = keras.models.load_model(\"rsna-v1.keras\")\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T16:39:50.306351Z","iopub.execute_input":"2023-09-01T16:39:50.307117Z","iopub.status.idle":"2023-09-01T16:39:51.599346Z","shell.execute_reply.started":"2023-09-01T16:39:50.307082Z","shell.execute_reply":"2023-09-01T16:39:51.598123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/rsna-2023-abdominal-trauma-detection\"\nIMAGE_DIR = \"/tmp/dataset/rsna-atd\"\nINPUT_MODEL_PATH = \"/kaggle/input/kerascv-starter-notebook-train/rsna-atd.keras\"\nMODEL_PATH = \"/kaggle/working/rsna-atd.keras\"\nSTRIDE = 10","metadata":{"execution":{"iopub.status.busy":"2023-09-01T16:40:15.150998Z","iopub.execute_input":"2023-09-01T16:40:15.151384Z","iopub.status.idle":"2023-09-01T16:40:15.156509Z","shell.execute_reply.started":"2023-09-01T16:40:15.151355Z","shell.execute_reply":"2023-09-01T16:40:15.155507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df = pd.read_csv(f\"{BASE_PATH}/test_series_meta.csv\")\n\n# Checking if patients are repeated by finding the number of unique patient IDs\nnum_rows = meta_df.shape[0]\nunique_patients = meta_df[\"patient_id\"].nunique()\n\nprint(f\"{num_rows=}\")\nprint(f\"{unique_patients=}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-01T16:40:25.260943Z","iopub.execute_input":"2023-09-01T16:40:25.261351Z","iopub.status.idle":"2023-09-01T16:40:25.279709Z","shell.execute_reply.started":"2023-09-01T16:40:25.26132Z","shell.execute_reply":"2023-09-01T16:40:25.278672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nfrom glob import glob\nmeta_df[\"dicom_folder\"] = BASE_PATH + \"/\" + \"test_images\"\\\n                                    + \"/\" + meta_df.patient_id.astype(str)\\\n                                    + \"/\" + meta_df.series_id.astype(str)\n\ntest_folders = meta_df.dicom_folder.tolist()\ntest_paths = []\nfor folder in tqdm(test_folders):\n    test_paths += sorted(glob(os.path.join(folder, \"*dcm\")))[::STRIDE]","metadata":{"execution":{"iopub.status.busy":"2023-09-01T16:40:46.7859Z","iopub.execute_input":"2023-09-01T16:40:46.786916Z","iopub.status.idle":"2023-09-01T16:40:46.834335Z","shell.execute_reply.started":"2023-09-01T16:40:46.786882Z","shell.execute_reply":"2023-09-01T16:40:46.833338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.DataFrame(test_paths, columns=[\"dicom_path\"])\ntest_df[\"patient_id\"] = test_df.dicom_path.map(lambda x: x.split(\"/\")[-3]).astype(int)\ntest_df[\"series_id\"] = test_df.dicom_path.map(lambda x: x.split(\"/\")[-2]).astype(int)\ntest_df[\"instance_number\"] = test_df.dicom_path.map(lambda x: x.split(\"/\")[-1].replace(\".dcm\",\"\")).astype(int)\n\ntest_df[\"image_path\"] = f\"{IMAGE_DIR}/test_images\"\\\n                    + \"/\" + test_df.patient_id.astype(str)\\\n                    + \"/\" + test_df.series_id.astype(str)\\\n                    + \"/\" + test_df.instance_number.astype(str) +\".png\"\n\ntest_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T16:41:09.165104Z","iopub.execute_input":"2023-09-01T16:41:09.165505Z","iopub.status.idle":"2023-09-01T16:41:09.187516Z","shell.execute_reply.started":"2023-09-01T16:41:09.165473Z","shell.execute_reply":"2023-09-01T16:41:09.186385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking if patients are repeated by finding the number of unique patient IDs\nnum_rows = test_df.shape[0]\nunique_patients = test_df[\"patient_id\"].nunique()\n\nprint(f\"{num_rows=}\")\nprint(f\"{unique_patients=}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-01T16:41:18.524996Z","iopub.execute_input":"2023-09-01T16:41:18.525387Z","iopub.status.idle":"2023-09-01T16:41:18.531519Z","shell.execute_reply.started":"2023-09-01T16:41:18.525357Z","shell.execute_reply":"2023-09-01T16:41:18.530554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"tensorflow\"\nimport keras_core as keras\nimport keras_cv\n\nimport gc\nimport cv2\nimport pydicom\nfrom joblib import Parallel, delayed\n\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nfrom glob import glob","metadata":{"execution":{"iopub.status.busy":"2023-09-01T16:56:53.492896Z","iopub.execute_input":"2023-09-01T16:56:53.493917Z","iopub.status.idle":"2023-09-01T16:56:53.500437Z","shell.execute_reply.started":"2023-09-01T16:56:53.49388Z","shell.execute_reply":"2023-09-01T16:56:53.499225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    IMAGE_SIZE = [256, 256]\n    RESIZE_DIM = 256\n    BATCH_SIZE = 64\n    AUTOTUNE = tf.data.AUTOTUNE\n    TARGET_COLS  = [\"bowel_healthy\", \"bowel_injury\", \"extravasation_healthy\",\n                   \"extravasation_injury\", \"kidney_healthy\", \"kidney_low\",\n                   \"kidney_high\", \"liver_healthy\", \"liver_low\", \"liver_high\",\n                   \"spleen_healthy\", \"spleen_low\", \"spleen_high\"]\n\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T17:05:05.425753Z","iopub.execute_input":"2023-09-01T17:05:05.426171Z","iopub.status.idle":"2023-09-01T17:05:05.432942Z","shell.execute_reply.started":"2023-09-01T17:05:05.426138Z","shell.execute_reply":"2023-09-01T17:05:05.431839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -r {IMAGE_DIR}\nos.makedirs(f\"{IMAGE_DIR}/train_images\", exist_ok=True)\nos.makedirs(f\"{IMAGE_DIR}/test_images\", exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T17:05:09.586104Z","iopub.execute_input":"2023-09-01T17:05:09.586495Z","iopub.status.idle":"2023-09-01T17:05:10.71094Z","shell.execute_reply.started":"2023-09-01T17:05:09.586465Z","shell.execute_reply":"2023-09-01T17:05:10.709533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def standardize_pixel_array(dcm):\n    # Correct DICOM pixel_array if PixelRepresentation == 1.\n    pixel_array = dcm.pixel_array\n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n    return pixel_array\n\ndef read_xray(path, fix_monochrome=True):\n    dicom = pydicom.dcmread(path)\n    data = standardize_pixel_array(dicom)\n    data = data - np.min(data)\n    data = data / (np.max(data) + 1e-5)\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = 1.0 - data\n    return data\n\ndef resize_and_save(file_path):\n    img = read_xray(file_path)\n    h, w = img.shape[:2]  # orig hw\n    img = cv2.resize(img, (config.RESIZE_DIM, config.RESIZE_DIM), cv2.INTER_LINEAR)\n    img = (img * 255).astype(np.uint8)\n    \n    sub_path = file_path.split(\"/\",4)[-1].split(\".dcm\")[0] + \".png\"\n    infos = sub_path.split(\"/\")\n    sub_path = file_path.split(\"/\",4)[-1].split(\".dcm\")[0] + \".png\"\n    infos = sub_path.split(\"/\")\n    pid = infos[-3]\n    sid = infos[-2]\n    iid = infos[-1]; iid = iid.replace(\".png\",\"\")\n    new_path = os.path.join(IMAGE_DIR, sub_path)\n    os.makedirs(new_path.rsplit(\"/\",1)[0], exist_ok=True)\n    cv2.imwrite(new_path, img)\n    return","metadata":{"execution":{"iopub.status.busy":"2023-09-01T17:05:18.73343Z","iopub.execute_input":"2023-09-01T17:05:18.733853Z","iopub.status.idle":"2023-09-01T17:05:18.748919Z","shell.execute_reply.started":"2023-09-01T17:05:18.733819Z","shell.execute_reply":"2023-09-01T17:05:18.747812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfile_paths = test_df.dicom_path.tolist()\n_ = Parallel(n_jobs=2, backend=\"threading\")(\n    delayed(resize_and_save)(file_path) for file_path in tqdm(file_paths, leave=True, position=0)\n)\n\ndel _; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T17:05:21.684788Z","iopub.execute_input":"2023-09-01T17:05:21.68519Z","iopub.status.idle":"2023-09-01T17:05:22.496609Z","shell.execute_reply.started":"2023-09-01T17:05:21.685157Z","shell.execute_reply":"2023-09-01T17:05:22.495538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(image_path):\n    file_bytes = tf.io.read_file(image_path)\n    image = tf.io.decode_png(file_bytes, channels=3, dtype=tf.uint8)\n    image = tf.image.resize(image, config.IMAGE_SIZE, method=\"bilinear\")\n    image = tf.cast(image, tf.float32) / 255.0\n    return image\n\ndef build_dataset(image_paths):\n    ds = (\n        tf.data.Dataset.from_tensor_slices(image_paths)\n        .map(decode_image, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n        .shuffle(config.BATCH_SIZE * 10)\n        .batch(config.BATCH_SIZE)\n        .prefetch(tf.data.experimental.AUTOTUNE)\n    )\n    return ds\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T17:05:28.012735Z","iopub.execute_input":"2023-09-01T17:05:28.013168Z","iopub.status.idle":"2023-09-01T17:05:28.020837Z","shell.execute_reply.started":"2023-09-01T17:05:28.013132Z","shell.execute_reply":"2023-09-01T17:05:28.019754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths  = test_df.image_path.tolist()\n\nds = build_dataset(paths)\nimages = next(iter(ds))\n\nimages.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-01T17:05:32.461729Z","iopub.execute_input":"2023-09-01T17:05:32.462132Z","iopub.status.idle":"2023-09-01T17:05:32.533921Z","shell.execute_reply.started":"2023-09-01T17:05:32.462097Z","shell.execute_reply":"2023-09-01T17:05:32.532992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def post_proc(pred):\n    proc_pred = np.empty((pred.shape[0], 2*2 + 3*3), dtype=\"float32\")\n\n    # bowel, extravasation\n    proc_pred[:, 0] = pred[:, 0]\n    proc_pred[:, 1] = 1 - proc_pred[:, 0]\n    proc_pred[:, 2] = pred[:, 1]\n    proc_pred[:, 3] = 1 - proc_pred[:, 2]\n    \n    # liver, kidney, sneel\n    proc_pred[:, 4:7] = pred[:, 2:5]\n    proc_pred[:, 7:10] = pred[:, 5:8]\n    proc_pred[:, 10:13] = pred[:, 8:11]\n\n    return proc_pred","metadata":{"execution":{"iopub.status.busy":"2023-09-01T17:05:34.625914Z","iopub.execute_input":"2023-09-01T17:05:34.627115Z","iopub.status.idle":"2023-09-01T17:05:34.635593Z","shell.execute_reply.started":"2023-09-01T17:05:34.62707Z","shell.execute_reply":"2023-09-01T17:05:34.633917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting unique patient IDs from test dataset\npatient_ids = test_df[\"patient_id\"].unique()\n\n# Initializing array to store predictions\npatient_preds = np.zeros(\n    shape=(len(patient_ids), 2*2 + 3*3),\n    dtype=\"float32\"\n)\n\n# Iterating over each patient\nfor pidx, patient_id in tqdm(enumerate(patient_ids), total=len(patient_ids), desc=\"Patients \"):\n    print(f\"Patient ID: {patient_id}\")\n    \n    # Query the dataframe for a particular patient\n    patient_df = test_df.query(\"patient_id == @patient_id\")\n    \n    # Getting image paths for a patient\n    patient_paths = patient_df.image_path.tolist()\n\n    # Building dataset for prediction\n    dtest = build_dataset(patient_paths)\n    \n    # Predicting with the model\n    pred = model.predict(dtest)\n    pred = np.concatenate(pred, axis=-1).astype(\"float32\")\n    pred = pred[:len(patient_paths), :]\n    pred = np.mean(pred.reshape(1, len(patient_paths), 11), axis=0)\n    pred = np.max(pred, axis=0, keepdims=True)\n    \n    patient_preds[pidx, :] += post_proc(pred)[0]\n    \n\n    # Deleting variables to free up memory \n    del patient_df, patient_paths, dtest, pred; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T17:05:36.554037Z","iopub.execute_input":"2023-09-01T17:05:36.55441Z","iopub.status.idle":"2023-09-01T17:05:39.480993Z","shell.execute_reply.started":"2023-09-01T17:05:36.55438Z","shell.execute_reply":"2023-09-01T17:05:39.479996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf {MODEL_PATH}","metadata":{"execution":{"iopub.status.busy":"2023-09-01T17:06:23.365671Z","iopub.execute_input":"2023-09-01T17:06:23.366081Z","iopub.status.idle":"2023-09-01T17:06:24.488821Z","shell.execute_reply.started":"2023-09-01T17:06:23.366046Z","shell.execute_reply":"2023-09-01T17:06:24.487226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Submission\npred_df = pd.DataFrame({\"patient_id\":patient_ids,})\npred_df[config.TARGET_COLS] = patient_preds.astype(\"float32\")\n\n# Align with sample submission\nsub_df = pd.read_csv(f\"{BASE_PATH}/sample_submission.csv\")\nsub_df = sub_df[[\"patient_id\"]]\nsub_df = sub_df.merge(pred_df, on=\"patient_id\", how=\"left\")\n\n# Store submission\nsub_df.to_csv(\"submission.csv\",index=False)\nsub_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T17:06:31.133543Z","iopub.execute_input":"2023-09-01T17:06:31.134078Z","iopub.status.idle":"2023-09-01T17:06:31.183224Z","shell.execute_reply.started":"2023-09-01T17:06:31.134031Z","shell.execute_reply":"2023-09-01T17:06:31.181929Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}