{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":52254,"databundleVersionId":9674523,"sourceType":"competition"},{"sourceId":6211844,"sourceType":"datasetVersion","datasetId":3567114}],"dockerImageVersionId":30554,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center>Multi-label classification of abdominal trauma from CT images</center>\n\n","metadata":{"papermill":{"duration":0.025483,"end_time":"2022-02-01T10:21:43.84374","exception":false,"start_time":"2022-02-01T10:21:43.818257","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## To commence this project, the neccesary libraries need to be installed.\n## We would be using the **TensorFlow** framework and the pretrained **EfficientNetB1**\n","metadata":{}},{"cell_type":"code","source":"# Imporitng libraries\nimport pandas as pd\nimport os\nimport random\nimport numpy as np\nimport tensorflow as tf\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom IPython.display import clear_output\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold\n\n\nrandom.seed(42)","metadata":{"id":"fUPKjZaaJHkM","papermill":{"duration":5.021373,"end_time":"2022-02-01T10:21:48.88737","exception":false,"start_time":"2022-02-01T10:21:43.865997","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-09T12:35:30.403775Z","iopub.execute_input":"2024-11-09T12:35:30.404199Z","iopub.status.idle":"2024-11-09T12:35:30.410736Z","shell.execute_reply.started":"2024-11-09T12:35:30.404167Z","shell.execute_reply":"2024-11-09T12:35:30.409728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading the training dataset\ntrain_img = \"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train_images\"","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:30.412712Z","iopub.execute_input":"2024-11-09T12:35:30.413216Z","iopub.status.idle":"2024-11-09T12:35:30.420422Z","shell.execute_reply.started":"2024-11-09T12:35:30.413177Z","shell.execute_reply":"2024-11-09T12:35:30.41936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in os.listdir(train_img):\n    for j in os.listdir(os.path.join(train_img,i)):\n        print(len(os.listdir(os.path.join(train_img,i,j))))","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-11-09T12:35:30.422387Z","iopub.execute_input":"2024-11-09T12:35:30.423183Z","iopub.status.idle":"2024-11-09T12:35:30.714879Z","shell.execute_reply.started":"2024-11-09T12:35:30.423145Z","shell.execute_reply":"2024-11-09T12:35:30.714025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install imbalanced-learn","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:30.716005Z","iopub.execute_input":"2024-11-09T12:35:30.716343Z","iopub.status.idle":"2024-11-09T12:35:42.90522Z","shell.execute_reply.started":"2024-11-09T12:35:30.716311Z","shell.execute_reply":"2024-11-09T12:35:42.903925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part A\n## Import and explore the data","metadata":{"papermill":{"duration":0.02193,"end_time":"2022-02-01T10:21:48.93356","exception":false,"start_time":"2022-02-01T10:21:48.91163","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Making a list containing all unique classes in the training set\n\n# Reading the training labels\ntraining_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")","metadata":{"id":"PBlH8ae3LbG0","papermill":{"duration":0.069238,"end_time":"2022-02-01T10:21:49.024476","exception":false,"start_time":"2022-02-01T10:21:48.955238","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-09T12:35:42.908953Z","iopub.execute_input":"2024-11-09T12:35:42.909302Z","iopub.status.idle":"2024-11-09T12:35:42.926913Z","shell.execute_reply.started":"2024-11-09T12:35:42.909272Z","shell.execute_reply":"2024-11-09T12:35:42.926115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Viewing the dataset in a structured format\ntraining_labels","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:42.928024Z","iopub.execute_input":"2024-11-09T12:35:42.928267Z","iopub.status.idle":"2024-11-09T12:35:42.947988Z","shell.execute_reply.started":"2024-11-09T12:35:42.928246Z","shell.execute_reply":"2024-11-09T12:35:42.947063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part B\n## In the next few cells, we chose to explore the dataframe to check for missing values","metadata":{}},{"cell_type":"code","source":"training_labels.columns","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:42.949206Z","iopub.execute_input":"2024-11-09T12:35:42.949468Z","iopub.status.idle":"2024-11-09T12:35:42.955909Z","shell.execute_reply.started":"2024-11-09T12:35:42.949444Z","shell.execute_reply":"2024-11-09T12:35:42.955023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['patient_id'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:42.95695Z","iopub.execute_input":"2024-11-09T12:35:42.95719Z","iopub.status.idle":"2024-11-09T12:35:42.966903Z","shell.execute_reply.started":"2024-11-09T12:35:42.957169Z","shell.execute_reply":"2024-11-09T12:35:42.965935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['bowel_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:42.968029Z","iopub.execute_input":"2024-11-09T12:35:42.968349Z","iopub.status.idle":"2024-11-09T12:35:42.979473Z","shell.execute_reply.started":"2024-11-09T12:35:42.968322Z","shell.execute_reply":"2024-11-09T12:35:42.978461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['bowel_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:42.983526Z","iopub.execute_input":"2024-11-09T12:35:42.983994Z","iopub.status.idle":"2024-11-09T12:35:42.992757Z","shell.execute_reply.started":"2024-11-09T12:35:42.983966Z","shell.execute_reply":"2024-11-09T12:35:42.991788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['extravasation_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:42.994066Z","iopub.execute_input":"2024-11-09T12:35:42.994415Z","iopub.status.idle":"2024-11-09T12:35:43.00331Z","shell.execute_reply.started":"2024-11-09T12:35:42.994387Z","shell.execute_reply":"2024-11-09T12:35:43.002255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['extravasation_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.004629Z","iopub.execute_input":"2024-11-09T12:35:43.005016Z","iopub.status.idle":"2024-11-09T12:35:43.015126Z","shell.execute_reply.started":"2024-11-09T12:35:43.004978Z","shell.execute_reply":"2024-11-09T12:35:43.014231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['kidney_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.016289Z","iopub.execute_input":"2024-11-09T12:35:43.016613Z","iopub.status.idle":"2024-11-09T12:35:43.026723Z","shell.execute_reply.started":"2024-11-09T12:35:43.016562Z","shell.execute_reply":"2024-11-09T12:35:43.025755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['kidney_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.028019Z","iopub.execute_input":"2024-11-09T12:35:43.028373Z","iopub.status.idle":"2024-11-09T12:35:43.039234Z","shell.execute_reply.started":"2024-11-09T12:35:43.028345Z","shell.execute_reply":"2024-11-09T12:35:43.038236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['kidney_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.040922Z","iopub.execute_input":"2024-11-09T12:35:43.041242Z","iopub.status.idle":"2024-11-09T12:35:43.050563Z","shell.execute_reply.started":"2024-11-09T12:35:43.041216Z","shell.execute_reply":"2024-11-09T12:35:43.049646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['liver_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.05168Z","iopub.execute_input":"2024-11-09T12:35:43.051961Z","iopub.status.idle":"2024-11-09T12:35:43.062085Z","shell.execute_reply.started":"2024-11-09T12:35:43.051938Z","shell.execute_reply":"2024-11-09T12:35:43.061143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['liver_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.063136Z","iopub.execute_input":"2024-11-09T12:35:43.063471Z","iopub.status.idle":"2024-11-09T12:35:43.074854Z","shell.execute_reply.started":"2024-11-09T12:35:43.063437Z","shell.execute_reply":"2024-11-09T12:35:43.073831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['liver_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.076269Z","iopub.execute_input":"2024-11-09T12:35:43.076979Z","iopub.status.idle":"2024-11-09T12:35:43.085528Z","shell.execute_reply.started":"2024-11-09T12:35:43.076946Z","shell.execute_reply":"2024-11-09T12:35:43.084556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['spleen_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.086769Z","iopub.execute_input":"2024-11-09T12:35:43.087641Z","iopub.status.idle":"2024-11-09T12:35:43.096184Z","shell.execute_reply.started":"2024-11-09T12:35:43.08758Z","shell.execute_reply":"2024-11-09T12:35:43.095233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['spleen_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.097211Z","iopub.execute_input":"2024-11-09T12:35:43.097481Z","iopub.status.idle":"2024-11-09T12:35:43.107495Z","shell.execute_reply.started":"2024-11-09T12:35:43.097457Z","shell.execute_reply":"2024-11-09T12:35:43.106521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['spleen_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.108736Z","iopub.execute_input":"2024-11-09T12:35:43.109101Z","iopub.status.idle":"2024-11-09T12:35:43.118318Z","shell.execute_reply.started":"2024-11-09T12:35:43.109065Z","shell.execute_reply":"2024-11-09T12:35:43.117428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels['any_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.120084Z","iopub.execute_input":"2024-11-09T12:35:43.12048Z","iopub.status.idle":"2024-11-09T12:35:43.129295Z","shell.execute_reply.started":"2024-11-09T12:35:43.120446Z","shell.execute_reply":"2024-11-09T12:35:43.128383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using the EfficientNetB1 model and tweaking the last two layers to suit our work\n#def create_model(decay_steps=10,warmup_steps=10):\n#    base_model = tf.keras.applications.EfficientNetB1(\n#    weights= \"imagenet\", include_top=False, input_shape= (512,512,3)\n#    )\n#    num_classes=61\n\n#    x = base_model.output\n#    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n#    x = tf.keras.layers.Dropout(0.2)(x)\n#    x_bowel = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_extra = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_liver = tf.keras.layers.Dense(32, activation='silu')(x)\n #   x_kidney = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_spleen = tf.keras.layers.Dense(32, activation='silu')(x)\n\n    # Define heads\n#    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel) # use sigmoid to convert predictions to [0-1]\n#    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra) # use sigmoid to convert predictions to [0-1]\n#    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver) # use softmax for the liver head\n#    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney) # use softmax for the kidney head\n#    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen) # use softmax for the spleen head\n\n#    model = tf.keras.Model(inputs = base_model.input, outputs = [out_bowel,out_extra,out_liver,out_kidney,out_spleen])\n        # Cosine Decay\n#    cosine_decay = tf.keras.optimizers.schedules.CosineDecay(\n#        initial_learning_rate=1e-4,\n#        decay_steps=decay_steps,\n#        alpha=0.0,\n        #warmup_target=1e-3,\n        #warmup_steps=warmup_steps,\n #   )\n\n    # Compile the model\n #   optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n #   loss = [\n #       tf.keras.losses.BinaryCrossentropy(),\n #       tf.keras.losses.BinaryCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy()]\n    \n #   metrics = [\n #       [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n #       [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n  #      [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n #       [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n #       [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]]\n    #    \"bowel\":[\"accuracy\"],\n    #    \"extra\":[\"accuracy\"],\n    #    \"liver\":[\"accuracy\"],\n    #    \"kidney\":[\"accuracy\"],\n    #    \"spleen\":[\"accuracy\"],\n    #}\n  #  print(\"[INFO] Compiling the model...\")\n #   model.compile(\n #       optimizer=optimizer,\n #     loss=loss,\n #     metrics=metrics\n #   )\n #   return model","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.130925Z","iopub.execute_input":"2024-11-09T12:35:43.131236Z","iopub.status.idle":"2024-11-09T12:35:43.144582Z","shell.execute_reply.started":"2024-11-09T12:35:43.131198Z","shell.execute_reply":"2024-11-09T12:35:43.143554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" + Sử dụng các optimizer tiên tiến như AdamW, RAdam, hoặc Lookahead:","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\n\n\ndef create_model(decay_steps=900, warmup_steps=60, fine_tune_from=100):\n    # Tải EfficientNetB1 làm base model\n    base_model = tf.keras.applications.EfficientNetB1(\n        weights=\"imagenet\", include_top=False, input_shape=(512, 512, 3)\n    )\n\n    # Fine-tune một phần của base model\n    for layer in base_model.layers[:fine_tune_from]:\n        layer.trainable = False\n\n    x = base_model.output\n\n    # Lớp Convolution với kernel 3x3 và tăng số lượng filters\n    x = tf.keras.layers.Conv2D(128, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n    \n    x = tf.keras.layers.Conv2D(256, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(512, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(1024, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    # Residual Block\n    residual = x\n    x = tf.keras.layers.Conv2D(1024, (2, 2), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Add()([x, residual])  # Skip connection\n\n    # Dense Block với dropout cao hơn và số lượng units lớn hơn\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dropout(0.6)(x)  # Tăng dropout\n\n    # Dense layers cho các đầu ra\n    x_bowel = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_extra = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_liver = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_kidney = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_spleen = tf.keras.layers.Dense(512, activation='relu')(x)\n\n    # Output Layers\n    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel)  # Bowel (binary)\n    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra)  # Extra (binary)\n    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver)  # Liver (multiclass)\n    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney)  # Kidney (multiclass)\n    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen)  # Spleen (multiclass)\n\n    # Create model\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n    \n    # Cosine Decay with a warmup phase\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecayRestarts(\n        initial_learning_rate=3e-4,  # Lower initial learning rate\n        first_decay_steps=warmup_steps,\n        t_mul=1.8,\n        m_mul=0.8,\n        alpha=0.2\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]\n    ]\n    \n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n        loss=loss,\n        metrics=metrics\n    )\n    \n    return model\n","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.145922Z","iopub.execute_input":"2024-11-09T12:35:43.14619Z","iopub.status.idle":"2024-11-09T12:35:43.169505Z","shell.execute_reply.started":"2024-11-09T12:35:43.146167Z","shell.execute_reply":"2024-11-09T12:35:43.168769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:43.175199Z","iopub.execute_input":"2024-11-09T12:35:43.176058Z","iopub.status.idle":"2024-11-09T12:35:46.420448Z","shell.execute_reply.started":"2024-11-09T12:35:43.176032Z","shell.execute_reply":"2024-11-09T12:35:46.419443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# In ra tóm tắt mô hình\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:46.421738Z","iopub.execute_input":"2024-11-09T12:35:46.422022Z","iopub.status.idle":"2024-11-09T12:35:47.271039Z","shell.execute_reply.started":"2024-11-09T12:35:46.421996Z","shell.execute_reply":"2024-11-09T12:35:47.270041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(\n    model,  # Sử dụng đúng tên biến của mô hình\n    to_file='model.png'\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:47.272336Z","iopub.execute_input":"2024-11-09T12:35:47.272715Z","iopub.status.idle":"2024-11-09T12:35:49.551199Z","shell.execute_reply.started":"2024-11-09T12:35:47.272679Z","shell.execute_reply":"2024-11-09T12:35:49.550235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels = training_labels.set_index('patient_id')\ntraining_labels.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:49.552799Z","iopub.execute_input":"2024-11-09T12:35:49.553098Z","iopub.status.idle":"2024-11-09T12:35:49.568533Z","shell.execute_reply.started":"2024-11-09T12:35:49.553071Z","shell.execute_reply":"2024-11-09T12:35:49.56757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part C\n## Balancing the imbalanced training dataset ","metadata":{}},{"cell_type":"code","source":"#Using histogram to get the distribution of the labels and to check if there are outliers and wrong labels\ntraining_labels.hist(figsize=(20,12),bins=2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:49.569783Z","iopub.execute_input":"2024-11-09T12:35:49.570137Z","iopub.status.idle":"2024-11-09T12:35:51.886157Z","shell.execute_reply.started":"2024-11-09T12:35:49.570107Z","shell.execute_reply":"2024-11-09T12:35:51.88512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#images ,labels","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:51.8874Z","iopub.execute_input":"2024-11-09T12:35:51.887708Z","iopub.status.idle":"2024-11-09T12:35:51.891811Z","shell.execute_reply.started":"2024-11-09T12:35:51.887682Z","shell.execute_reply":"2024-11-09T12:35:51.890818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img,i))[0]\n    file = os.listdir(os.path.join(train_img,i,folder))[0]\n    images.append(cv2.imread(os.path.join(train_img,i,folder,file), cv2.IMREAD_COLOR ))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:51.893158Z","iopub.execute_input":"2024-11-09T12:35:51.893484Z","iopub.status.idle":"2024-11-09T12:35:53.463899Z","shell.execute_reply.started":"2024-11-09T12:35:51.893452Z","shell.execute_reply":"2024-11-09T12:35:53.462977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Examples of images","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,12))\nfor i in range(10):\n    plt.subplot(2,5,i+1)\n    plt.imshow(images[i])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:53.465424Z","iopub.execute_input":"2024-11-09T12:35:53.465775Z","iopub.status.idle":"2024-11-09T12:35:55.747524Z","shell.execute_reply.started":"2024-11-09T12:35:53.465745Z","shell.execute_reply":"2024-11-09T12:35:55.746642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:55.748647Z","iopub.execute_input":"2024-11-09T12:35:55.748919Z","iopub.status.idle":"2024-11-09T12:35:55.753511Z","shell.execute_reply.started":"2024-11-09T12:35:55.748894Z","shell.execute_reply":"2024-11-09T12:35:55.752557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:55.754809Z","iopub.execute_input":"2024-11-09T12:35:55.755107Z","iopub.status.idle":"2024-11-09T12:35:55.820688Z","shell.execute_reply.started":"2024-11-09T12:35:55.755078Z","shell.execute_reply":"2024-11-09T12:35:55.819745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from imblearn.over_sampling import RandomOverSampler\n\n#oversampler = RandomOverSampler(random_state=42)\n#new_images, new_labels = oversampler.fit_resample(images, labels)","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:55.821917Z","iopub.execute_input":"2024-11-09T12:35:55.822184Z","iopub.status.idle":"2024-11-09T12:35:55.827016Z","shell.execute_reply.started":"2024-11-09T12:35:55.82216Z","shell.execute_reply":"2024-11-09T12:35:55.82595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Part D\n## Training the model with the training dataset","metadata":{}},{"cell_type":"code","source":"bowel_labels = labels[:,2]\nextravasation_labels = labels[:,4]\nkidney_labels = labels[:,4:7]\nliver_labels = labels[:,7:10]\nspleen_labels = labels[:,10:13]\nany_labels = labels[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:55.828464Z","iopub.execute_input":"2024-11-09T12:35:55.828751Z","iopub.status.idle":"2024-11-09T12:35:55.837154Z","shell.execute_reply.started":"2024-11-09T12:35:55.828727Z","shell.execute_reply":"2024-11-09T12:35:55.836366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bowel_val = y_val[:,2]\nextravasation_val = y_val[:,4]\nkidney_val = y_val[:,4:7]\nliver_val = y_val[:,7:10]\nspleen_val = y_val[:,10:13]\nany_val = y_val[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:55.838257Z","iopub.execute_input":"2024-11-09T12:35:55.838524Z","iopub.status.idle":"2024-11-09T12:35:55.847404Z","shell.execute_reply.started":"2024-11-09T12:35:55.8385Z","shell.execute_reply":"2024-11-09T12:35:55.846536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bowel_train = y_train[:,2]\nextravasation_train = y_train[:,4]\nkidney_train = y_train[:,4:7]\nliver_train = y_train[:,7:10]\nspleen_train = y_train[:,10:13]\nany_train = y_train[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:55.848514Z","iopub.execute_input":"2024-11-09T12:35:55.850427Z","iopub.status.idle":"2024-11-09T12:35:55.857223Z","shell.execute_reply.started":"2024-11-09T12:35:55.850402Z","shell.execute_reply":"2024-11-09T12:35:55.856302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#batch_size = 8\n#num_epoch = 2\n#history = model.fit(x=X_train,y=[bowel_train,extravasation_train,kidney_train,liver_train,spleen_train],batch_size=batch_size, epochs=num_epoch, verbose=1, validation_data=(X_val,[bowel_val,extravasation_val,kidney_val,liver_val,spleen_val]))\n\n\n#validation_data=(X_val,[bowel_val,extravasation_val,kidney_val,liver_val,spleen_val])","metadata":{"papermill":{"duration":1176.379334,"end_time":"2022-02-01T14:07:30.902597","exception":false,"start_time":"2022-02-01T13:47:54.523263","status":"completed"},"tags":[],"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-11-09T12:35:55.858346Z","iopub.execute_input":"2024-11-09T12:35:55.858644Z","iopub.status.idle":"2024-11-09T12:35:55.867372Z","shell.execute_reply.started":"2024-11-09T12:35:55.858611Z","shell.execute_reply":"2024-11-09T12:35:55.866505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nimport tensorflow as tf\n\n\n\ndef create_model(decay_steps=900, warmup_steps=60, fine_tune_from=100):\n    # Tải EfficientNetB1 làm base model\n    base_model = tf.keras.applications.EfficientNetB1(\n        weights=\"imagenet\", include_top=False, input_shape=(512, 512, 3)\n    )\n\n    # Fine-tune một phần của base model\n    for layer in base_model.layers[:fine_tune_from]:\n        layer.trainable = False\n\n    x = base_model.output\n\n    # Lớp Convolution với kernel 3x3 và tăng số lượng filters\n    x = tf.keras.layers.Conv2D(128, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n    \n    x = tf.keras.layers.Conv2D(256, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(512, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(1024, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    # Residual Block\n    residual = x\n    x = tf.keras.layers.Conv2D(1024, (2, 2), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Add()([x, residual])  # Skip connection\n\n    # Dense Block với dropout cao hơn và số lượng units lớn hơn\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dropout(0.6)(x)  # Tăng dropout\n\n    # Dense layers cho các đầu ra\n    x_bowel = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_extra = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_liver = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_kidney = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_spleen = tf.keras.layers.Dense(512, activation='relu')(x)\n\n    # Output Layers\n    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel)  # Bowel (binary)\n    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra)  # Extra (binary)\n    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver)  # Liver (multiclass)\n    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney)  # Kidney (multiclass)\n    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen)  # Spleen (multiclass)\n\n    # Create model\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n    \n    # Cosine Decay with a warmup phase\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecayRestarts(\n        initial_learning_rate=3e-4,  # Lower initial learning rate\n        first_decay_steps=warmup_steps,\n        t_mul=1.8,\n        m_mul=0.8,\n        alpha=0.2\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]\n    ]\n    \n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n        loss=loss,\n        metrics=metrics\n    )\n    \n    return model\n\n\n# Load images and labels\nimages = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img, i))[0]\n    file = os.listdir(os.path.join(train_img, i, folder))[0]\n    images.append(cv2.imread(os.path.join(train_img, i, folder, file), cv2.IMREAD_COLOR))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)\n\n# Split data\nX_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)\n\n# Prepare labels for each class\nbowel_labels = labels[:, 2]\nextravasation_labels = labels[:, 4]\nkidney_labels = labels[:, 4:7]\nliver_labels = labels[:, 7:10]\nspleen_labels = labels[:, 10:13]\nany_labels = labels[:, -1]\n\nbowel_val = y_val[:, 2]\nextravasation_val = y_val[:, 4]\nkidney_val = y_val[:, 4:7]\nliver_val = y_val[:, 7:10]\nspleen_val = y_val[:, 10:13]\nany_val = y_val[:, -1]\n\n# KFold Cross Validation\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Store results for each fold\nfold_results = []\n\n# Data Augmentation\ntrain_datagen = ImageDataGenerator(\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# Validation data generator (no augmentation)\nval_datagen = ImageDataGenerator()\n\nfor fold, (train_index, val_index) in enumerate(kf.split(images, any_labels)):\n    print(f\"Fold {fold}\")  # Changed to print fold number from 0 to 4\n\n    # Split data into train and validation according to fold\n    X_train_fold, X_val_fold = images[train_index], images[val_index]\n    y_train_fold, y_val_fold = labels[train_index], labels[val_index]\n\n    # Prepare labels for each class for the fold\n    bowel_train = y_train_fold[:, 2]\n    extravasation_train = y_train_fold[:, 4]\n    kidney_train = y_train_fold[:, 4:7]\n    liver_train = y_train_fold[:, 7:10]\n    spleen_train = y_train_fold[:, 10:13]\n\n    bowel_val = y_val_fold[:, 2]\n    extravasation_val = y_val_fold[:, 4]\n    kidney_val = y_val_fold[:, 4:7]\n    liver_val = y_val_fold[:, 7:10]\n    spleen_val = y_val_fold[:, 10:13]\n\n    # Create a new model for each fold\n    model = create_model()\n\n    # EarlyStopping and ModelCheckpoint\n    early_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n    checkpoint = ModelCheckpoint(f'best_model_fold_{fold}.h5', save_best_only=True, monitor='val_loss')\n\n    # Train the model\n    batch_size = 16\n    num_epoch = 40  # You can modify this value based on the training time and model convergence\n    \n    history = model.fit(\n        x=X_train_fold,\n        y=[bowel_train, extravasation_train, kidney_train, liver_train, spleen_train],\n        batch_size=batch_size,\n        epochs=num_epoch,\n        verbose=1,\n         validation_data=(X_val_fold, [bowel_val, extravasation_val, kidney_val, liver_val, spleen_val])\n    )\n\n    # Store results for this fold\n    fold_results.append(history.history)\n\n# Check results\nfor fold, result in enumerate(fold_results):\n    print(f\"Fold {fold} Results:\")  # Changed to print fold number from 0 to 4\n    for key in result.keys():\n        print(f\"{key}: {result[key][-1]}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-09T12:35:55.869783Z","iopub.execute_input":"2024-11-09T12:35:55.870083Z","iopub.status.idle":"2024-11-09T13:04:40.532162Z","shell.execute_reply.started":"2024-11-09T12:35:55.870045Z","shell.execute_reply":"2024-11-09T13:04:40.531272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history.history.keys()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T13:04:40.533639Z","iopub.execute_input":"2024-11-09T13:04:40.534005Z","iopub.status.idle":"2024-11-09T13:04:40.540701Z","shell.execute_reply.started":"2024-11-09T13:04:40.533971Z","shell.execute_reply":"2024-11-09T13:04:40.539618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_acc = ['val_bowel_bowel_binary_accuracy', 'val_extra_extra_binary_accuracy', 'val_liver_liver_cat_accuracy', 'val_kidney_kidney_cat_accuracy', 'val_spleen_spleen_cat_accuracy']","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-11-09T13:04:40.541967Z","iopub.execute_input":"2024-11-09T13:04:40.542381Z","iopub.status.idle":"2024-11-09T13:04:40.548859Z","shell.execute_reply.started":"2024-11-09T13:04:40.542348Z","shell.execute_reply":"2024-11-09T13:04:40.548093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,j in enumerate(val_acc):\n    print(j, np.asarray(history.history[val_acc[i]])[-1].round(2))\n#np.asarray(history.history['val_bowel_bowel_binary_accuracy'])[-1].round(2)","metadata":{"execution":{"iopub.status.busy":"2024-11-09T13:04:40.549782Z","iopub.execute_input":"2024-11-09T13:04:40.550012Z","iopub.status.idle":"2024-11-09T13:04:40.561554Z","shell.execute_reply.started":"2024-11-09T13:04:40.549992Z","shell.execute_reply":"2024-11-09T13:04:40.560639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"_loss\") and not i ==\"val_loss\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T13:04:40.562793Z","iopub.execute_input":"2024-11-09T13:04:40.563052Z","iopub.status.idle":"2024-11-09T13:04:40.924497Z","shell.execute_reply.started":"2024-11-09T13:04:40.56303Z","shell.execute_reply":"2024-11-09T13:04:40.923637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"accuracy\") and not i ==\"val_accuracy\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T13:04:40.925716Z","iopub.execute_input":"2024-11-09T13:04:40.926043Z","iopub.status.idle":"2024-11-09T13:04:41.232715Z","shell.execute_reply.started":"2024-11-09T13:04:40.926012Z","shell.execute_reply":"2024-11-09T13:04:41.23181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(history.history[\"loss\"], label=\"loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-09T13:04:41.233885Z","iopub.execute_input":"2024-11-09T13:04:41.234195Z","iopub.status.idle":"2024-11-09T13:04:41.503813Z","shell.execute_reply.started":"2024-11-09T13:04:41.234167Z","shell.execute_reply":"2024-11-09T13:04:41.502865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}