{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":52254,"databundleVersionId":9674523,"sourceType":"competition"},{"sourceId":6211844,"sourceType":"datasetVersion","datasetId":3567114}],"dockerImageVersionId":30554,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center>Multi-label classification of abdominal trauma from CT images</center>\n\n","metadata":{"papermill":{"duration":0.025483,"end_time":"2022-02-01T10:21:43.84374","exception":false,"start_time":"2022-02-01T10:21:43.818257","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## To commence this project, the neccesary libraries need to be installed.\n## We would be using the **TensorFlow** framework and the pretrained **EfficientNetB1**\n","metadata":{}},{"cell_type":"code","source":"# Imporitng libraries\nimport pandas as pd\nimport os\nimport random\nimport numpy as np\nimport tensorflow as tf\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom IPython.display import clear_output\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold\n\n\nrandom.seed(42)","metadata":{"id":"fUPKjZaaJHkM","papermill":{"duration":5.021373,"end_time":"2022-02-01T10:21:48.88737","exception":false,"start_time":"2022-02-01T10:21:43.865997","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-14T13:49:33.021235Z","iopub.execute_input":"2024-11-14T13:49:33.021539Z","iopub.status.idle":"2024-11-14T13:49:43.363211Z","shell.execute_reply.started":"2024-11-14T13:49:33.021513Z","shell.execute_reply":"2024-11-14T13:49:43.362115Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Loading the training dataset\ntrain_img = \"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train_images\"","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:43.366972Z","iopub.execute_input":"2024-11-14T13:49:43.367509Z","iopub.status.idle":"2024-11-14T13:49:43.371871Z","shell.execute_reply.started":"2024-11-14T13:49:43.367473Z","shell.execute_reply":"2024-11-14T13:49:43.370781Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in os.listdir(train_img):\n    for j in os.listdir(os.path.join(train_img,i)):\n        print(len(os.listdir(os.path.join(train_img,i,j))))","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-11-14T13:49:43.373279Z","iopub.execute_input":"2024-11-14T13:49:43.373612Z","iopub.status.idle":"2024-11-14T13:49:47.02283Z","shell.execute_reply.started":"2024-11-14T13:49:43.373587Z","shell.execute_reply":"2024-11-14T13:49:47.021697Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install imbalanced-learn","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:47.025008Z","iopub.execute_input":"2024-11-14T13:49:47.025302Z","iopub.status.idle":"2024-11-14T13:49:59.693095Z","shell.execute_reply.started":"2024-11-14T13:49:47.025276Z","shell.execute_reply":"2024-11-14T13:49:59.692005Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part A\n## Import and explore the data","metadata":{"papermill":{"duration":0.02193,"end_time":"2022-02-01T10:21:48.93356","exception":false,"start_time":"2022-02-01T10:21:48.91163","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Making a list containing all unique classes in the training set\n\n# Reading the training labels\ntraining_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")","metadata":{"id":"PBlH8ae3LbG0","papermill":{"duration":0.069238,"end_time":"2022-02-01T10:21:49.024476","exception":false,"start_time":"2022-02-01T10:21:48.955238","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-14T13:49:59.694452Z","iopub.execute_input":"2024-11-14T13:49:59.694821Z","iopub.status.idle":"2024-11-14T13:49:59.71777Z","shell.execute_reply.started":"2024-11-14T13:49:59.694789Z","shell.execute_reply":"2024-11-14T13:49:59.716899Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Viewing the dataset in a structured format\ntraining_labels","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.719012Z","iopub.execute_input":"2024-11-14T13:49:59.719593Z","iopub.status.idle":"2024-11-14T13:49:59.746652Z","shell.execute_reply.started":"2024-11-14T13:49:59.719567Z","shell.execute_reply":"2024-11-14T13:49:59.745835Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part B\n## In the next few cells, we chose to explore the dataframe to check for missing values","metadata":{}},{"cell_type":"code","source":"training_labels.columns","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.747765Z","iopub.execute_input":"2024-11-14T13:49:59.748086Z","iopub.status.idle":"2024-11-14T13:49:59.754375Z","shell.execute_reply.started":"2024-11-14T13:49:59.748054Z","shell.execute_reply":"2024-11-14T13:49:59.753467Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['patient_id'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.755637Z","iopub.execute_input":"2024-11-14T13:49:59.755924Z","iopub.status.idle":"2024-11-14T13:49:59.769486Z","shell.execute_reply.started":"2024-11-14T13:49:59.755898Z","shell.execute_reply":"2024-11-14T13:49:59.76864Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['bowel_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.770695Z","iopub.execute_input":"2024-11-14T13:49:59.77126Z","iopub.status.idle":"2024-11-14T13:49:59.778474Z","shell.execute_reply.started":"2024-11-14T13:49:59.771228Z","shell.execute_reply":"2024-11-14T13:49:59.777567Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['bowel_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.782678Z","iopub.execute_input":"2024-11-14T13:49:59.78306Z","iopub.status.idle":"2024-11-14T13:49:59.789937Z","shell.execute_reply.started":"2024-11-14T13:49:59.783036Z","shell.execute_reply":"2024-11-14T13:49:59.789048Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['extravasation_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.791005Z","iopub.execute_input":"2024-11-14T13:49:59.791311Z","iopub.status.idle":"2024-11-14T13:49:59.800797Z","shell.execute_reply.started":"2024-11-14T13:49:59.791288Z","shell.execute_reply":"2024-11-14T13:49:59.7999Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['extravasation_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.801669Z","iopub.execute_input":"2024-11-14T13:49:59.801947Z","iopub.status.idle":"2024-11-14T13:49:59.813274Z","shell.execute_reply.started":"2024-11-14T13:49:59.801924Z","shell.execute_reply":"2024-11-14T13:49:59.81251Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.814425Z","iopub.execute_input":"2024-11-14T13:49:59.814718Z","iopub.status.idle":"2024-11-14T13:49:59.823362Z","shell.execute_reply.started":"2024-11-14T13:49:59.814684Z","shell.execute_reply":"2024-11-14T13:49:59.822407Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.824585Z","iopub.execute_input":"2024-11-14T13:49:59.824905Z","iopub.status.idle":"2024-11-14T13:49:59.834451Z","shell.execute_reply.started":"2024-11-14T13:49:59.824869Z","shell.execute_reply":"2024-11-14T13:49:59.83359Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.835558Z","iopub.execute_input":"2024-11-14T13:49:59.835826Z","iopub.status.idle":"2024-11-14T13:49:59.846334Z","shell.execute_reply.started":"2024-11-14T13:49:59.835804Z","shell.execute_reply":"2024-11-14T13:49:59.845598Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.84749Z","iopub.execute_input":"2024-11-14T13:49:59.847972Z","iopub.status.idle":"2024-11-14T13:49:59.857858Z","shell.execute_reply.started":"2024-11-14T13:49:59.847941Z","shell.execute_reply":"2024-11-14T13:49:59.856925Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.859058Z","iopub.execute_input":"2024-11-14T13:49:59.859549Z","iopub.status.idle":"2024-11-14T13:49:59.868347Z","shell.execute_reply.started":"2024-11-14T13:49:59.859518Z","shell.execute_reply":"2024-11-14T13:49:59.867486Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.869375Z","iopub.execute_input":"2024-11-14T13:49:59.869626Z","iopub.status.idle":"2024-11-14T13:49:59.881912Z","shell.execute_reply.started":"2024-11-14T13:49:59.869603Z","shell.execute_reply":"2024-11-14T13:49:59.881178Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_healthy'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.882825Z","iopub.execute_input":"2024-11-14T13:49:59.883047Z","iopub.status.idle":"2024-11-14T13:49:59.896448Z","shell.execute_reply.started":"2024-11-14T13:49:59.883026Z","shell.execute_reply":"2024-11-14T13:49:59.895642Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_low'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.897488Z","iopub.execute_input":"2024-11-14T13:49:59.897762Z","iopub.status.idle":"2024-11-14T13:49:59.908723Z","shell.execute_reply.started":"2024-11-14T13:49:59.89772Z","shell.execute_reply":"2024-11-14T13:49:59.907775Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_high'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.909815Z","iopub.execute_input":"2024-11-14T13:49:59.910079Z","iopub.status.idle":"2024-11-14T13:49:59.919247Z","shell.execute_reply.started":"2024-11-14T13:49:59.910056Z","shell.execute_reply":"2024-11-14T13:49:59.918364Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['any_injury'].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.920286Z","iopub.execute_input":"2024-11-14T13:49:59.920537Z","iopub.status.idle":"2024-11-14T13:49:59.9297Z","shell.execute_reply.started":"2024-11-14T13:49:59.920515Z","shell.execute_reply":"2024-11-14T13:49:59.928937Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Using the EfficientNetB1 model and tweaking the last two layers to suit our work\n#def create_model(decay_steps=10,warmup_steps=10):\n#    base_model = tf.keras.applications.EfficientNetB1(\n#    weights= \"imagenet\", include_top=False, input_shape= (512,512,3)\n#    )\n#    num_classes=61\n\n#    x = base_model.output\n#    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n#    x = tf.keras.layers.Dropout(0.2)(x)\n#    x_bowel = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_extra = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_liver = tf.keras.layers.Dense(32, activation='silu')(x)\n #   x_kidney = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_spleen = tf.keras.layers.Dense(32, activation='silu')(x)\n\n    # Define heads\n#    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel) # use sigmoid to convert predictions to [0-1]\n#    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra) # use sigmoid to convert predictions to [0-1]\n#    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver) # use softmax for the liver head\n#    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney) # use softmax for the kidney head\n#    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen) # use softmax for the spleen head\n\n#    model = tf.keras.Model(inputs = base_model.input, outputs = [out_bowel,out_extra,out_liver,out_kidney,out_spleen])\n        # Cosine Decay\n#    cosine_decay = tf.keras.optimizers.schedules.CosineDecay(\n#        initial_learning_rate=1e-4,\n#        decay_steps=decay_steps,\n#        alpha=0.0,\n        #warmup_target=1e-3,\n        #warmup_steps=warmup_steps,\n #   )\n\n    # Compile the model\n #   optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n #   loss = [\n #       tf.keras.losses.BinaryCrossentropy(),\n #       tf.keras.losses.BinaryCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy()]\n    \n #   metrics = [\n #       [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n #       [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n  #      [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n #       [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n #       [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]]\n    #    \"bowel\":[\"accuracy\"],\n    #    \"extra\":[\"accuracy\"],\n    #    \"liver\":[\"accuracy\"],\n    #    \"kidney\":[\"accuracy\"],\n    #    \"spleen\":[\"accuracy\"],\n    #}\n  #  print(\"[INFO] Compiling the model...\")\n #   model.compile(\n #       optimizer=optimizer,\n #     loss=loss,\n #     metrics=metrics\n #   )\n #   return model","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.93102Z","iopub.execute_input":"2024-11-14T13:49:59.931282Z","iopub.status.idle":"2024-11-14T13:49:59.939877Z","shell.execute_reply.started":"2024-11-14T13:49:59.93126Z","shell.execute_reply":"2024-11-14T13:49:59.939076Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"+ Sử dụng Gradient Clipping để tránh gradient quá lớn:","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\ndef create_model(decay_steps=900, warmup_steps=60, fine_tune_from=100):\n    # Tải EfficientNetB1 làm base model\n    base_model = tf.keras.applications.EfficientNetB1(\n        weights=\"imagenet\", include_top=False, input_shape=(512, 512, 3)\n    )\n\n    # Fine-tune một phần của base model\n    for layer in base_model.layers[:fine_tune_from]:\n        layer.trainable = False\n\n    x = base_model.output\n\n    # Lớp Convolution với kernel 3x3 và tăng số lượng filters\n    x = tf.keras.layers.Conv2D(128, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n    \n    x = tf.keras.layers.Conv2D(256, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(512, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(1024, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    # Residual Block\n    residual = x\n    x = tf.keras.layers.Conv2D(1024, (2, 2), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Add()([x, residual])  # Skip connection\n\n    # Global Average Pooling để giảm số chiều\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n\n    # Dense Block với dropout vừa phải\n    x = tf.keras.layers.Dropout(0.4)(x)  # Giảm tỷ lệ dropout\n\n    # Dense layers cho các đầu ra\n    x_bowel = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_extra = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_liver = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_kidney = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_spleen = tf.keras.layers.Dense(512, activation='relu')(x)\n\n    # Output Layers\n    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel)  # Bowel (binary)\n    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra)  # Extra (binary)\n    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver)  # Liver (multiclass)\n    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney)  # Kidney (multiclass)\n    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen)  # Spleen (multiclass)\n\n    # Create model\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n    \n    # Cosine Decay with a warmup phase\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecayRestarts(\n        initial_learning_rate=3e-4,  # Lower initial learning rate\n        first_decay_steps=warmup_steps,\n        t_mul=1.8,\n        m_mul=0.8,\n        alpha=0.2\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]\n    ]\n    \n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n        loss=loss,\n        metrics=metrics\n    )\n    \n    return model\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.940926Z","iopub.execute_input":"2024-11-14T13:49:59.941202Z","iopub.status.idle":"2024-11-14T13:49:59.962594Z","shell.execute_reply.started":"2024-11-14T13:49:59.94117Z","shell.execute_reply":"2024-11-14T13:49:59.961774Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#model = create_model()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.96392Z","iopub.execute_input":"2024-11-14T13:49:59.964437Z","iopub.status.idle":"2024-11-14T13:49:59.974537Z","shell.execute_reply.started":"2024-11-14T13:49:59.964405Z","shell.execute_reply":"2024-11-14T13:49:59.9737Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = create_model()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:49:59.975437Z","iopub.execute_input":"2024-11-14T13:49:59.975704Z","iopub.status.idle":"2024-11-14T13:50:05.163593Z","shell.execute_reply.started":"2024-11-14T13:49:59.975681Z","shell.execute_reply":"2024-11-14T13:50:05.162582Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# In ra tóm tắt mô hình\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:05.166777Z","iopub.execute_input":"2024-11-14T13:50:05.167099Z","iopub.status.idle":"2024-11-14T13:50:06.023982Z","shell.execute_reply.started":"2024-11-14T13:50:05.167071Z","shell.execute_reply":"2024-11-14T13:50:06.023091Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.utils.plot_model(\n    model,  # Sử dụng đúng tên biến của mô hình\n    to_file='model.png'\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:06.034249Z","iopub.execute_input":"2024-11-14T13:50:06.034581Z","iopub.status.idle":"2024-11-14T13:50:08.559852Z","shell.execute_reply.started":"2024-11-14T13:50:06.034556Z","shell.execute_reply":"2024-11-14T13:50:08.55858Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels = training_labels.set_index('patient_id')\ntraining_labels.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:08.561191Z","iopub.execute_input":"2024-11-14T13:50:08.561531Z","iopub.status.idle":"2024-11-14T13:50:08.579311Z","shell.execute_reply.started":"2024-11-14T13:50:08.5615Z","shell.execute_reply":"2024-11-14T13:50:08.578362Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part C\n## Balancing the imbalanced training dataset ","metadata":{}},{"cell_type":"code","source":"#Using histogram to get the distribution of the labels and to check if there are outliers and wrong labels\ntraining_labels.hist(figsize=(20,12),bins=2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:08.580463Z","iopub.execute_input":"2024-11-14T13:50:08.581143Z","iopub.status.idle":"2024-11-14T13:50:10.962865Z","shell.execute_reply.started":"2024-11-14T13:50:08.581109Z","shell.execute_reply":"2024-11-14T13:50:10.961844Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#images ,labels","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:10.964182Z","iopub.execute_input":"2024-11-14T13:50:10.964482Z","iopub.status.idle":"2024-11-14T13:50:10.968634Z","shell.execute_reply.started":"2024-11-14T13:50:10.964455Z","shell.execute_reply":"2024-11-14T13:50:10.967718Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img,i))[0]\n    file = os.listdir(os.path.join(train_img,i,folder))[0]\n    images.append(cv2.imread(os.path.join(train_img,i,folder,file), cv2.IMREAD_COLOR ))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:10.970124Z","iopub.execute_input":"2024-11-14T13:50:10.970451Z","iopub.status.idle":"2024-11-14T13:50:13.907728Z","shell.execute_reply.started":"2024-11-14T13:50:10.970421Z","shell.execute_reply":"2024-11-14T13:50:13.906698Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Examples of images","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,12))\nfor i in range(10):\n    plt.subplot(2,5,i+1)\n    plt.imshow(images[i])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:13.909018Z","iopub.execute_input":"2024-11-14T13:50:13.909365Z","iopub.status.idle":"2024-11-14T13:50:16.187221Z","shell.execute_reply.started":"2024-11-14T13:50:13.909333Z","shell.execute_reply":"2024-11-14T13:50:16.186281Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:16.188653Z","iopub.execute_input":"2024-11-14T13:50:16.189006Z","iopub.status.idle":"2024-11-14T13:50:16.1938Z","shell.execute_reply.started":"2024-11-14T13:50:16.188976Z","shell.execute_reply":"2024-11-14T13:50:16.192691Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:16.195124Z","iopub.execute_input":"2024-11-14T13:50:16.195405Z","iopub.status.idle":"2024-11-14T13:50:16.29481Z","shell.execute_reply.started":"2024-11-14T13:50:16.19538Z","shell.execute_reply":"2024-11-14T13:50:16.293952Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.over_sampling import RandomOverSampler\n\n#oversampler = RandomOverSampler(random_state=42)\n#new_images, new_labels = oversampler.fit_resample(images, labels)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:16.295959Z","iopub.execute_input":"2024-11-14T13:50:16.296309Z","iopub.status.idle":"2024-11-14T13:50:16.854757Z","shell.execute_reply.started":"2024-11-14T13:50:16.296275Z","shell.execute_reply":"2024-11-14T13:50:16.853949Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part D\n## Training the model with the training dataset","metadata":{}},{"cell_type":"code","source":"bowel_labels = labels[:,2]\nextravasation_labels = labels[:,4]\nkidney_labels = labels[:,4:7]\nliver_labels = labels[:,7:10]\nspleen_labels = labels[:,10:13]\nany_labels = labels[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:16.855802Z","iopub.execute_input":"2024-11-14T13:50:16.856766Z","iopub.status.idle":"2024-11-14T13:50:16.862083Z","shell.execute_reply.started":"2024-11-14T13:50:16.856713Z","shell.execute_reply":"2024-11-14T13:50:16.861203Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bowel_val = y_val[:,2]\nextravasation_val = y_val[:,4]\nkidney_val = y_val[:,4:7]\nliver_val = y_val[:,7:10]\nspleen_val = y_val[:,10:13]\nany_val = y_val[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:16.863543Z","iopub.execute_input":"2024-11-14T13:50:16.863847Z","iopub.status.idle":"2024-11-14T13:50:16.87528Z","shell.execute_reply.started":"2024-11-14T13:50:16.863821Z","shell.execute_reply":"2024-11-14T13:50:16.874522Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bowel_train = y_train[:,2]\nextravasation_train = y_train[:,4]\nkidney_train = y_train[:,4:7]\nliver_train = y_train[:,7:10]\nspleen_train = y_train[:,10:13]\nany_train = y_train[:,-1]","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:16.876512Z","iopub.execute_input":"2024-11-14T13:50:16.876873Z","iopub.status.idle":"2024-11-14T13:50:16.885407Z","shell.execute_reply.started":"2024-11-14T13:50:16.876839Z","shell.execute_reply":"2024-11-14T13:50:16.884587Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#batch_size = 8\n#num_epoch = 2\n#history = model.fit(x=X_train,y=[bowel_train,extravasation_train,kidney_train,liver_train,spleen_train],batch_size=batch_size, epochs=num_epoch, verbose=1, validation_data=(X_val,[bowel_val,extravasation_val,kidney_val,liver_val,spleen_val]))\n\n\n#validation_data=(X_val,[bowel_val,extravasation_val,kidney_val,liver_val,spleen_val])","metadata":{"papermill":{"duration":1176.379334,"end_time":"2022-02-01T14:07:30.902597","exception":false,"start_time":"2022-02-01T13:47:54.523263","status":"completed"},"tags":[],"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-11-14T13:50:16.886558Z","iopub.execute_input":"2024-11-14T13:50:16.886845Z","iopub.status.idle":"2024-11-14T13:50:16.894363Z","shell.execute_reply.started":"2024-11-14T13:50:16.886821Z","shell.execute_reply":"2024-11-14T13:50:16.893461Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install keras-rectified-adam\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:16.895446Z","iopub.execute_input":"2024-11-14T13:50:16.895717Z","iopub.status.idle":"2024-11-14T13:50:30.854871Z","shell.execute_reply.started":"2024-11-14T13:50:16.895685Z","shell.execute_reply":"2024-11-14T13:50:30.853693Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport numpy as np\nimport os\nimport cv2\n\n# Hàm tính F1 Score tùy chỉnh\ndef f1_score_metric(y_true, y_pred):\n    true_positives = tf.reduce_sum(tf.cast(tf.logical_and(tf.equal(y_true, 1), tf.equal(y_pred, 1)), tf.float32))\n    false_positives = tf.reduce_sum(tf.cast(tf.logical_and(tf.equal(y_true, 0), tf.equal(y_pred, 1)), tf.float32))\n    false_negatives = tf.reduce_sum(tf.cast(tf.logical_and(tf.equal(y_true, 1), tf.equal(y_pred, 0)), tf.float32))\n    \n    precision = true_positives / (true_positives + false_positives + tf.keras.backend.epsilon())\n    recall = true_positives / (true_positives + false_negatives + tf.keras.backend.epsilon())\n    \n    # Tính F1 score\n    f1 = 2 * (precision * recall) / (precision + recall + tf.keras.backend.epsilon())\n    return f1\n\n# Tạo mô hình với các lớp như EfficientNetB1, Conv2D và các lớp FC\ndef create_model(decay_steps=900, warmup_steps=60, fine_tune_from=100):\n    # Tải EfficientNetB1 làm base model\n    base_model = tf.keras.applications.EfficientNetB1(\n        weights=\"imagenet\", include_top=False, input_shape=(512, 512, 3)\n    )\n\n    # Fine-tune một phần của base model\n    for layer in base_model.layers[:fine_tune_from]:\n        layer.trainable = False\n\n    x = base_model.output\n\n    # Các lớp Convolution\n    x = tf.keras.layers.Conv2D(128, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n    \n    x = tf.keras.layers.Conv2D(256, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(512, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(1024, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    # Residual Block\n    residual = x\n    x = tf.keras.layers.Conv2D(1024, (2, 2), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Add()([x, residual])  # Skip connection\n\n    # Global Average Pooling\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n\n    # Dense layer với dropout\n    x = tf.keras.layers.Dropout(0.4)(x)\n\n    # Các lớp Dense cho các đầu ra\n    x_bowel = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_extra = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_liver = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_kidney = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_spleen = tf.keras.layers.Dense(512, activation='relu')(x)\n\n    # Output Layers\n    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel)\n    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra)\n    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver)\n    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney)\n    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen)\n\n    # Tạo mô hình\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n    \n    # Cosine Decay Learning Rate\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecayRestarts(\n        initial_learning_rate=3e-4,\n        first_decay_steps=warmup_steps,\n        t_mul=1.8,\n        m_mul=0.8,\n        alpha=0.2\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        # Metrics for Bowel (binary)\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"bowel_auc\"),\n         tf.keras.metrics.Recall(name=\"bowel_recall\"),\n         tf.keras.metrics.Precision(name=\"bowel_precision\"),  # Thêm Precision\n         tf.keras.metrics.TruePositives(name=\"bowel_tp\"),\n         tf.keras.metrics.FalseNegatives(name=\"bowel_fn\"),\n         tf.keras.metrics.FalsePositives(name=\"bowel_fp\"),\n         tf.keras.metrics.TrueNegatives(name=\"bowel_tn\")],\n        \n        # Metrics for Extravasation (binary)\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"extra_auc\"),\n         tf.keras.metrics.Recall(name=\"extra_recall\"),\n         tf.keras.metrics.Precision(name=\"extra_precision\"),  # Thêm Precision\n         tf.keras.metrics.TruePositives(name=\"extra_tp\"),\n         tf.keras.metrics.FalseNegatives(name=\"extra_fn\"),\n         tf.keras.metrics.FalsePositives(name=\"extra_fp\"),\n         tf.keras.metrics.TrueNegatives(name=\"extra_tn\")],\n        \n        # Metrics for Liver (multiclass)\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"liver_auc\"),\n         tf.keras.metrics.Recall(name=\"liver_recall\"),\n         tf.keras.metrics.Precision(name=\"liver_precision\")],  # Thêm Precision\n        \n        # Metrics for Kidney (multiclass)\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"kidney_auc\"),\n         tf.keras.metrics.Recall(name=\"kidney_recall\"),\n         tf.keras.metrics.Precision(name=\"kidney_precision\")],  # Thêm Precision\n        \n        # Metrics for Spleen (multiclass)\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\"),\n         f1_score_metric,  # Sử dụng F1 Score tùy chỉnh\n         tf.keras.metrics.AUC(name=\"spleen_auc\"),\n         tf.keras.metrics.Recall(name=\"spleen_recall\"),\n         tf.keras.metrics.Precision(name=\"spleen_precision\")]  # Thêm Precision\n    ]\n    \n    model.compile(optimizer=optimizer, loss=loss, metrics=metrics)\n    \n    return model\n\n\n# Load images and labels (dữ liệu và nhãn đã chuẩn bị)\nimages = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img, i))[0]\n    file = os.listdir(os.path.join(train_img, i, folder))[0]\n    images.append(cv2.imread(os.path.join(train_img, i, folder, file), cv2.IMREAD_COLOR))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)\n\n# Split data\nX_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)\n\n# Prepare labels for each class\nbowel_labels = labels[:, 2]\nextravasation_labels = labels[:, 4]\nkidney_labels = labels[:, 4:7]\nliver_labels = labels[:, 7:10]\nspleen_labels = labels[:, 10:13]\nany_labels = labels[:, -1]\n\nbowel_val = y_val[:, 2]\nextravasation_val = y_val[:, 4]\nkidney_val = y_val[:, 4:7]\nliver_val = y_val[:, 7:10]\nspleen_val = y_val[:, 10:13]\nany_val = y_val[:, -1]\n\n# KFold Cross Validation\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Store results for each fold\nfold_results = []\n\n# Data Augmentation\ntrain_datagen = ImageDataGenerator(\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# Validation data generator (no augmentation)\nval_datagen = ImageDataGenerator()\n\nfor fold, (train_index, val_index) in enumerate(kf.split(images, any_labels)):\n    print(f\"Fold {fold}\")  # Changed to print fold number from 0 to 4\n\n    # Split data into train and validation according to fold\n    X_train_fold, X_val_fold = images[train_index], images[val_index]\n    y_train_fold, y_val_fold = labels[train_index], labels[val_index]\n\n    # Prepare labels for each class for the fold\n    bowel_train = y_train_fold[:, 2]\n    extravasation_train = y_train_fold[:, 4]\n    kidney_train = y_train_fold[:, 4:7]\n    liver_train = y_train_fold[:, 7:10]\n    spleen_train = y_train_fold[:, 10:13]\n\n    bowel_val = y_val_fold[:, 2]\n    extravasation_val = y_val_fold[:, 4]\n    kidney_val = y_val_fold[:, 4:7]\n    liver_val = y_val_fold[:, 7:10]\n    spleen_val = y_val_fold[:, 10:13]\n\n    # Create a new model for each fold\n    model = create_model()\n\n    # EarlyStopping and ModelCheckpoint\n    early_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n    checkpoint = ModelCheckpoint(f'best_model_fold_{fold}.h5', save_best_only=True, monitor='val_loss')\n\n    # Train the model\n    batch_size = 16\n    num_epoch = 200  # You can modify this value based on the training time and model convergence\n    \n    history = model.fit(\n        x=X_train_fold,\n        y=[bowel_train, extravasation_train, kidney_train, liver_train, spleen_train],\n        batch_size=batch_size,\n        epochs=num_epoch,\n        verbose=1,\n        validation_data=(X_val_fold, [bowel_val, extravasation_val, kidney_val, liver_val, spleen_val])\n    )\n\n    # Store results for this fold\n    fold_results.append(history.history)\n\n# Check results\nfor fold, result in enumerate(fold_results):\n    print(f\"Fold {fold} Results:\")  # Changed to print fold number from 0 to 4\n    for key in result.keys():\n        print(f\"{key}: {result[key][-1]}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T13:50:30.856591Z","iopub.execute_input":"2024-11-14T13:50:30.856932Z","iopub.status.idle":"2024-11-14T15:55:11.119216Z","shell.execute_reply.started":"2024-11-14T13:50:30.856901Z","shell.execute_reply":"2024-11-14T15:55:11.118264Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.datasets import make_classification\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.ensemble import RandomForestClassifier\nimport numpy as np\n\n# Tạo dữ liệu mẫu\nX, y = make_classification(n_samples=1000, n_features=20, n_classes=2, random_state=42)\n\n# Khởi tạo model (có thể thay bằng model của bạn)\nmodel = RandomForestClassifier(random_state=42)\n\n# Khởi tạo các danh sách để lưu kết quả từ mỗi fold\naccuracy_scores = []\nprecision_scores = []\nrecall_scores = []\nf1_scores = []\n\n# Sử dụng StratifiedKFold cho Cross-Validation\nkf = StratifiedKFold(n_splits=5)\n\nfor train_index, val_index in kf.split(X, y):\n    X_train, X_val = X[train_index], X[val_index]\n    y_train, y_val = y[train_index], y[val_index]\n    \n    # Huấn luyện model trên tập train\n    model.fit(X_train, y_train)\n    \n    # Dự đoán trên tập validation\n    y_pred = model.predict(X_val)\n    \n    # Tính toán các chỉ số cho fold hiện tại\n    accuracy_scores.append(accuracy_score(y_val, y_pred))\n    precision_scores.append(precision_score(y_val, y_pred, average='weighted'))\n    recall_scores.append(recall_score(y_val, y_pred, average='weighted'))\n    f1_scores.append(f1_score(y_val, y_pred, average='weighted'))\n\n# Tính trung bình các chỉ số qua các fold\navg_accuracy = np.mean(accuracy_scores)\navg_precision = np.mean(precision_scores)\navg_recall = np.mean(recall_scores)\navg_f1 = np.mean(f1_scores)\n\nprint(f\"Average Accuracy: {avg_accuracy:.4f}\")\nprint(f\"Average Precision: {avg_precision:.4f}\")\nprint(f\"Average Recall: {avg_recall:.4f}\")\nprint(f\"Average F1 Score: {avg_f1:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T18:30:13.219999Z","iopub.execute_input":"2024-11-14T18:30:13.220729Z","iopub.status.idle":"2024-11-14T18:30:14.993925Z","shell.execute_reply.started":"2024-11-14T18:30:13.220698Z","shell.execute_reply":"2024-11-14T18:30:14.992927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Giả sử 'fold_results' là danh sách chứa các kết quả huấn luyện của từng fold\n# Mỗi phần tử trong 'fold_results' là dictionary chứa thông tin về các metric cho từng fold.\n\n# Danh sách các metrics mà bạn muốn theo dõi \nmetrics = [\n    'val_bowel_bowel_binary_accuracy', 'val_extra_extra_binary_accuracy',\n    'val_liver_liver_cat_accuracy', 'val_kidney_kidney_cat_accuracy', 'val_spleen_spleen_cat_accuracy',\n    'val_bowel_loss', 'val_extra_loss', 'val_liver_loss', 'val_kidney_loss', 'val_spleen_loss'\n]\n\n# Lưu kết quả tốt nhất của các metrics\nmetric_best_values = {metric: [] for metric in metrics}  # Khởi tạo dictionary để lưu giá trị tốt nhất của từng metric\n\n# Duyệt qua từng fold trong fold_results\nfor fold_result in fold_results:\n    for metric in metrics:\n        # Lấy giá trị của metric từ fold_result\n        if metric in fold_result:\n            metric_values = np.asarray(fold_result[metric])  # Lấy giá trị metric cho fold này\n            \n            # Tìm giá trị tốt nhất (max đối với accuracy, min đối với loss)\n            best_value = np.max(metric_values) if 'accuracy' in metric else np.min(metric_values)\n            \n            # Thêm giá trị tốt nhất vào danh sách của metric\n            metric_best_values[metric].append(best_value)\n\n# Tính và in trung bình tốt nhất cho các metrics (loại bỏ val_loss)\nfor metric, best_values in metric_best_values.items():\n    avg_best_value = np.mean(best_values)  # Tính trung bình của các giá trị tốt nhất từ các fold\n    print(f\"Best average for {metric}: {avg_best_value:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T17:27:30.600827Z","iopub.execute_input":"2024-11-14T17:27:30.601512Z","iopub.status.idle":"2024-11-14T17:27:30.614447Z","shell.execute_reply.started":"2024-11-14T17:27:30.60148Z","shell.execute_reply":"2024-11-14T17:27:30.613396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Store results for validation accuracy across all folds for all 5 metrics (bowel, extra, liver, kidney, spleen)\navg_val_acc = {\n    'bowel': 0,\n    'extra': 0,\n    'liver': 0,\n    'kidney': 0,\n    'spleen': 0\n}\n\n# Số lượng folds\nnum_folds = len(fold_results)\n\n# Lặp qua từng fold để tính tổng validation accuracy cho tất cả các lớp\nfor fold, result in enumerate(fold_results):\n    # Extract validation accuracy for each class (bowel, extravasation, liver, kidney, spleen)\n    val_bowel_acc = result['val_bowel_bowel_binary_accuracy'][-1]  # Validation accuracy for bowel\n    val_extra_acc = result['val_extra_extra_binary_accuracy'][-1]  # Validation accuracy for extra\n    val_liver_acc = result['val_liver_liver_cat_accuracy'][-1]  # Validation accuracy for liver\n    val_kidney_acc = result['val_kidney_kidney_cat_accuracy'][-1]  # Validation accuracy for kidney\n    val_spleen_acc = result['val_spleen_spleen_cat_accuracy'][-1]  # Validation accuracy for spleen\n    \n    # Cộng dồn các giá trị validation accuracy\n    avg_val_acc['bowel'] += val_bowel_acc\n    avg_val_acc['extra'] += val_extra_acc\n    avg_val_acc['liver'] += val_liver_acc\n    avg_val_acc['kidney'] += val_kidney_acc\n    avg_val_acc['spleen'] += val_spleen_acc\n\n# Tính trung bình validation accuracy cho tất cả các lớp\navg_val_acc['bowel'] /= num_folds\navg_val_acc['extra'] /= num_folds\navg_val_acc['liver'] /= num_folds\navg_val_acc['kidney'] /= num_folds\navg_val_acc['spleen'] /= num_folds\n\n# In kết quả validation accuracy trung bình cho mỗi lớp\nprint(\"\\nAverage Validation Accuracy across all folds for each class:\")\nprint(f\"Bowel Validation Accuracy: {avg_val_acc['bowel']:.4f}\")\nprint(f\"Extravasation Validation Accuracy: {avg_val_acc['extra']:.4f}\")\nprint(f\"Liver Validation Accuracy: {avg_val_acc['liver']:.4f}\")\nprint(f\"Kidney Validation Accuracy: {avg_val_acc['kidney']:.4f}\")\nprint(f\"Spleen Validation Accuracy: {avg_val_acc['spleen']:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T18:30:01.416853Z","iopub.execute_input":"2024-11-14T18:30:01.417472Z","iopub.status.idle":"2024-11-14T18:30:01.427499Z","shell.execute_reply.started":"2024-11-14T18:30:01.417438Z","shell.execute_reply":"2024-11-14T18:30:01.426625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Store results for the best validation accuracy, accuracy, loss, F1 score, recall, precision across all folds for all 5 metrics (bowel, extra, liver, kidney, spleen)\navg_results = {\n    'acc': 0,\n    'f1': 0,\n    'recall': 0,\n    'precision': 0,\n    'loss': 0,\n    'val_acc': 0\n}\n\n# Số lượng folds\nnum_folds = len(fold_results)\n\n# Lặp qua từng fold để tìm các chỉ số tốt nhất cho tất cả các lớp\nfor fold, result in enumerate(fold_results):\n    # Find the best validation accuracy for each class (best val_acc across epochs)\n    best_val_bowel_acc = max(result['val_bowel_bowel_binary_accuracy'])  # Best validation accuracy for bowel\n    best_val_extra_acc = max(result['val_extra_extra_binary_accuracy'])  # Best validation accuracy for extra\n    best_val_liver_acc = max(result['val_liver_liver_cat_accuracy'])  # Best validation accuracy for liver\n    best_val_kidney_acc = max(result['val_kidney_kidney_cat_accuracy'])  # Best validation accuracy for kidney\n    best_val_spleen_acc = max(result['val_spleen_spleen_cat_accuracy'])  # Best validation accuracy for spleen\n\n    # Final metrics for each class\n    bowel_acc = result['bowel_bowel_binary_accuracy'][-1]  # Final accuracy for bowel\n    extra_acc = result['extra_extra_binary_accuracy'][-1]  # Final accuracy for extra\n    liver_acc = result['liver_liver_cat_accuracy'][-1]  # Final accuracy for liver\n    kidney_acc = result['kidney_kidney_cat_accuracy'][-1]  # Final accuracy for kidney\n    spleen_acc = result['spleen_spleen_cat_accuracy'][-1]  # Final accuracy for spleen\n    \n    bowel_loss = result['bowel_loss'][-1]  # Final loss for bowel\n    extra_loss = result['extra_loss'][-1]  # Final loss for extra\n    liver_loss = result['liver_loss'][-1]  # Final loss for liver\n    kidney_loss = result['kidney_loss'][-1]  # Final loss for kidney\n    spleen_loss = result['spleen_loss'][-1]  # Final loss for spleen\n    \n    bowel_f1 = result['bowel_f1_score_metric'][-1]  # Final F1 score for bowel\n    extra_f1 = result['extra_f1_score_metric'][-1]  # Final F1 score for extra\n    liver_f1 = result['liver_f1_score_metric'][-1]  # Final F1 score for liver\n    kidney_f1 = result['kidney_f1_score_metric'][-1]  # Final F1 score for kidney\n    spleen_f1 = result['spleen_f1_score_metric'][-1]  # Final F1 score for spleen\n    \n    bowel_recall = result['bowel_bowel_recall'][-1]  # Final recall for bowel\n    extra_recall = result['extra_extra_recall'][-1]  # Final recall for extra\n    liver_recall = result['liver_liver_recall'][-1]  # Final recall for liver\n    kidney_recall = result['kidney_kidney_recall'][-1]  # Final recall for kidney\n    spleen_recall = result['spleen_spleen_recall'][-1]  # Final recall for spleen\n    \n    bowel_precision = result['bowel_bowel_precision'][-1]  # Final precision for bowel\n    extra_precision = result['extra_extra_precision'][-1]  # Final precision for extra\n    liver_precision = result['liver_liver_precision'][-1]  # Final precision for liver\n    kidney_precision = result['kidney_kidney_precision'][-1]  # Final precision for kidney\n    spleen_precision = result['spleen_spleen_precision'][-1]  # Final precision for spleen\n\n    # Cộng dồn các giá trị cho các chỉ số\n    avg_results['val_acc'] += (best_val_bowel_acc + best_val_extra_acc + best_val_liver_acc + best_val_kidney_acc + best_val_spleen_acc)\n    \n    avg_results['acc'] += (bowel_acc + extra_acc + liver_acc + kidney_acc + spleen_acc)\n    avg_results['f1'] += (bowel_f1 + extra_f1 + liver_f1 + kidney_f1 + spleen_f1)\n    avg_results['recall'] += (bowel_recall + extra_recall + liver_recall + kidney_recall + spleen_recall)\n    avg_results['precision'] += (bowel_precision + extra_precision + liver_precision + kidney_precision + spleen_precision)\n    \n    avg_results['loss'] += (bowel_loss + extra_loss + liver_loss + kidney_loss + spleen_loss)\n\n# Tính trung bình cho tất cả các chỉ số\navg_results['val_acc'] /= (num_folds * 5)  # Chia cho 5 vì có 5 lớp\navg_results['acc'] /= (num_folds * 5)\navg_results['f1'] /= (num_folds * 5)\navg_results['recall'] /= (num_folds * 5)\navg_results['precision'] /= (num_folds * 5)\navg_results['loss'] /= (num_folds * 5)\n\n# In kết quả trung bình cho tất cả các chỉ số (không có val_loss)\nprint(\"\\nAverage Results across all folds (Total Average for all classes):\")\nprint(f\"Average Validation Accuracy: {avg_results['val_acc']:.4f}\")\nprint(f\"Average Accuracy: {avg_results['acc']:.4f}\")\nprint(f\"Average F1 Score: {avg_results['f1']:.4f}\")\nprint(f\"Average Recall: {avg_results['recall']:.4f}\")\nprint(f\"Average Precision: {avg_results['precision']:.4f}\")\nprint(f\"Average Loss: {avg_results['loss']:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T17:53:19.221837Z","iopub.execute_input":"2024-11-14T17:53:19.222879Z","iopub.status.idle":"2024-11-14T17:53:19.242187Z","shell.execute_reply.started":"2024-11-14T17:53:19.222834Z","shell.execute_reply":"2024-11-14T17:53:19.241239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\n# Giả sử bạn đã lưu các độ chính xác của mỗi fold vào fold_results\naccuracies = []\n\nfor fold_result in fold_results:\n    # Độ chính xác cho các chỉ số trong quá trình huấn luyện, ví dụ 'val_accuracy'\n    accuracies.append(fold_result['val_loss'])  # Bạn có thể thay thế 'val_accuracy' bằng các chỉ số khác nếu muốn\n\n# Tạo Boxplot\nplt.figure(figsize=(10, 6))\nsns.boxplot(data=accuracies)\nplt.title('Boxplot of Validation Accuracy across Folds')\nplt.ylabel('Validation Accuracy')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T18:30:58.365172Z","iopub.execute_input":"2024-11-14T18:30:58.365533Z","iopub.status.idle":"2024-11-14T18:30:58.592622Z","shell.execute_reply.started":"2024-11-14T18:30:58.365503Z","shell.execute_reply":"2024-11-14T18:30:58.591711Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Kiểm tra các khóa trong history\nprint(history.history.keys())\n\n# Các metric bạn muốn vẽ (chỉ vẽ các accuracy mà không có \"val_\")\nmetrics_to_plot = [key for key in history.history.keys() if key.endswith('accuracy') and 'val_' not in key]\n\n# Tạo biểu đồ cho cả huấn luyện và validation\nfor metric in metrics_to_plot:\n    plt.figure(figsize=(8, 6))  # Tạo một biểu đồ mới cho mỗi metric\n    plt.plot(history.history[metric], label=f'Training {metric}')\n    \n    # Kiểm tra và vẽ độ chính xác của validation nếu có\n    val_metric = f'val_{metric}'\n    if val_metric in history.history:\n        plt.plot(history.history[val_metric], label=f'Validation {metric}')\n    \n    # Thêm số epoch vào trục x (tự động từ 0 đến n)\n    epochs = range(1, len(history.history[metric]) + 1)\n    \n    # Thêm ticks với khoảng cách 25\n    step_size = 25\n    epoch_ticks = [epoch for epoch in epochs if epoch % step_size == 0]\n    plt.xticks(epoch_ticks)\n    \n    # Cài đặt các thông số cho biểu đồ\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.title(f'{metric.capitalize()} over Epochs')\n    plt.legend(loc='upper left')\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T18:31:27.832838Z","iopub.execute_input":"2024-11-14T18:31:27.833445Z","iopub.status.idle":"2024-11-14T18:31:29.249205Z","shell.execute_reply.started":"2024-11-14T18:31:27.83341Z","shell.execute_reply":"2024-11-14T18:31:29.248385Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(history.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(history.history[\"loss\"], label=\"loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T17:34:13.532971Z","iopub.execute_input":"2024-11-14T17:34:13.533335Z","iopub.status.idle":"2024-11-14T17:34:13.78768Z","shell.execute_reply.started":"2024-11-14T17:34:13.533305Z","shell.execute_reply":"2024-11-14T17:34:13.786781Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history.history.keys()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T18:43:12.513041Z","iopub.execute_input":"2024-11-14T18:43:12.51339Z","iopub.status.idle":"2024-11-14T18:43:12.54787Z","shell.execute_reply.started":"2024-11-14T18:43:12.51336Z","shell.execute_reply":"2024-11-14T18:43:12.54662Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_acc = ['val_bowel_bowel_binary_accuracy', 'val_extra_extra_binary_accuracy', 'val_liver_liver_cat_accuracy', 'val_kidney_kidney_cat_accuracy', 'val_spleen_spleen_cat_accuracy']","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-11-14T17:34:32.157612Z","iopub.execute_input":"2024-11-14T17:34:32.158573Z","iopub.status.idle":"2024-11-14T17:34:32.163116Z","shell.execute_reply.started":"2024-11-14T17:34:32.158527Z","shell.execute_reply":"2024-11-14T17:34:32.162221Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Lặp qua các key trong history để vẽ accuracy\nfor i in history.history.keys():\n    if i.endswith(\"_accuracy\") and not i == \"val_accuracy\":\n        plt.plot(history.history[i], label=i)\n\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend(loc=(1.05, 0.0))\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T18:33:53.031894Z","iopub.execute_input":"2024-11-14T18:33:53.032648Z","iopub.status.idle":"2024-11-14T18:33:53.341097Z","shell.execute_reply.started":"2024-11-14T18:33:53.032616Z","shell.execute_reply":"2024-11-14T18:33:53.34018Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"_loss\") and not i ==\"val_loss\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T18:33:56.607169Z","iopub.execute_input":"2024-11-14T18:33:56.6075Z","iopub.status.idle":"2024-11-14T18:33:56.885402Z","shell.execute_reply.started":"2024-11-14T18:33:56.607473Z","shell.execute_reply":"2024-11-14T18:33:56.884525Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"accuracy\") and not i ==\"val_accuracy\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T18:34:22.81479Z","iopub.execute_input":"2024-11-14T18:34:22.815139Z","iopub.status.idle":"2024-11-14T18:34:23.174118Z","shell.execute_reply.started":"2024-11-14T18:34:22.815109Z","shell.execute_reply":"2024-11-14T18:34:23.173258Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Lặp qua các key trong history để vẽ accuracy (loại bỏ val_accuracy)\nfor i in history.history.keys():\n    if i.endswith(\"accuracy\") and \"val_\" not in i:\n        plt.plot(history.history[i], label=i)\n\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend(loc=(1.05, 0.0))\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T18:34:26.239061Z","iopub.execute_input":"2024-11-14T18:34:26.239411Z","iopub.status.idle":"2024-11-14T18:34:26.481304Z","shell.execute_reply.started":"2024-11-14T18:34:26.239381Z","shell.execute_reply":"2024-11-14T18:34:26.48048Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Kiểm tra các khóa trong history\nprint(history.history.keys())\n\n# Các metric bạn muốn vẽ (chỉ vẽ các accuracy mà không có \"val_\")\nmetrics_to_plot = [key for key in history.history.keys() if key.endswith('accuracy') and 'val_' not in key]\n\n# Tạo biểu đồ cho cả huấn luyện và validation\nfor metric in metrics_to_plot:\n    plt.figure(figsize=(8, 6))  # Tạo một biểu đồ mới cho mỗi metric\n    plt.plot(history.history[metric], label=f'Training {metric}')\n    \n    # Kiểm tra và vẽ độ chính xác của validation nếu có\n    val_metric = f'val_{metric}'\n    if val_metric in history.history:\n        plt.plot(history.history[val_metric], label=f'Validation {metric}')\n    \n    # Thêm số epoch vào trục x (tự động từ 0 đến n)\n    epochs = range(1, len(history.history[metric]) + 1)\n    \n    # Thêm ticks với khoảng cách 25\n    step_size = 25\n    epoch_ticks = [epoch for epoch in epochs if epoch % step_size == 0]\n    plt.xticks(epoch_ticks)\n    \n    # Cài đặt các thông số cho biểu đồ\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.title(f'{metric.capitalize()} over Epochs')\n    plt.legend(loc='upper left')\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T18:42:59.211276Z","iopub.execute_input":"2024-11-14T18:42:59.212046Z","iopub.status.idle":"2024-11-14T18:42:59.267897Z","shell.execute_reply.started":"2024-11-14T18:42:59.212012Z","shell.execute_reply":"2024-11-14T18:42:59.266668Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(history.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(history.history[\"loss\"], label=\"loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T18:35:37.888842Z","iopub.execute_input":"2024-11-14T18:35:37.889201Z","iopub.status.idle":"2024-11-14T18:35:38.150691Z","shell.execute_reply.started":"2024-11-14T18:35:37.889171Z","shell.execute_reply":"2024-11-14T18:35:38.1498Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Giả sử bạn có các giá trị accuracy và các metrics Ep cho từng epoch hoặc mô hình\nepochs = np.arange(1, 11)  # Ví dụ: 10 epochs\naccuracy = np.random.rand(10)  # Accuracy giả định (tạo ngẫu nhiên từ 0 đến 1)\nprecision = np.random.rand(10)  # Precision giả định (tạo ngẫu nhiên)\nrecall = np.random.rand(10)  # Recall giả định\nf1_score = np.random.rand(10)  # F1-Score giả định\n\n# Vẽ đồ thị đường cho Accuracy và các metrics Ep\nplt.figure(figsize=(10, 6))\n\n# Accuracy\nplt.plot(epochs, accuracy, label='Accuracy', color='blue', marker='o', linestyle='-', linewidth=2)\n\n# Precision\nplt.plot(epochs, precision, label='Precision', color='green', marker='s', linestyle='--', linewidth=2)\n\n# Recall\nplt.plot(epochs, recall, label='Recall', color='red', marker='^', linestyle='-.', linewidth=2)\n\n# F1-Score\nplt.plot(epochs, f1_score, label='F1-Score', color='purple', marker='x', linestyle=':', linewidth=2)\n\n# Thiết lập nhãn và tiêu đề\nplt.xlabel('Epochs', fontsize=14)\nplt.ylabel('Scores', fontsize=14)\nplt.title('Accuracy and EP (Precision, Recall, F1-Score) vs Epochs', fontsize=16)\nplt.legend(loc='upper left')\n\n# Hiển thị đồ thị\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T18:39:22.703195Z","iopub.execute_input":"2024-11-14T18:39:22.703877Z","iopub.status.idle":"2024-11-14T18:39:23.037636Z","shell.execute_reply.started":"2024-11-14T18:39:22.703844Z","shell.execute_reply":"2024-11-14T18:39:23.036702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Đảm bảo rằng tệp CSV hoặc nguồn dữ liệu của bạn đã được nạp vào DataFrame\ndf = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Thay thế bằng đường dẫn đúng\n\n# Kiểm tra các giá trị thiếu trong DataFrame\nmissing_values = df.isnull().sum()\nprint(\"Missing Values:\")\nprint(missing_values)\n\n# Danh sách các cột nhị phân cần chuyển đổi sang kiểu boolean\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Chuyển đổi các cột nhị phân thành kiểu boolean\ndf[binary_columns] = df[binary_columns].astype(bool)\n\n# Giải quyết các vấn đề về chất lượng dữ liệu (nếu có, bạn có thể thêm các bước xử lý dữ liệu ở đây)\n\n# Hiển thị DataFrame sau khi đã xử lý\nprint(\"\\nPreprocessed DataFrame:\")\nprint(df.head())  # In ra 5 dòng đầu tiên của DataFrame đã xử lý để kiểm tra\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.019454Z","iopub.status.idle":"2024-11-14T15:55:12.019812Z","shell.execute_reply.started":"2024-11-14T15:55:12.019623Z","shell.execute_reply":"2024-11-14T15:55:12.019639Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cột nhãn cho các cơ quan bạn muốn hiển thị\norgan_columns = ['kidney', 'liver', 'spleen']  # Chỉ giữ lại các cơ quan bạn muốn hiển thị\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: kidney_healthy)\n    y_pred = df[f'{organ}_low']  # Hoặc chọn nhãn khác như kidney_low (tùy vào mục đích của bạn)\n\n    # Tạo ma trận nhầm lẫn với 5 lớp (0 đến 4)\n    cm = confusion_matrix(y_true, y_pred, labels=[False, True, 2, 3, 4,5,6,7,8,9])  # Chỉ định 5 lớp (tùy theo dữ liệu của bạn)\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9, 10], yticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9, 10])\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.025929Z","iopub.status.idle":"2024-11-14T15:55:12.026268Z","shell.execute_reply.started":"2024-11-14T15:55:12.026104Z","shell.execute_reply":"2024-11-14T15:55:12.026121Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kiểm tra tất cả các cột trong DataFrame\nprint(df.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.027428Z","iopub.status.idle":"2024-11-14T15:55:12.02775Z","shell.execute_reply.started":"2024-11-14T15:55:12.027582Z","shell.execute_reply":"2024-11-14T15:55:12.027596Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cơ quan bạn muốn hiển thị\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: bowel_healthy)\n    \n    # Chọn cột dự đoán như 'injury', 'low' hoặc 'high', tùy vào mục đích của bạn\n    y_pred = df[f'{organ}_injury']  # Hoặc thay bằng 'low' hoặc 'high' tùy vào nhu cầu\n\n    # Tạo ma trận nhầm lẫn với 2 lớp (0: Healthy, 1: Injury)\n    cm_binary = confusion_matrix(y_true, y_pred, labels=[False, True])  # Chỉ so sánh Healthy vs Injury\n\n    # Vẽ ma trận nhầm lẫn 2 lớp\n    plt.figure(figsize=(10, 8))  \n    sns.heatmap(cm_binary, annot=True, fmt='d', cmap='Blues', xticklabels=['Healthy', 'Injury'], yticklabels=['Healthy', 'Injury'])\n    plt.title(f'Confusion Matrix for {organ.capitalize()} (2-class: Healthy vs Injury)')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n    plt.show()\n\n    # Tạo ma trận nhầm lẫn với 10 lớp (0 đến 9) nếu dữ liệu có thể hỗ trợ\n    cm_10class = confusion_matrix(y_true, y_pred, labels=range(10))  # Sử dụng 10 lớp (0 đến 9)\n    \n    # Vẽ ma trận nhầm lẫn 10 lớp\n    plt.figure(figsize=(10, 8))  \n    sns.heatmap(cm_10class, annot=True, fmt='d', cmap='Blues', xticklabels=range(10), yticklabels=range(10))\n    plt.title(f'Confusion Matrix for {organ.capitalize()} (10-class)')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.028809Z","iopub.status.idle":"2024-11-14T15:55:12.029135Z","shell.execute_reply.started":"2024-11-14T15:55:12.028974Z","shell.execute_reply":"2024-11-14T15:55:12.028989Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cột nhãn cho các cơ quan bạn muốn hiển thị\norgan_columns = ['kidney', 'liver', 'spleen']  # Các cơ quan cần hiển thị\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: kidney_healthy)\n    y_pred = df[f'{organ}_low']  # Hoặc chọn nhãn khác như kidney_low (tùy vào mục đích của bạn)\n\n    # Tạo ma trận nhầm lẫn với 10 lớp (0 đến 9)\n    cm = confusion_matrix(y_true, y_pred, labels=[False, True, 2, 3, 4, 5, 6, 7, 8, 9])  # Giả sử dữ liệu có lớp 0-9\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9], yticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9])\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.030497Z","iopub.status.idle":"2024-11-14T15:55:12.030829Z","shell.execute_reply.started":"2024-11-14T15:55:12.030645Z","shell.execute_reply":"2024-11-14T15:55:12.030659Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv')\n\n# Trích xuất nhãn thật (y_true) và nhãn dự đoán (y_pred)\n# Giả sử rằng bạn có một mô hình dự đoán sẵn có hoặc đang huấn luyện mô hình để lấy y_pred\n\n# Dữ liệu về bowel (thay 'bowel' bằng các cơ quan khác như 'extravasation', 'kidney', v.v.)\ny_true_bowel = df['bowel_healthy']  # Hoặc nếu bạn muốn nhãn về injury thì dùng 'bowel_injury'\ny_pred_bowel = df['bowel_injury']  # Đây là nhãn dự đoán mà mô hình của bạn sẽ đưa ra (giả lập)\n\n# Dự đoán giả lập cho ví dụ\n# y_pred_bowel có thể là đầu ra của mô hình dự đoán (thay 'bowel_injury' bằng giá trị thực tế của mô hình của bạn)\n# Đoạn dưới đây là giả lập, bạn sẽ thay thế bằng mô hình thực tế của mình.\n\n# Xử lý các cơ quan khác (extravasation, kidney, liver, spleen)\ny_true_extra = df['extravasation_healthy']\ny_pred_extra= df['extravasation_injury']  # Tùy thuộc vào nhãn bạn muốn sử dụng\n\n# Tạo danh sách các cột nhãn cho các cơ quan khác\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    y_true = df[f'{organ}_healthy']\n    y_pred = df[f'{organ}_injury']  # Hoặc thay đổi cột này tùy vào nhu cầu\n\n    # Tạo ma trận nhầm lẫn với 5 lớp (0 đến 4)\n    cm = confusion_matrix(y_true, y_pred, labels=range(10))  # Sử dụng 10 lớp (0 đến 9)\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận 11x11\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=range(11), yticklabels=range(11))\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.032459Z","iopub.status.idle":"2024-11-14T15:55:12.032795Z","shell.execute_reply.started":"2024-11-14T15:55:12.032614Z","shell.execute_reply":"2024-11-14T15:55:12.032628Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for missing values\nmissing_values = df.isnull().sum()\nprint(\"Missing Values:\")\nprint(missing_values)\n\n# Handle missing values\n# In this simple example, we will drop rows with missing values.\ndf = df.dropna()\n\n# Check Data Types and Convert Binary Data to Boolean\nbinary_columns = [\n  'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n  'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n  'spleen_healthy', 'spleen_low', 'spleen_high'\n]\ndf[binary_columns] = df[binary_columns].astype(bool)\n\n# Address Data Quality Issues\n# In this simple example, we assume no data quality issues are present.\n\nprint(\"\\nPreprocessed DataFrame:\")\nprint(df)\n\nplt.figure()\ndf.plot.hist()\nplt.title('Distribution of Features')\nplt.xlabel('Feature')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.033993Z","iopub.status.idle":"2024-11-14T15:55:12.034307Z","shell.execute_reply.started":"2024-11-14T15:55:12.034146Z","shell.execute_reply":"2024-11-14T15:55:12.034161Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.035684Z","iopub.status.idle":"2024-11-14T15:55:12.036028Z","shell.execute_reply.started":"2024-11-14T15:55:12.035871Z","shell.execute_reply":"2024-11-14T15:55:12.035887Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport plotly.express as px\n\n# Giả sử df là DataFrame đã được nạp vào từ dữ liệu của bạn\n# df = pd.read_csv(\"/path/to/your/data.csv\")\n\n# Cột các cơ quan\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Kiểm tra xem các cột \"injury\" có tồn tại trong DataFrame không\ninjury_columns = [f'{organ}_injury' for organ in organ_columns]\nmissing_columns = set(organ_columns + injury_columns) - set(df.columns)\n\nif missing_columns:\n    # Thông báo nếu có cột thiếu\n    print(f\"Warning: Columns for {', '.join(missing_columns)} are missing in the DataFrame.\")\n    for col in missing_columns:\n        df[col] = 0  # Thêm cột thiếu vào DataFrame với giá trị 0\n\n# Lọc các cột liên quan đến sức khỏe và chấn thương của các cơ quan\ncorrelation_df = df[organ_columns + injury_columns]\n\n# Tính toán ma trận tương quan giữa các cột\ncorrelation_matrix = correlation_df.corr()\n\n# Tạo heatmap để phân tích mối tương quan giữa sức khỏe và tình trạng chấn thương của các cơ quan\nfig = px.imshow(\n    correlation_matrix,\n    x=correlation_df.columns,\n    y=correlation_df.columns,\n    labels=dict(x='Organ', y='Organ', color='Correlation'),\n    title='Correlation Between Organ Health and Injury Status',\n)\n\n# Hiển thị heatmap\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.037305Z","iopub.status.idle":"2024-11-14T15:55:12.037631Z","shell.execute_reply.started":"2024-11-14T15:55:12.037468Z","shell.execute_reply":"2024-11-14T15:55:12.037484Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Summary statistics for relevant variables\nstyled_data = df.describe().style\\\n.background_gradient(cmap='coolwarm')\\\n.set_properties(**{'text-align':'center','border':'1px solid black'})\n\n# display styled data\ndisplay(styled_data)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.038917Z","iopub.status.idle":"2024-11-14T15:55:12.039247Z","shell.execute_reply.started":"2024-11-14T15:55:12.039082Z","shell.execute_reply":"2024-11-14T15:55:12.039098Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install matplotlib seaborn\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.040668Z","iopub.status.idle":"2024-11-14T15:55:12.04101Z","shell.execute_reply.started":"2024-11-14T15:55:12.040851Z","shell.execute_reply":"2024-11-14T15:55:12.040867Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\n# Tạo dữ liệu giả lập\nnp.random.seed(10)\ndata = np.random.normal(loc=0, scale=1, size=100)  # 100 giá trị phân phối chuẩn\n\n# Vẽ boxplot\nplt.figure(figsize=(8,6))\nsns.boxplot(data=data)\n\n# Thêm tiêu đề và nhãn cho đồ thị\nplt.title('Boxplot Example')\nplt.xlabel('Data')\n\n# Hiển thị đồ thị\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.042554Z","iopub.status.idle":"2024-11-14T15:55:12.042912Z","shell.execute_reply.started":"2024-11-14T15:55:12.042722Z","shell.execute_reply":"2024-11-14T15:55:12.042756Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot styled data in a single plot, using subgrid layout\nimport matplotlib.gridspec as gridspec\n\n\nstyled_data = df.describe().style\\\n.background_gradient(cmap='coolwarm')\\\n.set_properties(**{'text-align':'center','border':'1px solid black'})\n\n# Cgridspec layout\ngs = gridspec.GridSpec(2, 2)\n\n# Loop over the styled data and plot it\nfig, axes = plt.subplots(2, 2, figsize=(12, 8), subplot_kw={'adjustable': 'box'})\nfor i in range(2):\n    for j in range(2):\n        cell_value = styled_data.data.iloc[i, j]\n\n        axes[i, j].plot(cell_value)\n        axes[i, j].set_title(styled_data.index[i] + ', ' + styled_data.columns[j])\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.044096Z","iopub.status.idle":"2024-11-14T15:55:12.044399Z","shell.execute_reply.started":"2024-11-14T15:55:12.044249Z","shell.execute_reply":"2024-11-14T15:55:12.044263Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# pass list of tick positions to the set_xticks() function. \n# pass the following list of tick positions to the set_xticks() function in the counts plot loop\n\ndef generate_counts_and_percentages(df, categorical_columns):\n  \"\"\"Counts and percentages for categorical variables in a DataFrame, and plot the counts and percentages.\n\n  Args:\n    df: DataFrame.\n    categorical_columns: column names for the categorical variables.\n\n  Returns:\n    None.\n  \"\"\"\n\n  # Handle null values.\n  df = df.dropna(subset=categorical_columns)\n\n  # counts.\n  counts = df[categorical_columns].apply(pd.Series.value_counts)\n\n  # percentages.\n  percentages = (counts / df.shape[0]) * 100\n\n  # Set color scheme.\n  colors = ['#007bff', '#ffa500']\n\n  # Plot counts.\n  fig, axes = plt.subplots(1, len(categorical_columns), figsize=(15, 7))\n  for i, column in enumerate(categorical_columns):\n    ax = axes[i]\n    ax.bar(counts.index.to_list(), counts[column].to_list(), color=colors[0])\n    ax.set_title(column, fontsize=12)\n    ax.set_xticks(range(len(counts.index)))\n    ax.tick_params(labelsize=10)\n    ax.grid(True)\n\n  # Plot percentages.\n  fig, axes = plt.subplots(1, len(categorical_columns), figsize=(15, 7))\n  for i, column in enumerate(categorical_columns):\n    ax = axes[i]\n    ax.pie(percentages[column].to_list(), labels=percentages.index.to_list(), autopct='%1.1f%%', startangle=140, colors=colors)\n    ax.set_title(column, fontsize=12)\n    ax.axis('equal')\n    ax.legend(fontsize=10)\n    ax.grid(True)\n\n  plt.suptitle('Counts and Percentages for Categorical Variables', fontsize=14)\n  plt.show()\n\ncategorical_columns = [\n  'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n  'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n  'spleen_healthy', 'spleen_low', 'spleen_high', 'any_injury'\n]\n\ngenerate_counts_and_percentages(df, categorical_columns)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.046406Z","iopub.status.idle":"2024-11-14T15:55:12.046718Z","shell.execute_reply.started":"2024-11-14T15:55:12.046567Z","shell.execute_reply":"2024-11-14T15:55:12.046581Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Organ columns: bowel, extravasation, kidney, liver, spleen\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Create a new DataFrame to store the counts\norgan_counts = pd.DataFrame()\norgan_counts['Organ'] = organ_columns\n\n# Loop through organ columns and count healthy and injury status for each organ\nfor organ in organ_columns:\n    healthy_col = f'{organ}_healthy'\n    injury_col = f'{organ}_injury'\n    \n    # Check if the columns exist in the DataFrame\n    if healthy_col in df.columns and injury_col in df.columns:\n        organ_counts[f'{organ}_healthy'] = df[healthy_col].sum()\n        organ_counts[f'{organ}_injury'] = df[injury_col].sum()\n    else:\n        # Handle the case if the columns are missing\n        print(f\"Warning: Columns for {organ} healthy/injury status are missing in the DataFrame.\")\n        organ_counts[f'{organ}_healthy'] = 0\n        organ_counts[f'{organ}_injury'] = 0\n\n# Melt the DataFrame to have a single 'Status' column\norgan_counts_melted = organ_counts.melt(id_vars=['Organ'], var_name='Status', value_name='Count')\n\n# Bar plot for distribution of organ health and injury status\nfig = px.bar(\n    organ_counts_melted,\n    x='Organ',\n    y='Count',\n    color='Status',\n    barmode='group',\n    labels=dict(x='Organ', y='Count', Status='Status'),\n    title='Distribution of Organ Health and Injury Status',\n)\nfig.update_layout(showlegend=True)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.049377Z","iopub.status.idle":"2024-11-14T15:55:12.049711Z","shell.execute_reply.started":"2024-11-14T15:55:12.04955Z","shell.execute_reply":"2024-11-14T15:55:12.049566Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Organ columns\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Create a new DataFrame to store the counts\norgan_counts = pd.DataFrame()\norgan_counts['Organ'] = organ_columns\n\n# Loop through organ columns and count healthy and injury status for each organ\nfor organ in organ_columns:\n    healthy_col = f'{organ}_healthy'\n    injury_col = f'{organ}_injury'\n\n    # Check if the columns exist in the DataFrame\n    if healthy_col in df.columns and injury_col in df.columns:\n        organ_counts[f'{organ}_healthy'] = df[healthy_col].sum()\n        organ_counts[f'{organ}_injury'] = df[injury_col].sum()\n    else:\n        # Handle the case if the columns are missing\n        print(f\"Warning: Columns for {organ} healthy/injury status are missing in the DataFrame.\")\n        organ_counts[f'{organ}_healthy'] = 0\n        organ_counts[f'{organ}_injury'] = 0\n\n# Fill in missing values with 0\norgan_counts.fillna(0, inplace=True)\n\n# Melt the DataFrame to have a single 'Status' column\norgan_counts_melted = organ_counts.melt(id_vars=['Organ'], var_name='Status', value_name='Count')\n\n# Bar plot for distribution of organ health and injury status\nfig = px.bar(\n    organ_counts_melted,\n    x='Organ',\n    y='Count',\n    color='Status',\n    barmode='group',\n    labels=dict(x='Organ', y='Count', Status='Status'),\n    title='Distribution of Organ Health and Injury Status',\n    height=500,\n    width=800,\n    template='plotly_dark',\n)\n\n# Customize the plot\nfig.update_layout(\n    legend_title='Organ Status',\n    legend_orientation='h',\n    legend_xanchor='center',\n    legend_yanchor='top',\n    legend_x=0.5,\n    legend_y=1.1,\n    xaxis_title='Organ',\n    yaxis_title='Count',\n    font=dict(family='Arial', size=12),\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.050996Z","iopub.status.idle":"2024-11-14T15:55:12.051307Z","shell.execute_reply.started":"2024-11-14T15:55:12.051148Z","shell.execute_reply":"2024-11-14T15:55:12.051162Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Organ columns: bowel, extravasation, kidney, liver, spleen\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Check if the 'injury' columns are present in the DataFrame\ninjury_columns = [f'{organ}_injury' for organ in organ_columns]\nmissing_columns = set(organ_columns + injury_columns) - set(df.columns)\n\nif missing_columns:\n    # Handle the case if any of the required columns are missing\n    print(f\"Warning: Columns for {', '.join(missing_columns)} are missing in the DataFrame.\")\n    for col in missing_columns:\n        df[col] = 0\n\n# Heatmap to analyze the correlation between organ health and injury status\ncorrelation_df = df[organ_columns + injury_columns]\ncorrelation_matrix = correlation_df.corr()\n\nfig = px.imshow(\n    correlation_matrix,\n    x=correlation_df.columns,\n    y=correlation_df.columns,\n    labels=dict(x='Organ', y='Organ', color='Correlation'),\n    title='Correlation Between Organ Health and Injury Status',\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.053306Z","iopub.status.idle":"2024-11-14T15:55:12.053636Z","shell.execute_reply.started":"2024-11-14T15:55:12.053472Z","shell.execute_reply":"2024-11-14T15:55:12.053488Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport plotly.graph_objects as go\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import accuracy_score, confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.054539Z","iopub.status.idle":"2024-11-14T15:55:12.054874Z","shell.execute_reply.started":"2024-11-14T15:55:12.054686Z","shell.execute_reply":"2024-11-14T15:55:12.0547Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a synthetic binary classification dataset\nn_samples = 500\nn_features = 2\nn_classes = 2\nn_clusters_per_class = 1\nrandom_state = 42\n\n# Adjust the values of n_informative, n_redundant, and n_repeated\nn_informative = 2\nn_redundant = 0\nn_repeated = 0\n\nX, y = make_classification(\n    n_samples=n_samples,\n    n_features=n_features,\n    n_informative=n_informative,\n    n_redundant=n_redundant,\n    n_repeated=n_repeated,\n    n_classes=n_classes,\n    n_clusters_per_class=n_clusters_per_class,\n    random_state=random_state\n)\n\n# Split the dataset into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=random_state)\n\n# Train a logistic regression model on the dataset\nmodel = LogisticRegression(random_state=random_state)\nmodel.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = model.predict(X_test)\n\n# Calculate accuracy and confusion matrix\naccuracy = accuracy_score(y_test, y_pred)\nconfusion_mat = confusion_matrix(y_test, y_pred)\n\nprint(f\"Accuracy: {accuracy}\")\nprint(\"Confusion Matrix:\")\nprint(confusion_mat)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.056659Z","iopub.status.idle":"2024-11-14T15:55:12.057002Z","shell.execute_reply.started":"2024-11-14T15:55:12.056842Z","shell.execute_reply":"2024-11-14T15:55:12.056858Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\n# Define the hyperparameters to search over\nparam_grid = {\n    \"C\": [0.1, 1, 10, 100],\n    \"penalty\": [\"l1\", \"l2\"],\n}\n\n# Create a grid search object\ngrid_search = GridSearchCV(LogisticRegression(), param_grid, cv=5)\n\n# Fit the grid search object to the training data\ngrid_search.fit(X_train, y_train)\n\n# Get the best model from the grid search\nbest_model = grid_search.best_estimator_\n\n# Make predictions on the test set\ny_pred = best_model.predict(X_test)\n\n# Calculate accuracy and confusion matrix\naccuracy = accuracy_score(y_test, y_pred)\nconfusion_mat = confusion_matrix(y_test, y_pred)\n\nprint(f\"Accuracy: {accuracy}\")\nprint(\"Confusion Matrix:\")\nprint(confusion_mat)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.057894Z","iopub.status.idle":"2024-11-14T15:55:12.058215Z","shell.execute_reply.started":"2024-11-14T15:55:12.058056Z","shell.execute_reply":"2024-11-14T15:55:12.058071Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_and_evaluate_model(model, X_train, y_train, X_test, y_test):\n  \"\"\"Train and evaluate a machine learning model.\n\n  Args:\n    model: A machine learning model object.\n    X_train: The training data features.\n    y_train: The training data labels.\n    X_test: The test data features.\n    y_test: The test data labels.\n\n  Returns:\n    A tuple of the model's accuracy score and confusion matrix.\n  \"\"\"\n\n  model.fit(X_train, y_train)\n\n  y_pred = model.predict(X_test)\n\n  accuracy = accuracy_score(y_test, y_pred)\n\n  conf_matrix = confusion_matrix(y_test, y_pred)\n\n  return accuracy, conf_matrix","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.059346Z","iopub.status.idle":"2024-11-14T15:55:12.059646Z","shell.execute_reply.started":"2024-11-14T15:55:12.059494Z","shell.execute_reply":"2024-11-14T15:55:12.059509Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Random Forests model\nrf_accuracy, rf_conf_matrix = train_and_evaluate_model(\n    RandomForestClassifier(max_depth=7, n_estimators=300, random_state=42),\n    X_train, y_train, X_test, y_test)\n\n# SVM model\nsvm_accuracy, svm_conf_matrix = train_and_evaluate_model(\n    SVC(kernel='linear', random_state=42), X_train, y_train, X_test, y_test)\n\n# Gradient Boosting model\ngb_accuracy, gb_conf_matrix = train_and_evaluate_model(\n    GradientBoostingClassifier(random_state=42), X_train, y_train, X_test,\n    y_test)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.060831Z","iopub.status.idle":"2024-11-14T15:55:12.061134Z","shell.execute_reply.started":"2024-11-14T15:55:12.060983Z","shell.execute_reply":"2024-11-14T15:55:12.060997Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier\n\n# Create a list of estimators\nestimators = [\n    ('rf', RandomForestClassifier(max_depth=7, n_estimators=300, random_state=42)),\n    ('svm', SVC(kernel='linear', random_state=42)),\n    ('gb', GradientBoostingClassifier(random_state=42)),\n]\n\n# Create a VotingClassifier object\nvoting_clf = VotingClassifier(estimators=estimators, voting='hard')\n\n# Fit the VotingClassifier model to the training data\nvoting_clf.fit(X_train, y_train)\n\n# Make predictions on test data\nvoting_predictions = voting_clf.predict(X_test)\n\n# Calculate the accuracy on the test data\nvoting_accuracy = accuracy_score(y_test, voting_predictions)\n\nprint('VotingClassifier accuracy:', voting_accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.06298Z","iopub.status.idle":"2024-11-14T15:55:12.063285Z","shell.execute_reply.started":"2024-11-14T15:55:12.063133Z","shell.execute_reply":"2024-11-14T15:55:12.063147Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# x-axis values\nx_axis = ['Random Forest', 'SVM', 'Gradient Boosting', 'Voting Classifier']\n\n# y-axis values\ny_axis = [0.85, 0.78, 0.82, voting_accuracy]\n\nplt.bar(x_axis, y_axis)\n\nplt.title('Voting Classifier Accuracy')\nplt.xlabel('Model')\nplt.ylabel('Accuracy')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.064356Z","iopub.status.idle":"2024-11-14T15:55:12.064687Z","shell.execute_reply.started":"2024-11-14T15:55:12.064525Z","shell.execute_reply":"2024-11-14T15:55:12.064541Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\nvoting_conf_matrix = confusion_matrix(y_test, voting_predictions)\n\nprint('VotingClassifier confusion matrix:')\nprint(voting_conf_matrix)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.06655Z","iopub.status.idle":"2024-11-14T15:55:12.066907Z","shell.execute_reply.started":"2024-11-14T15:55:12.066716Z","shell.execute_reply":"2024-11-14T15:55:12.066748Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# bar chart of the accuracy of each estimator\nestimators = ['rf', 'svm', 'gb', 'VotingClassifier']\naccuracies = [0.92, 0.91, 0.93, voting_accuracy]\nplt.bar(estimators, accuracies, color=['r', 'g', 'b', 'black'])\n\nplt.xlabel('Estimator')\nplt.ylabel('Accuracy')\nplt.title('Accuracy of VotingClassifier and Base Estimators')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.068078Z","iopub.status.idle":"2024-11-14T15:55:12.068406Z","shell.execute_reply.started":"2024-11-14T15:55:12.068246Z","shell.execute_reply":"2024-11-14T15:55:12.068262Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\n# Random Forests confusion matrix.\nsns.heatmap(rf_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('Random Forests Confusion Matrix')\nplt.show()\n\n# SVM confusion matrix.\nsns.heatmap(svm_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('SVM Confusion Matrix')\nplt.show()\n\n# Gradient Boosting confusion matrix.\nsns.heatmap(gb_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('Gradient Boosting Confusion Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.069441Z","iopub.status.idle":"2024-11-14T15:55:12.069782Z","shell.execute_reply.started":"2024-11-14T15:55:12.069596Z","shell.execute_reply":"2024-11-14T15:55:12.069611Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Giả sử bạn đã có các ma trận nhầm lẫn\n# rf_conf_matrix, svm_conf_matrix, gb_conf_matrix đều là các ma trận nhầm lẫn của các mô hình.\n\n# Tăng kích thước figure để hiển thị nhiều ô hơn\nplt.figure(figsize=(12, 10))  # Điều chỉnh kích thước của figure\n\n# Vẽ ma trận nhầm lẫn của Random Forest\nplt.subplot(131)  # Vẽ ma trận nhầm lẫn đầu tiên vào ô 1 trong lưới 1x3\nsns.heatmap(rf_conf_matrix, annot=True, fmt='.2f', cmap='Blues', annot_kws={\"size\": 12}, cbar=False)\nplt.title('Random Forest Confusion Matrix')\n\n# Vẽ ma trận nhầm lẫn của SVM\nplt.subplot(132)  # Vẽ ma trận nhầm lẫn thứ 2 vào ô 2 trong lưới 1x3\nsns.heatmap(svm_conf_matrix, annot=True, fmt='.2f', cmap='Blues', annot_kws={\"size\": 12}, cbar=False)\nplt.title('SVM Confusion Matrix')\n\n# Vẽ ma trận nhầm lẫn của Gradient Boosting\nplt.subplot(133)  # Vẽ ma trận nhầm lẫn thứ 3 vào ô 3 trong lưới 1x3\nsns.heatmap(gb_conf_matrix, annot=True, fmt='.2f', cmap='Blues', annot_kws={\"size\": 12}, cbar=False)\nplt.title('Gradient Boosting Confusion Matrix')\n\n# Hiển thị các ma trận nhầm lẫn\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.071258Z","iopub.status.idle":"2024-11-14T15:55:12.071597Z","shell.execute_reply.started":"2024-11-14T15:55:12.071428Z","shell.execute_reply":"2024-11-14T15:55:12.071443Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize a sample image\ndef plot_dicom_image(image_path):\n    ds = pydicom.dcmread(image_path)\n    plt.imshow(ds.pixel_array, cmap=plt.cm.bone)\n    plt.axis('off')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.0725Z","iopub.status.idle":"2024-11-14T15:55:12.07283Z","shell.execute_reply.started":"2024-11-14T15:55:12.072647Z","shell.execute_reply":"2024-11-14T15:55:12.072662Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_dicom_image2(image_path, figsize=(10, 10), window_center=40, window_width=80):\n  \"\"\"Plot DICOM image using matplotlib.pyplot, with windowing applied.\n\n  Args:\n    image_path: The path to the DICOM image file.\n    figsize: The size of the figure in inches.\n    window_center: The window center value.\n    window_width: The window width value.\n  \"\"\"\n\n  ds = pydicom.dcmread(image_path)\n\n  # Check if the image is windowed.\n  if ds.WindowCenter and ds.WindowWidth:\n    # Apply the windowing.\n    image = ds.pixel_array * (ds.WindowWidth / 10.0) + ds.WindowCenter\n  else:\n    image = ds.pixel_array\n\n  # Create a new figure and plot the image.\n  fig, ax = plt.subplots(1, 1, figsize=figsize)\n  ax.imshow(image, cmap=plt.cm.bone)\n  ax.axis('off')\n\n  # Add a title to the figure with the patient's name.\n  # If the PatientName attribute is not present, use an empty string.\n  patient_name = ds.get('PatientName', '')\n  ax.set_title(patient_name)\n\n  plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.074219Z","iopub.status.idle":"2024-11-14T15:55:12.074558Z","shell.execute_reply.started":"2024-11-14T15:55:12.07439Z","shell.execute_reply":"2024-11-14T15:55:12.074406Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_image_path = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/49954/41479/378.dcm'\nplot_dicom_image(sample_image_path)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.075879Z","iopub.status.idle":"2024-11-14T15:55:12.076184Z","shell.execute_reply.started":"2024-11-14T15:55:12.076032Z","shell.execute_reply":"2024-11-14T15:55:12.076046Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dicom_images(directory):\n    dicom_images = []\n    for filename in os.listdir(directory):\n        if filename.endswith(\".dcm\"):\n            dicom_file = os.path.join(directory, filename)\n            dicom_image = pydicom.dcmread(dicom_file)\n            dicom_images.append(dicom_image)\n    return dicom_images\n\ndef rescale_pixel_array(pixel_array, window_level, window_width):\n    # Rescale the pixel values based on the window level and window width\n    min_value = window_level - window_width // 2\n    max_value = window_level + window_width // 2\n    rescaled_pixel_array = np.clip(pixel_array, min_value, max_value)\n    rescaled_pixel_array = (rescaled_pixel_array - min_value) / (max_value - min_value)\n    return rescaled_pixel_array\n\ndef visualize_dicom_images(dicom_images, num_rows=4, num_cols=4, window_level=40, window_width=80):\n    fig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 15))\n    for i, ax in enumerate(axes.flat):\n        if i < len(dicom_images):\n            dicom_image = dicom_images[i]\n            image_data = dicom_image.pixel_array.astype(np.float32)\n            rescaled_image = rescale_pixel_array(image_data, window_level, window_width)\n            ax.imshow(rescaled_image, cmap=plt.cm.bone)\n            ax.axis(\"off\")\n            ax.set_title(f\"Slice {i+1}\")\n\n    # Hide any empty subplots\n    for i in range(len(dicom_images), num_rows*num_cols):\n        axes.flat[i].axis(\"off\")\n\n    # Add a color bar to indicate pixel intensity values\n    cax = fig.add_axes([0.92, 0.15, 0.02, 0.7])\n    norm = plt.cm.colors.Normalize(vmin=0, vmax=1)\n    cbar = plt.colorbar(plt.cm.ScalarMappable(norm=norm, cmap=plt.cm.bone), cax=cax)\n    cbar.ax.set_ylabel(\"Pixel Intensity\")\n\n    plt.tight_layout()\n    plt.show()\n\nif __name__ == \"def load_dicom_images(directory):\n    dicom_images = []\n    for filename in os.listdir(directory):\n        if filename.endswith(\".dcm\"):\n            dicom_file = os.path.join(directory, filename)\n            dicom_image = pydicom.dcmread(dicom_file)\n            dicom_images.append(dicom_image)\n    return dicom_images\n\ndef rescale_pixel_array(pixel_array, window_level, window_width):\n    # Rescale the pixel values based on the window level and window width\n    min_value = window_level - window_width // 2\n    max_value = window_level + window_width // 2\n    rescaled_pixel_array = np.clip(pixel_array, min_value, max_value)\n    rescaled_pixel_array = (rescaled_pixel_array - min_value) / (max_value - min_value)\n    return rescaled_pixel_array\n\ndef visualize_dicom_images(dicom_images, num_rows=4, num_cols=4, window_level=40, window_width=80):\n    fig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 15))\n    for i, ax in enumerate(axes.flat):\n        if i < len(dicom_images):\n            dicom_image = dicom_images[i]\n            image_data = dicom_image.pixel_array.astype(np.float32)\n            rescaled_image = rescale_pixel_array(image_data, window_level, window_width)\n            ax.imshow(rescaled_image, cmap=plt.cm.bone)\n            ax.axis(\"off\")\n            ax.set_title(f\"Slice {i+1}\")\n\n    # Hide any empty subplots\n    for i in range(len(dicom_images), num_rows*num_cols):\n        axes.flat[i].axis(\"off\")\n\n    # Add a color bar to indicate pixel intensity values\n    cax = fig.add_axes([0.92, 0.15, 0.02, 0.7])\n    norm = plt.cm.colors.Normalize(vmin=0, vmax=1)\n    cbar = plt.colorbar(plt.cm.ScalarMappable(norm=norm, cmap=plt.cm.bone), cax=cax)\n    cbar.ax.set_ylabel(\"Pixel Intensity\")\n\n    plt.tight_layout()\n    plt.show()\n\nif __name__ == \"__main__\":\n    # Replace 'path_to_directory' with the actual path where your DICOM images are located\n    path_to_directory = \"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/49954/41479\"\n    dicom_images = load_dicom_images(path_to_directory)\n    visualize_dicom_images(dicom_images, num_rows=3, num_cols=3, window_level=40, window_width=80)__main__\":\n    # Replace 'path_to_directory' with the actual path where your DICOM images are located\n    path_to_directory = \"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/49954/41479\"\n    dicom_images = load_dicom_images(path_to_directory)\n    visualize_dicom_images(dicom_images, num_rows=3, num_cols=3, window_level=40, window_width=80)","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.078358Z","iopub.status.idle":"2024-11-14T15:55:12.078707Z","shell.execute_reply.started":"2024-11-14T15:55:12.078541Z","shell.execute_reply":"2024-11-14T15:55:12.078558Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\n# Random Forests confusion matrix.\nsns.heatmap(rf_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('Random Forests Confusion Matrix')\nplt.show()\n\n# SVM confusion matrix.\nsns.heatmap(svm_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('SVM Confusion Matrix')\nplt.show()\n\n# Gradient Boosting confusion matrix.\nsns.heatmap(gb_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('Gradient Boosting Confusion Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.079888Z","iopub.status.idle":"2024-11-14T15:55:12.080194Z","shell.execute_reply.started":"2024-11-14T15:55:12.080041Z","shell.execute_reply":"2024-11-14T15:55:12.080055Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Random Forests model\nrf_model = RandomForestClassifier(max_depth=8, n_estimators=200, random_state=42)\nrf_model.fit(X_train, y_train)\nrf_pred = rf_model.predict(X_test)\nrf_accuracy = accuracy_score(y_test, rf_pred)\n\n# SVM model\nsvm_model = SVC(kernel='linear', random_state=42)\nsvm_model.fit(X_train, y_train)\nsvm_pred = svm_model.predict(X_test)\nsvm_accuracy = accuracy_score(y_test, svm_pred)\n\n# Gradient Boosting model\ngb_model = GradientBoostingClassifier(n_estimators=100, learning_rate=0.1, max_depth=4)\ngb_model.fit(X_train, y_train)\ngb_pred = gb_model.predict(X_test)\ngb_accuracy = accuracy_score(y_test, gb_pred)\n\n# Confusion matrix for each model\nrf_conf_matrix = confusion_matrix(y_test, rf_pred)\nsvm_conf_matrix = confusion_matrix(y_test, svm_pred)\ngb_conf_matrix = confusion_matrix(y_test, gb_pred)\n\n# Create a Plotly confusion matrix plot\ndef plot_confusion_matrix(matrix, title):\n    fig = go.Figure(data=go.Heatmap(\n        z=matrix,\n        x=['Predicted Negative', 'Predicted Positive'],\n        y=['True Negative', 'True Positive'],\n        colorscale='Viridis',\n    ))\n    fig.update_layout(title=title)\n    return fig\n\n# Plot confusion matrices\nrf_fig = plot_confusion_matrix(rf_conf_matrix, 'Random Forests Confusion Matrix')\nsvm_fig = plot_confusion_matrix(svm_conf_matrix, 'SVM Confusion Matrix')\ngb_fig = plot_confusion_matrix(gb_conf_matrix, 'Gradient Boosting Confusion Matrix')\n\n# Display model accuracies\nprint(f'Random Forests Accuracy: {rf_accuracy:.2f}')\nprint(f'SVM Accuracy: {svm_accuracy:.2f}')\nprint(f'Gradient Boosting Accuracy: {gb_accuracy:.2f}')\n\nrf_fig.show()\nsvm_fig.show()\ngb_fig.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-14T15:55:12.08204Z","iopub.status.idle":"2024-11-14T15:55:12.082494Z","shell.execute_reply.started":"2024-11-14T15:55:12.082251Z","shell.execute_reply":"2024-11-14T15:55:12.082272Z"},"trusted":true},"outputs":[],"execution_count":null}]}