{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":52254,"databundleVersionId":9674523,"sourceType":"competition"},{"sourceId":6211844,"sourceType":"datasetVersion","datasetId":3567114},{"sourceId":6267500,"sourceType":"datasetVersion","datasetId":3581068},{"sourceId":6605331,"sourceType":"datasetVersion","datasetId":3695414},{"sourceId":5838650,"sourceType":"datasetVersion","datasetId":3356068},{"sourceId":6642094,"sourceType":"datasetVersion","datasetId":3834301},{"sourceId":6440340,"sourceType":"datasetVersion","datasetId":3716919},{"sourceId":6437750,"sourceType":"datasetVersion","datasetId":3715314},{"sourceId":6694678,"sourceType":"datasetVersion","datasetId":3859765},{"sourceId":6694288,"sourceType":"datasetVersion","datasetId":3859591},{"sourceId":6666397,"sourceType":"datasetVersion","datasetId":3846730},{"sourceId":6367091,"sourceType":"datasetVersion","datasetId":3668143},{"sourceId":6695864,"sourceType":"datasetVersion","datasetId":3860267},{"sourceId":6678504,"sourceType":"datasetVersion","datasetId":3852993},{"sourceId":6681564,"sourceType":"datasetVersion","datasetId":3854378},{"sourceId":6681310,"sourceType":"datasetVersion","datasetId":3854197},{"sourceId":6670204,"sourceType":"datasetVersion","datasetId":3848762},{"sourceId":6533062,"sourceType":"datasetVersion","datasetId":3776927},{"sourceId":6504435,"sourceType":"datasetVersion","datasetId":3760213},{"sourceId":6678520,"sourceType":"datasetVersion","datasetId":3853000},{"sourceId":146675749,"sourceType":"kernelVersion"},{"sourceId":146649063,"sourceType":"kernelVersion"},{"sourceId":146650289,"sourceType":"kernelVersion"}],"dockerImageVersionId":30554,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center>Multi-label classification of abdominal trauma from CT images</center>\n\n","metadata":{"papermill":{"duration":0.025483,"end_time":"2022-02-01T10:21:43.84374","exception":false,"start_time":"2022-02-01T10:21:43.818257","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## To commence this project, the neccesary libraries need to be installed.\n## We would be using the **TensorFlow** framework and the pretrained **EfficientNetB1**\n","metadata":{}},{"cell_type":"code","source":"# Imporitng libraries\nimport pandas as pd\nimport os\nimport random\nimport numpy as np\nimport tensorflow as tf\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom IPython.display import clear_output\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold\n\n\nrandom.seed(42)","metadata":{"id":"fUPKjZaaJHkM","papermill":{"duration":5.021373,"end_time":"2022-02-01T10:21:48.88737","exception":false,"start_time":"2022-02-01T10:21:43.865997","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:21.910484Z","iopub.execute_input":"2024-11-23T05:17:21.910822Z","iopub.status.idle":"2024-11-23T05:17:39.350363Z","shell.execute_reply.started":"2024-11-23T05:17:21.910776Z","shell.execute_reply":"2024-11-23T05:17:39.349678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Loading the training dataset\ntrain_img = \"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train_images\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:39.352154Z","iopub.execute_input":"2024-11-23T05:17:39.352724Z","iopub.status.idle":"2024-11-23T05:17:39.356748Z","shell.execute_reply.started":"2024-11-23T05:17:39.352692Z","shell.execute_reply":"2024-11-23T05:17:39.355861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in os.listdir(train_img):\n    for j in os.listdir(os.path.join(train_img,i)):\n        print(len(os.listdir(os.path.join(train_img,i,j))))","metadata":{"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:39.357873Z","iopub.execute_input":"2024-11-23T05:17:39.358097Z","iopub.status.idle":"2024-11-23T05:17:43.148842Z","shell.execute_reply.started":"2024-11-23T05:17:39.358078Z","shell.execute_reply":"2024-11-23T05:17:43.147687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install imbalanced-learn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:43.151585Z","iopub.execute_input":"2024-11-23T05:17:43.152251Z","iopub.status.idle":"2024-11-23T05:17:54.032692Z","shell.execute_reply.started":"2024-11-23T05:17:43.152203Z","shell.execute_reply":"2024-11-23T05:17:54.031632Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part A\n## Import and explore the data","metadata":{"papermill":{"duration":0.02193,"end_time":"2022-02-01T10:21:48.93356","exception":false,"start_time":"2022-02-01T10:21:48.91163","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Making a list containing all unique classes in the training set\n\n# Reading the training labels\ntraining_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")","metadata":{"id":"PBlH8ae3LbG0","papermill":{"duration":0.069238,"end_time":"2022-02-01T10:21:49.024476","exception":false,"start_time":"2022-02-01T10:21:48.955238","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.034257Z","iopub.execute_input":"2024-11-23T05:17:54.034623Z","iopub.status.idle":"2024-11-23T05:17:54.065785Z","shell.execute_reply.started":"2024-11-23T05:17:54.034587Z","shell.execute_reply":"2024-11-23T05:17:54.065178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Viewing the dataset in a structured format\ntraining_labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.066751Z","iopub.execute_input":"2024-11-23T05:17:54.067065Z","iopub.status.idle":"2024-11-23T05:17:54.097001Z","shell.execute_reply.started":"2024-11-23T05:17:54.067042Z","shell.execute_reply":"2024-11-23T05:17:54.095973Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part B\n## In the next few cells, we chose to explore the dataframe to check for missing values","metadata":{}},{"cell_type":"code","source":"training_labels.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.098099Z","iopub.execute_input":"2024-11-23T05:17:54.098395Z","iopub.status.idle":"2024-11-23T05:17:54.105416Z","shell.execute_reply.started":"2024-11-23T05:17:54.098363Z","shell.execute_reply":"2024-11-23T05:17:54.104573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['patient_id'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.106625Z","iopub.execute_input":"2024-11-23T05:17:54.10732Z","iopub.status.idle":"2024-11-23T05:17:54.122571Z","shell.execute_reply.started":"2024-11-23T05:17:54.107278Z","shell.execute_reply":"2024-11-23T05:17:54.121795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['bowel_healthy'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.123479Z","iopub.execute_input":"2024-11-23T05:17:54.12371Z","iopub.status.idle":"2024-11-23T05:17:54.133007Z","shell.execute_reply.started":"2024-11-23T05:17:54.123691Z","shell.execute_reply":"2024-11-23T05:17:54.132265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['bowel_injury'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.135875Z","iopub.execute_input":"2024-11-23T05:17:54.136147Z","iopub.status.idle":"2024-11-23T05:17:54.146282Z","shell.execute_reply.started":"2024-11-23T05:17:54.136125Z","shell.execute_reply":"2024-11-23T05:17:54.145426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['extravasation_healthy'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.147193Z","iopub.execute_input":"2024-11-23T05:17:54.147493Z","iopub.status.idle":"2024-11-23T05:17:54.159649Z","shell.execute_reply.started":"2024-11-23T05:17:54.147473Z","shell.execute_reply":"2024-11-23T05:17:54.158986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['extravasation_injury'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.160524Z","iopub.execute_input":"2024-11-23T05:17:54.160746Z","iopub.status.idle":"2024-11-23T05:17:54.174166Z","shell.execute_reply.started":"2024-11-23T05:17:54.160727Z","shell.execute_reply":"2024-11-23T05:17:54.173331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_healthy'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.175373Z","iopub.execute_input":"2024-11-23T05:17:54.175705Z","iopub.status.idle":"2024-11-23T05:17:54.187585Z","shell.execute_reply.started":"2024-11-23T05:17:54.175682Z","shell.execute_reply":"2024-11-23T05:17:54.186853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_low'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.188535Z","iopub.execute_input":"2024-11-23T05:17:54.188794Z","iopub.status.idle":"2024-11-23T05:17:54.199248Z","shell.execute_reply.started":"2024-11-23T05:17:54.188764Z","shell.execute_reply":"2024-11-23T05:17:54.198476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['kidney_high'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.200105Z","iopub.execute_input":"2024-11-23T05:17:54.200295Z","iopub.status.idle":"2024-11-23T05:17:54.212143Z","shell.execute_reply.started":"2024-11-23T05:17:54.200279Z","shell.execute_reply":"2024-11-23T05:17:54.211229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_healthy'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.213072Z","iopub.execute_input":"2024-11-23T05:17:54.213294Z","iopub.status.idle":"2024-11-23T05:17:54.224367Z","shell.execute_reply.started":"2024-11-23T05:17:54.213276Z","shell.execute_reply":"2024-11-23T05:17:54.223596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_low'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.225572Z","iopub.execute_input":"2024-11-23T05:17:54.225891Z","iopub.status.idle":"2024-11-23T05:17:54.237684Z","shell.execute_reply.started":"2024-11-23T05:17:54.225864Z","shell.execute_reply":"2024-11-23T05:17:54.236867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['liver_high'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.238676Z","iopub.execute_input":"2024-11-23T05:17:54.238927Z","iopub.status.idle":"2024-11-23T05:17:54.248708Z","shell.execute_reply.started":"2024-11-23T05:17:54.238907Z","shell.execute_reply":"2024-11-23T05:17:54.248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_healthy'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.249581Z","iopub.execute_input":"2024-11-23T05:17:54.249867Z","iopub.status.idle":"2024-11-23T05:17:54.261068Z","shell.execute_reply.started":"2024-11-23T05:17:54.249841Z","shell.execute_reply":"2024-11-23T05:17:54.260351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_low'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.26195Z","iopub.execute_input":"2024-11-23T05:17:54.262361Z","iopub.status.idle":"2024-11-23T05:17:54.272283Z","shell.execute_reply.started":"2024-11-23T05:17:54.26234Z","shell.execute_reply":"2024-11-23T05:17:54.271576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['spleen_high'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.273304Z","iopub.execute_input":"2024-11-23T05:17:54.273586Z","iopub.status.idle":"2024-11-23T05:17:54.283484Z","shell.execute_reply.started":"2024-11-23T05:17:54.273557Z","shell.execute_reply":"2024-11-23T05:17:54.282898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels['any_injury'].isna().value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.284481Z","iopub.execute_input":"2024-11-23T05:17:54.284705Z","iopub.status.idle":"2024-11-23T05:17:54.299917Z","shell.execute_reply.started":"2024-11-23T05:17:54.284686Z","shell.execute_reply":"2024-11-23T05:17:54.299071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Using the EfficientNetB1 model and tweaking the last two layers to suit our work\n#def create_model(decay_steps=10,warmup_steps=10):\n#    base_model = tf.keras.applications.EfficientNetB1(\n#    weights= \"imagenet\", include_top=False, input_shape= (512,512,3)\n#    )\n#    num_classes=61\n\n#    x = base_model.output\n#    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n#    x = tf.keras.layers.Dropout(0.2)(x)\n#    x_bowel = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_extra = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_liver = tf.keras.layers.Dense(32, activation='silu')(x)\n #   x_kidney = tf.keras.layers.Dense(32, activation='silu')(x)\n#    x_spleen = tf.keras.layers.Dense(32, activation='silu')(x)\n\n    # Define heads\n#    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel) # use sigmoid to convert predictions to [0-1]\n#    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra) # use sigmoid to convert predictions to [0-1]\n#    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver) # use softmax for the liver head\n#    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney) # use softmax for the kidney head\n#    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen) # use softmax for the spleen head\n\n#    model = tf.keras.Model(inputs = base_model.input, outputs = [out_bowel,out_extra,out_liver,out_kidney,out_spleen])\n        # Cosine Decay\n#    cosine_decay = tf.keras.optimizers.schedules.CosineDecay(\n#        initial_learning_rate=1e-4,\n#        decay_steps=decay_steps,\n#        alpha=0.0,\n        #warmup_target=1e-3,\n        #warmup_steps=warmup_steps,\n #   )\n\n    # Compile the model\n #   optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n #   loss = [\n #       tf.keras.losses.BinaryCrossentropy(),\n #       tf.keras.losses.BinaryCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy(),\n #       tf.keras.losses.CategoricalCrossentropy()]\n    \n #   metrics = [\n #       [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n #       [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n  #      [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n #       [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n #       [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]]\n    #    \"bowel\":[\"accuracy\"],\n    #    \"extra\":[\"accuracy\"],\n    #    \"liver\":[\"accuracy\"],\n    #    \"kidney\":[\"accuracy\"],\n    #    \"spleen\":[\"accuracy\"],\n    #}\n  #  print(\"[INFO] Compiling the model...\")\n #   model.compile(\n #       optimizer=optimizer,\n #     loss=loss,\n #     metrics=metrics\n #   )\n #   return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.301275Z","iopub.execute_input":"2024-11-23T05:17:54.301569Z","iopub.status.idle":"2024-11-23T05:17:54.308992Z","shell.execute_reply.started":"2024-11-23T05:17:54.301542Z","shell.execute_reply":"2024-11-23T05:17:54.308353Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"+ Sử dụng Gradient Clipping để tránh gradient quá lớn:","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\n\n\ndef create_model(decay_steps=900, warmup_steps=100, fine_tune_from=100):\n    # Tải EfficientNetB1 làm base model\n    base_model = tf.keras.applications.EfficientNetB1(\n        weights=\"imagenet\", include_top=False, input_shape=(512, 512, 3)\n    )\n\n    # Fine-tune một phần của base model\n    for layer in base_model.layers[:fine_tune_from]:\n        layer.trainable = False\n\n    x = base_model.output\n\n    # Lớp Convolution với kernel 3x3 và tăng số lượng filters\n    x = tf.keras.layers.Conv2D(128, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n    \n    x = tf.keras.layers.Conv2D(256, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(512, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(1024, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    # Residual Block\n    residual = x\n    x = tf.keras.layers.Conv2D(1024, (2, 2), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Add()([x, residual])  # Skip connection\n\n    # Dense Block với dropout cao hơn và số lượng units lớn hơn\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dropout(0.6)(x)  # Tăng dropout\n\n    # Dense layers cho các đầu ra\n    x_bowel = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_extra = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_liver = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_kidney = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_spleen = tf.keras.layers.Dense(512, activation='relu')(x)\n\n    # Output Layers\n    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel)  # Bowel (binary)\n    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra)  # Extra (binary)\n    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver)  # Liver (multiclass)\n    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney)  # Kidney (multiclass)\n    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen)  # Spleen (multiclass)\n\n    # Create model\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n    \n    # Cosine Decay with a warmup phase\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecayRestarts(\n        initial_learning_rate=3e-4,  # Lower initial learning rate\n        first_decay_steps=warmup_steps,\n        t_mul=1.8,\n        m_mul=0.8,\n        alpha=0.2\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]\n    ]\n    \n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n        loss=loss,\n        metrics=metrics\n    )\n    \n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.310097Z","iopub.execute_input":"2024-11-23T05:17:54.310406Z","iopub.status.idle":"2024-11-23T05:17:54.327228Z","shell.execute_reply.started":"2024-11-23T05:17:54.310377Z","shell.execute_reply":"2024-11-23T05:17:54.32656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#model = create_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.327993Z","iopub.execute_input":"2024-11-23T05:17:54.328187Z","iopub.status.idle":"2024-11-23T05:17:54.344181Z","shell.execute_reply.started":"2024-11-23T05:17:54.328169Z","shell.execute_reply":"2024-11-23T05:17:54.343562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = create_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:54.344997Z","iopub.execute_input":"2024-11-23T05:17:54.345187Z","iopub.status.idle":"2024-11-23T05:17:59.948919Z","shell.execute_reply.started":"2024-11-23T05:17:54.345171Z","shell.execute_reply":"2024-11-23T05:17:59.948057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# In ra tóm tắt mô hình\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:17:59.949908Z","iopub.execute_input":"2024-11-23T05:17:59.95016Z","iopub.status.idle":"2024-11-23T05:18:00.712636Z","shell.execute_reply.started":"2024-11-23T05:17:59.950138Z","shell.execute_reply":"2024-11-23T05:18:00.711775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.utils.plot_model(\n    model,  # Sử dụng đúng tên biến của mô hình\n    to_file='model.png'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:00.71832Z","iopub.execute_input":"2024-11-23T05:18:00.718594Z","iopub.status.idle":"2024-11-23T05:18:03.211758Z","shell.execute_reply.started":"2024-11-23T05:18:00.71857Z","shell.execute_reply":"2024-11-23T05:18:03.210869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_labels = training_labels.set_index('patient_id')\ntraining_labels.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:03.213518Z","iopub.execute_input":"2024-11-23T05:18:03.213986Z","iopub.status.idle":"2024-11-23T05:18:03.230947Z","shell.execute_reply.started":"2024-11-23T05:18:03.213943Z","shell.execute_reply":"2024-11-23T05:18:03.230061Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part C\n## Balancing the imbalanced training dataset ","metadata":{}},{"cell_type":"code","source":"#Using histogram to get the distribution of the labels and to check if there are outliers and wrong labels\ntraining_labels.hist(figsize=(20,12),bins=2)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:03.232124Z","iopub.execute_input":"2024-11-23T05:18:03.232687Z","iopub.status.idle":"2024-11-23T05:18:05.245747Z","shell.execute_reply.started":"2024-11-23T05:18:03.232656Z","shell.execute_reply":"2024-11-23T05:18:05.244953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#images ,labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:05.247136Z","iopub.execute_input":"2024-11-23T05:18:05.247497Z","iopub.status.idle":"2024-11-23T05:18:05.252053Z","shell.execute_reply.started":"2024-11-23T05:18:05.247465Z","shell.execute_reply":"2024-11-23T05:18:05.251181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img,i))[0]\n    file = os.listdir(os.path.join(train_img,i,folder))[0]\n    images.append(cv2.imread(os.path.join(train_img,i,folder,file), cv2.IMREAD_COLOR ))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:05.253159Z","iopub.execute_input":"2024-11-23T05:18:05.253479Z","iopub.status.idle":"2024-11-23T05:18:09.193111Z","shell.execute_reply.started":"2024-11-23T05:18:05.253448Z","shell.execute_reply":"2024-11-23T05:18:09.191709Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Examples of images","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30,12))\nfor i in range(10):\n    plt.subplot(2,5,i+1)\n    plt.imshow(images[i])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:09.194758Z","iopub.execute_input":"2024-11-23T05:18:09.195205Z","iopub.status.idle":"2024-11-23T05:18:11.182681Z","shell.execute_reply.started":"2024-11-23T05:18:09.195145Z","shell.execute_reply":"2024-11-23T05:18:11.181714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:11.183908Z","iopub.execute_input":"2024-11-23T05:18:11.184197Z","iopub.status.idle":"2024-11-23T05:18:11.188672Z","shell.execute_reply.started":"2024-11-23T05:18:11.184174Z","shell.execute_reply":"2024-11-23T05:18:11.187645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:11.189889Z","iopub.execute_input":"2024-11-23T05:18:11.190191Z","iopub.status.idle":"2024-11-23T05:18:11.262518Z","shell.execute_reply.started":"2024-11-23T05:18:11.190167Z","shell.execute_reply":"2024-11-23T05:18:11.261485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.over_sampling import RandomOverSampler\n\n#oversampler = RandomOverSampler(random_state=42)\n#new_images, new_labels = oversampler.fit_resample(images, labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:11.263631Z","iopub.execute_input":"2024-11-23T05:18:11.263936Z","iopub.status.idle":"2024-11-23T05:18:12.096152Z","shell.execute_reply.started":"2024-11-23T05:18:11.263912Z","shell.execute_reply":"2024-11-23T05:18:12.095241Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Part D\n## Training the model with the training dataset","metadata":{}},{"cell_type":"code","source":"bowel_labels = labels[:,2]\nextravasation_labels = labels[:,4]\nkidney_labels = labels[:,4:7]\nliver_labels = labels[:,7:10]\nspleen_labels = labels[:,10:13]\nany_labels = labels[:,-1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:12.097406Z","iopub.execute_input":"2024-11-23T05:18:12.098883Z","iopub.status.idle":"2024-11-23T05:18:12.104865Z","shell.execute_reply.started":"2024-11-23T05:18:12.098836Z","shell.execute_reply":"2024-11-23T05:18:12.103903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bowel_val = y_val[:,2]\nextravasation_val = y_val[:,4]\nkidney_val = y_val[:,4:7]\nliver_val = y_val[:,7:10]\nspleen_val = y_val[:,10:13]\nany_val = y_val[:,-1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:12.106159Z","iopub.execute_input":"2024-11-23T05:18:12.106496Z","iopub.status.idle":"2024-11-23T05:18:12.11851Z","shell.execute_reply.started":"2024-11-23T05:18:12.106469Z","shell.execute_reply":"2024-11-23T05:18:12.117429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bowel_train = y_train[:,2]\nextravasation_train = y_train[:,4]\nkidney_train = y_train[:,4:7]\nliver_train = y_train[:,7:10]\nspleen_train = y_train[:,10:13]\nany_train = y_train[:,-1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:12.119999Z","iopub.execute_input":"2024-11-23T05:18:12.120364Z","iopub.status.idle":"2024-11-23T05:18:12.135496Z","shell.execute_reply.started":"2024-11-23T05:18:12.120328Z","shell.execute_reply":"2024-11-23T05:18:12.13465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#batch_size = 8\n#num_epoch = 2\n#history = model.fit(x=X_train,y=[bowel_train,extravasation_train,kidney_train,liver_train,spleen_train],batch_size=batch_size, epochs=num_epoch, verbose=1, validation_data=(X_val,[bowel_val,extravasation_val,kidney_val,liver_val,spleen_val]))\n\n\n#validation_data=(X_val,[bowel_val,extravasation_val,kidney_val,liver_val,spleen_val])","metadata":{"papermill":{"duration":1176.379334,"end_time":"2022-02-01T14:07:30.902597","exception":false,"start_time":"2022-02-01T13:47:54.523263","status":"completed"},"tags":[],"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:12.136726Z","iopub.execute_input":"2024-11-23T05:18:12.137023Z","iopub.status.idle":"2024-11-23T05:18:12.147316Z","shell.execute_reply.started":"2024-11-23T05:18:12.136999Z","shell.execute_reply":"2024-11-23T05:18:12.146035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install keras-rectified-adam\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:12.148751Z","iopub.execute_input":"2024-11-23T05:18:12.149073Z","iopub.status.idle":"2024-11-23T05:18:22.838769Z","shell.execute_reply.started":"2024-11-23T05:18:12.14905Z","shell.execute_reply":"2024-11-23T05:18:22.837594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nimport tensorflow as tf\n\n\n\ndef create_model(decay_steps=900, warmup_steps=60, fine_tune_from=100):\n    # Tải EfficientNetB1 làm base model\n    base_model = tf.keras.applications.EfficientNetB1(\n        weights=\"imagenet\", include_top=False, input_shape=(512, 512, 3)\n    )\n\n    # Fine-tune một phần của base model\n    for layer in base_model.layers[:fine_tune_from]:\n        layer.trainable = False\n\n    x = base_model.output\n\n    # Lớp Convolution với kernel 3x3 và tăng số lượng filters\n    x = tf.keras.layers.Conv2D(128, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n    \n    x = tf.keras.layers.Conv2D(256, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(512, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    x = tf.keras.layers.Conv2D(1024, (5, 5), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.MaxPooling2D((2, 2), padding='same')(x)\n\n    # Residual Block\n    residual = x\n    x = tf.keras.layers.Conv2D(1024, (2, 2), activation='relu', padding='same')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Add()([x, residual])  # Skip connection\n\n    # Dense Block với dropout cao hơn và số lượng units lớn hơn\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    x = tf.keras.layers.Dropout(0.6)(x)  # Tăng dropout\n\n    # Dense layers cho các đầu ra\n    x_bowel = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_extra = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_liver = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_kidney = tf.keras.layers.Dense(512, activation='relu')(x)\n    x_spleen = tf.keras.layers.Dense(512, activation='relu')(x)\n\n    # Output Layers\n    out_bowel = tf.keras.layers.Dense(1, name='bowel', activation='sigmoid')(x_bowel)  # Bowel (binary)\n    out_extra = tf.keras.layers.Dense(1, name='extra', activation='sigmoid')(x_extra)  # Extra (binary)\n    out_liver = tf.keras.layers.Dense(3, name='liver', activation='softmax')(x_liver)  # Liver (multiclass)\n    out_kidney = tf.keras.layers.Dense(3, name='kidney', activation='softmax')(x_kidney)  # Kidney (multiclass)\n    out_spleen = tf.keras.layers.Dense(3, name='spleen', activation='softmax')(x_spleen)  # Spleen (multiclass)\n\n    # Create model\n    model = tf.keras.Model(inputs=base_model.input, outputs=[out_bowel, out_extra, out_liver, out_kidney, out_spleen])\n    \n    # Cosine Decay with a warmup phase\n    cosine_decay = tf.keras.optimizers.schedules.CosineDecayRestarts(\n        initial_learning_rate=3e-4,  # Lower initial learning rate\n        first_decay_steps=warmup_steps,\n        t_mul=1.8,\n        m_mul=0.8,\n        alpha=0.2\n    )\n\n    # Compile the model\n    optimizer = tf.keras.optimizers.Adam(learning_rate=cosine_decay)\n    loss = [\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.BinaryCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy(),\n        tf.keras.losses.CategoricalCrossentropy()\n    ]\n    \n    metrics = [\n        [tf.keras.metrics.BinaryAccuracy(name=\"bowel_binary_accuracy\")],\n        [tf.keras.metrics.BinaryAccuracy(name=\"extra_binary_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"liver_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"kidney_cat_accuracy\")],\n        [tf.keras.metrics.CategoricalAccuracy(name=\"spleen_cat_accuracy\")]\n    ]\n    \n    print(\"[INFO] Compiling the model...\")\n    model.compile(\n        optimizer=optimizer,\n        loss=loss,\n        metrics=metrics\n    )\n    \n    return model\n\n\n# Load images and labels\nimages = []\nlabels = []\nfor i in sorted(os.listdir(train_img)):\n    folder = os.listdir(os.path.join(train_img, i))[0]\n    file = os.listdir(os.path.join(train_img, i, folder))[0]\n    images.append(cv2.imread(os.path.join(train_img, i, folder, file), cv2.IMREAD_COLOR))\n    labels.append(np.asarray(training_labels.loc[int(i)]))\nimages = np.asarray(images)\nlabels = np.asarray(labels)\n\n# Split data\nX_train, X_val, y_train, y_val = train_test_split(images, labels, test_size=0.25)\n\n# Prepare labels for each class\nbowel_labels = labels[:, 2]\nextravasation_labels = labels[:, 4]\nkidney_labels = labels[:, 4:7]\nliver_labels = labels[:, 7:10]\nspleen_labels = labels[:, 10:13]\nany_labels = labels[:, -1]\n\nbowel_val = y_val[:, 2]\nextravasation_val = y_val[:, 4]\nkidney_val = y_val[:, 4:7]\nliver_val = y_val[:, 7:10]\nspleen_val = y_val[:, 10:13]\nany_val = y_val[:, -1]\n\n# KFold Cross Validation\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Store results for each fold\nfold_results = []\n\n# Data Augmentation\ntrain_datagen = ImageDataGenerator(\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\n# Validation data generator (no augmentation)\nval_datagen = ImageDataGenerator()\n\nfor fold, (train_index, val_index) in enumerate(kf.split(images, any_labels)):\n    print(f\"Fold {fold}\")  # Changed to print fold number from 0 to 4\n\n    # Split data into train and validation according to fold\n    X_train_fold, X_val_fold = images[train_index], images[val_index]\n    y_train_fold, y_val_fold = labels[train_index], labels[val_index]\n\n    # Prepare labels for each class for the fold\n    bowel_train = y_train_fold[:, 2]\n    extravasation_train = y_train_fold[:, 4]\n    kidney_train = y_train_fold[:, 4:7]\n    liver_train = y_train_fold[:, 7:10]\n    spleen_train = y_train_fold[:, 10:13]\n\n    bowel_val = y_val_fold[:, 2]\n    extravasation_val = y_val_fold[:, 4]\n    kidney_val = y_val_fold[:, 4:7]\n    liver_val = y_val_fold[:, 7:10]\n    spleen_val = y_val_fold[:, 10:13]\n\n    # Create a new model for each fold\n    model = create_model()\n\n    # EarlyStopping and ModelCheckpoint\n    early_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n    checkpoint = ModelCheckpoint(f'best_model_fold_{fold}.h5', save_best_only=True, monitor='val_loss')\n\n    # Train the model\n    batch_size = 16\n    num_epoch = 200  # You can modify this value based on the training time and model convergence\n    \n    history = model.fit(\n        x=X_train_fold,\n        y=[bowel_train, extravasation_train, kidney_train, liver_train, spleen_train],\n        batch_size=batch_size,\n        epochs=num_epoch,\n        verbose=1,\n         validation_data=(X_val_fold, [bowel_val, extravasation_val, kidney_val, liver_val, spleen_val])\n    )\n\n    # Store results for this fold\n    fold_results.append(history.history)\n\n# Check results\nfor fold, result in enumerate(fold_results):\n    print(f\"Fold {fold} Results:\")  # Changed to print fold number from 0 to 4\n    for key in result.keys():\n        print(f\"{key}: {result[key][-1]}\")\n\n\n\nimport pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['bowel_healthy', 'bowel_injury', 'any_injury'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['bowel_healthy']\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T05:18:22.840648Z","iopub.execute_input":"2024-11-23T05:18:22.841312Z","iopub.status.idle":"2024-11-23T07:35:40.858929Z","shell.execute_reply.started":"2024-11-23T05:18:22.841276Z","shell.execute_reply":"2024-11-23T07:35:40.858067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history.history.keys()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:40.859998Z","iopub.execute_input":"2024-11-23T07:35:40.860264Z","iopub.status.idle":"2024-11-23T07:35:40.865905Z","shell.execute_reply.started":"2024-11-23T07:35:40.860241Z","shell.execute_reply":"2024-11-23T07:35:40.865032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_acc = ['val_bowel_bowel_binary_accuracy', 'val_extra_extra_binary_accuracy', 'val_liver_liver_cat_accuracy', 'val_kidney_kidney_cat_accuracy', 'val_spleen_spleen_cat_accuracy']","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:40.866929Z","iopub.execute_input":"2024-11-23T07:35:40.867155Z","iopub.status.idle":"2024-11-23T07:35:40.883617Z","shell.execute_reply.started":"2024-11-23T07:35:40.867135Z","shell.execute_reply":"2024-11-23T07:35:40.882845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv\")\n\n# Kiểm tra tên cột trong DataFrame để xác định các cột không phải số\nprint(train_labels.columns)\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['bowel_healthy', 'bowel_injury'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['bowel_healthy']\n\n# Chuyển đổi các cột không phải số (nếu có) thành số\n# Xác định các cột kiểu chuỗi và chuyển chúng thành giá trị số\ncategorical_columns = X.select_dtypes(include=['object']).columns\n\n# Áp dụng LabelEncoder cho tất cả các cột không phải số\nencoder = LabelEncoder()\nfor col in categorical_columns:\n    X[col] = encoder.fit_transform(X[col])\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:40.884634Z","iopub.execute_input":"2024-11-23T07:35:40.884854Z","iopub.status.idle":"2024-11-23T07:35:41.48538Z","shell.execute_reply.started":"2024-11-23T07:35:40.884835Z","shell.execute_reply":"2024-11-23T07:35:41.484522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv\")\n\n# Kiểm tra tên cột trong DataFrame để xác định các cột không phải số\nprint(train_labels.columns)\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['extravasation_healthy',\n       'extravasation_injury'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['extravasation_healthy']\n\n# Chuyển đổi các cột không phải số (nếu có) thành số\n# Xác định các cột kiểu chuỗi và chuyển chúng thành giá trị số\ncategorical_columns = X.select_dtypes(include=['object']).columns\n\n# Áp dụng LabelEncoder cho tất cả các cột không phải số\nencoder = LabelEncoder()\nfor col in categorical_columns:\n    X[col] = encoder.fit_transform(X[col])\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:41.486638Z","iopub.execute_input":"2024-11-23T07:35:41.48729Z","iopub.status.idle":"2024-11-23T07:35:42.032512Z","shell.execute_reply.started":"2024-11-23T07:35:41.487254Z","shell.execute_reply":"2024-11-23T07:35:42.031671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv\")\n\n# Kiểm tra tên cột trong DataFrame để xác định các cột không phải số\nprint(train_labels.columns)\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['kidney_healthy', 'kidney_low', 'kidney_high'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['kidney_healthy']\n\n# Chuyển đổi các cột không phải số (nếu có) thành số\n# Xác định các cột kiểu chuỗi và chuyển chúng thành giá trị số\ncategorical_columns = X.select_dtypes(include=['object']).columns\n\n# Áp dụng LabelEncoder cho tất cả các cột không phải số\nencoder = LabelEncoder()\nfor col in categorical_columns:\n    X[col] = encoder.fit_transform(X[col])\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:42.033663Z","iopub.execute_input":"2024-11-23T07:35:42.034032Z","iopub.status.idle":"2024-11-23T07:35:42.686048Z","shell.execute_reply.started":"2024-11-23T07:35:42.034Z","shell.execute_reply":"2024-11-23T07:35:42.685208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv\")\n\n# Kiểm tra tên cột trong DataFrame để xác định các cột không phải số\nprint(train_labels.columns)\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['liver_healthy', 'liver_low', 'liver_high'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['liver_healthy']\n\n# Chuyển đổi các cột không phải số (nếu có) thành số\n# Xác định các cột kiểu chuỗi và chuyển chúng thành giá trị số\ncategorical_columns = X.select_dtypes(include=['object']).columns\n\n# Áp dụng LabelEncoder cho tất cả các cột không phải số\nencoder = LabelEncoder()\nfor col in categorical_columns:\n    X[col] = encoder.fit_transform(X[col])\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:42.687162Z","iopub.execute_input":"2024-11-23T07:35:42.687423Z","iopub.status.idle":"2024-11-23T07:35:43.355006Z","shell.execute_reply.started":"2024-11-23T07:35:42.6874Z","shell.execute_reply":"2024-11-23T07:35:43.35413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv\")\n\n# Kiểm tra tên cột trong DataFrame để xác định các cột không phải số\nprint(train_labels.columns)\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['spleen_low', 'spleen_high', 'any_injury'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['spleen_low']\n\n# Chuyển đổi các cột không phải số (nếu có) thành số\n# Xác định các cột kiểu chuỗi và chuyển chúng thành giá trị số\ncategorical_columns = X.select_dtypes(include=['object']).columns\n\n# Áp dụng LabelEncoder cho tất cả các cột không phải số\nencoder = LabelEncoder()\nfor col in categorical_columns:\n    X[col] = encoder.fit_transform(X[col])\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:43.356548Z","iopub.execute_input":"2024-11-23T07:35:43.356982Z","iopub.status.idle":"2024-11-23T07:35:43.887155Z","shell.execute_reply.started":"2024-11-23T07:35:43.356944Z","shell.execute_reply":"2024-11-23T07:35:43.886283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['bowel_healthy', 'bowel_injury', 'any_injury'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['bowel_healthy']\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:43.888247Z","iopub.execute_input":"2024-11-23T07:35:43.888527Z","iopub.status.idle":"2024-11-23T07:35:44.205309Z","shell.execute_reply.started":"2024-11-23T07:35:43.888502Z","shell.execute_reply":"2024-11-23T07:35:44.204488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['extravasation_healthy', 'extravasation_injury'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['extravasation_healthy']\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:44.206387Z","iopub.execute_input":"2024-11-23T07:35:44.20669Z","iopub.status.idle":"2024-11-23T07:35:44.450002Z","shell.execute_reply.started":"2024-11-23T07:35:44.206653Z","shell.execute_reply":"2024-11-23T07:35:44.449088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['kidney_healthy', 'kidney_low', 'kidney_high',])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['kidney_healthy']\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:44.451456Z","iopub.execute_input":"2024-11-23T07:35:44.451723Z","iopub.status.idle":"2024-11-23T07:35:44.690306Z","shell.execute_reply.started":"2024-11-23T07:35:44.4517Z","shell.execute_reply":"2024-11-23T07:35:44.689606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['liver_healthy', 'liver_low', 'liver_high',])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['liver_healthy']\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:44.691231Z","iopub.execute_input":"2024-11-23T07:35:44.691465Z","iopub.status.idle":"2024-11-23T07:35:44.949399Z","shell.execute_reply.started":"2024-11-23T07:35:44.691444Z","shell.execute_reply":"2024-11-23T07:35:44.948419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['spleen_healthy', 'spleen_low', 'spleen_high'])  # Xóa các cột nhãn để chỉ lấy đặc trưng\ny_bowel_healthy = train_labels['spleen_healthy']\n\n# Chia dữ liệu thành tập huấn luyện và tập kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Khởi tạo và huấn luyện mô hình RandomForest\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\n\n# Dự đoán kết quả trên tập kiểm tra\ny_pred_bowel_healthy = model.predict(X_test)\n\n# Tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Tính toán các chỉ số đánh giá\nrecall_bowel_healthy, precision_bowel_healthy, f1_bowel_healthy, accuracy_bowel_healthy, specificity_bowel_healthy = calculate_metrics(y_test, y_pred_bowel_healthy)\n\n# In kết quả\nprint(f\"Results for bowel_healthy:\")\nprint(f\"Recall: {recall_bowel_healthy:.4f}\")\nprint(f\"Precision: {precision_bowel_healthy:.4f}\")\nprint(f\"F1-score: {f1_bowel_healthy:.4f}\")\nprint(f\"Accuracy: {accuracy_bowel_healthy:.4f}\")\nprint(f\"Specificity: {specificity_bowel_healthy:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:44.95071Z","iopub.execute_input":"2024-11-23T07:35:44.950993Z","iopub.status.idle":"2024-11-23T07:35:45.190337Z","shell.execute_reply.started":"2024-11-23T07:35:44.950969Z","shell.execute_reply":"2024-11-23T07:35:45.189493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix, recall_score, precision_score, f1_score, accuracy_score\nfrom sklearn.model_selection import train_test_split\n\n# Đọc dữ liệu nhãn từ file train\ntrain_labels = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Đảm bảo đường dẫn đúng\n\n# Tạo các đặc trưng (features) và nhãn (labels)\nX = train_labels.drop(columns=['bowel_healthy', 'bowel_injury', 'any_injury', \n                               'extravasation_healthy', 'extravasation_injury',\n                               'kidney_healthy', 'kidney_low', 'kidney_high',\n                               'liver_healthy', 'liver_low', 'liver_high',\n                               'spleen_healthy', 'spleen_low', 'spleen_high'])  # Xóa các cột nhãn không cần thiết\n\n# Định nghĩa các nhãn mục tiêu\ny_bowel_healthy = train_labels['bowel_healthy']\ny_extra_healthy = train_labels['extravasation_healthy']\ny_liver_healthy = train_labels['liver_healthy']\ny_kidney_healthy = train_labels['kidney_healthy']\ny_spleen_healthy = train_labels['spleen_healthy']\n\n# Tạo hàm tính toán các chỉ số đánh giá\ndef calculate_metrics(y_true, y_pred, num_classes=1):\n    if num_classes == 1:\n        cm = confusion_matrix(y_true, y_pred)\n        tn, fp, fn, tp = cm.ravel()\n        recall = tp / (tp + fn)  # Độ nhạy (Recall)\n        precision = tp / (tp + fp)  # Độ chính xác (Precision)\n        f1 = 2 * (precision * recall) / (precision + recall)  # F1-score\n        accuracy = (tp + tn) / (tp + tn + fp + fn)  # Độ chính xác (Accuracy)\n        specificity = tn / (tn + fp)  # Độ nhiễu (Specificity)\n    else:\n        recall = recall_score(y_true, y_pred, average='weighted', zero_division=0)\n        precision = precision_score(y_true, y_pred, average='weighted', zero_division=0)\n        f1 = f1_score(y_true, y_pred, average='weighted', zero_division=0)\n        accuracy = accuracy_score(y_true, y_pred)\n        specificity = None\n\n    return recall, precision, f1, accuracy, specificity\n\n# Hàm tính toán và in kết quả cho từng nhãn\ndef evaluate_model_for_label(X_train, X_test, y_train, y_test, label_name):\n    model = RandomForestClassifier(n_estimators=100, random_state=42)\n    model.fit(X_train, y_train)  # Huấn luyện mô hình\n\n    # Dự đoán trên tập kiểm tra\n    y_pred = model.predict(X_test)\n\n    # Tính toán các chỉ số đánh giá\n    recall, precision, f1, accuracy, specificity = calculate_metrics(y_test, y_pred)\n\n    # In kết quả cho từng nhãn\n    print(f\"Results for {label_name}:\")\n    print(f\"Recall: {recall:.4f}\")\n    print(f\"Precision: {precision:.4f}\")\n    print(f\"F1-score: {f1:.4f}\")\n    print(f\"Accuracy: {accuracy:.4f}\")\n    print(f\"Specificity: {specificity:.4f}\")\n    print()  # Dòng trống giữa các kết quả\n\n# Chia dữ liệu thành tập huấn luyện và kiểm tra\nX_train, X_test, y_train, y_test = train_test_split(X, y_bowel_healthy, test_size=0.2, random_state=42)\n\n# Đánh giá mô hình cho từng nhãn\nevaluate_model_for_label(X_train, X_test, y_train, y_test, \"bowel_healthy\")\n\n# Đánh giá cho các nhãn khác\nevaluate_model_for_label(X_train, X_test, y_train, y_test, \"extravasation_healthy\")\nevaluate_model_for_label(X_train, X_test, y_train, y_test, \"liver_healthy\")\nevaluate_model_for_label(X_train, X_test, y_train, y_test, \"kidney_healthy\")\nevaluate_model_for_label(X_train, X_test, y_train, y_test, \"spleen_healthy\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:45.191492Z","iopub.execute_input":"2024-11-23T07:35:45.191762Z","iopub.status.idle":"2024-11-23T07:35:46.922658Z","shell.execute_reply.started":"2024-11-23T07:35:45.19174Z","shell.execute_reply":"2024-11-23T07:35:46.921852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Lặp qua các key trong history để vẽ accuracy\nfor i in history.history.keys():\n    if i.endswith(\"_accuracy\") and not i == \"val_accuracy\":\n        plt.plot(history.history[i], label=i)\n\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend(loc=(1.05, 0.0))\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:46.923992Z","iopub.execute_input":"2024-11-23T07:35:46.924366Z","iopub.status.idle":"2024-11-23T07:35:47.265588Z","shell.execute_reply.started":"2024-11-23T07:35:46.924332Z","shell.execute_reply":"2024-11-23T07:35:47.264724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"_loss\") and not i ==\"val_loss\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:47.266548Z","iopub.execute_input":"2024-11-23T07:35:47.266791Z","iopub.status.idle":"2024-11-23T07:35:47.585472Z","shell.execute_reply.started":"2024-11-23T07:35:47.26677Z","shell.execute_reply":"2024-11-23T07:35:47.584492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in history.history.keys():\n    if i.endswith(\"accuracy\") and not i ==\"val_accuracy\":\n        plt.plot(history.history[i], label=i)\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.legend(loc=(1.05,0.0))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:47.586384Z","iopub.execute_input":"2024-11-23T07:35:47.586611Z","iopub.status.idle":"2024-11-23T07:35:47.925618Z","shell.execute_reply.started":"2024-11-23T07:35:47.586591Z","shell.execute_reply":"2024-11-23T07:35:47.924841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Lặp qua các key trong history để vẽ accuracy (loại bỏ val_accuracy)\nfor i in history.history.keys():\n    if i.endswith(\"accuracy\") and \"val_\" not in i:\n        plt.plot(history.history[i], label=i)\n\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.legend(loc=(1.05, 0.0))\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:47.926793Z","iopub.execute_input":"2024-11-23T07:35:47.927134Z","iopub.status.idle":"2024-11-23T07:35:48.209337Z","shell.execute_reply.started":"2024-11-23T07:35:47.927103Z","shell.execute_reply":"2024-11-23T07:35:48.208514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Kiểm tra các khóa trong history\nprint(history.history.keys())\n\n# Các metric bạn muốn vẽ (chỉ vẽ các accuracy mà không có \"val_\")\nmetrics_to_plot = [key for key in history.history.keys() if key.endswith('accuracy') and 'val_' not in key]\n\n# Tạo biểu đồ cho cả huấn luyện và validation\nfor metric in metrics_to_plot:\n    plt.figure(figsize=(8, 6))  # Tạo một biểu đồ mới cho mỗi metric\n    plt.plot(history.history[metric], label=f'Training {metric}')\n    \n    # Kiểm tra và vẽ độ chính xác của validation nếu có\n    val_metric = f'val_{metric}'\n    if val_metric in history.history:\n        plt.plot(history.history[val_metric], label=f'Validation {metric}')\n    \n    # Thêm số epoch vào trục x (tự động từ 0 đến n)\n    epochs = range(1, len(history.history[metric]) + 1)\n    \n    # Thêm ticks với khoảng cách 25\n    step_size = 25\n    epoch_ticks = [epoch for epoch in epochs if epoch % step_size == 0]\n    plt.xticks(epoch_ticks)\n    \n    # Cài đặt các thông số cho biểu đồ\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.title(f'{metric.capitalize()} over Epochs')\n    plt.legend(loc='upper left')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:48.210321Z","iopub.execute_input":"2024-11-23T07:35:48.210567Z","iopub.status.idle":"2024-11-23T07:35:49.54313Z","shell.execute_reply.started":"2024-11-23T07:35:48.210545Z","shell.execute_reply":"2024-11-23T07:35:49.542117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(history.history[\"val_loss\"], label=\"val_loss\")\nplt.plot(history.history[\"loss\"], label=\"loss\")\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:49.544341Z","iopub.execute_input":"2024-11-23T07:35:49.544626Z","iopub.status.idle":"2024-11-23T07:35:49.775781Z","shell.execute_reply.started":"2024-11-23T07:35:49.544603Z","shell.execute_reply":"2024-11-23T07:35:49.774913Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kiểm tra danh sách tên cột\nprint(df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:49.776895Z","iopub.execute_input":"2024-11-23T07:35:49.77714Z","iopub.status.idle":"2024-11-23T07:35:50.3156Z","shell.execute_reply.started":"2024-11-23T07:35:49.777118Z","shell.execute_reply":"2024-11-23T07:35:50.312957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Đảm bảo rằng tệp CSV hoặc nguồn dữ liệu của bạn đã được nạp vào DataFrame\ndf = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv\")  # Thay thế bằng đường dẫn đúng\n\n# Kiểm tra các giá trị thiếu trong DataFrame\nmissing_values = df.isnull().sum()\nprint(\"Missing Values:\")\nprint(missing_values)\n\n# Danh sách các cột nhị phân cần chuyển đổi sang kiểu boolean\nbinary_columns = [\n    'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n    'spleen_healthy', 'spleen_low', 'spleen_high'\n]\n\n# Chuyển đổi các cột nhị phân thành kiểu boolean\ndf[binary_columns] = df[binary_columns].astype(bool)\n\n# Giải quyết các vấn đề về chất lượng dữ liệu (nếu có, bạn có thể thêm các bước xử lý dữ liệu ở đây)\n\n# Hiển thị DataFrame sau khi đã xử lý\nprint(\"\\nPreprocessed DataFrame:\")\nprint(df.head())  # In ra 5 dòng đầu tiên của DataFrame đã xử lý để kiểm tra\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.316314Z","iopub.status.idle":"2024-11-23T07:35:50.316612Z","shell.execute_reply.started":"2024-11-23T07:35:50.316464Z","shell.execute_reply":"2024-11-23T07:35:50.316478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cột nhãn cho các cơ quan bạn muốn hiển thị\norgan_columns = ['kidney', 'liver', 'spleen']  # Chỉ giữ lại các cơ quan bạn muốn hiển thị\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: kidney_healthy)\n    y_pred = df[f'{organ}_low']  # Hoặc chọn nhãn khác như kidney_low (tùy vào mục đích của bạn)\n\n    # Tạo ma trận nhầm lẫn với 5 lớp (0 đến 4)\n    cm = confusion_matrix(y_true, y_pred, labels=[False, True, 2, 3, 4,5,6,7,8,9])  # Chỉ định 5 lớp (tùy theo dữ liệu của bạn)\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9, 10], yticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9, 10])\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.318124Z","iopub.status.idle":"2024-11-23T07:35:50.318423Z","shell.execute_reply.started":"2024-11-23T07:35:50.318283Z","shell.execute_reply":"2024-11-23T07:35:50.318297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kiểm tra tất cả các cột trong DataFrame\nprint(df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.319612Z","iopub.status.idle":"2024-11-23T07:35:50.319916Z","shell.execute_reply.started":"2024-11-23T07:35:50.319748Z","shell.execute_reply":"2024-11-23T07:35:50.31976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cơ quan bạn muốn hiển thị\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: bowel_healthy)\n    \n    # Chọn cột dự đoán như 'injury', 'low' hoặc 'high', tùy vào mục đích của bạn\n    y_pred = df[f'{organ}_injury']  # Hoặc thay bằng 'low' hoặc 'high' tùy vào nhu cầu\n\n    # Tạo ma trận nhầm lẫn với 2 lớp (0: Healthy, 1: Injury)\n    cm_binary = confusion_matrix(y_true, y_pred, labels=[False, True])  # Chỉ so sánh Healthy vs Injury\n\n    # Vẽ ma trận nhầm lẫn 2 lớp\n    plt.figure(figsize=(10, 8))  \n    sns.heatmap(cm_binary, annot=True, fmt='d', cmap='Blues', xticklabels=['Healthy', 'Injury'], yticklabels=['Healthy', 'Injury'])\n    plt.title(f'Confusion Matrix for {organ.capitalize()} (2-class: Healthy vs Injury)')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n    plt.show()\n\n    # Tạo ma trận nhầm lẫn với 10 lớp (0 đến 9) nếu dữ liệu có thể hỗ trợ\n    cm_10class = confusion_matrix(y_true, y_pred, labels=range(10))  # Sử dụng 10 lớp (0 đến 9)\n    \n    # Vẽ ma trận nhầm lẫn 10 lớp\n    plt.figure(figsize=(10, 8))  \n    sns.heatmap(cm_10class, annot=True, fmt='d', cmap='Blues', xticklabels=range(10), yticklabels=range(10))\n    plt.title(f'Confusion Matrix for {organ.capitalize()} (10-class)')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.321857Z","iopub.status.idle":"2024-11-23T07:35:50.322158Z","shell.execute_reply.started":"2024-11-23T07:35:50.322021Z","shell.execute_reply":"2024-11-23T07:35:50.322035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/train.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cột nhãn cho các cơ quan bạn muốn hiển thị\norgan_columns = ['kidney', 'liver', 'spleen']  # Các cơ quan cần hiển thị\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: kidney_healthy)\n    y_pred = df[f'{organ}_low']  # Hoặc chọn nhãn khác như kidney_low (tùy vào mục đích của bạn)\n\n    # Tạo ma trận nhầm lẫn với 10 lớp (0 đến 9)\n    cm = confusion_matrix(y_true, y_pred, labels=[False, True, 2, 3, 4, 5, 6, 7, 8, 9])  # Giả sử dữ liệu có lớp 0-9\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9], yticklabels=[0, 1, 2, 3, 4 , 5, 6, 7, 8, 9])\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.323863Z","iopub.status.idle":"2024-11-23T07:35:50.324165Z","shell.execute_reply.started":"2024-11-23T07:35:50.324027Z","shell.execute_reply":"2024-11-23T07:35:50.324041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv')\n\n# Trích xuất nhãn thật (y_true) và nhãn dự đoán (y_pred)\n# Giả sử rằng bạn có một mô hình dự đoán sẵn có hoặc đang huấn luyện mô hình để lấy y_pred\n\n# Dữ liệu về bowel (thay 'bowel' bằng các cơ quan khác như 'extravasation', 'kidney', v.v.)\ny_true_bowel = df['bowel_healthy']  # Hoặc nếu bạn muốn nhãn về injury thì dùng 'bowel_injury'\ny_pred_bowel = df['bowel_injury']  # Đây là nhãn dự đoán mà mô hình của bạn sẽ đưa ra (giả lập)\n\n# Dự đoán giả lập cho ví dụ\n# y_pred_bowel có thể là đầu ra của mô hình dự đoán (thay 'bowel_injury' bằng giá trị thực tế của mô hình của bạn)\n# Đoạn dưới đây là giả lập, bạn sẽ thay thế bằng mô hình thực tế của mình.\n\n# Xử lý các cơ quan khác (extravasation, kidney, liver, spleen)\ny_true_extra = df['extravasation_healthy']\ny_pred_extra= df['extravasation_injury']  # Tùy thuộc vào nhãn bạn muốn sử dụng\n\n# Tạo danh sách các cột nhãn cho các cơ quan khác\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    y_true = df[f'{organ}_healthy']\n    y_pred = df[f'{organ}_injury']  # Hoặc thay đổi cột này tùy vào nhu cầu\n\n    # Tạo ma trận nhầm lẫn với 5 lớp (0 đến 4)\n    cm = confusion_matrix(y_true, y_pred, labels=range(10))  # Sử dụng 10 lớp (0 đến 9)\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận 11x11\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=range(11), yticklabels=range(11))\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.32556Z","iopub.status.idle":"2024-11-23T07:35:50.325887Z","shell.execute_reply.started":"2024-11-23T07:35:50.325707Z","shell.execute_reply":"2024-11-23T07:35:50.325721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Đọc tệp CSV\ndf = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv')\n\n# Kiểm tra tất cả các cột trong DataFrame\nprint(df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.32735Z","iopub.status.idle":"2024-11-23T07:35:50.327638Z","shell.execute_reply.started":"2024-11-23T07:35:50.327503Z","shell.execute_reply":"2024-11-23T07:35:50.327517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv')\n\n# Loại bỏ khoảng trắng thừa trong tên cột (nếu có)\ndf.columns = df.columns.str.strip()\n\n# Tạo danh sách các cơ quan bạn muốn hiển thị\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Đảm bảo tên cột chính xác sau khi loại bỏ khoảng trắng\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế (ví dụ: bowel_healthy)\n    \n    # Chọn cột dự đoán phù hợp\n    if organ == 'kidney':  # kidney có 'low' và 'high' thay vì 'injury'\n        y_pred = df[f'{organ}_low']  # Thay đổi 'low' nếu bạn muốn dự đoán theo 'high'\n    elif organ == 'liver':  # liver cũng có 'low' và 'high'\n        y_pred = df[f'{organ}_low']\n    elif organ == 'spleen':  # spleen cũng có 'low' và 'high'\n        y_pred = df[f'{organ}_low']\n    else:  # Các cơ quan còn lại có 'injury' cột\n        y_pred = df[f'{organ}_injury']\n\n    # Tạo ma trận nhầm lẫn với 10 lớp (0 đến 9)\n    cm = confusion_matrix(y_true, y_pred, labels=range(10))  # Chỉ định 10 lớp (0 đến 9)\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận 10x10\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=range(10), yticklabels=range(10))\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.328884Z","iopub.status.idle":"2024-11-23T07:35:50.329176Z","shell.execute_reply.started":"2024-11-23T07:35:50.329042Z","shell.execute_reply":"2024-11-23T07:35:50.329056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Đọc tệp CSV chứa nhãn\ndf = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_2024.csv')\n\n# Trích xuất nhãn thật (y_true) và nhãn dự đoán (y_pred)\n# Giả sử rằng bạn có một mô hình dự đoán sẵn có hoặc đang huấn luyện mô hình để lấy y_pred\n\n# Tạo danh sách các cơ quan bạn muốn hiển thị\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Vẽ ma trận nhầm lẫn cho các cơ quan\nfor organ in organ_columns:\n    # Trích xuất nhãn thật (healthy) và nhãn dự đoán (injury)\n    y_true = df[f'{organ}_healthy']  # Nhãn thực tế\n    y_pred = df[f'{organ}_injury']  # Nhãn dự đoán (injury)\n\n    # Tạo ma trận nhầm lẫn với 10 lớp (0 đến 9)\n    cm = confusion_matrix(y_true, y_pred, labels=range(10))  # Sử dụng 10 lớp (0 đến 9)\n\n    # Vẽ ma trận nhầm lẫn\n    plt.figure(figsize=(10, 8))  # Thay đổi kích thước đồ thị cho phù hợp với ma trận\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=range(10), yticklabels=range(10))\n\n    # Thiết lập tiêu đề và nhãn\n    plt.title(f'Confusion Matrix for {organ.capitalize()}')\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n\n    # Hiển thị ma trận nhầm lẫn\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.330068Z","iopub.status.idle":"2024-11-23T07:35:50.330326Z","shell.execute_reply.started":"2024-11-23T07:35:50.330202Z","shell.execute_reply":"2024-11-23T07:35:50.330214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for missing values\nmissing_values = df.isnull().sum()\nprint(\"Missing Values:\")\nprint(missing_values)\n\n# Handle missing values\n# In this simple example, we will drop rows with missing values.\ndf = df.dropna()\n\n# Check Data Types and Convert Binary Data to Boolean\nbinary_columns = [\n  'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n  'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n  'spleen_healthy', 'spleen_low', 'spleen_high'\n]\ndf[binary_columns] = df[binary_columns].astype(bool)\n\n# Address Data Quality Issues\n# In this simple example, we assume no data quality issues are present.\n\nprint(\"\\nPreprocessed DataFrame:\")\nprint(df)\n\nplt.figure()\ndf.plot.hist()\nplt.title('Distribution of Features')\nplt.xlabel('Feature')\nplt.ylabel('Count')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.332224Z","iopub.status.idle":"2024-11-23T07:35:50.332623Z","shell.execute_reply.started":"2024-11-23T07:35:50.332419Z","shell.execute_reply":"2024-11-23T07:35:50.332438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.333971Z","iopub.status.idle":"2024-11-23T07:35:50.334369Z","shell.execute_reply.started":"2024-11-23T07:35:50.334165Z","shell.execute_reply":"2024-11-23T07:35:50.334184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Organ columns\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Create a new DataFrame to store the counts\norgan_counts = pd.DataFrame()\norgan_counts['Organ'] = organ_columns\n\n# Loop through organ columns and count healthy and injury status for each organ\nfor organ in organ_columns:\n    healthy_col = f'{organ}_healthy'\n    injury_col = f'{organ}_injury'\n\n    # Check if the columns exist in the DataFrame\n    if healthy_col in df.columns and injury_col in df.columns:\n        organ_counts[f'{organ}_healthy'] = df[healthy_col].sum()\n        organ_counts[f'{organ}_injury'] = df[injury_col].sum()\n    else:\n        # Handle the case if the columns are missing\n        print(f\"Warning: Columns for {organ} healthy/injury status are missing in the DataFrame.\")\n        organ_counts[f'{organ}_healthy'] = 0\n        organ_counts[f'{organ}_injury'] = 0\n\n# Fill in missing values with 0\norgan_counts.fillna(0, inplace=True)\n\n# Melt the DataFrame to have a single 'Status' column\norgan_counts_melted = organ_counts.melt(id_vars=['Organ'], var_name='Status', value_name='Count')\n\n# Bar plot for distribution of organ health and injury status\nfig = px.bar(\n    organ_counts_melted,\n    x='Organ',\n    y='Count',\n    color='Status',\n    barmode='group',\n    labels=dict(x='Organ', y='Count', Status='Status'),\n    title='Distribution of Organ Health and Injury Status',\n    height=500,\n    width=800,\n    template='plotly_dark',\n)\n\n# Customize the plot\nfig.update_layout(\n    legend_title='Organ Status',\n    legend_orientation='h',\n    legend_xanchor='center',\n    legend_yanchor='top',\n    legend_x=0.5,\n    legend_y=1.1,\n    xaxis_title='Organ',\n    yaxis_title='Count',\n    font=dict(family='Arial', size=12),\n)\n\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.33563Z","iopub.status.idle":"2024-11-23T07:35:50.33606Z","shell.execute_reply.started":"2024-11-23T07:35:50.335847Z","shell.execute_reply":"2024-11-23T07:35:50.335868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport plotly.express as px\n\n# Giả sử df là DataFrame đã được nạp vào từ dữ liệu của bạn\n# df = pd.read_csv(\"/path/to/your/data.csv\")\n\n# Cột các cơ quan\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Kiểm tra xem các cột \"injury\" có tồn tại trong DataFrame không\ninjury_columns = [f'{organ}_injury' for organ in organ_columns]\nmissing_columns = set(organ_columns + injury_columns) - set(df.columns)\n\nif missing_columns:\n    # Thông báo nếu có cột thiếu\n    print(f\"Warning: Columns for {', '.join(missing_columns)} are missing in the DataFrame.\")\n    for col in missing_columns:\n        df[col] = 0  # Thêm cột thiếu vào DataFrame với giá trị 0\n\n# Lọc các cột liên quan đến sức khỏe và chấn thương của các cơ quan\ncorrelation_df = df[organ_columns + injury_columns]\n\n# Tính toán ma trận tương quan giữa các cột\ncorrelation_matrix = correlation_df.corr()\n\n# Tạo heatmap để phân tích mối tương quan giữa sức khỏe và tình trạng chấn thương của các cơ quan\nfig = px.imshow(\n    correlation_matrix,\n    x=correlation_df.columns,\n    y=correlation_df.columns,\n    labels=dict(x='Organ', y='Organ', color='Correlation'),\n    title='Correlation Between Organ Health and Injury Status',\n)\n\n# Hiển thị heatmap\nfig.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.337635Z","iopub.status.idle":"2024-11-23T07:35:50.338067Z","shell.execute_reply.started":"2024-11-23T07:35:50.337855Z","shell.execute_reply":"2024-11-23T07:35:50.337875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Summary statistics for relevant variables\nstyled_data = df.describe().style\\\n.background_gradient(cmap='coolwarm')\\\n.set_properties(**{'text-align':'center','border':'1px solid black'})\n\n# display styled data\ndisplay(styled_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.339047Z","iopub.status.idle":"2024-11-23T07:35:50.33944Z","shell.execute_reply.started":"2024-11-23T07:35:50.339234Z","shell.execute_reply":"2024-11-23T07:35:50.339253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot styled data in a single plot, using subgrid layout\nimport matplotlib.gridspec as gridspec\n\n\nstyled_data = df.describe().style\\\n.background_gradient(cmap='coolwarm')\\\n.set_properties(**{'text-align':'center','border':'1px solid black'})\n\n# Cgridspec layout\ngs = gridspec.GridSpec(2, 2)\n\n# Loop over the styled data and plot it\nfig, axes = plt.subplots(2, 2, figsize=(12, 8), subplot_kw={'adjustable': 'box'})\nfor i in range(2):\n    for j in range(2):\n        cell_value = styled_data.data.iloc[i, j]\n\n        axes[i, j].plot(cell_value)\n        axes[i, j].set_title(styled_data.index[i] + ', ' + styled_data.columns[j])\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.341213Z","iopub.status.idle":"2024-11-23T07:35:50.341508Z","shell.execute_reply.started":"2024-11-23T07:35:50.341364Z","shell.execute_reply":"2024-11-23T07:35:50.341378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# pass list of tick positions to the set_xticks() function. \n# pass the following list of tick positions to the set_xticks() function in the counts plot loop\n\ndef generate_counts_and_percentages(df, categorical_columns):\n  \"\"\"Counts and percentages for categorical variables in a DataFrame, and plot the counts and percentages.\n\n  Args:\n    df: DataFrame.\n    categorical_columns: column names for the categorical variables.\n\n  Returns:\n    None.\n  \"\"\"\n\n  # Handle null values.\n  df = df.dropna(subset=categorical_columns)\n\n  # counts.\n  counts = df[categorical_columns].apply(pd.Series.value_counts)\n\n  # percentages.\n  percentages = (counts / df.shape[0]) * 100\n\n  # Set color scheme.\n  colors = ['#007bff', '#ffa500']\n\n  # Plot counts.\n  fig, axes = plt.subplots(1, len(categorical_columns), figsize=(15, 7))\n  for i, column in enumerate(categorical_columns):\n    ax = axes[i]\n    ax.bar(counts.index.to_list(), counts[column].to_list(), color=colors[0])\n    ax.set_title(column, fontsize=12)\n    ax.set_xticks(range(len(counts.index)))\n    ax.tick_params(labelsize=10)\n    ax.grid(True)\n\n  # Plot percentages.\n  fig, axes = plt.subplots(1, len(categorical_columns), figsize=(15, 7))\n  for i, column in enumerate(categorical_columns):\n    ax = axes[i]\n    ax.pie(percentages[column].to_list(), labels=percentages.index.to_list(), autopct='%1.1f%%', startangle=140, colors=colors)\n    ax.set_title(column, fontsize=12)\n    ax.axis('equal')\n    ax.legend(fontsize=10)\n    ax.grid(True)\n\n  plt.suptitle('Counts and Percentages for Categorical Variables', fontsize=14)\n  plt.show()\n\ncategorical_columns = [\n  'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury',\n  'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high',\n  'spleen_healthy', 'spleen_low', 'spleen_high', 'any_injury'\n]\n\ngenerate_counts_and_percentages(df, categorical_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.342617Z","iopub.status.idle":"2024-11-23T07:35:50.342939Z","shell.execute_reply.started":"2024-11-23T07:35:50.342767Z","shell.execute_reply":"2024-11-23T07:35:50.342782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Organ columns: bowel, extravasation, kidney, liver, spleen\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Create a new DataFrame to store the counts\norgan_counts = pd.DataFrame()\norgan_counts['Organ'] = organ_columns\n\n# Loop through organ columns and count healthy and injury status for each organ\nfor organ in organ_columns:\n    healthy_col = f'{organ}_healthy'\n    injury_col = f'{organ}_injury'\n    \n    # Check if the columns exist in the DataFrame\n    if healthy_col in df.columns and injury_col in df.columns:\n        organ_counts[f'{organ}_healthy'] = df[healthy_col].sum()\n        organ_counts[f'{organ}_injury'] = df[injury_col].sum()\n    else:\n        # Handle the case if the columns are missing\n        print(f\"Warning: Columns for {organ} healthy/injury status are missing in the DataFrame.\")\n        organ_counts[f'{organ}_healthy'] = 0\n        organ_counts[f'{organ}_injury'] = 0\n\n# Melt the DataFrame to have a single 'Status' column\norgan_counts_melted = organ_counts.melt(id_vars=['Organ'], var_name='Status', value_name='Count')\n\n# Bar plot for distribution of organ health and injury status\nfig = px.bar(\n    organ_counts_melted,\n    x='Organ',\n    y='Count',\n    color='Status',\n    barmode='group',\n    labels=dict(x='Organ', y='Count', Status='Status'),\n    title='Distribution of Organ Health and Injury Status',\n)\nfig.update_layout(showlegend=True)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.344252Z","iopub.status.idle":"2024-11-23T07:35:50.344526Z","shell.execute_reply.started":"2024-11-23T07:35:50.344382Z","shell.execute_reply":"2024-11-23T07:35:50.344394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Organ columns\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Create a new DataFrame to store the counts\norgan_counts = pd.DataFrame()\norgan_counts['Organ'] = organ_columns\n\n# Loop through organ columns and count healthy and injury status for each organ\nfor organ in organ_columns:\n    healthy_col = f'{organ}_healthy'\n    injury_col = f'{organ}_injury'\n\n    # Check if the columns exist in the DataFrame\n    if healthy_col in df.columns and injury_col in df.columns:\n        organ_counts[f'{organ}_healthy'] = df[healthy_col].sum()\n        organ_counts[f'{organ}_injury'] = df[injury_col].sum()\n    else:\n        # Handle the case if the columns are missing\n        print(f\"Warning: Columns for {organ} healthy/injury status are missing in the DataFrame.\")\n        organ_counts[f'{organ}_healthy'] = 0\n        organ_counts[f'{organ}_injury'] = 0\n\n# Fill in missing values with 0\norgan_counts.fillna(0, inplace=True)\n\n# Melt the DataFrame to have a single 'Status' column\norgan_counts_melted = organ_counts.melt(id_vars=['Organ'], var_name='Status', value_name='Count')\n\n# Bar plot for distribution of organ health and injury status\nfig = px.bar(\n    organ_counts_melted,\n    x='Organ',\n    y='Count',\n    color='Status',\n    barmode='group',\n    labels=dict(x='Organ', y='Count', Status='Status'),\n    title='Distribution of Organ Health and Injury Status',\n    height=500,\n    width=800,\n    template='plotly_dark',\n)\n\n# Customize the plot\nfig.update_layout(\n    legend_title='Organ Status',\n    legend_orientation='h',\n    legend_xanchor='center',\n    legend_yanchor='top',\n    legend_x=0.5,\n    legend_y=1.1,\n    xaxis_title='Organ',\n    yaxis_title='Count',\n    font=dict(family='Arial', size=12),\n)\n\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.346049Z","iopub.status.idle":"2024-11-23T07:35:50.346378Z","shell.execute_reply.started":"2024-11-23T07:35:50.346204Z","shell.execute_reply":"2024-11-23T07:35:50.346226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Organ columns: bowel, extravasation, kidney, liver, spleen\norgan_columns = ['bowel', 'extravasation', 'kidney', 'liver', 'spleen']\n\n# Check if the 'injury' columns are present in the DataFrame\ninjury_columns = [f'{organ}_injury' for organ in organ_columns]\nmissing_columns = set(organ_columns + injury_columns) - set(df.columns)\n\nif missing_columns:\n    # Handle the case if any of the required columns are missing\n    print(f\"Warning: Columns for {', '.join(missing_columns)} are missing in the DataFrame.\")\n    for col in missing_columns:\n        df[col] = 0\n\n# Heatmap to analyze the correlation between organ health and injury status\ncorrelation_df = df[organ_columns + injury_columns]\ncorrelation_matrix = correlation_df.corr()\n\nfig = px.imshow(\n    correlation_matrix,\n    x=correlation_df.columns,\n    y=correlation_df.columns,\n    labels=dict(x='Organ', y='Organ', color='Correlation'),\n    title='Correlation Between Organ Health and Injury Status',\n)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.347312Z","iopub.status.idle":"2024-11-23T07:35:50.347597Z","shell.execute_reply.started":"2024-11-23T07:35:50.347462Z","shell.execute_reply":"2024-11-23T07:35:50.347476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport plotly.graph_objects as go\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import accuracy_score, confusion_matrix","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.348821Z","iopub.status.idle":"2024-11-23T07:35:50.34909Z","shell.execute_reply.started":"2024-11-23T07:35:50.348962Z","shell.execute_reply":"2024-11-23T07:35:50.348975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a synthetic binary classification dataset\nn_samples = 500\nn_features = 2\nn_classes = 2\nn_clusters_per_class = 1\nrandom_state = 42\n\n# Adjust the values of n_informative, n_redundant, and n_repeated\nn_informative = 2\nn_redundant = 0\nn_repeated = 0\n\nX, y = make_classification(\n    n_samples=n_samples,\n    n_features=n_features,\n    n_informative=n_informative,\n    n_redundant=n_redundant,\n    n_repeated=n_repeated,\n    n_classes=n_classes,\n    n_clusters_per_class=n_clusters_per_class,\n    random_state=random_state\n)\n\n# Split the dataset into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=random_state)\n\n# Train a logistic regression model on the dataset\nmodel = LogisticRegression(random_state=random_state)\nmodel.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = model.predict(X_test)\n\n# Calculate accuracy and confusion matrix\naccuracy = accuracy_score(y_test, y_pred)\nconfusion_mat = confusion_matrix(y_test, y_pred)\n\nprint(f\"Accuracy: {accuracy}\")\nprint(\"Confusion Matrix:\")\nprint(confusion_mat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.350245Z","iopub.status.idle":"2024-11-23T07:35:50.350569Z","shell.execute_reply.started":"2024-11-23T07:35:50.350415Z","shell.execute_reply":"2024-11-23T07:35:50.350431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\n# Define the hyperparameters to search over\nparam_grid = {\n    \"C\": [0.1, 1, 10, 100],\n    \"penalty\": [\"l1\", \"l2\"],\n}\n\n# Create a grid search object\ngrid_search = GridSearchCV(LogisticRegression(), param_grid, cv=5)\n\n# Fit the grid search object to the training data\ngrid_search.fit(X_train, y_train)\n\n# Get the best model from the grid search\nbest_model = grid_search.best_estimator_\n\n# Make predictions on the test set\ny_pred = best_model.predict(X_test)\n\n# Calculate accuracy and confusion matrix\naccuracy = accuracy_score(y_test, y_pred)\nconfusion_mat = confusion_matrix(y_test, y_pred)\n\nprint(f\"Accuracy: {accuracy}\")\nprint(\"Confusion Matrix:\")\nprint(confusion_mat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.351672Z","iopub.status.idle":"2024-11-23T07:35:50.351975Z","shell.execute_reply.started":"2024-11-23T07:35:50.351832Z","shell.execute_reply":"2024-11-23T07:35:50.351851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_and_evaluate_model(model, X_train, y_train, X_test, y_test):\n  \"\"\"Train and evaluate a machine learning model.\n\n  Args:\n    model: A machine learning model object.\n    X_train: The training data features.\n    y_train: The training data labels.\n    X_test: The test data features.\n    y_test: The test data labels.\n\n  Returns:\n    A tuple of the model's accuracy score and confusion matrix.\n  \"\"\"\n\n  model.fit(X_train, y_train)\n\n  y_pred = model.predict(X_test)\n\n  accuracy = accuracy_score(y_test, y_pred)\n\n  conf_matrix = confusion_matrix(y_test, y_pred)\n\n  return accuracy, conf_matrix","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.353413Z","iopub.status.idle":"2024-11-23T07:35:50.353887Z","shell.execute_reply.started":"2024-11-23T07:35:50.353613Z","shell.execute_reply":"2024-11-23T07:35:50.353634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Random Forests model\nrf_accuracy, rf_conf_matrix = train_and_evaluate_model(\n    RandomForestClassifier(max_depth=7, n_estimators=300, random_state=42),\n    X_train, y_train, X_test, y_test)\n\n# SVM model\nsvm_accuracy, svm_conf_matrix = train_and_evaluate_model(\n    SVC(kernel='linear', random_state=42), X_train, y_train, X_test, y_test)\n\n# Gradient Boosting model\ngb_accuracy, gb_conf_matrix = train_and_evaluate_model(\n    GradientBoostingClassifier(random_state=42), X_train, y_train, X_test,\n    y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.355513Z","iopub.status.idle":"2024-11-23T07:35:50.356188Z","shell.execute_reply.started":"2024-11-23T07:35:50.355905Z","shell.execute_reply":"2024-11-23T07:35:50.355948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier\n\n# Create a list of estimators\nestimators = [\n    ('rf', RandomForestClassifier(max_depth=7, n_estimators=300, random_state=42)),\n    ('svm', SVC(kernel='linear', random_state=42)),\n    ('gb', GradientBoostingClassifier(random_state=42)),\n]\n\n# Create a VotingClassifier object\nvoting_clf = VotingClassifier(estimators=estimators, voting='hard')\n\n# Fit the VotingClassifier model to the training data\nvoting_clf.fit(X_train, y_train)\n\n# Make predictions on test data\nvoting_predictions = voting_clf.predict(X_test)\n\n# Calculate the accuracy on the test data\nvoting_accuracy = accuracy_score(y_test, voting_predictions)\n\nprint('VotingClassifier accuracy:', voting_accuracy)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.357865Z","iopub.status.idle":"2024-11-23T07:35:50.358301Z","shell.execute_reply.started":"2024-11-23T07:35:50.358068Z","shell.execute_reply":"2024-11-23T07:35:50.358089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# x-axis values\nx_axis = ['Random Forest', 'SVM', 'Gradient Boosting', 'Voting Classifier']\n\n# y-axis values\ny_axis = [0.85, 0.78, 0.82, voting_accuracy]\n\nplt.bar(x_axis, y_axis)\n\nplt.title('Voting Classifier Accuracy')\nplt.xlabel('Model')\nplt.ylabel('Accuracy')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.359858Z","iopub.status.idle":"2024-11-23T07:35:50.360258Z","shell.execute_reply.started":"2024-11-23T07:35:50.360052Z","shell.execute_reply":"2024-11-23T07:35:50.360071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\nvoting_conf_matrix = confusion_matrix(y_test, voting_predictions)\n\nprint('VotingClassifier confusion matrix:')\nprint(voting_conf_matrix)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.361762Z","iopub.status.idle":"2024-11-23T07:35:50.362062Z","shell.execute_reply.started":"2024-11-23T07:35:50.361931Z","shell.execute_reply":"2024-11-23T07:35:50.361945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# bar chart of the accuracy of each estimator\nestimators = ['rf', 'svm', 'gb', 'VotingClassifier']\naccuracies = [0.92, 0.91, 0.93, voting_accuracy]\nplt.bar(estimators, accuracies, color=['r', 'g', 'b', 'black'])\n\nplt.xlabel('Estimator')\nplt.ylabel('Accuracy')\nplt.title('Accuracy of VotingClassifier and Base Estimators')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.363167Z","iopub.status.idle":"2024-11-23T07:35:50.363469Z","shell.execute_reply.started":"2024-11-23T07:35:50.363324Z","shell.execute_reply":"2024-11-23T07:35:50.363339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\n# Random Forests confusion matrix.\nsns.heatmap(rf_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('Random Forests Confusion Matrix')\nplt.show()\n\n# SVM confusion matrix.\nsns.heatmap(svm_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('SVM Confusion Matrix')\nplt.show()\n\n# Gradient Boosting confusion matrix.\nsns.heatmap(gb_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('Gradient Boosting Confusion Matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.365145Z","iopub.status.idle":"2024-11-23T07:35:50.365453Z","shell.execute_reply.started":"2024-11-23T07:35:50.36531Z","shell.execute_reply":"2024-11-23T07:35:50.365324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Giả sử bạn đã có các ma trận nhầm lẫn\n# rf_conf_matrix, svm_conf_matrix, gb_conf_matrix đều là các ma trận nhầm lẫn của các mô hình.\n\n# Tăng kích thước figure để hiển thị nhiều ô hơn\nplt.figure(figsize=(12, 10))  # Điều chỉnh kích thước của figure\n\n# Vẽ ma trận nhầm lẫn của Random Forest\nplt.subplot(131)  # Vẽ ma trận nhầm lẫn đầu tiên vào ô 1 trong lưới 1x3\nsns.heatmap(rf_conf_matrix, annot=True, fmt='.2f', cmap='Blues', annot_kws={\"size\": 12}, cbar=False)\nplt.title('Random Forest Confusion Matrix')\n\n# Vẽ ma trận nhầm lẫn của SVM\nplt.subplot(132)  # Vẽ ma trận nhầm lẫn thứ 2 vào ô 2 trong lưới 1x3\nsns.heatmap(svm_conf_matrix, annot=True, fmt='.2f', cmap='Blues', annot_kws={\"size\": 12}, cbar=False)\nplt.title('SVM Confusion Matrix')\n\n# Vẽ ma trận nhầm lẫn của Gradient Boosting\nplt.subplot(133)  # Vẽ ma trận nhầm lẫn thứ 3 vào ô 3 trong lưới 1x3\nsns.heatmap(gb_conf_matrix, annot=True, fmt='.2f', cmap='Blues', annot_kws={\"size\": 12}, cbar=False)\nplt.title('Gradient Boosting Confusion Matrix')\n\n# Hiển thị các ma trận nhầm lẫn\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.366778Z","iopub.status.idle":"2024-11-23T07:35:50.367097Z","shell.execute_reply.started":"2024-11-23T07:35:50.366962Z","shell.execute_reply":"2024-11-23T07:35:50.366976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize a sample image\ndef plot_dicom_image(image_path):\n    ds = pydicom.dcmread(image_path)\n    plt.imshow(ds.pixel_array, cmap=plt.cm.bone)\n    plt.axis('off')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.371506Z","iopub.status.idle":"2024-11-23T07:35:50.371869Z","shell.execute_reply.started":"2024-11-23T07:35:50.371693Z","shell.execute_reply":"2024-11-23T07:35:50.371709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_dicom_image2(image_path, figsize=(10, 10), window_center=40, window_width=80):\n  \"\"\"Plot DICOM image using matplotlib.pyplot, with windowing applied.\n\n  Args:\n    image_path: The path to the DICOM image file.\n    figsize: The size of the figure in inches.\n    window_center: The window center value.\n    window_width: The window width value.\n  \"\"\"\n\n  ds = pydicom.dcmread(image_path)\n\n  # Check if the image is windowed.\n  if ds.WindowCenter and ds.WindowWidth:\n    # Apply the windowing.\n    image = ds.pixel_array * (ds.WindowWidth / 10.0) + ds.WindowCenter\n  else:\n    image = ds.pixel_array\n\n  # Create a new figure and plot the image.\n  fig, ax = plt.subplots(1, 1, figsize=figsize)\n  ax.imshow(image, cmap=plt.cm.bone)\n  ax.axis('off')\n\n  # Add a title to the figure with the patient's name.\n  # If the PatientName attribute is not present, use an empty string.\n  patient_name = ds.get('PatientName', '')\n  ax.set_title(patient_name)\n\n  plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.372978Z","iopub.status.idle":"2024-11-23T07:35:50.373275Z","shell.execute_reply.started":"2024-11-23T07:35:50.373133Z","shell.execute_reply":"2024-11-23T07:35:50.373148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_image_path = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/49954/41479/378.dcm'\nplot_dicom_image(sample_image_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.374407Z","iopub.status.idle":"2024-11-23T07:35:50.374723Z","shell.execute_reply.started":"2024-11-23T07:35:50.374581Z","shell.execute_reply":"2024-11-23T07:35:50.374594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dicom_images(directory):\n    dicom_images = []\n    for filename in os.listdir(directory):\n        if filename.endswith(\".dcm\"):\n            dicom_file = os.path.join(directory, filename)\n            dicom_image = pydicom.dcmread(dicom_file)\n            dicom_images.append(dicom_image)\n    return dicom_images\n\ndef rescale_pixel_array(pixel_array, window_level, window_width):\n    # Rescale the pixel values based on the window level and window width\n    min_value = window_level - window_width // 2\n    max_value = window_level + window_width // 2\n    rescaled_pixel_array = np.clip(pixel_array, min_value, max_value)\n    rescaled_pixel_array = (rescaled_pixel_array - min_value) / (max_value - min_value)\n    return rescaled_pixel_array\n\ndef visualize_dicom_images(dicom_images, num_rows=4, num_cols=4, window_level=40, window_width=80):\n    fig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 15))\n    for i, ax in enumerate(axes.flat):\n        if i < len(dicom_images):\n            dicom_image = dicom_images[i]\n            image_data = dicom_image.pixel_array.astype(np.float32)\n            rescaled_image = rescale_pixel_array(image_data, window_level, window_width)\n            ax.imshow(rescaled_image, cmap=plt.cm.bone)\n            ax.axis(\"off\")\n            ax.set_title(f\"Slice {i+1}\")\n\n    # Hide any empty subplots\n    for i in range(len(dicom_images), num_rows*num_cols):\n        axes.flat[i].axis(\"off\")\n\n    # Add a color bar to indicate pixel intensity values\n    cax = fig.add_axes([0.92, 0.15, 0.02, 0.7])\n    norm = plt.cm.colors.Normalize(vmin=0, vmax=1)\n    cbar = plt.colorbar(plt.cm.ScalarMappable(norm=norm, cmap=plt.cm.bone), cax=cax)\n    cbar.ax.set_ylabel(\"Pixel Intensity\")\n\n    plt.tight_layout()\n    plt.show()\n\nif __name__ == \"def load_dicom_images(directory):\n    dicom_images = []\n    for filename in os.listdir(directory):\n        if filename.endswith(\".dcm\"):\n            dicom_file = os.path.join(directory, filename)\n            dicom_image = pydicom.dcmread(dicom_file)\n            dicom_images.append(dicom_image)\n    return dicom_images\n\ndef rescale_pixel_array(pixel_array, window_level, window_width):\n    # Rescale the pixel values based on the window level and window width\n    min_value = window_level - window_width // 2\n    max_value = window_level + window_width // 2\n    rescaled_pixel_array = np.clip(pixel_array, min_value, max_value)\n    rescaled_pixel_array = (rescaled_pixel_array - min_value) / (max_value - min_value)\n    return rescaled_pixel_array\n\ndef visualize_dicom_images(dicom_images, num_rows=4, num_cols=4, window_level=40, window_width=80):\n    fig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 15))\n    for i, ax in enumerate(axes.flat):\n        if i < len(dicom_images):\n            dicom_image = dicom_images[i]\n            image_data = dicom_image.pixel_array.astype(np.float32)\n            rescaled_image = rescale_pixel_array(image_data, window_level, window_width)\n            ax.imshow(rescaled_image, cmap=plt.cm.bone)\n            ax.axis(\"off\")\n            ax.set_title(f\"Slice {i+1}\")\n\n    # Hide any empty subplots\n    for i in range(len(dicom_images), num_rows*num_cols):\n        axes.flat[i].axis(\"off\")\n\n    # Add a color bar to indicate pixel intensity values\n    cax = fig.add_axes([0.92, 0.15, 0.02, 0.7])\n    norm = plt.cm.colors.Normalize(vmin=0, vmax=1)\n    cbar = plt.colorbar(plt.cm.ScalarMappable(norm=norm, cmap=plt.cm.bone), cax=cax)\n    cbar.ax.set_ylabel(\"Pixel Intensity\")\n\n    plt.tight_layout()\n    plt.show()\n\nif __name__ == \"__main__\":\n    # Replace 'path_to_directory' with the actual path where your DICOM images are located\n    path_to_directory = \"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/49954/41479\"\n    dicom_images = load_dicom_images(path_to_directory)\n    visualize_dicom_images(dicom_images, num_rows=3, num_cols=3, window_level=40, window_width=80)__main__\":\n    # Replace 'path_to_directory' with the actual path where your DICOM images are located\n    path_to_directory = \"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/49954/41479\"\n    dicom_images = load_dicom_images(path_to_directory)\n    visualize_dicom_images(dicom_images, num_rows=3, num_cols=3, window_level=40, window_width=80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.376319Z","iopub.status.idle":"2024-11-23T07:35:50.376637Z","shell.execute_reply.started":"2024-11-23T07:35:50.376485Z","shell.execute_reply":"2024-11-23T07:35:50.376501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\n# Random Forests confusion matrix.\nsns.heatmap(rf_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('Random Forests Confusion Matrix')\nplt.show()\n\n# SVM confusion matrix.\nsns.heatmap(svm_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('SVM Confusion Matrix')\nplt.show()\n\n# Gradient Boosting confusion matrix.\nsns.heatmap(gb_conf_matrix, annot=True, fmt='.2f', cmap='Blues')\nplt.title('Gradient Boosting Confusion Matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.378288Z","iopub.status.idle":"2024-11-23T07:35:50.378596Z","shell.execute_reply.started":"2024-11-23T07:35:50.378451Z","shell.execute_reply":"2024-11-23T07:35:50.378466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Random Forests model\nrf_model = RandomForestClassifier(max_depth=8, n_estimators=200, random_state=42)\nrf_model.fit(X_train, y_train)\nrf_pred = rf_model.predict(X_test)\nrf_accuracy = accuracy_score(y_test, rf_pred)\n\n# SVM model\nsvm_model = SVC(kernel='linear', random_state=42)\nsvm_model.fit(X_train, y_train)\nsvm_pred = svm_model.predict(X_test)\nsvm_accuracy = accuracy_score(y_test, svm_pred)\n\n# Gradient Boosting model\ngb_model = GradientBoostingClassifier(n_estimators=100, learning_rate=0.1, max_depth=4)\ngb_model.fit(X_train, y_train)\ngb_pred = gb_model.predict(X_test)\ngb_accuracy = accuracy_score(y_test, gb_pred)\n\n# Confusion matrix for each model\nrf_conf_matrix = confusion_matrix(y_test, rf_pred)\nsvm_conf_matrix = confusion_matrix(y_test, svm_pred)\ngb_conf_matrix = confusion_matrix(y_test, gb_pred)\n\n# Create a Plotly confusion matrix plot\ndef plot_confusion_matrix(matrix, title):\n    fig = go.Figure(data=go.Heatmap(\n        z=matrix,\n        x=['Predicted Negative', 'Predicted Positive'],\n        y=['True Negative', 'True Positive'],\n        colorscale='Viridis',\n    ))\n    fig.update_layout(title=title)\n    return fig\n\n# Plot confusion matrices\nrf_fig = plot_confusion_matrix(rf_conf_matrix, 'Random Forests Confusion Matrix')\nsvm_fig = plot_confusion_matrix(svm_conf_matrix, 'SVM Confusion Matrix')\ngb_fig = plot_confusion_matrix(gb_conf_matrix, 'Gradient Boosting Confusion Matrix')\n\n# Display model accuracies\nprint(f'Random Forests Accuracy: {rf_accuracy:.2f}')\nprint(f'SVM Accuracy: {svm_accuracy:.2f}')\nprint(f'Gradient Boosting Accuracy: {gb_accuracy:.2f}')\n\nrf_fig.show()\nsvm_fig.show()\ngb_fig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:35:50.380129Z","iopub.status.idle":"2024-11-23T07:35:50.380427Z","shell.execute_reply.started":"2024-11-23T07:35:50.380287Z","shell.execute_reply":"2024-11-23T07:35:50.380302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!cp -r '/kaggle/input/contrails-libraries/pretrainedmodels-0.7.4/' './'\n!cp -r '/kaggle/input/contrails-libraries/efficientnet_pytorch-0.7.1/' './'\n\n!pip -q install /kaggle/input/dicomsdl--0-109-2/dicomsdl-0.109.2-cp310-cp310-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n!pip -q install '/kaggle/input/contrails-libraries/segmentation_models_pytorch-0.3.3-py3-none-any.whl' --no-deps\n!pip -q install /kaggle/input/contrails-model-def1/einops-0.6.1-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:58:17.336656Z","iopub.execute_input":"2024-11-23T07:58:17.33757Z","iopub.status.idle":"2024-11-23T07:58:38.657254Z","shell.execute_reply.started":"2024-11-23T07:58:17.337536Z","shell.execute_reply":"2024-11-23T07:58:38.656025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nsys.path.append('./pretrainedmodels-0.7.4/pretrainedmodels-0.7.4/')\nsys.path.append('./efficientnet_pytorch-0.7.1/efficientnet_pytorch-0.7.1/')\nsys.path.append(\"/kaggle/input/rsna-abd-models-classes/\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:58:38.659261Z","iopub.execute_input":"2024-11-23T07:58:38.659571Z","iopub.status.idle":"2024-11-23T07:58:38.664574Z","shell.execute_reply.started":"2024-11-23T07:58:38.659548Z","shell.execute_reply":"2024-11-23T07:58:38.663734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport copy\nimport time\nimport numpy as np\nimport pandas as pd\nfrom glob import glob\nfrom tqdm import tqdm\n\nimport cv2\nfrom PIL import Image\nimport pydicom\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport matplotlib.pyplot as plt\n\nimport torch\nfrom torch import nn\nimport torch.nn.functional as F\n\nimport timm\nimport segmentation_models_pytorch as smp\nfrom models import *\n\nimport dicomsdl\ndef __dataset__to_numpy_image(self, index=0):\n    info = self.getPixelDataInfo()\n    dtype = info['dtype']\n    if info['SamplesPerPixel'] != 1:\n        raise RuntimeError('SamplesPerPixel != 1')\n    else:\n        shape = [info['Rows'], info['Cols']]\n    outarr = np.empty(shape, dtype=dtype)\n    self.copyFrameData(index, outarr)\n    return outarr\ndicomsdl._dicomsdl.DataSet.to_numpy_image = __dataset__to_numpy_image   \n\n\ntorch.cuda.set_device('cuda:0')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:58:38.665669Z","iopub.execute_input":"2024-11-23T07:58:38.665928Z","iopub.status.idle":"2024-11-23T07:58:49.168084Z","shell.execute_reply.started":"2024-11-23T07:58:38.665907Z","shell.execute_reply":"2024-11-23T07:58:49.167278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def glob_sorted(path):\n    return sorted(glob(path), key=lambda x: int(x.split('/')[-1].split('.')[0]))\n\ndef get_rescaled_image(dcm, img):\n    resI, resS = dcm.RescaleIntercept, dcm.RescaleSlope\n    img = resS * img + resI\n    return img\n\ndef get_windowed_image(img, WL=50, WW=400):\n    upper, lower = WL+WW//2, WL-WW//2\n    X = np.clip(img.copy(), lower, upper)\n    X = X - np.min(X)\n    X = X / np.max(X)\n    X = (X*255.0).astype('uint8')\n    \n    return X\n\ndef standardize_pixel_array(dcm, pixel_array):\n    \"\"\"\n    Source : https://www.kaggle.com/competitions/rsna-2023-abdominal-trauma-detection/discussion/427217\n    \"\"\"\n    # Correct DICOM pixel_array if PixelRepresentation == 1.\n    #pixel_array = dcm.pixel_array\n    \n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        pixel_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n\n    intercept = float(dcm.RescaleIntercept)\n    slope = float(dcm.RescaleSlope)\n    center = int(dcm.WindowCenter)\n    width = int(dcm.WindowWidth)\n    low = center - width / 2\n    high = center + width / 2    \n    \n    pixel_array = (pixel_array * slope) + intercept\n    pixel_array = np.clip(pixel_array, low, high)\n\n    return pixel_array\n\ndef load_volume(dcms):\n    volume = []\n    pos_zs = []\n    \n    for dcm_path in dcms:\n        pydcm = pydicom.dcmread(dcm_path)\n        \n        pos_z = pydcm[(0x20, 0x32)].value[-1]\n        pos_zs.append(pos_z)\n        \n        dcm = dicomsdl.open(dcm_path)\n        \n        orig_image = dcm.to_numpy_image()\n        image = get_rescaled_image(dcm, orig_image)\n        image = get_windowed_image(image)\n        \n        if np.min(image)<0:\n            image = image + np.abs(np.min(image))\n        \n        image = image / image.max()\n        image = (image * 255).astype(np.uint8)\n        volume.append(image)\n    \n    return np.stack(volume)\n\n\ndef process_volume(volume):\n    volume = np.stack([cv2.resize(x, (128, 128)) for x in volume])\n    \n    volumes = []\n    cuts = [(x, x+32) for x in np.arange(0, volume.shape[0], 32)[:-1]]\n    \n    if cuts:\n        for cut in cuts:\n            volumes.append(volume[cut[0]:cut[1]])\n        volumes = np.stack(volumes)\n    else:\n        volumes = np.zeros((1, 32, 128, 128), dtype=np.uint8)\n        volumes[0, :len(volume)] = volume\n    \n    if cuts:\n        last_volume = np.zeros((1, 32, 128, 128), dtype=np.uint8)\n        last_volume[0, :volume[cuts[-1][1]:].shape[0]] =  volume[cuts[-1][1]:]\n        volumes = np.concatenate([volumes, last_volume])\n    \n    volumes = torch.as_tensor(volumes).float()\n    \n    return volumes\n\n\ndef get_volume_data(grd, step=96, stride=1, stride_cutoff=200):\n    volumes = []\n    \n    if len(grd)>stride_cutoff:\n        grd = grd[::stride]\n\n    take_last = False\n    if not str(len(grd)/step).endswith('.0'):\n        take_last = True\n\n    started = False\n    for i in range(len(grd)//step):\n        rows = grd[i*step:(i+1)*step]\n\n        if len(rows)!=step:\n            rows = pd.DataFrame([rows.iloc[int(x*len(rows))] for x in np.arange(0, 1, 1/step)])\n\n        volumes.append(rows)\n\n        started = True\n\n    if not started:\n        rows = grd\n        rows = pd.DataFrame([rows.iloc[int(x*len(rows))] for x in np.arange(0, 1, 1/step)])\n        volumes.append(rows)\n\n    if take_last:\n        rows = grd[-step:]\n        if len(rows)==step:\n            volumes.append(rows)\n\n    return volumes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:58:54.131575Z","iopub.execute_input":"2024-11-23T07:58:54.132401Z","iopub.status.idle":"2024-11-23T07:58:54.148193Z","shell.execute_reply.started":"2024-11-23T07:58:54.132372Z","shell.execute_reply":"2024-11-23T07:58:54.147434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMAGE_FOLDER = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/'\n\npatient = '26501'  # We only predict a single patient in this notebook\n\ntest_augs = A.Compose([\n    A.Resize(384, 384),\n    ToTensorV2()\n])\n\npatient","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:58:57.056217Z","iopub.execute_input":"2024-11-23T07:58:57.056532Z","iopub.status.idle":"2024-11-23T07:58:57.06305Z","shell.execute_reply.started":"2024-11-23T07:58:57.056507Z","shell.execute_reply":"2024-11-23T07:58:57.062094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = f'/kaggle/input/rsna-abd-models/try3_seg_resnet18d_v3/zip/0.pth'\nst = torch.load(path, map_location='cpu')\nmodel_3dseg = convert_3d(SegmentationModel())\nmodel_3dseg.load_state_dict(st)\nmodel_3dseg.eval()\nmodel_3dseg.cuda()\n    \n    \npath = f\"/kaggle/input/coatmed384ourdataseed6969/3.pth\"\nst = torch.load(path, map_location='cpu')\nmodel_organs = Model4(num_classes=10, seg_classes=4, arch='medium', mask_head=False)\nmodel_organs.load_state_dict(st)\nmodel_organs.cuda()\nmodel_organs.eval()\n\n\npath = f\"/kaggle/input/coatsmall384extravast4funet/3.pth\"\nst = torch.load(path, map_location='cpu')\nmodel_extrav = Model4(num_classes=2, seg_classes=4, arch='small', mask_head=False)\nmodel_extrav.load_state_dict(st)\nmodel_extrav.cuda()\nmodel_extrav.eval()\n\nprint('''We load 3 models here:\n1: 3d semantic segmentation model for segment organs\n2: 2.5d classification model for classify organs\n3: 2.5d classification model for classify extravasation\n''')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:59:05.97514Z","iopub.execute_input":"2024-11-23T07:59:05.975859Z","iopub.status.idle":"2024-11-23T07:59:12.130251Z","shell.execute_reply.started":"2024-11-23T07:59:05.975828Z","shell.execute_reply":"2024-11-23T07:59:12.129394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PATIENT_TO_PREDICTION = {}\nPATIENT_TO_PREDICTION2 = {}\n\nfinal_outputs = []\nfinal_outputs2 = []\n\nstudies = os.listdir(f'{IMAGE_FOLDER}/{patient}')\nfor study in studies:\n\n    files = glob_sorted(f\"{IMAGE_FOLDER}/{patient}/{study}/*\")\n\n    volume = load_volume(files)\n    file_to_volume = {file: vol for file, vol in zip(files, volume)}\n\n    volumes = process_volume(volume)\n    volumes_seg = predict_segmentation(volumes, [model_3dseg])\n    volume_seg = np.concatenate(volumes_seg.transpose(0, 2, 1, 3, 4))[:len(volume)]\n    \n    vis_seg_0 = volumes[volumes.shape[0]//2, 16].numpy().astype(np.uint8)\n    vis_seg_1 = volumes_seg[volumes_seg.shape[0]//2, :, 16]\n    vis_seg_1[0] += vis_seg_1[3]\n    vis_seg_1[1] += vis_seg_1[4]\n    vis_seg_1 = (vis_seg_1[:3].transpose(1,2,0).clip(0, 1) * 255).astype(np.uint8)\n\n#     print(volumes.shape, volumes_seg.shape)  # torch.Size([7, 32, 128, 128]) (7, 5, 32, 128, 128)\n\n    msk = volume_seg.max(0).max(0)\n    ys, xs = np.where(msk)\n    y1, y2, x1, x2 = np.min(ys) / 128, np.max(ys) / 128, np.min(xs) / 128, np.max(xs) / 128\n\n    files = pd.DataFrame({\"file\": files})\n    files_volumes = get_volume_data(files, step=96, stride=2, stride_cutoff=400)\n\n    first = True\n\n    del volumes, volumes_seg, volume_seg, volume\n    gc.collect()\n\n    for file_volume in files_volumes:\n        volume = np.stack([file_to_volume[file] for file in file_volume.file])\n\n        if first:\n            h, w = volume.shape[1:]\n            y1, y2, x1, x2 = int(y1*h), int(y2*h), int(x1*w), int(x2*w)\n        volume2 = volume\n        \n        #### CROPPED #####\n        volume = volume[:, y1:y2, x1:x2]\n\n        vols = []\n        NC = 3\n        for i in range(len(volume)//NC):\n            vols.append(volume[i*NC:(i+1)*NC])\n        vol = np.stack(vols, 0).transpose(0, 2, 3, 1)\n\n        volume_ = []\n        for image in vol:\n            image = image.astype(np.float32) / 255\n            transformed = test_augs(image=image)\n            image = transformed['image']\n            volume_.append(image)\n        volume = torch.stack(volume_).float()\n        volume = volume.cuda()\n\n        #### UNCROPPED #####\n        vols = []\n        NC = 3\n        for i in range(len(volume2)//NC):\n            vols.append(volume2[i*NC:(i+1)*NC])\n        vol = np.stack(vols, 0).transpose(0, 2, 3, 1)\n\n        volume_ = []\n        for image in vol:\n            image = image.astype(np.float32) / 255\n            transformed = test_augs(image=image)\n            image = transformed['image']\n            volume_.append(image)\n\n        volume2 = torch.stack(volume_).float()\n        volume2 = volume2.cuda()\n\n        outputs = []\n        outputs2 = []\n\n        with torch.no_grad():\n            with torch.cuda.amp.autocast(enabled=True):\n\n                outs = model_organs(volume.unsqueeze(0))\n                outs = outs.float().sigmoid()\n                outputs.append(outs)\n\n                outs = model_extrav(volume2.unsqueeze(0))[:, :, [1, 0]]\n                outs = outs.float().sigmoid()\n                outputs2.append(outs)\n\n        torch.cuda.empty_cache()\n\n        outputs = torch.stack(outputs)[:, 0].mean(0)\n        outputs2 = torch.stack(outputs2)[:, 0].mean(0)\n\n        final_outputs.append(outputs.detach().cpu().numpy())\n        final_outputs2.append(outputs2.detach().cpu().numpy())\n\n        first = False\n\n        torch.cuda.empty_cache()\n\n\nlast_final_outputs = final_outputs.copy()\nlast_final_outputs2 = final_outputs2.copy()\n\nfinal_outputs = np.concatenate(final_outputs)\nfinal_outputs2 = np.concatenate(final_outputs2)\n\nfinal_predictions = final_outputs.max(0)\nfinal_predictions2 = final_outputs2.max(0)\n\nPATIENT_TO_PREDICTION[patient] = final_predictions\nPATIENT_TO_PREDICTION2[patient] = final_predictions2\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T08:02:53.870099Z","iopub.execute_input":"2024-11-23T08:02:53.870428Z","iopub.status.idle":"2024-11-23T08:02:56.276617Z","shell.execute_reply.started":"2024-11-23T08:02:53.870401Z","shell.execute_reply":"2024-11-23T08:02:56.275333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Giả sử bạn có dữ liệu hình ảnh cho vis_seg_0 và vis_seg_1\n# Tạo dữ liệu ngẫu nhiên cho ví dụ (thay thế bằng dữ liệu thực tế của bạn)\nvis_seg_0 = np.random.rand(100, 100)  # Hình ảnh đầu tiên (100x100)\nvis_seg_1 = np.random.rand(100, 100)  # Hình ảnh thứ hai (100x100)\n\n# Tạo một figure với 2 subplot\nfig, axarr = plt.subplots(1, 2, figsize=(10, 4))\n\n# Hiển thị hình ảnh đầu tiên\naxarr[0].imshow(vis_seg_0, cmap='gray')  # Có thể thêm cmap='gray' nếu dữ liệu là hình ảnh xám\naxarr[0].axis('off')  # Tắt trục\n\n# Hiển thị hình ảnh thứ hai\naxarr[1].imshow(vis_seg_1, cmap='gray')  # Có thể thêm cmap='gray' nếu dữ liệu là hình ảnh xám\naxarr[1].axis('off')  # Tắt trục\n\n# Tinh chỉnh bố cục\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T08:01:19.971151Z","iopub.execute_input":"2024-11-23T08:01:19.971498Z","iopub.status.idle":"2024-11-23T08:01:20.071112Z","shell.execute_reply.started":"2024-11-23T08:01:19.971469Z","shell.execute_reply":"2024-11-23T08:01:20.07015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axarr = plt.subplots(1, 2, figsize=(10, 4))\n\naxarr[0].imshow(vis_seg_0)\naxarr[0].axis('off') \n\naxarr[1].imshow(vis_seg_1)\naxarr[1].axis('off') \n\nplt.tight_layout() \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-23T07:59:26.893249Z","iopub.execute_input":"2024-11-23T07:59:26.893578Z","iopub.status.idle":"2024-11-23T07:59:27.401463Z","shell.execute_reply.started":"2024-11-23T07:59:26.893553Z","shell.execute_reply":"2024-11-23T07:59:27.400328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}