{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":52254,"databundleVersionId":6863140,"sourceType":"competition"},{"sourceId":6983983,"sourceType":"datasetVersion","datasetId":4013814}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#pip install opencv-python\n#pip install tqdm\n","metadata":{"execution":{"iopub.status.busy":"2023-12-25T08:44:44.166583Z","iopub.execute_input":"2023-12-25T08:44:44.167055Z","iopub.status.idle":"2023-12-25T08:44:44.1724Z","shell.execute_reply.started":"2023-12-25T08:44:44.167024Z","shell.execute_reply":"2023-12-25T08:44:44.170835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Dataset Preparation ****","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nimport cv2\n\n# Table reading\nlabels_df = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')\n\n# List of image folder path\nimage_folder_path = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images'\nimage_paths = [os.path.join(image_folder_path, str(patient_id)) for patient_id in labels_df['patient_id']]\n\n# List of labels\nlabels = labels_df.drop('patient_id', axis=1).values\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-25T10:24:50.730816Z","iopub.execute_input":"2023-12-25T10:24:50.732409Z","iopub.status.idle":"2023-12-25T10:24:50.787352Z","shell.execute_reply.started":"2023-12-25T10:24:50.732361Z","shell.execute_reply":"2023-12-25T10:24:50.786036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(image_paths[1])\nprint(labels[1])\n","metadata":{"execution":{"iopub.status.busy":"2023-12-25T10:26:49.780298Z","iopub.execute_input":"2023-12-25T10:26:49.780675Z","iopub.status.idle":"2023-12-25T10:26:49.789245Z","shell.execute_reply.started":"2023-12-25T10:26:49.780647Z","shell.execute_reply":"2023-12-25T10:26:49.786794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Upload images and convert them to a suitable format\n\nimages = []\n#image_counter = 0\npatient_counter = 0\n\nfor image_path in image_paths:\n    series_folders = os.listdir(image_path)\n    patient_counter = patient_counter +1\n    sections = image_path.split('/')\n    last_section = sections[-1]\n    print(last_section , patient_counter)\n    for series_folder in series_folders:\n        series_path = os.path.join(image_path, series_folder)\n        \n        # Load and preprocess the DICOM images\n        for filename in os.listdir(series_path):\n            dicom_filepath = os.path.join(series_path, filename)\n            dicom_data = pydicom.dcmread(dicom_filepath)\n            image_array = dicom_data.pixel_array  # Assuming pixel_array contains the image data\n\n            # Resize the image to a consistent size (adjust the target size accordingly)\n            target_size = (224, 224)  # Change as needed\n            image_array = cv2.resize(image_array, target_size)\n\n            # Append the image and its label to the lists\n            images.append(image_array)\n\n            \n            # Increment the counter and display progress\n            #image_counter += 1\n            #print(f\"Processed {image_counter} images\")\n","metadata":{"execution":{"iopub.status.busy":"2023-12-25T10:27:39.732417Z","iopub.execute_input":"2023-12-25T10:27:39.732833Z","iopub.status.idle":"2023-12-25T10:28:06.799828Z","shell.execute_reply.started":"2023-12-25T10:27:39.732802Z","shell.execute_reply":"2023-12-25T10:28:06.79742Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-12-25T10:32:14.466237Z","iopub.execute_input":"2023-12-25T10:32:14.466619Z","iopub.status.idle":"2023-12-25T10:32:14.474555Z","shell.execute_reply.started":"2023-12-25T10:32:14.466589Z","shell.execute_reply":"2023-12-25T10:32:14.473041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encode labels using One-Hot Encoding\nmlb = MultiLabelBinarizer()\nencoded_labels = mlb.fit_transform(labels)\n\n# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(images, encoded_labels, test_size=0.2, random_state=42)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-25T10:32:56.339739Z","iopub.execute_input":"2023-12-25T10:32:56.340103Z","iopub.status.idle":"2023-12-25T10:32:56.581183Z","shell.execute_reply.started":"2023-12-25T10:32:56.340079Z","shell.execute_reply":"2023-12-25T10:32:56.57931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\n# Save training images and labels\nnp.save('X_train.npy', X_train)\nnp.save('y_train.npy', y_train)\n\n# Save testing images and labels\nnp.save('X_test.npy', X_test)\nnp.save('y_test.npy', y_test)\n\n\n\n\n# Convert encoded labels to a DataFrame for easier saving\nlabels_df = pd.DataFrame(encoded_labels, columns=mlb.classes_)\n\n# Save the DataFrame to a CSV file\nlabels_df.to_csv('encoded_labels.csv', index=False)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-12-25T09:45:58.039171Z","iopub.status.idle":"2023-12-25T09:45:58.039675Z","shell.execute_reply.started":"2023-12-25T09:45:58.03952Z","shell.execute_reply":"2023-12-25T09:45:58.039537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Training Section**","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.applications import ResNet50, InceptionV3, DenseNet121\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\nimport numpy as np\n\n# Load pre-trained models\ndef load_pretrained_model(model_name, input_shape, num_classes):\n    if model_name == 'ResNet50':\n        base_model = ResNet50(weights='imagenet', include_top=False, input_shape=input_shape)\n    elif model_name == 'InceptionV3':\n        base_model = InceptionV3(weights='imagenet', include_top=False, input_shape=input_shape)\n    elif model_name == 'DenseNet121':\n        base_model = DenseNet121(weights='imagenet', include_top=False, input_shape=input_shape)\n    else:\n        raise ValueError(\"Invalid model name\")\n\n    # Add custom classification head\n    x = GlobalAveragePooling2D()(base_model.output)\n    x = Dense(num_classes, activation='sigmoid')(x)\n\n    # Create the model\n    model = Model(inputs=base_model.input, outputs=x)\n    return model\n\n# Define the input shape and number of classes\ninput_shape = (224, 224, 3)  # Adjust for the model's input requirements\nnum_classes = len(labels_df.columns)\n\n# List of pre-trained models to try\npretrained_models = ['ResNet50', 'InceptionV3', 'DenseNet121']\n\n# Initialize dictionaries to store values and metrics\nall_values = {}\nall_metrics = {}\n\nfor model_name in pretrained_models:\n    # Load pre-trained model\n    model = load_pretrained_model(model_name, input_shape, num_classes)\n\n    # Compile the model with different loss functions\n    if model_name in ['ResNet50', 'InceptionV3']:\n        loss_function = 'categorical_crossentropy'\n    else:\n        #loss_function = 'binary_crossentropy'\n        loss_function = 'spare_binary_crossentropy'\n\n    model.compile(optimizer='adam', loss=loss_function, metrics=['accuracy'])\n\n    # Train the model (assuming you have X_train, y_train, X_test, and y_test)\n    history = model.fit(X_train, y_train, epochs=10, batch_size=32, validation_split=0.2)\n\n    # Store training history values\n    all_values[model_name] = {\n        'loss': history.history['loss'],\n        'val_loss': history.history['val_loss'],\n        'accuracy': history.history['accuracy'],\n        'val_accuracy': history.history['val_accuracy']\n    }\n\n    # Evaluate the model\n    y_pred = model.predict(X_test)\n\n    # Convert probabilities to binary predictions\n    threshold = 0.5\n    y_pred_binary = (y_pred > threshold).astype(int)\n\n    # Define evaluation metrics\n    accuracy = accuracy_score(y_test, y_pred_binary)\n    precision = precision_score(y_test, y_pred_binary, average='micro')\n    recall = recall_score(y_test, y_pred_binary, average='micro')\n    f1 = f1_score(y_test, y_pred_binary, average='micro')\n\n    # Store evaluation metrics\n    all_metrics[model_name] = {\n        'accuracy': accuracy,\n        'precision': precision,\n        'recall': recall,\n        'f1_score': f1\n    }\n\n    # Print evaluation metrics\n    print(f'Model: {model_name}')\n    print(f'Accuracy: {accuracy}')\n    print(f'Precision: {precision}')\n    print(f'Recall: {recall}')\n    print(f'F1 Score: {f1}')\n\n    # Save the trained model\n    model.save(f'{model_name}_abdominal_ct_model.h5')\n\n# Save values and metrics to files\nnp.save('all_values.npy', all_values)\nnp.save('all_metrics.npy', all_metrics)\n","metadata":{},"execution_count":null,"outputs":[]}]}