{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.simplefilter(action='ignore')\n\nimport os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom tqdm import tqdm\n\ntqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:58:39.014469Z","iopub.execute_input":"2023-05-24T14:58:39.014813Z","iopub.status.idle":"2023-05-24T14:58:40.003097Z","shell.execute_reply.started":"2023-05-24T14:58:39.014779Z","shell.execute_reply":"2023-05-24T14:58:40.00206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datadir = '/kaggle/input/cropped-images/images'","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:58:40.005331Z","iopub.execute_input":"2023-05-24T14:58:40.006023Z","iopub.status.idle":"2023-05-24T14:58:40.010713Z","shell.execute_reply.started":"2023-05-24T14:58:40.005988Z","shell.execute_reply":"2023-05-24T14:58:40.009871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = [[i, i.split('-')[1].split('.')[0]] for i in os.listdir(datadir)]","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:58:40.019104Z","iopub.execute_input":"2023-05-24T14:58:40.020151Z","iopub.status.idle":"2023-05-24T14:58:40.360738Z","shell.execute_reply.started":"2023-05-24T14:58:40.020119Z","shell.execute_reply":"2023-05-24T14:58:40.359727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.DataFrame(data, columns=['path', 'class'])\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:58:40.362378Z","iopub.execute_input":"2023-05-24T14:58:40.363059Z","iopub.status.idle":"2023-05-24T14:58:40.390374Z","shell.execute_reply.started":"2023-05-24T14:58:40.363004Z","shell.execute_reply":"2023-05-24T14:58:40.389273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['StudyInstanceUID'] = data.path.progress_apply(lambda x: x.split('-')[0].split('_')[0])","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:58:40.391853Z","iopub.execute_input":"2023-05-24T14:58:40.392559Z","iopub.status.idle":"2023-05-24T14:58:40.427957Z","shell.execute_reply.started":"2023-05-24T14:58:40.392524Z","shell.execute_reply":"2023-05-24T14:58:40.427093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train =pd.read_csv('/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:58:40.429405Z","iopub.execute_input":"2023-05-24T14:58:40.430059Z","iopub.status.idle":"2023-05-24T14:58:40.458811Z","shell.execute_reply.started":"2023-05-24T14:58:40.430009Z","shell.execute_reply":"2023-05-24T14:58:40.457924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_data = pd.merge(train, data, on='StudyInstanceUID', how='inner')\nfinal_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:58:40.460309Z","iopub.execute_input":"2023-05-24T14:58:40.460965Z","iopub.status.idle":"2023-05-24T14:58:40.496335Z","shell.execute_reply.started":"2023-05-24T14:58:40.460932Z","shell.execute_reply":"2023-05-24T14:58:40.495491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_df_by_cat(category):\n    cat = final_data[final_data['class']==category]\n    cat = cat[['path', category]]\n    cat.rename(columns = { category : 'Output'}, inplace = True)\n\n    return cat","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:58:40.500936Z","iopub.execute_input":"2023-05-24T14:58:40.503901Z","iopub.status.idle":"2023-05-24T14:58:40.511355Z","shell.execute_reply.started":"2023-05-24T14:58:40.503867Z","shell.execute_reply":"2023-05-24T14:58:40.5105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_C1 = get_df_by_cat('C1')\nfinal_C2 = get_df_by_cat('C2')\nfinal_C3 = get_df_by_cat('C3')\nfinal_C4 = get_df_by_cat('C4')\nfinal_C5 = get_df_by_cat('C5')\nfinal_C6 = get_df_by_cat('C6')\nfinal_C7 = get_df_by_cat('C7')","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:58:40.518311Z","iopub.execute_input":"2023-05-24T14:58:40.521417Z","iopub.status.idle":"2023-05-24T14:58:40.552996Z","shell.execute_reply.started":"2023-05-24T14:58:40.521378Z","shell.execute_reply":"2023-05-24T14:58:40.552181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, axes = plt.subplots(1, 7, figsize=(30, 6))\n\nfor i, j in enumerate([final_C1, final_C2, final_C3, final_C4, final_C5, final_C6, final_C7]):\n    ax = sns.countplot(j, x='Output', ax=axes[i])\n    ax.set(xlabel=f'C{i+1}')\n    ax.set(ylabel=None)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:58:40.557749Z","iopub.execute_input":"2023-05-24T14:58:40.560665Z","iopub.status.idle":"2023-05-24T14:58:41.576895Z","shell.execute_reply.started":"2023-05-24T14:58:40.560632Z","shell.execute_reply":"2023-05-24T14:58:41.57579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import losses, callbacks\nfrom tensorflow.keras.applications import ResNet50\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score\nfrom IPython.display import FileLink, display","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:58:41.581272Z","iopub.execute_input":"2023-05-24T14:58:41.585307Z","iopub.status.idle":"2023-05-24T14:58:48.185395Z","shell.execute_reply.started":"2023-05-24T14:58:41.585269Z","shell.execute_reply":"2023-05-24T14:58:48.184441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2023-05-24T14:58:48.186838Z","iopub.execute_input":"2023-05-24T14:58:48.187652Z","iopub.status.idle":"2023-05-24T14:58:48.194077Z","shell.execute_reply.started":"2023-05-24T14:58:48.187613Z","shell.execute_reply":"2023-05-24T14:58:48.193109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_confusion_matrix(y_test, y_pred, color=\"Oranges\"):\n    plt.figure(figsize=(3,3))\n    labels = np.unique(y_pred)\n    cm_df = pd.DataFrame(confusion_matrix(y_test, y_pred), index=labels, columns=labels)\n    sns.heatmap(cm_df, annot=True, fmt='g', cmap=color)\n    plt.xlabel(\"Actual\")\n    plt.ylabel(\"Predicted\")\n    plt.show()\n\n    \ndef build_model(data, tl_name='resnet', batch_size=16):\n    # Split data into train and test set\n    train, test = train_test_split(\n        data,\n        test_size=0.33, \n        shuffle=True\n    )\n    \n    # Loading train images as numpy array from path\n    train_images = train['path'].progress_apply(lambda x: np.load(f'{datadir}/{x}')).to_numpy()\n    train_images = np.stack(train_images)\n    train_images = np.reshape(train_images,(-1,512,512,1))\n    \n    # Loading Test Images as numpy array from path\n    test_images = test['path'].progress_apply(lambda x: np.load(f'{datadir}/{x}')).to_numpy()\n    test_images = np.stack(test_images)\n    test_images = np.reshape(test_images,(-1,512,512,1))\n    \n    # Loading labels\n    train_labels = train.Output.to_numpy()\n    test_labels = test.Output.to_numpy()\n    \n    cat = data.path.to_numpy()[0].split('-')[1].split('.')[0]\n    model_name = f'spine_{cat}.h5'\n        \n    print(f\"\"\"\n\n    ***************************** Category : {cat}, Output file name: {model_name} ****************************************\n\n    \"\"\")\n\n    # Define EarlyStopping\n    early_stopping = callbacks.EarlyStopping(\n        monitor='accuracy',\n        min_delta=0.001, # minimium amount of change to count as an improvement\n        patience=5, # how many epochs to wait before stopping\n        restore_best_weights=True,\n    )\n\n    if tl_name=='resnet':\n        # Load Resnet50 neural network\n        base_model = ResNet50(weights=None, include_top=False, input_shape=(512, 512, 1))\n        \n        # Fit base_model into model\n        model = tf.keras.models.Sequential()\n        model.add(base_model)\n        model.add(tf.keras.layers.Flatten())\n        model.add(tf.keras.layers.Dense(256, activation='relu'))\n        model.add(tf.keras.layers.Dense(1, activation='sigmoid'))\n        \n        # Print Model Summary\n        print(model.summary())\n        \n        # Compile the model\n        model.compile(optimizer=keras.optimizers.Adam(learning_rate=0.0005),\n                  loss=tf.keras.losses.BinaryCrossentropy(),\n                  metrics=['accuracy'],\n        )\n        \n        print(\"\"\"\n        \n        **********************************************  Training Data... **************************************************\n        \n        \"\"\")\n\n        history = model.fit(train_images, train_labels, batch_size=batch_size , epochs=50,use_multiprocessing=True,shuffle=True,callbacks=[early_stopping],validation_data=(test_images,test_labels))\n        \n        print(\"\"\"\n        \n        *******************************************  Dataset Trained Successfully  *****************************************\n        \n        \"\"\")\n        \n        predicted_labels = [1 if i>=0.5 else 0 for i in model.predict(np.array(test_images))]\n        \n        model.evaluate(test_images,test_labels)\n        \n        print(\"\"\"\n        \n        ***************************************** Training Loss vs Validation Loss *****************************************\n        \n        \"\"\")\n        \n        history_df = pd.DataFrame(history.history)\n\n        history_df.loc[:, ['loss', 'val_loss']].plot()\n        plt.show()\n        \n        print(\"\"\"\n        \n        ************************************ Training Accuracy vs Validation Accuracy **************************************\n        \n        \"\"\")\n        \n        history_df.loc[:, ['accuracy', 'val_accuracy']].plot()\n        plt.show()\n\n        print(\"\"\"\n        \n        ***************************************************** Result ********************************************************\n        \n        \"\"\")\n        \n        print(f\"Kappa Score\\t {cohen_kappa_score(test_labels, predicted_labels)}\")\n        print(\"Classification Report:\\n\")        \n        print(classification_report(test_labels, predicted_labels))\n        \n        print(\"\"\"\n        \n        ***************************************************** Plotting The Result ********************************************************\n        \n        \"\"\")\n        \n        pd.DataFrame(classification_report(test_labels, predicted_labels, output_dict=True)).transpose().head(2).drop(columns='support').plot.bar()\n        plt.show()\n        \n        print(\"\"\"\n        \n        ************************************************* Confusion Matrix **************************************************\n        \n        \"\"\")\n\n        # Plot Confusion Matrix\n        plot_confusion_matrix(test_labels, predicted_labels)\n        \n        # Save and Download the model\n        model.save(model_name)\n        print(\"\"\"\n        \n        ************************************************ Download the model **************************************************\n        \n        \"\"\")\n        display(FileLink(model_name))        ","metadata":{"execution":{"iopub.status.busy":"2023-05-24T15:05:00.88315Z","iopub.execute_input":"2023-05-24T15:05:00.883696Z","iopub.status.idle":"2023-05-24T15:05:00.906893Z","shell.execute_reply.started":"2023-05-24T15:05:00.883662Z","shell.execute_reply":"2023-05-24T15:05:00.905813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"build_model(final_C1)","metadata":{"execution":{"iopub.status.busy":"2023-05-24T15:05:01.47612Z","iopub.execute_input":"2023-05-24T15:05:01.476469Z","iopub.status.idle":"2023-05-24T15:09:54.158099Z","shell.execute_reply.started":"2023-05-24T15:05:01.476441Z","shell.execute_reply":"2023-05-24T15:09:54.156935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"build_model(final_C2)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T16:31:41.515824Z","iopub.execute_input":"2023-05-23T16:31:41.516219Z","iopub.status.idle":"2023-05-23T16:48:47.689398Z","shell.execute_reply.started":"2023-05-23T16:31:41.516188Z","shell.execute_reply":"2023-05-23T16:48:47.688385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"build_model(final_C3, batch_size=8)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T16:54:43.906443Z","iopub.execute_input":"2023-05-23T16:54:43.906904Z","iopub.status.idle":"2023-05-23T17:05:43.053607Z","shell.execute_reply.started":"2023-05-23T16:54:43.906867Z","shell.execute_reply":"2023-05-23T17:05:43.052447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"build_model(final_C4, batch_size=8)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T17:07:18.768347Z","iopub.execute_input":"2023-05-23T17:07:18.768751Z","iopub.status.idle":"2023-05-23T17:20:05.279102Z","shell.execute_reply.started":"2023-05-23T17:07:18.768714Z","shell.execute_reply":"2023-05-23T17:20:05.278035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"build_model(final_C5, batch_size=8)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T17:20:10.208145Z","iopub.execute_input":"2023-05-23T17:20:10.208529Z","iopub.status.idle":"2023-05-23T17:26:32.612133Z","shell.execute_reply.started":"2023-05-23T17:20:10.208498Z","shell.execute_reply":"2023-05-23T17:26:32.610551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"build_model(final_C6, batch_size=4)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T16:23:54.113698Z","iopub.status.idle":"2023-05-23T16:23:54.114551Z","shell.execute_reply.started":"2023-05-23T16:23:54.114296Z","shell.execute_reply":"2023-05-23T16:23:54.11432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"build_model(final_C7, batch_size=4)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T16:23:54.115972Z","iopub.status.idle":"2023-05-23T16:23:54.11683Z","shell.execute_reply.started":"2023-05-23T16:23:54.116569Z","shell.execute_reply":"2023-05-23T16:23:54.116591Z"},"trusted":true},"execution_count":null,"outputs":[]}]}