{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nimport pandas.api.types\nimport sklearn.metrics\nwarnings.filterwarnings('ignore')\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.applications import ResNet50, EfficientNetB7, ResNet50V2, VGG19 , InceptionV3\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, MaxPooling2D, Dropout, Flatten, BatchNormalization, Concatenate\nfrom tensorflow.keras import Input\nfrom tensorflow.keras.models import Sequential, Model, load_model\nfrom tensorflow.keras.losses import BinaryCrossentropy, CategoricalCrossentropy\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.optimizers import Adam, SGD\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, accuracy_score, confusion_matrix\n\nfrom PIL import Image\n\ndevices = tf.config.list_physical_devices('GPU')\n\nif len(devices)>0:\n    print(f'GPU is available with {len(devices)} slot(s)')\nelse:\n    print('GPU is not available')\n","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:28:45.965719Z","iopub.execute_input":"2023-10-06T15:28:45.966301Z","iopub.status.idle":"2023-10-06T15:28:56.145469Z","shell.execute_reply.started":"2023-10-06T15:28:45.966275Z","shell.execute_reply":"2023-10-06T15:28:56.144347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    competition = 'RSNA-ATD'\n    \n    model_name = 'Dr_OI_Model'\n    \n    base = '/kaggle/input/radimagenet/RadImageNet/RadImageNet-ResNet50_notop.h5'\n    \n    IsRadImageNet = False\n    \n    RadImageNetModel = \"/kaggle/input/radimagenet/RadImageNet/RadImageNet-InceptionV3_notop.h5\"\n    \n    random_state = 42\n    \n    img_shape = (224,224,3)\n    \n    batch_size = 64\n    \n    epochs = 8\n    \n    loss_cat = 'categorical_crossentropy'\n    \n    loss_bin = 'binary_crossentropy'\n    \n    optimizer = Adam(learning_rate = 0.1, epsilon= 1)\n    \n    target_col  =  ['bowel_injury', 'extravasation_injury', 'liver', 'kidney', 'spleen']\n    \n    rotation = 10\n    \n    width_shift = .1\n    \n    height_shift = .1\n    \n    zoom_range = .2\n    \n    h_flip = True\n    \n    brightness_range = [.9,1.3]\n    \n    fill_mode = 'constant'\n    ","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:28:56.146871Z","iopub.execute_input":"2023-10-06T15:28:56.14747Z","iopub.status.idle":"2023-10-06T15:28:56.236463Z","shell.execute_reply.started":"2023-10-06T15:28:56.147442Z","shell.execute_reply":"2023-10-06T15:28:56.235853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"OI_model = load_model(\"/kaggle/input/rsna-adt-oi-model-v2-h5/RSNA-ADT_OI_model.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:28:56.237308Z","iopub.execute_input":"2023-10-06T15:28:56.23837Z","iopub.status.idle":"2023-10-06T15:29:00.26623Z","shell.execute_reply.started":"2023-10-06T15:28:56.238319Z","shell.execute_reply":"2023-10-06T15:29:00.265268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_image_path(patient_id,series_id,instance_number,train_images = True):\n    if train_images:\n        return f\"/kaggle/input/rsna-atd-512x512-png-v2-dataset/train_images/{patient_id}/{series_id}/{instance_number}.png\"\n    else:\n        return f\"/kaggle/input/rsna-atd-512x512-png-v2-dataset/test_images/{patient_id}/{series_id}/{instance_number}.png\"\n    ","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:00.268418Z","iopub.execute_input":"2023-10-06T15:29:00.268663Z","iopub.status.idle":"2023-10-06T15:29:00.273863Z","shell.execute_reply.started":"2023-10-06T15:29:00.268644Z","shell.execute_reply":"2023-10-06T15:29:00.272575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List of Targets\nInjuries = ['bowel_healthy', 'bowel_injury', \n            'extravasation_healthy', 'extravasation_injury', \n            'kidney_healthy', 'kidney_low', 'kidney_high', \n            'liver_healthy', 'liver_low', 'liver_high', \n            'spleen_healthy', 'spleen_low', 'spleen_high',\n           'any_injury']\n\ndf1 = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')\n\ndf1[Injuries].describe()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:00.275367Z","iopub.execute_input":"2023-10-06T15:29:00.275689Z","iopub.status.idle":"2023-10-06T15:29:00.353937Z","shell.execute_reply.started":"2023-10-06T15:29:00.275663Z","shell.execute_reply":"2023-10-06T15:29:00.353196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ParticipantVisibleError(Exception):\n        pass\n\ndef normalize_probabilities_to_one(df: pd.DataFrame, group_columns: list) -> pd.DataFrame:\n    # Normalize the sum of each row's probabilities to 100%.\n    # 0.75, 0.75 => 0.5, 0.5\n    # 0.1, 0.1 => 0.5, 0.5\n    row_totals = df[group_columns].sum(axis=1)\n    if row_totals.min() == 0:\n        raise ParticipantVisibleError('All rows must contain at least one non-zero prediction')\n    for col in group_columns:\n        df[col] /= row_totals\n    return df\n\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str) -> float:\n    '''\n    Pseudocode:\n    1. For every label group (liver, bowel, etc):\n        - Normalize the sum of each row's probabilities to 100%.\n        - Calculate the sample weighted log loss.\n    2. Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    3. Calculate the sample weighted log loss for the new label group\n    4. Return the average of all of the label group log losses as the final score.\n    '''\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n\n    # Run basic QC checks on the inputs\n    if not pd.api.types.is_numeric_dtype(submission.values):\n        raise ParticipantVisibleError('All submission values must be numeric')\n\n    if not np.isfinite(submission.values).all():\n        raise ParticipantVisibleError('All submission values must be finite')\n\n    if solution.min().min() < 0:\n        raise ParticipantVisibleError('All labels must be at least zero')\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('All predictions must be at least zero')\n\n    # Calculate the label group log losses\n    binary_targets = ['bowel', 'extravasation']\n    triple_level_targets = ['kidney', 'liver', 'spleen']\n    all_target_categories = binary_targets + triple_level_targets\n\n    label_group_losses = []\n    for category in all_target_categories:\n        if category in binary_targets:\n            col_group = [f'{category}_healthy', f'{category}_injury']\n        else:\n            col_group = [f'{category}_healthy', f'{category}_low', f'{category}_high']\n\n        solution = normalize_probabilities_to_one(solution, col_group)\n\n        for col in col_group:\n            if col not in submission.columns:\n                raise ParticipantVisibleError(f'Missing submission column {col}')\n        submission = normalize_probabilities_to_one(submission, col_group)\n        label_group_losses.append(\n            sklearn.metrics.log_loss(\n                y_true=solution[col_group].values,\n                y_pred=submission[col_group].values,\n                sample_weight=solution[f'{category}_weight'].values\n            )\n        )\n\n    # Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    healthy_cols = [x + '_healthy' for x in all_target_categories]\n    any_injury_labels = (1 - solution[healthy_cols]).max(axis=1)\n    any_injury_predictions = (1 - submission[healthy_cols]).max(axis=1)\n    any_injury_loss = sklearn.metrics.log_loss(\n        y_true=any_injury_labels.values,\n        y_pred=any_injury_predictions.values,\n        sample_weight=solution['any_injury_weight'].values\n    )\n\n    label_group_losses.append(any_injury_loss)\n    return np.mean(label_group_losses)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:00.35539Z","iopub.execute_input":"2023-10-06T15:29:00.35574Z","iopub.status.idle":"2023-10-06T15:29:00.365552Z","shell.execute_reply.started":"2023-10-06T15:29:00.355713Z","shell.execute_reply":"2023-10-06T15:29:00.364632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign the appropriate weights to each category\ndef create_training_solution(y_train):\n    sol_train = y_train.copy()\n    \n    # bowel healthy|injury sample weight = 1|2\n    sol_train['bowel_weight'] = np.where(sol_train['bowel_injury'] == 1, 2, 1)\n    \n    # extravasation healthy/injury sample weight = 1|6\n    sol_train['extravasation_weight'] = np.where(sol_train['extravasation_injury'] == 1, 6, 1)\n    \n    # kidney healthy|low|high sample weight = 1|2|4\n    sol_train['kidney_weight'] = np.where(sol_train['kidney_low'] == 1, 2, np.where(sol_train['kidney_high'] == 1, 4, 1))\n    \n    # liver healthy|low|high sample weight = 1|2|4\n    sol_train['liver_weight'] = np.where(sol_train['liver_low'] == 1, 2, np.where(sol_train['liver_high'] == 1, 4, 1))\n    \n    # spleen healthy|low|high sample weight = 1|2|4\n    sol_train['spleen_weight'] = np.where(sol_train['spleen_low'] == 1, 2, np.where(sol_train['spleen_high'] == 1, 4, 1))\n    \n    # any healthy|injury sample weight = 1|6\n    sol_train['any_injury_weight'] = np.where(sol_train['any_injury'] == 1, 6, 1)\n    return sol_train","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:00.36645Z","iopub.execute_input":"2023-10-06T15:29:00.36678Z","iopub.status.idle":"2023-10-06T15:29:00.388189Z","shell.execute_reply.started":"2023-10-06T15:29:00.366763Z","shell.execute_reply":"2023-10-06T15:29:00.387236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Here testing begins!","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/rsna-atd-512x512-png-v2-dataset/test.csv')\ndf_test.sort_values(by = 'patient_id', inplace = True)\ndf_test.reset_index(drop = True,inplace = True)\n\npng_path = []\n\nfor i in range(df_test.shape[0]):\n    patient_id = df_test.iloc[i,:]['patient_id']\n    series_id = df_test.iloc[i,:]['series_id']\n    instance_number = df_test.iloc[i,:]['instance_number']\n    \n    png_path.append(get_image_path(patient_id, series_id, instance_number, train_images= False))\n    \n    \ndf_test['png_path'] = png_path\n\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:00.389226Z","iopub.execute_input":"2023-10-06T15:29:00.389559Z","iopub.status.idle":"2023-10-06T15:29:00.425994Z","shell.execute_reply.started":"2023-10-06T15:29:00.389541Z","shell.execute_reply":"2023-10-06T15:29:00.425219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['png_path'][0]","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:00.426857Z","iopub.execute_input":"2023-10-06T15:29:00.42704Z","iopub.status.idle":"2023-10-06T15:29:00.434107Z","shell.execute_reply.started":"2023-10-06T15:29:00.427024Z","shell.execute_reply":"2023-10-06T15:29:00.432607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_datagen = ImageDataGenerator(rescale=1./255)\n\ntest_generator = valid_datagen.flow_from_dataframe(\n    df_test,\n    directory=None,\n    x_col=\"png_path\",\n    y_col=None,\n    target_size=(CFG.img_shape[0], CFG.img_shape[1]),\n    color_mode=\"rgb\",\n    batch_size=CFG.batch_size,\n    class_mode= None,\n    shuffle = False,\n    seed=CFG.random_state\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:00.437269Z","iopub.execute_input":"2023-10-06T15:29:00.437503Z","iopub.status.idle":"2023-10-06T15:29:00.47391Z","shell.execute_reply.started":"2023-10-06T15:29:00.437485Z","shell.execute_reply":"2023-10-06T15:29:00.472536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def post_proc(pred):\n    proc_pred = np.empty((pred.shape[0], 2*2 + 3*3), dtype='float32')\n\n    # bowel, extravasation\n    proc_pred[:, 0] = 1 - pred[:, 0] # bowel-healthy\n    proc_pred[:, 1] = pred[:, 0] # bowel-injured\n    proc_pred[:, 2] = 1 - pred[:, 1] # extra-healthy\n    proc_pred[:, 3] = pred[:, 1] # extra-injured\n    \n    # kidney\n    proc_pred[:,4:7] = pred[:,5:8]\n    \n    #liver\n    proc_pred[:,7:10] = pred[:,2:5] \n    \n    # spleen\n    proc_pred[:,10:13] = pred[:,8:]\n\n    return proc_pred","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:00.475062Z","iopub.execute_input":"2023-10-06T15:29:00.475984Z","iopub.status.idle":"2023-10-06T15:29:00.482689Z","shell.execute_reply.started":"2023-10-06T15:29:00.475959Z","shell.execute_reply":"2023-10-06T15:29:00.481487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_probs = OI_model.predict(test_generator, steps = len(test_generator), \n                            verbose = 1)\n\npred = np.concatenate(y_pred_probs, axis = -1).astype('float32')\n \n\npred","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:00.483731Z","iopub.execute_input":"2023-10-06T15:29:00.483961Z","iopub.status.idle":"2023-10-06T15:29:02.097744Z","shell.execute_reply.started":"2023-10-06T15:29:00.483942Z","shell.execute_reply":"2023-10-06T15:29:02.096563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/sample_submission.csv\")\nsample_submission","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:02.099924Z","iopub.execute_input":"2023-10-06T15:29:02.100238Z","iopub.status.idle":"2023-10-06T15:29:02.126652Z","shell.execute_reply.started":"2023-10-06T15:29:02.100211Z","shell.execute_reply":"2023-10-06T15:29:02.124987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_sample = post_proc(pred)\n\nsubmission_df = pd.DataFrame(sample_submission['patient_id'], columns = ['patient_id'])\npred_df = pd.DataFrame(pred_sample, columns = sample_submission.columns[1:])\nsubmission_df = pd.concat([submission_df,pred_df], axis = 1)\n#submission_df.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:02.128749Z","iopub.execute_input":"2023-10-06T15:29:02.129147Z","iopub.status.idle":"2023-10-06T15:29:02.135773Z","shell.execute_reply.started":"2023-10-06T15:29:02.129111Z","shell.execute_reply":"2023-10-06T15:29:02.135097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Group by different sample weights\nscale_by_2 = ['bowel_injury','kidney_low','liver_low','spleen_low']\nscale_by_4 = ['kidney_high','liver_high','spleen_high']\nscale_by_6 = ['extravasation_injury','any_injury']\n\n# Scale factors based on described metric \nsf_2 = 2\nsf_4 = 4\nsf_6 = 6\n\n\nhealthy_cols = ['bowel_healthy', 'extravasation_healthy', 'kidney_healthy',\n               'liver_healthy', 'spleen_healthy']\nany_injury_predictions = (1 - submission_df[healthy_cols]).max(axis=1)\n\nsubmission_df['any_injury']  = any_injury_predictions\n\nfor i in range(len(submission_df)):\n    submission_df[Injuries].iloc[i] = submission_df[Injuries].iloc[i].mean().tolist()\n    \n#submission_df[scale_by_2] *=sf_2\n#submission_df[scale_by_4] *=sf_4\n#submission_df[scale_by_6] *=sf_6\n\n\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:02.136459Z","iopub.execute_input":"2023-10-06T15:29:02.136655Z","iopub.status.idle":"2023-10-06T15:29:02.171296Z","shell.execute_reply.started":"2023-10-06T15:29:02.136638Z","shell.execute_reply":"2023-10-06T15:29:02.170288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:02.172086Z","iopub.execute_input":"2023-10-06T15:29:02.172354Z","iopub.status.idle":"2023-10-06T15:29:02.185731Z","shell.execute_reply.started":"2023-10-06T15:29:02.172332Z","shell.execute_reply":"2023-10-06T15:29:02.184944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-10-06T15:29:02.188127Z","iopub.execute_input":"2023-10-06T15:29:02.188565Z","iopub.status.idle":"2023-10-06T15:29:02.229978Z","shell.execute_reply.started":"2023-10-06T15:29:02.188543Z","shell.execute_reply":"2023-10-06T15:29:02.22923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}