{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd  \nimport seaborn as sns \nimport matplotlib.pyplot as plt  \n\nimport pandas.api.types\nimport sklearn.metrics  \n\nimport nibabel as nib\n\nimport os\nimport pydicom\nfrom glob import glob\nfrom tqdm import tqdm, trange","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:41:44.158487Z","iopub.execute_input":"2023-09-18T09:41:44.158936Z","iopub.status.idle":"2023-09-18T09:41:45.887655Z","shell.execute_reply.started":"2023-09-18T09:41:44.1589Z","shell.execute_reply":"2023-09-18T09:41:45.886367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = \"/kaggle/input/rsna-2023-abdominal-trauma-detection\"","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:42:31.394484Z","iopub.execute_input":"2023-09-18T09:42:31.39499Z","iopub.status.idle":"2023-09-18T09:42:31.401668Z","shell.execute_reply.started":"2023-09-18T09:42:31.394951Z","shell.execute_reply":"2023-09-18T09:42:31.4003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Target_cols = [\"bowel_healthy\", \"bowel_injury\", \"extravasation_healthy\",\n                   \"extravasation_injury\", \"kidney_healthy\", \"kidney_low\",\n                   \"kidney_high\", \"liver_healthy\", \"liver_low\", \"liver_high\",\n                   \"spleen_healthy\", \"spleen_low\", \"spleen_high\",\"any_injury\"]","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:42:31.404019Z","iopub.execute_input":"2023-09-18T09:42:31.404417Z","iopub.status.idle":"2023-09-18T09:42:31.417686Z","shell.execute_reply.started":"2023-09-18T09:42:31.404383Z","shell.execute_reply":"2023-09-18T09:42:31.416636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = \"/kaggle/input/rsna-2023-abdominal-trauma-detection\"\ntrain_csv=f\"{file_path}/train.csv\" \ntrain=pd.read_csv(train_csv) \ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:42:31.549541Z","iopub.execute_input":"2023-09-18T09:42:31.550019Z","iopub.status.idle":"2023-09-18T09:42:31.602126Z","shell.execute_reply.started":"2023-09-18T09:42:31.549985Z","shell.execute_reply":"2023-09-18T09:42:31.600943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"organs_healthy = [ \n          'bowel_healthy',\n          'extravasation_healthy',\n          'kidney_healthy',\n          'liver_healthy',\n          'spleen_healthy'\n] \n#calculate and plot \ncorr_matrix1=train[organs_healthy].corr()\nsns.heatmap(corr_matrix1,annot=True); \nplt.title('Correlation Heatmap for healthy organs')","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:42:31.604157Z","iopub.execute_input":"2023-09-18T09:42:31.604507Z","iopub.status.idle":"2023-09-18T09:42:32.119428Z","shell.execute_reply.started":"2023-09-18T09:42:31.604477Z","shell.execute_reply":"2023-09-18T09:42:32.118493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"low_high = [\n            'bowel_injury',\n            'extravasation_injury',\n            'kidney_low',\n            'kidney_high',\n            'liver_high',\n            'liver_low' , \n            'spleen_low',\n            'spleen_high',\n            'any_injury'\n] \n#calculate and plot \ncorr_matrix2 = train[low_high].corr()\nsns.heatmap(corr_matrix2,annot=True , linewidths=1) \nplt.title('Correlation heatmap for injury organs')","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:43:23.1141Z","iopub.execute_input":"2023-09-18T09:43:23.114555Z","iopub.status.idle":"2023-09-18T09:43:24.021445Z","shell.execute_reply.started":"2023-09-18T09:43:23.114525Z","shell.execute_reply":"2023-09-18T09:43:24.020244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_series_meta = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_series_meta.csv')\ntrain_series_meta.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:43:24.853353Z","iopub.execute_input":"2023-09-18T09:43:24.854377Z","iopub.status.idle":"2023-09-18T09:43:24.879257Z","shell.execute_reply.started":"2023-09-18T09:43:24.85434Z","shell.execute_reply":"2023-09-18T09:43:24.878016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_series_meta=pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/test_series_meta.csv\") \ntest_series_meta.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:43:28.589095Z","iopub.execute_input":"2023-09-18T09:43:28.589581Z","iopub.status.idle":"2023-09-18T09:43:28.607335Z","shell.execute_reply.started":"2023-09-18T09:43:28.58954Z","shell.execute_reply":"2023-09-18T09:43:28.605951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_labels = pd.read_csv( \"/kaggle/input/rsna-2023-abdominal-trauma-detection/image_level_labels.csv\")\nimage_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:43:30.249318Z","iopub.execute_input":"2023-09-18T09:43:30.250433Z","iopub.status.idle":"2023-09-18T09:43:30.285385Z","shell.execute_reply.started":"2023-09-18T09:43:30.250389Z","shell.execute_reply":"2023-09-18T09:43:30.284113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ParticipantVisibleError(Exception):\n    pass\n","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:43:45.92877Z","iopub.execute_input":"2023-09-18T09:43:45.9293Z","iopub.status.idle":"2023-09-18T09:43:45.936333Z","shell.execute_reply.started":"2023-09-18T09:43:45.929264Z","shell.execute_reply":"2023-09-18T09:43:45.934423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalize_probabilities_to_one(df: pd.DataFrame, group_columns: list) -> pd.DataFrame:\n    # Normalize the sum of each row's probabilities to 100%.\n    # 0.75, 0.75 => 0.5, 0.5\n    # 0.1, 0.1 => 0.5, 0.5\n    row_totals = df[group_columns].sum(axis=1)\n    if row_totals.min() == 0:\n        raise ParticipantVisibleError('All rows must contain at least one non-zero prediction')\n    for col in group_columns:\n        df[col] /= row_totals\n    return df\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:43:54.639601Z","iopub.execute_input":"2023-09-18T09:43:54.640805Z","iopub.status.idle":"2023-09-18T09:43:54.648169Z","shell.execute_reply.started":"2023-09-18T09:43:54.640763Z","shell.execute_reply":"2023-09-18T09:43:54.646704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str) -> float:\n    '''\n    Pseudocode:\n    1. For every label group (liver, bowel, etc):\n        - Normalize the sum of each row's probabilities to 100%.\n        - Calculate the sample weighted log loss.\n    2. Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    3. Calculate the sample weighted log loss for the new label group\n    4. Return the average of all of the label group log losses as the final score.\n    '''\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n    # Run basic QC checks on the inputs\n    if not pandas.api.types.is_numeric_dtype(submission.values):\n        raise ParticipantVisibleError('All submission values must be numeric')\n\n    if not np.isfinite(submission.values).all():\n        raise ParticipantVisibleError('All submission values must be finite')\n\n    if solution.min().min() < 0:\n        raise ParticipantVisibleError('All labels must be at least zero')\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('All predictions must be at least zero')\n\n    # Calculate the label group log losses\n    binary_targets = ['bowel', 'extravasation']\n    triple_level_targets = ['kidney', 'liver', 'spleen']\n    all_target_categories = binary_targets + triple_level_targets\n\n    label_group_losses = []\n    for category in all_target_categories:\n        if category in binary_targets:\n            col_group = [f'{category}_healthy', f'{category}_injury']\n        else:\n            col_group = [f'{category}_healthy', f'{category}_low', f'{category}_high']\n            \n        solution = normalize_probabilities_to_one(solution, col_group)\n\n        for col in col_group:\n            if col not in submission.columns:\n                raise ParticipantVisibleError(f'Missing submission column {col}')\n        submission = normalize_probabilities_to_one(submission, col_group)\n        label_group_losses.append(\n            sklearn.metrics.log_loss(\n                y_true=solution[col_group].values,\n                y_pred=submission[col_group].values,\n                sample_weight=solution[f'{category}_weight'].values\n            )\n        )\n        \n    # Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    healthy_cols = [x + '_healthy' for x in all_target_categories]\n    any_injury_labels = (1 - solution[healthy_cols]).max(axis=1)\n    any_injury_predictions = (1 - submission[healthy_cols]).max(axis=1)\n    any_injury_loss = sklearn.metrics.log_loss(\n        y_true=any_injury_labels.values,\n        y_pred=any_injury_predictions.values,\n        sample_weight=solution['any_injury_weight'].values\n    )\n\n    label_group_losses.append(any_injury_loss)\n    return np.mean(label_group_losses) ","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:45:00.157547Z","iopub.execute_input":"2023-09-18T09:45:00.158282Z","iopub.status.idle":"2023-09-18T09:45:00.174038Z","shell.execute_reply.started":"2023-09-18T09:45:00.158245Z","shell.execute_reply":"2023-09-18T09:45:00.172841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign the appropriate weights to each category\ndef create_training_solution(y_train):\n    sol_train = y_train.copy()\n    \n    # bowel healthy|injury sample weight = 1|2/1\n    sol_train['bowel_weight'] = np.where(sol_train['bowel_injury'] == 1, 2, 1)\n    \n    # extravasation healthy/injury sample weight = 1|6/1\n    sol_train['extravasation_weight'] = np.where(sol_train['extravasation_injury'] == 1, 6, 1)\n    \n    # kidney healthy|low|high sample weight = 1|2|4\n    sol_train['kidney_weight'] = np.where(sol_train['kidney_low'] == 1, 2, np.where(sol_train['kidney_high'] == 1, 4, 1))\n    \n    # liver healthy|low|high sample weight = 1|2|4\n    sol_train['liver_weight'] = np.where(sol_train['liver_low'] == 1, 2, np.where(sol_train['liver_high'] == 1, 4, 1))\n    \n    # spleen healthy|low|high sample weight = 1|2|4\n    sol_train['spleen_weight'] = np.where(sol_train['spleen_low'] == 1, 2, np.where(sol_train['spleen_high'] == 1, 4, 1))\n    \n    # any healthy|injury sample weight = 1|6/1\n    sol_train['any_injury_weight'] = np.where(sol_train['any_injury'] == 1, 6, 1)\n    return sol_train","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:45:21.154582Z","iopub.execute_input":"2023-09-18T09:45:21.155095Z","iopub.status.idle":"2023-09-18T09:45:21.166091Z","shell.execute_reply.started":"2023-09-18T09:45:21.15506Z","shell.execute_reply":"2023-09-18T09:45:21.164801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"solution_train = create_training_solution(train)\n\n# predict a constant using the mean of the training data\ny_pred = train.copy()\ny_pred[Target_cols] = train[Target_cols].mean().tolist()\n\nno_scale_score = score(solution_train,y_pred,'patient_id')\nprint(f'Training score without scaling: {no_scale_score}')","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:45:33.709006Z","iopub.execute_input":"2023-09-18T09:45:33.709532Z","iopub.status.idle":"2023-09-18T09:45:33.799391Z","shell.execute_reply.started":"2023-09-18T09:45:33.709493Z","shell.execute_reply":"2023-09-18T09:45:33.798207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Group by different sample weights\nscale_by_2 = ['kidney_low','liver_low','spleen_low','spleen_high']\nscale_by_4 = ['bowel_injury','kidney_high','liver_high']\nscale_by_6 = ['extravasation_injury','any_injury']\nscale_healthy = ['bowel_healthy', 'extravasation_healthy', 'kidney_healthy', 'liver_healthy', 'spleen_healthy']\n\n# Scale factors based on described metric \nsf_2 = 2.8461531332 * 0.99999999\nsf_4 = 4.841531 * 0.99999999\nsf_6 = 20.81635153 * 0.99999999\nscale_h = 0.99519515313 * 0.99999999\n\n# The score function deletes the ID column so we remake it\nsolution_train = create_training_solution(train)\n\n# Reset the prediction\ny_pred = train.copy()\ny_pred[Target_cols] = train[Target_cols].mean().tolist()\n\n# Scale each target \ny_pred[scale_by_2] *=sf_2\ny_pred[scale_by_4] *=sf_4\ny_pred[scale_by_6] *=sf_6\ny_pred[scale_healthy] *=scale_h\n\nweight_scale_score = score(solution_train,y_pred,'patient_id')\nprint(f'Training score with weight scaling: {weight_scale_score}')","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:45:50.450243Z","iopub.execute_input":"2023-09-18T09:45:50.451082Z","iopub.status.idle":"2023-09-18T09:45:50.55364Z","shell.execute_reply.started":"2023-09-18T09:45:50.451029Z","shell.execute_reply":"2023-09-18T09:45:50.55248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Update scale factors to improve score\n# sf_2 = 4\n# sf_4 = 6\n# sf_6 = 28\n\n# The score function deletes the ID column so we remake it\nsolution_train = create_training_solution(train)\n\n# Reset the prediction, again\ny_pred = train.copy()\ny_pred[Target_cols] = train[Target_cols].mean().tolist()\n\n# Scale each target \ny_pred[scale_by_2] *=sf_2\ny_pred[scale_by_4] *=sf_4\ny_pred[scale_by_6] *=sf_6 \ny_pred[scale_healthy] *=scale_h\n\nimproved_scale_score = score(solution_train,y_pred,'patient_id')\nprint(f'Training score with better scaling: {improved_scale_score}')","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:46:03.830241Z","iopub.execute_input":"2023-09-18T09:46:03.830747Z","iopub.status.idle":"2023-09-18T09:46:03.916036Z","shell.execute_reply.started":"2023-09-18T09:46:03.830712Z","shell.execute_reply":"2023-09-18T09:46:03.914852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/sample_submission.csv')\n\n# Set output to mean of training data\nsubmission[Target_cols] = train[Target_cols].mean().tolist()\n\n# Scale each category by desired scale factor\nsubmission[scale_by_2] *=sf_2\nsubmission[scale_by_4] *=sf_4\nsubmission[scale_by_6] *=sf_6 \nsubmission[scale_healthy] *=scale_h\n\n# Save Submission!\nsubmission.to_csv('submission.csv', index=False) ","metadata":{"execution":{"iopub.status.busy":"2023-09-18T09:46:11.288796Z","iopub.execute_input":"2023-09-18T09:46:11.289267Z","iopub.status.idle":"2023-09-18T09:46:11.322352Z","shell.execute_reply.started":"2023-09-18T09:46:11.289233Z","shell.execute_reply":"2023-09-18T09:46:11.321366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}