{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Description\nThis notebook is an illustration of how the any_injury calculation as implemented in the [scoring notebook](https://www.kaggle.com/code/metric/rsna-trauma-metric/notebook)is very detrimental for making good predictions, necessitating counterintuitive post-processing to get good scores. \nThe basis for this notebook is the scale-h-implementation [here](https://www.kaggle.com/code/jasonheesanglee/rsna23-scale-h-implementation).\n\nI discussed the problem in more detail [here]().","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pandas.api.types\nimport sklearn.metrics\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-15T23:55:49.119492Z","iopub.execute_input":"2023-10-15T23:55:49.1199Z","iopub.status.idle":"2023-10-15T23:55:49.746258Z","shell.execute_reply.started":"2023-10-15T23:55:49.119867Z","shell.execute_reply":"2023-10-15T23:55:49.745422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"y_train = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')\ny_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:55:49.747755Z","iopub.execute_input":"2023-10-15T23:55:49.748755Z","iopub.status.idle":"2023-10-15T23:55:49.792335Z","shell.execute_reply.started":"2023-10-15T23:55:49.748713Z","shell.execute_reply":"2023-10-15T23:55:49.791289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Injuries = ['bowel_healthy', 'bowel_injury', \n            'extravasation_healthy', 'extravasation_injury', \n            'kidney_healthy', 'kidney_low', 'kidney_high', \n            'liver_healthy', 'liver_low', 'liver_high', \n            'spleen_healthy', 'spleen_low', 'spleen_high', \n            'any_injury']","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:55:50.415373Z","iopub.execute_input":"2023-10-15T23:55:50.416424Z","iopub.status.idle":"2023-10-15T23:55:50.421004Z","shell.execute_reply.started":"2023-10-15T23:55:50.416385Z","shell.execute_reply":"2023-10-15T23:55:50.420088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Target EDA","metadata":{}},{"cell_type":"code","source":"y_train[Injuries].describe()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:55:51.80565Z","iopub.execute_input":"2023-10-15T23:55:51.805995Z","iopub.status.idle":"2023-10-15T23:55:51.862621Z","shell.execute_reply.started":"2023-10-15T23:55:51.805968Z","shell.execute_reply":"2023-10-15T23:55:51.861571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# [Score](https://www.kaggle.com/code/metric/rsna-trauma-metric/notebook)","metadata":{}},{"cell_type":"markdown","source":"## Metric Implementation ","metadata":{}},{"cell_type":"code","source":"def normalize_probabilities_to_one(df: pd.DataFrame, group_columns: list) -> pd.DataFrame:\n    row_totals = df[group_columns].sum(axis=1)\n    if row_totals.min() == 0:\n        raise ParticipantVisibleError('All rows must contain at least one non-zero prediction')\n    for col in group_columns:\n        df[col] /= row_totals\n    return df\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str) -> float:\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n\n    if not pandas.api.types.is_numeric_dtype(submission.values):\n        raise ParticipantVisibleError('All submission values must be numeric')\n    if not np.isfinite(submission.values).all():\n        raise ParticipantVisibleError('All submission values must be finite')\n    if solution.min().min() < 0:\n        raise ParticipantVisibleError('All labels must be at least zero')\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('All predictions must be at least zero')\n\n    # Calculate the label group log losses\n    binary_targets = ['bowel', 'extravasation']\n    triple_level_targets = ['kidney', 'liver', 'spleen']\n    all_target_categories = binary_targets + triple_level_targets\n\n    label_group_losses = []\n    for category in all_target_categories:\n        if category in binary_targets:\n            col_group = [f'{category}_healthy', f'{category}_injury']\n        else:\n            col_group = [f'{category}_healthy', f'{category}_low', f'{category}_high']\n\n        solution = normalize_probabilities_to_one(solution, col_group)\n        for col in col_group:\n            if col not in submission.columns:\n                raise ParticipantVisibleError(f'Missing submission column {col}')\n        submission = normalize_probabilities_to_one(submission, col_group)\n        label_group_losses.append(\n            sklearn.metrics.log_loss(\n                y_true=solution[col_group].values,\n                y_pred=submission[col_group].values,\n                sample_weight=solution[f'{category}_weight'].values\n            )\n        )\n\n    # Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    healthy_cols = [x + '_healthy' for x in all_target_categories]\n    any_injury_labels = (1 - solution[healthy_cols]).max(axis=1)\n    any_injury_predictions = (1 - submission[healthy_cols]).max(axis=1)\n    any_injury_loss = sklearn.metrics.log_loss(\n        y_true=any_injury_labels.values,\n        y_pred=any_injury_predictions.values,\n        sample_weight=solution['any_injury_weight'].values\n    )\n\n    label_group_losses.append(any_injury_loss)\n    return label_group_losses","metadata":{"execution":{"iopub.status.busy":"2023-10-15T23:55:56.344216Z","iopub.execute_input":"2023-10-15T23:55:56.344685Z","iopub.status.idle":"2023-10-15T23:55:56.356791Z","shell.execute_reply.started":"2023-10-15T23:55:56.34465Z","shell.execute_reply":"2023-10-15T23:55:56.355435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign the appropriate weights to each category\ndef create_training_solution(y_train):\n    sol_train = y_train.copy()\n    sol_train['bowel_weight'] = np.where(sol_train['bowel_injury'] == 1, 2, 1)\n    sol_train['extravasation_weight'] = np.where(sol_train['extravasation_injury'] == 1, 6, 1)\n    sol_train['kidney_weight'] = np.where(sol_train['kidney_low'] == 1, 2, np.where(sol_train['kidney_high'] == 1, 4, 1))\n    sol_train['liver_weight'] = np.where(sol_train['liver_low'] == 1, 2, np.where(sol_train['liver_high'] == 1, 4, 1))\n    sol_train['spleen_weight'] = np.where(sol_train['spleen_low'] == 1, 2, np.where(sol_train['spleen_high'] == 1, 4, 1))\n    sol_train['any_injury_weight'] = np.where(sol_train['any_injury'] == 1, 6, 1)\n    return sol_train","metadata":{"execution":{"iopub.status.busy":"2023-10-15T21:28:35.615223Z","iopub.execute_input":"2023-10-15T21:28:35.615608Z","iopub.status.idle":"2023-10-15T21:28:35.622691Z","shell.execute_reply.started":"2023-10-15T21:28:35.615579Z","shell.execute_reply":"2023-10-15T21:28:35.621593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Optimal multiplier for the specific injury category equals it's weight","metadata":{}},{"cell_type":"code","source":"# Initialize lists to store the loss values and mult values\nbowel_loss_values = []\nextravasation_loss_values = []\nmult_values = []\n\n# predict a constant using the mean of the training data\nfor i in range(10,70):\n    mult = i * 0.1\n    solution_train = create_training_solution(y_train)\n    y_pred = y_train.copy()\n    y_pred[Injuries] = y_train[Injuries].mean().tolist()\n    y_pred['bowel_injury'] *= mult\n    y_pred['extravasation_injury'] *= mult\n    no_scale_score = score(solution_train, y_pred, 'patient_id')\n    bowel_loss = no_scale_score[0]\n    extravasation_loss = no_scale_score[1]  # Assuming the second item in the score tuple is extravasation_loss\n\n    # Append the current loss and mult value to the lists\n    bowel_loss_values.append(bowel_loss)\n    extravasation_loss_values.append(extravasation_loss)\n    mult_values.append(mult)\n\n# Plotting\nplt.figure(figsize=(12, 6))\n\nplt.subplot(1, 2, 1)\nplt.plot(mult_values[:30], bowel_loss_values[:30])\nplt.xlabel('Mult')\nplt.ylabel('Bowel Loss')\nplt.title('Bowel Loss over Mult')\n\nplt.subplot(1, 2, 2)\nplt.plot(mult_values[40:], extravasation_loss_values[40:])\nplt.xlabel('Mult')\nplt.ylabel('Extravasation Loss')\nplt.title('Extravasation Loss over Mult')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T21:34:57.398042Z","iopub.execute_input":"2023-10-15T21:34:57.398388Z","iopub.status.idle":"2023-10-15T21:35:01.316534Z","shell.execute_reply.started":"2023-10-15T21:34:57.398361Z","shell.execute_reply":"2023-10-15T21:35:01.315402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Optimal multiplier for the specific injury category is much higher than it's weight if you factor in any_injury loss","metadata":{}},{"cell_type":"markdown","source":"#### Weighted extravasation is the lowest healthy prediction determining any_injury","metadata":{}},{"cell_type":"code","source":"# Initialize lists to store the loss values and mult values\nextravasation_loss_values = []\nany_injury_loss_values = []\nsum_loss_values = []\nmult_values = []\n\n# predict a constant using the mean of the training data\nfor i in range(50,200):\n    mult = i * 0.1\n    solution_train = create_training_solution(y_train)\n    y_pred = y_train.copy()\n    y_pred[Injuries] = y_train[Injuries].mean().tolist()\n    y_pred['extravasation_injury'] *= mult\n    no_scale_score = score(solution_train, y_pred, 'patient_id')\n    extravasation_loss = no_scale_score[1]  # Assuming the second item in the score tuple is extravasation_loss\n    any_injury_loss = no_scale_score[-1]  # Assuming the last item in the score tuple is any_injury_loss\n\n    # Append the current loss and mult value to the lists\n    extravasation_loss_values.append(extravasation_loss)\n    any_injury_loss_values.append(any_injury_loss)\n    sum_loss_values.append(extravasation_loss + any_injury_loss)\n    mult_values.append(mult)\n\n# Plotting\nplt.figure(figsize=(18, 6))\n\nplt.subplot(1, 3, 1)\nplt.plot(mult_values, extravasation_loss_values)\nplt.xlabel('Mult')\nplt.ylabel('Extravasation Loss')\nplt.title('Extravasation Loss over Mult')\n\nplt.subplot(1, 3, 2)\nplt.plot(mult_values, any_injury_loss_values)\nplt.xlabel('Mult')\nplt.ylabel('Any Injury Loss')\nplt.title('Any Injury Loss over Mult')\n\nplt.subplot(1, 3, 3)\nplt.plot(mult_values, sum_loss_values)\nplt.xlabel('Mult')\nplt.ylabel('Sum of Losses')\nplt.title('Sum of Extravasation and Any Injury Losses over Mult')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T21:39:20.833215Z","iopub.execute_input":"2023-10-15T21:39:20.833625Z","iopub.status.idle":"2023-10-15T21:39:29.887576Z","shell.execute_reply.started":"2023-10-15T21:39:20.833589Z","shell.execute_reply":"2023-10-15T21:39:29.88624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### You can get lower total loss with any injury category though! You just need to surpass weighted extravasation first!","metadata":{}},{"cell_type":"code","source":"# Initialize lists to store the loss values and mult values\nbowel_loss_values = []\nany_injury_loss_values = []\nsum_loss_values = []\nmult_values = []\n\n# predict a constant using the mean of the training data\nfor i in range(10,200):\n    mult = i * 0.1\n    solution_train = create_training_solution(y_train)\n    y_pred = y_train.copy()\n    y_pred[Injuries] = y_train[Injuries].mean().tolist()\n    y_pred['bowel_injury'] *= mult\n    no_scale_score = score(solution_train, y_pred, 'patient_id')\n    bowel_loss = no_scale_score[0]  # Assuming the first item in the score tuple is bowel_loss\n    any_injury_loss = no_scale_score[-1]  # Assuming the last item in the score tuple is any_injury_loss\n\n    # Append the current loss and mult value to the lists\n    bowel_loss_values.append(bowel_loss)\n    any_injury_loss_values.append(any_injury_loss)\n    sum_loss_values.append(bowel_loss + any_injury_loss)\n    mult_values.append(mult)\n\n# Plotting\nplt.figure(figsize=(18, 6))\n\nplt.subplot(1, 3, 1)\nplt.plot(mult_values, bowel_loss_values)\nplt.xlabel('Mult')\nplt.ylabel('Bowel Loss')\nplt.title('Bowel Loss over Mult')\n\nplt.subplot(1, 3, 2)\nplt.plot(mult_values, any_injury_loss_values)\nplt.xlabel('Mult')\nplt.ylabel('Any Injury Loss')\nplt.title('Any Injury Loss over Mult')\n\nplt.subplot(1, 3, 3)\nplt.plot(mult_values, sum_loss_values)\nplt.xlabel('Mult')\nplt.ylabel('Sum of Losses')\nplt.title('Sum of Bowel and Any Injury Losses over Mult')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-15T21:41:53.611572Z","iopub.execute_input":"2023-10-15T21:41:53.611959Z","iopub.status.idle":"2023-10-15T21:42:05.272088Z","shell.execute_reply.started":"2023-10-15T21:41:53.611927Z","shell.execute_reply":"2023-10-15T21:42:05.271031Z"},"trusted":true},"execution_count":null,"outputs":[]}]}