{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pandas.api.types\nimport sklearn.metrics\n\nimport matplotlib.pyplot as plt\n\n\nclass ParticipantVisibleError(Exception):\n    pass\n\n\ndef normalize_probabilities_to_one(df: pd.DataFrame, group_columns: list) -> pd.DataFrame:\n    # Normalize the sum of each row's probabilities to 100%.\n    # 0.75, 0.75 => 0.5, 0.5\n    # 0.1, 0.1 => 0.5, 0.5\n    row_totals = df[group_columns].sum(axis=1)\n    if row_totals.min() == 0:\n        raise ParticipantVisibleError('All rows must contain at least one non-zero prediction')\n    for col in group_columns:\n        df[col] /= row_totals\n    return df\n\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str='patient_id') -> float:\n    '''\n    Pseudocode:\n    1. For every label group (liver, bowel, etc):\n        - Normalize the sum of each row's probabilities to 100%.\n        - Calculate the sample weighted log loss.\n    2. Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    3. Calculate the sample weighted log loss for the new label group\n    4. Return the average of all of the label group log losses as the final score.\n    '''\n    solution = solution.copy() # EDIT HERE TO AVOID SIDE EFFECTS\n    submission = submission.copy() # EDIT HERE TO AVOID SIDE EFFECTS\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n\n    # Run basic QC checks on the inputs\n    if not pandas.api.types.is_numeric_dtype(submission.values):\n        raise ParticipantVisibleError('All submission values must be numeric')\n\n    if not np.isfinite(submission.values).all():\n        raise ParticipantVisibleError('All submission values must be finite')\n\n    if solution.min().min() < 0:\n        raise ParticipantVisibleError('All labels must be at least zero')\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('All predictions must be at least zero')\n\n    # Calculate the label group log losses\n    binary_targets = ['bowel', 'extravasation']\n    triple_level_targets = ['kidney', 'liver', 'spleen']\n    all_target_categories = binary_targets + triple_level_targets\n\n    label_group_losses = []\n    for category in all_target_categories:\n        if category in binary_targets:\n            col_group = [f'{category}_healthy', f'{category}_injury']\n        else:\n            col_group = [f'{category}_healthy', f'{category}_low', f'{category}_high']\n\n        solution = normalize_probabilities_to_one(solution, col_group)\n\n        for col in col_group:\n            if col not in submission.columns:\n                raise ParticipantVisibleError(f'Missing submission column {col}')\n        submission = normalize_probabilities_to_one(submission, col_group)\n        label_group_losses.append(\n            sklearn.metrics.log_loss(\n                y_true=solution[col_group].values,\n                y_pred=submission[col_group].values,\n                sample_weight=solution[f'{category}_weight'].values\n            )\n        )\n\n    # Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    healthy_cols = [x + '_healthy' for x in all_target_categories]\n    any_injury_labels = (1 - solution[healthy_cols]).max(axis=1)\n    any_injury_predictions = (1 - submission[healthy_cols]).max(axis=1)\n    any_injury_loss = sklearn.metrics.log_loss(\n        y_true=any_injury_labels.values,\n        y_pred=any_injury_predictions.values,\n        sample_weight=solution['any_injury_weight'].values\n    )\n\n    label_group_losses.append(any_injury_loss)\n    return np.mean(label_group_losses)","metadata":{"_uuid":"e325914f-f2ce-4c67-9412-0f65de1118eb","_cell_guid":"ab74d9fc-4016-4a15-b3c1-7c03ae474e53","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-08-28T00:32:14.580779Z","iopub.execute_input":"2023-08-28T00:32:14.581113Z","iopub.status.idle":"2023-08-28T00:32:14.595555Z","shell.execute_reply.started":"2023-08-28T00:32:14.581086Z","shell.execute_reply":"2023-08-28T00:32:14.593979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_solution(train): \n    solution = train.copy()\n\n    binary_targets = ['bowel', 'extravasation'] \n    triple_level_targets = ['kidney', 'liver', 'spleen']\n    all_target_categories = binary_targets + triple_level_targets\n\n    for category in all_target_categories:\n        if category in binary_targets:\n            injury_weight = 2 if category == 'bowel' else 6\n            x = train[f'{category}_healthy']\n            solution[f'{category}_weight'] = np.where(x, 1, injury_weight)\n        else:\n            x = train[f'{category}_healthy']\n            y = train[f'{category}_low']\n            solution[f'{category}_weight'] = np.where(x, 1, np.where(y, 2, 4))\n            col_group = [f'{category}_healthy', f'{category}_low', f'{category}_high']\n    solution[f'any_injury_weight'] = np.where(train['any_injury'], 6, 1)\n    return solution","metadata":{"_uuid":"e7c31649-52e3-45f0-91ff-e1c759070968","_cell_guid":"4cbcb1e8-9a47-4875-9a7b-05cfb9780f96","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-08-28T00:23:56.220044Z","iopub.execute_input":"2023-08-28T00:23:56.220398Z","iopub.status.idle":"2023-08-28T00:23:56.228212Z","shell.execute_reply.started":"2023-08-28T00:23:56.22037Z","shell.execute_reply":"2023-08-28T00:23:56.226944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')\nsubmission = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/sample_submission.csv')\nsolution = make_solution(train)","metadata":{"_uuid":"2560dbbb-c407-4b60-a7e0-39519b2a031b","_cell_guid":"db706ead-d388-4ff9-a0be-623c29fb5b57","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-08-28T00:23:58.698187Z","iopub.execute_input":"2023-08-28T00:23:58.698542Z","iopub.status.idle":"2023-08-28T00:23:58.737423Z","shell.execute_reply.started":"2023-08-28T00:23:58.698518Z","shell.execute_reply":"2023-08-28T00:23:58.736117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score(solution, train, row_id_column_name='patient_id')","metadata":{"_uuid":"a4ce32a5-36f0-459f-b0d5-11bc5f0ce882","_cell_guid":"e777c7c6-2273-458c-a974-b4e698f16b42","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-08-28T00:24:00.14127Z","iopub.execute_input":"2023-08-28T00:24:00.141648Z","iopub.status.idle":"2023-08-28T00:24:00.208924Z","shell.execute_reply.started":"2023-08-28T00:24:00.141618Z","shell.execute_reply":"2023-08-28T00:24:00.207673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"category = 'bowel'\nx = train[[f'{category}_healthy', f'{category}_injury']]\nmean_x = x.mean()[1]\n\n# Calculate log loss scores for different constant predictions\nconst_pred = np.linspace(0, 1, 1000)\nscores = [sklearn.metrics.log_loss(x, \n                                   np.ones_like(x) * [1 - pred, pred], \n                                   sample_weight=solution[f'{category}_weight']\n                                  ) for pred in const_pred]\n\n# Find the index of the minimum score\nmin_score_idx = np.argmin(scores)\nmin_score = scores[min_score_idx]\nmin_score_pred = const_pred[min_score_idx]\n\n# Plot the log loss scores\nplt.plot(const_pred, scores)\nplt.axvline(x=mean_x, color='gray', linestyle='dashed', label=f'Mean x = {mean_x:.3f}')\nplt.axvline(x=min_score_pred, color='red', linestyle='dashed', label=f'Min Score = {min_score:.3f}')\nplt.xlabel('Constant Prediction')\nplt.ylabel('Log Loss Score')\nplt.title('Log Loss Scores for Constant Predictions')\nplt.legend()\nplt.show()\n\nprint('Mean x:', mean_x, 1 - mean_x)\nprint('Minimum Log Loss Score:', min_score)\nprint('Corresponding Prediction:', min_score_pred, 1 - min_score_pred)","metadata":{"_uuid":"cb490b77-2c21-468c-98be-1447c4ae4e90","_cell_guid":"f6789ecb-1d64-407c-bcb3-785b3b966dfe","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-08-28T00:24:04.944861Z","iopub.execute_input":"2023-08-28T00:24:04.945267Z","iopub.status.idle":"2023-08-28T00:24:11.216034Z","shell.execute_reply.started":"2023-08-28T00:24:04.945241Z","shell.execute_reply":"2023-08-28T00:24:11.215117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nbest_constant = {}\nfor category in ['bowel', 'extravasation']:\n    print('*' * 10, category, '*' * 10)\n    x = train[[f'{category}_healthy', f'{category}_injury']]\n    mean_x = x.mean()[1]\n\n    # Calculate log loss scores for different constant predictions\n    const_pred = np.linspace(mean_x, mean_x * 7, 1000)\n    scores = [sklearn.metrics.log_loss(x, \n                                       np.ones_like(x) * [1 - pred, pred], \n                                       sample_weight=solution[f'{category}_weight']\n                                      ) for pred in const_pred]\n    # Find the index of the minimum score\n    min_score_idx = np.argmin(scores)\n    min_score = scores[min_score_idx]\n    min_score_pred = const_pred[min_score_idx]\n\n    best_constant[category + '_healthy'] = 1 - min_score_pred\n    best_constant[category + '_injury'] = min_score_pred\n\n    print('Mean x:', mean_x, 1 - mean_x)\n    print('Minimum Log Loss Score:', min_score)\n    print('Corresponding Prediction:', min_score_pred, 1 - min_score_pred)","metadata":{"_uuid":"ae4533a0-1626-4a7c-a813-68c7200d049b","_cell_guid":"fc1ffadc-f8c8-4331-8810-77c3b0ee7718","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-08-28T00:24:20.529207Z","iopub.execute_input":"2023-08-28T00:24:20.529821Z","iopub.status.idle":"2023-08-28T00:24:32.593927Z","shell.execute_reply.started":"2023-08-28T00:24:20.529741Z","shell.execute_reply":"2023-08-28T00:24:32.592662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor category in ['kidney', 'liver', 'spleen']:\n    print('*' * 10, category, '*' * 10)\n    x = train[[f'{category}_healthy', f'{category}_low', f'{category}_high']]\n    mean_x_low = x.mean()[1]\n    mean_x_high = x.mean()[2]\n\n    # Calculate log loss scores for different constant predictions\n    const_pred = np.linspace(mean_x_low, mean_x_low * 3, 25)\n    const_pred_2 = np.linspace(mean_x_high, mean_x_high * 5, 25)\n    preds = []\n    scores = []\n    for pred in const_pred: \n        for pred_2 in const_pred_2:\n            p = [1 - pred - pred_2, pred, pred_2]\n            preds.append(p)\n            scores.append(sklearn.metrics.log_loss(x, \n                                       np.ones_like(x) * p, \n                                       sample_weight=solution[f'{category}_weight']))\n    assert len(preds) == len(scores)\n    # Find the index of the minimum score\n    min_score_idx = np.argmin(scores)\n    min_score = scores[min_score_idx]\n    min_score_pred = preds[min_score_idx]\n\n    best_constant[category + '_healthy'] = min_score_pred[0]\n    best_constant[category + '_low'] = min_score_pred[1]\n    best_constant[category + '_high'] = min_score_pred[2]\n\n    print('mean_x_low:', mean_x_low)\n    print('mean_x_high:', mean_x_high)\n    print('Minimum Log Loss Score:', min_score)\n    print('Corresponding Prediction:', min_score_pred)","metadata":{"_uuid":"a51698ce-d7bc-485d-9a01-b38cd1bf3178","_cell_guid":"72ea53f6-d0bf-4d0e-b58d-39e54cb34ef0","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-08-28T00:24:39.377824Z","iopub.execute_input":"2023-08-28T00:24:39.378219Z","iopub.status.idle":"2023-08-28T00:24:50.983863Z","shell.execute_reply.started":"2023-08-28T00:24:39.37819Z","shell.execute_reply":"2023-08-28T00:24:50.982634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t = train.copy()\ncols = t.columns[1:-1]\nt[cols] = cols.map(best_constant)\nt","metadata":{"execution":{"iopub.status.busy":"2023-08-28T00:28:51.394091Z","iopub.execute_input":"2023-08-28T00:28:51.394402Z","iopub.status.idle":"2023-08-28T00:28:51.419847Z","shell.execute_reply.started":"2023-08-28T00:28:51.394377Z","shell.execute_reply":"2023-08-28T00:28:51.418741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score(solution, t)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T00:32:46.18608Z","iopub.execute_input":"2023-08-28T00:32:46.186559Z","iopub.status.idle":"2023-08-28T00:32:46.242262Z","shell.execute_reply.started":"2023-08-28T00:32:46.186532Z","shell.execute_reply":"2023-08-28T00:32:46.241491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Maybe the any injury score is not being optimized enought","metadata":{}},{"cell_type":"code","source":"category = 'any_injury'\nx = train[category]\nmean_x = x.mean()\n\n# Calculate log loss scores for different constant predictions\nconst_pred = np.linspace(0, 1, 1000)\nscores = [sklearn.metrics.log_loss(x, \n                                   np.ones_like(x) * pred, \n                                   sample_weight=solution[f'{category}_weight']\n                                  ) for pred in const_pred]\n\n# Find the index of the minimum score\nmin_score_idx = np.argmin(scores)\nmin_score = scores[min_score_idx]\nany_injury_optimal = min_score_pred = const_pred[min_score_idx]\n\n# Plot the log loss scores\nplt.plot(const_pred, scores)\nplt.axvline(x=mean_x, color='gray', linestyle='dashed', label=f'Mean x = {mean_x:.3f}')\nplt.axvline(x=min_score_pred, color='red', linestyle='dashed', label=f'Min Score = {min_score:.3f}')\nplt.xlabel('Constant Prediction')\nplt.ylabel('Log Loss Score')\nplt.title('Log Loss Scores for Constant Predictions')\nplt.legend()\nplt.show()\n\nprint('Mean x:', mean_x, 1 - mean_x)\nprint('Minimum Log Loss Score:', min_score)\nprint('Corresponding Prediction:', min_score_pred, 1 - min_score_pred)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T00:46:50.493733Z","iopub.execute_input":"2023-08-28T00:46:50.494122Z","iopub.status.idle":"2023-08-28T00:46:52.249907Z","shell.execute_reply.started":"2023-08-28T00:46:50.494089Z","shell.execute_reply":"2023-08-28T00:46:52.249283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds, scores = [], []\nfor x in np.linspace(t['extravasation_injury'][0], .692, 100): \n    preds.append(x)\n    tt = t.copy()\n    tt['extravasation_injury'] = x\n    tt['extravasation_healthy'] = 1 - x\n    scores.append(score(solution, tt))\n    \nmin_score_idx = np.argmin(scores)\nmin_score = scores[min_score_idx]\nmin_score_pred = preds[min_score_idx]\noptimal_extravasation_injury = min_score_pred\n\n# Plot the log loss scores\nplt.plot(preds, scores)\nplt.axvline(x=mean_x, color='gray', linestyle='dashed', label=f'')\nplt.axvline(x=min_score_pred, color='red', linestyle='dashed', label=f'Min Score = {min_score:.3f}')\nplt.xlabel('Constant Prediction')\nplt.ylabel('Log Loss Score')\nplt.title('Log Loss Scores for Constant Predictions')\nplt.legend()\nplt.show()\n\nprint('Minimum Log Loss Score:', min_score)\nprint('Corresponding Prediction:', min_score_pred)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-28T00:58:40.526198Z","iopub.execute_input":"2023-08-28T00:58:40.526531Z","iopub.status.idle":"2023-08-28T00:58:45.178726Z","shell.execute_reply.started":"2023-08-28T00:58:40.526507Z","shell.execute_reply":"2023-08-28T00:58:45.177539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds, scores = [], []\nmax_delta = any_injury_optimal - t['spleen_low'][0] - t['spleen_high'][0]\nfor x in np.linspace(.01, max_delta, 100): \n    preds.append(x)\n    tt = t.copy()\n    tt['spleen_high'] += x\n    tt['spleen_healthy'] = 1 - tt['spleen_low'] - tt['spleen_high']\n    scores.append(score(solution, tt))\n    \nmin_score_idx = np.argmin(scores)\nmin_score = scores[min_score_idx]\nmin_score_pred = preds[min_score_idx] + t['spleen_high'][0]\n\n# Plot the log loss scores\nplt.plot([x + t['spleen_high'][0] for x in preds], scores)\n# plt.axvline(x=mean_x, color='gray', linestyle='dashed', label=f'')\nplt.axvline(x=min_score_pred, color='red', linestyle='dashed', label=f'Min Score = {min_score:.3f}')\nplt.xlabel('Constant Prediction')\nplt.ylabel('Log Loss Score')\nplt.title('Log Loss Scores for Constant Predictions')\nplt.legend()\nplt.show()\n\nprint('Minimum Log Loss Score:', min_score)\nprint('Corresponding Prediction:', min_score_pred)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T01:03:09.110815Z","iopub.execute_input":"2023-08-28T01:03:09.111317Z","iopub.status.idle":"2023-08-28T01:03:13.62309Z","shell.execute_reply.started":"2023-08-28T01:03:09.11128Z","shell.execute_reply":"2023-08-28T01:03:13.621928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(best_constant)","metadata":{"_uuid":"42a66f35-ffd4-4592-827a-fba5e7698cf0","_cell_guid":"4ef9a15d-f2c8-4efb-960d-33d354f14605","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-08-28T00:42:03.728438Z","iopub.execute_input":"2023-08-28T00:42:03.728746Z","iopub.status.idle":"2023-08-28T00:42:03.734823Z","shell.execute_reply.started":"2023-08-28T00:42:03.728721Z","shell.execute_reply":"2023-08-28T00:42:03.733978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1.779473 / (0.936447 + 1.779473) # checking .66 LB notebook's true extravasation\n# ratio","metadata":{"execution":{"iopub.status.busy":"2023-08-28T01:00:01.799832Z","iopub.execute_input":"2023-08-28T01:00:01.800345Z","iopub.status.idle":"2023-08-28T01:00:01.807026Z","shell.execute_reply.started":"2023-08-28T01:00:01.800305Z","shell.execute_reply":"2023-08-28T01:00:01.805385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[submission.columns[1:]] = submission.columns[1:].map(best_constant)\nsubmission['extravasation_injury'] = optimal_extravasation_injury\nsubmission['extravasation_healthy'] = 1 - optimal_extravasation_injury\ndisplay(submission)\ndisplay(best_constant)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"_uuid":"9b1ab9fd-4eed-41f1-af04-e672afb362f0","_cell_guid":"6336e189-e310-4cbf-8455-362266ad08bf","collapsed":false,"execution":{"iopub.status.busy":"2023-08-27T23:44:25.209054Z","iopub.execute_input":"2023-08-27T23:44:25.209508Z","iopub.status.idle":"2023-08-27T23:44:25.247382Z","shell.execute_reply.started":"2023-08-27T23:44:25.20947Z","shell.execute_reply":"2023-08-27T23:44:25.246082Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}