{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\n\nfrom torchvision.transforms import transforms as T\nfrom pydicom import dcmread\nfrom pathlib import Path\n\nimport pandas.api.types\nimport sklearn.metrics\nimport pandas as pd\nimport numpy as np\nimport torch","metadata":{"execution":{"iopub.status.busy":"2023-08-25T18:45:27.799795Z","iopub.execute_input":"2023-08-25T18:45:27.800123Z","iopub.status.idle":"2023-08-25T18:45:36.355359Z","shell.execute_reply.started":"2023-08-25T18:45:27.80007Z","shell.execute_reply":"2023-08-25T18:45:36.352477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_training_solution(y_train):\n    sol_train = y_train.copy()\n    \n    # bowel healthy|injury sample weight = 1|2\n    sol_train['bowel_weight'] = np.where(sol_train['bowel_injury'] == 1, 2, 1)\n    \n    # extravasation healthy/injury sample weight = 1|6\n    sol_train['extravasation_weight'] = np.where(sol_train['extravasation_injury'] == 1, 6, 1)\n    \n    # kidney healthy|low|high sample weight = 1|2|4\n    sol_train['kidney_weight'] = np.where(sol_train['kidney_low'] == 1, 2, np.where(sol_train['kidney_high'] == 1, 4, 1))\n    \n    # liver healthy|low|high sample weight = 1|2|4\n    sol_train['liver_weight'] = np.where(sol_train['liver_low'] == 1, 2, np.where(sol_train['liver_high'] == 1, 4, 1))\n    \n    # spleen healthy|low|high sample weight = 1|2|4\n    sol_train['spleen_weight'] = np.where(sol_train['spleen_low'] == 1, 2, np.where(sol_train['spleen_high'] == 1, 4, 1))\n    \n    # any healthy|injury sample weight = 1|6\n    sol_train['any_injury_weight'] = np.where(sol_train['any_injury'] == 1, 6, 1)\n    \n    return sol_train\n\nclass ParticipantVisibleError(Exception):\n    pass\n\ndef normalize_probabilities_to_one(df: pd.DataFrame, group_columns: list) -> pd.DataFrame:\n    row_totals = df[group_columns].sum(axis=1)\n    if row_totals.min() == 0:\n        raise ParticipantVisibleError('All rows must contain at least one non-zero prediction')\n    for col in group_columns:\n        df[col] /= row_totals\n    return df\n\ndef custom_log_loss(y_true, y_pred, sample_weight):\n    epsilon = 1e-15  # Small constant to avoid division by zero\n\n    # Clip the predicted probabilities to avoid log(0)\n    y_pred = np.clip(y_pred, epsilon, 1 - epsilon)\n\n    # Calculate the log loss for each sample and each output\n    log_loss = - (sample_weight * (y_true * np.log(y_pred) + (1 - y_true) * np.log(1 - y_pred))).sum()\n\n    # Normalize by the total weight of the samples\n    normalized_log_loss = log_loss / sample_weight.sum()\n\n    return normalized_log_loss\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str) -> float:\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n\n    if not pandas.api.types.is_numeric_dtype(submission.values):\n        raise ParticipantVisibleError('All submission values must be numeric')\n\n    if not np.isfinite(submission.values).all():\n        raise ParticipantVisibleError('All submission values must be finite')\n\n    if solution.min().min() < 0:\n        raise ParticipantVisibleError('All labels must be at least zero')\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('All predictions must be at least zero')\n\n    binary_targets = ['bowel', 'extravasation']\n    triple_level_targets = ['kidney', 'liver', 'spleen']\n    all_target_categories = binary_targets + triple_level_targets\n\n    label_group_losses = []\n    for category in all_target_categories:\n        if category in binary_targets:\n            col_group = [f'{category}_healthy', f'{category}_injury']\n        else:\n            col_group = [f'{category}_healthy', f'{category}_low', f'{category}_high']\n\n        solution = normalize_probabilities_to_one(solution, col_group)\n\n        for col in col_group:\n            if col not in submission.columns:\n                raise ParticipantVisibleError(f'Missing submission column {col}')\n        submission = normalize_probabilities_to_one(submission, col_group)\n\n        # Call your custom log loss function here for multioutput labels\n        label_group_loss = custom_log_loss(\n            y_true=solution[col_group].values,\n            y_pred=submission[col_group].values,\n            sample_weight=solution[f'{category}_weight'].values.reshape((-1, 1))\n        )\n\n        label_group_losses.append(label_group_loss)\n\n    # Calculate and append any_injury loss using the same custom log loss function\n    healthy_cols = [x + '_healthy' for x in all_target_categories]\n    any_injury_labels = (1 - solution[healthy_cols]).max(axis=1)\n    any_injury_predictions = (1 - submission[healthy_cols]).max(axis=1)\n    any_injury_loss = custom_log_loss(\n        y_true=any_injury_labels.values,\n        y_pred=any_injury_predictions.values,\n        sample_weight=solution['any_injury_weight'].values\n    )\n    label_group_losses.append(any_injury_loss)\n\n    return np.mean(label_group_losses)","metadata":{"execution":{"iopub.status.busy":"2023-08-25T18:45:36.356785Z","iopub.status.idle":"2023-08-25T18:45:36.35797Z","shell.execute_reply.started":"2023-08-25T18:45:36.357724Z","shell.execute_reply":"2023-08-25T18:45:36.357748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"custom_transforms = T.Compose(transforms=[\n    T.ToTensor(), \n    T.Resize(size=(512,512), antialias=True)\n])","metadata":{"execution":{"iopub.status.busy":"2023-08-25T18:45:36.35922Z","iopub.status.idle":"2023-08-25T18:45:36.360258Z","shell.execute_reply.started":"2023-08-25T18:45:36.36Z","shell.execute_reply":"2023-08-25T18:45:36.360022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_test_set() -> list[str]:\n    images = Path('/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images').rglob(pattern='*.dcm')\n   \n#     images = Path('/kaggle/input/rsna-7-patients-test/test_images').rglob(pattern='*.dcm')\n    \n    return images\n\ndef load_dicom(file_path: str | Path) -> np.ndarray:\n    with dcmread(fp=file_path) as dcm_file:\n        return dcm_file.pixel_array\n\ndef transform_data(file_path: str | Path) -> torch.Tensor:\n    img = load_dicom(file_path=file_path).astype(dtype=np.float32)\n    img_tensor = custom_transforms(img)\n    if len(img_tensor.size()) == 3:\n        img_tensor = img_tensor.unsqueeze(dim=0)\n    return img_tensor","metadata":{"execution":{"iopub.status.busy":"2023-08-25T18:45:36.361437Z","iopub.status.idle":"2023-08-25T18:45:36.362225Z","shell.execute_reply.started":"2023-08-25T18:45:36.361976Z","shell.execute_reply":"2023-08-25T18:45:36.361997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEVICE = 'cuda' if torch.cuda.is_available() else 'cpu'\nmodel = torch.jit.load(f='/kaggle/input/rsna-model/model.pt').to(device=DEVICE)\nmodel.eval()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T18:45:36.363612Z","iopub.status.idle":"2023-08-25T18:45:36.364633Z","shell.execute_reply.started":"2023-08-25T18:45:36.364397Z","shell.execute_reply":"2023-08-25T18:45:36.364427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.read_csv(filepath_or_buffer='/kaggle/input/rsna-2023-abdominal-trauma-detection/sample_submission.csv')\n# submission_df = pd.read_csv(filepath_or_buffer='/kaggle/input/rsna-7-patients-test/sample_submission.csv')\n\ntarget_columns = submission_df.select_dtypes(include='float64').columns","metadata":{"execution":{"iopub.status.busy":"2023-08-25T18:45:36.36579Z","iopub.status.idle":"2023-08-25T18:45:36.366375Z","shell.execute_reply.started":"2023-08-25T18:45:36.366145Z","shell.execute_reply":"2023-08-25T18:45:36.366167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df['any_injury'] = 0","metadata":{"execution":{"iopub.status.busy":"2023-08-25T18:45:36.36901Z","iopub.status.idle":"2023-08-25T18:45:36.370106Z","shell.execute_reply.started":"2023-08-25T18:45:36.369836Z","shell.execute_reply":"2023-08-25T18:45:36.369858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.columns","metadata":{"execution":{"iopub.status.busy":"2023-08-25T18:45:36.371272Z","iopub.status.idle":"2023-08-25T18:45:36.372072Z","shell.execute_reply.started":"2023-08-25T18:45:36.371833Z","shell.execute_reply":"2023-08-25T18:45:36.371854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = {\n    'patient_id': [], \n    'any_injury': []\n}\nfor col in target_columns:\n    submission[col] = []\n\nmy_submission = pd.DataFrame(data=submission_df)\n    \nwith torch.no_grad():\n    for test_file in load_test_set():\n        patient_id = int(test_file.parent.parent.name)\n        x = transform_data(file_path=test_file).to(device=DEVICE)\n        yhat = model(x)\n        prob = torch.sigmoid(input=yhat).detach().cpu().numpy().squeeze()\n        \n        submission['patient_id'].append(patient_id)\n        submission['any_injury'].append(prob[-1].item())\n        for col, value in zip(target_columns, prob):\n            submission[col].append(value.item())\n        \n        del x, yhat, prob\n        torch.cuda.empty_cache()\n        \n\nprint(my_submission.columns)\n\nmy_submission = create_training_solution(y_train=my_submission)\nid_col = my_submission['patient_id'].copy()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T18:45:36.374574Z","iopub.status.idle":"2023-08-25T18:45:36.375317Z","shell.execute_reply.started":"2023-08-25T18:45:36.375067Z","shell.execute_reply":"2023-08-25T18:45:36.375107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(my_submission.head())\ndisplay(submission_df.head())","metadata":{"execution":{"iopub.status.busy":"2023-08-25T18:45:36.376688Z","iopub.status.idle":"2023-08-25T18:45:36.377441Z","shell.execute_reply.started":"2023-08-25T18:45:36.377194Z","shell.execute_reply":"2023-08-25T18:45:36.377215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_score = score(solution=my_submission, submission=submission_df, row_id_column_name='patient_id')\nprint(my_score)","metadata":{"execution":{"iopub.status.busy":"2023-08-25T18:45:36.378885Z","iopub.status.idle":"2023-08-25T18:45:36.379725Z","shell.execute_reply.started":"2023-08-25T18:45:36.379464Z","shell.execute_reply":"2023-08-25T18:45:36.379487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_sub = my_submission.drop(labels=['any_injury', 'bowel_weight', 'extravasation_weight', 'kidney_weight', 'liver_weight', 'spleen_weight', 'any_injury_weight'], axis=1)\nfinal_sub['patient_id'] = id_col\nfinal_sub = final_sub[['patient_id', 'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury', 'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high', 'spleen_healthy', 'spleen_low', 'spleen_high']]\nfinal_sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-25T18:45:36.381095Z","iopub.status.idle":"2023-08-25T18:45:36.381863Z","shell.execute_reply.started":"2023-08-25T18:45:36.381625Z","shell.execute_reply":"2023-08-25T18:45:36.381647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-25T18:45:36.383235Z","iopub.status.idle":"2023-08-25T18:45:36.383983Z","shell.execute_reply.started":"2023-08-25T18:45:36.383745Z","shell.execute_reply":"2023-08-25T18:45:36.383767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}