{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nimport cv2\nimport glob\nimport pydicom\nfrom joblib import Parallel, delayed","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:40:40.838697Z","iopub.execute_input":"2023-08-29T10:40:40.839068Z","iopub.status.idle":"2023-08-29T10:40:41.247713Z","shell.execute_reply.started":"2023-08-29T10:40:40.839038Z","shell.execute_reply":"2023-08-29T10:40:41.246738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dicom_to_image(dicom_image):\n    \"\"\"\n    Read the dicom file and preprocess appropriately.\n    \"\"\"\n    pixel_array = dicom_image.pixel_array\n    \n    if dicom_image.PixelRepresentation == 1:\n        bit_shift = dicom_image.BitsAllocated - dicom_image.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dicom_image)\n    \n    if dicom_image.PhotometricInterpretation == \"MONOCHROME1\":\n        pixel_array = 1 - pixel_array\n    \n    # transform to hounsfield units\n    intercept = dicom_image.RescaleIntercept\n    slope = dicom_image.RescaleSlope\n    pixel_array = pixel_array * slope + intercept\n    \n    # windowing\n    window_center = int(dicom_image.WindowCenter)\n    window_width = int(dicom_image.WindowWidth)\n    img_min = window_center - window_width // 2\n    img_max = window_center + window_width // 2\n    pixel_array = pixel_array.copy()\n    pixel_array[pixel_array < img_min] = img_min\n    pixel_array[pixel_array > img_max] = img_max\n    \n    # normalization\n    pixel_array = (pixel_array - pixel_array.min())/(pixel_array.max() - pixel_array.min())\n    \n    return (pixel_array * 255).astype(np.uint8)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:40:41.252203Z","iopub.execute_input":"2023-08-29T10:40:41.252568Z","iopub.status.idle":"2023-08-29T10:40:41.265026Z","shell.execute_reply.started":"2023-08-29T10:40:41.25254Z","shell.execute_reply":"2023-08-29T10:40:41.263866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pids = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')['patient_id'].values\ntest_pids = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/sample_submission.csv')['patient_id'].values","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:40:45.526405Z","iopub.execute_input":"2023-08-29T10:40:45.526848Z","iopub.status.idle":"2023-08-29T10:40:45.56747Z","shell.execute_reply.started":"2023-08-29T10:40:45.526816Z","shell.execute_reply":"2023-08-29T10:40:45.566309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\n\nimport torch\nimport torchvision.models as models\nimport torchvision.transforms as transforms\nimport torchvision.datasets as datasets\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.autograd import Variable\nfrom torch.utils.data.dataset import Dataset\nimport torchvision\nimport torchvision.transforms as transforms\n\n# Define the transform\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),  # Resize the image to 224x224 pixels\n    transforms.ToTensor()  # Convert the image to a PyTorch tensor\n])\n\nmodel = torchvision.models.resnet34(False)\nmodel.fc = torch.nn.Linear(512, 14)\nmodel.conv1 = nn.Conv2d(1, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)\n\nmodel.load_state_dict(torch.load(\"/kaggle/input/rsna-training/model_0.pth\", map_location=torch.device('cpu')))\n\nmodel.cuda()","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:40:50.834151Z","iopub.execute_input":"2023-08-29T10:40:50.834563Z","iopub.status.idle":"2023-08-29T10:41:00.541941Z","shell.execute_reply.started":"2023-08-29T10:40:50.834531Z","shell.execute_reply":"2023-08-29T10:41:00.54092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RESIZE = True\nSIZE = 256\nTICK = 5\n\npred_list = []\nfor pid in train_pids[-200:]:    \n    pid_paths_dcm_paths = glob.glob('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/' + str(pid) + '/*/*')\n    pid_paths_dcm_paths.sort()\n    pid_paths_dcm_paths.sort()\n    pid_paths_dcm_paths = np.random.choice(pid_paths_dcm_paths, 15)\n    imgs = [Image.fromarray(dicom_to_image(pydicom.read_file(x))) for x in pid_paths_dcm_paths]\n    imgs = [transform(x) for x in imgs]\n    \n    imgs = torch.cat(imgs, 0)\n    imgs = imgs[:, None, :, :]\n\n    with torch.no_grad():\n        imgs = imgs.cuda()\n        output = model(imgs)[:, :-1]\n        pred = torch.sigmoid(output).data.cpu().numpy().round(3)\n        pred = pred.mean(0)\n        \n    pred_list.append(pred)\n    \npred_df = pd.DataFrame(np.vstack(pred_list))\npred_df.columns = ['bowel_healthy', 'bowel_injury', 'extravasation_healthy',\n       'extravasation_injury', 'kidney_healthy', 'kidney_low', 'kidney_high',\n       'liver_healthy', 'liver_low', 'liver_high', 'spleen_healthy',\n       'spleen_low', 'spleen_high']\n\npred_df['patient_id'] = train_pids[-200:]\npred_df[['patient_id', 'bowel_healthy', 'bowel_injury', 'extravasation_healthy',\n       'extravasation_injury', 'kidney_healthy', 'kidney_low', 'kidney_high',\n       'liver_healthy', 'liver_low', 'liver_high', 'spleen_healthy',\n       'spleen_low', 'spleen_high']].to_csv('submission.csv', index=None)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:56:17.884089Z","iopub.execute_input":"2023-08-29T10:56:17.884459Z","iopub.status.idle":"2023-08-29T10:57:58.675283Z","shell.execute_reply.started":"2023-08-29T10:56:17.884431Z","shell.execute_reply":"2023-08-29T10:57:58.674191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')\n# metrics weight\ntrain_csv['bowel_weight'] = train_csv['bowel_healthy'].map({1:1}).fillna(2)\ntrain_csv['extravasation_weight'] = train_csv['extravasation_healthy'].map({1:1}).fillna(6)\ntrain_csv['extravasation_weight'] = train_csv['extravasation_healthy'].map({1:1}).fillna(6)\n\nkidney_label = train_csv[['kidney_healthy', 'kidney_low','kidney_high']].values.argmax(1)\nkidney_label = pd.Series(kidney_label)\ntrain_csv['kidney_weight'] = kidney_label.map({0: 1, 1:2, 2:4})\n\nliver_label = train_csv[['liver_healthy', 'liver_low','liver_high']].values.argmax(1)\nliver_label = pd.Series(liver_label)\ntrain_csv['liver_weight'] = liver_label.map({0: 1, 1:2, 2:4})\n\nspleen_label = train_csv[['spleen_healthy', 'spleen_low','spleen_high']].values.argmax(1)\nspleen_label = pd.Series(spleen_label)\ntrain_csv['spleen_weight'] = spleen_label.map({0: 1, 1:2, 2:4})\ntrain_csv['any_injury_weight'] = train_csv['any_injury'].map({0:1, 1:6})","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:57:58.677385Z","iopub.execute_input":"2023-08-29T10:57:58.677769Z","iopub.status.idle":"2023-08-29T10:57:58.709918Z","shell.execute_reply.started":"2023-08-29T10:57:58.677728Z","shell.execute_reply":"2023-08-29T10:57:58.708865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pandas.api.types\nimport sklearn.metrics\n\ndef normalize_probabilities_to_one(df: pd.DataFrame, group_columns: list) -> pd.DataFrame:\n    # Normalize the sum of each row's probabilities to 100%.\n    # 0.75, 0.75 => 0.5, 0.5\n    # 0.1, 0.1 => 0.5, 0.5\n    row_totals = df[group_columns].sum(axis=1)\n    if row_totals.min() == 0:\n        raise ParticipantVisibleError('All rows must contain at least one non-zero prediction')\n    for col in group_columns:\n        df[col] /= row_totals\n    return df\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str) -> float:\n    '''\n    Pseudocode:\n    1. For every label group (liver, bowel, etc):\n        - Normalize the sum of each row's probabilities to 100%.\n        - Calculate the sample weighted log loss.\n    2. Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    3. Calculate the sample weighted log loss for the new label group\n    4. Return the average of all of the label group log losses as the final score.\n    '''\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n\n    # Run basic QC checks on the inputs\n    if not pandas.api.types.is_numeric_dtype(submission.values):\n        raise ParticipantVisibleError('All submission values must be numeric')\n\n    if not np.isfinite(submission.values).all():\n        raise ParticipantVisibleError('All submission values must be finite')\n\n    if solution.min().min() < 0:\n        raise ParticipantVisibleError('All labels must be at least zero')\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('All predictions must be at least zero')\n\n    # Calculate the label group log losses\n    binary_targets = ['bowel', 'extravasation']\n    triple_level_targets = ['kidney', 'liver', 'spleen']\n    all_target_categories = binary_targets + triple_level_targets\n\n    label_group_losses = []\n    for category in all_target_categories:\n        if category in binary_targets:\n            col_group = [f'{category}_healthy', f'{category}_injury']\n        else:\n            col_group = [f'{category}_healthy', f'{category}_low', f'{category}_high']\n\n        solution = normalize_probabilities_to_one(solution, col_group)\n\n        for col in col_group:\n            if col not in submission.columns:\n                raise ParticipantVisibleError(f'Missing submission column {col}')\n        submission = normalize_probabilities_to_one(submission, col_group)\n        label_group_losses.append(\n            sklearn.metrics.log_loss(\n                y_true=solution[col_group].values,\n                y_pred=submission[col_group].values,\n                sample_weight=solution[f'{category}_weight'].values\n            )\n        )\n\n    # Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    healthy_cols = [x + '_healthy' for x in all_target_categories]\n    any_injury_labels = (1 - solution[healthy_cols]).max(axis=1)\n    any_injury_predictions = (1 - submission[healthy_cols]).max(axis=1)\n    any_injury_loss = sklearn.metrics.log_loss(\n        y_true=any_injury_labels.values,\n        y_pred=any_injury_predictions.values,\n        sample_weight=solution['any_injury_weight'].values\n    )\n\n    label_group_losses.append(any_injury_loss)\n    return np.mean(label_group_losses)\n\n# Group by different sample weights\nscale_by_2 = ['bowel_injury','kidney_low','liver_low','spleen_low']\nscale_by_4 = ['kidney_high','liver_high','spleen_high']\nscale_by_6 = ['extravasation_injury']\n\nweight_result = [] \nfor sf_2 in range(1,2):\n    for sf_4 in range(1,10):\n        for sf_6 in range(1,10):\n            \n            pred_df2 = pred_df.copy()\n            pred_label = train_csv.iloc[-200:].copy()\n            \n            # Scale each target \n            pred_df2[scale_by_2] *=sf_2\n            pred_df2[scale_by_4] *=sf_4\n            pred_df2[scale_by_6] *=sf_6\n            \n            weight_result.append([sf_2, sf_4, sf_6, score(pred_label, pred_df2, 'patient_id')])","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:57:58.711592Z","iopub.execute_input":"2023-08-29T10:57:58.711994Z","iopub.status.idle":"2023-08-29T10:58:02.409923Z","shell.execute_reply.started":"2023-08-29T10:57:58.711959Z","shell.execute_reply":"2023-08-29T10:58:02.408817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weight_result[0]","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:58:02.412453Z","iopub.execute_input":"2023-08-29T10:58:02.412958Z","iopub.status.idle":"2023-08-29T10:58:02.421031Z","shell.execute_reply.started":"2023-08-29T10:58:02.412921Z","shell.execute_reply":"2023-08-29T10:58:02.419748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(weight_result).sort_values(by=3)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:58:02.423205Z","iopub.execute_input":"2023-08-29T10:58:02.424184Z","iopub.status.idle":"2023-08-29T10:58:02.441838Z","shell.execute_reply.started":"2023-08-29T10:58:02.424144Z","shell.execute_reply":"2023-08-29T10:58:02.440562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RESIZE = True\nSIZE = 256\nTICK = 5\n\npred_list = []\nfor pid in test_pids[:]:    \n    print(pid)\n    pid_paths_dcm_paths = glob.glob('/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/' + str(pid) + '/*/*')\n    pid_paths_dcm_paths.sort()\n    pid_paths_dcm_paths = np.random.choice(pid_paths_dcm_paths, 5)\n#     pid_paths_dcm_paths = pid_paths_dcm_paths[-25:]\n    imgs = [Image.fromarray(dicom_to_image(pydicom.read_file(x))) for x in pid_paths_dcm_paths]\n    imgs = [transform(x) for x in imgs]\n    \n    imgs = torch.cat(imgs, 0)\n    imgs = imgs[:, None, :, :]\n\n    with torch.no_grad():\n        imgs = imgs.cuda()\n        output = model(imgs)[:, :-1]\n        pred = torch.sigmoid(output).data.cpu().numpy().round(3)\n        pred = pred.mean(0)\n        \n    pred_list.append(pred)\n\npred_df = pd.DataFrame(np.vstack(pred_list))\npred_df.columns = ['bowel_healthy', 'bowel_injury', 'extravasation_healthy',\n       'extravasation_injury', 'kidney_healthy', 'kidney_low', 'kidney_high',\n       'liver_healthy', 'liver_low', 'liver_high', 'spleen_healthy',\n       'spleen_low', 'spleen_high']\n\npred_df['patient_id'] = test_pids[:]\n\nscale_by_6 = ['extravasation_injury']\npred_df2[scale_by_6] *= 3\n\npred_df[['patient_id', 'bowel_healthy', 'bowel_injury', 'extravasation_healthy',\n       'extravasation_injury', 'kidney_healthy', 'kidney_low', 'kidney_high',\n       'liver_healthy', 'liver_low', 'liver_high', 'spleen_healthy',\n       'spleen_low', 'spleen_high']].to_csv('submission.csv', index=None)","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:58:22.674847Z","iopub.execute_input":"2023-08-29T10:58:22.675262Z","iopub.status.idle":"2023-08-29T10:58:22.976897Z","shell.execute_reply.started":"2023-08-29T10:58:22.67523Z","shell.execute_reply":"2023-08-29T10:58:22.975338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head /kaggle/working/submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-08-29T10:58:23.984658Z","iopub.execute_input":"2023-08-29T10:58:23.985053Z","iopub.status.idle":"2023-08-29T10:58:25.041411Z","shell.execute_reply.started":"2023-08-29T10:58:23.985021Z","shell.execute_reply":"2023-08-29T10:58:25.039983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}