{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":52254,"databundleVersionId":6863140},{"sourceType":"modelInstanceVersion","sourceId":6091,"databundleVersionId":7429345,"modelInstanceId":4623}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1><b>Imports</b></h1>\n<i>All libraries that will be used later in the code below</i>","metadata":{}},{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"tensorflow\"\nimport keras_core as keras\nimport keras_cv\n\nimport gc\nimport cv2\nimport pydicom\nfrom joblib import Parallel, delayed\n\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nfrom glob import glob","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:27.219393Z","iopub.execute_input":"2024-02-12T08:15:27.219905Z","iopub.status.idle":"2024-02-12T08:15:27.227522Z","shell.execute_reply.started":"2024-02-12T08:15:27.219849Z","shell.execute_reply":"2024-02-12T08:15:27.226057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3><b>Constants that will be used frequently</b></h3>","metadata":{}},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/rsna-2023-abdominal-trauma-detection\"\nIMAGE_DIR = \"/tmp/dataset/rsna-atd\"\nINPUT_MODEL_PATH = \"/kaggle/input/kerascv-starter-notebook-train/rsna-atd.keras\"\nMODEL_PATH = \"/kaggle/working/rsna-atd.keras\"\nSTRIDE = 10","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:27.231978Z","iopub.execute_input":"2024-02-12T08:15:27.233202Z","iopub.status.idle":"2024-02-12T08:15:27.245315Z","shell.execute_reply.started":"2024-02-12T08:15:27.233148Z","shell.execute_reply":"2024-02-12T08:15:27.243921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    IMAGE_SIZE = [256, 256]\n    RESIZE_DIM = 256\n    BATCH_SIZE = 64\n    AUTOTUNE = tf.data.AUTOTUNE\n    TARGET_COLS  = [\"bowel_healthy\", \"bowel_injury\", \"extravasation_healthy\",\n                   \"extravasation_injury\", \"kidney_healthy\", \"kidney_low\",\n                   \"kidney_high\", \"liver_healthy\", \"liver_low\", \"liver_high\",\n                   \"spleen_healthy\", \"spleen_low\", \"spleen_high\"]\n\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:27.295981Z","iopub.execute_input":"2024-02-12T08:15:27.297487Z","iopub.status.idle":"2024-02-12T08:15:27.30533Z","shell.execute_reply.started":"2024-02-12T08:15:27.297437Z","shell.execute_reply":"2024-02-12T08:15:27.303925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1>Load The ResNet 50 Model</h1>","metadata":{}},{"cell_type":"code","source":"model = keras_cv.models.ResNetBackbone.from_preset(\"resnet50_imagenet\")","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:27.311917Z","iopub.execute_input":"2024-02-12T08:15:27.312737Z","iopub.status.idle":"2024-02-12T08:15:32.735692Z","shell.execute_reply.started":"2024-02-12T08:15:27.312693Z","shell.execute_reply":"2024-02-12T08:15:32.734374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:32.738239Z","iopub.execute_input":"2024-02-12T08:15:32.738611Z","iopub.status.idle":"2024-02-12T08:15:33.342266Z","shell.execute_reply.started":"2024-02-12T08:15:32.738579Z","shell.execute_reply":"2024-02-12T08:15:33.341348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking if patients are repeated by finding the number of unique patient IDs","metadata":{}},{"cell_type":"code","source":"meta_df = pd.read_csv(f\"{BASE_PATH}/test_series_meta.csv\")\n\nnum_rows = meta_df.shape[0]\nunique_patients = meta_df[\"patient_id\"].nunique()\n\nprint(f\"{num_rows=}\")\nprint(f\"{unique_patients=}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:33.343393Z","iopub.execute_input":"2024-02-12T08:15:33.343737Z","iopub.status.idle":"2024-02-12T08:15:33.354711Z","shell.execute_reply.started":"2024-02-12T08:15:33.343706Z","shell.execute_reply":"2024-02-12T08:15:33.353025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df[\"dicom_folder\"] = BASE_PATH + \"/\" + \"test_images\"\\\n                                    + \"/\" + meta_df.patient_id.astype(str)\\\n                                    + \"/\" + meta_df.series_id.astype(str)\n\ntest_folders = meta_df.dicom_folder.tolist()\ntest_paths = []\nfor folder in tqdm(test_folders):\n    test_paths += sorted(glob(os.path.join(folder, \"*dcm\")))[::STRIDE]","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:33.356704Z","iopub.execute_input":"2024-02-12T08:15:33.357108Z","iopub.status.idle":"2024-02-12T08:15:33.3986Z","shell.execute_reply.started":"2024-02-12T08:15:33.357076Z","shell.execute_reply":"2024-02-12T08:15:33.397213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.DataFrame(test_paths, columns=[\"dicom_path\"])\ntest_df[\"patient_id\"] = test_df.dicom_path.map(lambda x: x.split(\"/\")[-3]).astype(int)\ntest_df[\"series_id\"] = test_df.dicom_path.map(lambda x: x.split(\"/\")[-2]).astype(int)\ntest_df[\"instance_number\"] = test_df.dicom_path.map(lambda x: x.split(\"/\")[-1].replace(\".dcm\",\"\")).astype(int)\n\ntest_df[\"image_path\"] = f\"{IMAGE_DIR}/test_images\"\\\n                    + \"/\" + test_df.patient_id.astype(str)\\\n                    + \"/\" + test_df.series_id.astype(str)\\\n                    + \"/\" + test_df.instance_number.astype(str) +\".png\"\n\ntest_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:33.402223Z","iopub.execute_input":"2024-02-12T08:15:33.402665Z","iopub.status.idle":"2024-02-12T08:15:33.428271Z","shell.execute_reply.started":"2024-02-12T08:15:33.402629Z","shell.execute_reply":"2024-02-12T08:15:33.427283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking if patients are repeated by finding the number of unique patient IDs\nnum_rows = test_df.shape[0]\nunique_patients = test_df[\"patient_id\"].nunique()\n\nprint(f\"{num_rows=}\")\nprint(f\"{unique_patients=}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:33.430315Z","iopub.execute_input":"2024-02-12T08:15:33.431592Z","iopub.status.idle":"2024-02-12T08:15:33.43942Z","shell.execute_reply.started":"2024-02-12T08:15:33.431536Z","shell.execute_reply":"2024-02-12T08:15:33.437932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs(f\"{IMAGE_DIR}/train_images\", exist_ok=True)\nos.makedirs(f\"{IMAGE_DIR}/test_images\", exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:33.440825Z","iopub.execute_input":"2024-02-12T08:15:33.441238Z","iopub.status.idle":"2024-02-12T08:15:33.450778Z","shell.execute_reply.started":"2024-02-12T08:15:33.441205Z","shell.execute_reply":"2024-02-12T08:15:33.449858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def standardize_pixel_array(dcm):\n    # Correct DICOM pixel_array if PixelRepresentation == 1.\n    pixel_array = dcm.pixel_array\n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n    return pixel_array\n\ndef read_xray(path, fix_monochrome=True):\n    dicom = pydicom.dcmread(path)\n    data = standardize_pixel_array(dicom)\n    data = data - np.min(data)\n    data = data / (np.max(data) + 1e-5)\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = 1.0 - data\n    return data\n\ndef resize_and_save(file_path):\n    img = read_xray(file_path)\n    h, w = img.shape[:2]  # orig hw\n    img = cv2.resize(img, (config.RESIZE_DIM, config.RESIZE_DIM), cv2.INTER_LINEAR)\n    img = (img * 255).astype(np.uint8)\n    \n    sub_path = file_path.split(\"/\",4)[-1].split(\".dcm\")[0] + \".png\"\n    infos = sub_path.split(\"/\")\n    sub_path = file_path.split(\"/\",4)[-1].split(\".dcm\")[0] + \".png\"\n    infos = sub_path.split(\"/\")\n    pid = infos[-3]\n    sid = infos[-2]\n    iid = infos[-1]; iid = iid.replace(\".png\",\"\")\n    new_path = os.path.join(IMAGE_DIR, sub_path)\n    os.makedirs(new_path.rsplit(\"/\",1)[0], exist_ok=True)\n    cv2.imwrite(new_path, img)\n    return","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:33.451868Z","iopub.execute_input":"2024-02-12T08:15:33.452201Z","iopub.status.idle":"2024-02-12T08:15:33.467658Z","shell.execute_reply.started":"2024-02-12T08:15:33.452172Z","shell.execute_reply":"2024-02-12T08:15:33.466603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfile_paths = test_df.dicom_path.tolist()\n_ = Parallel(n_jobs=2, backend=\"threading\")(\n    delayed(resize_and_save)(file_path) for file_path in tqdm(file_paths, leave=True, position=0)\n)\n\ndel _; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:33.468879Z","iopub.execute_input":"2024-02-12T08:15:33.469475Z","iopub.status.idle":"2024-02-12T08:15:34.455792Z","shell.execute_reply.started":"2024-02-12T08:15:33.469443Z","shell.execute_reply":"2024-02-12T08:15:34.454466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(image_path):\n    file_bytes = tf.io.read_file(image_path)\n    image = tf.io.decode_png(file_bytes, channels=3, dtype=tf.uint8)\n    image = tf.image.resize(image, config.IMAGE_SIZE, method=\"bilinear\")\n    image = tf.cast(image, tf.float32) / 255.0\n    return image\n\ndef build_dataset(image_paths):\n    ds = (\n        tf.data.Dataset.from_tensor_slices(image_paths)\n        .map(decode_image, num_parallel_calls=config.AUTOTUNE)\n        .shuffle(config.BATCH_SIZE * 10)\n        .batch(config.BATCH_SIZE)\n        .prefetch(config.AUTOTUNE)\n    )\n    return ds","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:34.457325Z","iopub.execute_input":"2024-02-12T08:15:34.457675Z","iopub.status.idle":"2024-02-12T08:15:34.468261Z","shell.execute_reply.started":"2024-02-12T08:15:34.457645Z","shell.execute_reply":"2024-02-12T08:15:34.46692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths  = test_df.image_path.tolist()\n\nds = build_dataset(paths)\nimages = next(iter(ds))\n\nimages.shape","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:34.470311Z","iopub.execute_input":"2024-02-12T08:15:34.471606Z","iopub.status.idle":"2024-02-12T08:15:34.538368Z","shell.execute_reply.started":"2024-02-12T08:15:34.471554Z","shell.execute_reply":"2024-02-12T08:15:34.537049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras_cv.visualization.plot_image_gallery(\n    images=images,\n    value_range=(0, 1),\n    rows=1,\n    cols=3,\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:34.540584Z","iopub.execute_input":"2024-02-12T08:15:34.541093Z","iopub.status.idle":"2024-02-12T08:15:34.868606Z","shell.execute_reply.started":"2024-02-12T08:15:34.541047Z","shell.execute_reply":"2024-02-12T08:15:34.867278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def post_proc(pred):\n    proc_pred = np.empty((pred.shape[0], 2*2 + 3*3), dtype=\"float32\")\n\n    # bowel, extravasation\n    proc_pred[:, 0] = pred[:, 0]\n    proc_pred[:, 1] = 1 - proc_pred[:, 0]\n    proc_pred[:, 2] = pred[:, 1]\n    proc_pred[:, 3] = 1 - proc_pred[:, 2]\n    \n    # liver, kidney, sneel\n    proc_pred[:, 4:7] = pred[:, 2:5]\n    proc_pred[:, 7:10] = pred[:, 5:8]\n    proc_pred[:, 10:13] = pred[:, 8:11]\n\n    return proc_pred","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:15:34.87048Z","iopub.execute_input":"2024-02-12T08:15:34.870882Z","iopub.status.idle":"2024-02-12T08:15:34.880124Z","shell.execute_reply.started":"2024-02-12T08:15:34.870829Z","shell.execute_reply":"2024-02-12T08:15:34.878697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting unique patient IDs from test dataset\npatient_ids = test_df[\"patient_id\"].unique()\n\n# Initializing array to store predictions\npatient_preds = np.zeros(\n    shape=(len(patient_ids), 2*2 + 3*3),\n    dtype=\"float32\"\n)\n\n# Iterating over each patient\nfor pidx, patient_id in tqdm(enumerate(patient_ids), total=len(patient_ids), desc=\"Patients \"):\n    print(f\"Patient ID: {patient_id}\")\n    \n    # Query the dataframe for a particular patient\n    patient_df = test_df.query(\"patient_id == @patient_id\")\n    \n    # Getting image paths for a patient\n    patient_paths = patient_df.image_path.tolist()\n\n    # Building dataset for prediction\n    dtest = build_dataset(patient_paths)\n    \n    # Predicting with the model\n    pred = model.predict(dtest)\n    pred = np.concatenate(pred, axis=-1).astype(\"float32\")\n    pred = pred[:len(patient_paths), :]\n    pred = np.mean(pred, axis=0, keepdims=True)\n\n    # Correcting reshape operation\n    pred = pred.reshape(len(patient_paths), -1)\n\n    # Update the array of patient predictions\n    patient_preds[pidx, :] += post_proc(pred)[0]\n    \n\n    # Deleting variables to free up memory \n    del patient_df, patient_paths, dtest, pred; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:16:24.449126Z","iopub.execute_input":"2024-02-12T08:16:24.449717Z","iopub.status.idle":"2024-02-12T08:16:27.60966Z","shell.execute_reply.started":"2024-02-12T08:16:24.449669Z","shell.execute_reply":"2024-02-12T08:16:27.608357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Submission\npred_df = pd.DataFrame({\"patient_id\":patient_ids,})\npred_df[config.TARGET_COLS] = patient_preds.astype(\"float32\")\n\n# Align with sample submission\npre_df = pd.read_csv(f\"{BASE_PATH}/sample_submission.csv\")\npre_df = pre_df[[\"patient_id\"]]\npre_df = pre_df.merge(pred_df, on=\"patient_id\", how=\"left\")\n\n# Store submission\npre_df.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:17:12.042143Z","iopub.execute_input":"2024-02-12T08:17:12.04275Z","iopub.status.idle":"2024-02-12T08:17:12.06938Z","shell.execute_reply.started":"2024-02-12T08:17:12.042697Z","shell.execute_reply":"2024-02-12T08:17:12.067977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **In the first row**\nthe patient with `patient_id` 48843 is predicted to have bowel injury with a probability of 1.0, extravasation injury with a probability of 1.0, low kidney function with a probability of approximately 0.07, and high spleen function with a probability of approximately 0.44.","metadata":{}},{"cell_type":"code","source":"pre_df","metadata":{"execution":{"iopub.status.busy":"2024-02-12T08:17:13.786848Z","iopub.execute_input":"2024-02-12T08:17:13.787325Z","iopub.status.idle":"2024-02-12T08:17:13.815318Z","shell.execute_reply.started":"2024-02-12T08:17:13.787292Z","shell.execute_reply":"2024-02-12T08:17:13.813625Z"},"trusted":true},"execution_count":null,"outputs":[]}]}