{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q pydicom pillow imageio","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:09:51.093827Z","iopub.execute_input":"2023-08-19T09:09:51.094221Z","iopub.status.idle":"2023-08-19T09:10:04.768275Z","shell.execute_reply.started":"2023-08-19T09:09:51.094189Z","shell.execute_reply":"2023-08-19T09:10:04.766833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport imageio\nimport pydicom\nfrom PIL import Image\n\nimport matplotlib.pyplot as plt\nfrom glob import glob\nfrom tqdm import tqdm\nimport cv2\nimport math\nimport os\nimport shutil\n\nfrom joblib import Parallel,delayed\nfrom zipfile import ZipFile","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:04.770326Z","iopub.execute_input":"2023-08-19T09:10:04.771193Z","iopub.status.idle":"2023-08-19T09:10:05.294985Z","shell.execute_reply.started":"2023-08-19T09:10:04.771159Z","shell.execute_reply":"2023-08-19T09:10:05.293942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sample Readin","metadata":{}},{"cell_type":"code","source":"dicom_path= \"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/10004/21057/1001.dcm\"\nvol_arr= pydicom.dcmread(dicom_path).pixel_array\nvol_arr.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:05.297581Z","iopub.execute_input":"2023-08-19T09:10:05.297949Z","iopub.status.idle":"2023-08-19T09:10:05.337471Z","shell.execute_reply.started":"2023-08-19T09:10:05.297917Z","shell.execute_reply":"2023-08-19T09:10:05.33644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(vol_arr,cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:05.339947Z","iopub.execute_input":"2023-08-19T09:10:05.340275Z","iopub.status.idle":"2023-08-19T09:10:05.724672Z","shell.execute_reply.started":"2023-08-19T09:10:05.340246Z","shell.execute_reply":"2023-08-19T09:10:05.72383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Readin","metadata":{}},{"cell_type":"code","source":"ROOT_PATH= \"/kaggle/input/rsna-2023-abdominal-trauma-detection\"\nOUT_PATH=\"/kaggle/working/RSNA-ATD/\"","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:05.726037Z","iopub.execute_input":"2023-08-19T09:10:05.726576Z","iopub.status.idle":"2023-08-19T09:10:05.73127Z","shell.execute_reply.started":"2023-08-19T09:10:05.72653Z","shell.execute_reply":"2023-08-19T09:10:05.730498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_dim=[512,512]\nresize_dim=512\n\ntrain_df= pd.read_csv(f\"{ROOT_PATH}/train.csv\")\nimage_labels_df= pd.read_csv(f\"{ROOT_PATH}/image_level_labels.csv\")\n\ntrain_df.head()\n    ","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:05.732485Z","iopub.execute_input":"2023-08-19T09:10:05.732813Z","iopub.status.idle":"2023-08-19T09:10:05.791837Z","shell.execute_reply.started":"2023-08-19T09:10:05.732786Z","shell.execute_reply":"2023-08-19T09:10:05.790679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_labels_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:05.793601Z","iopub.execute_input":"2023-08-19T09:10:05.793983Z","iopub.status.idle":"2023-08-19T09:10:05.804843Z","shell.execute_reply.started":"2023-08-19T09:10:05.793953Z","shell.execute_reply":"2023-08-19T09:10:05.803725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we can merge both to get a singular dataframe","metadata":{}},{"cell_type":"code","source":"train_df= train_df.merge(image_labels_df,how=\"right\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:05.806049Z","iopub.execute_input":"2023-08-19T09:10:05.806831Z","iopub.status.idle":"2023-08-19T09:10:05.842157Z","shell.execute_reply.started":"2023-08-19T09:10:05.806798Z","shell.execute_reply":"2023-08-19T09:10:05.841019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since for each line in the dataframe the image is present at \n\n{train_images or test_images}/{patient_id}/{series_id}/{instance_number}.dcm\n\nLet us create a column in the dataframe that creates the path","metadata":{}},{"cell_type":"code","source":"train_df[\"image_path\"]= f'{ROOT_PATH}/train_images'+'/'+train_df.patient_id.astype(str)+'/'+train_df.series_id.astype(str)+'/'+train_df.instance_number.astype(str)+'.dcm'\ntrain_df.head()\n                ","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:05.843379Z","iopub.execute_input":"2023-08-19T09:10:05.84397Z","iopub.status.idle":"2023-08-19T09:10:05.902393Z","shell.execute_reply.started":"2023-08-19T09:10:05.843939Z","shell.execute_reply":"2023-08-19T09:10:05.90134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:05.906441Z","iopub.execute_input":"2023-08-19T09:10:05.906808Z","iopub.status.idle":"2023-08-19T09:10:05.913147Z","shell.execute_reply.started":"2023-08-19T09:10:05.906778Z","shell.execute_reply":"2023-08-19T09:10:05.912028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## For Test data\n\nWe need to create a test data dataframe but we do have access to the files","metadata":{}},{"cell_type":"code","source":"test_paths= glob(f'{ROOT_PATH}/test_images/*/*/*.dcm')\ntest_df= pd.DataFrame(test_paths,columns=[\"image_path\"])\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:05.914738Z","iopub.execute_input":"2023-08-19T09:10:05.91505Z","iopub.status.idle":"2023-08-19T09:10:05.950581Z","shell.execute_reply.started":"2023-08-19T09:10:05.915022Z","shell.execute_reply":"2023-08-19T09:10:05.949675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.image_path[1]","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:05.951857Z","iopub.execute_input":"2023-08-19T09:10:05.952154Z","iopub.status.idle":"2023-08-19T09:10:05.958595Z","shell.execute_reply.started":"2023-08-19T09:10:05.952128Z","shell.execute_reply":"2023-08-19T09:10:05.957597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[\"patient_id\"]=test_df.image_path.map(lambda x: x.split('/')[-3])\ntest_df[\"series_id\"]=test_df.image_path.map(lambda x: x.split('/')[-2])\ntest_df[\"instance\"]= test_df.image_path.map(lambda x:x.split('/')[-1].replace('.dcm','')).astype(int)\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:05.959806Z","iopub.execute_input":"2023-08-19T09:10:05.960081Z","iopub.status.idle":"2023-08-19T09:10:05.977551Z","shell.execute_reply.started":"2023-08-19T09:10:05.960056Z","shell.execute_reply":"2023-08-19T09:10:05.976472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Time to convert .DCM to .PNG","metadata":{}},{"cell_type":"code","source":"!rm -r {OUT_PATH}\nos.makedirs(f'{OUT_PATH}/train_images', exist_ok = True)\nos.makedirs(f'{OUT_PATH}/test_images', exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:05.979036Z","iopub.execute_input":"2023-08-19T09:10:05.979545Z","iopub.status.idle":"2023-08-19T09:10:07.068737Z","shell.execute_reply.started":"2023-08-19T09:10:05.979506Z","shell.execute_reply":"2023-08-19T09:10:07.067614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def standardize_pixel_array(dcm: pydicom.dataset.FileDataset) -> np.ndarray:\n    pixel_array= dcm.pixel_array\n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n    return pixel_array\n\ndef read_xray(path, fix_monochrome = True):\n    dicom = pydicom.dcmread(path)\n    data = standardize_pixel_array(dicom)\n    data = data - np.min(data)\n    data = data / (np.max(data) + 1e-6)\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = 1.0 - data\n    return data\n\ndef resize_and_save(file_path):\n    img = read_xray(file_path)\n    h, w = img.shape[:2]  # original height and width\n    img = cv2.resize(img, (resize_dim, resize_dim), cv2.INTER_LINEAR)\n    img = (img * 255).astype(np.uint8)\n    \n    sub_path = file_path.split(\"/\",4)[-1].split('.dcm')[0] + '.png'\n    infos = sub_path.split('/')\n    pid = infos[-3]\n    sid = infos[-2]\n    iid = infos[-1]; iid = iid.replace('.png','')\n    new_path = os.path.join(OUT_PATH, sub_path)\n    os.makedirs(new_path.rsplit('/',1)[0], exist_ok=True)\n    cv2.imwrite(new_path, img,\n#                 [cv2.IMWRITE_PNG_COMPRESSION, 1],\n               )\n    return pid,sid,iid,w,h","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:07.070677Z","iopub.execute_input":"2023-08-19T09:10:07.071892Z","iopub.status.idle":"2023-08-19T09:10:07.085545Z","shell.execute_reply.started":"2023-08-19T09:10:07.071845Z","shell.execute_reply":"2023-08-19T09:10:07.084462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"With threading we can convert the image data folders from dcm to png data","metadata":{}},{"cell_type":"code","source":"%%time\nfrom joblib import Parallel, delayed\nIDX = 0\nPARTS = 1\nSIZE = -(-len(train_df) // PARTS)\nfile_paths = train_df.image_path.tolist()[IDX*SIZE:(IDX+1)*SIZE]\nimgsize_train = Parallel(n_jobs=-1,backend='threading')(delayed(resize_and_save)(file_path)\\\n                                                  for file_path in tqdm(file_paths, leave=True, position=0))\npid, sid, iid, width, height = list(zip(*imgsize_train))\nmeta_df = pd.DataFrame({'patient_id':pid,\n                        'series_id':sid,\n                       'instance_number':iid,\n                       'width':width,\n                       'height':height})\nmeta_df[['patient_id','series_id','instance_number']] = meta_df[['patient_id','series_id','instance_number']].astype(int)\ntrain_df = train_df.merge(meta_df, on=['patient_id','series_id','instance_number'], how='right')\ntrain_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:10:07.0868Z","iopub.execute_input":"2023-08-19T09:10:07.087191Z","iopub.status.idle":"2023-08-19T09:14:25.512434Z","shell.execute_reply.started":"2023-08-19T09:10:07.087163Z","shell.execute_reply":"2023-08-19T09:14:25.511694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom joblib import Parallel, delayed\nfile_paths = test_df.image_path.tolist()\nimgsize_test = Parallel(n_jobs=-1,backend='threading')(delayed(resize_and_save)(file_path)\\\n                                                  for file_path in tqdm(file_paths))\n\npid, sid, iid, width, height = list(zip(*imgsize_test))\nmeta_df = pd.DataFrame({'patient_id':pid,\n                        'series_id':sid,\n                       'instance_number':iid,\n                       'width':width,\n                       'height':height})\nmeta_df[['patient_id','series_id','instance_number']] = meta_df[['patient_id','series_id','instance_number']].astype(int)\n#test_df = test_df.merge(meta_df, on=['patient_id','series_id','instance_number'], how='right')\n#test_df.head(2)\nmeta_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:32:54.907519Z","iopub.execute_input":"2023-08-19T09:32:54.907975Z","iopub.status.idle":"2023-08-19T09:32:55.00759Z","shell.execute_reply.started":"2023-08-19T09:32:54.907943Z","shell.execute_reply":"2023-08-19T09:32:55.0068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_csv(f'{OUT_PATH}/train.csv',index = False)\ntest_df.to_csv(f'{OUT_PATH}/test.csv',index = False)\n\nshutil.copy(f'{ROOT_PATH}/train_series_meta.csv',f'{OUT_PATH}/')\nshutil.copy(f'{ROOT_PATH}/test_series_meta.csv',f'{OUT_PATH}/')\nshutil.copy(f'{ROOT_PATH}/sample_submission.csv',f'{OUT_PATH}/')","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:33:03.134207Z","iopub.execute_input":"2023-08-19T09:33:03.134651Z","iopub.status.idle":"2023-08-19T09:33:03.333234Z","shell.execute_reply.started":"2023-08-19T09:33:03.134603Z","shell.execute_reply":"2023-08-19T09:33:03.332038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"zipObj = ZipFile(f'/kaggle/working/rsna-acd.zip', 'w')\n\nfile_paths = glob(f'{OUT_PATH}/**/*',recursive = True)\nprint(f'Total Files:{len(file_paths)}')\nprint('Zippping...')\nfor file_path in tqdm(file_paths):\n    zipObj.write(file_path, file_path[len(OUT_PATH):])\n    os.remove(file_path) if os.path.isfile(file_path) else None\nzipObj.close()","metadata":{"execution":{"iopub.status.busy":"2023-08-19T09:33:36.356496Z","iopub.execute_input":"2023-08-19T09:33:36.35692Z","iopub.status.idle":"2023-08-19T09:33:41.325768Z","shell.execute_reply.started":"2023-08-19T09:33:36.356889Z","shell.execute_reply":"2023-08-19T09:33:41.324647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}