{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Forked from [RSNA-ATD: 512x512 PNG v2 Data](https://www.kaggle.com/awsaf49/rsna-atd-512x512-png-v2-data)\n\nThe only difference is that the images are first cropped to be square, and then are then  resized\n\nAlso the pngs are stored on 16 bits","metadata":{}},{"cell_type":"code","source":"!pip install -q python-gdcm\n!pip install -q pylibjpeg\n# !pip install -q dicomsdl","metadata":{"execution":{"iopub.status.busy":"2023-09-10T12:50:17.2367Z","iopub.execute_input":"2023-09-10T12:50:17.237156Z","iopub.status.idle":"2023-09-10T12:50:45.070076Z","shell.execute_reply.started":"2023-09-10T12:50:17.23712Z","shell.execute_reply":"2023-09-10T12:50:45.068356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os,shutil, pandas as pd\nfrom glob import glob\nfrom tqdm import tqdm\nimport zipfile\nimport cv2\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-09-10T12:50:45.072817Z","iopub.execute_input":"2023-09-10T12:50:45.073187Z","iopub.status.idle":"2023-09-10T12:50:45.07856Z","shell.execute_reply.started":"2023-09-10T12:50:45.073144Z","shell.execute_reply":"2023-09-10T12:50:45.07761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_PATH = '/kaggle/input/rsna-2023-abdominal-trauma-detection'\n# IMG_DIR = '/kaggle/working'\nIMG_DIR = '/tmp/Dataset/rsna-atd'","metadata":{"execution":{"iopub.status.busy":"2023-09-10T12:50:45.07986Z","iopub.execute_input":"2023-09-10T12:50:45.080235Z","iopub.status.idle":"2023-09-10T12:50:45.09921Z","shell.execute_reply.started":"2023-09-10T12:50:45.080205Z","shell.execute_reply":"2023-09-10T12:50:45.098043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resize_dim = 512\nimg_size = [resize_dim, resize_dim]\nIDX = 0\nPARTS = 1\nprint(f\"Image Size: {img_size}\")\nprint(f\"Dim: {np.prod(img_size)**0.5: 0.2f}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-10T12:50:45.101017Z","iopub.execute_input":"2023-09-10T12:50:45.101475Z","iopub.status.idle":"2023-09-10T12:50:45.115978Z","shell.execute_reply.started":"2023-09-10T12:50:45.10143Z","shell.execute_reply":"2023-09-10T12:50:45.115003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Meta Data","metadata":{}},{"cell_type":"code","source":"# train_df = pd.read_csv(f'{ROOT_PATH}/train.csv')\n# img_lvl_df = pd.read_csv(f'{ROOT_PATH}/image_level_labels.csv')\n# train_df = train_df.merge(img_lvl_df, on=['patient_id'], how='right')\n\n# train_df['image_path'] = (f'{ROOT_PATH}/train_images' + '/'\n#                           + train_df.patient_id.astype(str)\n#                           + '/' + train_df.series_id.astype(str)\n#                           + '/' + train_df.instance_number.astype(str) + '.dcm')\n\n# print('Train:')\n# print(f'# Size: {len(train_df)}')\n# display(train_df.head(2))","metadata":{"execution":{"iopub.status.busy":"2023-09-10T12:50:45.119426Z","iopub.execute_input":"2023-09-10T12:50:45.119812Z","iopub.status.idle":"2023-09-10T12:50:45.135396Z","shell.execute_reply.started":"2023-09-10T12:50:45.119781Z","shell.execute_reply":"2023-09-10T12:50:45.134401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_paths = glob('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/*/*/*dcm')\n\ntrain_df = pd.DataFrame(train_paths, columns=[\"image_path\"])\ntrain_df['patient_id'] = train_df.image_path.map(lambda x: x.split('/')[-3]).astype(int)\ntrain_df['series_id'] = train_df.image_path.map(lambda x: x.split('/')[-2]).astype(int)\ntrain_df['instance_number'] = train_df.image_path.map(lambda x: x.split('/')[-1].replace('.dcm','')).astype(int)\nprint('Train:')\nprint(f'# Size: {len(train_df)}')\ndisplay(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2023-09-10T12:50:45.13662Z","iopub.execute_input":"2023-09-10T12:50:45.137416Z","iopub.status.idle":"2023-09-10T12:56:53.283351Z","shell.execute_reply.started":"2023-09-10T12:50:45.137382Z","shell.execute_reply":"2023-09-10T12:56:53.281915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_paths = glob('/kaggle/input/rsna-2023-abdominal-trauma-detection/test_images/*/*/*dcm')\n\ntest_df = pd.DataFrame(test_paths, columns=[\"image_path\"])\ntest_df['patient_id'] = test_df.image_path.map(lambda x: x.split('/')[-3]).astype(int)\ntest_df['series_id'] = test_df.image_path.map(lambda x: x.split('/')[-2]).astype(int)\ntest_df['instance_number'] = test_df.image_path.map(lambda x: x.split('/')[-1].replace('.dcm','')).astype(int)\nprint('Test:')\nprint(f'# Size: {len(test_df)}')\ndisplay(test_df.head())","metadata":{"execution":{"iopub.status.busy":"2023-09-10T12:56:53.285437Z","iopub.execute_input":"2023-09-10T12:56:53.285886Z","iopub.status.idle":"2023-09-10T12:56:53.359728Z","shell.execute_reply.started":"2023-09-10T12:56:53.285844Z","shell.execute_reply":"2023-09-10T12:56:53.358949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# `.dcm` to `.png`","metadata":{}},{"cell_type":"code","source":"!rm -r {IMG_DIR}\nos.makedirs(f'{IMG_DIR}/train_images', exist_ok = True)\nos.makedirs(f'{IMG_DIR}/test_images', exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2023-09-10T12:56:53.361142Z","iopub.execute_input":"2023-09-10T12:56:53.361471Z","iopub.status.idle":"2023-09-10T12:56:54.469731Z","shell.execute_reply.started":"2023-09-10T12:56:53.361442Z","shell.execute_reply":"2023-09-10T12:56:54.468336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nimport math\nimport cv2\n    \ndef standardize_pixel_array(dcm: pydicom.dataset.FileDataset) -> np.ndarray:\n    # Correct DICOM pixel_array if PixelRepresentation == 1.\n    pixel_array = dcm.pixel_array\n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n    return pixel_array\n\ndef read_xray(path, fix_monochrome = True):\n    dicom = pydicom.dcmread(path)\n    data = standardize_pixel_array(dicom)\n    data = data - np.min(data)\n    data = data / (np.max(data) + 1e-5)\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = 1.0 - data\n    return data\n\ndef resize_and_save(file_path):\n    img = read_xray(file_path)\n    h, w = img.shape[:2]  # orig hw\n    # crop\n    min_dim = min(h, w)\n    h_tail = (h - min_dim) // 2\n    w_tail = (w - min_dim) // 2\n    img = img[h_tail: h-h_tail, w_tail: w-w_tail]\n    \n    img = cv2.resize(img, (resize_dim, resize_dim), cv2.INTER_LINEAR)\n    img = (img * 255).astype(np.uint16)\n    \n    sub_path = file_path.split(\"/\",4)[-1].split('.dcm')[0] + '.png'\n    infos = sub_path.split('/')\n    pid = infos[-3]\n    sid = infos[-2]\n    iid = infos[-1]; iid = iid.replace('.png','')\n    new_path = os.path.join(IMG_DIR, sub_path)\n    os.makedirs(new_path.rsplit('/',1)[0], exist_ok=True)\n    cv2.imwrite(new_path, img)\n    return pid,sid,iid,w,h","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-10T12:56:54.472278Z","iopub.execute_input":"2023-09-10T12:56:54.473141Z","iopub.status.idle":"2023-09-10T12:56:54.798818Z","shell.execute_reply.started":"2023-09-10T12:56:54.473061Z","shell.execute_reply":"2023-09-10T12:56:54.79769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"%%time\nfrom joblib import Parallel, delayed\nSIZE = -(-len(train_df) // PARTS)\nfile_paths = train_df.image_path.tolist()[IDX*SIZE:(IDX+1)*SIZE]\nimgsize_train = Parallel(n_jobs=-1,backend='threading')(delayed(resize_and_save)(file_path)\\\n                                                  for file_path in tqdm(file_paths, leave=True, position=0))","metadata":{"execution":{"iopub.status.busy":"2023-09-10T12:56:54.801283Z","iopub.execute_input":"2023-09-10T12:56:54.801785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pid, sid, iid, width, height = list(zip(*imgsize_train))\nmeta_df = pd.DataFrame({'patient_id':pid,\n                        'series_id':sid,\n                       'instance_number':iid,\n                       'width':width,\n                       'height':height})\nmeta_df[['patient_id','series_id','instance_number']] = meta_df[['patient_id','series_id','instance_number']].astype(int)\ntrain_df = train_df.merge(meta_df, on=['patient_id','series_id','instance_number'], how='right')\ntrain_df.head(2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test","metadata":{}},{"cell_type":"code","source":"%%time\nfrom joblib import Parallel, delayed\nfile_paths = test_df.image_path.tolist()\nimgsize_test = Parallel(n_jobs=-1,backend='threading')(delayed(resize_and_save)(file_path)\\\n                                                  for file_path in tqdm(file_paths))","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pid, sid, iid, width, height = list(zip(*imgsize_test))\nmeta_df = pd.DataFrame({'patient_id':pid,\n                        'series_id':sid,\n                       'instance_number':iid,\n                       'width':width,\n                       'height':height})\nmeta_df[['patient_id','series_id','instance_number']] = meta_df[['patient_id','series_id','instance_number']].astype(int)\ntest_df = test_df.merge(meta_df, on=['patient_id','series_id','instance_number'], how='right')\ntest_df.head(2)","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Check","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize=(8,8))\nplt.scatter(train_df['width'], train_df['height'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check Image","metadata":{}},{"cell_type":"code","source":"path = train_df[train_df.width<700].image_path.iloc[0].replace(ROOT_PATH, IMG_DIR).replace('.dcm','.png')\nimg = cv2.imread(path, cv2.IMREAD_UNCHANGED)\nplt.figure(figsize=(10,10))\nplt.imshow(img, cmap='gray')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = train_df[train_df.width>700].image_path.iloc[0].replace(ROOT_PATH, IMG_DIR).replace('.dcm','.png')\nimg = cv2.imread(path, cv2.IMREAD_UNCHANGED)\nplt.figure(figsize=(10,10))\nplt.imshow(img, cmap='gray')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CSV","metadata":{}},{"cell_type":"code","source":"train_df.to_csv(f'{IMG_DIR}/train.csv',index = False)\ntest_df.to_csv(f'{IMG_DIR}/test.csv',index = False)\n\nshutil.copy(f'{ROOT_PATH}/train_series_meta.csv',f'{IMG_DIR}/')\nshutil.copy(f'{ROOT_PATH}/test_series_meta.csv',f'{IMG_DIR}/')\nshutil.copy(f'{ROOT_PATH}/sample_submission.csv',f'{IMG_DIR}/')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -al {IMG_DIR}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from zipfile import ZipFile\n\nzipObj = ZipFile(f'/kaggle/working/rsna-acd.zip', 'w')\n\nfile_paths = glob(f'{IMG_DIR}/**/*',recursive = True)\nprint(f'Total Files:{len(file_paths)}')\nprint('Zippping...')\nfor file_path in tqdm(file_paths):\n    zipObj.write(file_path, file_path[len(IMG_DIR):])\n    os.remove(file_path) if os.path.isfile(file_path) else None\nzipObj.close()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -al {IMG_DIR}","metadata":{"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}