{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":24800,"databundleVersionId":1831594,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pydicom","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:58:36.319423Z","iopub.execute_input":"2024-03-12T15:58:36.319872Z","iopub.status.idle":"2024-03-12T15:59:10.396617Z","shell.execute_reply.started":"2024-03-12T15:58:36.31983Z","shell.execute_reply":"2024-03-12T15:59:10.395544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:59:10.399094Z","iopub.execute_input":"2024-03-12T15:59:10.399547Z","iopub.status.idle":"2024-03-12T15:59:10.903406Z","shell.execute_reply.started":"2024-03-12T15:59:10.399503Z","shell.execute_reply":"2024-03-12T15:59:10.902011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##converting dicon images to numpy array\nimport numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef read_images(path, voi_lut=True, fix_monochrome=True):\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT is used to transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:59:10.90505Z","iopub.execute_input":"2024-03-12T15:59:10.906061Z","iopub.status.idle":"2024-03-12T15:59:11.096942Z","shell.execute_reply.started":"2024-03-12T15:59:10.906026Z","shell.execute_reply":"2024-03-12T15:59:11.096063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##resizing the image\ndef resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:59:11.099282Z","iopub.execute_input":"2024-03-12T15:59:11.099721Z","iopub.status.idle":"2024-03-12T15:59:11.105844Z","shell.execute_reply.started":"2024-03-12T15:59:11.099685Z","shell.execute_reply":"2024-03-12T15:59:11.104787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id_train = []\ndim0train = []\ndim1train = []\nimage_id_test = []\ndim0test = []\ndim1test = []\n\nfor split in ['train', 'test']:\n    load_dir = f'../input/vinbigdata-chest-xray-abnormalities-detection/{split}/'\n    save_dir = f'/kaggle/working/{split}/'\n\n    os.makedirs(save_dir, exist_ok=True)\n\n    for file in tqdm(os.listdir(load_dir)):\n        # set keep_ratio=True to have original aspect ratio\n        xray = read_images(load_dir + file)\n        im = resize(xray, size=512)  \n        im.save(save_dir + file.replace('dicom', 'png'))\n        \n        if split == 'train':\n            image_id_train.append(file.replace('.dicom', ''))\n            dim0train.append(xray.shape[0])\n            dim1train.append(xray.shape[1])\n            \n        if split == 'test':\n            image_id_test.append(file.replace('.dicom', ''))\n            dim0test.append(xray.shape[0])\n            dim1test.append(xray.shape[1])","metadata":{"execution":{"iopub.status.busy":"2024-03-12T15:59:11.107016Z","iopub.execute_input":"2024-03-12T15:59:11.107297Z","iopub.status.idle":"2024-03-12T23:02:54.912585Z","shell.execute_reply.started":"2024-03-12T15:59:11.107273Z","shell.execute_reply":"2024-03-12T23:02:54.90739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_meta = pd.DataFrame.from_dict({'image_id': image_id_train, 'dim0': dim0train, 'dim1': dim1train})\ndf_train_meta.to_csv('train_meta.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T00:26:25.643973Z","iopub.execute_input":"2024-03-13T00:26:25.645484Z","iopub.status.idle":"2024-03-13T00:26:25.725599Z","shell.execute_reply.started":"2024-03-13T00:26:25.645402Z","shell.execute_reply":"2024-03-13T00:26:25.724324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.DataFrame.from_dict({'image_id': image_id_test, 'dim0': dim0test, 'dim1': dim1test})\ndf_test.to_csv('test.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T00:26:37.518765Z","iopub.execute_input":"2024-03-13T00:26:37.51985Z","iopub.status.idle":"2024-03-13T00:26:37.539552Z","shell.execute_reply.started":"2024-03-13T00:26:37.5198Z","shell.execute_reply":"2024-03-13T00:26:37.538223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!tar -zcf train.tar.gz -C \"/kaggle/working/train/\" .\n!tar -zcf test.tar.gz -C \"/kaggle/working/test/\" .","metadata":{"execution":{"iopub.status.busy":"2024-03-12T23:45:41.768824Z","iopub.execute_input":"2024-03-12T23:45:41.769928Z","iopub.status.idle":"2024-03-12T23:47:22.976364Z","shell.execute_reply.started":"2024-03-12T23:45:41.769875Z","shell.execute_reply":"2024-03-12T23:47:22.973886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(f\"\")\ntrain_df.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}