{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# https://www.kaggle.com/code/deannahedges/mammography-challenge-dicom-to-png\n# results: https://www.kaggle.com/datasets/deannahedges/mammography-challenge-pngs\n\n# Sources:\n    # To go from Dicom -> PNG:\n        # https://www.kaggle.com/code/radek1/how-to-process-dicom-images-to-pngs/notebook?scriptVersionId=113529850\n    # To load the data, configure for performance, and build model in keras:\n        # https://www.tensorflow.org/tutorials/load_data/images#:~:text=This%20tutorial%20shows%20how%20to%20load%20and%20preprocess,from%20the%20large%20catalog%20available%20in%20TensorFlow%20Datasets.\n    # To augment the data:\n        # https://www.tensorflow.org/tutorials/images/data_augmentation\n    # To make the submission notebook:\n        # https://www.kaggle.com/code/radek1/fast-ai-starter-pack-train-inference/notebook\n        \n\nimport numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-02-27T20:04:39.119787Z","iopub.execute_input":"2023-02-27T20:04:39.120808Z","iopub.status.idle":"2023-02-27T20:04:39.126262Z","shell.execute_reply.started":"2023-02-27T20:04:39.120673Z","shell.execute_reply":"2023-02-27T20:04:39.124985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploring training csv with labels","metadata":{}},{"cell_type":"code","source":"train_file = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ntrain_file.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T20:04:39.144983Z","iopub.execute_input":"2023-02-27T20:04:39.145291Z","iopub.status.idle":"2023-02-27T20:04:39.269673Z","shell.execute_reply.started":"2023-02-27T20:04:39.145264Z","shell.execute_reply":"2023-02-27T20:04:39.268301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_file['cancer'] = train_file['cancer'].astype('float32')\ntrain_file.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T20:04:42.256067Z","iopub.execute_input":"2023-02-27T20:04:42.256626Z","iopub.status.idle":"2023-02-27T20:04:42.314842Z","shell.execute_reply.started":"2023-02-27T20:04:42.256577Z","shell.execute_reply":"2023-02-27T20:04:42.313572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_file['cancer'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T20:04:45.73391Z","iopub.execute_input":"2023-02-27T20:04:45.735046Z","iopub.status.idle":"2023-02-27T20:04:45.749157Z","shell.execute_reply.started":"2023-02-27T20:04:45.734998Z","shell.execute_reply":"2023-02-27T20:04:45.747686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating function to categorize images as \"positive\" or \"negative\" based on file path","metadata":{}},{"cell_type":"code","source":"def pos_or_neg(img_directory):\n    img_id = str(img_directory).split('/')[-1][:-4]\n    diagnosis = train_file.loc[train_file['image_id']==int(img_id), 'cancer'].values[0]\n    if diagnosis == 0:\n        return \"negative\"\n    else:\n        return \"positive\"","metadata":{"execution":{"iopub.status.busy":"2023-02-27T17:41:58.870682Z","iopub.execute_input":"2023-02-27T17:41:58.871058Z","iopub.status.idle":"2023-02-27T17:41:58.877645Z","shell.execute_reply.started":"2023-02-27T17:41:58.871024Z","shell.execute_reply":"2023-02-27T17:41:58.876164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install dicomsdl","metadata":{"execution":{"iopub.status.busy":"2023-02-27T17:42:00.567415Z","iopub.execute_input":"2023-02-27T17:42:00.568152Z","iopub.status.idle":"2023-02-27T17:42:11.057884Z","shell.execute_reply.started":"2023-02-27T17:42:00.568112Z","shell.execute_reply":"2023-02-27T17:42:11.056505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Transforming images from DICOM format to PNG and sorting them into a \"positive\" and \"negative\" folder","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport cv2\nimport os\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\nfrom pathlib import Path\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport dicomsdl\nimport sys\nimport time\n\nRESIZE_TO = (256, 256)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n!mkdir -p /kaggle/working/train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}/positive/\n!mkdir -p /kaggle/working/train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}/negative/\n\n# https://www.kaggle.com/code/tanlikesmath/brain-tumor-radiogenomic-classification-eda/notebook\ndef dicom_file_to_ary(path):\n    dcm_file = dicomsdl.open(str(path))\n    data = dcm_file.pixelData()\n\n    data = (data - data.min()) / (data.max() - data.min())\n\n    if dcm_file.getPixelDataInfo()['PhotometricInterpretation'] == \"MONOCHROME1\":\n        data = 1 - data\n\n    data = cv2.resize(data, RESIZE_TO)\n    data = (data * 255).astype(np.uint8)\n    return data\n\nimage_directories = []\nfor patient_dir in Path('/kaggle/input/rsna-breast-cancer-detection/train_images/').iterdir():\n    for pic_dir in patient_dir.iterdir():\n#         if pic_dir.stem not in done_ids:\n        image_directories.append(pic_dir)\nprint(len(image_directories))\n\ndef process_directory(directory_path):\n    parent_directory = pos_or_neg(directory_path)\n    \n    processed_ary = dicom_file_to_ary(directory_path)\n        \n    cv2.imwrite(\n        f'train_images_processed_cv2_dicomsdl_{RESIZE_TO[0]}/{parent_directory}/{directory_path.stem}.png',\n        processed_ary\n    )\npos_dir = Path(\"/kaggle/working/train_images_processed_cv2_dicomsdl_256/positive/\")\n    \nimport multiprocessing as mp\n\nwith mp.Pool(64) as p:\n    p.map(process_directory, image_directories)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-27T19:10:10.432834Z","iopub.execute_input":"2023-02-27T19:10:10.433377Z","iopub.status.idle":"2023-02-27T19:15:11.518525Z","shell.execute_reply.started":"2023-02-27T19:10:10.433319Z","shell.execute_reply":"2023-02-27T19:15:11.516011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Insuring that the final number of images matches the original number","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\ndata_dir = Path(\"/kaggle/working/train_images_processed_cv2_dicomsdl_256/\")\ndone_paths = list(data_dir.glob('*/*.png'))\nimage_count = len(list(data_dir.glob('*/*.png')))\nprint(image_count)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T19:16:47.429288Z","iopub.execute_input":"2023-02-27T19:16:47.430024Z","iopub.status.idle":"2023-02-27T19:16:47.466316Z","shell.execute_reply.started":"2023-02-27T19:16:47.429982Z","shell.execute_reply":"2023-02-27T19:16:47.464966Z"},"trusted":true},"execution_count":null,"outputs":[]}]}