{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\ni = 0\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        i += 1\n        if i > 5: break\n    if i > 5: break\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip download pylibjpeg gdcm python-gdcm pylibjpeg-libjpeg pylibjpeg-openjpeg pydicom -d ./pip_packages/","metadata":{"execution":{"iopub.status.busy":"2022-12-04T16:59:55.597436Z","iopub.execute_input":"2022-12-04T16:59:55.59811Z","iopub.status.idle":"2022-12-04T16:59:55.60518Z","shell.execute_reply.started":"2022-12-04T16:59:55.598059Z","shell.execute_reply":"2022-12-04T16:59:55.6037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# from zipfile import ZipFile\n\n# dirName = \"./pip_packages/\"\n# zipName = \"packages.zip\"\n\n# # Create a ZipFile Object\n# with ZipFile(zipName, 'w') as zipObj:\n#     # Iterate over all the files in directory\n#     for folderName, subfolders, filenames in os.walk(dirName):\n#         for filename in filenames:\n#             if (filename != zipName):\n#                 # create complete filepath of file in directory\n#                 filePath = os.path.join(folderName, filename)\n#                 # Add file to zip\n#                 zipObj.write(filePath)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:00:00.117968Z","iopub.execute_input":"2022-12-04T17:00:00.118363Z","iopub.status.idle":"2022-12-04T17:00:00.124354Z","shell.execute_reply.started":"2022-12-04T17:00:00.118329Z","shell.execute_reply":"2022-12-04T17:00:00.123013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -larth ../input/eda-fastai-timm-approach/pip_packages/","metadata":{"execution":{"iopub.status.busy":"2022-12-04T17:00:57.009408Z","iopub.execute_input":"2022-12-04T17:00:57.009826Z","iopub.status.idle":"2022-12-04T17:00:58.130463Z","shell.execute_reply.started":"2022-12-04T17:00:57.009789Z","shell.execute_reply":"2022-12-04T17:00:58.128927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pylibjpeg gdcm python-gdcm pylibjpeg-libjpeg pylibjpeg-openjpeg pydicom  --no-index --find-links=file:///kaggle/input/eda-fastai-timm-approach/pip_packages/\n\nimport pkg_resources\nlist(pkg_resources.WorkingSet(None).iter_entry_points(\"pylibjpeg.pixel_data_decoders\"))\n\n# Monkey patch in pylibjpeg\nimport pylibjpeg.utils\npylibjpeg.utils.iter_entry_points = pkg_resources.WorkingSet(None).iter_entry_points\n# Prove monkey patch worked\npylibjpeg.utils.get_pixel_data_decoders()\n\n# Proof monkey patch worked all the way\nimport pydicom.pixel_data_handlers.pylibjpeg_handler\npydicom.pixel_data_handlers.pylibjpeg_handler._DECODERS\n\nfrom PIL import Image\nfrom fastai.vision.widgets import *\nimport pandas as pd\nfrom fastcore.all import *\nfrom fastai.vision.all import *\nfrom fastai.medical.imaging import *\nimport seaborn as sns\n\ntrain_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_df.describe()\n\ntrain_df['image_id'].duplicated().sum()\n\nimport os\npath = Path('/kaggle/input/rsna-breast-cancer-detection/train_images/')\n\n# From debugging here: https://forums.fast.ai/t/fastai2-problems-with-medical-images-dicom/76138/5\nfrom pydicom.filereader import dcmread\ndef dcmread2(fn):\n    dcm = dcmread(fn)\n    intercept=dcm[0x0028, 0x1052].value\n    slope=dcm[0x0028, 0x1053].value    \n    arr=dcm.pixel_array\n    return intercept + arr*slope\n\nclass PILDicom2(PILBase):\n    \"same as PILDicom but changed pixel type to np.int32 as np.int16 cannot be handled by PIL\"\n    _open_args,_tensor_cls,_show_args = {},TensorDicom,TensorDicom._show_args\n    @classmethod\n    def create(cls, fn:(Path,str,bytes), mode=None)->None:\n        \"Open a `DICOM file` from path `fn` or bytes `fn` and load it as a `PIL Image`\"\n        # print(fn)\n        if isinstance(fn,bytes): im = Image.fromarray(dcmread2(pydicom.filebase.DicomBytesIO(fn)))\n        if isinstance(fn,(Path,str)): im = Image.fromarray(dcmread2(fn).astype(np.int32)) # images are np.int16, but this cannont be handled by PIL. Will throw wrong mode error. \n        im.load()\n        im = im._new(im.im)\n        return cls(im.convert(mode) if mode else im)\n    \nclass PILDicom3(PILBase):\n    \"same as PILDicom but changed pixel type to np.int32 as np.int16 cannot be handled by PIL\"\n    _open_args,_tensor_cls,_show_args = {},TensorDicom,TensorDicom._show_args\n    @classmethod\n    def create(cls, fn:(Path,str,bytes), mode=None)->None:\n        \"Open a `DICOM file` from path `fn` or bytes `fn` and load it as a `PIL Image`\"\n        if isinstance(fn, str):\n            fn = Path(fn)\n        im = fn.dcmread()\n        # Convert the pixels into an array using numpy\n        return np.array(im.pixels, dtype=np.int32)\n        \ndatablock = DataBlock(\n    blocks=(ImageBlock(cls=PILDicom2), CategoryBlock),\n    get_x=lambda x: path/f\"{x[1]}/{x[2]}.dcm\",\n    get_y=lambda x:x[6],\n    item_tfms=[Resize(224)],\n    batch_tfms=[\n        *aug_transforms(size=224),\n        Normalize.from_stats(*imagenet_stats)\n    ]\n)\n\ndls = datablock.dataloaders(train_df.sample(1_000, random_state=42+1), num_workers=0)\n\nlearn = vision_learner(dls, resnet18, metrics=accuracy)\nlearn.fine_tune(0)\n\ntest_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\n# path = Path(\"/kaggle/input/rsna-breast-cancer-detection/test_images/\")\ntest_datablock = DataBlock(\n    blocks=(ImageBlock(cls=PILDicom2), CategoryBlock),\n    get_x=lambda x: f\"/kaggle/input/rsna-breast-cancer-detection/test_images/{x[1]}/{x[2]}.dcm\",\n    get_y=lambda x:x[6],\n    item_tfms=[Resize(224)],\n    batch_tfms=[\n        *aug_transforms(size=224),\n        Normalize.from_stats(*imagenet_stats)\n    ],\n)\ntest_dl = test_datablock.dataloaders(test_df, num_workers=0, bs=min(len(test_df), 32)) # Create a test dataloader\npreds = learn.get_preds(dl = test_dl) # Make predictions on it\n\ntest_df['cancer'] = preds[1]\ntest_df['prediction_id'] = test_df['patient_id'].astype(str) + \"_\" + test_df['laterality']\ntest_df.groupby('prediction_id')[['cancer']].max().reset_index().to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-04T14:30:40.291091Z","iopub.execute_input":"2022-12-04T14:30:40.291594Z","iopub.status.idle":"2022-12-04T14:30:52.350413Z","shell.execute_reply.started":"2022-12-04T14:30:40.291545Z","shell.execute_reply":"2022-12-04T14:30:52.348855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}