{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip download pylibjpeg pylibjpeg-libjpeg pydicom python-gdcm","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:19:56.810234Z","iopub.execute_input":"2023-02-04T19:19:56.810792Z","iopub.status.idle":"2023-02-04T19:19:56.837844Z","shell.execute_reply.started":"2023-02-04T19:19:56.810587Z","shell.execute_reply":"2023-02-04T19:19:56.836844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install ../input/rsna-python-libraries/pydicom-2.3.1-py3-none-any.whl\n!pip install ../input/rsna-python-libraries/pylibjpeg-1.4.0-py3-none-any.whl\n!pip install ../input/rsna-python-libraries/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n!pip install ../input/rsna-python-libraries/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n!pip install ../input/rsna-python-libraries/python_gdcm-3.0.21-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n!pip install ../input/rsna-python-libraries/pylibjpeg_libjpeg-1.3.3-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n!pip install ../input/rsna-python-libraries/pylibjpeg_openjpeg-1.3.1-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:19:56.839899Z","iopub.execute_input":"2023-02-04T19:19:56.840724Z","iopub.status.idle":"2023-02-04T19:23:20.21073Z","shell.execute_reply.started":"2023-02-04T19:19:56.840689Z","shell.execute_reply":"2023-02-04T19:23:20.209763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gdcm\n\nimport importlib\nimportlib.reload(__import__(\"gdcm\"))\n\nfrom gdcm import DataElement\nimport pandas as pd\nimport os\nfrom pathlib import Path","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:23:20.211834Z","iopub.execute_input":"2023-02-04T19:23:20.212281Z","iopub.status.idle":"2023-02-04T19:23:20.25816Z","shell.execute_reply.started":"2023-02-04T19:23:20.212238Z","shell.execute_reply":"2023-02-04T19:23:20.25683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_name = \"rsna-breast-cancer-detection\"\njson = False\nlines = False\nsubset_rows = None\nfile_path = f\"/kaggle/input/{competition_name}\"\n\niskaggle = os.environ.get('KAGGLE_KERNEL_RUN_TYPE', '')\nif iskaggle:\n    path = Path(file_path)\nelse:\n    path = Path('titanic')\n    if not path.exists():\n        import zipfile\n        import kaggle\n        kaggle.api.competition_download_cli(str(path))\n        zipfile.ZipFile(f'{path}.zip').extractall(path)\n\n\n# load test and train data\n# [train/test]_images/[patient_id]/[image_id].dcm \nif json:\n    train = pd.read_json(f\"{path}/train.json\", lines=lines, nrows=subset_rows)\n    test = pd.read_json(f\"{path}/test.json\", lines=lines, nrows=subset_rows)\nelse:\n    train_csv = pd.read_csv(f\"{path}/train.csv\", nrows=subset_rows)\n    test_csv = pd.read_csv(f\"{path}/test.csv\", nrows=subset_rows)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:23:20.261229Z","iopub.execute_input":"2023-02-04T19:23:20.261691Z","iopub.status.idle":"2023-02-04T19:23:20.37804Z","shell.execute_reply.started":"2023-02-04T19:23:20.261656Z","shell.execute_reply":"2023-02-04T19:23:20.376755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv['test'] = False\ntest_csv['test'] = True","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:23:20.380528Z","iopub.execute_input":"2023-02-04T19:23:20.380995Z","iopub.status.idle":"2023-02-04T19:23:20.395661Z","shell.execute_reply.started":"2023-02-04T19:23:20.380959Z","shell.execute_reply":"2023-02-04T19:23:20.393992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fastai.basics import *\nfrom fastai.callback.all import *\nfrom fastai.vision.all import *\nfrom fastai.medical.imaging import *\n\nimport pydicom\n\nimport pandas as pd\n\nfrom pydicom import dcmread\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport gdcm","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:23:20.397505Z","iopub.execute_input":"2023-02-04T19:23:20.397979Z","iopub.status.idle":"2023-02-04T19:23:24.945872Z","shell.execute_reply.started":"2023-02-04T19:23:20.397954Z","shell.execute_reply":"2023-02-04T19:23:24.944381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row = 1\ndcm_path = f'{path}/test_images/{test_csv.loc[row,\"patient_id\"]}/{test_csv.loc[row, \"image_id\"]}.dcm'\ndcm = dcmread(dcm_path)#, force=True)\n# transfer syntaxes https://pydicom.github.io/pydicom/stable/old/image_data_handlers.html\n\nimage = Image.fromarray(dcm.pixel_array.astype(float))\nplt.imshow(dcm.pixel_array, cmap=plt.cm.bone)","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:23:24.947213Z","iopub.execute_input":"2023-02-04T19:23:24.947836Z","iopub.status.idle":"2023-02-04T19:23:26.155103Z","shell.execute_reply.started":"2023-02-04T19:23:24.94781Z","shell.execute_reply":"2023-02-04T19:23:26.154191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"patient_id_col = test_csv.columns.get_loc('patient_id')\nimage_id_col = test_csv.columns.get_loc('image_id')\nprint(image_id_col)","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:31:06.265173Z","iopub.execute_input":"2023-02-04T19:31:06.265536Z","iopub.status.idle":"2023-02-04T19:31:06.271179Z","shell.execute_reply.started":"2023-02-04T19:31:06.265509Z","shell.execute_reply":"2023-02-04T19:31:06.270157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\n\ndef get_x(x):\n    sub_path = 'test_images'  if x[-1] else 'train_images'\n    print(x)\n    return f\"{path}/{sub_path}/{x[patient_id_col]}/{x[image_id_col]}.dcm\"\n\ndef get_y(y):\n    return y['cancer']\n    \n# cancer = DataBlock(\n#         blocks=(\n#             ImageBlock(cls=PILDicom),\n#             CategoryBlock\n#         ),\n#         get_x=get_x,\n#         get_y=get_y,\n#         item_tfms=Resize(224),\n#         batch_tfms=[\n#             *aug_transforms(size=224),\n#             Normalize.from_stats(*imagenet_stats)\n#         ]\n#     )]\n\n# class PILDicomCustom(PILBase):\n#     _open_args,_tensor_cls,_show_args = {},TensorDicom,TensorDicom._show_args\n#     @classmethod\n#     def create(cls, fn:Path|str|bytes, mode=None)->None:\n#         \"Open a `DICOM file` from path `fn` or bytes `fn` and load it as a `PIL Image`\"\n#         if isinstance(fn,bytes): im = Image.fromarray(pydicom.dcmread(pydicom.filebase.DicomBytesIO(fn)).pixel_array)\n#         if isinstance(fn,(Path,str)): im = Image.fromarray(pydicom.dcmread(fn).pixel_array)\n#         im.load()\n#         im = im._new(im.im)\n#         return cls(im.convert(mode) if mode else im)\n\n\ncancer = DataBlock(\n        blocks=(\n            ImageBlock(cls=PILDicom),\n            CategoryBlock\n        ),\n        get_x=get_x,\n        get_y=get_y,\n        item_tfms=[Resize(224, resamples= (Image.Resampling.NEAREST,0))],\n        batch_tfms=[\n            IntToFloatTensor(div=2**16-1),\n            *aug_transforms(size=224),\n            Normalize.from_stats(*imagenet_stats)\n        ]\n    )\n","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:31:21.222688Z","iopub.execute_input":"2023-02-04T19:31:21.223045Z","iopub.status.idle":"2023-02-04T19:31:21.23755Z","shell.execute_reply.started":"2023-02-04T19:31:21.223019Z","shell.execute_reply":"2023-02-04T19:31:21.236071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import datetime\nprint('_________________________')\nprint (datetime.datetime.now())\nprint('dataloaders')\nprint('_________________________')","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:23:26.182759Z","iopub.status.idle":"2023-02-04T19:23:26.18307Z","shell.execute_reply.started":"2023-02-04T19:23:26.182916Z","shell.execute_reply":"2023-02-04T19:23:26.182931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# download https://download.pytorch.org/models/resnet34-b627a593.pth & upload as data source / https://www.kaggle.com/datasets/pytorch/resnet34\nimport os\nif not os.path.exists('/root/.cache/torch/hub/checkpoints/'):\n        os.makedirs('/root/.cache/torch/hub/checkpoints/')\n!cp '../input/resnet34/resnet34.pth' '/root/.cache/torch/hub/checkpoints/resnet34-b627a593.pth'","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:31:32.439968Z","iopub.execute_input":"2023-02-04T19:31:32.440289Z","iopub.status.idle":"2023-02-04T19:31:34.044785Z","shell.execute_reply.started":"2023-02-04T19:31:32.440266Z","shell.execute_reply":"2023-02-04T19:31:34.043157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import datetime\nprint('_________________________')\nprint (datetime.datetime.now())\nprint('lr_find')\nprint('_________________________')","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:31:34.048517Z","iopub.execute_input":"2023-02-04T19:31:34.049143Z","iopub.status.idle":"2023-02-04T19:31:34.056426Z","shell.execute_reply.started":"2023-02-04T19:31:34.049104Z","shell.execute_reply":"2023-02-04T19:31:34.054912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('hi')\nlearn = load_learner('../input/version-27-rsna/learner.pkl', cpu=False)\n\n# version27-rsna","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:31:34.15746Z","iopub.execute_input":"2023-02-04T19:31:34.157798Z","iopub.status.idle":"2023-02-04T19:31:35.045441Z","shell.execute_reply.started":"2023-02-04T19:31:34.157773Z","shell.execute_reply":"2023-02-04T19:31:35.044475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import datetime\nprint('_________________________')\nprint (datetime.datetime.now())\nprint('interp')\nprint('_________________________')","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:31:35.047716Z","iopub.execute_input":"2023-02-04T19:31:35.048256Z","iopub.status.idle":"2023-02-04T19:31:35.053395Z","shell.execute_reply.started":"2023-02-04T19:31:35.048228Z","shell.execute_reply":"2023-02-04T19:31:35.052193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_csv","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:31:35.787869Z","iopub.execute_input":"2023-02-04T19:31:35.788231Z","iopub.status.idle":"2023-02-04T19:31:35.807181Z","shell.execute_reply.started":"2023-02-04T19:31:35.788205Z","shell.execute_reply":"2023-02-04T19:31:35.805853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dl = learn.dls.test_dl(test_csv.values)\npredictions, _, decoded = learn.get_preds(dl=test_dl, with_decoded=True)\nprint('predictions:')\nprint(predictions)\nprint('decoded')\nprint(decoded)","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:31:36.663855Z","iopub.execute_input":"2023-02-04T19:31:36.664189Z","iopub.status.idle":"2023-02-04T19:31:38.505144Z","shell.execute_reply.started":"2023-02-04T19:31:36.664164Z","shell.execute_reply":"2023-02-04T19:31:38.504167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# submit_df = predictions[['prediction_id', 'cancer']]\n# submission = pd.DataFrame([])\n# submission['cancer'] = pd.DataFrame(predictions.numpy())[0]\n# submission['prediction_id']  = test_csv['patient_id'].astype(str) + \"-\" + test_csv['laterality']\n# # submission.columns = ['predictions_id', 'cancer']\n# submission.sort_index()\n# # subsmission.groupby('prediction_id')","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:23:26.197115Z","iopub.status.idle":"2023-02-04T19:23:26.197533Z","shell.execute_reply.started":"2023-02-04T19:23:26.197328Z","shell.execute_reply":"2023-02-04T19:23:26.197348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame([])\nsubmission['prediction_id'] = test_csv['patient_id'].astype(str) + \"_\" + test_csv['laterality']\nsubmission['cancer'] = pd.DataFrame(predictions.numpy())[0]\nsubmission = submission.groupby('prediction_id').mean().reset_index()\nsubmission.to_csv('submission.csv', index=False)\nsubmission.head()\n! head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:31:38.613385Z","iopub.execute_input":"2023-02-04T19:31:38.615053Z","iopub.status.idle":"2023-02-04T19:31:38.937976Z","shell.execute_reply.started":"2023-02-04T19:31:38.614992Z","shell.execute_reply":"2023-02-04T19:31:38.936486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/working\n","metadata":{"execution":{"iopub.status.busy":"2023-02-04T19:31:40.748623Z","iopub.execute_input":"2023-02-04T19:31:40.748972Z","iopub.status.idle":"2023-02-04T19:31:41.048694Z","shell.execute_reply.started":"2023-02-04T19:31:40.748944Z","shell.execute_reply":"2023-02-04T19:31:41.04761Z"},"trusted":true},"execution_count":null,"outputs":[]}]}