{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":4619805,"sourceType":"datasetVersion","datasetId":2688675},{"sourceId":4679582,"sourceType":"datasetVersion","datasetId":2711917}],"dockerImageVersionId":30636,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Description\nA pre-trained model for breast-density classification.\n\n# Model Overview\nThis model is trained using transfer learning on InceptionV3. The model weights were fine tuned using the Mayo Clinic Data. The details of training and data is outlined in https://arxiv.org/abs/2202.08238. The images should be resampled to a size [299, 299, 3] for training.\nA training pipeline will be added to the model zoo in near future.\nThe bundle does not support torchscript.\n\n\n# Input and Output Formats\nThe input image should have the size [299, 299, 3]. For a dicom image which are single channel. The channel can be repeated 3 times.\nThe output is an array with probabilities for each of the four class.\n","metadata":{}},{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg monai","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:46:47.204513Z","iopub.execute_input":"2024-01-25T11:46:47.204868Z","iopub.status.idle":"2024-01-25T11:47:02.474809Z","shell.execute_reply.started":"2024-01-25T11:46:47.204835Z","shell.execute_reply":"2024-01-25T11:47:02.473727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport sys\nimport json\nimport glob\nimport gdcm\nimport torch\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed\nfrom monai.bundle.config_parser import ConfigParser\n\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression ","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:02.47662Z","iopub.execute_input":"2024-01-25T11:47:02.476925Z","iopub.status.idle":"2024-01-25T11:47:43.656698Z","shell.execute_reply.started":"2024-01-25T11:47:02.476899Z","shell.execute_reply":"2024-01-25T11:47:43.655698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"preprocessing ","metadata":{}},{"cell_type":"code","source":"def process(f, size=512, save_folder=\"\", extension=\"png\"):\n    patient = f.split('/')[-2]\n    image = f.split('/')[-1][:-4]\n\n    dicom = pydicom.dcmread(f)\n    img = dicom.pixel_array\n\n    img = (img - img.min()) / (img.max() - img.min())\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n\n    img = cv2.resize(img, (size, size))\n\n    cv2.imwrite(save_folder + f\"{patient}_{image}.{extension}\", (img * 255).astype(np.uint8))","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:43.658145Z","iopub.execute_input":"2024-01-25T11:47:43.658969Z","iopub.status.idle":"2024-01-25T11:47:43.665853Z","shell.execute_reply.started":"2024-01-25T11:47:43.65894Z","shell.execute_reply":"2024-01-25T11:47:43.664801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = glob.glob(\"/kaggle/input/rsna-breast-cancer-detection/test_images/*/*.dcm\")\n\nSAVE_FOLDER = \"/kaggle/working/output/\"\nSIZE = 512\nEXTENSION = \"png\"\n\nos.makedirs(SAVE_FOLDER, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:43.667923Z","iopub.execute_input":"2024-01-25T11:47:43.668227Z","iopub.status.idle":"2024-01-25T11:47:43.760401Z","shell.execute_reply.started":"2024-01-25T11:47:43.668203Z","shell.execute_reply":"2024-01-25T11:47:43.75953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = Parallel(n_jobs=2)(\n    delayed(process)(uid, size=SIZE, save_folder=SAVE_FOLDER, extension=EXTENSION)\n    for uid in tqdm(images)\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:43.761494Z","iopub.execute_input":"2024-01-25T11:47:43.761764Z","iopub.status.idle":"2024-01-25T11:47:46.022116Z","shell.execute_reply.started":"2024-01-25T11:47:43.761741Z","shell.execute_reply":"2024-01-25T11:47:46.021117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"monai density","metadata":{}},{"cell_type":"code","source":"MODEL_PATH = \"/kaggle/input/monai-breast-density-classification/breast_density_classification/\"","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:46.023513Z","iopub.execute_input":"2024-01-25T11:47:46.023822Z","iopub.status.idle":"2024-01-25T11:47:46.028464Z","shell.execute_reply.started":"2024-01-25T11:47:46.023794Z","shell.execute_reply":"2024-01-25T11:47:46.027527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cp -r $MODEL_PATH ./","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:46.029738Z","iopub.execute_input":"2024-01-25T11:47:46.030094Z","iopub.status.idle":"2024-01-25T11:47:48.686163Z","shell.execute_reply.started":"2024-01-25T11:47:46.030061Z","shell.execute_reply":"2024-01-25T11:47:48.685022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd breast_density_classification","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:48.687791Z","iopub.execute_input":"2024-01-25T11:47:48.688113Z","iopub.status.idle":"2024-01-25T11:47:48.694566Z","shell.execute_reply.started":"2024-01-25T11:47:48.688083Z","shell.execute_reply":"2024-01-25T11:47:48.693722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_dataset(base_dir: str, output_file: str):\n    list_classes = [\"A\", \"B\", \"C\", \"D\"]\n\n    output_list = []\n    for _class in list_classes:\n        data_dir = os.path.join(base_dir, _class)\n        list_files = os.listdir(data_dir)\n        if _class == \"A\":\n            _label = [1, 0, 0, 0]\n        elif _class == \"B\":\n            _label = [0, 1, 0, 0]\n        elif _class == \"C\":\n            _label = [0, 0, 1, 0]\n        elif _class == \"D\":\n            _label = [0, 0, 0, 1]\n\n        for _file in list_files:\n            _out = {\"image\": os.path.join(data_dir, _file), \"label\": _label}\n            output_list.append(_out)\n\n    data_dict = {\"Test\": output_list}\n\n    fid = open(output_file, \"w\")\n    json.dump(data_dict, fid, indent=1)","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:48.695926Z","iopub.execute_input":"2024-01-25T11:47:48.696208Z","iopub.status.idle":"2024-01-25T11:47:48.704624Z","shell.execute_reply.started":"2024-01-25T11:47:48.696178Z","shell.execute_reply":"2024-01-25T11:47:48.703684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_dataset(\"sample_data\", \"configs/sample_image_data.json\")","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:48.70932Z","iopub.execute_input":"2024-01-25T11:47:48.709749Z","iopub.status.idle":"2024-01-25T11:47:48.717775Z","shell.execute_reply.started":"2024-01-25T11:47:48.709724Z","shell.execute_reply":"2024-01-25T11:47:48.717045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"config","metadata":{}},{"cell_type":"code","source":"CONFIG_FILE = \"/kaggle/working/breast_density_classification/configs/inference.json\"","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:48.718956Z","iopub.execute_input":"2024-01-25T11:47:48.719423Z","iopub.status.idle":"2024-01-25T11:47:48.725822Z","shell.execute_reply.started":"2024-01-25T11:47:48.719386Z","shell.execute_reply":"2024-01-25T11:47:48.72491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile $CONFIG_FILE\n\n{\n    \"import\": [\n        \"$import glob\",\n        \"$import os\",\n        \"$import torchvision\"\n    ],\n    \"bundle_root\": \".\",\n    \"model_dir\": \"$@bundle_root + '/models'\",\n    \"output_dir\": \"$@bundle_root + '/output'\",\n    \"data\": {\n        \"_target_\": \"createList.CreateImageLabelList\",\n        \"filename\": \"../configs/sample_image_data.json\"\n    },\n    \"test_imagelist\": \"$@data.create_dataset('Test')[0]\",\n    \"test_labellist\": \"$@data.create_dataset('Test')[1]\",\n    \"dataset\": {\n        \"_target_\": \"CacheDataset\",\n        \"data\": \"$[{'image': i, 'label': l} for i, l in zip(@test_imagelist, @test_labellist)]\",\n        \"transform\": \"@preprocessing\",\n        \"cache_rate\": 1,\n        \"num_workers\": 2\n    },\n    \"dataloader\": {\n        \"_target_\": \"DataLoader\",\n        \"dataset\": \"@dataset\",\n        \"batch_size\": 16,\n        \"shuffle\": false,\n        \"num_workers\": 2\n    },\n    \"device\": \"$torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\",\n    \"network_def\": {\n        \"_target_\": \"TorchVisionFCModel\",\n        \"model_name\": \"inception_v3\",\n        \"num_classes\": 4,\n        \"pool\": null,\n        \"use_conv\": false,\n        \"bias\": true,\n        \"pretrained\": true\n    },\n    \"network\": \"$@network_def.to(@device)\",\n    \"preprocessing\": {\n        \"_target_\": \"Compose\",\n        \"transforms\": [\n            {\n                \"_target_\": \"LoadImaged\",\n                \"keys\": \"image\"\n            },\n            {\n                \"_target_\": \"EnsureChannelFirstd\",\n                \"keys\": \"image\",\n                \"channel_dim\": 2\n            },\n            {\n                \"_target_\": \"ScaleIntensityd\",\n                \"keys\": \"image\",\n                \"minv\": 0.0,\n                \"maxv\": 1.0\n            },\n            {\n                \"_target_\": \"Resized\",\n                \"keys\": \"image\",\n                \"spatial_size\": [\n                    299,\n                    299\n                ]\n            }\n        ]\n    },\n    \"inferer\": {\n        \"_target_\": \"SimpleInferer\"\n    },\n    \"postprocessing\": {\n        \"_target_\": \"Compose\",\n        \"transforms\": [\n            {\n                \"_target_\": \"Activationsd\",\n                \"keys\": \"pred\",\n                \"sigmoid\": false\n            }\n        ]\n    },\n    \"handlers\": [\n        {\n            \"_target_\": \"CheckpointLoader\",\n            \"load_path\": \"$@model_dir + '/model.pt'\",\n            \"load_dict\": {\n                \"model\": \"@network\"\n            }\n        },\n        {\n            \"_target_\": \"StatsHandler\",\n            \"iteration_log\": false,\n            \"output_transform\": \"$lambda x: None\"\n        },\n        {\n            \"_target_\": \"ClassificationSaver\",\n            \"output_dir\": \"@output_dir\",\n            \"batch_transform\": \"$monai.handlers.from_engine(['image_meta_dict'])\",\n            \"output_transform\": \"$monai.handlers.from_engine(['pred'])\"\n        }\n    ],\n    \"evaluator\": {\n        \"_target_\": \"SupervisedEvaluator\",\n        \"device\": \"@device\",\n        \"val_data_loader\": \"@dataloader\",\n        \"network\": \"@network\",\n        \"inferer\": \"@inferer\",\n        \"postprocessing\": \"@postprocessing\",\n        \"val_handlers\": \"@handlers\",\n        \"amp\": true\n    },\n    \"evaluating\": [\n        \"$setattr(torch.backends.cudnn, 'benchmark', True)\",\n        \"$@evaluator.run()\"\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:48.727751Z","iopub.execute_input":"2024-01-25T11:47:48.728033Z","iopub.status.idle":"2024-01-25T11:47:48.738539Z","shell.execute_reply.started":"2024-01-25T11:47:48.728003Z","shell.execute_reply":"2024-01-25T11:47:48.737715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"parser","metadata":{}},{"cell_type":"code","source":"cd /kaggle/working/breast_density_classification/scripts","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:48.739834Z","iopub.execute_input":"2024-01-25T11:47:48.740476Z","iopub.status.idle":"2024-01-25T11:47:48.751103Z","shell.execute_reply.started":"2024-01-25T11:47:48.74045Z","shell.execute_reply":"2024-01-25T11:47:48.750254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parser = ConfigParser()\n\nparser.read_config(CONFIG_FILE)\n\ndata = parser.get_parsed_content(\"data\")\ndevice = parser.get_parsed_content(\"device\")","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:48.752211Z","iopub.execute_input":"2024-01-25T11:47:48.754539Z","iopub.status.idle":"2024-01-25T11:47:48.832338Z","shell.execute_reply.started":"2024-01-25T11:47:48.754508Z","shell.execute_reply":"2024-01-25T11:47:48.831535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd /kaggle/working/breast_density_classification/","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:48.833407Z","iopub.execute_input":"2024-01-25T11:47:48.833675Z","iopub.status.idle":"2024-01-25T11:47:48.83971Z","shell.execute_reply.started":"2024-01-25T11:47:48.833652Z","shell.execute_reply":"2024-01-25T11:47:48.838909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"inf","metadata":{}},{"cell_type":"code","source":"def predict(parser):\n    inference = parser.get_parsed_content(\"inferer\")\n    loader = parser.get_parsed_content(\"dataloader\")\n    network = parser.get_parsed_content(\"network_def\")\n    \n    state_dict = torch.load(\"/kaggle/input/monai-breast-density-classification/breast_density_classification/models/model.pt\")\n    network.load_state_dict(state_dict, strict=True)\n\n    preds = []\n    network.eval()\n    with torch.no_grad():\n        for batch in tqdm(loader):\n            pred = inference(batch['image'], network)\n            pred = pred.softmax(-1)\n            preds.append(pred.detach().cpu().numpy())\n\n    return np.concatenate(preds)","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:48.840954Z","iopub.execute_input":"2024-01-25T11:47:48.841243Z","iopub.status.idle":"2024-01-25T11:47:48.851336Z","shell.execute_reply.started":"2024-01-25T11:47:48.84122Z","shell.execute_reply":"2024-01-25T11:47:48.850457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npreds = predict(parser)","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:48.852366Z","iopub.execute_input":"2024-01-25T11:47:48.852618Z","iopub.status.idle":"2024-01-25T11:47:58.602488Z","shell.execute_reply.started":"2024-01-25T11:47:48.852595Z","shell.execute_reply":"2024-01-25T11:47:58.601478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"RSNA","metadata":{}},{"cell_type":"code","source":"def create_rsna_dataset(base_dir: str, output_file: str, num_files=0):\n    output_list = []\n\n    for _file in glob.glob(base_dir + \"*.png\"):\n        _out = {\"image\": _file, \"label\": [0, 0, 0, 0]}\n        output_list.append(_out)\n\n    if num_files:\n        output_list = output_list[:num_files]\n\n    data_dict = {\"Test\": output_list}\n\n    fid = open(output_file, \"w\")\n    json.dump(data_dict, fid, indent=1)","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:58.604043Z","iopub.execute_input":"2024-01-25T11:47:58.604764Z","iopub.status.idle":"2024-01-25T11:47:58.611284Z","shell.execute_reply.started":"2024-01-25T11:47:58.604725Z","shell.execute_reply":"2024-01-25T11:47:58.610241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_rsna_dataset(SAVE_FOLDER, \"configs/rsna_test.json\")","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:58.612447Z","iopub.execute_input":"2024-01-25T11:47:58.612786Z","iopub.status.idle":"2024-01-25T11:47:58.66485Z","shell.execute_reply.started":"2024-01-25T11:47:58.612755Z","shell.execute_reply":"2024-01-25T11:47:58.664113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"config","metadata":{}},{"cell_type":"code","source":"CONFIG_FILE = \"/kaggle/working/breast_density_classification/configs/inference_rsna_test.json\"","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:58.665983Z","iopub.execute_input":"2024-01-25T11:47:58.666329Z","iopub.status.idle":"2024-01-25T11:47:58.673635Z","shell.execute_reply.started":"2024-01-25T11:47:58.666298Z","shell.execute_reply":"2024-01-25T11:47:58.672822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile $CONFIG_FILE\n\n{\n    \"import\": [\n        \"$import glob\",\n        \"$import os\",\n        \"$import torchvision\"\n    ],\n    \"bundle_root\": \".\",\n    \"model_dir\": \"$@bundle_root + '/models'\",\n    \"output_dir\": \"$@bundle_root + '/output'\",\n    \"data\": {\n        \"_target_\": \"createList.CreateImageLabelList\",\n        \"filename\": \"../configs/rsna_test.json\"\n    },\n    \"test_imagelist\": \"$@data.create_dataset('Test')[0]\",\n    \"test_labellist\": \"$@data.create_dataset('Test')[1]\",\n    \"dataset\": {\n        \"_target_\": \"CacheDataset\",\n        \"data\": \"$[{'image': i, 'label': l} for i, l in zip(@test_imagelist, @test_labellist)]\",\n        \"transform\": \"@preprocessing\",\n        \"cache_rate\": 1,\n        \"num_workers\": 2\n    },\n    \"dataloader\": {\n        \"_target_\": \"DataLoader\",\n        \"dataset\": \"@dataset\",\n        \"batch_size\": 16,\n        \"shuffle\": false,\n        \"num_workers\": 2\n    },\n    \"device\": \"$torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\",\n    \"network_def\": {\n        \"_target_\": \"TorchVisionFCModel\",\n        \"model_name\": \"inception_v3\",\n        \"num_classes\": 4,\n        \"pool\": null,\n        \"use_conv\": false,\n        \"bias\": true,\n        \"pretrained\": true\n    },\n    \"network\": \"$@network_def.to(@device)\",\n    \"preprocessing\": {\n        \"_target_\": \"Compose\",\n        \"transforms\": [\n            {\n                \"_target_\": \"LoadImaged\",\n                \"reader\": \"PILReader\",\n                \"converter\" : \"$lambda img: img.convert('RGB')\",\n                \"keys\": \"image\"\n            },\n            {\n                \"_target_\": \"EnsureChannelFirstd\",\n                \"keys\": \"image\",\n                \"channel_dim\": 2\n            },\n            {\n                \"_target_\": \"ScaleIntensityd\",\n                \"keys\": \"image\",\n                \"minv\": 0.0,\n                \"maxv\": 1.0\n            },\n            {\n                \"_target_\": \"Resized\",\n                \"keys\": \"image\",\n                \"spatial_size\": [\n                    299,\n                    299\n                ]\n            }\n        ]\n    },\n    \"inferer\": {\n        \"_target_\": \"SimpleInferer\"\n    },\n    \"postprocessing\": {\n        \"_target_\": \"Compose\",\n        \"transforms\": [\n            {\n                \"_target_\": \"Activationsd\",\n                \"keys\": \"pred\",\n                \"sigmoid\": false\n            }\n        ]\n    },\n    \"handlers\": [\n        {\n            \"_target_\": \"CheckpointLoader\",\n            \"load_path\": \"$@model_dir + '/model.pt'\",\n            \"load_dict\": {\n                \"model\": \"@network\"\n            }\n        },\n        {\n            \"_target_\": \"StatsHandler\",\n            \"iteration_log\": false,\n            \"output_transform\": \"$lambda x: None\"\n        },\n        {\n            \"_target_\": \"ClassificationSaver\",\n            \"output_dir\": \"@output_dir\",\n            \"batch_transform\": \"$monai.handlers.from_engine(['image_meta_dict'])\",\n            \"output_transform\": \"$monai.handlers.from_engine(['pred'])\"\n        }\n    ],\n    \"evaluator\": {\n        \"_target_\": \"SupervisedEvaluator\",\n        \"device\": \"@device\",\n        \"val_data_loader\": \"@dataloader\",\n        \"network\": \"@network\",\n        \"inferer\": \"@inferer\",\n        \"postprocessing\": \"@postprocessing\",\n        \"val_handlers\": \"@handlers\",\n        \"amp\": true\n    },\n    \"evaluating\": [\n        \"$setattr(torch.backends.cudnn, 'benchmark', True)\",\n        \"$@evaluator.run()\"\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:58.675164Z","iopub.execute_input":"2024-01-25T11:47:58.675494Z","iopub.status.idle":"2024-01-25T11:47:58.685145Z","shell.execute_reply.started":"2024-01-25T11:47:58.675463Z","shell.execute_reply":"2024-01-25T11:47:58.684317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"inf","metadata":{}},{"cell_type":"code","source":"cd /kaggle/working/breast_density_classification/scripts","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:58.686243Z","iopub.execute_input":"2024-01-25T11:47:58.686916Z","iopub.status.idle":"2024-01-25T11:47:58.696767Z","shell.execute_reply.started":"2024-01-25T11:47:58.686884Z","shell.execute_reply":"2024-01-25T11:47:58.695914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parser = ConfigParser()\n\nparser.read_config(CONFIG_FILE)\n\ndata = parser.get_parsed_content(\"data\")\ndevice = parser.get_parsed_content(\"device\")","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:58.697776Z","iopub.execute_input":"2024-01-25T11:47:58.698119Z","iopub.status.idle":"2024-01-25T11:47:58.746637Z","shell.execute_reply.started":"2024-01-25T11:47:58.698096Z","shell.execute_reply":"2024-01-25T11:47:58.745735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd /kaggle/working/breast_density_classification/","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:58.747785Z","iopub.execute_input":"2024-01-25T11:47:58.748131Z","iopub.status.idle":"2024-01-25T11:47:58.753641Z","shell.execute_reply.started":"2024-01-25T11:47:58.7481Z","shell.execute_reply":"2024-01-25T11:47:58.752724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npreds = predict(parser)","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:58.754675Z","iopub.execute_input":"2024-01-25T11:47:58.754923Z","iopub.status.idle":"2024-01-25T11:47:59.953704Z","shell.execute_reply.started":"2024-01-25T11:47:58.754901Z","shell.execute_reply":"2024-01-25T11:47:59.952763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"RSNA train","metadata":{}},{"cell_type":"code","source":"'''\nSplitting into two if I want to run through all photos\n\nimport shutil\n\ndef split_folder(input_folder, output_folder1, output_folder2, split_ratio):\n    # Create output folders if they don't exist\n    os.makedirs(output_folder1, exist_ok=True)\n    os.makedirs(output_folder2, exist_ok=True)\n\n    # List all files in the input folder\n    files = os.listdir(input_folder)\n\n    # Calculate the split point\n    split_point = int(len(files) * split_ratio)\n\n    # Copy files to the first output folder\n    for file in files[:split_point]:\n        src_path = os.path.join(input_folder, file)\n        dest_path = os.path.join(output_folder1, file)\n        shutil.copy(src_path, dest_path)\n\n    # Copy files to the second output folder\n    for file in files[split_point:]:\n        src_path = os.path.join(input_folder, file)\n        dest_path = os.path.join(output_folder2, file)\n        shutil.copy(src_path, dest_path)\n\n# Example usage\ninput_folder = '/kaggle/input/rsna-breast-cancer-512-pngs'\noutput_folder1 = '/kaggle/working/folder1'\noutput_folder2 = '/kaggle/working/folder2'\nsplit_ratio = 0.5  # Adjust the ratio as needed\n\nsplit_folder(input_folder, output_folder1, output_folder2, split_ratio)'''","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:59.955173Z","iopub.execute_input":"2024-01-25T11:47:59.955483Z","iopub.status.idle":"2024-01-25T11:47:59.964457Z","shell.execute_reply.started":"2024-01-25T11:47:59.955453Z","shell.execute_reply":"2024-01-25T11:47:59.963549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#25000 for faster \ncreate_rsna_dataset(\"/kaggle/input/rsna-breast-cancer-512-pngs/\", \"configs/rsna_train.json\", num_files=25000)  \n#create_rsna_dataset(\"/kaggle/working/folder1\", \"configs/rsna_train.json\")  ","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:47:59.965924Z","iopub.execute_input":"2024-01-25T11:47:59.966214Z","iopub.status.idle":"2024-01-25T11:48:04.913594Z","shell.execute_reply.started":"2024-01-25T11:47:59.966192Z","shell.execute_reply":"2024-01-25T11:48:04.912543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CONFIG_FILE = \"/kaggle/working/breast_density_classification/configs/inference_rsna_train.json\"","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:48:04.91885Z","iopub.execute_input":"2024-01-25T11:48:04.919167Z","iopub.status.idle":"2024-01-25T11:48:04.923626Z","shell.execute_reply.started":"2024-01-25T11:48:04.919141Z","shell.execute_reply":"2024-01-25T11:48:04.922643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile $CONFIG_FILE\n\n{\n    \"import\": [\n        \"$import glob\",\n        \"$import os\",\n        \"$import torchvision\"\n    ],\n    \"bundle_root\": \".\",\n    \"model_dir\": \"$@bundle_root + '/models'\",\n    \"output_dir\": \"$@bundle_root + '/output'\",\n    \"data\": {\n        \"_target_\": \"createList.CreateImageLabelList\",\n        \"filename\": \"../configs/rsna_train.json\"\n    },\n    \"test_imagelist\": \"$@data.create_dataset('Test')[0]\",\n    \"test_labellist\": \"$@data.create_dataset('Test')[1]\",\n    \"dataset\": {\n        \"_target_\": \"CacheDataset\",\n        \"data\": \"$[{'image': i, 'label': l} for i, l in zip(@test_imagelist, @test_labellist)]\",\n        \"transform\": \"@preprocessing\",\n        \"copy_cache\": false,\n        \"cache_rate\": 0.1,\n        \"num_workers\": 2\n    },\n    \"dataloader\": {\n        \"_target_\": \"DataLoader\",\n        \"dataset\": \"@dataset\",\n        \"batch_size\": 16,\n        \"shuffle\": false,\n        \"num_workers\": 2\n    },\n    \"device\": \"$torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\",\n    \"network_def\": {\n        \"_target_\": \"TorchVisionFCModel\",\n        \"model_name\": \"inception_v3\",\n        \"num_classes\": 4,\n        \"pool\": null,\n        \"use_conv\": false,\n        \"bias\": true,\n        \"pretrained\": true\n    },\n    \"network\": \"$@network_def.to(@device)\",\n    \"preprocessing\": {\n        \"_target_\": \"Compose\",\n        \"transforms\": [\n            {\n                \"_target_\": \"LoadImaged\",\n                \"reader\": \"PILReader\",\n                \"converter\" : \"$lambda img: img.convert('RGB')\",\n                \"keys\": \"image\"\n            },\n            {\n                \"_target_\": \"EnsureChannelFirstd\",\n                \"keys\": \"image\",\n                \"channel_dim\": 2\n            },\n            {\n                \"_target_\": \"ScaleIntensityd\",\n                \"keys\": \"image\",\n                \"minv\": 0.0,\n                \"maxv\": 1.0\n            },\n            {\n                \"_target_\": \"Resized\",\n                \"keys\": \"image\",\n                \"spatial_size\": [\n                    299,\n                    299\n                ]\n            }\n        ]\n    },\n    \"inferer\": {\n        \"_target_\": \"SimpleInferer\"\n    },\n    \"postprocessing\": {\n        \"_target_\": \"Compose\",\n        \"transforms\": [\n            {\n                \"_target_\": \"Activationsd\",\n                \"keys\": \"pred\",\n                \"sigmoid\": false\n            }\n        ]\n    },\n    \"handlers\": [\n        {\n            \"_target_\": \"CheckpointLoader\",\n            \"load_path\": \"$@model_dir + '/model.pt'\",\n            \"load_dict\": {\n                \"model\": \"@network\"\n            }\n        },\n        {\n            \"_target_\": \"StatsHandler\",\n            \"iteration_log\": false,\n            \"output_transform\": \"$lambda x: None\"\n        },\n        {\n            \"_target_\": \"ClassificationSaver\",\n            \"output_dir\": \"@output_dir\",\n            \"batch_transform\": \"$monai.handlers.from_engine(['image_meta_dict'])\",\n            \"output_transform\": \"$monai.handlers.from_engine(['pred'])\"\n        }\n    ],\n    \"evaluator\": {\n        \"_target_\": \"SupervisedEvaluator\",\n        \"device\": \"@device\",\n        \"val_data_loader\": \"@dataloader\",\n        \"network\": \"@network\",\n        \"inferer\": \"@inferer\",\n        \"postprocessing\": \"@postprocessing\",\n        \"val_handlers\": \"@handlers\",\n        \"amp\": true\n    },\n    \"evaluating\": [\n        \"$setattr(torch.backends.cudnn, 'benchmark', True)\",\n        \"$@evaluator.run()\"\n    ]\n}","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:48:04.924887Z","iopub.execute_input":"2024-01-25T11:48:04.925311Z","iopub.status.idle":"2024-01-25T11:48:04.939197Z","shell.execute_reply.started":"2024-01-25T11:48:04.925285Z","shell.execute_reply":"2024-01-25T11:48:04.938272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd /kaggle/working/breast_density_classification/scripts","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:48:04.94024Z","iopub.execute_input":"2024-01-25T11:48:04.94051Z","iopub.status.idle":"2024-01-25T11:48:04.952936Z","shell.execute_reply.started":"2024-01-25T11:48:04.940481Z","shell.execute_reply":"2024-01-25T11:48:04.952053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parser = ConfigParser()\n\nparser.read_config(CONFIG_FILE)\n\ndata = parser.get_parsed_content(\"data\")\ndevice = parser.get_parsed_content(\"device\")","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:48:04.954091Z","iopub.execute_input":"2024-01-25T11:48:04.954417Z","iopub.status.idle":"2024-01-25T11:48:05.039999Z","shell.execute_reply.started":"2024-01-25T11:48:04.954385Z","shell.execute_reply":"2024-01-25T11:48:05.039207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd /kaggle/working/breast_density_classification/","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:48:05.041118Z","iopub.execute_input":"2024-01-25T11:48:05.041402Z","iopub.status.idle":"2024-01-25T11:48:05.046907Z","shell.execute_reply.started":"2024-01-25T11:48:05.041378Z","shell.execute_reply":"2024-01-25T11:48:05.046042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npreds = predict(parser)","metadata":{"execution":{"iopub.status.busy":"2024-01-25T11:48:05.048135Z","iopub.execute_input":"2024-01-25T11:48:05.048383Z","iopub.status.idle":"2024-01-25T12:43:05.102612Z","shell.execute_reply.started":"2024-01-25T11:48:05.048361Z","shell.execute_reply":"2024-01-25T12:43:05.101656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('/kaggle/working/preds_density.npy', preds)","metadata":{"execution":{"iopub.status.busy":"2024-01-25T12:43:05.103942Z","iopub.execute_input":"2024-01-25T12:43:05.104278Z","iopub.status.idle":"2024-01-25T12:43:05.109808Z","shell.execute_reply.started":"2024-01-25T12:43:05.104247Z","shell.execute_reply":"2024-01-25T12:43:05.108939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.savetxt('/kaggle/working/preds_density.csv',preds)","metadata":{"execution":{"iopub.status.busy":"2024-01-25T12:43:05.110931Z","iopub.execute_input":"2024-01-25T12:43:05.111223Z","iopub.status.idle":"2024-01-25T12:43:05.256557Z","shell.execute_reply.started":"2024-01-25T12:43:05.111199Z","shell.execute_reply":"2024-01-25T12:43:05.25566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"RESULTS","metadata":{}},{"cell_type":"code","source":"CLASSES = [\"A\", \"B\", \"C\", \"D\"]","metadata":{"execution":{"iopub.status.busy":"2024-01-25T12:43:05.257852Z","iopub.execute_input":"2024-01-25T12:43:05.258545Z","iopub.status.idle":"2024-01-25T12:43:05.262872Z","shell.execute_reply.started":"2024-01-25T12:43:05.25851Z","shell.execute_reply":"2024-01-25T12:43:05.261921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#remove rows with missing data in the density column so we can create a confusion matrix \ndef remove_rows_with_missing_data(csv_file_path, column_name, new_csv_file_path):\n    # Read the CSV file into a pandas DataFrame\n    df = pd.read_csv(csv_file_path)\n\n    # Drop rows where the specified column has missing values\n    df = df.dropna(subset=[column_name])\n\n    # Save the modified DataFrame to a new CSV file\n    df.to_csv(new_csv_file_path, index=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-25T10:54:33.097604Z","iopub.execute_input":"2024-01-25T10:54:33.098063Z","iopub.status.idle":"2024-01-25T10:54:33.108285Z","shell.execute_reply.started":"2024-01-25T10:54:33.098022Z","shell.execute_reply":"2024-01-25T10:54:33.107408Z"}}},{"cell_type":"markdown","source":"original_csv_file = \"/kaggle/input/rsna-breast-cancer-detection/train.csv\"\ncolumn_name = \"density\"\nnew_csv_file = \"/kaggle/working/predswithoutmissing.csv\"\n\nremove_rows_with_missing_data(original_csv_file, column_name, new_csv_file)","metadata":{"execution":{"iopub.status.busy":"2024-01-25T10:54:33.118329Z","iopub.execute_input":"2024-01-25T10:54:33.11873Z","iopub.status.idle":"2024-01-25T10:54:33.434854Z","shell.execute_reply.started":"2024-01-25T10:54:33.118699Z","shell.execute_reply":"2024-01-25T10:54:33.433956Z"}}},{"cell_type":"code","source":"#create a list of image IDs \ndata = json.load(open('configs/rsna_train.json', 'r'))['Test']\n#splits at _, takes last element and removes .png \nimg_ids = [int(d['image'].split('_')[-1][:-4]) for d in data] \n\n#create dataframe for confusion matrix \ndf = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n#include only the rows corresponding to the specified image IDs in the img_ids list\ndf = df.set_index('image_id').loc[img_ids]\n\n#add preds column\ndf['pred'] = [CLASSES[p] for p in preds.argmax(-1)]\n#drop all other columns \ndf = df[['density', 'pred']].dropna(axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-01-25T12:43:05.26402Z","iopub.execute_input":"2024-01-25T12:43:05.264348Z","iopub.status.idle":"2024-01-25T12:43:05.487078Z","shell.execute_reply.started":"2024-01-25T12:43:05.264315Z","shell.execute_reply":"2024-01-25T12:43:05.486285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ny_test=df['density']\ny_pred=df['pred']\n# Create a confusion matrix\ncm = confusion_matrix(y_test, y_pred)\n\n# Calculate accuracy\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Accuracy: {accuracy:.2f}\")\n\n# Plot the confusion matrix\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\", cbar=False,\n            xticklabels=[\"A\", \"B\",\"C\",\"D\"],\n            yticklabels=[\"A\", \"B\",\"C\",\"D\"])\n\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-25T12:43:05.488231Z","iopub.execute_input":"2024-01-25T12:43:05.488521Z","iopub.status.idle":"2024-01-25T12:43:06.269181Z","shell.execute_reply.started":"2024-01-25T12:43:05.488497Z","shell.execute_reply":"2024-01-25T12:43:06.268166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}