{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! pip install -U pylibjpeg pylibjpeg-openjpeg pylibjpeg-libjpeg pydicom python-gdcm\n! pip install --upgrade pydicom","metadata":{"papermill":{"duration":22.534607,"end_time":"2023-02-13T09:01:20.916251","exception":false,"start_time":"2023-02-13T09:00:58.381644","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:11.504456Z","iopub.execute_input":"2023-06-12T02:43:11.505069Z","iopub.status.idle":"2023-06-12T02:43:43.39068Z","shell.execute_reply.started":"2023-06-12T02:43:11.504964Z","shell.execute_reply":"2023-06-12T02:43:43.389488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    import pylibjpeg\nexcept:\n    !pip install /kaggle/input/rsna-2022-whl/{pydicom-2.3.0-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"papermill":{"duration":0.030413,"end_time":"2023-02-13T09:01:20.962274","exception":false,"start_time":"2023-02-13T09:01:20.931861","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:43.394329Z","iopub.execute_input":"2023-06-12T02:43:43.394715Z","iopub.status.idle":"2023-06-12T02:43:43.407945Z","shell.execute_reply.started":"2023-06-12T02:43:43.394681Z","shell.execute_reply":"2023-06-12T02:43:43.406994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# basic libraries\nimport numpy as np\nimport pandas as pd\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nimport os\nimport cv2\nfrom sklearn.metrics import confusion_matrix\nimport random\nfrom PIL import Image\n\n# specific for medical image data\nimport pydicom\npydicom.__version__\n\n# PyTorch libraries\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nimport torchvision.transforms as transforms\nimport torchvision.models as models\nfrom torch.utils.data import Dataset","metadata":{"papermill":{"duration":3.033682,"end_time":"2023-02-13T09:01:24.010944","exception":false,"start_time":"2023-02-13T09:01:20.977262","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:43.409537Z","iopub.execute_input":"2023-06-12T02:43:43.410124Z","iopub.status.idle":"2023-06-12T02:43:46.164948Z","shell.execute_reply.started":"2023-06-12T02:43:43.410085Z","shell.execute_reply":"2023-06-12T02:43:46.163952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the path to the image data\nRSNA_2022_path = '/kaggle/input/rsna-breast-cancer-detection/train_images'","metadata":{"papermill":{"duration":0.023374,"end_time":"2023-02-13T09:01:24.079866","exception":false,"start_time":"2023-02-13T09:01:24.056492","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.167751Z","iopub.execute_input":"2023-06-12T02:43:46.168961Z","iopub.status.idle":"2023-06-12T02:43:46.174692Z","shell.execute_reply.started":"2023-06-12T02:43:46.168922Z","shell.execute_reply":"2023-06-12T02:43:46.172745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the csv data.\ndf_train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndf_train.head()","metadata":{"papermill":{"duration":0.140708,"end_time":"2023-02-13T09:01:24.235195","exception":false,"start_time":"2023-02-13T09:01:24.094487","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.176581Z","iopub.execute_input":"2023-06-12T02:43:46.17707Z","iopub.status.idle":"2023-06-12T02:43:46.324454Z","shell.execute_reply.started":"2023-06-12T02:43:46.177031Z","shell.execute_reply":"2023-06-12T02:43:46.323216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of total patients\nlen(df_train)","metadata":{"papermill":{"duration":0.024468,"end_time":"2023-02-13T09:01:24.275418","exception":false,"start_time":"2023-02-13T09:01:24.25095","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.326071Z","iopub.execute_input":"2023-06-12T02:43:46.32756Z","iopub.status.idle":"2023-06-12T02:43:46.335503Z","shell.execute_reply.started":"2023-06-12T02:43:46.327509Z","shell.execute_reply":"2023-06-12T02:43:46.334281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patients having implant\nlen(df_train[df_train['implant'] == 1])","metadata":{"papermill":{"duration":0.025413,"end_time":"2023-02-13T09:01:24.406329","exception":false,"start_time":"2023-02-13T09:01:24.380916","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.337589Z","iopub.execute_input":"2023-06-12T02:43:46.338569Z","iopub.status.idle":"2023-06-12T02:43:46.35371Z","shell.execute_reply.started":"2023-06-12T02:43:46.338526Z","shell.execute_reply":"2023-06-12T02:43:46.352832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patients without malignant cancer\nlen(df_train[df_train['cancer'] == 0])","metadata":{"papermill":{"duration":0.033745,"end_time":"2023-02-13T09:01:24.32428","exception":false,"start_time":"2023-02-13T09:01:24.290535","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.356206Z","iopub.execute_input":"2023-06-12T02:43:46.356828Z","iopub.status.idle":"2023-06-12T02:43:46.369087Z","shell.execute_reply.started":"2023-06-12T02:43:46.356792Z","shell.execute_reply":"2023-06-12T02:43:46.368091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patient who took biopsy\nlen(df_train[df_train['biopsy'] == 1])","metadata":{"papermill":{"duration":0.026251,"end_time":"2023-02-13T09:01:24.365749","exception":false,"start_time":"2023-02-13T09:01:24.339498","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.370366Z","iopub.execute_input":"2023-06-12T02:43:46.3708Z","iopub.status.idle":"2023-06-12T02:43:46.379217Z","shell.execute_reply.started":"2023-06-12T02:43:46.370762Z","shell.execute_reply":"2023-06-12T02:43:46.378109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patients having malignant cancer\nlen(df_train[df_train['cancer'] == 1])","metadata":{"papermill":{"duration":0.025413,"end_time":"2023-02-13T09:01:24.406329","exception":false,"start_time":"2023-02-13T09:01:24.380916","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.384548Z","iopub.execute_input":"2023-06-12T02:43:46.384825Z","iopub.status.idle":"2023-06-12T02:43:46.392341Z","shell.execute_reply.started":"2023-06-12T02:43:46.384799Z","shell.execute_reply":"2023-06-12T02:43:46.391386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of patients whose malignant cancer is invasive\nlen(df_train[df_train['invasive'] == 1])","metadata":{"papermill":{"duration":0.025425,"end_time":"2023-02-13T09:01:24.44691","exception":false,"start_time":"2023-02-13T09:01:24.421485","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.393747Z","iopub.execute_input":"2023-06-12T02:43:46.39462Z","iopub.status.idle":"2023-06-12T02:43:46.402539Z","shell.execute_reply.started":"2023-06-12T02:43:46.394585Z","shell.execute_reply":"2023-06-12T02:43:46.401456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the same as above\nlen(df_train[(df_train['cancer'] == 1) & (df_train['invasive'] == 1)])","metadata":{"papermill":{"duration":0.026607,"end_time":"2023-02-13T09:01:24.488965","exception":false,"start_time":"2023-02-13T09:01:24.462358","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.403929Z","iopub.execute_input":"2023-06-12T02:43:46.404637Z","iopub.status.idle":"2023-06-12T02:43:46.413566Z","shell.execute_reply.started":"2023-06-12T02:43:46.4046Z","shell.execute_reply":"2023-06-12T02:43:46.412541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Most of the cases are normal or not-malignant cancer. Thus, physicians sometimes overlook cancer.\ndata = pd.DataFrame(np.concatenate([['Total'] * len(df_train) , ['Maglignant Cancer'] *  len(df_train[df_train['cancer'] == 1]), ['Invasive Cancer'] *  len(df_train[(df_train['cancer'] == 1) & (df_train['invasive'] == 1)])]), columns = [\"class\"])\n\nsns.countplot(x = 'class', data = data)","metadata":{"papermill":{"duration":0.252709,"end_time":"2023-02-13T09:01:24.757149","exception":false,"start_time":"2023-02-13T09:01:24.50444","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.414891Z","iopub.execute_input":"2023-06-12T02:43:46.415362Z","iopub.status.idle":"2023-06-12T02:43:46.695352Z","shell.execute_reply.started":"2023-06-12T02:43:46.415326Z","shell.execute_reply":"2023-06-12T02:43:46.69433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Around 3000 patient took biopsy and malignant cancer was found from some of them.\ndata = pd.DataFrame(np.concatenate([['Biopsy'] * len(df_train[df_train['biopsy'] == 1]) , ['Malignant Cancer'] *  len(df_train[df_train['cancer'] == 1]), ['Invasive Cancer'] *  len(df_train[(df_train['cancer'] == 1) & (df_train['invasive'] == 1)])]), columns = [\"class\"])\n\nsns.countplot(x = 'class', data = data)","metadata":{"papermill":{"duration":0.196823,"end_time":"2023-02-13T09:01:24.970072","exception":false,"start_time":"2023-02-13T09:01:24.773249","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.697095Z","iopub.execute_input":"2023-06-12T02:43:46.698995Z","iopub.status.idle":"2023-06-12T02:43:46.882089Z","shell.execute_reply.started":"2023-06-12T02:43:46.698955Z","shell.execute_reply":"2023-06-12T02:43:46.88118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of not-malignant cancer cases from biopsy\nlen(df_train[(df_train['biopsy'] == 1) & (df_train['cancer'] == 0)])","metadata":{"papermill":{"duration":0.027406,"end_time":"2023-02-13T09:01:25.013858","exception":false,"start_time":"2023-02-13T09:01:24.986452","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.88345Z","iopub.execute_input":"2023-06-12T02:43:46.883771Z","iopub.status.idle":"2023-06-12T02:43:46.894041Z","shell.execute_reply.started":"2023-06-12T02:43:46.883744Z","shell.execute_reply":"2023-06-12T02:43:46.892772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the number of malignant cancer cases from biopsy\nlen(df_train[(df_train['biopsy'] == 1) & (df_train['cancer'] == 1)])","metadata":{"papermill":{"duration":0.026787,"end_time":"2023-02-13T09:01:25.056814","exception":false,"start_time":"2023-02-13T09:01:25.030027","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.896029Z","iopub.execute_input":"2023-06-12T02:43:46.896723Z","iopub.status.idle":"2023-06-12T02:43:46.904982Z","shell.execute_reply.started":"2023-06-12T02:43:46.896688Z","shell.execute_reply":"2023-06-12T02:43:46.904207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 60% of biopsy resulted in not-malignanct cancer.\ndata = pd.DataFrame(np.concatenate([['Biopsy but Not Malignant'] * len(df_train[(df_train['biopsy'] == 1) & (df_train['cancer'] == 0)]) , ['Malignant Cancer'] *  len(df_train[df_train['cancer'] == 1]), ['Invasive Cancer'] *  len(df_train[(df_train['cancer'] == 1) & (df_train['invasive'] == 1)])]), columns = [\"class\"])\n\nsns.countplot(x = 'class', data = data)","metadata":{"papermill":{"duration":0.199887,"end_time":"2023-02-13T09:01:25.27297","exception":false,"start_time":"2023-02-13T09:01:25.073083","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:46.906663Z","iopub.execute_input":"2023-06-12T02:43:46.907512Z","iopub.status.idle":"2023-06-12T02:43:47.091093Z","shell.execute_reply.started":"2023-06-12T02:43:46.907363Z","shell.execute_reply":"2023-06-12T02:43:47.090155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The not-malignant cancer cases were limited into biopsy cases. \nDF_train = df_train[df_train['biopsy'] == 1].reset_index(drop = True)\nDF_train.head()","metadata":{"papermill":{"duration":0.038103,"end_time":"2023-02-13T09:01:25.328245","exception":false,"start_time":"2023-02-13T09:01:25.290142","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:47.09298Z","iopub.execute_input":"2023-06-12T02:43:47.093745Z","iopub.status.idle":"2023-06-12T02:43:47.114152Z","shell.execute_reply.started":"2023-06-12T02:43:47.093709Z","shell.execute_reply":"2023-06-12T02:43:47.113295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The number of positive (malignant) and negative (not-malignat) cases should be the same\n# to create a balanced dataset.\nDF_train = DF_train.groupby(['cancer']).apply(lambda x: x.sample(1158, replace = True)\n                                                      ).reset_index(drop = True)\nprint('New Data Size:', DF_train.shape[0])","metadata":{"papermill":{"duration":0.034133,"end_time":"2023-02-13T09:01:25.380147","exception":false,"start_time":"2023-02-13T09:01:25.346014","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:47.115302Z","iopub.execute_input":"2023-06-12T02:43:47.115956Z","iopub.status.idle":"2023-06-12T02:43:47.134849Z","shell.execute_reply.started":"2023-06-12T02:43:47.11592Z","shell.execute_reply":"2023-06-12T02:43:47.13392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generally, invasive cancer is confirmed by biopsy, not by mammography.\n# Maybe it is also extremely difficult for AI to detect invasive cancer from mammography.\ndata = pd.DataFrame(np.concatenate([['Biopsy but Not Malignant'] * len(DF_train[(DF_train['biopsy'] == 1) & (DF_train['cancer'] == 0)]) , ['Malignant Cancer'] *  len(DF_train[DF_train['cancer'] == 1]), ['Invasive Cancer'] *  len(DF_train[(DF_train['cancer'] == 1) & (DF_train['invasive'] == 1)])]), columns = [\"class\"])\n\nsns.countplot(x = 'class', data = data)","metadata":{"papermill":{"duration":0.19691,"end_time":"2023-02-13T09:01:25.593941","exception":false,"start_time":"2023-02-13T09:01:25.397031","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:47.13648Z","iopub.execute_input":"2023-06-12T02:43:47.136846Z","iopub.status.idle":"2023-06-12T02:43:47.320128Z","shell.execute_reply.started":"2023-06-12T02:43:47.136811Z","shell.execute_reply":"2023-06-12T02:43:47.319182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dcmfnm = '/kaggle/input/rsna-breast-cancer-detection/train_images/10006/1459541791.dcm'\n\nds = pydicom.dcmread(dcmfnm, force = True)\nprint(\"Display Meta Information\\n\", ds)\n\n# Get information with keyword.\np_id = ds.PatientID\nprint(\"\\n>Patient ID=\", p_id, type(p_id))","metadata":{"papermill":{"duration":0.109334,"end_time":"2023-02-13T09:01:25.756228","exception":false,"start_time":"2023-02-13T09:01:25.646894","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:47.321482Z","iopub.execute_input":"2023-06-12T02:43:47.321924Z","iopub.status.idle":"2023-06-12T02:43:47.447297Z","shell.execute_reply.started":"2023-06-12T02:43:47.321888Z","shell.execute_reply":"2023-06-12T02:43:47.446243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This is the way to visualize a medical image.\nimg = ds.pixel_array\nplt.imshow(img, cmap = 'gray')\nplt.show()","metadata":{"papermill":{"duration":2.50666,"end_time":"2023-02-13T09:01:28.280395","exception":false,"start_time":"2023-02-13T09:01:25.773735","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:47.448606Z","iopub.execute_input":"2023-06-12T02:43:47.449208Z","iopub.status.idle":"2023-06-12T02:43:50.09113Z","shell.execute_reply.started":"2023-06-12T02:43:47.449171Z","shell.execute_reply":"2023-06-12T02:43:50.090153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img","metadata":{"papermill":{"duration":0.027203,"end_time":"2023-02-13T09:01:28.325651","exception":false,"start_time":"2023-02-13T09:01:28.298448","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:50.092887Z","iopub.execute_input":"2023-06-12T02:43:50.093602Z","iopub.status.idle":"2023-06-12T02:43:50.101604Z","shell.execute_reply.started":"2023-06-12T02:43:50.09356Z","shell.execute_reply":"2023-06-12T02:43:50.100381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The size of medical image is extremely huge.\nimg.shape","metadata":{"papermill":{"duration":0.026516,"end_time":"2023-02-13T09:01:28.369805","exception":false,"start_time":"2023-02-13T09:01:28.343289","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:50.102989Z","iopub.execute_input":"2023-06-12T02:43:50.103516Z","iopub.status.idle":"2023-06-12T02:43:50.1141Z","shell.execute_reply.started":"2023-06-12T02:43:50.103476Z","shell.execute_reply":"2023-06-12T02:43:50.113033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Take another image.\nds = pydicom.dcmread(os.path.join(RSNA_2022_path + '/' + str(DF_train.loc[0, 'patient_id']) + '/' + str(DF_train.loc[0, 'image_id']) + '.dcm'), force = True)\nimg = ds.pixel_array.astype(np.float32)\nimg","metadata":{"papermill":{"duration":0.523099,"end_time":"2023-02-13T09:01:28.910574","exception":false,"start_time":"2023-02-13T09:01:28.387475","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:50.115642Z","iopub.execute_input":"2023-06-12T02:43:50.116231Z","iopub.status.idle":"2023-06-12T02:43:50.750934Z","shell.execute_reply.started":"2023-06-12T02:43:50.116197Z","shell.execute_reply":"2023-06-12T02:43:50.749757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.shape","metadata":{"papermill":{"duration":0.031077,"end_time":"2023-02-13T09:01:28.960236","exception":false,"start_time":"2023-02-13T09:01:28.929159","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:50.752365Z","iopub.execute_input":"2023-06-12T02:43:50.752833Z","iopub.status.idle":"2023-06-12T02:43:50.76015Z","shell.execute_reply.started":"2023-06-12T02:43:50.752794Z","shell.execute_reply":"2023-06-12T02:43:50.758974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(img, cmap = 'gray')\nplt.show()","metadata":{"papermill":{"duration":0.704853,"end_time":"2023-02-13T09:01:29.713707","exception":false,"start_time":"2023-02-13T09:01:29.008854","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:50.761927Z","iopub.execute_input":"2023-06-12T02:43:50.76237Z","iopub.status.idle":"2023-06-12T02:43:51.427007Z","shell.execute_reply.started":"2023-06-12T02:43:50.762327Z","shell.execute_reply":"2023-06-12T02:43:51.426004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The size of image must be resized to reduce computational costs for machine learning.\nimg = np.resize(img, (1024, 1024))","metadata":{"papermill":{"duration":0.046846,"end_time":"2023-02-13T09:01:29.779323","exception":false,"start_time":"2023-02-13T09:01:29.732477","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:51.435737Z","iopub.execute_input":"2023-06-12T02:43:51.436692Z","iopub.status.idle":"2023-06-12T02:43:51.461107Z","shell.execute_reply.started":"2023-06-12T02:43:51.436651Z","shell.execute_reply":"2023-06-12T02:43:51.460196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# But nothing can be seen by human eyes after the resizing.\nplt.imshow(img, cmap = 'gray')\nplt.show()","metadata":{"papermill":{"duration":0.253293,"end_time":"2023-02-13T09:01:30.050805","exception":false,"start_time":"2023-02-13T09:01:29.797512","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:51.46264Z","iopub.execute_input":"2023-06-12T02:43:51.463028Z","iopub.status.idle":"2023-06-12T02:43:51.716301Z","shell.execute_reply.started":"2023-06-12T02:43:51.462991Z","shell.execute_reply":"2023-06-12T02:43:51.715298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = transforms.Compose(\n    [transforms.ToTensor(),\n     transforms.Normalize((0.5), (0.5))])","metadata":{"papermill":{"duration":0.025884,"end_time":"2023-02-13T09:01:30.132433","exception":false,"start_time":"2023-02-13T09:01:30.106549","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:51.717579Z","iopub.execute_input":"2023-06-12T02:43:51.718207Z","iopub.status.idle":"2023-06-12T02:43:51.723855Z","shell.execute_reply.started":"2023-06-12T02:43:51.718169Z","shell.execute_reply":"2023-06-12T02:43:51.722719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RSNA_Dataset(Dataset):\n    def __init__(self, img_data, img_path, transform = None):\n        self.img_path = img_path\n        self.transform = transform\n        self.img_data = img_data\n        \n    def __len__(self):\n        return len(self.img_data)\n    \n    def __getitem__(self, index):\n        img_name = os.path.join(RSNA_2022_path + '/' + str(self.img_data.loc[index, 'patient_id']) + '/' + str(self.img_data.loc[index, 'image_id']) + '.dcm')\n        ds = pydicom.dcmread(img_name, force = True) # for DICOM data\n        image = ds.pixel_array.astype(np.float32)\n        # The data augmentation process begins.\n        # Convert to PIL image.\n        image = Image.fromarray(image)\n        # random crop\n        i, j, h, w = transforms.RandomCrop.get_params(image, output_size = (256, 256))\n        image = transforms.functional.crop(image, i, j, h, w)\n        # random horizontal flipping\n        if random.random() > 0.5:\n            image = transforms.functional.hflip(image)\n        # Convert back to NumPy array\n        image = np.array(image)\n        # The data augmentation process ends. It can be cut for validation data set.\n        # resize\n        image = np.resize(image, (300, 300))\n        #image = cv2.cvtColor(image, cv2.COLOR_GRAY2RGB)\n        label = torch.tensor(self.img_data.loc[index, 'cancer'])\n        if self.transform is not None:\n            image = self.transform(image)\n        return image, label","metadata":{"execution":{"iopub.status.busy":"2023-06-12T02:43:51.725286Z","iopub.execute_input":"2023-06-12T02:43:51.725721Z","iopub.status.idle":"2023-06-12T02:43:51.740338Z","shell.execute_reply.started":"2023-06-12T02:43:51.725685Z","shell.execute_reply":"2023-06-12T02:43:51.739366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = RSNA_Dataset(DF_train, RSNA_2022_path, transform)","metadata":{"papermill":{"duration":0.025802,"end_time":"2023-02-13T09:01:30.222871","exception":false,"start_time":"2023-02-13T09:01:30.197069","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:51.741833Z","iopub.execute_input":"2023-06-12T02:43:51.742309Z","iopub.status.idle":"2023-06-12T02:43:51.754228Z","shell.execute_reply.started":"2023-06-12T02:43:51.742273Z","shell.execute_reply":"2023-06-12T02:43:51.753303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(dataset)","metadata":{"papermill":{"duration":0.026849,"end_time":"2023-02-13T09:01:30.267929","exception":false,"start_time":"2023-02-13T09:01:30.24108","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:51.755531Z","iopub.execute_input":"2023-06-12T02:43:51.755979Z","iopub.status.idle":"2023-06-12T02:43:51.766199Z","shell.execute_reply.started":"2023-06-12T02:43:51.755943Z","shell.execute_reply":"2023-06-12T02:43:51.765182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x, t = dataset[0]","metadata":{"papermill":{"duration":0.488539,"end_time":"2023-02-13T09:01:30.774638","exception":false,"start_time":"2023-02-13T09:01:30.286099","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:51.768137Z","iopub.execute_input":"2023-06-12T02:43:51.768928Z","iopub.status.idle":"2023-06-12T02:43:52.309712Z","shell.execute_reply.started":"2023-06-12T02:43:51.768891Z","shell.execute_reply":"2023-06-12T02:43:52.308675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x","metadata":{"papermill":{"duration":0.035786,"end_time":"2023-02-13T09:01:30.833129","exception":false,"start_time":"2023-02-13T09:01:30.797343","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:52.311056Z","iopub.execute_input":"2023-06-12T02:43:52.31211Z","iopub.status.idle":"2023-06-12T02:43:52.324976Z","shell.execute_reply.started":"2023-06-12T02:43:52.312068Z","shell.execute_reply":"2023-06-12T02:43:52.323759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(x), x.dtype, x.shape","metadata":{"papermill":{"duration":0.028839,"end_time":"2023-02-13T09:01:30.88166","exception":false,"start_time":"2023-02-13T09:01:30.852821","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:52.32646Z","iopub.execute_input":"2023-06-12T02:43:52.326897Z","iopub.status.idle":"2023-06-12T02:43:52.334279Z","shell.execute_reply.started":"2023-06-12T02:43:52.326857Z","shell.execute_reply":"2023-06-12T02:43:52.333163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t","metadata":{"papermill":{"duration":0.028762,"end_time":"2023-02-13T09:01:30.928714","exception":false,"start_time":"2023-02-13T09:01:30.899952","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:52.335804Z","iopub.execute_input":"2023-06-12T02:43:52.336279Z","iopub.status.idle":"2023-06-12T02:43:52.345272Z","shell.execute_reply.started":"2023-06-12T02:43:52.336217Z","shell.execute_reply":"2023-06-12T02:43:52.344033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train, Val = torch.utils.data.random_split(dataset = dataset, lengths = [1800, 516], generator = torch.Generator().manual_seed(42))","metadata":{"papermill":{"duration":0.026665,"end_time":"2023-02-13T09:01:30.973799","exception":false,"start_time":"2023-02-13T09:01:30.947134","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:52.347008Z","iopub.execute_input":"2023-06-12T02:43:52.347403Z","iopub.status.idle":"2023-06-12T02:43:52.35384Z","shell.execute_reply.started":"2023-06-12T02:43:52.347357Z","shell.execute_reply":"2023-06-12T02:43:52.352805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train), len(Val)","metadata":{"papermill":{"duration":0.028316,"end_time":"2023-02-13T09:01:31.020946","exception":false,"start_time":"2023-02-13T09:01:30.99263","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:52.355325Z","iopub.execute_input":"2023-06-12T02:43:52.35596Z","iopub.status.idle":"2023-06-12T02:43:52.365807Z","shell.execute_reply.started":"2023-06-12T02:43:52.355923Z","shell.execute_reply":"2023-06-12T02:43:52.364751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RSNA_Valset(Dataset):\n    def __init__(self, img_data, img_path, transform = None):\n        self.img_path = img_path\n        self.transform = transform\n        self.img_data = img_data\n        \n    def __len__(self):\n        return len(self.img_data)\n    \n    def __getitem__(self, index):\n        img_name = os.path.join(RSNA_2022_path + '/' + str(self.img_data.loc[index, 'patient_id']) + '/' + str(self.img_data.loc[index, 'image_id']) + '.dcm')\n        ds = pydicom.dcmread(img_name, force = True) # for DICOM data\n        image = ds.pixel_array.astype(np.float32)\n        # resize\n        image = np.resize(image, (300, 300))\n        #image = cv2.cvtColor(image, cv2.COLOR_GRAY2RGB)\n        label = torch.tensor(self.img_data.loc[index, 'cancer'])\n        if self.transform is not None:\n            image = self.transform(image)\n        return image, label","metadata":{"execution":{"iopub.status.busy":"2023-06-12T02:43:52.367666Z","iopub.execute_input":"2023-06-12T02:43:52.36808Z","iopub.status.idle":"2023-06-12T02:43:52.377999Z","shell.execute_reply.started":"2023-06-12T02:43:52.368046Z","shell.execute_reply":"2023-06-12T02:43:52.377048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_val = RSNA_Valset(DF_train, RSNA_2022_path, transform)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T02:43:52.379593Z","iopub.execute_input":"2023-06-12T02:43:52.38021Z","iopub.status.idle":"2023-06-12T02:43:52.391537Z","shell.execute_reply.started":"2023-06-12T02:43:52.380175Z","shell.execute_reply":"2023-06-12T02:43:52.390401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(dataset_val)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T02:43:52.393231Z","iopub.execute_input":"2023-06-12T02:43:52.39366Z","iopub.status.idle":"2023-06-12T02:43:52.403529Z","shell.execute_reply.started":"2023-06-12T02:43:52.393624Z","shell.execute_reply":"2023-06-12T02:43:52.40232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Train, val = torch.utils.data.random_split(dataset = dataset_val, lengths = [1800, 516], generator = torch.Generator().manual_seed(42))","metadata":{"execution":{"iopub.status.busy":"2023-06-12T02:43:52.405092Z","iopub.execute_input":"2023-06-12T02:43:52.405487Z","iopub.status.idle":"2023-06-12T02:43:52.413226Z","shell.execute_reply.started":"2023-06-12T02:43:52.405452Z","shell.execute_reply":"2023-06-12T02:43:52.412373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(Train), len(val)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T02:43:52.414701Z","iopub.execute_input":"2023-06-12T02:43:52.415549Z","iopub.status.idle":"2023-06-12T02:43:52.424567Z","shell.execute_reply.started":"2023-06-12T02:43:52.415513Z","shell.execute_reply":"2023-06-12T02:43:52.423719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**They are what we exactly need!**","metadata":{}},{"cell_type":"code","source":"# They are what we exactly need!\nlen(train), len(val)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T02:43:52.425941Z","iopub.execute_input":"2023-06-12T02:43:52.427144Z","iopub.status.idle":"2023-06-12T02:43:52.434996Z","shell.execute_reply.started":"2023-06-12T02:43:52.427107Z","shell.execute_reply":"2023-06-12T02:43:52.433928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check that the dataset is balanced with not-malignant and malignant cases.\ncan = 0\nfor i in range(len(val)):\n    x, t = val[i]\n    if t == 1:\n        can += 1\nprint(can / len(val))        ","metadata":{"papermill":{"duration":368.124207,"end_time":"2023-02-13T09:07:39.163921","exception":false,"start_time":"2023-02-13T09:01:31.039714","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:43:52.436407Z","iopub.execute_input":"2023-06-12T02:43:52.43756Z","iopub.status.idle":"2023-06-12T02:50:10.061017Z","shell.execute_reply.started":"2023-06-12T02:43:52.437525Z","shell.execute_reply":"2023-06-12T02:50:10.060015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader = torch.utils.data.DataLoader(train, batch_size = 128, shuffle = False, drop_last = True)\nval_loader = torch.utils.data.DataLoader(val, batch_size = 128)","metadata":{"papermill":{"duration":0.026605,"end_time":"2023-02-13T09:07:39.210197","exception":false,"start_time":"2023-02-13T09:07:39.183592","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:50:10.062477Z","iopub.execute_input":"2023-06-12T02:50:10.062946Z","iopub.status.idle":"2023-06-12T02:50:10.069358Z","shell.execute_reply.started":"2023-06-12T02:50:10.062906Z","shell.execute_reply":"2023-06-12T02:50:10.06828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RSNA_Dataset_Visual(Dataset):\n    def __init__(self, img_data, img_path, transform = None):\n        self.img_path = img_path\n        self.transform = transform\n        self.img_data = img_data\n        \n    def __len__(self):\n        return len(self.img_data)\n    \n    def __getitem__(self, index):\n        img_name = os.path.join(RSNA_2022_path + '/' + str(self.img_data.loc[index, 'patient_id']) + '/' + str(self.img_data.loc[index, 'image_id']) + '.dcm')\n        ds = pydicom.dcmread(img_name, force = True)\n        image = ds.pixel_array.astype(np.float32)\n        #image = np.resize(image, (300, 300)) # no risize\n        #image = cv2.cvtColor(image, cv2.COLOR_GRAY2RGB)\n        label = torch.tensor(self.img_data.loc[index, 'cancer'])\n        if self.transform is not None:\n            image = self.transform(image)\n        return image, label","metadata":{"papermill":{"duration":0.029029,"end_time":"2023-02-13T09:07:39.295699","exception":false,"start_time":"2023-02-13T09:07:39.26667","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:50:10.070942Z","iopub.execute_input":"2023-06-12T02:50:10.071351Z","iopub.status.idle":"2023-06-12T02:50:10.083706Z","shell.execute_reply.started":"2023-06-12T02:50:10.071274Z","shell.execute_reply":"2023-06-12T02:50:10.082529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_visual = RSNA_Dataset_Visual(DF_train, RSNA_2022_path, transform)","metadata":{"papermill":{"duration":0.02565,"end_time":"2023-02-13T09:07:39.339867","exception":false,"start_time":"2023-02-13T09:07:39.314217","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:50:10.085437Z","iopub.execute_input":"2023-06-12T02:50:10.085799Z","iopub.status.idle":"2023-06-12T02:50:10.096932Z","shell.execute_reply.started":"2023-06-12T02:50:10.085765Z","shell.execute_reply":"2023-06-12T02:50:10.095758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use the same random seed to create the same dataset as before except for the resizing. \ntrain_visual, val_visual = torch.utils.data.random_split(dataset = dataset_visual, lengths = [1800, 516], generator = torch.Generator().manual_seed(42))","metadata":{"papermill":{"duration":0.026751,"end_time":"2023-02-13T09:07:39.385564","exception":false,"start_time":"2023-02-13T09:07:39.358813","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:50:10.100281Z","iopub.execute_input":"2023-06-12T02:50:10.100827Z","iopub.status.idle":"2023-06-12T02:50:10.108091Z","shell.execute_reply.started":"2023-06-12T02:50:10.100799Z","shell.execute_reply":"2023-06-12T02:50:10.107051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def img_display(img):\n    img = img / 2 + 0.5 # unnormalize\n    npimg = img.numpy()\n    #npimg = np.transpose(npimg, (1, 2, 0))\n    return npimg","metadata":{"papermill":{"duration":0.025764,"end_time":"2023-02-13T09:07:39.429904","exception":false,"start_time":"2023-02-13T09:07:39.40414","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:50:10.109721Z","iopub.execute_input":"2023-06-12T02:50:10.110274Z","iopub.status.idle":"2023-06-12T02:50:10.117216Z","shell.execute_reply.started":"2023-06-12T02:50:10.110218Z","shell.execute_reply":"2023-06-12T02:50:10.115609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get some training images.\narthopod_types = {0: 'Not Malignant', 1: 'Malignant Cancer'}\n# viewing data examples used for training\nfig, axis = plt.subplots(2, 5, figsize = (15, 10))\nfor i, ax in enumerate(axis.flat):\n    with torch.no_grad():\n        image, label = train_visual[i]\n        ax.imshow(img_display(image).squeeze(0)) # add image\n        ax.set(title = f\"{arthopod_types[label.item()]}\") # add label","metadata":{"papermill":{"duration":9.015798,"end_time":"2023-02-13T09:07:48.464326","exception":false,"start_time":"2023-02-13T09:07:39.448528","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:50:10.119128Z","iopub.execute_input":"2023-06-12T02:50:10.119522Z","iopub.status.idle":"2023-06-12T02:50:23.451081Z","shell.execute_reply.started":"2023-06-12T02:50:10.119484Z","shell.execute_reply":"2023-06-12T02:50:23.45018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Net(nn.Module):\n    def __init__(self):\n        super(Net, self).__init__()\n        # 1 input image channel, 16 output channels, 3x3 square convolution kernel\n        self.conv1 = nn.Conv2d(1, 16, kernel_size = 3,stride = 2,padding = 1)\n        self.conv2 = nn.Conv2d(16, 32, kernel_size = 3,stride = 2, padding = 1)\n        self.conv3 = nn.Conv2d(32, 64, kernel_size = 3,stride = 2, padding = 1)\n        self.conv4 = nn.Conv2d(64, 64, kernel_size = 3,stride = 2, padding = 1)\n        self.pool = nn.MaxPool2d(2, 2)\n        self.dropout = nn.Dropout2d(0.4)\n        self.batchnorm1 = nn.BatchNorm2d(16)\n        self.batchnorm2 = nn.BatchNorm2d(32)\n        self.batchnorm3 = nn.BatchNorm2d(64)\n        self.fc1 = nn.Linear(64 * 5 * 5, 512 )\n        self.fc2 = nn.Linear(512, 256)\n        self.fc3 = nn.Linear(256, 1)\n        \n    def forward(self, x):\n        x = self.batchnorm1(F.relu(self.conv1(x)))\n        x = self.batchnorm2(F.relu(self.conv2(x)))\n        x = self.dropout(self.batchnorm2(self.pool(x)))\n        x = self.batchnorm3(self.pool(F.relu(self.conv3(x))))\n        x = self.dropout(self.conv4(x))\n        x = x.view(-1, 64 * 5 * 5) # Flatten layer\n        x = self.dropout(self.fc1(x))\n        x = self.dropout(self.fc2(x))\n        x = F.logsigmoid(self.fc3(x))\n        return x","metadata":{"papermill":{"duration":0.056163,"end_time":"2023-02-13T09:07:48.58822","exception":false,"start_time":"2023-02-13T09:07:48.532057","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:50:23.452109Z","iopub.execute_input":"2023-06-12T02:50:23.452475Z","iopub.status.idle":"2023-06-12T02:50:23.470627Z","shell.execute_reply.started":"2023-06-12T02:50:23.452438Z","shell.execute_reply":"2023-06-12T02:50:23.469466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model = Net() # On CPU\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nmodel = Net().to(device) # On GPU\nprint(model)","metadata":{"papermill":{"duration":0.107934,"end_time":"2023-02-13T09:07:48.726673","exception":false,"start_time":"2023-02-13T09:07:48.618739","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:50:23.47206Z","iopub.execute_input":"2023-06-12T02:50:23.47262Z","iopub.status.idle":"2023-06-12T02:50:26.494306Z","shell.execute_reply.started":"2023-06-12T02:50:23.472582Z","shell.execute_reply":"2023-06-12T02:50:26.493087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"criterion = nn.BCELoss()\noptimizer = optim.Adam(model.parameters(), lr = 0.001)\nscheduler = torch.optim.lr_scheduler.LinearLR(optimizer, start_factor = 1, end_factor = 0.1, total_iters = 8)","metadata":{"papermill":{"duration":0.030542,"end_time":"2023-02-13T09:07:48.824331","exception":false,"start_time":"2023-02-13T09:07:48.793789","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:50:26.495999Z","iopub.execute_input":"2023-06-12T02:50:26.497013Z","iopub.status.idle":"2023-06-12T02:50:26.503945Z","shell.execute_reply.started":"2023-06-12T02:50:26.496964Z","shell.execute_reply":"2023-06-12T02:50:26.502619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_epochs = 6\nvalid_loss_min = np.Inf\nval_loss = []\nval_acc = []\ntrain_loss = []\ntrain_acc = []\ntotal_step = len(train_loader)\n# training\nfor epoch in range(1, n_epochs + 1):\n    running_loss = 0.0\n    scheduler.step(epoch)\n    correct = 0\n    total = 0\n    print(f'Epoch {epoch}\\n')\n    for batch_idx, (data_, target_) in enumerate(train_loader):\n        data_, target_ = data_.to(device), target_.to(device) # on GPU\n        # zero the parameter gradients\n        optimizer.zero_grad()\n        # forward + backward + optimize\n        outputs = model(data_)\n        pred = torch.sigmoid(outputs)\n        target = target_.unsqueeze(1).float()\n        loss = criterion(pred, target)\n        loss.backward()\n        optimizer.step()\n        # print statistics\n        running_loss += loss.item()\n        pred = pred > 0.40 # normally 0.5\n        accuracy = (target == pred).sum().item() / target.size(0)\n\n        if (batch_idx) % 3 == 0:\n            print ('Epoch [{}/{}], Step [{}/{}], Loss: {:.4f}' \n                   .format(epoch, n_epochs, batch_idx, total_step, loss.item()))\n    train_acc.append(100 * accuracy)\n    train_loss.append(running_loss / total_step)\n    print(f'\\ntrain loss: {np.mean(train_loss):.4f}, train acc: {(100 * accuracy):.4f}')\n    batch_loss = 0\n    total_t = 0\n    correct_t = 0\n# validation\n    with torch.no_grad():\n        model.eval()\n        for data_t, target_t in (val_loader):\n            data_t, target_t = data_t.to(device), target_t.to(device) # on GPU\n            outputs_t = model(data_t)\n            pred_t = torch.sigmoid(outputs_t)\n            target_t = target_t.unsqueeze(1).float()\n            loss_t = criterion(pred_t, target_t)\n            batch_loss += loss_t.item()\n            pred_t = pred_t > 0.40 # normally 0.5\n            accuracy_t = (target_t == pred_t).sum().item() / target_t.size(0)\n        val_acc.append(100 * accuracy_t)\n        val_loss.append(batch_loss / len(val_loader))\n        network_learned = batch_loss < valid_loss_min\n        print(f'validation loss: {np.mean(val_loss):.4f}, validation acc: {(100 * accuracy_t):.4f}\\n')\n        # Saving the best weight. \n        if network_learned:\n            valid_loss_min = batch_loss\n            torch.save(model.state_dict(), 'cancer_classification.pt')\n            print('Detected network improvement, saving current model')\n    scheduler.step()\n    model.train()","metadata":{"papermill":{"duration":8611.131065,"end_time":"2023-02-13T11:31:19.97735","exception":false,"start_time":"2023-02-13T09:07:48.846285","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T02:50:26.505462Z","iopub.execute_input":"2023-06-12T02:50:26.505884Z","iopub.status.idle":"2023-06-12T03:43:59.146466Z","shell.execute_reply.started":"2023-06-12T02:50:26.505848Z","shell.execute_reply":"2023-06-12T03:43:59.145324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize = (20, 10))\nplt.title(\"Train - Validation Accuracy\")\nplt.plot(train_acc, label = 'train')\nplt.plot(val_acc, label = 'validation')\nplt.xlabel('num_epochs', fontsize = 12)\nplt.ylabel('accuracy', fontsize = 12)\nplt.legend(loc = 'best')","metadata":{"papermill":{"duration":0.28155,"end_time":"2023-02-13T11:31:20.331883","exception":false,"start_time":"2023-02-13T11:31:20.050333","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:43:59.15033Z","iopub.execute_input":"2023-06-12T03:43:59.15253Z","iopub.status.idle":"2023-06-12T03:43:59.456146Z","shell.execute_reply.started":"2023-06-12T03:43:59.152499Z","shell.execute_reply":"2023-06-12T03:43:59.455153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize = (20, 10))\nplt.title(\"Train - Validation Loss\")\nplt.plot(train_loss, label = 'train')\nplt.plot(val_loss, label = 'validation')\nplt.xlabel('num_epochs', fontsize = 12)\nplt.ylabel('loss', fontsize = 12)\nplt.legend(loc = 'best')","metadata":{"papermill":{"duration":0.26598,"end_time":"2023-02-13T11:31:20.623086","exception":false,"start_time":"2023-02-13T11:31:20.357106","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:43:59.457796Z","iopub.execute_input":"2023-06-12T03:43:59.458469Z","iopub.status.idle":"2023-06-12T03:43:59.722598Z","shell.execute_reply.started":"2023-06-12T03:43:59.45843Z","shell.execute_reply":"2023-06-12T03:43:59.721295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# importing trained network with better loss of validation\nmodel.load_state_dict(torch.load('cancer_classification.pt'))","metadata":{"papermill":{"duration":0.042298,"end_time":"2023-02-13T11:31:20.69137","exception":false,"start_time":"2023-02-13T11:31:20.649072","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:43:59.725431Z","iopub.execute_input":"2023-06-12T03:43:59.726215Z","iopub.status.idle":"2023-06-12T03:43:59.775149Z","shell.execute_reply.started":"2023-06-12T03:43:59.726168Z","shell.execute_reply":"2023-06-12T03:43:59.774141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \"True\" and \"False\" mean that the prediction by the model is correct or wrong, respectively.\nmodel = model.cpu() # CPU\ndataiter = iter(val_loader)\nimages, labels = dataiter.next()\narthopod_types = {0: 'Not Malignant', 1: 'Malignant Cancer'}\n# viewing data examples used for training\nfig, axis = plt.subplots(2, 5, figsize = (15, 10))\nfor i, ax in enumerate(axis.flat):\n    with torch.no_grad():\n        model.eval()\n        image_visual, label_visual = val_visual[i]\n        image, label = images[i], labels[i]\n        ax.imshow(img_display(image_visual.squeeze(0))) # add image\n        image_tensor = image.unsqueeze_(0)\n        output_ = model(image_tensor)\n        output_ = torch.sigmoid(output_)\n        if output_.item() > 0.40: # normally 0.5\n            k = (label.item() == 1)\n        else:\n            k = (label.item() == 0)\n        ax.set_title(str(arthopod_types[label.item()]) + \":\" + str(k)) # add label","metadata":{"papermill":{"duration":85.178115,"end_time":"2023-02-13T11:32:45.895386","exception":false,"start_time":"2023-02-13T11:31:20.717271","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:43:59.77644Z","iopub.execute_input":"2023-06-12T03:43:59.776784Z","iopub.status.idle":"2023-06-12T03:45:32.392999Z","shell.execute_reply.started":"2023-06-12T03:43:59.776748Z","shell.execute_reply":"2023-06-12T03:45:32.392044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = model.cpu() # CPU\nmodel.eval()\ny_pred = []\ny_true = []\n\n# Iterate over test data.\nwith torch.no_grad():\n    for inputs, labels in val_loader:\n            outputs = model(inputs) # Feed Network.\n            outputs = torch.sigmoid(outputs)\n\n            for i in range(len(outputs)):\n                output = outputs[i]            \n                if output.item() > 0.40: # normally 0.5\n                    y_pred.append(int(1)) # Save prediction.\n                else:\n                    y_pred.append(int(0)) # Save prediction.\n\n            labels = labels.data.cpu().numpy()\n            y_true.extend(labels) # Save truth.\n\n# constant for classes\nclasses = ('Not Malignant', 'Malignant Cancer')","metadata":{"papermill":{"duration":319.676138,"end_time":"2023-02-13T11:38:05.601677","exception":false,"start_time":"2023-02-13T11:32:45.925539","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:45:32.394502Z","iopub.execute_input":"2023-06-12T03:45:32.395564Z","iopub.status.idle":"2023-06-12T03:51:14.706387Z","shell.execute_reply.started":"2023-06-12T03:45:32.395524Z","shell.execute_reply":"2023-06-12T03:51:14.705314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build confusion matrix.\ncm = confusion_matrix(y_true, y_pred)","metadata":{"papermill":{"duration":0.040908,"end_time":"2023-02-13T11:38:05.67339","exception":false,"start_time":"2023-02-13T11:38:05.632482","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:51:14.707951Z","iopub.execute_input":"2023-06-12T03:51:14.708369Z","iopub.status.idle":"2023-06-12T03:51:14.719098Z","shell.execute_reply.started":"2023-06-12T03:51:14.708325Z","shell.execute_reply":"2023-06-12T03:51:14.718048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_name = {'Not Malignant': 0, 'Malignant Cancer': 1}\nprint('Confusion Matrix')\nplt.figure(figsize = (10, 10))\n_ = sns.heatmap(cm.T, annot = True, fmt = 'd', cbar = True, square = True, xticklabels = target_name.keys(),\n             yticklabels = target_name.keys())\nplt.xlabel('Truth')\nplt.ylabel('Predicted')","metadata":{"papermill":{"duration":0.273102,"end_time":"2023-02-13T11:38:05.975466","exception":false,"start_time":"2023-02-13T11:38:05.702364","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:51:14.720588Z","iopub.execute_input":"2023-06-12T03:51:14.721298Z","iopub.status.idle":"2023-06-12T03:51:14.97926Z","shell.execute_reply.started":"2023-06-12T03:51:14.721241Z","shell.execute_reply":"2023-06-12T03:51:14.978183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RSNA_test_path = '/kaggle/input/rsna-breast-cancer-detection/test_images'","metadata":{"papermill":{"duration":0.036539,"end_time":"2023-02-13T11:38:06.100689","exception":false,"start_time":"2023-02-13T11:38:06.06415","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:51:14.981224Z","iopub.execute_input":"2023-06-12T03:51:14.98222Z","iopub.status.idle":"2023-06-12T03:51:14.986947Z","shell.execute_reply.started":"2023-06-12T03:51:14.982182Z","shell.execute_reply":"2023-06-12T03:51:14.985739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ndf_test.head()","metadata":{"papermill":{"duration":0.061541,"end_time":"2023-02-13T11:38:06.191505","exception":false,"start_time":"2023-02-13T11:38:06.129964","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:51:14.988342Z","iopub.execute_input":"2023-06-12T03:51:14.988772Z","iopub.status.idle":"2023-06-12T03:51:15.030893Z","shell.execute_reply.started":"2023-06-12T03:51:14.988728Z","shell.execute_reply":"2023-06-12T03:51:15.029988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RSNA_Testset(Dataset):\n    def __init__(self, img_data, img_path, transform = None):\n        self.img_path = img_path\n        self.transform = transform\n        self.img_data = img_data\n        \n    def __len__(self):\n        return len(self.img_data)\n    \n    def __getitem__(self, index):\n        img_name = os.path.join(RSNA_test_path + '/' + str(self.img_data.loc[index, 'patient_id']) + '/' + str(self.img_data.loc[index, 'image_id']) + '.dcm')\n        ds = pydicom.dcmread(img_name, force = True)\n        image = ds.pixel_array.astype(np.float32)\n        image = np.resize(image, (300, 300))\n        #image = cv2.cvtColor(image, cv2.COLOR_GRAY2RGB)\n        #label = torch.tensor(self.img_data.loc[index, 'cancer'])\n        if self.transform is not None:\n            image = self.transform(image)\n        return image#, label","metadata":{"papermill":{"duration":0.040559,"end_time":"2023-02-13T11:38:06.261478","exception":false,"start_time":"2023-02-13T11:38:06.220919","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:51:15.032136Z","iopub.execute_input":"2023-06-12T03:51:15.032724Z","iopub.status.idle":"2023-06-12T03:51:15.042404Z","shell.execute_reply.started":"2023-06-12T03:51:15.032686Z","shell.execute_reply":"2023-06-12T03:51:15.041392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testset = RSNA_Testset(df_test, RSNA_test_path, transform)","metadata":{"papermill":{"duration":0.036619,"end_time":"2023-02-13T11:38:06.327251","exception":false,"start_time":"2023-02-13T11:38:06.290632","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:51:15.043769Z","iopub.execute_input":"2023-06-12T03:51:15.044224Z","iopub.status.idle":"2023-06-12T03:51:15.056955Z","shell.execute_reply.started":"2023-06-12T03:51:15.044189Z","shell.execute_reply":"2023-06-12T03:51:15.055922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Without data loader","metadata":{}},{"cell_type":"code","source":"model = model.to(device) # On GPU\nmodel.eval()\ncancer = []\nwith torch.no_grad():\n    for i in range(len(testset)):\n        out = testset[i].unsqueeze_(0)\n        out = out.to(device)\n        out = model(out)\n        out = torch.sigmoid(out)\n        cancer.append(out.item())\ncancer","metadata":{"papermill":{"duration":2.345762,"end_time":"2023-02-13T11:38:08.702254","exception":false,"start_time":"2023-02-13T11:38:06.356492","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:51:15.058511Z","iopub.execute_input":"2023-06-12T03:51:15.058838Z","iopub.status.idle":"2023-06-12T03:51:17.406101Z","shell.execute_reply.started":"2023-06-12T03:51:15.058812Z","shell.execute_reply":"2023-06-12T03:51:17.405085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## With data loader","metadata":{}},{"cell_type":"code","source":"# Load the test set.\ntestset = RSNA_Testset(df_test, RSNA_test_path, transform)\n\n# Create a DataLoader for the test set.\ntestloader = torch.utils.data.DataLoader(testset, batch_size = 32, shuffle = False)\n\n# Move the model to the GPU (if available).\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nmodel = model.to(device)\n\n# Put the model in evaluation mode.\nmodel.eval()\n\n# Make predictions on the test set.\ncancer = []\nwith torch.no_grad():\n    for images in testloader:\n        # Move the batch of images to the GPU (if available).\n        images = images.to(device)\n\n        # Pass the batch of images through the model to get the predicted cancer probabilities.\n        out = model(images)\n\n        # Apply a sigmoid function to get a probability between 0 and 1.\n        out = torch.sigmoid(out)\n\n        # Add the predicted probabilities to the list\n        cancer.extend(out.cpu().numpy())\n        \n        cancer_probs = np.concatenate(cancer)\n        \n        cancer = [x[0] for x in cancer]\n\n# Print the predicted probabilities.\nprint(cancer)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T03:51:17.407702Z","iopub.execute_input":"2023-06-12T03:51:17.408078Z","iopub.status.idle":"2023-06-12T03:51:20.20768Z","shell.execute_reply.started":"2023-06-12T03:51:17.40804Z","shell.execute_reply":"2023-06-12T03:51:20.206607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['cancer'] = cancer\ndf_test","metadata":{"papermill":{"duration":0.045937,"end_time":"2023-02-13T11:38:08.778934","exception":false,"start_time":"2023-02-13T11:38:08.732997","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:51:20.209294Z","iopub.execute_input":"2023-06-12T03:51:20.209658Z","iopub.status.idle":"2023-06-12T03:51:20.224889Z","shell.execute_reply.started":"2023-06-12T03:51:20.209619Z","shell.execute_reply":"2023-06-12T03:51:20.223711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = df_test.loc[:, 'prediction_id':'cancer']\nsubmission","metadata":{"papermill":{"duration":0.051298,"end_time":"2023-02-13T11:38:08.860074","exception":false,"start_time":"2023-02-13T11:38:08.808776","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:51:20.226503Z","iopub.execute_input":"2023-06-12T03:51:20.227187Z","iopub.status.idle":"2023-06-12T03:51:20.251481Z","shell.execute_reply.started":"2023-06-12T03:51:20.227148Z","shell.execute_reply":"2023-06-12T03:51:20.250315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Choose mean, max, min, or something else.\nsubmission = submission.groupby('prediction_id').mean().reset_index()\nsubmission","metadata":{"papermill":{"duration":0.04781,"end_time":"2023-02-13T11:38:08.937687","exception":false,"start_time":"2023-02-13T11:38:08.889877","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:51:20.253398Z","iopub.execute_input":"2023-06-12T03:51:20.253929Z","iopub.status.idle":"2023-06-12T03:51:20.274999Z","shell.execute_reply.started":"2023-06-12T03:51:20.253746Z","shell.execute_reply":"2023-06-12T03:51:20.273915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(data = {'prediction_id': df_test['prediction_id'], 'cancer': cancer})\nsubmission","metadata":{"papermill":{"duration":0.043929,"end_time":"2023-02-13T11:38:09.012057","exception":false,"start_time":"2023-02-13T11:38:08.968128","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:51:20.276412Z","iopub.execute_input":"2023-06-12T03:51:20.277189Z","iopub.status.idle":"2023-06-12T03:51:20.288529Z","shell.execute_reply.started":"2023-06-12T03:51:20.277154Z","shell.execute_reply":"2023-06-12T03:51:20.287409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Attention\nI prefer max() to mean() or min(), because it is possible that **cancer can be observed from one view (high score) and cannot be observed from the other view (low score)**. If cancer is positive from one view, this case should be positive regardless of the other view.","metadata":{}},{"cell_type":"code","source":"submission = submission.groupby('prediction_id').max().reset_index()\nsubmission","metadata":{"papermill":{"duration":0.048208,"end_time":"2023-02-13T11:38:09.091172","exception":false,"start_time":"2023-02-13T11:38:09.042964","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:51:20.290192Z","iopub.execute_input":"2023-06-12T03:51:20.290591Z","iopub.status.idle":"2023-06-12T03:51:20.308522Z","shell.execute_reply.started":"2023-06-12T03:51:20.290557Z","shell.execute_reply":"2023-06-12T03:51:20.30751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index = False)","metadata":{"papermill":{"duration":0.043742,"end_time":"2023-02-13T11:38:09.164927","exception":false,"start_time":"2023-02-13T11:38:09.121185","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-12T03:51:20.310187Z","iopub.execute_input":"2023-06-12T03:51:20.310596Z","iopub.status.idle":"2023-06-12T03:51:20.32098Z","shell.execute_reply.started":"2023-06-12T03:51:20.310561Z","shell.execute_reply":"2023-06-12T03:51:20.320284Z"},"trusted":true},"execution_count":null,"outputs":[]}]}