{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":4619805,"sourceType":"datasetVersion","datasetId":2688675},{"sourceId":4696088,"sourceType":"datasetVersion","datasetId":2687741},{"sourceId":4976318,"sourceType":"datasetVersion","datasetId":2693468}],"dockerImageVersionId":30302,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### ROI Extractor \n\nDuring the visual data analysis I noticed that there is a large variation in the arrangement of the object in the images. In addition, some objects occupy only a small part of the image. By converting (resizing) without ROI extraction we have a very inefficient use of the reduced image. Most of the picture is blank.\n\n**Solution:**\n- annotate data - I annotated about 500 images in a human in the loop technique (3 models were created - I started from 300 images and ended up about 500) \n- train object detector - I used yolov5 (small - balance between accuracy and speed) \n\n**Result on train DS:**\n* 54601 images processed successfully \n* 105 images - detection failed\n\nModel performance (on my validation DS):\n* mAP@50 -> 0.995      \n* mAP50-95 -> 0.914\n\n<div class=\"alert alert-warning\">If you are interested in:\n    <ul>\n        <li>Annotation dataset</li>\n        <li>yolov5 training notebook</li>\n    </ul>\n    <p>&nbsp;</p>\nLet me know in comment. I will provide it as well in separate notebooks.</div>\n\n**Next steps:**\n* generate ROI based dataset for training (768 pix) -> is available here: https://www.kaggle.com/datasets/remekkinas/rsna-breast-cancer-detection-poi-images\n* use ROI extractor in inference part ","metadata":{}},{"cell_type":"code","source":"%%capture \n\n# Clone yolov5 repository\n!git clone https://github.com/ultralytics/yolov5","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:45:09.392866Z","iopub.execute_input":"2024-01-30T09:45:09.393351Z","iopub.status.idle":"2024-01-30T09:45:12.082959Z","shell.execute_reply.started":"2024-01-30T09:45:09.393257Z","shell.execute_reply":"2024-01-30T09:45:12.081469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport glob\nimport random\nimport cv2\nimport matplotlib.pyplot as plt","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-01-30T15:57:20.786391Z","iopub.execute_input":"2024-01-30T15:57:20.787019Z","iopub.status.idle":"2024-01-30T15:57:20.797678Z","shell.execute_reply.started":"2024-01-30T15:57:20.786978Z","shell.execute_reply":"2024-01-30T15:57:20.796595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install tensorflow --upgrade","metadata":{"execution":{"iopub.status.busy":"2024-01-30T15:53:48.555298Z","iopub.execute_input":"2024-01-30T15:53:48.555642Z","iopub.status.idle":"2024-01-30T15:54:00.112165Z","shell.execute_reply.started":"2024-01-30T15:53:48.555604Z","shell.execute_reply":"2024-01-30T15:54:00.110913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TensorFlow libraries\nimport tensorflow as tf\nfrom tensorflow.keras.applications.resnet_v2 import ResNet50V2\nfrom tensorflow import keras\n# from tensorflow.keras.applications.ConvNeXtSmall import ConvNeXtSmall\nfrom tensorflow.keras.optimizers import RMSprop\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\n\n# basic libraries\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport os\n\nimport glob\nfrom glob import glob","metadata":{"execution":{"iopub.status.busy":"2024-01-30T15:57:29.377663Z","iopub.execute_input":"2024-01-30T15:57:29.378484Z","iopub.status.idle":"2024-01-30T15:57:35.476473Z","shell.execute_reply.started":"2024-01-30T15:57:29.378448Z","shell.execute_reply":"2024-01-30T15:57:35.475642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load trained model\nmodel = torch.hub.load('./yolov5', 'custom', path='/kaggle/input/rsna-breast-cancer-detection-roi-model/rsna-roi-003.pt', source='local')","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:51:31.317642Z","iopub.execute_input":"2024-01-30T09:51:31.31877Z","iopub.status.idle":"2024-01-30T09:52:03.627304Z","shell.execute_reply.started":"2024-01-30T09:51:31.318722Z","shell.execute_reply":"2024-01-30T09:52:03.62614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the path to the image data\nRSNA_512_path = '/kaggle/input/rsna-breast-cancer-512-pngs'","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:57:29.182787Z","iopub.execute_input":"2024-01-30T09:57:29.18325Z","iopub.status.idle":"2024-01-30T09:57:29.188747Z","shell.execute_reply.started":"2024-01-30T09:57:29.183212Z","shell.execute_reply":"2024-01-30T09:57:29.187509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the csv data.\ndf_train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:57:31.927506Z","iopub.execute_input":"2024-01-30T09:57:31.927936Z","iopub.status.idle":"2024-01-30T09:57:32.036304Z","shell.execute_reply.started":"2024-01-30T09:57:31.927893Z","shell.execute_reply":"2024-01-30T09:57:32.035156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Most of the cases are normal or not-malignant cancer. Thus, physicians sometimes overlook cancer.\ndata = pd.DataFrame(np.concatenate([['Total'] * len(df_train) , ['Maglignant Cancer'] *  len(df_train[df_train['cancer'] == 1]), ['Invasive Cancer'] *  len(df_train[(df_train['cancer'] == 1) & (df_train['invasive'] == 1)])]), columns = [\"class\"])\n\nsns.countplot(x = 'class', data = data)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T15:57:01.110587Z","iopub.execute_input":"2024-01-30T15:57:01.11101Z","iopub.status.idle":"2024-01-30T15:57:01.209714Z","shell.execute_reply.started":"2024-01-30T15:57:01.110928Z","shell.execute_reply":"2024-01-30T15:57:01.208288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The not-malignant cancer cases were limited into biopsy cases.\nDF_train = df_train[df_train['biopsy'] == 1].reset_index(drop = True)\nDF_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:57:43.303588Z","iopub.execute_input":"2024-01-30T09:57:43.304016Z","iopub.status.idle":"2024-01-30T09:57:43.328945Z","shell.execute_reply.started":"2024-01-30T09:57:43.30398Z","shell.execute_reply":"2024-01-30T09:57:43.327665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The number of positive (malignant) and negative (not-malignat) cases should be the same\n# to create a balanced dataset.\nDF_train = DF_train.groupby(['cancer']).apply(lambda x: x.sample(1158, replace = True)\n                                                      ).reset_index(drop = True)\nprint('New Data Size:', DF_train.shape[0])","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:57:47.196871Z","iopub.execute_input":"2024-01-30T09:57:47.19774Z","iopub.status.idle":"2024-01-30T09:57:47.460831Z","shell.execute_reply.started":"2024-01-30T09:57:47.197703Z","shell.execute_reply":"2024-01-30T09:57:47.459632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generally, invasive cancer is confirmed by biopsy, not by mammography.\n# Maybe it is also extremely difficult for AI to detect invasive cancer from mammography.\ndata = pd.DataFrame(np.concatenate([['Biopsy but Not Malignant'] * len(DF_train[(DF_train['biopsy'] == 1) & (DF_train['cancer'] == 0)]) , ['Malignant Cancer'] *  len(DF_train[DF_train['cancer'] == 1]), ['Invasive Cancer'] *  len(DF_train[(DF_train['cancer'] == 1) & (DF_train['invasive'] == 1)])]), columns = [\"class\"])\n\nsns.countplot(x = 'class', data = data)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:57:52.762892Z","iopub.execute_input":"2024-01-30T09:57:52.763353Z","iopub.status.idle":"2024-01-30T09:57:53.001726Z","shell.execute_reply.started":"2024-01-30T09:57:52.763316Z","shell.execute_reply":"2024-01-30T09:57:53.000618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the path to each image.\nfor i in range(len(DF_train)):\n    DF_train.loc[i, 'path'] = os.path.join(RSNA_512_path + '/' + str(DF_train.loc[i, 'patient_id']) + '_' + str(DF_train.loc[i, 'image_id']) + '.png')\nDF_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:58:16.383884Z","iopub.execute_input":"2024-01-30T09:58:16.384946Z","iopub.status.idle":"2024-01-30T09:58:17.41341Z","shell.execute_reply.started":"2024-01-30T09:58:16.384893Z","shell.execute_reply":"2024-01-30T09:58:17.412242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a sample path\nDF_train.loc[0, 'path']","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:09:04.459171Z","iopub.execute_input":"2024-01-26T18:09:04.459574Z","iopub.status.idle":"2024-01-26T18:09:04.466658Z","shell.execute_reply.started":"2024-01-26T18:09:04.459542Z","shell.execute_reply":"2024-01-26T18:09:04.465676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a sample image\nimg = cv2.imread(DF_train.loc[3, 'path'])\nplt.imshow(img, cmap = 'gray')","metadata":{"execution":{"iopub.status.busy":"2024-01-30T09:58:51.10273Z","iopub.execute_input":"2024-01-30T09:58:51.103744Z","iopub.status.idle":"2024-01-30T09:58:51.406583Z","shell.execute_reply.started":"2024-01-30T09:58:51.103698Z","shell.execute_reply":"2024-01-30T09:58:51.405471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:09:55.600363Z","iopub.execute_input":"2024-01-26T18:09:55.601412Z","iopub.status.idle":"2024-01-26T18:09:55.608513Z","shell.execute_reply.started":"2024-01-26T18:09:55.60137Z","shell.execute_reply":"2024-01-26T18:09:55.607491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:09:25.62728Z","iopub.execute_input":"2024-01-26T18:09:25.62824Z","iopub.status.idle":"2024-01-26T18:09:25.634789Z","shell.execute_reply.started":"2024-01-26T18:09:25.628199Z","shell.execute_reply":"2024-01-26T18:09:25.633634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normal and cancer images must be equally distrubuted.\ntrain_df, val_df = train_test_split(DF_train, \n                                   test_size = 0.20, \n                                   random_state = 2018,\n                                   stratify = DF_train[['cancer']])\n\nprint('train', train_df.shape[0], 'validation', val_df.shape[0])\nprint('train', train_df['cancer'].value_counts())\nprint('validation', val_df['cancer'].value_counts())\ntrain_df.sample(1)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:09:29.448686Z","iopub.execute_input":"2024-01-26T18:09:29.44964Z","iopub.status.idle":"2024-01-26T18:09:29.490014Z","shell.execute_reply.started":"2024-01-26T18:09:29.449599Z","shell.execute_reply":"2024-01-26T18:09:29.489102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training data\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:09:33.513336Z","iopub.execute_input":"2024-01-26T18:09:33.513773Z","iopub.status.idle":"2024-01-26T18:09:33.532546Z","shell.execute_reply.started":"2024-01-26T18:09:33.513742Z","shell.execute_reply":"2024-01-26T18:09:33.531492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# validation data\nval_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:09:42.952557Z","iopub.execute_input":"2024-01-26T18:09:42.952969Z","iopub.status.idle":"2024-01-26T18:09:42.976888Z","shell.execute_reply.started":"2024-01-26T18:09:42.952934Z","shell.execute_reply":"2024-01-26T18:09:42.975617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up normal images from the training data.\ntrain_df_normal = train_df[train_df['cancer'] == 0].reset_index(drop = True)\nprint(len(train_df_normal))\ntrain_df_normal.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:10:12.264574Z","iopub.execute_input":"2024-01-26T18:10:12.265502Z","iopub.status.idle":"2024-01-26T18:10:12.285944Z","shell.execute_reply.started":"2024-01-26T18:10:12.265461Z","shell.execute_reply":"2024-01-26T18:10:12.284597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up cancer images from the training data.\ntrain_df_cancer = train_df[train_df['cancer'] == 1].reset_index(drop = True)\nprint(len(train_df_cancer))\ntrain_df_cancer.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:10:14.337869Z","iopub.execute_input":"2024-01-26T18:10:14.338286Z","iopub.status.idle":"2024-01-26T18:10:14.359789Z","shell.execute_reply.started":"2024-01-26T18:10:14.338256Z","shell.execute_reply":"2024-01-26T18:10:14.358774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up normal images from the validation data.\nval_df_normal = val_df[val_df['cancer'] == 0].reset_index(drop = True)\nprint(len(val_df_normal))\nval_df_normal.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:10:17.136704Z","iopub.execute_input":"2024-01-26T18:10:17.137369Z","iopub.status.idle":"2024-01-26T18:10:17.158239Z","shell.execute_reply.started":"2024-01-26T18:10:17.137336Z","shell.execute_reply":"2024-01-26T18:10:17.157127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up cancer images from the validation data.\nval_df_cancer = val_df[val_df['cancer'] == 1].reset_index(drop = True)\nprint(len(val_df_cancer))\nval_df_cancer.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:10:19.090675Z","iopub.execute_input":"2024-01-26T18:10:19.091637Z","iopub.status.idle":"2024-01-26T18:10:19.110564Z","shell.execute_reply.started":"2024-01-26T18:10:19.0916Z","shell.execute_reply":"2024-01-26T18:10:19.109374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n# Define the destination directory.\ndestination_dir = '/kaggle/working/train'\ndestination_dir_sub = '/kaggle/working/train/normal'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in train_df_normal['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:10:20.975765Z","iopub.execute_input":"2024-01-26T18:10:20.976876Z","iopub.status.idle":"2024-01-26T18:10:24.186438Z","shell.execute_reply.started":"2024-01-26T18:10:20.976837Z","shell.execute_reply":"2024-01-26T18:10:24.185568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:24:42.893356Z","iopub.execute_input":"2024-01-26T18:24:42.894419Z","iopub.status.idle":"2024-01-26T18:24:42.901485Z","shell.execute_reply.started":"2024-01-26T18:24:42.894376Z","shell.execute_reply":"2024-01-26T18:24:42.900478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/train'\ndestination_dir_sub = '/kaggle/working/train/cancer'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in train_df_cancer['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:10:42.81691Z","iopub.execute_input":"2024-01-26T18:10:42.817619Z","iopub.status.idle":"2024-01-26T18:10:45.855828Z","shell.execute_reply.started":"2024-01-26T18:10:42.817583Z","shell.execute_reply":"2024-01-26T18:10:45.854939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/val'\ndestination_dir_sub = '/kaggle/working/val/normal'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in val_df_normal['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:11:00.136875Z","iopub.execute_input":"2024-01-26T18:11:00.137545Z","iopub.status.idle":"2024-01-26T18:11:00.61605Z","shell.execute_reply.started":"2024-01-26T18:11:00.137512Z","shell.execute_reply":"2024-01-26T18:11:00.615196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/val'\ndestination_dir_sub = '/kaggle/working/val/cancer'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in val_df_cancer['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:11:06.472853Z","iopub.execute_input":"2024-01-26T18:11:06.473245Z","iopub.status.idle":"2024-01-26T18:11:06.949118Z","shell.execute_reply.started":"2024-01-26T18:11:06.473213Z","shell.execute_reply":"2024-01-26T18:11:06.94794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nnormal_train_images = glob.glob('/kaggle/working/train/normal/*.png')\ncancer_train_images = glob.glob('/kaggle/working/train/cancer/*.png')","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:11:25.800895Z","iopub.execute_input":"2024-01-26T18:11:25.801331Z","iopub.status.idle":"2024-01-26T18:11:25.811028Z","shell.execute_reply.started":"2024-01-26T18:11:25.801296Z","shell.execute_reply":"2024-01-26T18:11:25.810139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See normal images from the training dataset.\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in enumerate(axes.flat):\n    img = cv2.imread(normal_train_images[i])\n    ax.imshow(img)\n    ax.set_title('Normal')\nfig.tight_layout()    \n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:13:02.832752Z","iopub.execute_input":"2024-01-26T18:13:02.833122Z","iopub.status.idle":"2024-01-26T18:13:03.903314Z","shell.execute_reply.started":"2024-01-26T18:13:02.833096Z","shell.execute_reply":"2024-01-26T18:13:03.902364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See cancer images from the training dataset.\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in enumerate(axes.flat):\n    img = cv2.imread(cancer_train_images[i])\n    ax.imshow(img)\n    ax.set_title('Cancer')\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:13:07.834393Z","iopub.execute_input":"2024-01-26T18:13:07.835024Z","iopub.status.idle":"2024-01-26T18:13:08.52434Z","shell.execute_reply.started":"2024-01-26T18:13:07.834986Z","shell.execute_reply":"2024-01-26T18:13:08.523356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\nimages = []\n\nfor img_file in random.sample(cancer_train_images, 25):  # it is fixed to 25 random predictions - if you want to change it remeber to change plot_roi as well\n    \n    # Read file from file\n    frame = cv2.imread(img_file)\n    \n    # Make prediction\n    detections = model(frame)\n    \n    # Convert results to Pandas style\n    results = detections.pandas().xyxy[0].to_dict(orient=\"records\")\n    \n    # Plot result (in 99.99% it predicts only one instance - certainly you can assure that only best prediction is used)\n    for result in results:\n        images.append(cv2.rectangle(frame, (int(result['xmin']), int(result['ymin'])), (int(result['xmax']), int(result['ymax'])), (255,0,0), 4))\n\n# Plot result\nfig, axes = plt.subplots(5, 5, figsize=(20,20))\n    \nfor idx, image in enumerate(images):\n    i = idx % 5 \n    j = idx // 5 \n    axes[i, j].imshow(image)\n\nplt.subplots_adjust(wspace=0, hspace=.2)\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndestination_dir = '/kaggle/working/ctrain'\ndestination_dir_sub = '/kaggle/working/ctrain/cnormal'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# # Copy the images to the destination directory.\n# for path in train_df_normal['path']:\n#     shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T19:24:55.835597Z","iopub.execute_input":"2024-01-26T19:24:55.83607Z","iopub.status.idle":"2024-01-26T19:24:55.843055Z","shell.execute_reply.started":"2024-01-26T19:24:55.836036Z","shell.execute_reply":"2024-01-26T19:24:55.842044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/ctrain'\ndestination_dir_sub = '/kaggle/working/ctrain/ccancer'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\n# for path in train_df_cancer['path']:\n#     shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T19:26:03.563687Z","iopub.execute_input":"2024-01-26T19:26:03.564099Z","iopub.status.idle":"2024-01-26T19:26:03.57072Z","shell.execute_reply.started":"2024-01-26T19:26:03.564064Z","shell.execute_reply":"2024-01-26T19:26:03.569628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/cval'\ndestination_dir_sub = '/kaggle/working/cval/cnormal'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\n# for path in val_df_normal['path']:\n#     shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T19:29:45.872313Z","iopub.execute_input":"2024-01-26T19:29:45.872968Z","iopub.status.idle":"2024-01-26T19:29:45.879182Z","shell.execute_reply.started":"2024-01-26T19:29:45.87293Z","shell.execute_reply":"2024-01-26T19:29:45.878029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/cval'\ndestination_dir_sub = '/kaggle/working/cval/ccancer'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\n# for path in val_df_normal['path']:\n#     shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T19:29:13.712816Z","iopub.execute_input":"2024-01-26T19:29:13.713817Z","iopub.status.idle":"2024-01-26T19:29:13.719853Z","shell.execute_reply.started":"2024-01-26T19:29:13.713779Z","shell.execute_reply":"2024-01-26T19:29:13.718861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_directories(dir_paths):\n    for dir_path in dir_paths:\n        if not os.path.exists(dir_path):\n            os.makedirs(dir_path)\n\n# Define directories\nworking_dir = '/kaggle/working'\ntrain_dir = os.path.join(working_dir, 'train')\nval_dir = os.path.join(working_dir, 'val')\nctrain_dir = os.path.join(working_dir, 'ctrain')\ncval_dir = os.path.join(working_dir, 'cval')\nccancer_train_dir = os.path.join(ctrain_dir, 'ccancer')\ncnormal_train_dir = os.path.join(ctrain_dir, 'cnormal')\nccancer_val_dir = os.path.join(cval_dir, 'ccancer')\ncnormal_val_dir = os.path.join(cval_dir, 'cnormal')\n\n\n\ndef detect_and_crop_images(image_paths, output_dir_cancer, output_dir_normal):\n    for image_path in image_paths:\n        # Read the image\n        image = cv2.imread(image_path)\n        \n        # Perform object detection\n        detections = model(image)\n        results = detections.pandas().xyxy[0].to_dict(orient=\"records\")\n        \n        # Create output directory based on cancer or normal\n        if 'cancer' in image_path:\n            output_dir = output_dir_cancer\n        else:\n            output_dir = output_dir_normal\n        \n        # Iterate over each detected object\n        for idx, result in enumerate(results):\n            # Extract bounding box coordinates\n            xmin, ymin, xmax, ymax = int(result['xmin']), int(result['ymin']), int(result['xmax']), int(result['ymax'])\n            \n            # Crop the image\n            cropped_image = image[ymin:ymax, xmin:xmax]\n            \n            # Save the cropped image\n            cv2.imwrite(os.path.join(output_dir, f\"{os.path.basename(image_path)}_{idx}.png\"), cropped_image)\n\n\n# Perform object detection and crop images for training set\ndetect_and_crop_images(glob.glob(os.path.join(train_dir, 'cancer', '*.png')), ccancer_train_dir, cnormal_train_dir)\ndetect_and_crop_images(glob.glob(os.path.join(train_dir, 'normal', '*.png')), ccancer_train_dir, cnormal_train_dir)\n\n# Perform object detection and crop images for validation set\ndetect_and_crop_images(glob.glob(os.path.join(val_dir, 'cancer', '*.png')), ccancer_val_dir, cnormal_val_dir)\ndetect_and_crop_images(glob.glob(os.path.join(val_dir, 'normal', '*.png')), ccancer_val_dir, cnormal_val_dir)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T19:44:59.083783Z","iopub.execute_input":"2024-01-26T19:44:59.084597Z","iopub.status.idle":"2024-01-26T19:45:23.695927Z","shell.execute_reply.started":"2024-01-26T19:44:59.084561Z","shell.execute_reply":"2024-01-26T19:45:23.695079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = glob.glob('/kaggle/working/ctrain/cnormal/*.png')\n# See normal images from the training dataset.\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in enumerate(axes.flat):\n    img = cv2.imread(images[i])\n    ax.imshow(img)\n    ax.set_title('Normal')\nfig.tight_layout()    \n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-26T19:51:31.010496Z","iopub.execute_input":"2024-01-26T19:51:31.011619Z","iopub.status.idle":"2024-01-26T19:51:31.936793Z","shell.execute_reply.started":"2024-01-26T19:51:31.011575Z","shell.execute_reply":"2024-01-26T19:51:31.935771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale = 1./255.,\n                                   zoom_range = 0.2)\nval_datagen = ImageDataGenerator(rescale = 1./255.,)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T19:53:23.634877Z","iopub.execute_input":"2024-01-26T19:53:23.635689Z","iopub.status.idle":"2024-01-26T19:53:23.641217Z","shell.execute_reply.started":"2024-01-26T19:53:23.635649Z","shell.execute_reply":"2024-01-26T19:53:23.64013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#do not run by hisham!!!!!!!\ntrain_path = '/kaggle/working/train'\nval_path = '/kaggle/working/val'\n\ntrain_generator = train_datagen.flow_from_directory(\n    train_path,\n    target_size = (512, 512),\n    batch_size = 32,\n    class_mode = 'binary'\n)\nvalidation_generator = val_datagen.flow_from_directory(\n        val_path,\n        target_size = (512, 512),\n        batch_size = 16,\n        class_mode = 'binary'\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '/kaggle/working/ctrain'\nval_path = '/kaggle/working/cval'\n\ntrain_generator = train_datagen.flow_from_directory(\n    train_path,\n    target_size = (512, 512),\n    batch_size = 32,\n    class_mode = 'binary'\n)\nvalidation_generator = val_datagen.flow_from_directory(\n        val_path,\n        target_size = (512, 512),\n        batch_size = 16,\n        class_mode = 'binary'\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T19:53:37.958131Z","iopub.execute_input":"2024-01-26T19:53:37.958929Z","iopub.status.idle":"2024-01-26T19:53:38.172203Z","shell.execute_reply.started":"2024-01-26T19:53:37.958893Z","shell.execute_reply":"2024-01-26T19:53:38.17122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create list of input files \n# I use Radek Osmulski dataset -> https://www.kaggle.com/datasets/radek1/rsna-mammography-images-as-pngs \n\n# file_list = glob.glob('/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_768/train_images_processed_768/*/*.png')","metadata":{"execution":{"iopub.status.busy":"2022-12-01T09:15:19.313994Z","iopub.execute_input":"2022-12-01T09:15:19.314895Z","iopub.status.idle":"2022-12-01T09:16:09.854016Z","shell.execute_reply.started":"2022-12-01T09:15:19.314847Z","shell.execute_reply":"2022-12-01T09:16:09.852888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install tensorflow --upgrade","metadata":{"execution":{"iopub.status.busy":"2024-01-30T15:53:36.730135Z","iopub.execute_input":"2024-01-30T15:53:36.730445Z","iopub.status.idle":"2024-01-30T15:53:48.553735Z","shell.execute_reply.started":"2024-01-30T15:53:36.730412Z","shell.execute_reply":"2024-01-30T15:53:48.552509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-01-30T15:59:59.126766Z","iopub.execute_input":"2024-01-30T15:59:59.127469Z","iopub.status.idle":"2024-01-30T15:59:59.155648Z","shell.execute_reply.started":"2024-01-30T15:59:59.127435Z","shell.execute_reply":"2024-01-30T15:59:59.15438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model= tf.keras.applications.convnext.ConvNeXtSmall(\n    model_name='convnext_small',\n    include_top=True,\n    include_preprocessing=True,\n    weights='imagenet',\n    input_tensor=None,\n    input_shape=None,\n    pooling=None,\n    classes=2,\n    classifier_activation='softmax'\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T15:59:05.406014Z","iopub.execute_input":"2024-01-30T15:59:05.406369Z","iopub.status.idle":"2024-01-30T15:59:05.435219Z","shell.execute_reply.started":"2024-01-30T15:59:05.40634Z","shell.execute_reply":"2024-01-30T15:59:05.433659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = ResNet50V2(weights = 'imagenet', input_shape = (512, 512, 3), include_top = False)\n\nfor layer in base_model.layers:\n    layer.trainable = False\n    \nmodel = Sequential()\nmodel.add(base_model)\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Dense(128, activation = 'relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(1, activation = 'sigmoid'))\n\nmodel.compile(optimizer = \"adam\", loss = 'binary_crossentropy', metrics = [\"accuracy\"])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-01-26T19:55:18.81705Z","iopub.execute_input":"2024-01-26T19:55:18.817465Z","iopub.status.idle":"2024-01-26T19:55:18.832983Z","shell.execute_reply.started":"2024-01-26T19:55:18.817432Z","shell.execute_reply":"2024-01-26T19:55:18.832003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callback = tf.keras.callbacks.EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 4)\n\nhistory = model.fit(train_generator, validation_data = validation_generator, steps_per_epoch = 20, epochs = 15, callbacks = callback)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T15:35:35.885638Z","iopub.execute_input":"2024-01-30T15:35:35.886014Z","iopub.status.idle":"2024-01-30T15:35:35.999404Z","shell.execute_reply.started":"2024-01-30T15:35:35.885979Z","shell.execute_reply":"2024-01-30T15:35:35.998118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('mammography_pred_model.h5')","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:05:12.675457Z","iopub.execute_input":"2024-01-26T20:05:12.675937Z","iopub.status.idle":"2024-01-26T20:05:13.130984Z","shell.execute_reply.started":"2024-01-26T20:05:12.675887Z","shell.execute_reply":"2024-01-26T20:05:13.129869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = history.history['accuracy']\nval_accuracy = history.history['val_accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:05:21.926306Z","iopub.execute_input":"2024-01-26T20:05:21.92726Z","iopub.status.idle":"2024-01-26T20:05:21.932136Z","shell.execute_reply.started":"2024-01-26T20:05:21.927223Z","shell.execute_reply":"2024-01-26T20:05:21.931122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15,10))\n\nplt.subplot(2, 2, 1)\nplt.plot(accuracy, label = \"Training Accuracy\")\nplt.plot(val_accuracy, label = \"Validation Accuracy\")\nplt.ylim(0.4, 1)\nplt.legend(['Train', 'Validation'], loc = 'upper left')\nplt.title(\"Training vs Validation Accuracy\")\nplt.xlabel('epoch')\nplt.ylabel('accuracy')\n\n\nplt.subplot(2, 2, 2)\nplt.plot(loss, label = \"Training Loss\")\nplt.plot(val_loss, label = \"Validation Loss\")\nplt.legend(['Train', 'Validation'], loc = 'upper left')\nplt.title(\"Training vs Validation Loss\")\nplt.xlabel('epoch')\nplt.ylabel('loss')","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:05:25.0191Z","iopub.execute_input":"2024-01-26T20:05:25.019468Z","iopub.status.idle":"2024-01-26T20:05:25.355897Z","shell.execute_reply.started":"2024-01-26T20:05:25.01944Z","shell.execute_reply":"2024-01-26T20:05:25.354936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\nmodel = load_model('/kaggle/working/mammography_pred_model.h5')","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:05:40.171617Z","iopub.execute_input":"2024-01-26T20:05:40.172026Z","iopub.status.idle":"2024-01-26T20:05:41.849608Z","shell.execute_reply.started":"2024-01-26T20:05:40.171994Z","shell.execute_reply":"2024-01-26T20:05:41.848781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(validation_generator)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:05:43.890135Z","iopub.execute_input":"2024-01-26T20:05:43.890954Z","iopub.status.idle":"2024-01-26T20:05:50.457023Z","shell.execute_reply.started":"2024-01-26T20:05:43.890917Z","shell.execute_reply":"2024-01-26T20:05:50.456094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:05:54.427085Z","iopub.execute_input":"2024-01-26T20:05:54.427857Z","iopub.status.idle":"2024-01-26T20:05:54.441389Z","shell.execute_reply.started":"2024-01-26T20:05:54.427821Z","shell.execute_reply":"2024-01-26T20:05:54.440403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = []\nfor prob in pred:\n    if prob >= 0.5:\n        y_pred.append(1)\n    else:\n        y_pred.append(0)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:06:09.153532Z","iopub.execute_input":"2024-01-26T20:06:09.154333Z","iopub.status.idle":"2024-01-26T20:06:09.160906Z","shell.execute_reply.started":"2024-01-26T20:06:09.154298Z","shell.execute_reply":"2024-01-26T20:06:09.15969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:06:13.894648Z","iopub.execute_input":"2024-01-26T20:06:13.895073Z","iopub.status.idle":"2024-01-26T20:06:13.900014Z","shell.execute_reply.started":"2024-01-26T20:06:13.895037Z","shell.execute_reply":"2024-01-26T20:06:13.899016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(y_pred).value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:06:20.641233Z","iopub.execute_input":"2024-01-26T20:06:20.641653Z","iopub.status.idle":"2024-01-26T20:06:20.65067Z","shell.execute_reply.started":"2024-01-26T20:06:20.64162Z","shell.execute_reply":"2024-01-26T20:06:20.649572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = validation_generator.classes","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:06:29.060194Z","iopub.execute_input":"2024-01-26T20:06:29.060956Z","iopub.status.idle":"2024-01-26T20:06:29.065453Z","shell.execute_reply.started":"2024-01-26T20:06:29.060921Z","shell.execute_reply":"2024-01-26T20:06:29.064398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_true)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:06:32.536309Z","iopub.execute_input":"2024-01-26T20:06:32.537166Z","iopub.status.idle":"2024-01-26T20:06:32.543969Z","shell.execute_reply.started":"2024-01-26T20:06:32.537128Z","shell.execute_reply":"2024-01-26T20:06:32.542967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:06:39.056686Z","iopub.execute_input":"2024-01-26T20:06:39.057097Z","iopub.status.idle":"2024-01-26T20:06:39.063596Z","shell.execute_reply.started":"2024-01-26T20:06:39.057057Z","shell.execute_reply":"2024-01-26T20:06:39.062551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the class names.\nclass_names = ['Normal', 'Cancer']\n\n# Create the heatmap with class names as tick labels.\nax = sns.heatmap(cm, annot = True, fmt = '.0f', cmap = \"Blues\", annot_kws = {\"size\": 16},\\\n           xticklabels = class_names, yticklabels = class_names)\n\n# Set the axis labels.\nax.set_xlabel(\"Prediction\")\nax.set_ylabel(\"Truth\")","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:06:43.643467Z","iopub.execute_input":"2024-01-26T20:06:43.643895Z","iopub.status.idle":"2024-01-26T20:06:43.904804Z","shell.execute_reply.started":"2024-01-26T20:06:43.643858Z","shell.execute_reply":"2024-01-26T20:06:43.903922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_true, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-01-26T20:06:49.558298Z","iopub.execute_input":"2024-01-26T20:06:49.558692Z","iopub.status.idle":"2024-01-26T20:06:49.571415Z","shell.execute_reply.started":"2024-01-26T20:06:49.558658Z","shell.execute_reply":"2024-01-26T20:06:49.570364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-01-26T18:12:40.414351Z","iopub.execute_input":"2024-01-26T18:12:40.415284Z","iopub.status.idle":"2024-01-26T18:12:49.070668Z","shell.execute_reply.started":"2024-01-26T18:12:40.415242Z","shell.execute_reply":"2024-01-26T18:12:49.06977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thank you! This is my first contribution to RSNA Screening Mammography Breast Cancer Detection.\n\nHave a nice day and competition.","metadata":{}}]}