{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":6243,"databundleVersionId":868544,"sourceType":"competition"},{"sourceId":5967,"sourceType":"datasetVersion","datasetId":3748},{"sourceId":2052874,"sourceType":"datasetVersion","datasetId":981292},{"sourceId":4732089,"sourceType":"datasetVersion","datasetId":2738335},{"sourceId":4810680,"sourceType":"datasetVersion","datasetId":2785611},{"sourceId":5205373,"sourceType":"datasetVersion","datasetId":3027365},{"sourceId":5209690,"sourceType":"datasetVersion","datasetId":3030283}],"dockerImageVersionId":30302,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"I use 4 sets of mammography datasets, including: `RSNA`, `mini-ddsm`, `mias`, `inbreast`.\n\nI ROI crop by image segmentation and yolov8 model.\n\nmodel yolov8: https://colab.research.google.com/drive/1WFSiMeTixqm4anSMNUY_FdzK8TdHB_W5?usp=sharing","metadata":{}},{"cell_type":"code","source":"!pip install ultralytics\n!pip install xlrd\n!pip install /kaggle/input/rsnawhl/{pydicom-2.3.0-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}\n!pip install --upgrade tensorflow\n","metadata":{"execution":{"execution_failed":"2024-12-20T16:46:54.883Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport tensorflow as tf\nprint(tf.config.list_physical_devices('GPU'))\nfrom tensorflow.keras import backend as K\nK.set_image_data_format('channels_last')  # Use NHWC (default for TensorFlow)\nimport os\nos.environ[\"CUDA_VISIBLE_DEVICES\"] = \"-1\"\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:02:56.426972Z","iopub.execute_input":"2024-12-20T16:02:56.427268Z","iopub.status.idle":"2024-12-20T16:03:00.223842Z","shell.execute_reply.started":"2024-12-20T16:02:56.427241Z","shell.execute_reply":"2024-12-20T16:03:00.222923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nfrom glob import glob\nimport cv2\nfrom keras.optimizers import Adam\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut, apply_windowing\nfrom tqdm.notebook import tqdm\nimport time\nfrom datetime import datetime\nfrom IPython import display\nimport os\nimport torch\nimport multiprocessing as mp\nimport warnings\nwarnings.filterwarnings('ignore')\n# import required libraries\nfrom PIL import Image\nfrom PIL import ImageFile\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport tensorflow as tf\n# from keras.preprocessing.image import ImageDataGenerator\n# from keras.utils.np_utils import to_categorical\nfrom keras.applications.vgg16 import VGG16\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Flatten , Dropout\nfrom keras.optimizers import Adam\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau,ModelCheckpoint \nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nImageFile.LOAD_TRUNCATED_IMAGES = True\n\n%matplotlib inline\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\ncores = mp.cpu_count()\n\nplt.rcParams.update({'font.size': 10})\nplt.rcParams['figure.figsize'] = (8, 6)\n\nprint('Cores:', cores)\nprint('Device:', device)\nprint('Day: ', datetime.now())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:00.225618Z","iopub.execute_input":"2024-12-20T16:03:00.226356Z","iopub.status.idle":"2024-12-20T16:03:03.448125Z","shell.execute_reply.started":"2024-12-20T16:03:00.226318Z","shell.execute_reply":"2024-12-20T16:03:03.447246Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA rsna","metadata":{}},{"cell_type":"code","source":"# %cd '/kaggle'\n# %pwd","metadata":{"execution":{"iopub.status.busy":"2024-12-20T16:03:03.449373Z","iopub.execute_input":"2024-12-20T16:03:03.449728Z","iopub.status.idle":"2024-12-20T16:03:03.453677Z","shell.execute_reply.started":"2024-12-20T16:03:03.449694Z","shell.execute_reply":"2024-12-20T16:03:03.452806Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# input_path = \"input/rsna-breast-cancer-detection/\"\n# # df_samp = pd.read_csv(input_path+'sample_submission.csv')\n# df_train = pd.read_csv(input_path+'train.csv')\n# df_test = pd.read_csv(input_path+'test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-12-20T16:03:03.455761Z","iopub.execute_input":"2024-12-20T16:03:03.456025Z","iopub.status.idle":"2024-12-20T16:03:03.464025Z","shell.execute_reply.started":"2024-12-20T16:03:03.456002Z","shell.execute_reply":"2024-12-20T16:03:03.463153Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# count_cancer=df_train.groupby(by=\"cancer\").count()[\"patient_id\"]\n# fig = px.pie(values = count_cancer.values, names = count_cancer.index)\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-20T16:03:03.464983Z","iopub.execute_input":"2024-12-20T16:03:03.46525Z","iopub.status.idle":"2024-12-20T16:03:03.472467Z","shell.execute_reply.started":"2024-12-20T16:03:03.465227Z","shell.execute_reply":"2024-12-20T16:03:03.471776Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> Data cancer and normal are imbalance","metadata":{}},{"cell_type":"markdown","source":"# all dataset","metadata":{}},{"cell_type":"code","source":"# # Custom colors\n# class clr:\n#     S = '\\033[1m' + '\\033[91m'\n#     E = '\\033[0m'","metadata":{"execution":{"iopub.status.busy":"2024-12-20T16:03:03.473504Z","iopub.execute_input":"2024-12-20T16:03:03.473766Z","iopub.status.idle":"2024-12-20T16:03:03.480507Z","shell.execute_reply.started":"2024-12-20T16:03:03.473744Z","shell.execute_reply":"2024-12-20T16:03:03.479721Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def get_df(name_dataset):\n#     df = None\n#     if name_dataset == \"RSNA\":\n#         df = pd.read_csv(f\"input/rsna-breast-cancer-detection/train.csv\")\n#         df.columns = df.columns.str.capitalize()\n#         df['Path'] = df[['Patient_id', 'Image_id']].apply(lambda x: '/kaggle/input/rsna-breast-cancer-detection/train_images/'+\\\n#                                     str(x['Patient_id']) + \"/\" + str(x['Image_id']) + \".dcm\", axis=1)\n#         for i in tqdm(range(len(df)), desc=\"Loading RSNA dataset\"):\n#             i + 1\n\n#     elif name_dataset == \"DDSM\":\n#         is_cancer = {'Benign': 1, 'Cancer': 1, 'Normal': 0}\n#         data = {'Filename': [], 'Age':[], 'Density': [], 'Cancer':[], 'View': [], 'Laterality': [], 'Path': []}\n#         # extract the path and view\n#         paths = glob(f\"input/miniddsm2/MINI-DDSM-Complete-PNG-16/*/*\")\n#         for head in tqdm(paths, desc=\"Loading MINI-DDSM dataset\"):\n#             path = glob(f\"{head}/*\")\n#             path_ics = [x for x in path if \"ics\" in x][0]\n#             path_img = [x for x in path if (\"png\" in x and 'Mask' not in x)]\n#             if len(path_img) >= 1:\n#                 # get information from file *.png\n#                 for txt in path_img:\n#                     view = txt.split('.')[-2].split('_')[1]\n#                     laterality = txt.split('.')[-2].split('_')[0]\n#                     data['View'].append(view)\n#                     data['Laterality'].append('L' if laterality=='LEFT' else 'R')\n#                     data['Path'].append(txt)\n#                     data['Cancer'].append(is_cancer[head.split('/')[-2]])\n\n#                     # get information from file *.ics\n#                     f = open(path_ics, \"r\")\n#                     ics_text = f.read().strip().split(\"\\n\")\n#                     for txt in ics_text:\n#                         if txt.split()[0].upper() == 'FILENAME':\n#                             data['Filename'].append(txt.split()[1] if len(txt.split()) > 1 else 'NaN')\n#                         if txt.split()[0].upper() == 'PATIENT_AGE':\n#                             data['Age'].append(txt.split()[1] if len(txt.split()) > 1 else 'NaN')\n#                         if txt.split()[0].upper() == 'DENSITY':\n#                             data['Density'].append(txt.split()[1] if len(txt.split()) > 1 else 'NaN')\n#         df = pd.DataFrame(data)\n\n#     elif name_dataset == \"MIAS\":\n#         df = pd.read_csv(f'input/mias-mammography/Info.txt', sep=\" \").drop('Unnamed: 7',axis=1)\n#         df.columns = df.columns.str.capitalize()\n#         df['Path'] = df['Refnum'].apply(lambda x: '/kaggle/input/mias-mammography' + \"/\" + \"all-mias\" + \"/\" + x + \".pgm\")\n#         df['Cancer'] = df['Class'].apply(lambda x: 0 if x.upper() == 'NORM' else 1)\n#         for i in tqdm(range(len(df)), desc=\"Loading MIAS dataset\"):\n#             i+1\n#     elif name_dataset == \"INBREAST\":\n#         df = pd.read_excel(f'input/inbreast-2023/INbreast.xls', skipfooter=2)\n#         df.columns = df.columns.str.capitalize()\n#         paths = glob(f\"input/inbreast-2023/ALL-IMGS/*.dcm\")\n#         df['Path'] = df['File name'].apply(lambda x: [path for path in paths if path.split('/')[-1].split('_')[0] == str(x)][0])\n#         df['Lesion annotation status'].fillna('cancer', inplace=True)\n#         df['Lesion annotation status'] = df['Lesion annotation status'].str.upper()\n#         df['Cancer'] = df['Lesion annotation status'].apply(lambda x: 0 if x == 'NO ANNOTATION (NORMAL)' else 1)\n#         for i in tqdm(range(len(df)), desc=\"Loading INBREAST dataset\"):\n#             i+1\n#     else:\n#         print(\"Dataset not found\")\n#     return df\n\n# data_breast = [\"RSNA\", \"DDSM\", \"MIAS\", 'INBREAST']\n# rsna = get_df(data_breast[0])[['Path', 'Cancer', 'View', 'Laterality']]\n# ddsm = get_df(data_breast[1])[['Path', 'Cancer', 'View', 'Laterality']]\n# mias = get_df(data_breast[2])[['Path', 'Cancer']]\n# inbreast = get_df(data_breast[3])[['Path', 'Cancer', 'View', 'Laterality']]","metadata":{"execution":{"iopub.status.busy":"2024-12-20T16:03:03.481662Z","iopub.execute_input":"2024-12-20T16:03:03.481918Z","iopub.status.idle":"2024-12-20T16:03:03.49118Z","shell.execute_reply.started":"2024-12-20T16:03:03.481895Z","shell.execute_reply":"2024-12-20T16:03:03.490321Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# total_cancer = {'RSNA': rsna['Cancer'].value_counts(), 'MINI-DDSM': ddsm['Cancer'].value_counts(), 'MIAS': mias['Cancer'].value_counts(), 'INBREAST': inbreast['Cancer'].value_counts()}\n# frame_cancer = pd.DataFrame(total_cancer).T\n\n# # total data for each dataset\n# dataset_sum = frame_cancer.sum(axis=1)\n# for name_dataset, total in dataset_sum.to_dict().items():\n#     print(f'- {name_dataset}: {clr.S}{total}{clr.E}')\n\n# plt.pie(dataset_sum, labels=dataset_sum.index, autopct='%1.1f%%', startangle=90)\n# plt.title(\"Percent dataset\")\n# plt.legend()\n# plt.show()\n\n# cancer_sum = frame_cancer.sum(axis=0)\n# plt.pie(cancer_sum, labels=cancer_sum.index, autopct='%1.1f%%', startangle=90)\n# plt.title(\"Percent cancer and non-cancer\")\n# plt.legend(['non-cancer', 'cancer'])\n# plt.show()\n\n# frame_cancer.reset_index(inplace=True)\n# frame_cancer.rename(columns={'index': 'dataset', 1: 'cancer', 0: 'non-cancer'}, inplace=True)\n# # plot double bar chart \n# x = np.arange(len(frame_cancer.dataset))\n# w = 0.4\n# plt.bar(x, frame_cancer.cancer, label='cancer', width=w)\n# plt.bar(x+w, frame_cancer['non-cancer'], label='non-cancer', width=w)\n# plt.xticks(x+w/2, frame_cancer.dataset)\n\n# plt.title(\"Number of cancer and non-cancer\")\n# plt.legend()\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-20T16:03:03.492186Z","iopub.execute_input":"2024-12-20T16:03:03.492427Z","iopub.status.idle":"2024-12-20T16:03:03.50256Z","shell.execute_reply.started":"2024-12-20T16:03:03.492406Z","shell.execute_reply":"2024-12-20T16:03:03.501667Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ROI crop with process basic","metadata":{}},{"cell_type":"code","source":"# class Process_Image():\n#     def __init__(self, df, head = 'train_images'):\n#         self.df = df\n#         self.path = None\n#         self.head = head\n    \n#     def set_path(self, src):\n#         if isinstance(src, int):\n#             desc = self.df.iloc[src]\n#             src = input_path+self.head+\"/\"+str(desc.patient_id)+\"/\"+str(desc.image_id)+\".dcm\"\n#         self.path = src\n        \n#     def get_target(self):\n#         x = self.path.split(\"/\")[-2:] #'input/rsna-breast-cancer-detection/train_images/10589/195400299.dcm'\n#         pat_id = x[0]\n#         img_id = x[1][:-4]\n#         target = self.df.loc[(self.df.patient_id==int(pat_id)) & (self.df.image_id==int(img_id))].cancer.values\n#         return target[0]\n    \n#     def load_image(self, img_path, voi_lut=False, noInterpretation=False):\n#         dataset = pydicom.dcmread(img_path)\n#         img = dataset.pixel_array\n#         if voi_lut:\n#             img = apply_voi_lut(img, dataset)\n#         if noInterpretation:\n#             return img\n#         if dataset.PhotometricInterpretation == \"MONOCHROME1\":\n#             img = np.amax(img) - img\n#         return img\n\n#     def cvtRGB_image(self, img):\n#         img = img-np.amin(img)\n#         img = (img/np.amax(img))*255\n#         return img\n\n#     def crop_image(self, img):\n#         # threshold image\n#         threshold = ((img > np.mean(img))*255).astype(np.uint8)\n#         # bounding box\n#         contours, hierarchy = cv2.findContours(threshold, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_NONE)\n#         c = max(contours, key = cv2.contourArea)\n#         x,y,w,h = cv2.boundingRect(c)\n#         # crop\n#         return img[y:y+h, x:x+w]\n    \n#     def process(self, src, voi_lut = True, aug = 'clahe,2,8', sz_size=(512, 512)):\n#         self.set_path(src)\n#         img = self.load_image(self.path, voi_lut=voi_lut)\n#         img = self.cvtRGB_image(img)\n#         img = self.crop_image(img)\n#         if aug[:5]=='clahe':\n#             x = aug.split(',')\n#             clm = float(x[1])\n#             tgs = int(x[2])\n#             clahe = cv2.createCLAHE(clipLimit=clm, tileGridSize=(tgs, tgs))\n#             img = clahe.apply(img.astype('uint8'))\n#         img = cv2.resize(img, sz_size)\n#         return img\n        \n#     def __show__(self, cm='gray', voi_lut = True, aug='clahe,2,8'):\n#         fig, ax = plt.subplots(1, 2)\n#         origin_img = self.load_image(self.path, noInterpretation = True)\n#         start = time.time()\n#         processed = self.process(self.path, voi_lut, aug)\n#         end = time.time()\n#         target = {0:'no-cancer', 1:'cancer'}\n#         fig.suptitle(\"Path: \"+self.path.split(\"/\", 2)[-1]+\"\\n\"+f\"Time processed: {end-start}\"+\"\\n\"+f\"Target: {target[self.get_target()]}\")\n#         ax[0].imshow(origin_img, cmap=cm)\n#         ax[0].set_title(f'origin, shape: {origin_img.shape}')\n#         ax[0].axis('off')\n#         ax[1].imshow(processed, cmap=cm)\n#         ax[1].set_title(f'processed, shape: {processed.shape}')\n#         ax[1].axis('off')\n#         fig.tight_layout()\n#         plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-20T16:03:03.504004Z","iopub.execute_input":"2024-12-20T16:03:03.504338Z","iopub.status.idle":"2024-12-20T16:03:03.516425Z","shell.execute_reply.started":"2024-12-20T16:03:03.504306Z","shell.execute_reply":"2024-12-20T16:03:03.515577Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# pI = Process_Image(df_train, head='train_images')\n# pI.set_path(4)\n# pI.__show__()","metadata":{"execution":{"iopub.status.busy":"2024-12-20T16:03:03.517476Z","iopub.execute_input":"2024-12-20T16:03:03.517792Z","iopub.status.idle":"2024-12-20T16:03:03.527402Z","shell.execute_reply.started":"2024-12-20T16:03:03.517758Z","shell.execute_reply":"2024-12-20T16:03:03.526573Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cancer = df_train.loc[df_train.cancer==1][10:15]\n# no_cancer = df_train.loc[df_train.cancer==0][10:15]\n# test = pd.concat([cancer, no_cancer])","metadata":{"execution":{"iopub.status.busy":"2024-12-20T16:03:03.528615Z","iopub.execute_input":"2024-12-20T16:03:03.529203Z","iopub.status.idle":"2024-12-20T16:03:03.535482Z","shell.execute_reply.started":"2024-12-20T16:03:03.529169Z","shell.execute_reply":"2024-12-20T16:03:03.534726Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ROI crop with yolo","metadata":{}},{"cell_type":"code","source":"# yolo_model = YOLO(\"/kaggle/input/checkpoint-yolov8l/best.pt\")","metadata":{"execution":{"iopub.status.busy":"2024-12-20T16:03:03.536391Z","iopub.execute_input":"2024-12-20T16:03:03.536641Z","iopub.status.idle":"2024-12-20T16:03:03.546146Z","shell.execute_reply.started":"2024-12-20T16:03:03.536609Z","shell.execute_reply":"2024-12-20T16:03:03.545249Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# class extract_ROI():\n#     def __init__(self, df):\n#         self.df = df\n#         self.detect_model = yolo_model\n#         self.folder = None\n#         self.df_loc = None\n#         self.count_access = 0\n#         self.count_error = 0\n\n#     def __len__(self):\n#         return len(self.df)\n\n#     def load_image(self, idx=0):\n#         '''\n#         Method to load the image\n#         Parameter:\n#             - path (int or str): index or path of the image\n#         Return (numpy.ndarray): image 8-bit 3-channel\n#         '''\n#         path = idx\n#         if isinstance(idx, int):\n#             self.df_loc = self.df.iloc[idx]\n#             path = self.df_loc.Path\n            \n#         else:\n#             self.df_loc = self.df[self.df['Path']==idx]\n#         mode = path.split('.')[-1]\n#         img = None\n#         if mode == 'dcm':\n#             ds = pydicom.dcmread(path)\n#             img2d = ds.pixel_array\n#             # apply voi_lut\n#             voi_lut = apply_voi_lut(img2d, ds)\n#             if np.sum(voi_lut) > 0:\n#                 img2d = voi_lut\n#             # min-max scale\n#             img2d = (img2d - img2d.min()) / (img2d.max() - img2d.min())\n#             # convert to uint8\n#             img2d = (img2d * 255).astype(np.uint8)\n#             # convert to float to avoid overflow or underflow losses.\n#             if ds.PhotometricInterpretation == 'MONOCHROME1':\n#                 img2d = np.invert(img2d)\n#             # convert to 3-channel\n#             img = cv2.cvtColor(img2d, cv2.COLOR_GRAY2BGR)\n#         else:\n#             img = cv2.imread(path)\n#         return img\n    \n    \n#     def crop(self, img):\n#         results = self.detect_model(img)\n#         boxes = results[0].boxes\n#         box = boxes[0]\n#         xy = box.xyxy\n#         x1 = int(xy[0][0].item())\n#         y1 = int(xy[0][1].item())\n#         x2 = int(xy[0][2].item())\n#         y2 = int(xy[0][3].item())\n#         img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n#         img = img[y1: y2, x1:x2]\n#         return img\n\n#     def plot_image(self, idx=0):\n#         img = self.load_image(idx)\n#         # convert to grayscale\n#         img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n#         # plot\n#         fig, ax = plt.subplots(1, 1)\n#         fig.suptitle(f\"Path: {self.df_loc.Path}\")\n#         ax.imshow(img, cmap=plt.cm.gray)\n#         ax.set_title(f'Image, shape: {img.shape}')\n#         ax.axis('off')\n#         fig.tight_layout()\n#         plt.show()\n\n#     def plot(self, idx=0):\n#         img = self.load_image(idx)\n#         cropped = self.crop(img)\n#         # convert to grayscale\n#         img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n#         # plot\n#         fig, ax = plt.subplots(1, 2)\n#         fig.suptitle(f\"Path: {self.df_loc.Path}\")\n#         ax[0].imshow(img, cmap=plt.cm.gray)\n#         ax[0].set_title(f'Image, shape: {img.shape}')\n#         ax[0].axis('off')\n#         ax[1].imshow(cropped, cmap=plt.cm.gray)\n#         ax[1].set_title(f'Cropped, shape: {cropped.shape}')\n#         ax[1].axis('off')\n#         fig.tight_layout()\n#         plt.show()\n        \n#     def plot_sample(self, resize=256):\n#         '''\n#         Method to plot a sample of the images\n#         Parameter:\n#             - cropped (bool): True is origin image, False is cropped image\n#             - resize (int): resize the image\n#         '''\n#         imgs = []\n#         df_sample = self.df.sample(100, random_state=42)\n#         for path in tqdm(df_sample.Path.values):\n#             print(path)\n#             img = self.load_image(path)\n#             cropped = self.crop(img)\n#             display.clear_output(wait=True)\n#             imgs.append(cropped)\n#         if resize:\n#             imgs = [cv2.resize(img, (resize, resize)) for img in imgs]\n#         # plot\n#         n_cols = 10\n#         n_rows = 10\n#         fig, ax = plt.subplots(n_rows, n_cols, figsize=(n_cols*5,n_rows*5))\n#         i = 0\n#         for r in tqdm(range(0,n_rows)):\n#             for c in range(0,n_cols):\n#                 idx = r*n_cols + c\n#                 ax_idx = ax[r,c]\n#                 ax_idx.imshow(imgs[i],cmap=plt.cm.gray)\n#                 i+=1\n#                 ax_idx.axis('off')\n#         plt.tight_layout()\n#         plt.show()\n\n#         fig, ax = plt.subplots(n_rows, n_cols, figsize=(n_cols*5,n_rows*5))\n#         i = 0\n#         for r in tqdm(range(0,n_rows)):\n#             for c in range(0,n_cols):\n#                 idx = r*n_cols + c\n#                 ax_idx = ax[r,c]\n#                 ax_idx.imshow(imgs[i],cmap=plt.cm.gray)\n#                 i+=1\n#                 ax_idx.axis('off')\n#         plt.tight_layout()\n#         plt.show()\n\n#     def init_folder_save(self, folder_struc):\n#         self.folder = folder_struc[0]\n#         for path in folder_struc:\n#             os.makedirs(f'{path}', exist_ok=True)\n#         print(\"--- Created folder structure successfully!\")\n#         df_save = self.df.copy()\n#         df_save['Path'] = df_save['Path'].apply(lambda x: x.split('\\\\')[-1][:-3]+'png')\n#         df_save.to_csv(f'{self.folder}/description.csv', index=False)\n#         print('--- Saved description.csv successfully!')\n        \n\n#     def save_image(self, idx):\n#         img = self.load_image(idx)\n#         label_dict = {1:'Cancer', 0:'Normal'}\n#         try:\n#             crop = self.crop(img)\n#             # resize\n#             percent = 1280 / max(crop.shape)\n#             crop = cv2.resize(crop, (int(crop.shape[1]*percent), int(crop.shape[0]*percent)))\n#             # save\n#             path_save = f'RSNA-ROI-Mammography\\\\{label_dict[self.df_loc.Cancer]}\\\\{self.df_loc.Path_save}'\n#             cv2.imwrite(path_save, crop)\n#             self.count_access += 1\n#         except:\n#             # path_save = f'{self.folder}\\\\No_detect\\\\{self.df_loc.Path_save}'\n#             path_save = f'RSNA-ROI-Mammography\\\\No_detect\\\\{self.df_loc.Path_save}'\n#             cv2.imwrite(path_save, img)\n#             self.count_error += 1\n        \n#         display.clear_output(wait=True)\n    \n#     def save_all(self):\n#         for i in tqdm(range(self.__len__()), desc=\"Extracting ROI\"):\n#             self.save_image(i)\n#             print(f'--- Saved: {self.count_access+self.count_error}/{self.__len__()}')\n#             print(f'--- Detected: {self.count_access}')\n#             print(f'--- No detected: {self.count_error}')","metadata":{"execution":{"iopub.status.busy":"2024-12-20T16:03:03.549635Z","iopub.execute_input":"2024-12-20T16:03:03.549885Z","iopub.status.idle":"2024-12-20T16:03:03.560735Z","shell.execute_reply.started":"2024-12-20T16:03:03.549862Z","shell.execute_reply":"2024-12-20T16:03:03.560038Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# roi = extract_ROI(rsna)\n# print(roi.__len__())\n# # roi.df.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-20T16:03:03.561647Z","iopub.execute_input":"2024-12-20T16:03:03.561878Z","iopub.status.idle":"2024-12-20T16:03:03.572482Z","shell.execute_reply.started":"2024-12-20T16:03:03.561857Z","shell.execute_reply":"2024-12-20T16:03:03.571576Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# for i in range (10):\n#     roi.plot(i)","metadata":{"execution":{"iopub.status.busy":"2024-12-20T16:03:03.573586Z","iopub.execute_input":"2024-12-20T16:03:03.573827Z","iopub.status.idle":"2024-12-20T16:03:03.581265Z","shell.execute_reply.started":"2024-12-20T16:03:03.573805Z","shell.execute_reply":"2024-12-20T16:03:03.580618Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport pandas as pd\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport cv2\nimport time\n\n# === Load cervical cancer dataset ===\nroot_dir = '../input/intel-mobileodt-cervical-cancer-screening'\ntrain_dir = os.path.join(root_dir, 'train', 'train')\n\ntype1_dir = os.path.join(train_dir, 'Type_1')\ntype2_dir = os.path.join(train_dir, 'Type_2')\ntype3_dir = os.path.join(train_dir, 'Type_3')\n\ntrain_type1_files = glob.glob(type1_dir + '/*.jpg')\ntrain_type2_files = glob.glob(type2_dir + '/*.jpg')\ntrain_type3_files = glob.glob(type3_dir + '/*.jpg')\n\nadded_type1_files = glob.glob(os.path.join(root_dir, \"additional_Type_1_v2\", \"Type_1\") + '/*.jpg')\nadded_type2_files = glob.glob(os.path.join(root_dir, \"additional_Type_2_v2\", \"Type_2\") + '/*.jpg')\nadded_type3_files = glob.glob(os.path.join(root_dir, \"additional_Type_3_v2\", \"Type_3\") + '/*.jpg')\n\ntype1_files = train_type1_files + added_type1_files\ntype2_files = train_type2_files + added_type2_files\ntype3_files = train_type3_files + added_type3_files\n\nprint(f\"\"\"Type 1 files for training: {len(type1_files)} \nType 2 files for training: {len(type2_files)} \nType 3 files for training: {len(type3_files)}\"\"\")\n\n# === Create dataframe with filepaths and labels ===\nfiles = {\n    'filepath': type1_files + type2_files + type3_files,\n    'label': ['Type 1'] * len(type1_files) + ['Type 2'] * len(type2_files) + ['Type 3'] * len(type3_files)\n}\n\nfiles_df = pd.DataFrame(files).sample(frac=1, random_state=1).reset_index(drop=True)\n\n# === Check and remove damaged files ===\nbad_files = []\nfor path in files_df['filepath'].values:\n    try:\n        img = Image.open(path)\n    except Exception as e:\n        index = files_df[files_df['filepath'] == path].index.values[0]\n        bad_files.append(index)\n        print(f\"Bad file: {path}, Error: {e}\")\n\nfiles_df.drop(bad_files, inplace=True)\n\n# === Split the data into train, validation, and test sets ===\ntrain_df, eval_df = train_test_split(files_df, test_size=0.2, stratify=files_df['label'], random_state=1)\nval_df, test_df = train_test_split(eval_df, test_size=0.5, stratify=eval_df['label'], random_state=1)\nprint(len(train_df), len(val_df), len(test_df))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(files_df[files_df.duplicated(subset=['filepath'])])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.582325Z","iopub.execute_input":"2024-12-20T16:03:03.582562Z","iopub.status.idle":"2024-12-20T16:03:03.969194Z","shell.execute_reply.started":"2024-12-20T16:03:03.582541Z","shell.execute_reply":"2024-12-20T16:03:03.967658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check for damaged files\nbad_files = []\nfor path in (files_df['filepath'].values):\n    try:\n        img = Image.open(path)\n    except:\n        index = files_df[files_df['filepath']==path].index.values[0]\n        bad_files.append(index)\nprint(len(bad_files))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.970037Z","iopub.status.idle":"2024-12-20T16:03:03.970393Z","shell.execute_reply.started":"2024-12-20T16:03:03.97024Z","shell.execute_reply":"2024-12-20T16:03:03.970256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# drop the damaged files\nfiles_df.drop(bad_files, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.972004Z","iopub.status.idle":"2024-12-20T16:03:03.97233Z","shell.execute_reply.started":"2024-12-20T16:03:03.972179Z","shell.execute_reply":"2024-12-20T16:03:03.972194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check length of files in dataframe\nlen(files_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.974072Z","iopub.status.idle":"2024-12-20T16:03:03.974638Z","shell.execute_reply.started":"2024-12-20T16:03:03.974391Z","shell.execute_reply":"2024-12-20T16:03:03.974415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check unique labels\nfiles_df['label'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.975807Z","iopub.status.idle":"2024-12-20T16:03:03.976145Z","shell.execute_reply.started":"2024-12-20T16:03:03.975968Z","shell.execute_reply":"2024-12-20T16:03:03.975984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get count of each type \ntype_count = pd.DataFrame(files_df['label'].value_counts()).rename(columns= {'label': 'Num_Values'})\ntype_count","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.977449Z","iopub.status.idle":"2024-12-20T16:03:03.977762Z","shell.execute_reply.started":"2024-12-20T16:03:03.977613Z","shell.execute_reply":"2024-12-20T16:03:03.977628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === ROI Processing for Cervical Cancer ===\nclass ProcessCervicalImage:\n    def __init__(self, df):\n        self.df = df\n        self.path = None\n    \n    def set_path(self, src):\n        if isinstance(src, int):\n            self.path = self.df.iloc[src]['filepath']\n        else:\n            self.path = src\n\n    def load_image(self, img_path):\n        img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)\n        if img is None:\n            raise ValueError(f\"Image not found or unsupported format: {img_path}\")\n        return img\n\n    def crop_image(self, img):\n        # Apply threshold to find contours\n        _, thresh = cv2.threshold(img, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n        contours, _ = cv2.findContours(thresh, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n        if contours:\n            c = max(contours, key=cv2.contourArea)\n            x, y, w, h = cv2.boundingRect(c)\n            cropped_img = img[y:y+h, x:x+w]\n            return cropped_img\n        return img  # Return original if no contours found\n\n    def preprocess_image(self, src, output_size=(224, 224)):\n        self.set_path(src)\n        img = self.load_image(self.path)\n        img = self.crop_image(img)\n        img = cv2.resize(img, output_size)\n        return img\n\n    def visualize_processing(self, src):\n        self.set_path(src)\n        img_original = self.load_image(self.path)\n        img_processed = self.preprocess_image(self.path)\n\n        fig, axes = plt.subplots(1, 2, figsize=(10, 5))\n        axes[0].imshow(img_original, cmap='gray')\n        axes[0].set_title(\"Original Image\")\n        axes[1].imshow(img_processed, cmap='gray')\n        axes[1].set_title(\"Processed Image (Cropped and Resized)\")\n        plt.show()\n\n# === Example Usage ===\nprocessor = ProcessCervicalImage(files_df)\nprocessor.visualize_processing(0)  # Visualize processing for the first image in the dataset\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.979542Z","iopub.status.idle":"2024-12-20T16:03:03.979866Z","shell.execute_reply.started":"2024-12-20T16:03:03.979715Z","shell.execute_reply":"2024-12-20T16:03:03.97973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# display pieplot of label distribution\n#pie_plot = go.Pie(labels= type_count.index.to_list(), values= type_count.values.flatten(),hole= 0.2, text= type_count.index.to_list(), textposition='auto')\n#fig = go.Figure([pie_plot])\n#fig.update_layout(title_text='Pie Plot of Type Distribution')\n#fig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.98099Z","iopub.status.idle":"2024-12-20T16:03:03.981312Z","shell.execute_reply.started":"2024-12-20T16:03:03.98116Z","shell.execute_reply":"2024-12-20T16:03:03.981176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# display sample images of types\nfor label in ('Type 1', 'Type 2', 'Type 3'):\n    filepaths = files_df[files_df['label']==label]['filepath'].values[:5]\n    fig = plt.figure(figsize= (15, 6))\n    for i, path in enumerate(filepaths):\n        img = cv2.imread(path)\n        img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)\n        img = cv2.resize(img, (224, 224))\n        fig.add_subplot(1, 5, i+1)\n        plt.imshow(img)\n        plt.subplots_adjust(hspace=0.5)\n        plt.axis(False)\n        plt.title(label)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.982065Z","iopub.status.idle":"2024-12-20T16:03:03.982395Z","shell.execute_reply.started":"2024-12-20T16:03:03.982249Z","shell.execute_reply":"2024-12-20T16:03:03.982264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  split the data into train  and validation set\ntrain_df, eval_df = train_test_split(files_df, test_size= 0.2, stratify= files_df['label'], random_state= 1)\nval_df, test_df = train_test_split(eval_df, test_size= 0.5, stratify= eval_df['label'], random_state= 1)\nprint(len(train_df), len(val_df), len(test_df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.984663Z","iopub.status.idle":"2024-12-20T16:03:03.985135Z","shell.execute_reply.started":"2024-12-20T16:03:03.984879Z","shell.execute_reply":"2024-12-20T16:03:03.9849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# loads images from dataframe\ndef load_images(dataframe):\n    features = []\n    filepaths = dataframe['filepath'].values\n    labels = dataframe['label'].values\n    \n    for path in filepaths:\n        img = cv2.imread(path)\n        resized_img = cv2.resize(img, (180, 180))\n        features.append(np.array(resized_img))\n    return np.array(features), np.array(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.986281Z","iopub.status.idle":"2024-12-20T16:03:03.986715Z","shell.execute_reply.started":"2024-12-20T16:03:03.986488Z","shell.execute_reply":"2024-12-20T16:03:03.986509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# load training and evaluation data\ntrain_features, train_labels = load_images(train_df)\nval_features, val_labels = load_images(val_df)\ntest_features, test_labels = load_images(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.988204Z","iopub.status.idle":"2024-12-20T16:03:03.988482Z","shell.execute_reply.started":"2024-12-20T16:03:03.988347Z","shell.execute_reply":"2024-12-20T16:03:03.988361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check lengths of training and evaluation  sets\nlen(train_features), len(train_labels), len(test_features), len(test_labels), len(test_features), len(test_labels) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.989785Z","iopub.status.idle":"2024-12-20T16:03:03.990062Z","shell.execute_reply.started":"2024-12-20T16:03:03.989927Z","shell.execute_reply":"2024-12-20T16:03:03.98994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get image shape\nInputShape = train_features[0].shape\nprint(InputShape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.990954Z","iopub.status.idle":"2024-12-20T16:03:03.991284Z","shell.execute_reply.started":"2024-12-20T16:03:03.991129Z","shell.execute_reply":"2024-12-20T16:03:03.991148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# normalize the features\nX_train = train_features/255\nX_val  = val_features/255\nX_test  = test_features/255","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.992355Z","iopub.status.idle":"2024-12-20T16:03:03.992636Z","shell.execute_reply.started":"2024-12-20T16:03:03.992495Z","shell.execute_reply":"2024-12-20T16:03:03.992508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# encode the labels\nle = LabelEncoder().fit(['Type 1', 'Type 2', 'Type 3'])\ny_train = le.transform(train_labels)\ny_val = le.transform(val_labels)\ny_test = le.transform(test_labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.994501Z","iopub.status.idle":"2024-12-20T16:03:03.994778Z","shell.execute_reply.started":"2024-12-20T16:03:03.994643Z","shell.execute_reply":"2024-12-20T16:03:03.994656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check unique labels\nnp.unique(y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.996239Z","iopub.status.idle":"2024-12-20T16:03:03.996543Z","shell.execute_reply.started":"2024-12-20T16:03:03.996398Z","shell.execute_reply":"2024-12-20T16:03:03.996412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n# initialize image data generator for training and evaluation sets\ntrain_datagen = ImageDataGenerator(\n                                rotation_range = 40,\n                                zoom_range = 0.2,\n                                width_shift_range=0.2,\n                                height_shift_range=0.2,\n                                shear_range=0.2,\n                                horizontal_flip=True,\n                                vertical_flip = True)\n\neval_datagen = ImageDataGenerator()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.9982Z","iopub.status.idle":"2024-12-20T16:03:03.998502Z","shell.execute_reply.started":"2024-12-20T16:03:03.998358Z","shell.execute_reply":"2024-12-20T16:03:03.998372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# apply data augmentation to features\nBATCH_SIZE= 32\ntrain_gen = train_datagen.flow(X_train, y_train, batch_size= BATCH_SIZE)\nval_gen = eval_datagen.flow(X_val, y_val, batch_size= BATCH_SIZE)\ntest_gen = eval_datagen.flow(X_test, y_test, batch_size= BATCH_SIZE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:03.999662Z","iopub.status.idle":"2024-12-20T16:03:03.999963Z","shell.execute_reply.started":"2024-12-20T16:03:03.999811Z","shell.execute_reply":"2024-12-20T16:03:03.999825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# show shape of each  batch\nfor data_batch, labels_batch in train_gen:\n    print('data batch shape: {} \\n labels batch shape: {}'.format(data_batch.shape, labels_batch.shape))\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:04.001436Z","iopub.status.idle":"2024-12-20T16:03:04.001741Z","shell.execute_reply.started":"2024-12-20T16:03:04.001588Z","shell.execute_reply":"2024-12-20T16:03:04.001609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# initialize pretrained vgg model base\n# conv_base = VGG16(weights= 'imagenet', include_top= False, input_shape= (180, 180, 3))\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense\nfrom tensorflow.keras.models import Model\n\n# Define the EfficientNetB0 base with pre-trained weights\nconv_base = EfficientNetB0(weights='imagenet', include_top=False, input_shape=(180, 180, 3))\n\n# Add global average pooling layer and a fully connected layer for classification\nx = conv_base.output\nx = GlobalAveragePooling2D()(x)  # Reduce the feature maps to a single vector\nx = Dense(256, activation='relu')(x)  # Fully connected layer with 256 units\noutput = Dense(1, activation='sigmoid')(x)  # Adjust to your number of classes\n\n# Build the final model\nmodel = Model(inputs=conv_base.input, outputs=output)\n\n# Freeze the layers in the EfficientNet base\nfor layer in conv_base.layers:\n    layer.trainable = False\n\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:04.002828Z","iopub.status.idle":"2024-12-20T16:03:04.003124Z","shell.execute_reply.started":"2024-12-20T16:03:04.002965Z","shell.execute_reply":"2024-12-20T16:03:04.002978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# show trainable layers before freezing\nprint('This is the number of trainable weights '\n'before freezing layers in the conv base:', len(conv_base.trainable_weights))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:04.004005Z","iopub.status.idle":"2024-12-20T16:03:04.004322Z","shell.execute_reply.started":"2024-12-20T16:03:04.004176Z","shell.execute_reply":"2024-12-20T16:03:04.004191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# freeze few layers of pretrained model\nfor layer in conv_base.layers[:-5]:\n    layer.trainable= False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:04.005734Z","iopub.status.idle":"2024-12-20T16:03:04.006043Z","shell.execute_reply.started":"2024-12-20T16:03:04.005895Z","shell.execute_reply":"2024-12-20T16:03:04.00591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# show trainable layers after freezing\nprint('This is the number of trainable weights '\n'after freezing layers in the conv base:', len(conv_base.trainable_weights))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:04.007043Z","iopub.status.idle":"2024-12-20T16:03:04.007359Z","shell.execute_reply.started":"2024-12-20T16:03:04.007216Z","shell.execute_reply":"2024-12-20T16:03:04.007231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# build model \nmodel = Sequential([conv_base, \n                    Flatten(),\n                   Dropout(0.5),\n                   Dense(3, activation='softmax')])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:04.009041Z","iopub.status.idle":"2024-12-20T16:03:04.009401Z","shell.execute_reply.started":"2024-12-20T16:03:04.009254Z","shell.execute_reply":"2024-12-20T16:03:04.009269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# compile model\nmodel.compile(optimizer= Adam(0.0001), loss= 'sparse_categorical_crossentropy', metrics= ['accuracy'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:04.010816Z","iopub.status.idle":"2024-12-20T16:03:04.011095Z","shell.execute_reply.started":"2024-12-20T16:03:04.010959Z","shell.execute_reply":"2024-12-20T16:03:04.010972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# show model summary\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:04.01219Z","iopub.status.idle":"2024-12-20T16:03:04.012464Z","shell.execute_reply.started":"2024-12-20T16:03:04.012328Z","shell.execute_reply":"2024-12-20T16:03:04.012341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# define training steps\nTRAIN_STEPS = len(train_df)//BATCH_SIZE\nVAL_STEPS = len(val_df)//BATCH_SIZE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:04.013604Z","iopub.status.idle":"2024-12-20T16:03:04.013902Z","shell.execute_reply.started":"2024-12-20T16:03:04.013756Z","shell.execute_reply":"2024-12-20T16:03:04.013771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# initialize callbacks\nreduceLR = ReduceLROnPlateau(monitor='val_loss', patience=10, verbose= 1, mode='min', factor=  0.2, min_lr = 1e-5)\n\nearly_stopping = EarlyStopping(monitor='val_loss', patience = 20, verbose=1, mode='min', restore_best_weights= True)\n\ncheckpoint = ModelCheckpoint('cervicalModel.weights.keras', monitor='val_loss', verbose=1,save_best_only=True, mode= 'min')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:04.01501Z","iopub.status.idle":"2024-12-20T16:03:04.015326Z","shell.execute_reply.started":"2024-12-20T16:03:04.015179Z","shell.execute_reply":"2024-12-20T16:03:04.015194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train model\nhistory = model.fit(train_gen, steps_per_epoch= TRAIN_STEPS, validation_data=val_gen, validation_steps=VAL_STEPS, epochs= 100,\n                   callbacks= [reduceLR, early_stopping, checkpoint])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:04.016616Z","iopub.status.idle":"2024-12-20T16:03:04.016923Z","shell.execute_reply.started":"2024-12-20T16:03:04.016776Z","shell.execute_reply":"2024-12-20T16:03:04.01679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# yolo_model = YOLO(\"/kaggle/input/checkpoint-yolov8l/best.pt\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:04.018065Z","iopub.status.idle":"2024-12-20T16:03:04.018375Z","shell.execute_reply.started":"2024-12-20T16:03:04.018235Z","shell.execute_reply":"2024-12-20T16:03:04.01825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# class ExtractCervicalROI:\n#     def __init__(self, df, detect_model):\n#         \"\"\"\n#         Initialize the ROI extraction class.\n        \n#         Parameters:\n#         - df (pd.DataFrame): DataFrame containing image file paths and labels.\n#         - detect_model (YOLO): Pre-trained YOLO model for ROI detection.\n#         \"\"\"\n#         self.df = df\n#         self.detect_model = detect_model\n#         self.df_loc = None\n\n#     def __len__(self):\n#         return len(self.df)\n\n#     def load_image(self, idx=0):\n#         \"\"\"\n#         Load the image given an index or a path.\n        \n#         Parameters:\n#         - idx (int or str): Index in the DataFrame or the file path of the image.\n\n    #     Returns:\n    #     - img (numpy.ndarray): Loaded image as a NumPy array.\n    #     \"\"\"\n    #     if isinstance(idx, str):\n    #         path = idx\n    #         self.df_loc = self.df[self.df['filepath'] == path].iloc[0]  # Ensure df_loc is a DataFrame row\n    #     else:\n    #         path = self.df.iloc[idx]['filepath']\n    #         self.df_loc = self.df.iloc[idx]\n\n    #     img = cv2.imread(path)\n    #     if img is None:\n    #         raise FileNotFoundError(f\"Image at path {path} not found.\")\n    #     return cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    # def crop(self, img):\n    #     \"\"\"\n    #     Crop the region of interest (ROI) detected by YOLO.\n        \n    #     Parameters:\n    #     - img (numpy.ndarray): Original image.\n\n    #     Returns:\n    #     - cropped_img (numpy.ndarray): Cropped image containing the ROI.\n    #     \"\"\"\n    #     results = self.detect_model(img)\n    #     boxes = results[0].boxes\n    #     if len(boxes) == 0:\n    #         raise ValueError(\"No ROI detected in the image.\")\n        \n    #     # Extract the first bounding box (modify if multiple boxes need to be handled)\n    #     box = boxes[0]\n    #     x1, y1, x2, y2 = map(int, box.xyxy[0])\n    #     cropped_img = img[y1:y2, x1:x2]\n    #     return cropped_img\n\n    # def plot_image(self, idx=0):\n    #     \"\"\"\n    #     Plot the original image and the detected ROI.\n    \n    #     Parameters:\n    #     - idx (int): Index in the DataFrame or path of the image.\n    #     \"\"\"\n    #     img = self.load_image(idx)\n    #     cropped_img = self.crop(img)\n\n    #     # Get file path for the suptitle\n    #     path = self.df_loc['filepath'] if isinstance(self.df_loc, dict) else self.df_loc['filepath']\n\n    #     # Plot original and cropped image\n    #     fig, ax = plt.subplots(1, 2, figsize=(10, 5))\n    #     fig.suptitle(f\"Path: {path}\")\n    #     ax[0].imshow(img)\n    #     ax[0].set_title(f'Original Image, shape: {img.shape}')\n    #     ax[0].axis('off')\n    #     ax[1].imshow(cropped_img)\n    #     ax[1].set_title(f'Cropped ROI, shape: {cropped_img.shape}')\n    #     ax[1].axis('off')\n    #     plt.tight_layout()\n#         plt.show()\n\n#     def process_and_save(self, output_dir):\n#         \"\"\"\n#         Process all images, extract ROIs, and save them.\n\n#         Parameters:\n#         - output_dir (str): Directory to save the cropped images.\n#         \"\"\"\n#         os.makedirs(output_dir, exist_ok=True)\n#         for idx in tqdm(range(len(self.df)), desc=\"Processing images\"):\n#             img = self.load_image(idx)\n#             try:\n#                 cropped_img = self.crop(img)\n#                 save_path = os.path.join(output_dir, f\"roi_{idx}.jpg\")\n#                 cv2.imwrite(save_path, cv2.cvtColor(cropped_img, cv2.COLOR_RGB2BGR))\n#             except Exception as e:\n#                 print(f\"Error processing image {self.df.iloc[idx]['filepath']}: {e}\")\n\n# # Example usage\n# # Assuming `files_df` is the DataFrame created from cervical cancer dataset\n# # and `yolo_model` is the trained YOLO model for cervical cancer detection.\n\n# roi_extractor = ExtractCervicalROI(files_df, detect_model=yolo_model)\n\n# # Plot the ROI for a sample image\n# roi_extractor.plot_image(0)\n\n# # Process all images and save the ROIs\n# output_directory = \"./cervical_cancer_rois\"\n# roi_extractor.process_and_save(output_directory)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T16:03:04.020212Z","iopub.status.idle":"2024-12-20T16:03:04.020523Z","shell.execute_reply.started":"2024-12-20T16:03:04.020376Z","shell.execute_reply":"2024-12-20T16:03:04.020391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}