{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Import lib","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport scipy\nimport imghdr\nimport torch\nimport pytorch_lightning as pl\nfrom torch.utils.data import DataLoader, Dataset\nimport torchvision.transforms as transforms\nimport pydicom\nimport os\nimport numpy as np\nimport pandas as pd\nimport numpy as np\nimport cv2\nfrom skimage import data\nfrom skimage import filters\nfrom skimage.color import rgb2gray","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-03T09:49:02.883623Z","iopub.execute_input":"2022-12-03T09:49:02.884087Z","iopub.status.idle":"2022-12-03T09:49:08.685526Z","shell.execute_reply.started":"2022-12-03T09:49:02.883997Z","shell.execute_reply":"2022-12-03T09:49:08.683612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Load data","metadata":{}},{"cell_type":"code","source":"path_train = \"/kaggle/input/rsna-breast-cancer-detection/train.csv\"\npath_test = \"/kaggle/input/rsna-breast-cancer-detection/train.csv\"\ndf_train = pd.read_csv(path_train)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T09:49:08.689185Z","iopub.execute_input":"2022-12-03T09:49:08.691077Z","iopub.status.idle":"2022-12-03T09:49:08.852469Z","shell.execute_reply.started":"2022-12-03T09:49:08.691004Z","shell.execute_reply":"2022-12-03T09:49:08.851657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Preprocessing in one image","metadata":{}},{"cell_type":"markdown","source":"## a. read_image","metadata":{}},{"cell_type":"code","source":"def read_image(path: str):\n    img = pydicom.dcmread(path)\n    img = np.array(img.pixel_array, dtype = np.int64)\n    return img","metadata":{"execution":{"iopub.status.busy":"2022-12-03T09:49:08.853973Z","iopub.execute_input":"2022-12-03T09:49:08.854324Z","iopub.status.idle":"2022-12-03T09:49:08.860236Z","shell.execute_reply.started":"2022-12-03T09:49:08.854292Z","shell.execute_reply":"2022-12-03T09:49:08.858915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## b. CLAHE image","metadata":{}},{"cell_type":"code","source":"\nplt.rcParams.update({'font.size': 15})\nplt.rcParams['figure.figsize'] = (30, 10)\ndef show_hist(gray, equ, categori, cdf1 = [], cdf2 = []):\n    m = int(np.max(img))\n    f = plt.figure(figsize=(15, 15))\n    f.add_subplot(2, 2, 1, xticks = [], yticks=[]).set_title('Gray Image')\n    plt.imshow(gray, cmap = 'gray')\n    f.add_subplot(2, 2, 2).set_title('Histogram Gray Image')\n    histogram, bin_edges = np.histogram(gray, bins=m+1, range=(0, m+1))\n    plt.plot(bin_edges[:-1], histogram/np.max(histogram))\n    if len(cdf1) > 0:\n        plt.plot(cdf1, color = 'red')\n    plt.xlabel('pixel')\n    plt.ylabel('percent')\n\n\n    f.add_subplot(2, 2, 3, xticks = [], yticks=[]).set_title(f'{categori}')\n    plt.imshow(equ, cmap = 'gray')\n    f.add_subplot(2, 2, 4).set_title(f'Histogram {categori}')\n    histogram, bin_edges = np.histogram(equ, bins=256, range=(0, 256))\n    plt.plot(bin_edges[:-1], histogram/np.max(histogram))\n    if len(cdf2) > 0:\n        plt.plot(cdf2, color = 'red')\n    plt.xlabel('pixel')\n    plt.ylabel('percent')\n    plt.show()\n    \ndef get_cdf(img):\n    m = int(np.max(img))\n    hist = np.histogram(img, bins=m+1, range=(0, m+1))[0]\n    hist = hist/img.size\n    cdf = np.cumsum(hist)\n    return cdf\n\n# numpy\ndef histogram_equalization(img):\n    m = int(np.max(img))\n    hist = np.histogram(img, bins=m+1, range=(0, m+1))[0]\n    # bước 1: tính pdf\n    hist = hist/img.size\n    # bước 2: tính cdf\n    cdf = np.cumsum(hist)\n    # bước 3: lập bảng thay thế\n    s_k = (255 * cdf)\n    # ảnh mới\n    img_new = np.array([s_k[i] for i in img.ravel()]).reshape(img.shape)\n    return img_new\npath = \"/kaggle/input/rsna-breast-cancer-detection/train_images/10179/2015481666.dcm\"\nimg = read_image(path)\nimg_equ= histogram_equalization(img)\ncdf1 = get_cdf(img)\ncdf2 = get_cdf(img_equ)\nshow_hist(img, img_equ, 'Equalized with numpy', cdf1=cdf1, cdf2=cdf2)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-12-03T09:49:08.86324Z","iopub.execute_input":"2022-12-03T09:49:08.863746Z","iopub.status.idle":"2022-12-03T09:49:13.544982Z","shell.execute_reply.started":"2022-12-03T09:49:08.863698Z","shell.execute_reply":"2022-12-03T09:49:13.542416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-03T09:49:13.545951Z","iopub.execute_input":"2022-12-03T09:49:13.546636Z","iopub.status.idle":"2022-12-03T09:49:13.553025Z","shell.execute_reply.started":"2022-12-03T09:49:13.546602Z","shell.execute_reply":"2022-12-03T09:49:13.552155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## c. crop image","metadata":{}},{"cell_type":"markdown","source":"### get max component:\n","metadata":{}},{"cell_type":"code","source":"import scipy\nimport pandas as pd\ndef get_masks_and_sizes_of_connected_components(img_mask):\n    \"\"\"\n    Finds the connected components from the mask of the image\n    \"\"\"\n    mask, num_labels = scipy.ndimage.label(img_mask)\n\n    mask_pixels_dict = {}\n    for i in range(num_labels+1):\n        this_mask = (mask == i)\n        if img_mask[this_mask][0] != 0:\n            # Exclude the 0-valued mask\n            mask_pixels_dict[i] = np.sum(this_mask)\n        \n    return mask, mask_pixels_dict\n\n\ndef get_mask_of_largest_connected_component(img_mask):\n    \"\"\"\n    Finds the largest connected component from the mask of the image\n    \"\"\"\n    mask, mask_pixels_dict = get_masks_and_sizes_of_connected_components(img_mask)\n    largest_mask_index = pd.Series(mask_pixels_dict).idxmax()\n    largest_mask = mask == largest_mask_index\n    return largest_mask","metadata":{"execution":{"iopub.status.busy":"2022-12-03T09:49:13.553916Z","iopub.execute_input":"2022-12-03T09:49:13.554218Z","iopub.status.idle":"2022-12-03T09:49:13.616889Z","shell.execute_reply.started":"2022-12-03T09:49:13.554192Z","shell.execute_reply":"2022-12-03T09:49:13.615643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### preprocess before crop","metadata":{}},{"cell_type":"code","source":"def show(img, title = 'Gray Image', rgb = False, fs = 12, dp = (10, 10)):\n    plt.rcParams.update({'font.size': fs})\n    plt.rcParams['figure.figsize'] = dp\n    if rgb:\n        plt.imshow(img[:,:,::-1])\n    else:\n        plt.imshow(img, cmap=plt.cm.gray)\n    plt.axis('off')\n    plt.title(f'{title}')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T09:49:13.618143Z","iopub.execute_input":"2022-12-03T09:49:13.619Z","iopub.status.idle":"2022-12-03T09:49:13.636742Z","shell.execute_reply.started":"2022-12-03T09:49:13.618962Z","shell.execute_reply":"2022-12-03T09:49:13.635622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Image_procescing(img):\n    #check_img_convert_gray\n    if len(img.shape)==3:\n        img = rgb2gray(img)\n    #convert to bin\n    \n    threshold = filters.threshold_isodata(img)\n    bin_img = (img > threshold)*1\n    kernel = np.ones((5, 5), np.uint8)\n    bin_img = bin_img.astype('uint8')\n    bin_img = cv2.erode(bin_img, kernel, iterations=-2)\n    \n    #most mask\n    img_mask = get_mask_of_largest_connected_component(bin_img)\n    #crop_image\n    \n    farest_pixel = np.max(list(zip(*np.where(img_mask == 1))), axis=0)\n    nearest_pixel = np.min(list(zip(*np.where(img_mask == 1))), axis=0)\n    croped =  img[nearest_pixel[0]:farest_pixel[0], nearest_pixel[1]:farest_pixel[1]]\n    return croped","metadata":{"execution":{"iopub.status.busy":"2022-12-03T09:49:13.638237Z","iopub.execute_input":"2022-12-03T09:49:13.638697Z","iopub.status.idle":"2022-12-03T09:49:13.6513Z","shell.execute_reply.started":"2022-12-03T09:49:13.638663Z","shell.execute_reply":"2022-12-03T09:49:13.65043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_multi_img(img_list,labels=None,columns=1,rows=1):\n    fig = plt.figure(figsize=(8, 8))\n    for i in range(1, columns+rows):\n        img = images[i-1]\n        label = labels[i-1]\n        fig.add_subplot(rows, columns, i)\n        plt.title(label)\n        plt.axis(\"off\")\n        plt.imshow(img,cmap='gray')\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-03T09:49:13.652932Z","iopub.execute_input":"2022-12-03T09:49:13.65366Z","iopub.status.idle":"2022-12-03T09:49:13.661836Z","shell.execute_reply.started":"2022-12-03T09:49:13.653612Z","shell.execute_reply":"2022-12-03T09:49:13.66096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/rsna-breast-cancer-detection/train_images/10179/2015481666.dcm\"\nimg = read_image(path)\nimages=[img,Image_procescing(img),histogram_equalization(Image_procescing(img))]\nlabels=['Before Procescing','After Processcing','After CLAHE']\nshow_multi_img(images,labels,3,1)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T09:52:05.032829Z","iopub.execute_input":"2022-12-03T09:52:05.033281Z","iopub.status.idle":"2022-12-03T09:52:16.413528Z","shell.execute_reply.started":"2022-12-03T09:52:05.033245Z","shell.execute_reply":"2022-12-03T09:52:16.412671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Class dataset","metadata":{}},{"cell_type":"code","source":"class BreastCancerDataset(Dataset):\n    def __init__(self, img_dir):\n        self.img_dir = img_dir\n        self.img_data = pd.read_csv(img_dir)\n        \n    def __len__(self):\n        return len(self.img_data['image_id'])\n    \n    def __getitem__(self, idx):\n        img_path = os.path.join(self.img_dir, str(self.img_data.iloc[idx]['patient_id']), str(self.img_data.iloc[idx]['image_id'])+\".dcm\")\n        img = pydicom.dcmread(img_path)\n        img = np.array(img.pixel_array, dtype = np.float32)\n        img = torch.from_numpy(img)\n        label = self.img_data['cancer'][idx]\n        return img, label","metadata":{"execution":{"iopub.status.busy":"2022-12-03T09:49:20.205654Z","iopub.execute_input":"2022-12-03T09:49:20.206655Z","iopub.status.idle":"2022-12-03T09:49:20.215935Z","shell.execute_reply.started":"2022-12-03T09:49:20.206611Z","shell.execute_reply":"2022-12-03T09:49:20.214799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = \"/kaggle/input/rsna-breast-cancer-detection/train.csv\"\ntrain_dataset = BreastCancerDataset(train_dir)\nlen(train_dataset)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T09:49:20.217402Z","iopub.execute_input":"2022-12-03T09:49:20.217982Z","iopub.status.idle":"2022-12-03T09:49:20.295585Z","shell.execute_reply.started":"2022-12-03T09:49:20.21795Z","shell.execute_reply":"2022-12-03T09:49:20.294431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}