{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# The preprocess of 66th (LB 10th) solution.\n\n[Here](https://www.kaggle.com/competitions/rsna-breast-cancer-detection/discussion/391088) is my solution.\n\nAbout my preprocess (after windowing):\n1. Background noise reduction (kmeans)\n   I apply kmeans-clustering into each image using pixel values, then the pixel values of a cluster which has the lowest summation change into zero. To save time, I used the kmeans-pytorch.\n   This method gave me almost same CV score, but it boosted my LB (~0.58 --> ~0.63).\n   \n2. Image crop\n   I cropped images using below code.\n   This approach gave me slightly better CV and LB score than cropping using YOLO .","metadata":{}},{"cell_type":"code","source":"!pip install /kaggle/input/kmeans-pytorch/kmeans_pytorch","metadata":{"execution":{"iopub.status.busy":"2023-02-28T11:19:03.646616Z","iopub.execute_input":"2023-02-28T11:19:03.647065Z","iopub.status.idle":"2023-02-28T11:19:40.024482Z","shell.execute_reply.started":"2023-02-28T11:19:03.646987Z","shell.execute_reply":"2023-02-28T11:19:40.023281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! ls /kaggle/input/kmeans-pytorch/kmeans_pytorch","metadata":{"execution":{"iopub.status.busy":"2023-02-28T11:19:40.027031Z","iopub.execute_input":"2023-02-28T11:19:40.027764Z","iopub.status.idle":"2023-02-28T11:19:41.196938Z","shell.execute_reply.started":"2023-02-28T11:19:40.027711Z","shell.execute_reply":"2023-02-28T11:19:41.195743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append(\"/kaggle/input/kmeans-pytorch/kmeans_pytorch\")\nfrom kmeans_pytorch import kmeans","metadata":{"execution":{"iopub.status.busy":"2023-02-28T11:19:41.20183Z","iopub.execute_input":"2023-02-28T11:19:41.204412Z","iopub.status.idle":"2023-02-28T11:19:44.068636Z","shell.execute_reply.started":"2023-02-28T11:19:41.204362Z","shell.execute_reply":"2023-02-28T11:19:44.067627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport cv2, os, gc\n\nfrom glob import glob\n##from tqdm import tqdm\nfrom tqdm.notebook import tqdm\n\nimport torch\n\nimport matplotlib\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-02-28T11:26:10.721891Z","iopub.execute_input":"2023-02-28T11:26:10.722264Z","iopub.status.idle":"2023-02-28T11:26:10.729555Z","shell.execute_reply.started":"2023-02-28T11:26:10.722232Z","shell.execute_reply":"2023-02-28T11:26:10.726575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the data and pass it onto the training function\npng_dir = \"/kaggle/input/rsna-breast-cancer-1024-pngs/output\"\nnum = 25\n\ndf = pd.read_csv(\"../input/rsna-breast-cancer-detection/train.csv\").sample(25)\n\ndf['img_name'] = png_dir + \"/\" + df['patient_id'].astype(str) + \"_\" + df['image_id'].astype(str) + \".png\"\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T11:27:34.374156Z","iopub.execute_input":"2023-02-28T11:27:34.374653Z","iopub.status.idle":"2023-02-28T11:27:34.484687Z","shell.execute_reply.started":"2023-02-28T11:27:34.37461Z","shell.execute_reply":"2023-02-28T11:27:34.483398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tqdm\nimport copy\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\ndef kmeans_set_zero(img, dsize=(1536,960), num_clusters=4):\n    img_1d = torch.tensor(img.reshape(-1,1), device=device)\n    cluster_ids_x, cluster_centers = kmeans(X=img_1d, num_clusters=num_clusters,distance='euclidean', device=device, tqdm_flag=False, seed=0)\n    \n    smallest_sum = np.inf\n    smallest_id = 0\n    for cluster_id in range(num_clusters):\n        cluster_sum = img_1d[cluster_ids_x == cluster_id].sum().to(\"cpu\") / (100000)\n        if cluster_sum < smallest_sum:\n            smallest_sum = cluster_sum\n            smallest_id = cluster_id\n            \n    img_1d[cluster_ids_x == smallest_id] = 0    \n    kmeans_img = img_1d.to(\"cpu\").numpy().reshape(dsize)\n    return kmeans_img\n\n\ndef extraxtor_from_roi_box(d, frame, dsize=(960, 1536), num_clusters=4):\n    h, w = frame.shape\n    org_dsize = (h, w)\n\n    frame = kmeans_set_zero(frame, dsize=org_dsize, num_clusters=num_clusters) ##2048,1536 --> 2048,1536\n    frame_org = copy.copy(frame)\n    \n    thres1 = np.min(frame)+68 # 50 #frame.sum() / (h*w)\n    np.place(frame, frame < thres1, 0)\n\n    thres2 = frame_org.sum() / (h*w)\n    # print(i,\"thres1:\",thres1,\"thres2:\",thres2)\n    vertical_not_zero = [True if frame[:,idx].sum() > thres2 else False for idx in range(w)]\n    horizontal_not_zero = [True if frame[idx,:].sum() > thres2 else False for idx in range(h)]\n\n    crop = frame_org[horizontal_not_zero,:]\n    crop = crop[:,vertical_not_zero]     \n\n    crop = cv2.resize(crop, dsize=dsize, interpolation=cv2.INTER_LINEAR) # 2048,1536 --> 1536,960\n        \n    return crop\n\ndsize = (640, 1024)\norg_images = []\nimages = []\nnum = 5\nfor idx in tqdm.tqdm(range(len(df))):\n    d = df.iloc[idx]\n    img_file = d.img_name        \n    # Read file from file\n    frame = cv2.imread(img_file, cv2.IMREAD_GRAYSCALE)\n    org_frame = frame.copy()\n    crop = extraxtor_from_roi_box(d, frame, dsize=dsize)\n#         print(\"crop size:\",crop.shape)\n    org_images.append(org_frame)\n    images.append(crop)\n\n\nfig, axes = plt.subplots(num, num, figsize=(20,20))\nfor idx, image in enumerate(org_images):\n    i = idx % num\n    j = idx // num \n    axes[i, j].imshow(image)\n\nplt.subplots_adjust(wspace=0, hspace=.2)\nplt.title(\"original images\")\nplt.show()        \n\nfig, axes = plt.subplots(num, num, figsize=(20,20))\nfor idx, image in enumerate(images):\n    i = idx % num\n    j = idx // num \n    axes[i, j].imshow(image)\n\nplt.subplots_adjust(wspace=0, hspace=.2)\nplt.title(\"original images + CROP\")\nplt.show()       \n\n\n\ndel frame, crop, images, image\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T11:27:35.304378Z","iopub.execute_input":"2023-02-28T11:27:35.304764Z","iopub.status.idle":"2023-02-28T11:27:48.130626Z","shell.execute_reply.started":"2023-02-28T11:27:35.304732Z","shell.execute_reply":"2023-02-28T11:27:48.129806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Thanks for your reading","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}