{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":18647,"databundleVersionId":1126921},{"sourceType":"datasetVersion","sourceId":1113957,"datasetId":624783,"databundleVersionId":1144242}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:20:17.556808Z","iopub.execute_input":"2026-03-15T12:20:17.557173Z","iopub.status.idle":"2026-03-15T12:20:29.344394Z","shell.execute_reply.started":"2026-03-15T12:20:17.557142Z","shell.execute_reply":"2026-03-15T12:20:29.343294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install staintools","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:22:50.762268Z","iopub.execute_input":"2026-03-15T12:22:50.762616Z","iopub.status.idle":"2026-03-15T12:22:54.923857Z","shell.execute_reply.started":"2026-03-15T12:22:50.762588Z","shell.execute_reply":"2026-03-15T12:22:54.922333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" !pip install spams-bin","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:23:00.195503Z","iopub.execute_input":"2026-03-15T12:23:00.196876Z","iopub.status.idle":"2026-03-15T12:23:04.351018Z","shell.execute_reply.started":"2026-03-15T12:23:00.19681Z","shell.execute_reply":"2026-03-15T12:23:04.349625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import spams","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:23:10.529053Z","iopub.execute_input":"2026-03-15T12:23:10.529375Z","iopub.status.idle":"2026-03-15T12:23:10.533999Z","shell.execute_reply.started":"2026-03-15T12:23:10.529349Z","shell.execute_reply":"2026-03-15T12:23:10.533006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!apt-get install -y openslide-tools\n!pip install openslide-python","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:23:14.44235Z","iopub.execute_input":"2026-03-15T12:23:14.442711Z","iopub.status.idle":"2026-03-15T12:23:21.847188Z","shell.execute_reply.started":"2026-03-15T12:23:14.44268Z","shell.execute_reply":"2026-03-15T12:23:21.846032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install staintools\n# staintools -> target에 대한 color를 orginal의 색으로 해주기 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:23:23.221993Z","iopub.execute_input":"2026-03-15T12:23:23.22238Z","iopub.status.idle":"2026-03-15T12:23:27.422282Z","shell.execute_reply.started":"2026-03-15T12:23:23.222342Z","shell.execute_reply":"2026-03-15T12:23:27.420976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pathology의 경우 아주 큰 영상과 다양한 color distribution이 있음 \n# 대표적으로 patch, color normalization에 대해 배워보도록 함","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Path관련 library\nfrom pathlib import Path # data path관련 \n\n## WSI관련 library\nimport openslide # WSI 읽어오는 library \nimport cv2  # 기본적인 CV library \nimport seaborn as sns # data analysis library\n\n## etc\nimport pandas as pd # csv, dataframe 사용 library\nimport numpy as np \nimport matplotlib.pyplot as plt # image visulization library\n\n## color normalize\nimport staintools","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:23:30.0553Z","iopub.execute_input":"2026-03-15T12:23:30.055684Z","iopub.status.idle":"2026-03-15T12:23:30.939252Z","shell.execute_reply.started":"2026-03-15T12:23:30.055651Z","shell.execute_reply":"2026-03-15T12:23:30.938024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 우선적으로 datset을 읽어와줍니다. \nclass cfg: \n    # Location of the training images\n    # (개인) kaggle의 path은 옆에 업로드 된 것에서 copy하기 \n    BASE_PATH = \"/kaggle/input/competitions/prostate-cancer-grade-assessment\"\n    # image and mask directories\n    data_dir = f'{BASE_PATH}/train_images'\n    mask_dir = f'{BASE_PATH}/train_label_masks'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:23:38.509023Z","iopub.execute_input":"2026-03-15T12:23:38.509384Z","iopub.status.idle":"2026-03-15T12:23:38.514572Z","shell.execute_reply.started":"2026-03-15T12:23:38.509354Z","shell.execute_reply":"2026-03-15T12:23:38.513731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(cfg.BASE_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:23:42.606337Z","iopub.execute_input":"2026-03-15T12:23:42.606925Z","iopub.status.idle":"2026-03-15T12:23:42.612588Z","shell.execute_reply.started":"2026-03-15T12:23:42.60688Z","shell.execute_reply":"2026-03-15T12:23:42.611546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls /kaggle/input","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:23:46.351815Z","iopub.execute_input":"2026-03-15T12:23:46.352174Z","iopub.status.idle":"2026-03-15T12:23:46.480896Z","shell.execute_reply.started":"2026-03-15T12:23:46.352137Z","shell.execute_reply":"2026-03-15T12:23:46.479611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# csv 파일을 읽어 data의 경로를 확인합니다 \ntrain = pd.read_csv(f'{cfg.BASE_PATH}/train.csv')\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:23:48.126965Z","iopub.execute_input":"2026-03-15T12:23:48.127348Z","iopub.status.idle":"2026-03-15T12:23:48.171768Z","shell.execute_reply.started":"2026-03-15T12:23:48.127312Z","shell.execute_reply":"2026-03-15T12:23:48.170592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['img_path'] = train['image_id'].apply(lambda x : Path(cfg.data_dir)/ (x+'.tiff'))\ntrain.head()\n# (개인) image id로만 wsi를 가지고 올 수 없어서 경로 연결해주기 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:23:51.012788Z","iopub.execute_input":"2026-03-15T12:23:51.013143Z","iopub.status.idle":"2026-03-15T12:23:51.072155Z","shell.execute_reply.started":"2026-03-15T12:23:51.013111Z","shell.execute_reply":"2026-03-15T12:23:51.070918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['img_path'][0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:23:53.690957Z","iopub.execute_input":"2026-03-15T12:23:53.691312Z","iopub.status.idle":"2026-03-15T12:23:53.698075Z","shell.execute_reply.started":"2026-03-15T12:23:53.691282Z","shell.execute_reply":"2026-03-15T12:23:53.697108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image = openslide.OpenSlide(train.loc[0,'img_path']) # openslid로 tiff 읽기 \nmax_level = image.level_count   # 가장 해상도가 적은 image 순서 가져오기  \nprint(image.level_dimensions, image.level_count)\nlevel = max_level - 3\nsmall_patch = image.read_region((7100,10000), level, (1024,1024)) #r \n# (개인) read_region(location, level, size)\n\nlevel = max_level - 2\nminum_patch = image.read_region((7100//pow(2,1),10000//pow(2,1)), level, (1024,1024)) #r \n\nlevel = max_level - 1\nmin_size = image.level_dimensions[level]\nlarge_patch = image.read_region((7100//pow(2,3),10000//pow(2,3)), level, (1024,1024)) #r \n\n# (개인) width x height -> level이 올라갈수록 dim 작아짐, 최대 level은 3까지 \n# (개인) 아래 20000 크기를 가지고 오면 kernel이 터짐. 그래서 우리는 crop해서 사용 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:23:56.523549Z","iopub.execute_input":"2026-03-15T12:23:56.524683Z","iopub.status.idle":"2026-03-15T12:23:57.067288Z","shell.execute_reply.started":"2026-03-15T12:23:56.524635Z","shell.execute_reply":"2026-03-15T12:23:57.066441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"f, ax = plt.subplots(1,3, figsize=(15,15))\nax[0].imshow(small_patch) # (개인) 고화질 -> 여기서 patch를 잘라서 쓰기 \nax[1].imshow(minum_patch) # (개인) 큰 조직에 대해서 \nax[2].imshow(large_patch) # (개인) 전반 모양 보기, 그리고 white space가 많음 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:01.320483Z","iopub.execute_input":"2026-03-15T12:24:01.320864Z","iopub.status.idle":"2026-03-15T12:24:02.354954Z","shell.execute_reply.started":"2026-03-15T12:24:01.320836Z","shell.execute_reply":"2026-03-15T12:24:02.353809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 전처리\n# Tile (Patching)\n# 우리가 아는 openslide의 경우 다양한 image를 읽어올수 있지만 읽어오는 속도가 느리다. \n# 이를 해결하고자 `skimage`를 사용하여 이미지를 읽어오도록 한다. \n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import skimage # (개인) 실제 CPU 읽어오는 속도가 빠름 \nimport openslide \n%time image_a = openslide.OpenSlide(train.loc[0,'img_path'])\n%time image_b = skimage.io.MultiImage(str(train.loc[0,'img_path']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:06.927264Z","iopub.execute_input":"2026-03-15T12:24:06.927625Z","iopub.status.idle":"2026-03-15T12:24:07.034344Z","shell.execute_reply.started":"2026-03-15T12:24:06.927594Z","shell.execute_reply":"2026-03-15T12:24:07.032984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Patch로 자르기 이전에 pathology image에서는 white즉 흰공간이 너무 많다.\n# 이를 제거해주는 간편한 코드를 사용해보자. ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"level = max_level - 1\nmin_size = image_a.level_dimensions[level]\nlarge_patch = image_a.read_region((0,0), level, min_size) #r ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:12.94562Z","iopub.execute_input":"2026-03-15T12:24:12.945952Z","iopub.status.idle":"2026-03-15T12:24:13.073901Z","shell.execute_reply.started":"2026-03-15T12:24:12.945923Z","shell.execute_reply":"2026-03-15T12:24:13.072969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## int8로 되어있음. \nfig, ax = plt.subplots(1,3, figsize=(10,2))\nsns.histplot(np.array(large_patch)[...,0].ravel(), ax=ax[0])\nsns.histplot(np.array(large_patch)[...,1].ravel(), ax=ax[1])\nsns.histplot(np.array(large_patch)[...,2].ravel(), ax=ax[2])\n# (개인) RGB에 대한 histogram 사용 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:17.560183Z","iopub.execute_input":"2026-03-15T12:24:17.560553Z","iopub.status.idle":"2026-03-15T12:24:24.233625Z","shell.execute_reply.started":"2026-03-15T12:24:17.560521Z","shell.execute_reply":"2026-03-15T12:24:24.232822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## crop white region\ndef crop_white(image: np.ndarray) -> np.ndarray:\n    assert image.shape[2] == 3\n    assert image.dtype == np.uint8\n    ys, = (image.min((1, 2)) != 255).nonzero()\n    xs, = (image.min(0).min(1) != 255).nonzero()\n    if len(xs) == 0 or len(ys) == 0:\n        return image\n    return image[ys.min():ys.max() + 1, xs.min():xs.max() + 1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:26.333889Z","iopub.execute_input":"2026-03-15T12:24:26.334227Z","iopub.status.idle":"2026-03-15T12:24:26.341312Z","shell.execute_reply.started":"2026-03-15T12:24:26.334199Z","shell.execute_reply":"2026-03-15T12:24:26.3401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rm_white_img = crop_white(np.array(large_patch)[...,:3])\nfig, ax = plt.subplots(1,2, figsize=(10,2))\nax[0].imshow(large_patch)\nax[1].imshow(rm_white_img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:27.616809Z","iopub.execute_input":"2026-03-15T12:24:27.617177Z","iopub.status.idle":"2026-03-15T12:24:28.347963Z","shell.execute_reply.started":"2026-03-15T12:24:27.617145Z","shell.execute_reply":"2026-03-15T12:24:28.346957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import math\nimport cv2\nsz = 256 # (개인) 어떤 patch로 자를건지?\npad = 128\nN=9 # (개인) 9개 patch만 뽑아내기 \ndef tile(img):\n    shape = img.shape\n    pad0,pad1 = (sz - shape[0]%sz)%sz, (sz - shape[1]%sz)%sz\n    img = np.pad(img,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]],\n                 constant_values=255)\n    print([[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]])\n    img = img.reshape(img.shape[0]//sz,sz,img.shape[1]//sz,sz,3)\n    img = img.transpose(0,2,1,3,4).reshape(-1,sz,sz,3)\n    if len(img) < N:\n        img = np.pad(img,[[0,N-len(img)],[0,0],[0,0],[0,0]],constant_values=255)\n    idxs = np.argsort(img.reshape(img.shape[0],-1).sum(-1))[:N]\n    img = img[idxs]\n    return img\n\n# (개인) 9개의 patch를 잘랐는데, 3x3으로 concat \ndef concate_images(img): \n    assert len(img.shape) == 4\n    assert img.shape[0] == N\n    ax_size = int(math.sqrt(N))\n    order_index = np.arange(0,N).reshape(ax_size,ax_size)\n    hconcat = [cv2.hconcat(img[i]) for i in order_index]\n    con_img = cv2.vconcat(hconcat)\n    return con_img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:31.157626Z","iopub.execute_input":"2026-03-15T12:24:31.158105Z","iopub.status.idle":"2026-03-15T12:24:31.180216Z","shell.execute_reply.started":"2026-03-15T12:24:31.15806Z","shell.execute_reply":"2026-03-15T12:24:31.178971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install imagecodecs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:33.758695Z","iopub.execute_input":"2026-03-15T12:24:33.759095Z","iopub.status.idle":"2026-03-15T12:24:37.832148Z","shell.execute_reply.started":"2026-03-15T12:24:33.759063Z","shell.execute_reply":"2026-03-15T12:24:37.830666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q imagecodecs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:37.834427Z","iopub.execute_input":"2026-03-15T12:24:37.8348Z","iopub.status.idle":"2026-03-15T12:24:41.950973Z","shell.execute_reply.started":"2026-03-15T12:24:37.834757Z","shell.execute_reply":"2026-03-15T12:24:41.949383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import imagecodecs\nprint(\"imagecodecs installed\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:43.35963Z","iopub.execute_input":"2026-03-15T12:24:43.359984Z","iopub.status.idle":"2026-03-15T12:24:43.365763Z","shell.execute_reply.started":"2026-03-15T12:24:43.359948Z","shell.execute_reply":"2026-03-15T12:24:43.364532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"remove_image_b = crop_white(image_b[0]) # white 의 값을 제거 \ntile_images = tile(remove_image_b) #Patch로 나눔\nconcate_img = concate_images(tile_images) #다시 vis concate","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:44.696037Z","iopub.execute_input":"2026-03-15T12:24:44.69639Z","iopub.status.idle":"2026-03-15T12:24:48.819263Z","shell.execute_reply.started":"2026-03-15T12:24:44.696361Z","shell.execute_reply":"2026-03-15T12:24:48.817949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig,axs = plt.subplots(1,2)\naxs[0].imshow(remove_image_b)\naxs[1].imshow(concate_img)\n\ndel remove_image_b, image_b, concate_img\n\n# (개인) 왼쪽은 5000 x 20000의 크기가 됨 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:49.617444Z","iopub.execute_input":"2026-03-15T12:24:49.618335Z","iopub.status.idle":"2026-03-15T12:24:55.832674Z","shell.execute_reply.started":"2026-03-15T12:24:49.618292Z","shell.execute_reply":"2026-03-15T12:24:55.831726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n## Color normalization\n# 일반적으로 Pathology image는 색의 분포가 다르다. \n# 이를 맞춰주는 작업도 필요하다.(학습시 필수는 아니다) \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:57.632414Z","iopub.execute_input":"2026-03-15T12:24:57.632781Z","iopub.status.idle":"2026-03-15T12:24:57.637704Z","shell.execute_reply.started":"2026-03-15T12:24:57.632749Z","shell.execute_reply":"2026-03-15T12:24:57.636605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_type = train['data_provider'].unique()\nprint(data_type)\n\nkar_img_path = train[train['data_provider'] == data_type[0]].reset_index(drop=True).loc[0,'img_path']\nrad_img_path = train[train['data_provider'] == data_type[1]].reset_index(drop=True).loc[0,'img_path']\n\nkar_img = skimage.io.MultiImage(str(kar_img_path))\nrad_img = skimage.io.MultiImage(str(rad_img_path))\n\nremove_image_b = crop_white(np.array(kar_img)[0])\ntile_kar_images = concate_images(tile(remove_image_b))\n\nremove_image_b = crop_white(np.array(rad_img)[0])\ntile_rad_images = concate_images(tile(remove_image_b))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:24:59.291982Z","iopub.execute_input":"2026-03-15T12:24:59.292362Z","iopub.status.idle":"2026-03-15T12:25:05.840266Z","shell.execute_reply.started":"2026-03-15T12:24:59.292332Z","shell.execute_reply":"2026-03-15T12:25:05.839254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig,axs = plt.subplots(1,2)\naxs[0].imshow(tile_kar_images)\naxs[1].imshow(tile_rad_images)\n\n# (개인) 아래 왼쪽은 karolinska, 오른쪽은 radboud은 색이 다름. DL은 이것을 다른 것으로 인식할 수 있어서 color normalization을 해야 함. ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:25:05.841921Z","iopub.execute_input":"2026-03-15T12:25:05.842245Z","iopub.status.idle":"2026-03-15T12:25:06.239215Z","shell.execute_reply.started":"2026-03-15T12:25:05.842217Z","shell.execute_reply":"2026-03-15T12:25:06.237912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"normalizer = staintools.ReinhardColorNormalizer()\nnormalizer.fit(np.array(tile_kar_images))\nreinhard_normalized = normalizer.transform(np.array(tile_rad_images))\nfig,axs = plt.subplots(1,2)\naxs[0].imshow(tile_kar_images)\naxs[1].imshow(reinhard_normalized)\n\n# (개인) 아래 오른쪽을 보면 위에서 같은 이미지에서 색만 바꾸기 \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:25:13.472579Z","iopub.execute_input":"2026-03-15T12:25:13.47293Z","iopub.status.idle":"2026-03-15T12:25:14.105735Z","shell.execute_reply.started":"2026-03-15T12:25:13.4729Z","shell.execute_reply":"2026-03-15T12:25:14.104317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"normalizer = staintools.ReinhardColorNormalizer()\nnormalizer.fit(np.array(tile_rad_images))\nreinhard_normalized = normalizer.transform(np.array(tile_kar_images))\nfig,axs = plt.subplots(1,2)\naxs[0].imshow(reinhard_normalized)\naxs[1].imshow(tile_rad_images)\n\n# (개인) 여러 색깔이 있는 aug여서 좋음. 그렇지만 학습속도가 불안정할 수 있음. \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T12:25:15.967938Z","iopub.execute_input":"2026-03-15T12:25:15.968312Z","iopub.status.idle":"2026-03-15T12:25:16.418615Z","shell.execute_reply.started":"2026-03-15T12:25:15.968283Z","shell.execute_reply":"2026-03-15T12:25:16.41761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# (개인) WSI에서 image patch마다 나타나는 특색 중 하나가 Gleason pattern \n# 3+3 -> 1, 3+4 -> 2, 4+3 -> 3, 4+4 -> 4, 3+5 -> 4, 5+3 -> 4, 4+5 -> 5, 5+4 -> 5, 5+5 -> 5\n# Prostate Cancer? 전립선에서 발생하는 암. 남성에게 흔한 암중에 하나로, 천천히 자라나지만 급격하게 빠르게 퍼지기 \n# Gleason Score? biopsy를 통해서 암의 grade 정도를 판독하기, grade 높을수록 전이가 높음 \n# Gleason Score? 1) 생검을 하고 난후 조직에 H&E 염색을 함 2) WSI에서 흰색 구멍 또는 속이 빈 구조의 경우 Gleason score로 판독 ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Load image ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install openslide-python","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T13:56:29.40384Z","iopub.execute_input":"2026-03-15T13:56:29.404196Z","iopub.status.idle":"2026-03-15T13:56:34.616159Z","shell.execute_reply.started":"2026-03-15T13:56:29.404155Z","shell.execute_reply":"2026-03-15T13:56:34.614979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Path 관련 libary \nfrom pathlib import Path \n\n## WSI 관련 library \nimport openslide \nimport cv2 # (개인) 기본 imaging processing tool \n\n## etc\nimport pandas as pd \nimport numpy as np \nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T13:56:59.959088Z","iopub.execute_input":"2026-03-15T13:56:59.959946Z","iopub.status.idle":"2026-03-15T13:57:00.691482Z","shell.execute_reply.started":"2026-03-15T13:56:59.959903Z","shell.execute_reply":"2026-03-15T13:57:00.690761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 우선적으로 datset을 읽어와줍니다. \nclass cfg: \n    # Location of the training images\n    # (개인) kaggle의 path은 옆에 업로드 된 것에서 copy하기 \n    BASE_PATH = \"/kaggle/input/competitions/prostate-cancer-grade-assessment\"\n    # image and mask directories\n    data_dir = f'{BASE_PATH}/train_images'\n    mask_dir = f'{BASE_PATH}/train_label_masks'\n    train_zip_path = 'train.zip'\n    mask_zip_path = 'mask.zip'\n    sz = 256\n    N = 16","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T13:58:26.348538Z","iopub.execute_input":"2026-03-15T13:58:26.350161Z","iopub.status.idle":"2026-03-15T13:58:26.355388Z","shell.execute_reply.started":"2026-03-15T13:58:26.350121Z","shell.execute_reply":"2026-03-15T13:58:26.354432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Location of training labels\ntrain = pd.read_csv(f'{cfg.BASE_PATH}/train.csv')\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T13:59:10.066574Z","iopub.execute_input":"2026-03-15T13:59:10.06773Z","iopub.status.idle":"2026-03-15T13:59:10.133429Z","shell.execute_reply.started":"2026-03-15T13:59:10.067634Z","shell.execute_reply":"2026-03-15T13:59:10.132746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['img_path'] = train['image_id'].apply(lambda x : Path(cfg.data_dir)/ (x+'.tiff'))\ntrain['mask_path'] = train['image_id'].apply(lambda x : Path(cfg.mask_dir)/ (x+'_mask.tiff'))\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:00:26.680806Z","iopub.execute_input":"2026-03-15T14:00:26.681204Z","iopub.status.idle":"2026-03-15T14:00:26.855932Z","shell.execute_reply.started":"2026-03-15T14:00:26.681163Z","shell.execute_reply":"2026-03-15T14:00:26.854961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# chek NAN value \ntrain.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:00:39.183304Z","iopub.execute_input":"2026-03-15T14:00:39.183607Z","iopub.status.idle":"2026-03-15T14:00:39.194955Z","shell.execute_reply.started":"2026-03-15T14:00:39.183582Z","shell.execute_reply":"2026-03-15T14:00:39.193805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. EDA \n# 기본적인 data의 distribution의 정보를 파악하는 것이 몹시 중요 \n# 데이터의 따른 학습방법을 설정하고 시험을 진행할 수 있음 \n# dataframe에 대해서 본격적인 분석을 진행해보자 ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns # data에 대한 distribution 등 다양한 feature를 분석하기 위해 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:01:58.16759Z","iopub.execute_input":"2026-03-15T14:01:58.168456Z","iopub.status.idle":"2026-03-15T14:01:59.386154Z","shell.execute_reply.started":"2026-03-15T14:01:58.168423Z","shell.execute_reply":"2026-03-15T14:01:59.385119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:11:51.916393Z","iopub.execute_input":"2026-03-15T14:11:51.917158Z","iopub.status.idle":"2026-03-15T14:11:51.927502Z","shell.execute_reply.started":"2026-03-15T14:11:51.917125Z","shell.execute_reply":"2026-03-15T14:11:51.926699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save a palette to a variable:\npalette = sns.color_palette(\"Set2\")\n\n# Use palplot and pass in the variable:\nsns.set_palette(\"Set2\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:11:53.092215Z","iopub.execute_input":"2026-03-15T14:11:53.092519Z","iopub.status.idle":"2026-03-15T14:11:53.098274Z","shell.execute_reply.started":"2026-03-15T14:11:53.092493Z","shell.execute_reply":"2026-03-15T14:11:53.097183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 우선적으로 data provider 즉 data가 2가지 정도 있는데 이는 몇 개의 이미지가 있는지 분석 \nsns.countplot(x = \"data_provider\", data = train, hue=\"data_provider\")\n# 그림의 결과 별 차이 없음 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:13:19.450822Z","iopub.execute_input":"2026-03-15T14:13:19.451551Z","iopub.status.idle":"2026-03-15T14:13:19.611178Z","shell.execute_reply.started":"2026-03-15T14:13:19.451521Z","shell.execute_reply":"2026-03-15T14:13:19.610141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# gt의 label은 균등하게 분포 되어 있는지 확인 \n\nfig, axs = plt.subplots(1,2,figsize = (15,6))\nsns.countplot(y = \"gleason_score\", data = train, ax = axs[0],  hue = 'gleason_score')\naxs[0].set_title('gleason_score')\nsns.countplot(y='gleason_score', hue = 'data_provider', data= train, ax = axs[1])\naxs[1].set_title('gleason_score by data provider')\n\n# (개인) 중증 데이터 5+4는 데이터의 분포가 적음 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:14:15.650823Z","iopub.execute_input":"2026-03-15T14:14:15.651785Z","iopub.status.idle":"2026-03-15T14:14:16.209749Z","shell.execute_reply.started":"2026-03-15T14:14:15.651744Z","shell.execute_reply":"2026-03-15T14:14:16.208732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axs = plt.subplots(1,2,figsize = (10,6))\nsns.countplot(x = \"isup_grade\", data = train, ax = axs[0])\naxs[0].set_title('isup_grade')\nsns.countplot(x='isup_grade', hue = 'data_provider', data= train, ax = axs[1])\naxs[1].set_title('isup grade by data provider')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:13:21.795783Z","iopub.execute_input":"2026-03-15T14:13:21.796587Z","iopub.status.idle":"2026-03-15T14:13:22.229262Z","shell.execute_reply.started":"2026-03-15T14:13:21.796553Z","shell.execute_reply":"2026-03-15T14:13:22.228386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## group마다 distribution을 살펴보자\ndata = [[\"Gleason Score\", \"ISUP Grade\"],\n        [\"0+0\", \"0\"], [\"negative\", \"0\"],\n        [\"3+3\", \"1\"], [\"3+4\", \"2\"], [\"4+3\", \"3\"],\n        [\"4+4\", \"4\"], [\"3+5\", \"4\"], [\"5+3\", \"4\"],\n        [\"4+5\", \"5\"], [\"5+4\", \"5\"], [\"5+5\", \"5\"]]\n## show sample images\n\ntmp = train.groupby('data_provider')['gleason_score'].value_counts()\ndf = pd.DataFrame(data={'Exams': tmp.values}, index=tmp.index).reset_index()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:13:30.44106Z","iopub.execute_input":"2026-03-15T14:13:30.441552Z","iopub.status.idle":"2026-03-15T14:13:30.454231Z","shell.execute_reply.started":"2026-03-15T14:13:30.44152Z","shell.execute_reply":"2026-03-15T14:13:30.453143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axs = plt.subplots(1, 2, figsize=(15,6))\nsns.barplot(x='data_provider', y='Exams', hue='gleason_score', data=df, ax=axs[0])\nsns.countplot(y='gleason_score', data=train, ax=axs[1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:13:32.105904Z","iopub.execute_input":"2026-03-15T14:13:32.106843Z","iopub.status.idle":"2026-03-15T14:13:32.688319Z","shell.execute_reply.started":"2026-03-15T14:13:32.106796Z","shell.execute_reply":"2026-03-15T14:13:32.687589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# result\n# 살펴본 결과, karolinska의 경우 negative, low score일 때 데이터가 많았으며, radboud의 경우 high score 데이터가 일반적으로 많았다. 또한 최종적으로 전체 데이터에서는 3+3의 데이터가 많았고 3+5의 데이터는 작았는데 이를 ㅌ오해서 fold의 나눔에 도움을 줄 것이다 ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. Load image and Analysis \n# 데이터의 분포를 파악했으니 이제 sample 이미지를 읽어보고 어떻게 처리할지 고민 \n# 1. data provider마다 sample image 읽어보기 \n# 2. Gleason score마다 data 읽어보기 \n# 3. mask annotation과 함께 읽어보기 \n# 4. mask와 함께 데이터 tilling을 해보기 ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def multi_plot(df, region=(0,0), level=-1, crop_size=(256,256)): # (개인) level = -1 가장 res가 낮음 \n    f, ax = plt.subplots(3,3, figsize=(15,15))\n    for i, row in enumerate(df.itertuples()):\n        image = openslide.OpenSlide(row.img_path)\n        slevel = image.level_count -1 if level == -1 else level\n        patch = image.read_region(region, slevel, crop_size)\n        ax[i//3, i%3].imshow(patch)\n        image.close()\n\n        ax[i//3, i%3].axis(\"off\")\n\n        image_id = row.image_id\n        data_provider = row.data_provider\n        isup_grade = row.isup_grade\n        gleason_score = row.gleason_score\n\n        ax[i//3, i%3].set_title(\n            f'ID: {image_id}\\nSource: {data_provider} ISUP: {isup_grade} Gleason: {gleason_score}'\n        )\n\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:20:09.305549Z","iopub.execute_input":"2026-03-15T14:20:09.306638Z","iopub.status.idle":"2026-03-15T14:20:09.313778Z","shell.execute_reply.started":"2026-03-15T14:20:09.306603Z","shell.execute_reply":"2026-03-15T14:20:09.312549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df = train.sample(9)\nmulti_plot(sample_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:20:36.646638Z","iopub.execute_input":"2026-03-15T14:20:36.64715Z","iopub.status.idle":"2026-03-15T14:20:38.144218Z","shell.execute_reply.started":"2026-03-15T14:20:36.647119Z","shell.execute_reply":"2026-03-15T14:20:38.143137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:20:50.634413Z","iopub.execute_input":"2026-03-15T14:20:50.635427Z","iopub.status.idle":"2026-03-15T14:20:50.646388Z","shell.execute_reply.started":"2026-03-15T14:20:50.635391Z","shell.execute_reply":"2026-03-15T14:20:50.645374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"id_list = [\n    '037504061b9fba71ef6e24c48c6df44d',\n    '035b1edd3d1aeefcf77ce5d248a01a53',\n    '059cbf902c5e42972587c8d17d49efed',\n    '06a0cbd8fd6320ef1aa6f19342af2e68',\n    '06eda4a6faca84e84a781fee2d5f47e1',\n    '0a4b7a7499ed55c71033cefb0765e93d',\n    '0838c82917cd9af681df249264d2769c',\n    '046b35ae95374bfb48cdca8d7c83233f',\n    '074c3e01525681a275a42282cd21cbde',\n]\n\nslect_df = train[train['image_id'].isin(id_list)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:23:21.909211Z","iopub.execute_input":"2026-03-15T14:23:21.910503Z","iopub.status.idle":"2026-03-15T14:23:21.917451Z","shell.execute_reply.started":"2026-03-15T14:23:21.910465Z","shell.execute_reply":"2026-03-15T14:23:21.916331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nmulti_plot(slect_df, (1780, 1950), 0, (256, 256))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:23:26.41776Z","iopub.execute_input":"2026-03-15T14:23:26.41822Z","iopub.status.idle":"2026-03-15T14:23:28.231289Z","shell.execute_reply.started":"2026-03-15T14:23:26.418189Z","shell.execute_reply":"2026-03-15T14:23:28.229728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib\n\ndef mask_multi_plot(df, region=(0,0), level=-1, crop_size=(256,256)):\n    f, ax = plt.subplots(3,3, figsize=(15,15))\n    for i, row in enumerate(df.itertuples()):\n        image = openslide.OpenSlide(row.mask_path)\n        slevel = image.level_count - 1 if level == -1 else level\n        patch = image.read_region(region, slevel, crop_size)\n\n        cmap = matplotlib.colors.ListedColormap(\n            ['black', 'gray', 'green', 'yellow', 'orange', 'red']\n        # (개인) background black, foreground grey, 등등\n        )\n\n        ax[i//3, i%3].imshow(\n            np.asarray(patch)[:,:,0],\n            cmap=cmap,\n            interpolation=\"nearest\",\n            vmin=0,\n            vmax=5\n        )\n\n        image.close()\n        ax[i//3, i%3].axis(\"off\")\n\n        image_id = row.image_id\n        data_provider = row.data_provider\n        isup_grade = row.isup_grade\n        gleason_score = row.gleason_score\n\n        ax[i//3, i%3].set_title(\n            f'ID: {image_id}\\nSource: {data_provider} ISUP: {isup_grade} Gleason: {gleason_score}'\n        )\n\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:25:07.85691Z","iopub.execute_input":"2026-03-15T14:25:07.858214Z","iopub.status.idle":"2026-03-15T14:25:07.86663Z","shell.execute_reply.started":"2026-03-15T14:25:07.858175Z","shell.execute_reply":"2026-03-15T14:25:07.865726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mask_multi_plot(sample_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:25:10.774314Z","iopub.execute_input":"2026-03-15T14:25:10.774614Z","iopub.status.idle":"2026-03-15T14:25:12.045302Z","shell.execute_reply.started":"2026-03-15T14:25:10.774588Z","shell.execute_reply":"2026-03-15T14:25:12.044391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mask_multi_plot(slect_df, (1780, 1950), 0, (256,256))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:27:01.603793Z","iopub.execute_input":"2026-03-15T14:27:01.604734Z","iopub.status.idle":"2026-03-15T14:27:02.792775Z","shell.execute_reply.started":"2026-03-15T14:27:01.604634Z","shell.execute_reply":"2026-03-15T14:27:02.792005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import PIL\n#https://www.kaggle.com/code/spidyweb/prostate-cancer-panda-insight-eda# reference :\ndef override_image(df, region=(0,0), level=-1, crop_size=(256,256), center='radboud', alpha=0.6, max_size=(800, 800)):\n    f, ax = plt.subplots(3,3, figsize=(15,15))\n    for i, row in enumerate(df.itertuples()):\n\n        slide = openslide.OpenSlide(row.img_path)\n        mask = openslide.OpenSlide(row.mask_path)\n        slevel = slide.level_count - 1 if level == -1 else level\n        slide_data = slide.read_region(region, slevel, crop_size)\n        mask_data = mask.read_region(region, slevel, crop_size)\n        mask_data = mask_data.split()[0]\n\n        # Create alpha mask\n        alpha_int = int(round(255*alpha))\n        if center == 'radboud':\n            alpha_content = np.less(mask_data.split()[0], 2).astype('uint8') * alpha_int + (255 - alpha_int)\n        elif center == 'karolinska':\n            alpha_content = np.less(mask_data.split()[0], 1).astype('uint8') * alpha_int + (255 - alpha_int)\n\n        alpha_content = PIL.Image.fromarray(alpha_content)\n        preview_palette = np.zeros(shape=768, dtype=int)\n\n        if center == 'radboud':\n            # Mapping: {0: background, 1: stroma, 2: benign epithelium, 3: Gleason 3, 4: Gleason 4, 5: Gleason 5}\n            preview_palette[0:18] = (np.array([0, 0, 0, 0.5, 0.5, 0.5, 0, 1, 0, 1, 1, 0.7, 1, 0.5, 0, 1, 0, 0]) * 255).astype(int)\n        elif center == 'karolinska':\n            # Mapping: {0: background, 1: benign, 2: cancer}\n            preview_palette[0:9] = (np.array([0, 0, 0, 1, 0, 1, 0, 0]) * 255).astype(int)\n\n        mask_data.putpalette(data=preview_palette.tolist())\n        mask_rgb = mask_data.convert(mode='RGB')\n        overlayed_image = PIL.Image.composite(image1=slide_data, image2=mask_rgb, mask=alpha_content)\n        overlayed_image.thumbnail(size=max_size, resample=0)\n\n        ax[i//3, i%3].imshow(overlayed_image)\n        slide.close()\n        mask.close()\n        ax[i//3, i%3].axis('off')\n\n        image_id = row.image_id\n        data_provider = row.data_provider\n        isup_grade = row.isup_grade\n        gleason_score = row.gleason_score\n        ax[i//3, i%3].set_title(f\"ID: {image_id}\\nSource: {data_provider} ISUP: {isup_grade} Gleason: {gleason_score}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:29:56.497103Z","iopub.execute_input":"2026-03-15T14:29:56.497812Z","iopub.status.idle":"2026-03-15T14:29:56.509893Z","shell.execute_reply.started":"2026-03-15T14:29:56.49778Z","shell.execute_reply":"2026-03-15T14:29:56.508843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"override_image(sample_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:29:59.407266Z","iopub.execute_input":"2026-03-15T14:29:59.407905Z","iopub.status.idle":"2026-03-15T14:30:01.283351Z","shell.execute_reply.started":"2026-03-15T14:29:59.407871Z","shell.execute_reply":"2026-03-15T14:30:01.282148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"override_image(slect_df, (1780, 1950), 0, (256,256))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:30:43.456585Z","iopub.execute_input":"2026-03-15T14:30:43.457151Z","iopub.status.idle":"2026-03-15T14:30:45.387753Z","shell.execute_reply.started":"2026-03-15T14:30:43.457118Z","shell.execute_reply":"2026-03-15T14:30:45.386443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def print_slide_details(slide, show_thumbnail=True, max_size=(600,400)):\n    \"\"\"Print some basic information about a slide\"\"\"\n    # Generate a small image thumbnail\n    if show_thumbnail:\n        display(slide.get_thumbnail(size=max_size))\n\n    # Here we compute the 'pixel spacing': the physical size of a pixel in the image.\n    # OpenSlide gives the resolution in centimeters so we convert this to microns.\n    spacing = 1 / (float(slide.properties['tiff.XResolution'])) / 10000\n\n    print(f\"File id: {slide}\")\n    print(f\"Dimensions: {slide.dimensions}\")\n    print(f\"Microns per pixel / pixel spacing: {spacing:.3f}\")\n    print(f\"Number of levels in the image: {slide.level_count}\")\n    print(f\"Downsample factor per level: {slide.level_downsamples}\")\n    print(f\"Dimensions of levels: {slide.level_dimensions}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:33:01.430912Z","iopub.execute_input":"2026-03-15T14:33:01.431443Z","iopub.status.idle":"2026-03-15T14:33:01.438561Z","shell.execute_reply.started":"2026-03-15T14:33:01.431409Z","shell.execute_reply":"2026-03-15T14:33:01.437523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for row in sample_df.itertuples():\n    biopsy = openslide.OpenSlide(row.img_path)\n    print_slide_details(biopsy)\n    biopsy.close()\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:33:03.620322Z","iopub.execute_input":"2026-03-15T14:33:03.621088Z","iopub.status.idle":"2026-03-15T14:33:03.884328Z","shell.execute_reply.started":"2026-03-15T14:33:03.621051Z","shell.execute_reply":"2026-03-15T14:33:03.883243Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get all meta data\nfrom tqdm import tqdm\nmeta_dict = {'width':[], 'height':[], 'spacing':[], 'level_count':[]}\nfor row in tqdm(train.itertuples(), total=len(train)):\n    slide = openslide.OpenSlide(row.img_path)\n\n    spacing = 1 / (float(slide.properties['tiff.XResolution'])) / 10000\n\n    meta_dict['width'].append(slide.dimensions[0])\n    meta_dict['height'].append(slide.dimensions[1])\n    meta_dict['spacing'].append(spacing)\n    meta_dict['level_count'].append(slide.level_count)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:34:08.674588Z","iopub.execute_input":"2026-03-15T14:34:08.675302Z","iopub.status.idle":"2026-03-15T14:47:37.211634Z","shell.execute_reply.started":"2026-03-15T14:34:08.675266Z","shell.execute_reply":"2026-03-15T14:47:37.210916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.concat([train, pd.DataFrame(meta_dict)], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:49:06.460244Z","iopub.execute_input":"2026-03-15T14:49:06.460979Z","iopub.status.idle":"2026-03-15T14:49:06.483198Z","shell.execute_reply.started":"2026-03-15T14:49:06.460946Z","shell.execute_reply":"2026-03-15T14:49:06.482275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = plt.figure(figsize=(12,6))\nax = sns.scatterplot(x='width', y='height', data=train_df, alpha=0.3)\nplt.title(\"height(y) width(x) scatter plot\")\nplt.show()\n\n# (개인) 한 이미지당 width가 10000의 size, height은 45000 정도 \n# (개인) outlier들도 있음을 확인 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:49:08.016167Z","iopub.execute_input":"2026-03-15T14:49:08.017089Z","iopub.status.idle":"2026-03-15T14:49:08.257748Z","shell.execute_reply.started":"2026-03-15T14:49:08.017054Z","shell.execute_reply":"2026-03-15T14:49:08.256713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = plt.figure(figsize=(12, 6))\nax = sns.scatterplot(x='width', y='height', hue='isup_grade', data=train_df, alpha=0.6)\nplt.title(\"height(y) width(x) scatter plot with target\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:49:10.598345Z","iopub.execute_input":"2026-03-15T14:49:10.598854Z","iopub.status.idle":"2026-03-15T14:49:11.340441Z","shell.execute_reply.started":"2026-03-15T14:49:10.598822Z","shell.execute_reply":"2026-03-15T14:49:11.339416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make Tile \n# (개인) tiling 관련해서는 challenge dataset에서 검색창에 tiling 관련해서 사람들이 사용한 방식 확인하고 진행 \ndef tile(img, mask):\n    result = []\n    shape = img.shape\n    pad0, pad1 = (cfg.sz - shape[0] % cfg.sz) % cfg.sz, (cfg.sz - shape[1] % cfg.sz) % cfg.sz\n    img = np.pad(img, [[pad0//2, pad0 - pad0//2], [pad1//2, pad1 - pad1//2], [0,0]])\n    mask = np.pad(mask, [[pad0//2, pad0 - pad0//2], [pad1//2, pad1 - pad1//2], [0,0]],\n                  constant_values=0)\n\n    img = img.reshape(img.shape[0]//cfg.sz, cfg.sz, img.shape[1]//cfg.sz, cfg.sz, 3)\n    img = np.transpose(img, (0,2,1,3,4)).reshape(-1, cfg.sz, cfg.sz, 3)\n    mask = mask.reshape(mask.shape[0]//cfg.sz, cfg.sz, mask.shape[1]//cfg.sz, cfg.sz, 3)\n    mask = np.transpose(mask, (0,2,1,3,4)).reshape(-1, cfg.sz, cfg.sz, 3)\n\n    if len(img) < cfg.N:\n        mask = np.pad(mask, [[0, cfg.N-len(img)], [0,0], [0,0], [0,0]], constant_values=0)\n        img = np.pad(img, [[0, cfg.N-len(img)], [0,0], [0,0], [0,0]], constant_values=255)\n\n    idxs = np.argsort(img.reshape(img.shape[0], -1).sum(-1))[:cfg.N]\n    img = img[idxs]\n    mask = mask[idxs]\n\n    for i in range(len(img)):\n        result.append({'img':img[i], 'mask':mask[i], 'idx':i})\n\n    return result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:49:19.005442Z","iopub.execute_input":"2026-03-15T14:49:19.005812Z","iopub.status.idle":"2026-03-15T14:49:19.016351Z","shell.execute_reply.started":"2026-03-15T14:49:19.005783Z","shell.execute_reply":"2026-03-15T14:49:19.015195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\nimport skimage\n\ndef save_image(row):\n    img = skimage.io.MultiImage(str(row.img_path))[-1]\n    mask = skimage.io.MultiImage(str(row.mask_path))[-1]\n    tiles = tile(img,mask)\n    for t in tiles:\n        img,mask,idx = t['img'],t['mask'],t['idx']\n        x_tot.append((img/255.0).reshape(-1,3).mean(0))\n        x2_tot.append(((img/255.0)**2).reshape(-1,3).mean(0))\n        #if read with PIL RGB turns to BGR\n        img = cv2.imencode('.png', cv2.cvtColor(img, cv2.COLOR_RGB2BGR))[1]\n        img_out.writestr(f'{row.image_id}_{idx}.png', img)\n        mask = cv2.imencode('.png', mask[:,:,0])[1]\n        mask_out.writestr(f'{row.image_id}_{idx}.png', mask)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:49:21.276408Z","iopub.execute_input":"2026-03-15T14:49:21.277102Z","iopub.status.idle":"2026-03-15T14:49:21.286865Z","shell.execute_reply.started":"2026-03-15T14:49:21.277069Z","shell.execute_reply":"2026-03-15T14:49:21.286016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from joblib import Parallel, delayed\n# faster = False\n# x_tot,x2_tot = [],[]\n\n# with zipfile.ZipFile(cfg.train_zip_path, 'w') as img_out, \\\n#      zipfile.ZipFile(cfg.mask_zip_path, 'w') as mask_out:\n\n#     pbar = tqdm(train_df.iterrows(), total=len(train_df))\n#     [save_image(row) for row in pbar]\n         \n# (개인) 이것 돌리면 하루 반정도 걸림","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CH04-06.\n\n# deep learning model 관련 \nimport torch \nimport random \nimport os \nimport skimage \nimport sys \n\nfrom tqdm.notebook import tqdm\nfrom sklearn.metrics import cohen_kappa_score # metric \ntqdm.pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T14:50:10.572613Z","iopub.execute_input":"2026-03-15T14:50:10.573088Z","iopub.status.idle":"2026-03-15T14:50:10.753992Z","shell.execute_reply.started":"2026-03-15T14:50:10.573046Z","shell.execute_reply":"2026-03-15T14:50:10.753215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# tiling -> pathology 하나에 대해서 몇 장의 이미지를 쓰는지에 관해 \nclass cfg:\n    img_size = 128\n    sz = 128\n    bs = 2\n    n_tiles = 36\n\n    BASE_PATH = '/kaggle/input/competitions/prostate-cancer-grade-assessment'\n\n    data_dir = f'{BASE_PATH}/train_images'\n    mask_dir = f'{BASE_PATH}/train_label_masks'\n\n    tile_img_path = '/kaggle/input/datasets/iafoss/panda-16x128x128-tiles-data/train'\n    tile_mask_path = '/kaggle/input/datasets/iafoss/panda-16x128x128-tiles-data/masks'\n    make_patch = False\n    do_train = True\n    do_infer = True\n    n_folds = 6\n    fold = 0\n    out_cls = 5\n    seed = 2024\n    lr = 3e-4\n    warmup_epo = 1\n    n_epochs = 10\n    tile_size = 256","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:03:29.560199Z","iopub.execute_input":"2026-03-15T15:03:29.560909Z","iopub.status.idle":"2026-03-15T15:03:29.567332Z","shell.execute_reply.started":"2026-03-15T15:03:29.560876Z","shell.execute_reply":"2026-03-15T15:03:29.566376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Location of training labels\ntrain = pd.read_csv(f'{cfg.BASE_PATH}/train.csv')\ntrain.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:03:31.669963Z","iopub.execute_input":"2026-03-15T15:03:31.670261Z","iopub.status.idle":"2026-03-15T15:03:31.697074Z","shell.execute_reply.started":"2026-03-15T15:03:31.670235Z","shell.execute_reply":"2026-03-15T15:03:31.696253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['img_path'] = train['image_id'].apply(lambda x : Path(cfg.data_dir) / (x+'.tiff'))\ntrain['mask_path'] = train['image_id'].apply(lambda x : Path(cfg.mask_dir) / (x+'.tiff'))\ntrain.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:03:34.005367Z","iopub.execute_input":"2026-03-15T15:03:34.005868Z","iopub.status.idle":"2026-03-15T15:03:34.336011Z","shell.execute_reply.started":"2026-03-15T15:03:34.005836Z","shell.execute_reply":"2026-03-15T15:03:34.335128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. Make dataset & Data loader\n\n# • pathology처럼 큰 이미지의 경우 patch로 잘라서 학습을 하는 과정을 거친다.\n# • 따로 저장을 하여서 사용하는 방법도 있으며 바로 iteration마다 patch를 잘라서 학습을 하는 방법이 있다.\n# • kaggle 에서는 side의 bar에서 이전에 저장했던 데이터를 쉽게 불러오고 이미 작업을 하였던 데이터에 대해서 쉽게 불러올 수 있습니다.\n\n# 2.1 Splite kfold\n\n# • 학습을 하기 위해서는 내부 validation과 traind을 fold를 나누어서 학습을 해야한다.","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\ndf = train.set_index('image_id')\nfiles = sorted(set([str(p.name)[:32] for p in list(Path(cfg.tile_img_path).glob('*'))]))\ndf = df.loc[files]\ndf = df.reset_index()\n\nsplits = StratifiedKFold(n_splits=cfg.n_folds, shuffle=True)\nsplits = list(splits.split(df, df.isup_grade))\n\nfolds_splits = np.zeros(len(df)).astype(int)\n\nfor i in range(cfg.n_folds):\n    folds_splits[splits[i][1]] = i\n\ndf['split'] = folds_splits","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:04:38.199375Z","iopub.execute_input":"2026-03-15T15:04:38.199906Z","iopub.status.idle":"2026-03-15T15:04:38.223149Z","shell.execute_reply.started":"2026-03-15T15:04:38.199871Z","shell.execute_reply":"2026-03-15T15:04:38.222003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(cfg.tile_img_path)\nprint(Path(cfg.tile_img_path).exists())\nprint(len(list(Path(cfg.tile_img_path).glob('*'))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:05:22.673933Z","iopub.execute_input":"2026-03-15T15:05:22.674784Z","iopub.status.idle":"2026-03-15T15:05:27.048415Z","shell.execute_reply.started":"2026-03-15T15:05:22.674742Z","shell.execute_reply":"2026-03-15T15:05:27.047523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nprint(len(list(Path(cfg.tile_img_path).glob('*'))))\nprint(list(Path(cfg.tile_img_path).glob('*'))[:10])\nprint(train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:05:50.355319Z","iopub.execute_input":"2026-03-15T15:05:50.355621Z","iopub.status.idle":"2026-03-15T15:05:58.845064Z","shell.execute_reply.started":"2026-03-15T15:05:50.355596Z","shell.execute_reply":"2026-03-15T15:05:58.844191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\ndf = train.set_index('image_id')\nfiles = sorted(set([str(p.name)[:32] for p in list(Path(cfg.tile_img_path).glob('*'))]))\n\nprint('num tile files:', len(list(Path(cfg.tile_img_path).glob('*'))))\nprint('num unique image ids from tiles:', len(files))\nprint('first 5 file ids:', files[:5])\nprint('first 5 train ids:', train['image_id'].head().tolist())\nprint('overlap:', len(set(files) & set(train['image_id'])))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:06:46.325157Z","iopub.execute_input":"2026-03-15T15:06:46.325862Z","iopub.status.idle":"2026-03-15T15:06:54.128987Z","shell.execute_reply.started":"2026-03-15T15:06:46.325827Z","shell.execute_reply":"2026-03-15T15:06:54.128068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom pathlib import Path\nimport numpy as np\n\ndf = train.set_index('image_id')\nfiles = sorted(set([p.name[:32] for p in Path(cfg.tile_img_path).glob('*')]))\ndf = df.loc[df.index.intersection(files)].reset_index()\n\nprint(df.shape)\n\nsplits = StratifiedKFold(n_splits=cfg.n_folds, shuffle=True, random_state=cfg.seed)\nsplits = list(splits.split(df, df.isup_grade))\n\nfolds_splits = np.zeros(len(df), dtype=int)\nfor i in range(cfg.n_folds):\n    folds_splits[splits[i][1]] = i\n\ndf['split'] = folds_splits\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:07:45.584815Z","iopub.execute_input":"2026-03-15T15:07:45.58515Z","iopub.status.idle":"2026-03-15T15:07:48.870721Z","shell.execute_reply.started":"2026-03-15T15:07:45.585123Z","shell.execute_reply":"2026-03-15T15:07:48.869933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing = sorted(set(train['image_id']) - set(files))\nprint(len(missing))\nprint(missing[:10])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:07:58.775702Z","iopub.execute_input":"2026-03-15T15:07:58.776016Z","iopub.status.idle":"2026-03-15T15:07:58.785207Z","shell.execute_reply.started":"2026-03-15T15:07:58.77599Z","shell.execute_reply":"2026-03-15T15:07:58.784102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom pathlib import Path\nimport numpy as np\n\ndf = train.set_index('image_id')\nfiles = sorted(set([p.name[:32] for p in Path(cfg.tile_img_path).glob('*')]))\ndf = df.loc[df.index.intersection(files)].reset_index()\n\nsplits = StratifiedKFold(n_splits=cfg.n_folds, shuffle=True, random_state=cfg.seed)\nsplits = list(splits.split(df, df.isup_grade))\n\nfolds_splits = np.zeros(len(df), dtype=int)\nfor i in range(cfg.n_folds):\n    folds_splits[splits[i][1]] = i\n\ndf['split'] = folds_splits\nprint(df.shape)\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:08:44.802221Z","iopub.execute_input":"2026-03-15T15:08:44.802822Z","iopub.status.idle":"2026-03-15T15:08:48.030709Z","shell.execute_reply.started":"2026-03-15T15:08:44.80279Z","shell.execute_reply":"2026-03-15T15:08:48.029728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DEBUG = False\ncfg.n_epophs = 1 if DEBUG else 30 \ndf = df.sample(100).reset_index(drop = True) if DEBUG else df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:10:16.207476Z","iopub.execute_input":"2026-03-15T15:10:16.208555Z","iopub.status.idle":"2026-03-15T15:10:16.213207Z","shell.execute_reply.started":"2026-03-15T15:10:16.208518Z","shell.execute_reply":"2026-03-15T15:10:16.212319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:10:21.648181Z","iopub.execute_input":"2026-03-15T15:10:21.648962Z","iopub.status.idle":"2026-03-15T15:10:21.659342Z","shell.execute_reply.started":"2026-03-15T15:10:21.648917Z","shell.execute_reply":"2026-03-15T15:10:21.65812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\n\nseed_everything(cfg.seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:11:31.131514Z","iopub.execute_input":"2026-03-15T15:11:31.132409Z","iopub.status.idle":"2026-03-15T15:11:31.144356Z","shell.execute_reply.started":"2026-03-15T15:11:31.132373Z","shell.execute_reply":"2026-03-15T15:11:31.143333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import DataLoader, Dataset\nfrom torch.utils.data.sampler import SubsetRandomSampler, RandomSampler, SequentialSampler\nimport albumentations as A","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-15T15:11:33.916328Z","iopub.execute_input":"2026-03-15T15:11:33.917381Z","iopub.status.idle":"2026-03-15T15:11:35.616721Z","shell.execute_reply.started":"2026-03-15T15:11:33.917338Z","shell.execute_reply":"2026-03-15T15:11:35.615947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}