{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Library","metadata":{}},{"cell_type":"code","source":"!conda install ../input/pyvips-install/pyvips/*.tar.bz2","metadata":{"execution":{"iopub.status.busy":"2022-09-11T07:15:56.545499Z","iopub.execute_input":"2022-09-11T07:15:56.546452Z","iopub.status.idle":"2022-09-11T07:16:48.767604Z","shell.execute_reply.started":"2022-09-11T07:15:56.546337Z","shell.execute_reply":"2022-09-11T07:16:48.766356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport zipfile\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\n\nimport cv2\nimport pyvips\nimport scipy.stats\nimport random","metadata":{"execution":{"iopub.status.busy":"2022-09-11T07:16:48.769777Z","iopub.execute_input":"2022-09-11T07:16:48.770149Z","iopub.status.idle":"2022-09-11T07:16:49.940448Z","shell.execute_reply.started":"2022-09-11T07:16:48.770112Z","shell.execute_reply":"2022-09-11T07:16:49.939205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_datasets import KaggleDatasets\nKaggleDatasets().get_gcs_path(\"mayo-jpg-dataset-4x-downsampled\")","metadata":{"execution":{"iopub.status.busy":"2022-09-11T07:16:49.942477Z","iopub.execute_input":"2022-09-11T07:16:49.943029Z","iopub.status.idle":"2022-09-11T07:17:12.739548Z","shell.execute_reply.started":"2022-09-11T07:16:49.942975Z","shell.execute_reply":"2022-09-11T07:17:12.738328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Load","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/mayo-jpg-dataset-4x-downsampled/train.csv')\nother = pd.read_csv('../input/mayo-jpg-dataset-4x-downsampled/other.csv')\n# test = pd.read_csv('../input/mayo-jpg-dataset-4x-downsampled/test.csv')\n# submission = pd.read_csv('../input/mayo-jpg-dataset-4x-downsampled/sample_submission.csv')\n\nprint(train.shape)\ndisplay(train.head())\nprint(other.shape)\ndisplay(other.head())\n# print(test.shape)\n# display(test.head())\n# print(submission.shape)\n# display(submission.head())","metadata":{"execution":{"iopub.status.busy":"2022-09-11T07:17:12.74268Z","iopub.execute_input":"2022-09-11T07:17:12.743073Z","iopub.status.idle":"2022-09-11T07:17:12.818429Z","shell.execute_reply.started":"2022-09-11T07:17:12.743039Z","shell.execute_reply":"2022-09-11T07:17:12.817603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare dataset","metadata":{}},{"cell_type":"markdown","source":"Images are very large, so tile method that worked [previous competition](https://www.kaggle.com/competitions/prostate-cancer-grade-assessment/discussion/146855) would be effective.","metadata":{}},{"cell_type":"code","source":"def tile(img, sz=128, N=16):\n    shape = img.shape\n    pad0,pad1 = (sz - shape[0]%sz)%sz, (sz - shape[1]%sz)%sz\n    img = np.pad(img,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]],constant_values=255)\n    img = img.reshape(img.shape[0]//sz,sz,img.shape[1]//sz,sz,3)\n    img = img.transpose(0,2,1,3,4).reshape(-1,sz,sz,3)\n    if len(img) < N:\n        img = np.pad(img,[[0,N-len(img)],[0,0],[0,0],[0,0]],constant_values=255)\n    scores = []\n    for im in img:\n        scores.append(len(cv2.imencode(\".jpg\", im)[1]))\n#     idxs = np.argsort(img.reshape(img.shape[0],-1).sum(-1))[:N]\n#     img = img[idxs]\n    scores, img = zip(*sorted(zip(scores, img), reverse=True, key=lambda x: x[0]))\n    high_info_ind = pd.Series(scores).where(pd.Series(scores) > 30000).idxmin()\n#     print(high_info_ind is not np.nan)\n    \n    bg_ind = pd.Series(scores).where(pd.Series(scores) > 10000).idxmin()\n    bg_cand = img[bg_ind:]\n    for bgi, bg in enumerate(bg_cand):\n        bg = bg.reshape(bg.shape[0] * bg.shape[1], bg.shape[2])\n        white, _ = scipy.stats.mode(bg, axis=0)\n        diff_to_white = (255,255,255) - white\n        if sum(diff_to_white[0]) < 128*3:\n            break\n        else:\n            diff_to_white = [[0,0,0]]\n    \n    if high_info_ind < N or high_info_ind is np.nan:\n        return img[:N], bg_cand[bgi], diff_to_white\n    else:\n        high_info_indexes = random.sample(list(range(high_info_ind)), N)\n        img2 = []\n        for i in high_info_indexes:\n            img2.append(img[i])\n        return img2, bg_cand[bgi], diff_to_white\n        ","metadata":{"execution":{"iopub.status.busy":"2022-09-11T07:17:12.82003Z","iopub.execute_input":"2022-09-11T07:17:12.820728Z","iopub.status.idle":"2022-09-11T07:17:12.839699Z","shell.execute_reply.started":"2022-09-11T07:17:12.820691Z","shell.execute_reply":"2022-09-11T07:17:12.83811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# format_to_dtype = {\n#    'uchar': np.uint8,\n#    'char': np.int8,\n#    'ushort': np.uint16,\n#    'short': np.int16,\n#    'uint': np.uint32,\n#    'int': np.int32,\n#    'float': np.float32,\n#    'double': np.float64,\n#    'complex': np.complex64,\n#    'dpcomplex': np.complex128,\n# }\n\n# def vips2numpy(vi):\n#     return np.ndarray(\n#         buffer=vi.write_to_memory(),\n#         dtype=format_to_dtype[vi.format],\n#         shape=[vi.height, vi.width, vi.bands])\n# sz=128\n# N=6\n# img = pyvips.Image.thumbnail(f'../input/mayo-jpg-dataset-4x-downsampled/train/006388_0.jpg', 20000)\n# img = vips2numpy(img)\n# shape = img.shape\n# pad0,pad1 = (sz - shape[0]%sz)%sz, (sz - shape[1]%sz)%sz\n# img = np.pad(img,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]],constant_values=255)\n# img = img.reshape(img.shape[0]//sz,sz,img.shape[1]//sz,sz,3)\n# img = img.transpose(0,2,1,3,4).reshape(-1,sz,sz,3)\n# if len(img) < N:\n#     img = np.pad(img,[[0,N-len(img)],[0,0],[0,0],[0,0]],constant_values=255)\n# scores = []\n# for im in img:\n#     scores.append(len(cv2.imencode(\".jpg\", im)[1]))\n# #     idxs = np.argsort(img.reshape(img.shape[0],-1).sum(-1))[:N]\n# #     img = img[idxs]\n# scores, img = zip(*sorted(zip(scores, img), reverse=True, key=lambda x: x[0]))\n","metadata":{"execution":{"iopub.status.busy":"2022-09-11T07:17:12.841668Z","iopub.execute_input":"2022-09-11T07:17:12.842075Z","iopub.status.idle":"2022-09-11T07:17:12.857354Z","shell.execute_reply.started":"2022-09-11T07:17:12.842039Z","shell.execute_reply":"2022-09-11T07:17:12.856048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def vips2numpy(vi):\n#     return np.ndarray(\n#         buffer=vi.write_to_memory(),\n#         dtype=format_to_dtype[vi.format],\n#         shape=[vi.height, vi.width, vi.bands])\n\n# format_to_dtype = {\n#    'uchar': np.uint8,\n#    'char': np.int8,\n#    'ushort': np.uint16,\n#    'short': np.int16,\n#    'uint': np.uint32,\n#    'int': np.int32,\n#    'float': np.float32,\n#    'double': np.float64,\n#    'complex': np.complex64,\n#    'dpcomplex': np.complex128,\n# }\n\n# image = pyvips.Image.thumbnail(f'../input/mayo-jpg-dataset-4x-downsampled/train/7b9aaa_0.jpg', 30000)\n# image = vips2numpy(image)\n# tile(image, sz=384, N=64)","metadata":{"execution":{"iopub.status.busy":"2022-09-11T07:17:12.859323Z","iopub.execute_input":"2022-09-11T07:17:12.86012Z","iopub.status.idle":"2022-09-11T07:17:12.873978Z","shell.execute_reply.started":"2022-09-11T07:17:12.86008Z","shell.execute_reply":"2022-09-11T07:17:12.872915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_dataset(\n    df: pd.DataFrame, \n    N=16,\n    max_size=30000, \n    crop_size=384, \n    image_dir='../input/mayo-jpg-dataset-4x-downsampled/train', \n    out_dir='train_images.zip',\n):\n    format_to_dtype = {\n       'uchar': np.uint8,\n       'char': np.int8,\n       'ushort': np.uint16,\n       'short': np.int16,\n       'uint': np.uint32,\n       'int': np.int32,\n       'float': np.float32,\n       'double': np.float64,\n       'complex': np.complex64,\n       'dpcomplex': np.complex128,\n    }\n    def vips2numpy(vi):\n        return np.ndarray(\n            buffer=vi.write_to_memory(),\n            dtype=format_to_dtype[vi.format],\n            shape=[vi.height, vi.width, vi.bands])\n    with zipfile.ZipFile(out_dir, \"w\") as out_image:\n        tk0 = tqdm(enumerate(df[\"image_id\"].values), total=len(df))\n        for i, image_id in tk0:\n            print(f\"[{i+1}/{len(df)}] image_id: {image_id}\")\n            image = pyvips.Image.thumbnail(f'{image_dir}/{image_id}.tif', max_size)\n#             image = pyvips.Image.thumbnail(f'{image_dir}/{image_id}.jpg', max_size)\n\n            image = vips2numpy(image)\n            width, height, c = image.shape\n            print(f\"Input width: {width} height: {height}\")\n            images, bg, diff_to_white = tile(image, sz=crop_size, N=N)\n            out_image.writestr(f\"{image_id}_bg.jpg\", cv2.imencode(\".jpg\", bg, [cv2.IMWRITE_JPEG_QUALITY, 100])[1])\n            print(diff_to_white)\n            for idx, img in enumerate(images):\n                img = img + diff_to_white\n                img = img / img.max()\n                img = np.clip(img * 255, a_min = 0, a_max = 255).astype(np.uint8)\n                img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)\n                img = cv2.imencode(\".jpg\", img, [cv2.IMWRITE_JPEG_QUALITY, 100])[1]\n                out_image.writestr(f\"{image_id}_{idx}.jpg\", img)\n            del img, image, images; gc.collect()\n\n\nfor i in range(1, 9):\n# for i in range(8, 9):\n\n    if i == 8:\n        df = train[(i-1)*100:]\n    else:\n        df = train[(i-1)*100:i*100]\n\n    save_dataset(\n        df,\n        N=64, \n        max_size=20000,\n        crop_size=384, \n        image_dir='../input/mayo-clinic-strip-ai/train', \n        out_dir=f'train_images_{i}.zip'\n    )","metadata":{"execution":{"iopub.status.busy":"2022-09-11T07:17:57.365292Z","iopub.execute_input":"2022-09-11T07:17:57.366196Z","iopub.status.idle":"2022-09-11T08:24:22.968188Z","shell.execute_reply.started":"2022-09-11T07:17:57.366134Z","shell.execute_reply":"2022-09-11T08:24:22.965719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir train_images","metadata":{"execution":{"iopub.status.busy":"2022-09-11T07:17:13.05876Z","iopub.status.idle":"2022-09-11T07:17:13.059151Z","shell.execute_reply.started":"2022-09-11T07:17:13.058968Z","shell.execute_reply":"2022-09-11T07:17:13.058986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mv *.zip ./train_images/","metadata":{"execution":{"iopub.status.busy":"2022-09-11T07:17:13.060973Z","iopub.status.idle":"2022-09-11T07:17:13.061574Z","shell.execute_reply.started":"2022-09-11T07:17:13.061263Z","shell.execute_reply":"2022-09-11T07:17:13.06129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar -cvf train_images.tar train_images","metadata":{"execution":{"iopub.status.busy":"2022-09-11T07:17:13.063116Z","iopub.status.idle":"2022-09-11T07:17:13.063692Z","shell.execute_reply.started":"2022-09-11T07:17:13.063407Z","shell.execute_reply":"2022-09-11T07:17:13.063432Z"},"trusted":true},"execution_count":null,"outputs":[]}]}