{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Library","metadata":{}},{"cell_type":"code","source":"!conda install -y --channel conda-forge pyvips","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:32:41.879228Z","iopub.execute_input":"2022-07-11T16:32:41.879724Z","iopub.status.idle":"2022-07-11T16:35:07.787095Z","shell.execute_reply.started":"2022-07-11T16:32:41.879619Z","shell.execute_reply":"2022-07-11T16:35:07.785248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport zipfile\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\n\nimport cv2\nimport pyvips","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:35:07.790848Z","iopub.execute_input":"2022-07-11T16:35:07.791213Z","iopub.status.idle":"2022-07-11T16:35:08.333342Z","shell.execute_reply.started":"2022-07-11T16:35:07.791178Z","shell.execute_reply":"2022-07-11T16:35:08.332112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Load","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\nother = pd.read_csv('../input/mayo-clinic-strip-ai/other.csv')\ntest = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\nsubmission = pd.read_csv('../input/mayo-clinic-strip-ai/sample_submission.csv')\n\nprint(train.shape)\ndisplay(train.head())\nprint(other.shape)\ndisplay(other.head())\nprint(test.shape)\ndisplay(test.head())\nprint(submission.shape)\ndisplay(submission.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:35:08.335217Z","iopub.execute_input":"2022-07-11T16:35:08.335646Z","iopub.status.idle":"2022-07-11T16:35:08.427769Z","shell.execute_reply.started":"2022-07-11T16:35:08.335601Z","shell.execute_reply":"2022-07-11T16:35:08.426623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare dataset","metadata":{}},{"cell_type":"markdown","source":"Images are very large, so tile method that worked [previous competition](https://www.kaggle.com/competitions/prostate-cancer-grade-assessment/discussion/146855) would be effective.","metadata":{}},{"cell_type":"code","source":"def tile(img, sz=128, N=16):\n    shape = img.shape\n    pad0,pad1 = (sz - shape[0]%sz)%sz, (sz - shape[1]%sz)%sz\n    img = np.pad(img,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]],constant_values=255)\n    img = img.reshape(img.shape[0]//sz,sz,img.shape[1]//sz,sz,3)\n    img = img.transpose(0,2,1,3,4).reshape(-1,sz,sz,3)\n    if len(img) < N:\n        img = np.pad(img,[[0,N-len(img)],[0,0],[0,0],[0,0]],constant_values=255)\n    idxs = np.argsort(img.reshape(img.shape[0],-1).sum(-1))[:N]\n    img = img[idxs]\n    return img","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:35:08.430546Z","iopub.execute_input":"2022-07-11T16:35:08.43097Z","iopub.status.idle":"2022-07-11T16:35:08.442246Z","shell.execute_reply.started":"2022-07-11T16:35:08.430928Z","shell.execute_reply":"2022-07-11T16:35:08.440971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_dataset(\n    df: pd.DataFrame, \n    N=16,\n    max_size=20000, \n    crop_size=1024, \n    image_dir='../input/mayo-clinic-strip-ai/train', \n    out_dir='train_images.zip',\n):\n    format_to_dtype = {\n       'uchar': np.uint8,\n       'char': np.int8,\n       'ushort': np.uint16,\n       'short': np.int16,\n       'uint': np.uint32,\n       'int': np.int32,\n       'float': np.float32,\n       'double': np.float64,\n       'complex': np.complex64,\n       'dpcomplex': np.complex128,\n    }\n    def vips2numpy(vi):\n        return np.ndarray(\n            buffer=vi.write_to_memory(),\n            dtype=format_to_dtype[vi.format],\n            shape=[vi.height, vi.width, vi.bands])\n    with zipfile.ZipFile(out_dir, \"w\") as out_image:\n        tk0 = tqdm(enumerate(df[\"image_id\"].values), total=len(df))\n        for i, image_id in tk0:\n            print(f\"[{i+1}/{len(df)}] image_id: {image_id}\")\n            image = pyvips.Image.thumbnail(f'{image_dir}/{image_id}.tif', max_size)\n            image = vips2numpy(image)\n            width, height, c = image.shape\n            print(f\"Input width: {width} height: {height}\")\n            images = tile(image, sz=crop_size, N=N)\n            for idx, img in enumerate(images):\n                img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)\n                img = cv2.imencode(\".jpg\", img, [cv2.IMWRITE_JPEG_QUALITY, 100])[1]\n                out_image.writestr(f\"{image_id}_{idx}.jpg\", img)\n            del img, image, images; gc.collect()\n\ni = 1 # 1~8\n\nif i == 8:\n    df = train[(i-1)*100:]\nelse:\n    df = train[(i-1)*100:i*100]\n\nsave_dataset(\n    df,\n    N=16, \n    max_size=20000,\n    crop_size=1024, \n    image_dir='../input/mayo-clinic-strip-ai/train', \n    out_dir=f'train_images_{i}.zip'\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:35:08.444335Z","iopub.execute_input":"2022-07-11T16:35:08.444643Z"},"trusted":true},"execution_count":null,"outputs":[]}]}