{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Downscale Train Images\n\nThis notebook converts the train slides into downscaled (x30) png images using OpenSlide regions and multiprocessing.\n\n[Dataset](https://www.kaggle.com/datasets/basfest/mayo-clinic-strip-ai-train-downscale-x30-png)","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nimport sys\nimport math\nfrom glob import glob\n\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nfrom openslide import OpenSlide\n\nimport time\nimport multiprocessing\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-14T14:30:50.296714Z","iopub.execute_input":"2022-07-14T14:30:50.297146Z","iopub.status.idle":"2022-07-14T14:30:50.423538Z","shell.execute_reply.started":"2022-07-14T14:30:50.297059Z","shell.execute_reply":"2022-07-14T14:30:50.422359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# current directory\ncurrent_path = os.getcwd()\nprint(\"Current Directory:\", current_path)\n\n# parent directory\nparent_path = os.path.abspath(os.path.join(current_path, os.pardir))\nprint(\"Parent Directory:\", parent_path)\n\n# data directory\ndata_path = os.path.join(parent_path, 'input', 'mayo-clinic-strip-ai')\nprint(\"Data Directory:\", data_path)\n\n# images directory\nimages_path = os.path.join(current_path, 'mayo-clinic-strip-ai-png', 'train')\nprint(\"Images Directory:\", images_path)\n\n# dataset directory\ndataset_path = os.path.join(parent_path, 'input', 'mayo-clinic-strip-ai-preprocessing', 'mayo-clinic-strip-ai-png', 'train')\nprint(\"Dataset Directory:\", dataset_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:34:29.097886Z","iopub.execute_input":"2022-07-14T14:34:29.098271Z","iopub.status.idle":"2022-07-14T14:34:29.106525Z","shell.execute_reply.started":"2022-07-14T14:34:29.098239Z","shell.execute_reply":"2022-07-14T14:34:29.105277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists(images_path):\n    os.makedirs(images_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:58:52.562959Z","iopub.execute_input":"2022-07-13T16:58:52.563697Z","iopub.status.idle":"2022-07-13T16:58:52.569032Z","shell.execute_reply.started":"2022-07-13T16:58:52.563658Z","shell.execute_reply":"2022-07-13T16:58:52.567827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(os.path.join(data_path, 'train.csv'))\ntest_df = pd.read_csv(os.path.join(data_path, 'test.csv'))\nother_df = pd.read_csv(os.path.join(data_path, 'other.csv'))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:33:14.52324Z","iopub.execute_input":"2022-07-14T14:33:14.523727Z","iopub.status.idle":"2022-07-14T14:33:14.55283Z","shell.execute_reply.started":"2022-07-14T14:33:14.523674Z","shell.execute_reply":"2022-07-14T14:33:14.55162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images = glob(os.path.join(data_path, 'train', '*.tif'))\nlen(train_images)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:33:15.650266Z","iopub.execute_input":"2022-07-14T14:33:15.650664Z","iopub.status.idle":"2022-07-14T14:33:15.833207Z","shell.execute_reply.started":"2022-07-14T14:33:15.650632Z","shell.execute_reply":"2022-07-14T14:33:15.831931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Resize Slides\n\nResizing the slides by region (crop) allows processing very large images without memory issues.","metadata":{}},{"cell_type":"code","source":"def resize_slide(image_path, level=0, region_size=(15000, 15000), factor=30):\n\n    with OpenSlide(image_path) as slide:\n        \n        width, height = slide.dimensions\n        w, h = region_size\n        resize_w, resize_h = w // factor, h // factor\n        \n        resized_image = Image.new('RGB', (width // factor, height // factor))\n        \n        if slide.level_count != 1:\n            print(\"level count != 1: \" + image_path)\n        \n        x_offset = 0\n        y_offset = 0\n        \n        while y_offset < height:\n            region = slide.read_region((x_offset, y_offset), level, region_size)\n            resized_region = region.resize((resize_w, resize_h), Image.Resampling.LANCZOS)\n            \n            resized_image.paste(resized_region, (x_offset // factor, y_offset // factor))\n            x_offset += w\n            \n            if x_offset > width:\n                x_offset = 0\n                y_offset += h\n        \n        return resized_image\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:33:17.112273Z","iopub.execute_input":"2022-07-14T14:33:17.112653Z","iopub.status.idle":"2022-07-14T14:33:17.122967Z","shell.execute_reply.started":"2022-07-14T14:33:17.112621Z","shell.execute_reply":"2022-07-14T14:33:17.122096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def task(image_paths):\n    for image_path in image_paths:\n        resized_image = resize_slide(image_path)\n        _, filename = os.path.split(image_path)\n        filename, _ = os.path.splitext(filename)\n        resized_image.save(os.path.join(images_path, f\"{filename}.png\"), \"PNG\")","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:33:35.288569Z","iopub.execute_input":"2022-07-14T14:33:35.288939Z","iopub.status.idle":"2022-07-14T14:33:35.294334Z","shell.execute_reply.started":"2022-07-14T14:33:35.288909Z","shell.execute_reply":"2022-07-14T14:33:35.293604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Multiprocessing\n\nParallelization of the slide downscaling process to make sure it runs within the limit of 12 hours.","metadata":{}},{"cell_type":"code","source":"start_time = time.perf_counter()\nprocesses = []\n\n# creates 8 processes then starts them\nfor i in range(8):\n    # 95 images per process ensures that we downscale all 754 slides\n    p = multiprocessing.Process(target = lambda: task(train_images[i*95:(i+1)*95]))\n    p.start()\n    processes.append(p)\n\n# joins all the processes\nfor p in processes:\n    p.join()\n\nfinish_time = time.perf_counter()\n\nprint(f\"Program finished in {finish_time-start_time} seconds\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:24:26.642488Z","iopub.execute_input":"2022-07-13T16:24:26.643206Z","iopub.status.idle":"2022-07-13T16:30:49.649102Z","shell.execute_reply.started":"2022-07-13T16:24:26.643165Z","shell.execute_reply":"2022-07-13T16:30:49.647015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = glob(os.path.join(dataset_path, '*.png'))\nprint(len(images))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:34:36.427484Z","iopub.execute_input":"2022-07-14T14:34:36.427873Z","iopub.status.idle":"2022-07-14T14:34:36.524876Z","shell.execute_reply.started":"2022-07-14T14:34:36.42784Z","shell.execute_reply":"2022-07-14T14:34:36.523775Z"},"trusted":true},"execution_count":null,"outputs":[]}]}