{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n\n# http://www.ajnr.org/content/36/9/1756   clot histology study\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom PIL import Image\nfrom openslide import open_slide\nimport openslide\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport matplotlib.pyplot as plt\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\nINPUT_DIR = \"/kaggle/input/mayo-clinic-strip-ai/\"\n\n# DATA_DIR_other = \"/kaggle/input/mayo-clinic-strip-ai/other\"\nDATA_DIR_train = \"/kaggle/input/mayo-clinic-strip-ai/train\"\n# DATA_DIR_test = \"/kaggle/input/mayo-clinic-strip-ai/test\"\ntiff_list = []\n# for dirname, _, filenames in os.walk(DATA_DIR_other):\n#     for filename in filenames:\n#         tiff_list.append([filename[:-4],os.path.join(dirname, filename)])\n# print(tiff_list[-5:])\n\nfor dirname, _, filenames in os.walk(DATA_DIR_train):\n    for filename in filenames:\n        tiff_list.append([filename[:-4],os.path.join(dirname, filename)])\nprint(tiff_list[-5:])\n\n# for dirname, _, filenames in os.walk(DATA_DIR_test):\n#     for filename in filenames:\n#         tiff_list.append([filename[:-4],os.path.join(dirname, filename)])\n# print(tiff_list[-5:])\n!pip install openslide-python\nprocessed_path = '/kaggle/working/processed'\nif not os.path.exists(processed_path):\n    os.makedirs(processed_path, exist_ok=False)\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-08T06:35:16.905395Z","iopub.execute_input":"2022-08-08T06:35:16.906345Z","iopub.status.idle":"2022-08-08T06:35:31.263617Z","shell.execute_reply.started":"2022-08-08T06:35:16.906231Z","shell.execute_reply":"2022-08-08T06:35:31.261826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T06:50:44.677826Z","iopub.execute_input":"2022-08-06T06:50:44.6782Z","iopub.status.idle":"2022-08-06T06:50:44.805191Z","shell.execute_reply.started":"2022-08-06T06:50:44.678166Z","shell.execute_reply":"2022-08-06T06:50:44.804038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Load the slide file (svs) into an object.\n\ndef process_slide(slideid,input_image_path):\n    slide = open_slide(input_image_path)\n    slide_out_path = f'{processed_path}/{slideid}'\n    if not os.path.exists(slide_out_path):\n        os.makedirs(slide_out_path, exist_ok=False)\n\n    dims = slide.level_dimensions[0]\n    print(dims)\n    tile_size = 900\n    counter = 0\n    for x in range(0,dims[0],tile_size):\n        for y in range(0,dims[1],tile_size):\n            if not FILE_REPLACE:\n                image_path  = f\"{slide_out_path}/{x}_{y}.jpg\"\n                if os.path.exists(image_path):\n                    continue\n                \n            level_img = slide.read_region((x,y), 0, (tile_size,tile_size)).convert('RGB')\n            level_img =  level_img.resize((300,300),Image.ANTIALIAS)\n            [a1,a2] = level_img.convert(\"L\").getextrema()\n#             Here if the image's max and min value is less then 10 then avpid since this may be white background images\n            if abs(a1-a2) < 10:\n                break\n#             print(a1,a2, abs(a1-a2))\n            image_path  = f\"{slide_out_path}/{x}_{y}.jpg\"\n            if os.path.exists(image_path):\n                os.remove(image_path)\n            level_img.save(image_path)\n            counter +=1\n#             return\n           \n#             im_array = np.asarray(level_img)\n    #         plt.imshow(im_array)\n    #         plt.show()\n#             level_img.save(f\"{slide_out_path}/{x}_{y}.jpg\")\n    print(f\"Written {counter} images\")\n# process_slide(slideid)\n\n#chk file existance\ncounterC = 0\nfor slide in tiff_list:\n        if not os.path.isfile(slide[1]):\n            print(slide[1])\n        else:\n            counterC += 1\n            \nprint(f\"Counter: {counterC}  total file: {len(tiff_list)}\")\nFILE_TO_PROCESS_FLAG = True\nFILE_REPLACE = False\ncounterC = 0\nif FILE_TO_PROCESS_FLAG:\n    for slide in tiff_list:\n        counterC += 1\n        print(f\"Slide: {slide[0]} :: {counterC}/{len(tiff_list)} \")\n        process_slide(slide[0],slide[1])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:35:49.306583Z","iopub.execute_input":"2022-08-08T06:35:49.307101Z","iopub.status.idle":"2022-08-08T06:35:50.269552Z","shell.execute_reply.started":"2022-08-08T06:35:49.307052Z","shell.execute_reply":"2022-08-08T06:35:50.268371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # level_img = slide.read_region((4000,3000), 0, (900,900))\n# # im_array = np.asarray(level_img)\n# # plt.imshow(im_array)\n# # plt.show()\n\n# # to find out if same patient id has different diagnosis\n# import pandas as pd\n# import numpy as np\n# data = pd.read_csv(\"../input/mayo-clinic-strip-ai/train.csv\")\n# print(data.head())\n# np_data = data.to_numpy()\n# a = {}\n# counter = 0\n# for d in np_data:\n#     if (d[2] in a):\n#         counter += 1\n#         # if the data has privious entry is check if it is different diagnosis\n#         if (a[d[2]] != d[4]):\n#             print(d)\n#     else:\n#         a[d[2]] = d[4]\n\n# print(len(a),len(np_data),counter)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T13:13:03.089324Z","iopub.execute_input":"2022-07-30T13:13:03.090238Z","iopub.status.idle":"2022-07-30T13:13:03.212056Z","shell.execute_reply.started":"2022-07-30T13:13:03.090186Z","shell.execute_reply":"2022-07-30T13:13:03.210933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}