{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<img src=\"https://storage.googleapis.com/kaggle-competitions/kaggle/37333/logos/header.png?t=2022-06-29-00-47-20\">\n\n<h1><center>STRIP AI - EDA & Data Preparation</center></h1>\n\nAs the name suggests, this notebook encompasses two main goals: EDA and data preparation.\n\nIf you're interested only in the latter part, feel free to go to the dataset directly: **[STRIP AI - 256x256 PNG Tiles][1]**.\n\n<h2>Table of Contents</h2>\n\n- [Setup][2]\n- [Images EDA][3]\n- [Tabular Data EDA][4]\n- [Data Preparation][5]\n\n<h1><center>TL;DR</center></h1>\n\n<font size=\"4\">\n    This is a binary classification problem: having 754 train huge size train images of blood clots from a patient that had experienced an acute ischemic stroke you have to train the model to predict the etiology to be either CE (Cardioembolic) or LAA (Large Artery Atherosclerosis).\n</font>\n\n\n[1]: https://www.kaggle.com/datasets/nickuzmenkov/strip-ai-256x256-png-tiles\n[2]: #setup\n[3]: #images_eda\n[4]: #tabular_data_eda\n[5]: #data_preparation","metadata":{"_uuid":"2aa1e8ce-d5c1-4595-9b26-3eddad82f735","_cell_guid":"1343f9eb-8afe-4715-a743-158604416530","trusted":true}},{"cell_type":"markdown","source":"<h1><center id=\"setup\">Setup</center></h1>","metadata":{"_uuid":"f881e7af-366f-498d-b497-e28a25cd62c7","_cell_guid":"6174cfce-a1cd-4933-bc7a-31c7e2123437","trusted":true}},{"cell_type":"code","source":"import itertools\nimport os\nimport typing\nimport warnings\nimport zipfile\n\nimport cv2\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport plotly.graph_objects as go\nimport rasterio\nimport tensorflow as tf\nfrom rasterio.windows import Window\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom tqdm.notebook import tqdm","metadata":{"_uuid":"bbd7db67-1184-4339-8900-bdac15a788cb","_cell_guid":"3f703e75-170e-4848-91e0-1fd39533fff3","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:31:46.833741Z","iopub.execute_input":"2022-07-10T14:31:46.834327Z","iopub.status.idle":"2022-07-10T14:31:56.266862Z","shell.execute_reply.started":"2022-07-10T14:31:46.834217Z","shell.execute_reply":"2022-07-10T14:31:56.265736Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RANDOM_STATE = 42\nIMG_SIZE = 256\nREDUCE = 16\nFOLDS = 5\nFILES_PER_FOLD = 16\nCUT_SIZE = IMG_SIZE * REDUCE\nINPUT_PATH = \"../input/mayo-clinic-strip-ai\"\nMIN_SATURATION = 40\nMIN_PIXELS = 1000\n\nwarnings.filterwarnings(\"ignore\")\nnp.random.seed(RANDOM_STATE)","metadata":{"_uuid":"adb3fd19-80e6-4cfd-9554-ec8e36b9ea70","_cell_guid":"b9cf7490-b6b6-41fb-9212-80477692a118","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:48:10.825753Z","iopub.execute_input":"2022-07-10T14:48:10.826134Z","iopub.status.idle":"2022-07-10T14:48:10.832765Z","shell.execute_reply.started":"2022-07-10T14:48:10.826103Z","shell.execute_reply":"2022-07-10T14:48:10.831972Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TilesCount(typing.NamedTuple):\n    h_tiles: int\n    v_tiles: int\n    total: int\n\n\ndef count_tiles(path: str) -> TilesCount:\n    with rasterio.open(os.path.join(INPUT_PATH, path), num_threads=\"all_cpus\") as image:\n        h_tiles = len(list(range(0, image.width // CUT_SIZE * CUT_SIZE, CUT_SIZE)))\n        v_tiles = len(list(range(0, image.height // CUT_SIZE * CUT_SIZE, CUT_SIZE)))\n\n        return TilesCount(h_tiles=h_tiles, v_tiles=v_tiles, total=h_tiles * v_tiles)\n\n\ndef empty_tile(tile: np.array) -> bool:\n    hsv = cv2.cvtColor(tile, cv2.COLOR_RGB2HSV)\n    _, s, _ = cv2.split(hsv)\n\n    return ((s > MIN_SATURATION).sum() < MIN_PIXELS) or (tile.sum() < MIN_PIXELS)\n\n\ndef read_tiles(\n    path: str,\n    filter_empty: bool = False,\n    verbose: bool = True,\n) -> typing.Generator[typing.Union[np.array, None], None, None]:\n    with rasterio.open(os.path.join(INPUT_PATH, path), num_threads=\"all_cpus\") as image:\n        grid = itertools.product(\n            range(0, image.height // CUT_SIZE * CUT_SIZE, CUT_SIZE),\n            range(0, image.width // CUT_SIZE * CUT_SIZE, CUT_SIZE),\n        )\n\n        for i, j in tqdm(\n            grid,\n            desc=\"Reading image\",\n            total=count_tiles(path).total,\n            disable=not verbose,\n        ):\n            tile = image.read(\n                [1, 2, 3],\n                window=Window.from_slices((i, i + CUT_SIZE), (j, j + CUT_SIZE)),\n            )\n            tile = np.transpose(tile, (1, 2, 0))\n            tile = cv2.resize(tile, (IMG_SIZE, IMG_SIZE))\n            if filter_empty and empty_tile(tile):\n                yield None\n            else:\n                yield tile\n\n\ndef show_tiles(path: str, filter_empty: bool = False) -> None:\n    tiles_count = count_tiles(path)\n    tiles = read_tiles(path, filter_empty=filter_empty)\n\n    figure, axes = plt.subplots(\n        tiles_count.v_tiles,\n        tiles_count.h_tiles,\n        figsize=(tiles_count.h_tiles, tiles_count.v_tiles),\n    )\n    axes = np.ravel(axes)\n\n    for i, tile in enumerate(tiles):\n        if tile is not None:\n            axes[i].imshow(tile)\n        else:\n            axes[i].imshow(np.zeros((IMG_SIZE, IMG_SIZE, 3)))\n        axes[i].axis(\"off\")\n\n    figure.suptitle(os.path.basename(path))\n    figure.subplots_adjust(wspace=0, hspace=0.05)\n    figure.show()\n\n\ndef serialize_sample(image: bytes, label: int) -> bytes:\n    feature = {\n        \"image\": tf.train.Feature(bytes_list=tf.train.BytesList(value=[image])),\n        \"label\": tf.train.Feature(int64_list=tf.train.Int64List(value=[label])),\n    }\n    sample = tf.train.Example(features=tf.train.Features(feature=feature))\n    return sample.SerializeToString()\n\n\ndef serialize(image_names: typing.List[str], labels: typing.List[int], path: str) -> None:\n    with tf.io.TFRecordWriter(path) as writer, zipfile.ZipFile(\"train.zip\") as file:\n        for image_name, label in zip(image_names, labels):\n            image = file.read(image_name)\n            writer.write(serialize_sample(image, label))","metadata":{"_uuid":"7e70d79b-22ca-410b-a783-b665a93ee824","_cell_guid":"9d77d236-8e41-4ca4-ad00-153b6a1993fd","collapsed":false,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-10T14:31:56.282785Z","iopub.execute_input":"2022-07-10T14:31:56.283622Z","iopub.status.idle":"2022-07-10T14:31:56.328401Z","shell.execute_reply.started":"2022-07-10T14:31:56.283584Z","shell.execute_reply":"2022-07-10T14:31:56.327176Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(os.path.join(INPUT_PATH, \"train.csv\"))\nother_df = pd.read_csv(os.path.join(INPUT_PATH, \"other.csv\"))\ntest_df = pd.read_csv(os.path.join(INPUT_PATH, \"test.csv\"))\nsample_submission = pd.read_csv(os.path.join(INPUT_PATH, \"sample_submission.csv\"))","metadata":{"_uuid":"15ecc3da-2c9b-42fa-90fe-a6b212315684","_cell_guid":"1dd9e375-ec5a-41af-b3f1-0c87418b7844","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:31:56.329968Z","iopub.execute_input":"2022-07-10T14:31:56.331054Z","iopub.status.idle":"2022-07-10T14:31:56.365467Z","shell.execute_reply.started":"2022-07-10T14:31:56.331008Z","shell.execute_reply":"2022-07-10T14:31:56.364503Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1><center id=\"images_eda\">Images EDA</center></h1>\n\n<font size=\"4\">\n    We have 350 Gb of data. And only 754 train images\n</font>","metadata":{"_uuid":"57efad22-a9b6-4dde-913a-5746e09c763c","_cell_guid":"efb12846-295f-4c3a-a935-52c9be5ddc4e","trusted":true}},{"cell_type":"code","source":"train_paths = os.listdir(os.path.join(INPUT_PATH, \"train\"))\ntest_paths = os.listdir(os.path.join(INPUT_PATH, \"test\"))\nother_paths = os.listdir(os.path.join(INPUT_PATH, \"other\"))\n\nprint(f\"Train images: {len(train_paths)}\")\nprint(f\"Test images: {len(test_paths)}\")\nprint(f\"Other images: {len(other_paths)}\")","metadata":{"_uuid":"3b572efb-1f4c-4f7a-8cc8-654b4ae7a455","_cell_guid":"e81c96ce-8148-4dca-9154-5055169eb7ef","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:31:56.367873Z","iopub.execute_input":"2022-07-10T14:31:56.368465Z","iopub.status.idle":"2022-07-10T14:31:56.455958Z","shell.execute_reply.started":"2022-07-10T14:31:56.36843Z","shell.execute_reply":"2022-07-10T14:31:56.455124Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    But if we cut those into 4096x4096 tiles, we'll get a decent 35,396 images dataset\n</font>","metadata":{"_uuid":"c930b403-634e-4d8f-a705-02b00355bc38","_cell_guid":"d58ad9df-c730-485a-a075-c092a4bcccbc","trusted":true}},{"cell_type":"code","source":"train_tiles = sum(count_tiles(os.path.join(\"train\", x)).total for x in train_paths)\ntest_tiles = sum(count_tiles(os.path.join(\"test\", x)).total for x in test_paths)\nother_tiles = sum(count_tiles(os.path.join(\"other\", x)).total for x in other_paths)\n\nprint(f\"Train tiles: {train_tiles:,d}\")\nprint(f\"Test tiles: {test_tiles:,d}\")\nprint(f\"Other tiles: {other_tiles:,d}\")","metadata":{"_uuid":"2ff993ba-41dd-4a9b-83ac-3611a9452c5c","_cell_guid":"9571b2ea-037c-43bc-81fa-d7fd763a42b5","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:31:56.457118Z","iopub.execute_input":"2022-07-10T14:31:56.457617Z","iopub.status.idle":"2022-07-10T14:32:08.392458Z","shell.execute_reply.started":"2022-07-10T14:31:56.457587Z","shell.execute_reply":"2022-07-10T14:32:08.391072Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    We can now look at a whole image without OOM\n</font>","metadata":{"_uuid":"7b68fa3f-f5d7-483d-ac24-519da112404e","_cell_guid":"98627a48-674f-46e2-beb2-5c395559eb67","trusted":true}},{"cell_type":"code","source":"path = \"train/01adc5_0.tif\"\nshow_tiles(path)","metadata":{"_uuid":"64961f4e-2838-4500-b59c-0fbf75dc219e","_cell_guid":"5315d95e-b448-4f68-97cd-05d304abc8d6","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:32:08.394702Z","iopub.execute_input":"2022-07-10T14:32:08.395018Z","iopub.status.idle":"2022-07-10T14:32:43.604556Z","shell.execute_reply.started":"2022-07-10T14:32:08.394989Z","shell.execute_reply":"2022-07-10T14:32:43.603279Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    Closer look at CE images\n</font>","metadata":{"_uuid":"99a9ad1e-bf41-4ecc-b753-0e667017d088","_cell_guid":"de6fc9c6-2dfb-4282-ab18-6c2c19c725cd","trusted":true}},{"cell_type":"code","source":"for image_id in train_df[train_df[\"label\"] == \"CE\"].sample(3)[\"image_id\"]:\n    show_tiles(f\"train/{image_id}.tif\")","metadata":{"_uuid":"e02426f6-3504-4318-ba4f-3957dcfb4bce","_cell_guid":"066daa58-a965-4786-9543-167cf87fb592","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:32:43.605887Z","iopub.execute_input":"2022-07-10T14:32:43.60662Z","iopub.status.idle":"2022-07-10T14:33:08.523457Z","shell.execute_reply.started":"2022-07-10T14:32:43.60658Z","shell.execute_reply":"2022-07-10T14:33:08.522323Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    Closer look at LAA images\n</font>","metadata":{"_uuid":"5712de80-22b1-4552-a8cb-a18efe29a4a3","_cell_guid":"96d88329-5257-424d-b1c4-8311700352d7","trusted":true}},{"cell_type":"code","source":"for image_id in train_df[train_df[\"label\"] == \"LAA\"].sample(3)[\"image_id\"]:\n    show_tiles(f\"train/{image_id}.tif\")","metadata":{"_uuid":"26f58f89-ea8d-4b57-89ed-18eb00491931","_cell_guid":"6892e719-44a6-4901-88ea-f463460c6eb1","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:33:08.524958Z","iopub.execute_input":"2022-07-10T14:33:08.525393Z","iopub.status.idle":"2022-07-10T14:33:29.264449Z","shell.execute_reply.started":"2022-07-10T14:33:08.52535Z","shell.execute_reply":"2022-07-10T14:33:29.262829Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    A lot of images have duplicate areas within the image\n</font>\n#\n(see **[topic](https://www.kaggle.com/competitions/mayo-clinic-strip-ai/discussion/336063)** on this problem)","metadata":{"_uuid":"1c7e3ddb-3ce8-4d2e-a507-ceeece74530d","_cell_guid":"2a1f0e20-80bc-46a0-9f1e-4422c114503f","trusted":true}},{"cell_type":"code","source":"show_tiles(\"train/611a50_0.tif\")","metadata":{"_uuid":"9b1392f0-8ed5-435a-a558-4bae320f2f7f","_cell_guid":"2839db03-0bcb-43e4-8212-9a9b8b3b27f9","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:33:29.266753Z","iopub.execute_input":"2022-07-10T14:33:29.267283Z","iopub.status.idle":"2022-07-10T14:33:50.528135Z","shell.execute_reply.started":"2022-07-10T14:33:29.267229Z","shell.execute_reply":"2022-07-10T14:33:50.526943Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    CE images are twice the size of LAA images (on average)\n</font>","metadata":{"_uuid":"b1540d22-1aa3-4384-bf1d-c0a5f1106d31","_cell_guid":"4ea42e13-9295-4f96-bbce-0631bb521957","trusted":true}},{"cell_type":"code","source":"ce_paths = [f\"train/{x}.tif\" for x in train_df[train_df[\"label\"] == \"CE\"][\"image_id\"]]\nce_tiles = sum(count_tiles(x).total for x in ce_paths)\n\nlaa_paths = [f\"train/{x}.tif\" for x in train_df[train_df[\"label\"] == \"LAA\"][\"image_id\"]]\nlaa_tiles = sum(count_tiles(x).total for x in laa_paths)\n\nprint(f\"Total number of CE tiles: {ce_tiles:,d}\")\nprint(f\"Total number of CE tiles: {laa_tiles:,d}\")","metadata":{"_uuid":"6ef2c713-6f04-4acf-bf24-270ebfb27965","_cell_guid":"9519eb57-938c-4051-8b83-fb71c1fef694","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:33:50.530242Z","iopub.execute_input":"2022-07-10T14:33:50.531053Z","iopub.status.idle":"2022-07-10T14:33:56.147087Z","shell.execute_reply.started":"2022-07-10T14:33:50.531008Z","shell.execute_reply":"2022-07-10T14:33:56.145656Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1><center id=\"tabular_data_eda\">Tabular Data EDA</center></h1>\n\n<font size=\"4\">\n    There are 2 extra features available: medical center id and patient id. Both should be used for grouping samples within folds\n</font>","metadata":{"_uuid":"491382f3-e927-42d2-84c0-b56cb5f6ccd8","_cell_guid":"d0d367db-22b9-4b10-b8ef-f00166df0c2b","trusted":true}},{"cell_type":"code","source":"train_df.head()","metadata":{"_uuid":"bdfc0170-18b8-4ce6-acae-7fd2e71ab82f","_cell_guid":"e7c2cb8f-4a30-47a4-ac7a-0b68a1d58c82","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:33:56.148808Z","iopub.execute_input":"2022-07-10T14:33:56.149359Z","iopub.status.idle":"2022-07-10T14:33:56.174971Z","shell.execute_reply.started":"2022-07-10T14:33:56.149315Z","shell.execute_reply":"2022-07-10T14:33:56.173873Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    Labels are imbalanced: 72% are CE, whereas only 28% are LAA\n</font>","metadata":{"_uuid":"ef6a6f5e-1ceb-4353-9d1f-76b497b4bad9","_cell_guid":"50cc5694-be1a-4d5b-acf7-ca3de2405f1e","trusted":true}},{"cell_type":"code","source":"diagnoses = train_df[\"label\"].value_counts(normalize=True)\n\ngo.Figure(\n    data=(go.Bar(x=diagnoses.index, y=diagnoses.values * 100)),\n    layout=dict(\n        width=600,\n        title_text=\"Label distribution\",\n        xaxis_title_text=\"Label\",\n        yaxis_title_text=\"%\",\n        font=dict(size=16),\n    ),\n)","metadata":{"_uuid":"0f1e6fb3-2746-42ec-b26b-9021a1d919fe","_cell_guid":"46fdb0b7-7f82-4811-baf2-685124c7eb60","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:33:56.176606Z","iopub.execute_input":"2022-07-10T14:33:56.177269Z","iopub.status.idle":"2022-07-10T14:33:56.34776Z","shell.execute_reply.started":"2022-07-10T14:33:56.177225Z","shell.execute_reply":"2022-07-10T14:33:56.346841Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    85% patients have only 1 record, but some have up to 5\n</font>","metadata":{"_uuid":"6d4276f7-2deb-49e7-9042-ce7f468d3968","_cell_guid":"884456c4-49b9-423b-83f2-ab74eee31efc","trusted":true}},{"cell_type":"code","source":"patient_records = train_df.groupby(\"patient_id\")[\"label\"].count().value_counts(normalize=True)\n\ngo.Figure(\n    data=(go.Bar(x=patient_records.index, y=patient_records.values * 100)),\n    layout=dict(\n        width=600,\n        title_text=\"Records per patient\",\n        xaxis_title_text=\"Number of records\",\n        yaxis_title_text=\"%\",\n        font=dict(size=16),\n    ),\n)","metadata":{"_uuid":"ab936321-66ad-4ef8-993d-8598f31003ea","_cell_guid":"e44ec40d-dbbe-49fd-9b3a-ad6f769de1db","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:33:56.35167Z","iopub.execute_input":"2022-07-10T14:33:56.352308Z","iopub.status.idle":"2022-07-10T14:33:56.377549Z","shell.execute_reply.started":"2022-07-10T14:33:56.352274Z","shell.execute_reply":"2022-07-10T14:33:56.376702Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    If patient has multiple records, all diagnoses are always the same\n</font>","metadata":{"_uuid":"cc17e58e-ef72-40d1-b6af-c7a84fb11615","_cell_guid":"85c9f566-4442-4d59-af1d-13481f86fa99","trusted":true}},{"cell_type":"code","source":"non_matching = (train_df.groupby(\"patient_id\")[\"label\"].nunique() != 1).sum()\n\nprint(\n    \"Number of patients with non-matching diagnoses in multiple cases:\",\n    non_matching,\n)","metadata":{"_uuid":"129c0a3e-c903-47c1-9a03-f1659993fd03","_cell_guid":"10ad1bc1-97cb-4e76-980b-380ce2ba1f7d","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:33:56.378886Z","iopub.execute_input":"2022-07-10T14:33:56.379458Z","iopub.status.idle":"2022-07-10T14:33:56.388868Z","shell.execute_reply.started":"2022-07-10T14:33:56.379424Z","shell.execute_reply":"2022-07-10T14:33:56.387633Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    Closer look at records of the same patient\n</font>","metadata":{"_uuid":"87116de3-cbbf-48c7-9872-574668e91317","_cell_guid":"e28b6e12-4c1e-4801-9951-34647bb8d804","trusted":true}},{"cell_type":"code","source":"record_counts = train_df.groupby(\"patient_id\")[\"patient_id\"].count()\npatient_id = record_counts[record_counts == 3].sample(1).index[0]\npatient_image_ids = train_df[train_df[\"patient_id\"] == patient_id][\"image_id\"]\n\nfor image_id in patient_image_ids:\n    show_tiles(f\"train/{image_id}.tif\")","metadata":{"_uuid":"cfd10b40-24d4-4130-8d34-6eaf4c5fb12a","_cell_guid":"9bc06140-67c2-4e2b-abc0-b7ac9e5c17cf","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:33:56.390662Z","iopub.execute_input":"2022-07-10T14:33:56.391087Z","iopub.status.idle":"2022-07-10T14:34:38.842009Z","shell.execute_reply.started":"2022-07-10T14:33:56.391046Z","shell.execute_reply":"2022-07-10T14:34:38.839388Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    There are exactly 11 medical centers. Most train images and all available test images are from the 11th center\n</font>","metadata":{"_uuid":"bfd8b3c4-e324-4c0a-890d-9f816e18ff7c","_cell_guid":"2dfe1ae5-fb62-41a7-a08d-52c2f763b06f","trusted":true}},{"cell_type":"code","source":"medical_centers = train_df[\"center_id\"].value_counts(normalize=True).sort_index()\n\ngo.Figure(\n    data=(go.Bar(x=medical_centers.index.astype(str), y=medical_centers.values * 100)),\n    layout=dict(\n        title_text=\"Records per medical center\",\n        xaxis_title_text=\"Medical center\",\n        yaxis_title_text=\"%\",\n        font=dict(size=16),\n    ),\n)","metadata":{"_uuid":"37228da4-710e-4051-90c6-56d8b9a97826","_cell_guid":"ed53a95e-c309-46be-87b8-983172625b7b","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:34:38.843328Z","iopub.execute_input":"2022-07-10T14:34:38.843651Z","iopub.status.idle":"2022-07-10T14:34:38.866446Z","shell.execute_reply.started":"2022-07-10T14:34:38.843624Z","shell.execute_reply":"2022-07-10T14:34:38.865369Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    84% images in the additional dataset provided by organizers are unlabeled. The rest belong to one of 9 other types of diseases\n</font>","metadata":{"_uuid":"26675f3a-7aa8-46cf-a5b4-bea6c87cd473","_cell_guid":"b7acc2d1-5b8e-454c-93cb-d818d0ce0d85","trusted":true}},{"cell_type":"code","source":"other_diagnoses = other_df[\"other_specified\"].value_counts(\n    normalize=True, ascending=True, dropna=False\n)\n\ngo.Figure(\n    data=(\n        go.Bar(\n            y=other_diagnoses.rename({np.nan: \"Unknown\"}).index,\n            x=other_diagnoses.values * 100,\n            orientation=\"h\",\n        )\n    ),\n    layout=dict(\n        width=600,\n        title_text=\"Diagnoses in additional data\",\n        xaxis_title_text=\"%\",\n        font=dict(size=16),\n    ),\n)","metadata":{"_uuid":"ca78897c-67fe-4a83-acff-c53678bba9ce","_cell_guid":"ff41e444-3fcf-4ade-a862-953cd540db67","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:34:38.868112Z","iopub.execute_input":"2022-07-10T14:34:38.86913Z","iopub.status.idle":"2022-07-10T14:34:38.891309Z","shell.execute_reply.started":"2022-07-10T14:34:38.869086Z","shell.execute_reply":"2022-07-10T14:34:38.890095Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    Closer look at additional unlabeled images\n</font>","metadata":{"_uuid":"dcca153d-3832-45e6-ad94-e9d3893257cd","_cell_guid":"623a4245-2dcf-4a25-a430-030d8a613cf3","trusted":true}},{"cell_type":"code","source":"image_ids = other_df[\"image_id\"].sample(3)\n\nfor image_id in image_ids:\n    show_tiles(f\"other/{image_id}.tif\")","metadata":{"_uuid":"e91e8921-dadc-4657-97d7-c5b5d6cd545f","_cell_guid":"5b224ee5-86c5-4088-ac0f-ac1d1586fe5a","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:34:38.893172Z","iopub.execute_input":"2022-07-10T14:34:38.8936Z","iopub.status.idle":"2022-07-10T14:36:17.6775Z","shell.execute_reply.started":"2022-07-10T14:34:38.893558Z","shell.execute_reply":"2022-07-10T14:36:17.676265Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1><center id=\"data_preparation\">Data Preparation</center></h1>\n\n<font size=\"4\">\n    Here we'll save 4096x4096 train image tiles downscaled to 256x256 as PNG images for further training. Blank areas will be excluded, e.g. for this image\n</font>","metadata":{"_uuid":"223de019-eaa6-4a4d-9496-5cf7f478eda8","_cell_guid":"96c67fb0-f241-430e-b0e4-e356023879b0","trusted":true}},{"cell_type":"code","source":"image_id = train_df[\"image_id\"].sample(1).item()\nshow_tiles(f\"train/{image_id}.tif\")","metadata":{"_uuid":"80e1f6a1-288e-4ace-806f-d421ea307221","_cell_guid":"2a795781-e720-4a46-8177-9dbf96e8dac6","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:36:17.679199Z","iopub.execute_input":"2022-07-10T14:36:17.679511Z","iopub.status.idle":"2022-07-10T14:36:28.69818Z","shell.execute_reply.started":"2022-07-10T14:36:17.679481Z","shell.execute_reply":"2022-07-10T14:36:28.696958Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    Tiles replaced by solid black squares won't be saved\n</font>","metadata":{"_uuid":"51e06e31-8d78-4ce6-bb28-db6c98029372","_cell_guid":"1bc06f59-70b6-48c8-b9c9-a0b6907380e2","trusted":true}},{"cell_type":"code","source":"show_tiles(f\"train/{image_id}.tif\", filter_empty=True)","metadata":{"_uuid":"c5ad2008-4f46-4f14-820a-87c903374100","_cell_guid":"164273dd-7df2-42af-8145-65b5da207a7f","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:36:28.699824Z","iopub.execute_input":"2022-07-10T14:36:28.700223Z","iopub.status.idle":"2022-07-10T14:36:38.931165Z","shell.execute_reply.started":"2022-07-10T14:36:28.700188Z","shell.execute_reply":"2022-07-10T14:36:38.92996Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    Save image tiles in PNG format\n</font>","metadata":{"_uuid":"5022d6ca-e873-4ce0-8530-063e96926824","_cell_guid":"ff2b723f-41fa-4fc2-855f-b4cc4d740116","trusted":true}},{"cell_type":"code","source":"with zipfile.ZipFile(\"train.zip\", \"w\") as file, tqdm(total=train_tiles) as bar:\n    for image_id in train_df[\"image_id\"]:\n        for i, tile in enumerate(\n            read_tiles(f\"train/{image_id}.tif\", filter_empty=True, verbose=False)\n        ):\n            if tile is not None:\n                tile = cv2.imencode(\".png\", cv2.cvtColor(tile, cv2.COLOR_RGB2BGR))[1]\n                file.writestr(f\"{image_id}-{i:04d}.png\", tile)\n            bar.update(1)","metadata":{"_uuid":"e8020188-fdf4-428c-aa19-587ef101006c","_cell_guid":"1f605ec1-6751-4e2e-831f-bbeb29842b20","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:45:54.062572Z","iopub.execute_input":"2022-07-10T14:45:54.062979Z","iopub.status.idle":"2022-07-10T14:45:58.874645Z","shell.execute_reply.started":"2022-07-10T14:45:54.062944Z","shell.execute_reply":"2022-07-10T14:45:58.873266Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    Save image tile reference in CSV format\n</font>","metadata":{"_uuid":"6f96329a-dab1-4774-a07a-c7f01acf2b26","_cell_guid":"48ac325a-f17f-4a88-bf40-ec43088c82bc","trusted":true}},{"cell_type":"code","source":"with zipfile.ZipFile(\"train.zip\") as file:\n    filenames = file.namelist()\n\nfilenames = pd.DataFrame({\"filename\": filenames})\nfilenames[\"image_id\"] = filenames[\"filename\"].map(lambda x: x.split(\"-\")[0])\n\ntile_df = train_df.merge(filenames, on=\"image_id\")\ntile_df.to_csv(\"train.csv\", index=False)\ntile_df.head()","metadata":{"_uuid":"17a60e1c-ea85-4f2f-9433-27a4d9622f36","_cell_guid":"4615976f-bc09-4530-8050-5219a85d719c","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:46:52.462356Z","iopub.execute_input":"2022-07-10T14:46:52.462812Z","iopub.status.idle":"2022-07-10T14:46:52.492304Z","shell.execute_reply.started":"2022-07-10T14:46:52.462775Z","shell.execute_reply":"2022-07-10T14:46:52.491292Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=\"4\">\n    Save image tiles in TFRecord fomat (for TensorFlow users)\n</font>\n\nBecause TFRecord is basically a collection of several images in one binary file (just like a ZIP archive), we should split them into folds beforehand for convenient usage. We use stratified group KFold to balance labels and avoid data leakage between folds.","metadata":{"_uuid":"3fd65d21-b253-4bcb-9885-4c6c39da45a3","_cell_guid":"277962dc-ec09-4bfd-82a9-f4ab6189266c","trusted":true}},{"cell_type":"code","source":"os.mkdir(\"tfrec\")\n\ntile_df[\"label\"] = (tile_df[\"label\"] == \"LAA\").astype(\"uint8\")\nkfold = StratifiedGroupKFold(n_splits=FOLDS, shuffle=True, random_state=RANDOM_STATE)\n\nwith tqdm(total=FOLDS * FILES_PER_FOLD, desc=\"Serialization\") as bar:\n    for i, (_, index) in enumerate(\n        kfold.split(\n            tile_df[\"image_id\"],\n            tile_df[\"label\"],\n            groups=tile_df[\"patient_id\"],\n        ),\n    ):\n        os.mkdir(f\"tfrec/{i}\")\n\n        for j, fold in enumerate(np.array_split(tile_df.iloc[index], FILES_PER_FOLD)):\n            serialize(\n                fold[\"filename\"],\n                fold[\"label\"],\n                path=f\"tfrec/{i}/{j:02d}-{len(fold):04d}.tfrec\",\n            )\n            bar.update(1)","metadata":{"_uuid":"9f1a9f70-3290-42c0-947b-02bfcb0d3113","_cell_guid":"6a0c9264-fc92-405f-b85f-6ae81a8ab761","collapsed":false,"execution":{"iopub.status.busy":"2022-07-10T14:48:34.550044Z","iopub.execute_input":"2022-07-10T14:48:34.550526Z","iopub.status.idle":"2022-07-10T14:48:34.633249Z","shell.execute_reply.started":"2022-07-10T14:48:34.550487Z","shell.execute_reply":"2022-07-10T14:48:34.6316Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1><center id=\"summary\">Good luck with the competition!</center></h1>","metadata":{"_uuid":"a960b965-07f7-441a-be73-151e1c742cf2","_cell_guid":"09136286-75e2-4505-b5b7-afcce6d71556","trusted":true}}]}