{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🔬Stroke Blood Clot Origin: 🔎 Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"import os, glob\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nDATASET_FOLDER = \"/kaggle/input/mayo-clinic-strip-ai/\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-25T12:24:36.056985Z","iopub.execute_input":"2022-07-25T12:24:36.057858Z","iopub.status.idle":"2022-07-25T12:24:36.085638Z","shell.execute_reply.started":"2022-07-25T12:24:36.057733Z","shell.execute_reply":"2022-07-25T12:24:36.084422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"How to download pyvips: https://www.kaggle.com/code/analokamus/how-to-use-pyvips-offline","metadata":{}},{"cell_type":"code","source":"!rm -rf conda-pkgs\n!mkdir conda-pkgs\n # Set the location where conda package will be downloaded\n!conda config --add pkgs_dirs ./conda-pkgs\n # Download pyvips and dependencies\n!conda install --download-only -y \"pyvips>=2.2.0\"\n!rm -rf ./conda-pkgs/cache","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T12:24:36.087804Z","iopub.execute_input":"2022-07-25T12:24:36.088525Z","iopub.status.idle":"2022-07-25T12:35:44.371385Z","shell.execute_reply.started":"2022-07-25T12:24:36.088485Z","shell.execute_reply":"2022-07-25T12:35:44.369815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!conda install ./conda-pkgs/*.tar.bz2 --offline","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T12:35:44.373787Z","iopub.execute_input":"2022-07-25T12:35:44.374577Z","iopub.status.idle":"2022-07-25T12:36:17.109332Z","shell.execute_reply.started":"2022-07-25T12:35:44.374523Z","shell.execute_reply":"2022-07-25T12:36:17.107857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explore metadata 🗄️","metadata":{}},{"cell_type":"code","source":"path_csv = os.path.join(DATASET_FOLDER, \"train.csv\")\ndf_train = pd.read_csv(path_csv)\ndisplay(df_train.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:36:17.11202Z","iopub.execute_input":"2022-07-25T12:36:17.112435Z","iopub.status.idle":"2022-07-25T12:36:17.159821Z","shell.execute_reply.started":"2022-07-25T12:36:17.112398Z","shell.execute_reply":"2022-07-25T12:36:17.15858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_= df_train[[\"label\"]].value_counts().plot.pie(autopct='%1.1f%%', ylabel=\"label\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:36:17.161477Z","iopub.execute_input":"2022-07-25T12:36:17.162368Z","iopub.status.idle":"2022-07-25T12:36:17.473371Z","shell.execute_reply.started":"2022-07-25T12:36:17.162312Z","shell.execute_reply":"2022-07-25T12:36:17.471808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=1, ncols=2, figsize=(9, 4))\nfor i, col in enumerate([\"image_num\", \"center_id\"]):\n    _= df_train[[col]].value_counts().sort_index().plot.bar(ax=axes[i], ylabel=col, grid=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:36:17.480184Z","iopub.execute_input":"2022-07-25T12:36:17.4843Z","iopub.status.idle":"2022-07-25T12:36:17.881739Z","shell.execute_reply.started":"2022-07-25T12:36:17.484214Z","shell.execute_reply":"2022-07-25T12:36:17.880277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_= df_train[[\"patient_id\"]].value_counts().plot.bar(ylabel=\"patient_id\", grid=True, figsize=(12, 3))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:36:17.88392Z","iopub.execute_input":"2022-07-25T12:36:17.884469Z","iopub.status.idle":"2022-07-25T12:36:26.059917Z","shell.execute_reply.started":"2022-07-25T12:36:17.884419Z","shell.execute_reply":"2022-07-25T12:36:26.058924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explore image dataset 🗃️","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nfrom tqdm.auto import tqdm\n\nImage.MAX_IMAGE_PIXELS = 5_000_000_000\n\nsizes = []\nfor name in tqdm(df_train[\"image_id\"]):\n    img = Image.open(os.path.join(DATASET_FOLDER, \"train\", f\"{name}.tif\"))\n    sizes.append({\"img_height\": img.height, \"img_width\": img.width})\n\ndf_sizes = pd.DataFrame(sizes)\nfor col in df_sizes.columns:\n    df_train[col] = df_sizes[col]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:36:26.061262Z","iopub.execute_input":"2022-07-25T12:36:26.062781Z","iopub.status.idle":"2022-07-25T12:36:58.212178Z","shell.execute_reply.started":"2022-07-25T12:36:26.062726Z","shell.execute_reply":"2022-07-25T12:36:58.210789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=2, ncols=1, figsize=(8, 5))\nfor i, col in enumerate([\"img_height\", \"img_width\"]):\n    _= df_train[[col]].hist(ax=axes[i], bins=35)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:36:58.214146Z","iopub.execute_input":"2022-07-25T12:36:58.214876Z","iopub.status.idle":"2022-07-25T12:36:58.624082Z","shell.execute_reply.started":"2022-07-25T12:36:58.214823Z","shell.execute_reply":"2022-07-25T12:36:58.622638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n\ndf_sizes = df_train[[\"img_width\", \"img_height\", \"label\"]]\nfor col in (\"img_width\", \"img_height\"):\n    df_sizes[col] = [round(i / 2000) * 2000 for i in df_sizes[col]]\n\ndf_sizes = df_sizes.groupby([\"img_width\", \"img_height\", \"label\"], as_index=False).size()\n# display(df_sizes.head())\nfig = px.scatter(df_sizes, x=\"img_width\", y=\"img_height\", size=\"size\", color=\"label\", height=600, width=600)\nfig.update_xaxes(range=[1_000, 120_000])\nfig.update_yaxes(scaleanchor=\"x\", scaleratio=1, range=[1_000, 120_000])\nfig.show() ","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:36:58.62772Z","iopub.execute_input":"2022-07-25T12:36:58.628259Z","iopub.status.idle":"2022-07-25T12:37:00.879287Z","shell.execute_reply.started":"2022-07-25T12:36:58.628209Z","shell.execute_reply":"2022-07-25T12:37:00.877912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"pixels\"] = df_train[\"img_width\"] * df_train[\"img_height\"]\ndf_train.sort_values(\"pixels\", inplace=True)\ndisplay(df_train.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:37:00.881222Z","iopub.execute_input":"2022-07-25T12:37:00.881988Z","iopub.status.idle":"2022-07-25T12:37:00.901526Z","shell.execute_reply.started":"2022-07-25T12:37:00.881939Z","shell.execute_reply":"2022-07-25T12:37:00.899836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample images 🖼️\n\n- noralize images by removing background\n- pruning rmpty columns and rows in image","metadata":{}},{"cell_type":"code","source":"def prune_image_rows_cols(im, mask, thr=0.990):\n    # delete empty columns\n    for l in reversed(range(im.shape[1])):\n        if (np.sum(mask[:, l]) / float(mask.shape[0])) > thr:\n            im = np.delete(im, l, 1)\n    # delete empty rows\n    for l in reversed(range(im.shape[0])):\n        if (np.sum(mask[l, :]) / float(mask.shape[1])) > thr:\n            im = np.delete(im, l, 0)\n    return im\n\n\ndef mask_median(im, val=255):\n    masks = [None] * 3\n    for c in range(3):\n        masks[c] = im[..., c] >= np.median(im[:, :, c]) - 5\n    mask = np.logical_and(*masks)\n    im[mask, :] = val\n    return im, mask","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:37:00.903338Z","iopub.execute_input":"2022-07-25T12:37:00.904086Z","iopub.status.idle":"2022-07-25T12:37:00.914905Z","shell.execute_reply.started":"2022-07-25T12:37:00.90404Z","shell.execute_reply":"2022-07-25T12:37:00.913953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pyvips\n\nos.environ['VIPS_CONCURRENCY'] = '4'\nos.environ['VIPS_DISC_THRESHOLD'] = '8gb'\n\nfor lb, dfg in df_train.groupby(\"label\"):\n    fig, axes = plt.subplots(nrows=5, ncols=2, figsize=(16, 20))\n    for i, name in enumerate(dfg[\"image_id\"].sample(5)):\n        img_path = os.path.join(DATASET_FOLDER, \"train\", f\"{name}.tif\")\n        # img = Image.open(img_path)\n        # for gap see: https://pillow.readthedocs.io/en/stable/reference/Image.html#PIL.Image.Image.resize\n        # The smaller reducing_gap, the faster resizing.\n        # With reducing_gap greater or equal to 3.0, the result is indistinguishable from fair resampling.\n        # The default value for thumbnail() is 2.0, which is very close to fair resampling while still being faster in many cases.\n        # UnSupported gap issue: https://github.com/libvips/pyvips/issues/339\n        img = pyvips.Image.thumbnail(img_path, 2000).numpy()\n        if img.shape[0] > img.shape[1]:\n            img = np.rollaxis(img, 0, 2)\n        axes[i, 0].imshow(img)\n        axes[i, 0].set_title(f\"{lb}: {name}\")\n        img, mask = mask_median(np.array(img))\n        img = prune_image_rows_cols(img, mask)\n        #print(clr_mean)\n        axes[i, 1].imshow(img)\n        axes[i, 1].set_title(\"normalized\")\n    fig.show()\ndel fig, axes, img","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:37:00.916317Z","iopub.execute_input":"2022-07-25T12:37:00.916717Z","iopub.status.idle":"2022-07-25T12:46:02.574597Z","shell.execute_reply.started":"2022-07-25T12:37:00.916682Z","shell.execute_reply":"2022-07-25T12:46:02.573228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Export images 🚇","metadata":{}},{"cell_type":"code","source":"!mkdir -p train_images\n\ndef image_load_scale_norm(img_path, prune_thr=0.990, bg_val=255):\n    img = Image.open(img_path)\n    if (img.width * img.height) > 4_000_000_000:\n        print(f\"width: {img.width}, height: {img.height}, pixels: {img.width * img.height}\")\n        return None\n    scale = min(img.height / 2e3, img.width / 2e3)\n    if scale > 1:\n        tmp_size = int(img.width / scale), int(img.height / scale)\n        img.thumbnail(tmp_size, resample=Image.Resampling.BILINEAR, reducing_gap=2.0)\n    img, mask = mask_median(np.array(img), val=bg_val)\n    img = prune_image_rows_cols(img, mask, thr=prune_thr)\n    img = Image.fromarray(img)\n    scale = min(img.height / 1e3, img.width / 1e3)\n    if scale > 1:\n        img = img.resize((int(img.width / scale), int(img.height / scale)), Image.ANTIALIAS)\n    return img","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:46:02.579374Z","iopub.execute_input":"2022-07-25T12:46:02.579839Z","iopub.status.idle":"2022-07-25T12:46:03.405794Z","shell.execute_reply.started":"2022-07-25T12:46:02.579799Z","shell.execute_reply":"2022-07-25T12:46:03.404462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\nfor name in tqdm(df_train[\"image_id\"]):\n    img_path = os.path.join(DATASET_FOLDER, \"train\", f\"{name}.tif\")\n    img = image_load_scale_norm(img_path)\n    if not img:\n        continue\n    img.save(os.path.join(\"train_images\", f\"{name}.png\"))\n    del img\n    gc.collect()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-25T12:46:03.407773Z","iopub.execute_input":"2022-07-25T12:46:03.408877Z"},"trusted":true},"execution_count":null,"outputs":[]}]}