{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport openslide\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nfrom matplotlib import pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"!ls /kaggle/input/prostate-cancer-grade-assessment","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"BASE_DIR = '/kaggle/input/prostate-cancer-grade-assessment'","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Glimpse of the dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ntrain_df = pd.read_csv(os.path.join(BASE_DIR, 'train.csv'))\ntest_df = pd.read_csv(os.path.join(BASE_DIR, 'test.csv'))\nsample_sub_df = pd.read_csv(os.path.join(BASE_DIR, 'sample_submission.csv'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f'Number of training images: {len(os.listdir(os.path.join(BASE_DIR, \"train_images\")))}')\nprint(f'Number of segmentation masks for training: {len(os.listdir(os.path.join(BASE_DIR, \"train_label_masks\")))}')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Segmentation mask is given for almost all the training images."},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_sub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f'Size of train_df: {train_df.shape}')\nprint(f'Size of test_df: {test_df.shape}')\nprint(f'Size of sample_sub_df: {sample_sub_df.shape}')","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":false,"trusted":true},"cell_type":"code","source":"sns.set(rc={'figure.figsize':(11,8)})\nsns.set(style=\"whitegrid\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Distribution of target variable"},{"metadata":{"trusted":true},"cell_type":"code","source":"np.sort(pd.unique(train_df['isup_grade']))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ax = sns.barplot(np.sort(pd.unique(train_df['isup_grade'])), train_df['isup_grade'].value_counts().sort_values(ascending=False))\nax.set(xlabel='ISUP Grades', ylabel='# of records', title='ISUP Grades vs. # of records')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Distribution of Data Providers"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['data_provider'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ax = sns.barplot(np.sort(pd.unique(train_df['data_provider'])), train_df['data_provider'].value_counts().sort_values(ascending=False))\nax.set(xlabel='Data Providers', ylabel='# of records', title='Data Providers vs. # of records')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Target variable distribution based on data providers"},{"metadata":{"trusted":true},"cell_type":"code","source":"counts_karolinska = train_df[train_df['data_provider'] == 'karolinska']['isup_grade'].value_counts(ascending=False)\ncounts_radboud = train_df[~(train_df['data_provider'] == 'karolinska')]['isup_grade'].value_counts()\n\nkarolinska_df = pd.DataFrame({\n    '# of records': counts_karolinska,\n    'isup_grades': np.sort(pd.unique(train_df[train_df['data_provider'] == 'karolinska']['isup_grade'])),\n    'data_provider': 'karolinska'\n})\nradboud_df = pd.DataFrame({\n    '# of records': counts_radboud,\n    'isup_grades': np.sort(pd.unique(train_df[~(train_df['data_provider'] == 'karolinska')]['isup_grade'])),\n    'data_provider': 'radboud'\n})\nsns.factorplot(x='isup_grades', y='# of records', hue='data_provider', data=pd.concat([karolinska_df, radboud_df], ignore_index=True), kind='bar', height=7, aspect=1.5)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualizing a few training images"},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# Visualize few samples of current training dataset\nfig, ax = plt.subplots(nrows=3, ncols=4, figsize=(20, 20))\ncount=0\nfor row in ax:\n    for col in row:\n        img = os.path.join(BASE_DIR, 'train_images', f'{train_df[\"image_id\"].iloc[count]}.tiff')\n        img = openslide.OpenSlide(img)\n        patch = img.read_region((0, 0), 2, img.level_dimensions[-1])\n        col.title.set_text(f'Source: {train_df[\"data_provider\"].iloc[count]} \\n ISUP grade: {train_df[\"isup_grade\"].iloc[count]} \\n gleason score: {train_df[\"gleason_score\"].iloc[count]}')\n        col.grid(False)\n        col.set_xticks([])\n        col.set_yticks([])\n        col.imshow(patch)\n        count += 1\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Segmentation masks for above images"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"def print_mask_details(slide, center='radboud', show_thumbnail=True, max_size=(400,400)):\n    \"\"\"Print some basic information about a slide\"\"\"\n\n    if center not in ['radboud', 'karolinska']:\n        raise Exception(\"Unsupported palette, should be one of [radboud, karolinska].\")\n\n    # Generate a small image thumbnail\n    if show_thumbnail:\n        # Read in the mask data from the highest level\n        # We cannot use thumbnail() here because we need to load the raw label data.\n        mask_data = slide.read_region((0,0), slide.level_count - 1, slide.level_dimensions[-1])\n        # Mask data is present in the R channel\n        mask_data = mask_data.split()[0]\n\n        # To show the masks we map the raw label values to RGB values\n        preview_palette = np.zeros(shape=768, dtype=int)\n        if center == 'radboud':\n            # Mapping: {0: background, 1: stroma, 2: benign epithelium, 3: Gleason 3, 4: Gleason 4, 5: Gleason 5}\n            preview_palette[0:18] = (np.array([0, 0, 0, 0.5, 0.5, 0.5, 0, 1, 0, 1, 1, 0.7, 1, 0.5, 0, 1, 0, 0]) * 255).astype(int)\n        elif center == 'karolinska':\n            # Mapping: {0: background, 1: benign, 2: cancer}\n            preview_palette[0:9] = (np.array([0, 0, 0, 0.5, 0.5, 0.5, 1, 0, 0]) * 255).astype(int)\n        mask_data.putpalette(data=preview_palette.tolist())\n        mask_data = mask_data.convert(mode='RGB')\n        mask_data.thumbnail(size=max_size, resample=0)\n        return mask_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"fig, ax = plt.subplots(nrows=3, ncols=4, figsize=(20, 20))\ncount=0\nfor row in ax:\n    for col in row:\n        mask = os.path.join(BASE_DIR, 'train_label_masks', f'{train_df[\"image_id\"].iloc[count]}_mask.tiff')\n        mask = openslide.OpenSlide(mask)\n        mask = print_mask_details(mask, center='radboud')\n        col.imshow(mask)\n        col.title.set_text(f'Source: {train_df[\"data_provider\"].iloc[count]} \\n ISUP grade: {train_df[\"isup_grade\"].iloc[count]} \\n gleason score: {train_df[\"gleason_score\"].iloc[count]}')\n        col.grid(False)\n        col.set_xticks([])\n        col.set_yticks([])\n        mask.close()\n        count += 1\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualizing some of the image patches from training dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_patch(img, x, y, level=0, width=512, height=512):\n    biopsy = openslide.OpenSlide(os.path.join(BASE_DIR, 'train_images', img))\n    region = biopsy.read_region((x, y), level, (width, height))\n    display(region)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_patch('00928370e2dfeb8a507667ef1d4efcbb.tiff', 5150, 21000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_patch('0005f7aaab2800f6170c399693a96917.tiff', 6000, 18000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_patch('0018ae58b01bdadc8e347995b69f99aa.tiff', 1500, 6000)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visualizing biopsies having different ISUP grandes (Cancer Severitirs)"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"def plot_biopsy_grid(df):\n    fig, ax = plt.subplots(nrows=2, ncols=2, figsize=(20, 20))\n    count=0\n    for row in ax:\n        for col in row:\n            img = os.path.join(BASE_DIR, 'train_images', f'{df[\"image_id\"].iloc[count]}.tiff')\n            img = openslide.OpenSlide(img)\n            patch = img.read_region((0, 0), 2, img.level_dimensions[-1])\n            col.title.set_text(f'Source: {df[\"data_provider\"].iloc[count]} \\n ISUP grade: {df[\"isup_grade\"].iloc[count]} \\n gleason score: {df[\"gleason_score\"].iloc[count]}')\n            col.grid(False)\n            col.set_xticks([])\n            col.set_yticks([])\n            col.imshow(patch)\n            count += 1\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### ISUP Grade: 0 (Risk Group: Healthy)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_biopsy_grid(train_df[train_df['isup_grade'] == 0][:4])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## ISUP Grade 1: (Risk Group: Low)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_biopsy_grid(train_df[train_df['isup_grade'] == 1][:4])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## ISUP Grade 2: (Risk Group: Intermediate Favorable)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_biopsy_grid(train_df[train_df['isup_grade'] == 2][:4])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## ISUP Grade 3: (Risk Group: Intermediate Unfavorable)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_biopsy_grid(train_df[train_df['isup_grade'] == 3][:4])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## ISUP Grade 4: (Risk Group: High)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_biopsy_grid(train_df[train_df['isup_grade'] == 4][:4])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## ISUP Grade 5: (Risk Group: High)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_biopsy_grid(train_df[train_df['isup_grade'] == 5][:4])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Stay Tuned (In Progress)"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}