{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# Imports\nimport os\n\nimport openslide\nfrom IPython.display import Image, display\n#     Allows viewing of images\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Files and Directories","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Setting dataset directories to variables\ntrain_dir = \"/kaggle/input/prostate-cancer-grade-assessment/train_images/\"\n#     train_images\nmask_dir = \"/kaggle/input/prostate-cancer-grade-assessment/train_label_masks/\"\n#     train_masks\n\ntrain_csv = pd.read_csv(\"/kaggle/input/prostate-cancer-grade-assessment/train.csv\")\n#     train_csv","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# **Data Munging train_csv (from PANDA Challenge Part 2)**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.drop(index=7273, inplace=True)\n#     Dropping 7273\n\ntrain_csv['gleason_score'] = train_csv['gleason_score'].apply(lambda x: '0+0' if x == 'negative' else x)\n#     Converting gleason_score = negative to gleason_score = 0+0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Confirming\ntry:\n    train_csv.loc[7273]\nexcept:\n    print('index 7273 not found')\n    \nprint(train_csv[train_csv['isup_grade']==0]['gleason_score'].value_counts())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# EDA of Dataset","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.dtypes","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"It's better to change isup_grade, daa_provider, and gleason_score into category data types","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv = train_csv.astype({'data_provider': 'category',\n                 'isup_grade': 'category',\n                 'gleason_score': 'category'})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(data=train_csv, x='data_provider', palette='viridis')\n\nplt.title(\"Samples by Provider\", fontdict={'fontsize': 12})\nplt.ylabel(\"No. of Samples\", labelpad=6.5, fontdict={'fontsize': 12})\nplt.xlabel(\"Institution\", labelpad=7, fontdict={'fontsize': 13})\nplt.yticks()\nsns.despine(left=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# There are similar sample amounts coming from karolinska and radboud \nprint(train_csv.data_provider.value_counts())\nprint(train_csv.data_provider.value_counts(normalize=True))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(data=train_csv, x='isup_grade', palette=\"cividis\")\n\nplt.title(\"Samples by ISUP Grade\", fontdict={'fontsize': 12})\nplt.ylabel(\"No. of Samples\", labelpad=6.5, fontdict={'fontsize': 12})\nplt.xlabel(\"ISUP Grade\", labelpad=7, fontdict={'fontsize': 13})\nplt.yticks()\nsns.despine(left=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# There are similar sample counts for samples ISUP Grade >= 2\n# 0 ISUP Grade Samples (Negative for Cancer) is the largest category, 1 is the second highest. Both almost double the other categories\nprint(train_csv.isup_grade.value_counts())\nprint(train_csv.isup_grade.value_counts(normalize=True))\nprint(\"\\nNegative Samples Take Up More Than 27% of the Dataset. Potentially be a source of class imbalance\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(7,5))\nsns.countplot(data=train_csv, x='gleason_score', palette=\"hot\")\n\nplt.title(\"Samples by Gleason Score\", fontdict={'fontsize': 14})\nplt.ylabel(\"No. of Samples\", labelpad=6.5, fontdict={'fontsize': 12})\nplt.xlabel(\"Gleason Score\", labelpad=7, fontdict={'fontsize': 13})\nplt.yticks()\nsns.despine(left=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# The Same Class Inbalance is observed because Gleason Score is tied to ISUP Grades \n# Although there are very few samples for 3+5 and 5+3 samples\n# Lack of 3+5 samples impacts predicting ISUP Grade 4 samples and predicting Gleason-Score 3+5 samples \n# Lack of 5+3 samples impacts predicting ISUP Grade 5 samples and predicting Gleason-Score 5+3 samples \n\n# With a Gleason-Score predictor, what changes needs to be made?\n#     This predictor would predict poorly serious cancers (Grade 5), so there is a potential case of misclassifing serious cancers as not so serioius.\n#     In a medical domain, false negative for a serious diagnosis isn't premissable\n# With a ISUP Grade predictor, what changes needs to be made?\n#     This predictor would have prediction inconsistencies for both Grade 4 and Grade 5 cancers as it will 'miss' a category of samples at\n#     that grade.\n\n# The model has to be designed to account for these flaws in the dataset.\n#     IF:\n#         A ISUP Model predicts : 5\n#         A Glea Model predicts : 5+3, 4+4, 3+5 (A 4)\n#     What do I conclude?\n#         If the G-Model predicts a 3+5: make the final model predict 5. It's a measure to lower false negatives\n#         If the G-Model preidcts a 5+3 or 4+4: Check prediction weightings, if high and ISUP Model has a low weighting for 5 probs let it be a 4..","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(8,5))\nsns.countplot(data=train_csv, x='isup_grade', hue='gleason_score', palette=\"icefire\")\n\nplt.title(\"Samples by ISUP Grade grouped by Gleason Score\", fontdict={'fontsize': 14})\nplt.ylabel(\"No. of Samples\", labelpad=6.5, fontdict={'fontsize': 12})\nplt.xlabel(\"ISUP Grade\", labelpad=7, fontdict={'fontsize': 13})\nplt.yticks()\nplt.legend(loc='upper left', bbox_to_anchor=(1,1))\nsns.despine(left=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Oh No, the problem doesn't extend to 3+5 and and 5+3. It extends to 5+4, 5+5\n# But the point still stands, I may need the intelligence of both a G-score and I-grade model to make my final predictions.\n\n# Graph proves that data munging cleaned up the dataset (No rows with mismathed ISUP Grades and Gleason Scores)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(8,5))\nsns.countplot(data=train_csv, x='isup_grade', hue='data_provider', palette=\"icefire\")\n\nplt.title(\"Samples by ISUP Grade grouped by Provider\", fontdict={'fontsize': 14})\nplt.ylabel(\"No. of Samples\", labelpad=6.5, fontdict={'fontsize': 12})\nplt.xlabel(\"ISUP Grade\", labelpad=7, fontdict={'fontsize': 13})\nplt.yticks()\nplt.legend(loc='upper left', bbox_to_anchor=(1,1))\nsns.despine(left=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# It seems radboud provides more serious biospies\n# While karolinska provides more beigin biospies\n\n# I hope my models don't fit for variations due to the provider (image size, color due to staining, microscopes), because of provider inbalance","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(8,5))\nsns.countplot(data=train_csv, x='gleason_score', hue='data_provider', palette=\"icefire\")\n\nplt.title(\"Samples by Gleason Score grouped by Provider\", fontdict={'fontsize': 14})\nplt.ylabel(\"No. of Samples\", labelpad=6.5, fontdict={'fontsize': 12})\nplt.xlabel(\"Gleason Score\", labelpad=7, fontdict={'fontsize': 13})\nplt.yticks()\nplt.legend(loc='upper left', bbox_to_anchor=(1,1))\nsns.despine(left=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Shows the same, but by Gleason Score.\n\n# I'll need to find what's different between karolinska and radboud images to conclude possible issues with the data.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.boxplot(data=train_csv.astype({'isup_grade': 'int8'}), x='data_provider', y='isup_grade', palette='cool')\n\nplt.title(\"ISUP Grade Distrubtion by Institute\", fontdict={'fontsize': 14})\nplt.ylabel(\"ISUP Grade\", labelpad=6.5, fontdict={'fontsize': 12})\nplt.xlabel(\"Institute\", labelpad=7, fontdict={'fontsize': 13})\nplt.yticks()\nplt.tick_params(axis='x', length=0)\nsns.despine(left=True, bottom=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Boxplot further shows the inbalance","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Key Findings from EDA of train.csv\n- Well there are a relatively equal amount of samples provided from both Institutions, Radboud however provides more serious cancer biospies than Karolinska. I don't know how this will impact my model\n- There is a serious class inbalance in the dataset. ISUP Grade 0 and 1 samples makeup 52% of the samples. But there are relatively equal sample amounts across ISUP Grade 2-5 samples.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# EDA on Images\n- With my (non)existing knowledge of visual machine learning I'm pretty ignorant of the requirements of image EDA (apart from inspecting each image individually and comparing them to train_csv.\n\n- Granted their are many kaggles that have made some interesting observations\n     - https://www.kaggle.com/rohitsingh9990/panda-eda-better-visualization-simple-baseline has identified some of images with markings\n     - https://www.kaggle.com/akensert/panda-removal-of-pen-marks used AI to remove the pen marks\n     - https://www.kaggle.com/c/prostate-cancer-grade-assessment/discussion/151323 has put in the effort to idenfity >600 suspicious images, ranging from pen marks, blank images, to blank images with a grading score of >0 (wow). ","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Images with Markers \n- In the Data Description, it is noted that some of the training test sets have markings on them\n- Thanks to rohitsingh9990, he has found some of them. https://www.kaggle.com/rohitsingh9990/panda-eda-better-visualization-simple-baseline","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"pen_marked_images = [\n    'fd6fe1a3985b17d067f2cb4d5bc1e6e1',\n    'ebb6a080d72e09f6481721ef9f88c472',\n    'ebb6d5ca45942536f78beb451ee43cc4',\n    'ea9d52d65500acc9b9d89eb6b82cdcdf',\n    'e726a8eac36c3d91c3c4f9edba8ba713',\n    'e90abe191f61b6fed6d6781c8305fe4b',\n    'fd0bb45eba479a7f7d953f41d574bf9f',\n    'ff10f937c3d52eff6ad4dd733f2bc3ac',\n    'feee2e895355a921f2b75b54debad328',\n    'feac91652a1c5accff08217d19116f1c',\n    'fb01a0a69517bb47d7f4699b6217f69d',\n    'f00ec753b5618cfb30519db0947fe724',\n    'e9a4f528b33479412ee019e155e1a197',\n    'f062f6c1128e0e9d51a76747d9018849',\n    'f39bf22d9a2f313425ee201932bac91a',\n]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Display One of Them\nimage_name = pen_marked_images[0]+\".tiff\"\nslide = openslide.OpenSlide(os.path.join(train_dir,image_name))\ndisplay(slide.get_thumbnail(size=(400,500)))\nslide.close()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# What's Next?\n- Become more familar with the tools of image machine learning\n- Explore Kaggle Community Discussions on how to wrangle image data\n- Creating a working model asap is a high priority","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}