{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":18647,"databundleVersionId":1126921,"sourceType":"competition"},{"sourceId":1101206,"sourceType":"datasetVersion","datasetId":615046}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        os.path.join(dirname, filename)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-16T01:40:07.11713Z","iopub.execute_input":"2024-04-16T01:40:07.117755Z","iopub.status.idle":"2024-04-16T01:41:05.717787Z","shell.execute_reply.started":"2024-04-16T01:40:07.117723Z","shell.execute_reply":"2024-04-16T01:41:05.716772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:05.719814Z","iopub.execute_input":"2024-04-16T01:41:05.720245Z","iopub.status.idle":"2024-04-16T01:41:05.762106Z","shell.execute_reply.started":"2024-04-16T01:41:05.720218Z","shell.execute_reply":"2024-04-16T01:41:05.761253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:05.763183Z","iopub.execute_input":"2024-04-16T01:41:05.763493Z","iopub.status.idle":"2024-04-16T01:41:05.784924Z","shell.execute_reply.started":"2024-04-16T01:41:05.763469Z","shell.execute_reply":"2024-04-16T01:41:05.78397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_gleason_score(score):\n    try:\n        parts = score.split('+')\n        return int(parts[0]) + int(parts[1])\n    except (ValueError, AttributeError):\n        return np.nan\n\ntrain_df['gleason_score_numeric'] = train_df['gleason_score'].apply(convert_gleason_score)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:05.786113Z","iopub.execute_input":"2024-04-16T01:41:05.786536Z","iopub.status.idle":"2024-04-16T01:41:05.807358Z","shell.execute_reply.started":"2024-04-16T01:41:05.786499Z","shell.execute_reply":"2024-04-16T01:41:05.806582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:05.810244Z","iopub.execute_input":"2024-04-16T01:41:05.8105Z","iopub.status.idle":"2024-04-16T01:41:05.824994Z","shell.execute_reply.started":"2024-04-16T01:41:05.810478Z","shell.execute_reply":"2024-04-16T01:41:05.824167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:05.826077Z","iopub.execute_input":"2024-04-16T01:41:05.826343Z","iopub.status.idle":"2024-04-16T01:41:07.126938Z","shell.execute_reply.started":"2024-04-16T01:41:05.826322Z","shell.execute_reply":"2024-04-16T01:41:07.126207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:07.128058Z","iopub.execute_input":"2024-04-16T01:41:07.128435Z","iopub.status.idle":"2024-04-16T01:41:07.132982Z","shell.execute_reply.started":"2024-04-16T01:41:07.12841Z","shell.execute_reply":"2024-04-16T01:41:07.13193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nplt.subplot(1, 2, 1)\nsns.countplot(x='isup_grade', data=train_df)\nplt.title('Distribution of ISUP Grade')\n\nfor p in plt.gca().patches:\n    plt.gca().annotate(f\"{p.get_height()}\", (p.get_x() + p.get_width() / 2., p.get_height()), ha='center', va='center', fontsize=10, color='black', xytext=(0, 5), textcoords='offset points')\n\nplt.subplot(1, 2, 2)\nsns.countplot(x='gleason_score_numeric', data=train_df)\nplt.title('Distribution of Gleason Score (Numeric)')\n\nfor p in plt.gca().patches:\n    plt.gca().annotate(f\"{p.get_height()}\", (p.get_x() + p.get_width() / 2., p.get_height()), ha='center', va='center', fontsize=10, color='black', xytext=(0, 5), textcoords='offset points')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:07.134082Z","iopub.execute_input":"2024-04-16T01:41:07.134356Z","iopub.status.idle":"2024-04-16T01:41:07.790413Z","shell.execute_reply.started":"2024-04-16T01:41:07.134334Z","shell.execute_reply":"2024-04-16T01:41:07.789563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:07.791646Z","iopub.execute_input":"2024-04-16T01:41:07.791984Z","iopub.status.idle":"2024-04-16T01:41:08.017027Z","shell.execute_reply.started":"2024-04-16T01:41:07.791952Z","shell.execute_reply":"2024-04-16T01:41:08.016337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_dir = '/kaggle/input/prostate-cancer-grade-assessment/train_images/'","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:08.017969Z","iopub.execute_input":"2024-04-16T01:41:08.018235Z","iopub.status.idle":"2024-04-16T01:41:08.022449Z","shell.execute_reply.started":"2024-04-16T01:41:08.018201Z","shell.execute_reply":"2024-04-16T01:41:08.021452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_size = 256","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:08.023511Z","iopub.execute_input":"2024-04-16T01:41:08.023768Z","iopub.status.idle":"2024-04-16T01:41:08.032045Z","shell.execute_reply.started":"2024-04-16T01:41:08.023737Z","shell.execute_reply":"2024-04-16T01:41:08.03121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_ids = train_df['image_id'].iloc[:9].tolist()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:08.033055Z","iopub.execute_input":"2024-04-16T01:41:08.03332Z","iopub.status.idle":"2024-04-16T01:41:08.042301Z","shell.execute_reply.started":"2024-04-16T01:41:08.033289Z","shell.execute_reply":"2024-04-16T01:41:08.041561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import openslide","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:08.043251Z","iopub.execute_input":"2024-04-16T01:41:08.043506Z","iopub.status.idle":"2024-04-16T01:41:08.14387Z","shell.execute_reply.started":"2024-04-16T01:41:08.043485Z","shell.execute_reply":"2024-04-16T01:41:08.143209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = train_df['image_id'].iloc[0]\nfull_image_path = image_dir + image_id + '.tiff'\n\ntry:\n\n    example = openslide.OpenSlide(full_image_path)\n\n    clipped_example = example.read_region((5000, 5000), 0, (image_size, image_size))\n\n    plt.imshow(clipped_example)\n    plt.title(f\"Image ID: {image_id}\")\n    plt.axis('off')\n\n    example.close()\nexcept Exception as e:\n    print(f\"Error loading image {image_id}: {e}\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:08.148728Z","iopub.execute_input":"2024-04-16T01:41:08.149024Z","iopub.status.idle":"2024-04-16T01:41:08.444806Z","shell.execute_reply.started":"2024-04-16T01:41:08.149001Z","shell.execute_reply":"2024-04-16T01:41:08.44399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"image_path\"] = [image_dir+image_id+\".tiff\" for image_id in train_df[\"image_id\"]]","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:08.445792Z","iopub.execute_input":"2024-04-16T01:41:08.446026Z","iopub.status.idle":"2024-04-16T01:41:08.456222Z","shell.execute_reply.started":"2024-04-16T01:41:08.446005Z","shell.execute_reply":"2024-04-16T01:41:08.455257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:08.457123Z","iopub.execute_input":"2024-04-16T01:41:08.457388Z","iopub.status.idle":"2024-04-16T01:41:08.477887Z","shell.execute_reply.started":"2024-04-16T01:41:08.457365Z","shell.execute_reply":"2024-04-16T01:41:08.477065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='isup_grade')\nplt.title('Distribution of ISUP grades')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:08.478869Z","iopub.execute_input":"2024-04-16T01:41:08.479171Z","iopub.status.idle":"2024-04-16T01:41:08.660988Z","shell.execute_reply.started":"2024-04-16T01:41:08.479148Z","shell.execute_reply":"2024-04-16T01:41:08.66021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nplt.pie(train_df['isup_grade'].value_counts(), labels=train_df['isup_grade'].unique(), autopct='%1.1f%%', startangle=140)\nplt.title('Distribution of ISUP grades')\nplt.axis('equal')  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:08.662174Z","iopub.execute_input":"2024-04-16T01:41:08.662471Z","iopub.status.idle":"2024-04-16T01:41:08.856758Z","shell.execute_reply.started":"2024-04-16T01:41:08.662435Z","shell.execute_reply":"2024-04-16T01:41:08.855634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='data_provider')\nplt.title('Distribution of Data Providers')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:08.858086Z","iopub.execute_input":"2024-04-16T01:41:08.858756Z","iopub.status.idle":"2024-04-16T01:41:09.052091Z","shell.execute_reply.started":"2024-04-16T01:41:08.858721Z","shell.execute_reply":"2024-04-16T01:41:09.051221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nplt.pie(train_df['data_provider'].value_counts(), labels=train_df['data_provider'].unique(), autopct='%1.1f%%', startangle=140)\nplt.title('Distribution of Data Providers')\nplt.axis('equal')  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:09.053035Z","iopub.execute_input":"2024-04-16T01:41:09.053288Z","iopub.status.idle":"2024-04-16T01:41:09.166315Z","shell.execute_reply.started":"2024-04-16T01:41:09.053266Z","shell.execute_reply":"2024-04-16T01:41:09.16546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='gleason_score_numeric')\nplt.title('Distribution of Gleason Scores')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:09.167357Z","iopub.execute_input":"2024-04-16T01:41:09.167916Z","iopub.status.idle":"2024-04-16T01:41:09.347996Z","shell.execute_reply.started":"2024-04-16T01:41:09.167891Z","shell.execute_reply":"2024-04-16T01:41:09.346979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_matrix = train_df[['isup_grade', 'gleason_score_numeric']].corr()\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm')\nplt.title('Correlation Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:09.34937Z","iopub.execute_input":"2024-04-16T01:41:09.349777Z","iopub.status.idle":"2024-04-16T01:41:09.627816Z","shell.execute_reply.started":"2024-04-16T01:41:09.349746Z","shell.execute_reply":"2024-04-16T01:41:09.62684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_item_count = int(len(train_df) * 0.8)\nvalidation_df = train_df[training_item_count:]\ntrain_df = train_df[:training_item_count]","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:09.628919Z","iopub.execute_input":"2024-04-16T01:41:09.629177Z","iopub.status.idle":"2024-04-16T01:41:09.633794Z","shell.execute_reply.started":"2024-04-16T01:41:09.629154Z","shell.execute_reply":"2024-04-16T01:41:09.632974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_single_sample(image_path, image_size=256, training=False, display=False):\n    image = openslide.OpenSlide(image_path)\n    mask_path = image_path.replace(\"train_images\", \"train_label_masks\").replace(\".tiff\", \"_mask.tiff\")\n    mask = openslide.OpenSlide(mask_path)\n    \n    stacked_image = []\n    groundtruth_per_image = []\n    \n    maximum_iteration = 0\n    selected_sample = False\n    while not selected_sample:\n        sampling_start_x = randint(image_size, image.dimensions[0] - image_size)\n        sampling_start_y = randint(image_size, image.dimensions[1] - image_size)\n\n        clipped_sample = image.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n        clipped_array = np.asarray(clipped_sample)\n        \n        if (not np.all(clipped_array == 255) and np.std(clipped_array) > 20) or maximum_iteration > 200:\n            if display:\n                plt.imshow(clipped_sample)\n                plt.show()\n                \n            sampled_image = clipped_array[:, :, :3]\n            \n            if training:\n                clipped_mask = mask.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n                groundtruth_per_image.append(np.mean(np.asarray(clipped_mask)[:, :, 0]))\n            \n            selected_sample = True\n        maximum_iteration += 1\n    \n    if training: \n        return np.array(sampled_image), np.array(groundtruth_per_image)\n    else:\n        return np.array(sampled_image)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:09.634925Z","iopub.execute_input":"2024-04-16T01:41:09.635231Z","iopub.status.idle":"2024-04-16T01:41:09.646317Z","shell.execute_reply.started":"2024-04-16T01:41:09.635202Z","shell.execute_reply":"2024-04-16T01:41:09.645576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_random_samples(image_path, image_size=256, display=False):\n    image = openslide.OpenSlide(image_path)\n    stacked_image = []\n    \n    selected_samples = 0\n    maximum_iteration = 0\n    while selected_samples < 3:\n        sampling_start_x = randint(image_size, image.dimensions[0] - image_size)\n        sampling_start_y = randint(image_size, image.dimensions[1] - image_size)\n\n        clipped_sample = image.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n        clipped_array = np.asarray(clipped_sample)\n        \n        if (not np.all(clipped_array == 255) and np.std(clipped_array) > 20) or maximum_iteration > 200:\n            if display:\n                plt.imshow(clipped_sample)\n                plt.show()\n\n            stacked_image.append(clipped_array[:, :, :3])\n            selected_samples += 1\n        maximum_iteration += 1\n    return np.array(stacked_image)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:09.64736Z","iopub.execute_input":"2024-04-16T01:41:09.647674Z","iopub.status.idle":"2024-04-16T01:41:09.66016Z","shell.execute_reply.started":"2024-04-16T01:41:09.647644Z","shell.execute_reply":"2024-04-16T01:41:09.659421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_single_image_generator(image_path_list, batch_size=16):\n    while True:\n        for start in range(0, len(image_path_list), batch_size):\n            X_batch = []\n            Y_batch = []\n            end = min(start + batch_size, training_item_count)\n\n            image_info_list = [get_single_sample(image_path, training=True) for image_path in image_path_list[start:end]]\n            X_batch = np.array([image_info[0]/255. for image_info in image_info_list])\n            Y_batch = np.array([image_info[1] for image_info in image_info_list])\n            \n            yield X_batch, Y_batch ","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:09.661276Z","iopub.execute_input":"2024-04-16T01:41:09.661836Z","iopub.status.idle":"2024-04-16T01:41:09.673651Z","shell.execute_reply.started":"2024-04-16T01:41:09.661806Z","shell.execute_reply":"2024-04-16T01:41:09.672748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from random import randint ","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:09.674737Z","iopub.execute_input":"2024-04-16T01:41:09.675049Z","iopub.status.idle":"2024-04-16T01:41:09.6821Z","shell.execute_reply.started":"2024-04-16T01:41:09.675019Z","shell.execute_reply":"2024-04-16T01:41:09.681331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_samples = get_random_samples(train_df.iloc[0].image_path, display=True)\nprint(\"Random samples shape:\", random_samples.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:09.683107Z","iopub.execute_input":"2024-04-16T01:41:09.683349Z","iopub.status.idle":"2024-04-16T01:41:10.887277Z","shell.execute_reply.started":"2024-04-16T01:41:09.683326Z","shell.execute_reply":"2024-04-16T01:41:10.886419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_path = train_df.iloc[0].image_path  \nsample_image, groundtruth = get_single_sample(image_path, training=True, display=True)\n\nprint(\"Sample Image Shape:\", sample_image.shape)\nprint(\"Ground Truth:\", groundtruth)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:10.888368Z","iopub.execute_input":"2024-04-16T01:41:10.888636Z","iopub.status.idle":"2024-04-16T01:41:12.157052Z","shell.execute_reply.started":"2024-04-16T01:41:10.888613Z","shell.execute_reply":"2024-04-16T01:41:12.15591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport PIL\nfrom IPython.display import Image, display\nfrom keras.applications.vgg16 import VGG16,preprocess_input\nimport plotly.graph_objs as go\nfrom sklearn.model_selection import train_test_split\nfrom keras.models import Sequential, Model,load_model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator, load_img, img_to_array\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Dense, Dropout, Input, Flatten,BatchNormalization,Activation\nfrom keras.layers import GlobalMaxPooling2D\nfrom keras.models import Model\nfrom keras.optimizers import Adam, SGD, RMSprop\nfrom keras.callbacks import ModelCheckpoint, Callback, EarlyStopping\nfrom keras.utils import to_categorical\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport gc\nimport skimage.io\nfrom sklearn.model_selection import KFold\nimport tensorflow as tf\nfrom tensorflow.python.keras import backend as K\nsess = K.get_session()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:12.158488Z","iopub.execute_input":"2024-04-16T01:41:12.158824Z","iopub.status.idle":"2024-04-16T01:41:25.543624Z","shell.execute_reply.started":"2024-04-16T01:41:12.158792Z","shell.execute_reply":"2024-04-16T01:41:25.542798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img=openslide.OpenSlide('/kaggle/input/prostate-cancer-grade-assessment/train_images/2fd1c7dc4a0f3a546a59717d8e9d28c3.tiff')\ndisplay(img.get_thumbnail(size=(512,512)))","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:25.544648Z","iopub.execute_input":"2024-04-16T01:41:25.545171Z","iopub.status.idle":"2024-04-16T01:41:25.830828Z","shell.execute_reply.started":"2024-04-16T01:41:25.545147Z","shell.execute_reply":"2024-04-16T01:41:25.829883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.dimensions","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:25.831878Z","iopub.execute_input":"2024-04-16T01:41:25.832131Z","iopub.status.idle":"2024-04-16T01:41:25.837904Z","shell.execute_reply.started":"2024-04-16T01:41:25.832108Z","shell.execute_reply":"2024-04-16T01:41:25.837134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['isup_grade'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:25.838876Z","iopub.execute_input":"2024-04-16T01:41:25.839142Z","iopub.status.idle":"2024-04-16T01:41:25.85108Z","shell.execute_reply.started":"2024-04-16T01:41:25.839112Z","shell.execute_reply":"2024-04-16T01:41:25.850131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels=[]\ndata=[]\ndata_dir='/kaggle/input/panda-resized-train-data-512x512/train_images/train_images/'\nfor i in range(train_df.shape[0]):\n    data.append(data_dir + train_df['image_id'].iloc[i]+'.png')\n    labels.append(train_df['isup_grade'].iloc[i])\ndf=pd.DataFrame(data)\ndf.columns=['images']\ndf['isup_grade']=labels","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:25.852472Z","iopub.execute_input":"2024-04-16T01:41:25.852838Z","iopub.status.idle":"2024-04-16T01:41:26.142868Z","shell.execute_reply.started":"2024-04-16T01:41:25.852807Z","shell.execute_reply":"2024-04-16T01:41:26.14221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:26.144065Z","iopub.execute_input":"2024-04-16T01:41:26.144452Z","iopub.status.idle":"2024-04-16T01:41:26.154692Z","shell.execute_reply.started":"2024-04-16T01:41:26.14442Z","shell.execute_reply":"2024-04-16T01:41:26.153857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(df['images'],df['isup_grade'], test_size=0.2, random_state=41)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:26.155838Z","iopub.execute_input":"2024-04-16T01:41:26.156169Z","iopub.status.idle":"2024-04-16T01:41:26.166025Z","shell.execute_reply.started":"2024-04-16T01:41:26.156139Z","shell.execute_reply":"2024-04-16T01:41:26.165302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.DataFrame(X_train)\ntrain.columns=['images']\ntrain['isup_grade']=y_train\n\nvalidation=pd.DataFrame(X_val)\nvalidation.columns=['images']\nvalidation['isup_grade']=y_val\n\ntrain['isup_grade']=train['isup_grade'].astype(str)\nvalidation['isup_grade']=validation['isup_grade'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:26.167113Z","iopub.execute_input":"2024-04-16T01:41:26.16744Z","iopub.status.idle":"2024-04-16T01:41:26.181621Z","shell.execute_reply.started":"2024-04-16T01:41:26.167411Z","shell.execute_reply":"2024-04-16T01:41:26.180852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale=1./255,rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,horizontal_flip=True)\nval_datagen=train_datagen = ImageDataGenerator(rescale=1./255)\ntrain_generator = train_datagen.flow_from_dataframe(\n    train,\n    x_col='images',\n    y_col='isup_grade',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='categorical')\n\nvalidation_generator = val_datagen.flow_from_dataframe(\n    validation,\n    x_col='images',\n    y_col='isup_grade',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='categorical')","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:26.18265Z","iopub.execute_input":"2024-04-16T01:41:26.182907Z","iopub.status.idle":"2024-04-16T01:41:30.253008Z","shell.execute_reply.started":"2024-04-16T01:41:26.182879Z","shell.execute_reply":"2024-04-16T01:41:30.252245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import VGG16\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Flatten\nfrom tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:30.254073Z","iopub.execute_input":"2024-04-16T01:41:30.25436Z","iopub.status.idle":"2024-04-16T01:41:30.265531Z","shell.execute_reply.started":"2024-04-16T01:41:30.254335Z","shell.execute_reply":"2024-04-16T01:41:30.264611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = VGG16(weights='imagenet', include_top=False, input_shape=(224, 224, 3))","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:30.266403Z","iopub.execute_input":"2024-04-16T01:41:30.266634Z","iopub.status.idle":"2024-04-16T01:41:31.349769Z","shell.execute_reply.started":"2024-04-16T01:41:30.266614Z","shell.execute_reply":"2024-04-16T01:41:31.348967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = base_model.output\nx = Flatten()(x)\nx = Dense(1024, activation='relu')(x)  \nx = Dropout(0.5)(x)  \nx = BatchNormalization()(x)  \nx = Dense(512, activation='relu')(x)  \nx = Dropout(0.5)(x)  \nx = BatchNormalization()(x)  \npredictions = Dense(6, activation='softmax')(x)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:31.356665Z","iopub.execute_input":"2024-04-16T01:41:31.35695Z","iopub.status.idle":"2024-04-16T01:41:31.408555Z","shell.execute_reply.started":"2024-04-16T01:41:31.356926Z","shell.execute_reply":"2024-04-16T01:41:31.407843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model(inputs=base_model.input, outputs=predictions)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:31.409764Z","iopub.execute_input":"2024-04-16T01:41:31.410348Z","iopub.status.idle":"2024-04-16T01:41:31.419991Z","shell.execute_reply.started":"2024-04-16T01:41:31.410316Z","shell.execute_reply":"2024-04-16T01:41:31.419141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:31.421135Z","iopub.execute_input":"2024-04-16T01:41:31.421434Z","iopub.status.idle":"2024-04-16T01:41:31.429119Z","shell.execute_reply.started":"2024-04-16T01:41:31.421405Z","shell.execute_reply":"2024-04-16T01:41:31.428166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=Adam(learning_rate=0.0001), loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:31.430245Z","iopub.execute_input":"2024-04-16T01:41:31.430798Z","iopub.status.idle":"2024-04-16T01:41:31.446095Z","shell.execute_reply.started":"2024-04-16T01:41:31.430769Z","shell.execute_reply":"2024-04-16T01:41:31.445232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    steps_per_epoch=len(train_generator),\n    epochs=3,  \n    validation_data=validation_generator,\n    validation_steps=len(validation_generator)\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:41:31.446999Z","iopub.execute_input":"2024-04-16T01:41:31.4484Z","iopub.status.idle":"2024-04-16T01:46:54.251071Z","shell.execute_reply.started":"2024-04-16T01:41:31.448378Z","shell.execute_reply":"2024-04-16T01:46:54.2503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, accuracy = model.evaluate(validation_generator)\nprint(\"Validation Loss:\", loss)\nprint(\"Validation Accuracy:\", accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:46:54.252344Z","iopub.execute_input":"2024-04-16T01:46:54.252624Z","iopub.status.idle":"2024-04-16T01:47:05.647824Z","shell.execute_reply.started":"2024-04-16T01:46:54.2526Z","shell.execute_reply":"2024-04-16T01:47:05.646888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\n\ny_pred = model.predict(validation_generator)\ny_pred_classes = np.argmax(y_pred, axis=1)\n\ny_true = validation_generator.classes\n\nconf_matrix = confusion_matrix(y_true, y_pred_classes)\n\nclass_report = classification_report(y_true, y_pred_classes)\n\nprint(\"Confusion Matrix:\")\nprint(conf_matrix)\n\nprint(\"\\nClassification Report:\")\nprint(class_report)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:47:05.648951Z","iopub.execute_input":"2024-04-16T01:47:05.649234Z","iopub.status.idle":"2024-04-16T01:47:17.971041Z","shell.execute_reply.started":"2024-04-16T01:47:05.649201Z","shell.execute_reply":"2024-04-16T01:47:17.970088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.legend()\nplt.show()\n\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.title('Training and Validation Loss')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:47:17.972326Z","iopub.execute_input":"2024-04-16T01:47:17.972636Z","iopub.status.idle":"2024-04-16T01:47:18.45163Z","shell.execute_reply.started":"2024-04-16T01:47:17.972612Z","shell.execute_reply":"2024-04-16T01:47:18.450751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import VGG19","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:47:18.452792Z","iopub.execute_input":"2024-04-16T01:47:18.453038Z","iopub.status.idle":"2024-04-16T01:47:18.457144Z","shell.execute_reply.started":"2024-04-16T01:47:18.453017Z","shell.execute_reply":"2024-04-16T01:47:18.456244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = VGG19(weights='imagenet', include_top=False, input_shape=(224, 224, 3))","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:47:18.458121Z","iopub.execute_input":"2024-04-16T01:47:18.458399Z","iopub.status.idle":"2024-04-16T01:47:19.676579Z","shell.execute_reply.started":"2024-04-16T01:47:18.458377Z","shell.execute_reply":"2024-04-16T01:47:19.675738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = base_model.output\nx = Flatten()(x)\nx = Dense(1024, activation='relu')(x)  \nx = Dropout(0.5)(x)  \nx = BatchNormalization()(x)  \nx = Dense(512, activation='relu')(x)  \nx = Dropout(0.5)(x)  \nx = BatchNormalization()(x)  \npredictions = Dense(6, activation='softmax')(x)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:47:19.677595Z","iopub.execute_input":"2024-04-16T01:47:19.677844Z","iopub.status.idle":"2024-04-16T01:47:19.723341Z","shell.execute_reply.started":"2024-04-16T01:47:19.677823Z","shell.execute_reply":"2024-04-16T01:47:19.72248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model(inputs=base_model.input, outputs=predictions)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:47:19.724421Z","iopub.execute_input":"2024-04-16T01:47:19.724685Z","iopub.status.idle":"2024-04-16T01:47:19.734348Z","shell.execute_reply.started":"2024-04-16T01:47:19.724663Z","shell.execute_reply":"2024-04-16T01:47:19.733403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=Adam(0.001), loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:47:19.735491Z","iopub.execute_input":"2024-04-16T01:47:19.735797Z","iopub.status.idle":"2024-04-16T01:47:19.745944Z","shell.execute_reply.started":"2024-04-16T01:47:19.735775Z","shell.execute_reply":"2024-04-16T01:47:19.745205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    steps_per_epoch=len(train_generator),\n    epochs=3, \n    validation_data=validation_generator,\n    validation_steps=len(validation_generator)\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:47:19.746859Z","iopub.execute_input":"2024-04-16T01:47:19.747095Z","iopub.status.idle":"2024-04-16T01:51:17.950821Z","shell.execute_reply.started":"2024-04-16T01:47:19.747074Z","shell.execute_reply":"2024-04-16T01:51:17.950026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, accuracy = model.evaluate(validation_generator)\nprint(\"Validation Loss:\", loss)\nprint(\"Validation Accuracy:\", accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:51:17.952229Z","iopub.execute_input":"2024-04-16T01:51:17.952528Z","iopub.status.idle":"2024-04-16T01:51:29.056424Z","shell.execute_reply.started":"2024-04-16T01:51:17.952501Z","shell.execute_reply":"2024-04-16T01:51:29.055514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(validation_generator)\ny_pred_classes = np.argmax(y_pred, axis=1)\n\ny_true = validation_generator.classes\n\nconf_matrix = confusion_matrix(y_true, y_pred_classes)\n\nclass_report = classification_report(y_true, y_pred_classes)\n\nprint(\"Confusion Matrix:\")\nprint(conf_matrix)\n\nprint(\"\\nClassification Report:\")\nprint(class_report)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:51:29.057773Z","iopub.execute_input":"2024-04-16T01:51:29.058381Z","iopub.status.idle":"2024-04-16T01:51:41.521314Z","shell.execute_reply.started":"2024-04-16T01:51:29.058345Z","shell.execute_reply":"2024-04-16T01:51:41.520447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import InceptionV3","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:51:41.522605Z","iopub.execute_input":"2024-04-16T01:51:41.522974Z","iopub.status.idle":"2024-04-16T01:51:41.527786Z","shell.execute_reply.started":"2024-04-16T01:51:41.522942Z","shell.execute_reply":"2024-04-16T01:51:41.526899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = InceptionV3(weights='imagenet', include_top=False, input_shape=(224, 224, 3))","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:51:41.528897Z","iopub.execute_input":"2024-04-16T01:51:41.529236Z","iopub.status.idle":"2024-04-16T01:51:43.821718Z","shell.execute_reply.started":"2024-04-16T01:51:41.529206Z","shell.execute_reply":"2024-04-16T01:51:43.820921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import GlobalAveragePooling2D","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:51:43.823043Z","iopub.execute_input":"2024-04-16T01:51:43.823344Z","iopub.status.idle":"2024-04-16T01:51:43.827626Z","shell.execute_reply.started":"2024-04-16T01:51:43.82332Z","shell.execute_reply":"2024-04-16T01:51:43.826729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = base_model.output\nx = GlobalAveragePooling2D()(x)\nx = Dense(1024, activation='relu')(x)\npredictions = Dense(6, activation='softmax')(x)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:51:43.828816Z","iopub.execute_input":"2024-04-16T01:51:43.829108Z","iopub.status.idle":"2024-04-16T01:51:43.853585Z","shell.execute_reply.started":"2024-04-16T01:51:43.829085Z","shell.execute_reply":"2024-04-16T01:51:43.852876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model(inputs=base_model.input, outputs=predictions)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:51:43.854532Z","iopub.execute_input":"2024-04-16T01:51:43.85479Z","iopub.status.idle":"2024-04-16T01:51:43.899443Z","shell.execute_reply.started":"2024-04-16T01:51:43.85477Z","shell.execute_reply":"2024-04-16T01:51:43.898605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in base_model.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:51:43.900489Z","iopub.execute_input":"2024-04-16T01:51:43.900818Z","iopub.status.idle":"2024-04-16T01:51:43.916358Z","shell.execute_reply.started":"2024-04-16T01:51:43.900788Z","shell.execute_reply":"2024-04-16T01:51:43.915587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=Adam(0.0001), loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-04-16T01:51:43.917387Z","iopub.execute_input":"2024-04-16T01:51:43.917639Z","iopub.status.idle":"2024-04-16T01:51:43.92973Z","shell.execute_reply.started":"2024-04-16T01:51:43.917618Z","shell.execute_reply":"2024-04-16T01:51:43.92897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    steps_per_epoch=len(train_generator),\n    epochs=20,  \n    validation_data=validation_generator,\n    validation_steps=len(validation_generator)\n)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T02:05:23.448472Z","iopub.execute_input":"2024-04-16T02:05:23.449343Z","iopub.status.idle":"2024-04-16T02:14:35.081205Z","shell.execute_reply.started":"2024-04-16T02:05:23.44931Z","shell.execute_reply":"2024-04-16T02:14:35.080242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, accuracy = model.evaluate(validation_generator)\nprint(\"Validation Loss:\", loss)\nprint(\"Validation Accuracy:\", accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T02:14:35.083395Z","iopub.execute_input":"2024-04-16T02:14:35.084427Z","iopub.status.idle":"2024-04-16T02:14:46.076135Z","shell.execute_reply.started":"2024-04-16T02:14:35.084387Z","shell.execute_reply":"2024-04-16T02:14:46.075238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(validation_generator)\ny_pred_classes = np.argmax(y_pred, axis=1)\n\ny_true = validation_generator.classes\n\nconf_matrix = confusion_matrix(y_true, y_pred_classes)\n\nclass_report = classification_report(y_true, y_pred_classes)\n\nprint(\"Confusion Matrix:\")\nprint(conf_matrix)\n\nprint(\"\\nClassification Report:\")\nprint(class_report)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T02:14:46.077297Z","iopub.execute_input":"2024-04-16T02:14:46.077581Z","iopub.status.idle":"2024-04-16T02:14:57.146709Z","shell.execute_reply.started":"2024-04-16T02:14:46.077558Z","shell.execute_reply":"2024-04-16T02:14:57.14582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.legend()\nplt.show()\n\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.title('Training and Validation Loss')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-16T02:14:57.148315Z","iopub.execute_input":"2024-04-16T02:14:57.148593Z","iopub.status.idle":"2024-04-16T02:14:57.795386Z","shell.execute_reply.started":"2024-04-16T02:14:57.148568Z","shell.execute_reply":"2024-04-16T02:14:57.794422Z"},"trusted":true},"execution_count":null,"outputs":[]}]}