{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":18647,"databundleVersionId":1126921,"sourceType":"competition"},{"sourceId":1101206,"sourceType":"datasetVersion","datasetId":615046}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-06T06:06:21.752214Z","iopub.execute_input":"2024-06-06T06:06:21.752561Z","iopub.status.idle":"2024-06-06T06:07:06.550303Z","shell.execute_reply.started":"2024-06-06T06:06:21.752531Z","shell.execute_reply":"2024-06-06T06:07:06.548998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:06.552099Z","iopub.execute_input":"2024-06-06T06:07:06.552763Z","iopub.status.idle":"2024-06-06T06:07:06.591123Z","shell.execute_reply.started":"2024-06-06T06:07:06.552731Z","shell.execute_reply":"2024-06-06T06:07:06.589803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:06.592476Z","iopub.execute_input":"2024-06-06T06:07:06.592778Z","iopub.status.idle":"2024-06-06T06:07:06.615407Z","shell.execute_reply.started":"2024-06-06T06:07:06.592752Z","shell.execute_reply":"2024-06-06T06:07:06.614292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_gleason_score(score):\n    try:\n        parts = score.split('+')\n        return int(parts[0]) + int(parts[1])\n    except (ValueError, AttributeError):\n        return np.nan\n\ntrain_df['gleason_score_numeric'] = train_df['gleason_score'].apply(convert_gleason_score)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:06.617452Z","iopub.execute_input":"2024-06-06T06:07:06.617729Z","iopub.status.idle":"2024-06-06T06:07:06.639207Z","shell.execute_reply.started":"2024-06-06T06:07:06.617706Z","shell.execute_reply":"2024-06-06T06:07:06.63832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:06.640354Z","iopub.execute_input":"2024-06-06T06:07:06.640695Z","iopub.status.idle":"2024-06-06T06:07:06.661644Z","shell.execute_reply.started":"2024-06-06T06:07:06.640665Z","shell.execute_reply":"2024-06-06T06:07:06.660759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:06.663128Z","iopub.execute_input":"2024-06-06T06:07:06.663963Z","iopub.status.idle":"2024-06-06T06:07:07.93375Z","shell.execute_reply.started":"2024-06-06T06:07:06.663935Z","shell.execute_reply":"2024-06-06T06:07:07.93292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:07.934863Z","iopub.execute_input":"2024-06-06T06:07:07.935315Z","iopub.status.idle":"2024-06-06T06:07:07.940005Z","shell.execute_reply.started":"2024-06-06T06:07:07.935284Z","shell.execute_reply":"2024-06-06T06:07:07.938963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nplt.subplot(1, 2, 1)\nsns.countplot(x='isup_grade', data=train_df)\nplt.title('Distribution of ISUP Grade')\n\nfor p in plt.gca().patches:\n    plt.gca().annotate(f\"{p.get_height()}\", (p.get_x() + p.get_width() / 2., p.get_height()), ha='center', va='center', fontsize=10, color='black', xytext=(0, 5), textcoords='offset points')\n\nplt.subplot(1, 2, 2)\nsns.countplot(x='gleason_score_numeric', data=train_df)\nplt.title('Distribution of Gleason Score (Numeric)')\n\nfor p in plt.gca().patches:\n    plt.gca().annotate(f\"{p.get_height()}\", (p.get_x() + p.get_width() / 2., p.get_height()), ha='center', va='center', fontsize=10, color='black', xytext=(0, 5), textcoords='offset points')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:07.941162Z","iopub.execute_input":"2024-06-06T06:07:07.941494Z","iopub.status.idle":"2024-06-06T06:07:08.644573Z","shell.execute_reply.started":"2024-06-06T06:07:07.941464Z","shell.execute_reply":"2024-06-06T06:07:08.643478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:08.645957Z","iopub.execute_input":"2024-06-06T06:07:08.64693Z","iopub.status.idle":"2024-06-06T06:07:08.875733Z","shell.execute_reply.started":"2024-06-06T06:07:08.646895Z","shell.execute_reply":"2024-06-06T06:07:08.874864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_dir = '/kaggle/input/prostate-cancer-grade-assessment/train_images/'","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:08.878938Z","iopub.execute_input":"2024-06-06T06:07:08.879211Z","iopub.status.idle":"2024-06-06T06:07:08.883376Z","shell.execute_reply.started":"2024-06-06T06:07:08.879187Z","shell.execute_reply":"2024-06-06T06:07:08.882325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_size = 128","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:08.884579Z","iopub.execute_input":"2024-06-06T06:07:08.884837Z","iopub.status.idle":"2024-06-06T06:07:08.894676Z","shell.execute_reply.started":"2024-06-06T06:07:08.884815Z","shell.execute_reply":"2024-06-06T06:07:08.893821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_ids = train_df['image_id'].iloc[:9].tolist()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:08.895718Z","iopub.execute_input":"2024-06-06T06:07:08.895982Z","iopub.status.idle":"2024-06-06T06:07:08.905763Z","shell.execute_reply.started":"2024-06-06T06:07:08.89596Z","shell.execute_reply":"2024-06-06T06:07:08.904967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import openslide","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:08.906792Z","iopub.execute_input":"2024-06-06T06:07:08.90705Z","iopub.status.idle":"2024-06-06T06:07:09.002152Z","shell.execute_reply.started":"2024-06-06T06:07:08.907028Z","shell.execute_reply":"2024-06-06T06:07:09.001314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = train_df['image_id'].iloc[0]\nfull_image_path = image_dir + image_id + '.tiff'\n\ntry:\n\n    example = openslide.OpenSlide(full_image_path)\n\n    clipped_example = example.read_region((5000, 5000), 0, (image_size, image_size))\n\n    plt.imshow(clipped_example)\n    plt.title(f\"Image ID: {image_id}\")\n    plt.axis('off')\n\n    example.close()\nexcept Exception as e:\n    print(f\"Error loading image {image_id}: {e}\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:09.00334Z","iopub.execute_input":"2024-06-06T06:07:09.003714Z","iopub.status.idle":"2024-06-06T06:07:09.337647Z","shell.execute_reply.started":"2024-06-06T06:07:09.003681Z","shell.execute_reply":"2024-06-06T06:07:09.336731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"image_path\"] = [image_dir+image_id+\".tiff\" for image_id in train_df[\"image_id\"]]","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:09.33883Z","iopub.execute_input":"2024-06-06T06:07:09.339125Z","iopub.status.idle":"2024-06-06T06:07:09.350601Z","shell.execute_reply.started":"2024-06-06T06:07:09.339101Z","shell.execute_reply":"2024-06-06T06:07:09.349626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:09.351763Z","iopub.execute_input":"2024-06-06T06:07:09.352061Z","iopub.status.idle":"2024-06-06T06:07:09.373061Z","shell.execute_reply.started":"2024-06-06T06:07:09.352035Z","shell.execute_reply":"2024-06-06T06:07:09.371929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='isup_grade')\nplt.title('Distribution of ISUP grades')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:09.374764Z","iopub.execute_input":"2024-06-06T06:07:09.375103Z","iopub.status.idle":"2024-06-06T06:07:09.64286Z","shell.execute_reply.started":"2024-06-06T06:07:09.375076Z","shell.execute_reply":"2024-06-06T06:07:09.641833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nplt.pie(train_df['isup_grade'].value_counts(), labels=train_df['isup_grade'].unique(), autopct='%1.1f%%', startangle=140)\nplt.title('Distribution of ISUP grades')\nplt.axis('equal')  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:09.644563Z","iopub.execute_input":"2024-06-06T06:07:09.644903Z","iopub.status.idle":"2024-06-06T06:07:09.858637Z","shell.execute_reply.started":"2024-06-06T06:07:09.644872Z","shell.execute_reply":"2024-06-06T06:07:09.856856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='data_provider')\nplt.title('Distribution of Data Providers')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:09.861417Z","iopub.execute_input":"2024-06-06T06:07:09.861977Z","iopub.status.idle":"2024-06-06T06:07:10.115281Z","shell.execute_reply.started":"2024-06-06T06:07:09.861927Z","shell.execute_reply":"2024-06-06T06:07:10.114303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nplt.pie(train_df['data_provider'].value_counts(), labels=train_df['data_provider'].unique(), autopct='%1.1f%%', startangle=140)\nplt.title('Distribution of Data Providers')\nplt.axis('equal')  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:10.116599Z","iopub.execute_input":"2024-06-06T06:07:10.116956Z","iopub.status.idle":"2024-06-06T06:07:10.278449Z","shell.execute_reply.started":"2024-06-06T06:07:10.116922Z","shell.execute_reply":"2024-06-06T06:07:10.277213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='gleason_score_numeric')\nplt.title('Distribution of Gleason Scores')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:10.280582Z","iopub.execute_input":"2024-06-06T06:07:10.281126Z","iopub.status.idle":"2024-06-06T06:07:10.568318Z","shell.execute_reply.started":"2024-06-06T06:07:10.281083Z","shell.execute_reply":"2024-06-06T06:07:10.567281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_matrix = train_df[['isup_grade', 'gleason_score_numeric']].corr()\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm')\nplt.title('Correlation Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:10.569785Z","iopub.execute_input":"2024-06-06T06:07:10.57024Z","iopub.status.idle":"2024-06-06T06:07:10.824927Z","shell.execute_reply.started":"2024-06-06T06:07:10.5702Z","shell.execute_reply":"2024-06-06T06:07:10.823807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_item_count = int(len(train_df) * 0.8)\nvalidation_df = train_df[training_item_count:]\ntrain_df = train_df[:training_item_count]","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:10.826098Z","iopub.execute_input":"2024-06-06T06:07:10.826408Z","iopub.status.idle":"2024-06-06T06:07:10.832375Z","shell.execute_reply.started":"2024-06-06T06:07:10.826366Z","shell.execute_reply":"2024-06-06T06:07:10.831462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_single_sample(image_path, image_size=256, training=False, display=False):\n    image = openslide.OpenSlide(image_path)\n    mask_path = image_path.replace(\"train_images\", \"train_label_masks\").replace(\".tiff\", \"_mask.tiff\")\n    mask = openslide.OpenSlide(mask_path)\n    \n    stacked_image = []\n    groundtruth_per_image = []\n    \n    maximum_iteration = 0\n    selected_sample = False\n    while not selected_sample:\n        sampling_start_x = randint(image_size, image.dimensions[0] - image_size)\n        sampling_start_y = randint(image_size, image.dimensions[1] - image_size)\n\n        clipped_sample = image.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n        clipped_array = np.asarray(clipped_sample)\n        \n        if (not np.all(clipped_array == 255) and np.std(clipped_array) > 20) or maximum_iteration > 200:\n            if display:\n                plt.imshow(clipped_sample)\n                plt.show()\n                \n            sampled_image = clipped_array[:, :, :3]\n            \n            if training:\n                clipped_mask = mask.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n                groundtruth_per_image.append(np.mean(np.asarray(clipped_mask)[:, :, 0]))\n            \n            selected_sample = True\n        maximum_iteration += 1\n    \n    if training: \n        return np.array(sampled_image), np.array(groundtruth_per_image)\n    else:\n        return np.array(sampled_image)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:10.833548Z","iopub.execute_input":"2024-06-06T06:07:10.833804Z","iopub.status.idle":"2024-06-06T06:07:10.84509Z","shell.execute_reply.started":"2024-06-06T06:07:10.833781Z","shell.execute_reply":"2024-06-06T06:07:10.844177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_random_samples(image_path, image_size=256, display=False):\n    image = openslide.OpenSlide(image_path)\n    stacked_image = []\n    \n    selected_samples = 0\n    maximum_iteration = 0\n    while selected_samples < 3:\n        sampling_start_x = randint(image_size, image.dimensions[0] - image_size)\n        sampling_start_y = randint(image_size, image.dimensions[1] - image_size)\n\n        clipped_sample = image.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n        clipped_array = np.asarray(clipped_sample)\n        \n        if (not np.all(clipped_array == 255) and np.std(clipped_array) > 20) or maximum_iteration > 200:\n            if display:\n                plt.imshow(clipped_sample)\n                plt.show()\n\n            stacked_image.append(clipped_array[:, :, :3])\n            selected_samples += 1\n        maximum_iteration += 1\n    return np.array(stacked_image)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:10.846342Z","iopub.execute_input":"2024-06-06T06:07:10.847071Z","iopub.status.idle":"2024-06-06T06:07:10.86197Z","shell.execute_reply.started":"2024-06-06T06:07:10.847036Z","shell.execute_reply":"2024-06-06T06:07:10.861002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_single_image_generator(image_path_list, batch_size=16):\n    while True:\n        for start in range(0, len(image_path_list), batch_size):\n            X_batch = []\n            Y_batch = []\n            end = min(start + batch_size, training_item_count)\n\n            image_info_list = [get_single_sample(image_path, training=True) for image_path in image_path_list[start:end]]\n            X_batch = np.array([image_info[0]/255. for image_info in image_info_list])\n            Y_batch = np.array([image_info[1] for image_info in image_info_list])\n            \n            yield X_batch, Y_batch ","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:10.863322Z","iopub.execute_input":"2024-06-06T06:07:10.863695Z","iopub.status.idle":"2024-06-06T06:07:10.879289Z","shell.execute_reply.started":"2024-06-06T06:07:10.863662Z","shell.execute_reply":"2024-06-06T06:07:10.878257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from random import randint ","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:10.880413Z","iopub.execute_input":"2024-06-06T06:07:10.880688Z","iopub.status.idle":"2024-06-06T06:07:10.891077Z","shell.execute_reply.started":"2024-06-06T06:07:10.880665Z","shell.execute_reply":"2024-06-06T06:07:10.889978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_samples = get_random_samples(train_df.iloc[0].image_path, display=True)\nprint(\"Random samples shape:\", random_samples.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:10.899637Z","iopub.execute_input":"2024-06-06T06:07:10.899969Z","iopub.status.idle":"2024-06-06T06:07:12.96703Z","shell.execute_reply.started":"2024-06-06T06:07:10.899944Z","shell.execute_reply":"2024-06-06T06:07:12.96606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_path = train_df.iloc[0].image_path  \nsample_image, groundtruth = get_single_sample(image_path, training=True, display=True)\n\nprint(\"Sample Image Shape:\", sample_image.shape)\nprint(\"Ground Truth:\", groundtruth)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:12.968316Z","iopub.execute_input":"2024-06-06T06:07:12.968643Z","iopub.status.idle":"2024-06-06T06:07:13.592297Z","shell.execute_reply.started":"2024-06-06T06:07:12.968618Z","shell.execute_reply":"2024-06-06T06:07:13.591216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport PIL\nfrom IPython.display import Image, display\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import resample\nfrom keras.models import Sequential, Model,load_model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator, load_img, img_to_array\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Dense, Dropout, Input, Flatten,BatchNormalization,Activation\nfrom keras.layers import GlobalMaxPooling2D\nfrom keras.models import Model\nfrom keras.optimizers import Adam, SGD, RMSprop\nfrom keras.callbacks import ModelCheckpoint, Callback, EarlyStopping\nfrom keras.utils import to_categorical\nimport gc\nimport skimage.io\nfrom sklearn.model_selection import KFold\nimport tensorflow as tf\nfrom tensorflow.python.keras import backend as K\nsess = K.get_session()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:13.593664Z","iopub.execute_input":"2024-06-06T06:07:13.594005Z","iopub.status.idle":"2024-06-06T06:07:26.504011Z","shell.execute_reply.started":"2024-06-06T06:07:13.593975Z","shell.execute_reply":"2024-06-06T06:07:26.502954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img=openslide.OpenSlide('/kaggle/input/prostate-cancer-grade-assessment/train_images/2fd1c7dc4a0f3a546a59717d8e9d28c3.tiff')\ndisplay(img.get_thumbnail(size=(512,512)))","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:26.505271Z","iopub.execute_input":"2024-06-06T06:07:26.50587Z","iopub.status.idle":"2024-06-06T06:07:26.791583Z","shell.execute_reply.started":"2024-06-06T06:07:26.505843Z","shell.execute_reply":"2024-06-06T06:07:26.790618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.dimensions","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:26.792917Z","iopub.execute_input":"2024-06-06T06:07:26.793735Z","iopub.status.idle":"2024-06-06T06:07:26.800191Z","shell.execute_reply.started":"2024-06-06T06:07:26.793697Z","shell.execute_reply":"2024-06-06T06:07:26.799142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['isup_grade'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:26.801594Z","iopub.execute_input":"2024-06-06T06:07:26.801933Z","iopub.status.idle":"2024-06-06T06:07:26.814506Z","shell.execute_reply.started":"2024-06-06T06:07:26.801902Z","shell.execute_reply":"2024-06-06T06:07:26.813449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels=[]\ndata=[]\ndata_dir='/kaggle/input/panda-resized-train-data-512x512/train_images/train_images/'\nfor i in range(train_df.shape[0]):\n    data.append(data_dir + train_df['image_id'].iloc[i]+'.png')\n    labels.append(train_df['isup_grade'].iloc[i])\ndf=pd.DataFrame(data)\ndf.columns=['images']\ndf['isup_grade']=labels","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:26.815642Z","iopub.execute_input":"2024-06-06T06:07:26.815916Z","iopub.status.idle":"2024-06-06T06:07:27.127059Z","shell.execute_reply.started":"2024-06-06T06:07:26.815893Z","shell.execute_reply":"2024-06-06T06:07:27.126223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:27.128078Z","iopub.execute_input":"2024-06-06T06:07:27.128346Z","iopub.status.idle":"2024-06-06T06:07:27.139612Z","shell.execute_reply.started":"2024-06-06T06:07:27.128323Z","shell.execute_reply":"2024-06-06T06:07:27.13857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_images(image_paths, num_images=5):\n    fig, axes = plt.subplots(1, num_images, figsize=(15, 5))\n    for i in range(num_images):\n        image_path = image_paths[i]\n        image = cv2.imread(image_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)  \n        axes[i].imshow(image)\n        axes[i].axis('off')\n        axes[i].set_title(f'Image {i+1}')\n    plt.show()\n\nimage_paths = df['images'].tolist()\n\ndisplay_images(image_paths, num_images=5)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:27.140765Z","iopub.execute_input":"2024-06-06T06:07:27.141072Z","iopub.status.idle":"2024-06-06T06:07:27.759879Z","shell.execute_reply.started":"2024-06-06T06:07:27.141045Z","shell.execute_reply":"2024-06-06T06:07:27.758886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('/kaggle/working/prostrate.csv')","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:27.761116Z","iopub.execute_input":"2024-06-06T06:07:27.761424Z","iopub.status.idle":"2024-06-06T06:07:27.8203Z","shell.execute_reply.started":"2024-06-06T06:07:27.761386Z","shell.execute_reply":"2024-06-06T06:07:27.819514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:27.821374Z","iopub.execute_input":"2024-06-06T06:07:27.821682Z","iopub.status.idle":"2024-06-06T06:07:27.825995Z","shell.execute_reply.started":"2024-06-06T06:07:27.821657Z","shell.execute_reply":"2024-06-06T06:07:27.82515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_size = 900\n\ngrouped = df.groupby('isup_grade')\n\ndf_new = pd.DataFrame(columns=df.columns)\n\nfor isup_grade, group in grouped:\n    if len(group) >= sample_size:\n        sampled_group = group.sample(n=sample_size, random_state=42)\n    else:\n        sampled_group = group.sample(n=len(group), random_state=42)\n    df_new = pd.concat([df_new, sampled_group])\n\ndf_new.reset_index(drop=True, inplace=True)\n\nprint(\"Shape of df_new:\", df_new.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:27.827264Z","iopub.execute_input":"2024-06-06T06:07:27.827573Z","iopub.status.idle":"2024-06-06T06:07:27.849963Z","shell.execute_reply.started":"2024-06-06T06:07:27.827538Z","shell.execute_reply":"2024-06-06T06:07:27.848937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_new","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:27.851074Z","iopub.execute_input":"2024-06-06T06:07:27.851363Z","iopub.status.idle":"2024-06-06T06:07:27.861964Z","shell.execute_reply.started":"2024-06-06T06:07:27.851339Z","shell.execute_reply":"2024-06-06T06:07:27.860987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_val_df, test_df = train_test_split(df_new, test_size=0.15, random_state=42, stratify=df_new['isup_grade'])","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:27.863231Z","iopub.execute_input":"2024-06-06T06:07:27.863626Z","iopub.status.idle":"2024-06-06T06:07:27.877984Z","shell.execute_reply.started":"2024-06-06T06:07:27.863589Z","shell.execute_reply":"2024-06-06T06:07:27.877108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, val_df = train_test_split(train_val_df, test_size=0.1765, random_state=42, stratify=train_val_df['isup_grade'])","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:27.87905Z","iopub.execute_input":"2024-06-06T06:07:27.879302Z","iopub.status.idle":"2024-06-06T06:07:27.898114Z","shell.execute_reply.started":"2024-06-06T06:07:27.87928Z","shell.execute_reply":"2024-06-06T06:07:27.897026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Total samples: {len(df_new)}\")\nprint(f\"Training samples: {len(train_df)}\")\nprint(f\"Validation samples: {len(val_df)}\")\nprint(f\"Test samples: {len(test_df)}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:27.899293Z","iopub.execute_input":"2024-06-06T06:07:27.899704Z","iopub.status.idle":"2024-06-06T06:07:27.905545Z","shell.execute_reply.started":"2024-06-06T06:07:27.899678Z","shell.execute_reply":"2024-06-06T06:07:27.904508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nimport shutil\nimport pathlib\nimport itertools\nfrom PIL import Image\n\nimport cv2\nimport seaborn as sns\nsns.set_style('darkgrid')\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, classification_report\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam, Adamax\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Activation, Dropout, BatchNormalization\nfrom tensorflow.keras import regularizers","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:27.906683Z","iopub.execute_input":"2024-06-06T06:07:27.906988Z","iopub.status.idle":"2024-06-06T06:07:27.921382Z","shell.execute_reply.started":"2024-06-06T06:07:27.906957Z","shell.execute_reply":"2024-06-06T06:07:27.920586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['isup_grade'] = train_df['isup_grade'].astype(str)\nval_df['isup_grade'] = val_df['isup_grade'].astype(str)\ntest_df['isup_grade'] = test_df['isup_grade'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:27.922451Z","iopub.execute_input":"2024-06-06T06:07:27.922787Z","iopub.status.idle":"2024-06-06T06:07:27.930493Z","shell.execute_reply.started":"2024-06-06T06:07:27.922755Z","shell.execute_reply":"2024-06-06T06:07:27.929597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32\nimg_size = (224, 224)\nchannels = 3\nimg_shape = (img_size[0], img_size[1], channels)\n\ntr_gen = ImageDataGenerator(rescale=1./255,rotation_range=20, width_shift_range=0.2,height_shift_range=0.2,horizontal_flip=True)\nts_gen = ImageDataGenerator(rescale=1./255)\n\ntrain_gen = tr_gen.flow_from_dataframe(train_df, x_col= 'images', y_col= 'isup_grade', target_size= img_size, class_mode= 'categorical',\n                                    color_mode= 'rgb', shuffle= True, batch_size= batch_size)\n\nvalid_gen = ts_gen.flow_from_dataframe(val_df, x_col= 'images', y_col= 'isup_grade', target_size= img_size, class_mode= 'categorical',\n                                    color_mode= 'rgb', shuffle= True, batch_size= batch_size)\n\ntest_gen = ts_gen.flow_from_dataframe(test_df, x_col= 'images', y_col= 'isup_grade', target_size= img_size, class_mode= 'categorical',\n                                    color_mode= 'rgb', shuffle= False, batch_size= batch_size)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:27.93156Z","iopub.execute_input":"2024-06-06T06:07:27.931872Z","iopub.status.idle":"2024-06-06T06:07:30.282543Z","shell.execute_reply.started":"2024-06-06T06:07:27.931848Z","shell.execute_reply":"2024-06-06T06:07:30.281444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:30.283777Z","iopub.execute_input":"2024-06-06T06:07:30.284102Z","iopub.status.idle":"2024-06-06T06:07:30.28944Z","shell.execute_reply.started":"2024-06-06T06:07:30.284075Z","shell.execute_reply":"2024-06-06T06:07:30.288433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"policy = tf.keras.mixed_precision.Policy('mixed_float16')\ntf.keras.mixed_precision.set_global_policy(policy)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:30.290838Z","iopub.execute_input":"2024-06-06T06:07:30.291696Z","iopub.status.idle":"2024-06-06T06:07:30.30063Z","shell.execute_reply.started":"2024-06-06T06:07:30.29166Z","shell.execute_reply":"2024-06-06T06:07:30.2997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import InceptionV3","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:30.301812Z","iopub.execute_input":"2024-06-06T06:07:30.302162Z","iopub.status.idle":"2024-06-06T06:07:30.313258Z","shell.execute_reply.started":"2024-06-06T06:07:30.302137Z","shell.execute_reply":"2024-06-06T06:07:30.312452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.layers import GlobalAveragePooling2D\n\ndef inceptionv3_model(num_classes=None):\n    model = InceptionV3(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n    x = GlobalAveragePooling2D()(model.output)  \n    output = Dense(num_classes, activation='softmax')(x)\n    model = Model(model.input, output)\n    return model\n\ninceptionv3_conv = inceptionv3_model(6)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:30.314516Z","iopub.execute_input":"2024-06-06T06:07:30.314798Z","iopub.status.idle":"2024-06-06T06:07:36.465684Z","shell.execute_reply.started":"2024-06-06T06:07:30.31476Z","shell.execute_reply":"2024-06-06T06:07:36.464667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:36.466899Z","iopub.execute_input":"2024-06-06T06:07:36.467191Z","iopub.status.idle":"2024-06-06T06:07:36.471618Z","shell.execute_reply.started":"2024-06-06T06:07:36.467161Z","shell.execute_reply":"2024-06-06T06:07:36.470713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import Callback","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:36.472801Z","iopub.execute_input":"2024-06-06T06:07:36.473143Z","iopub.status.idle":"2024-06-06T06:07:36.483769Z","shell.execute_reply.started":"2024-06-06T06:07:36.47311Z","shell.execute_reply":"2024-06-06T06:07:36.4828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class KappaScoreCallback(Callback):\n    def __init__(self, validation_data):\n        super().__init__()\n        self.validation_data = validation_data\n\n    def on_epoch_end(self, epoch, logs=None):\n        val_gen = self.validation_data\n        val_true = []\n        val_pred = []\n        for i in range(len(val_gen)):\n            x_val, y_val = val_gen[i]\n            val_true.extend(tf.argmax(y_val, axis=1).numpy())\n            val_pred.extend(tf.argmax(self.model.predict(x_val), axis=1).numpy())\n\n        kappa = cohen_kappa_score(val_true, val_pred)\n        print(f\"\\nEpoch {epoch + 1} - Cohen's Kappa: {kappa:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:36.484854Z","iopub.execute_input":"2024-06-06T06:07:36.485136Z","iopub.status.idle":"2024-06-06T06:07:36.493034Z","shell.execute_reply.started":"2024-06-06T06:07:36.485114Z","shell.execute_reply":"2024-06-06T06:07:36.492094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam\nopt = Adam(learning_rate=0.001)\ninceptionv3_conv.compile(loss='categorical_crossentropy', optimizer=opt, metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:36.494082Z","iopub.execute_input":"2024-06-06T06:07:36.494385Z","iopub.status.idle":"2024-06-06T06:07:36.52151Z","shell.execute_reply.started":"2024-06-06T06:07:36.494355Z","shell.execute_reply":"2024-06-06T06:07:36.520749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"opt = SGD(learning_rate=0.001)\ninceptionv3_conv.compile(loss='categorical_crossentropy',optimizer=opt,metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-06-05T06:46:10.829609Z","iopub.execute_input":"2024-06-05T06:46:10.829865Z","iopub.status.idle":"2024-06-05T06:46:10.855661Z","shell.execute_reply.started":"2024-06-05T06:46:10.829843Z","shell.execute_reply":"2024-06-05T06:46:10.854956Z"}}},{"cell_type":"code","source":"nb_epochs = 75\nbatch_size=16\nnb_train_steps = train_df.shape[0]//batch_size\nnb_val_steps=val_df.shape[0]//batch_size\nprint(\"Number of training and validation steps: {} and {}\".format(nb_train_steps,nb_val_steps))","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:36.522423Z","iopub.execute_input":"2024-06-06T06:07:36.522656Z","iopub.status.idle":"2024-06-06T06:07:36.527667Z","shell.execute_reply.started":"2024-06-06T06:07:36.522635Z","shell.execute_reply":"2024-06-06T06:07:36.526827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = inceptionv3_conv.fit(\n    train_gen,\n    steps_per_epoch=nb_train_steps,\n    epochs=nb_epochs,\n    validation_data=valid_gen,\n    validation_steps=nb_val_steps)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T06:07:36.528768Z","iopub.execute_input":"2024-06-06T06:07:36.529081Z","iopub.status.idle":"2024-06-06T07:32:36.082548Z","shell.execute_reply.started":"2024-06-06T06:07:36.529055Z","shell.execute_reply":"2024-06-06T07:32:36.081736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(14, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Train Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Model Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend(loc='upper left')\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Model Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend(loc='upper left')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:32:36.083784Z","iopub.execute_input":"2024-06-06T07:32:36.08406Z","iopub.status.idle":"2024-06-06T07:32:36.934425Z","shell.execute_reply.started":"2024-06-06T07:32:36.084036Z","shell.execute_reply":"2024-06-06T07:32:36.933364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:32:36.935545Z","iopub.execute_input":"2024-06-06T07:32:36.935828Z","iopub.status.idle":"2024-06-06T07:32:36.940053Z","shell.execute_reply.started":"2024-06-06T07:32:36.935802Z","shell.execute_reply":"2024-06-06T07:32:36.939216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gen.reset()\nY_pred = inceptionv3_conv.predict(test_gen, steps=len(test_gen), verbose=1)\ny_pred = np.argmax(Y_pred, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:32:36.941022Z","iopub.execute_input":"2024-06-06T07:32:36.941251Z","iopub.status.idle":"2024-06-06T07:33:04.477093Z","shell.execute_reply.started":"2024-06-06T07:32:36.94123Z","shell.execute_reply":"2024-06-06T07:33:04.476144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = test_gen.classes","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:04.478371Z","iopub.execute_input":"2024-06-06T07:33:04.478729Z","iopub.status.idle":"2024-06-06T07:33:04.48285Z","shell.execute_reply.started":"2024-06-06T07:33:04.478702Z","shell.execute_reply":"2024-06-06T07:33:04.481906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels = list(test_gen.class_indices.keys())","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:04.48401Z","iopub.execute_input":"2024-06-06T07:33:04.484285Z","iopub.status.idle":"2024-06-06T07:33:04.551695Z","shell.execute_reply.started":"2024-06-06T07:33:04.484254Z","shell.execute_reply":"2024-06-06T07:33:04.550976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conf_matrix = confusion_matrix(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:04.552661Z","iopub.execute_input":"2024-06-06T07:33:04.552896Z","iopub.status.idle":"2024-06-06T07:33:04.569049Z","shell.execute_reply.started":"2024-06-06T07:33:04.552875Z","shell.execute_reply":"2024-06-06T07:33:04.568093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conf_matrix","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:04.570428Z","iopub.execute_input":"2024-06-06T07:33:04.570746Z","iopub.status.idle":"2024-06-06T07:33:04.605056Z","shell.execute_reply.started":"2024-06-06T07:33:04.57069Z","shell.execute_reply":"2024-06-06T07:33:04.604163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues', xticklabels=class_labels, yticklabels=class_labels)\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:04.606064Z","iopub.execute_input":"2024-06-06T07:33:04.606337Z","iopub.status.idle":"2024-06-06T07:33:05.0666Z","shell.execute_reply.started":"2024-06-06T07:33:04.606313Z","shell.execute_reply":"2024-06-06T07:33:05.065722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_report = classification_report(y_true, y_pred, target_names=class_labels)\nprint('Classification Report')\nprint(class_report)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:05.067835Z","iopub.execute_input":"2024-06-06T07:33:05.0681Z","iopub.status.idle":"2024-06-06T07:33:05.101159Z","shell.execute_reply.started":"2024-06-06T07:33:05.068077Z","shell.execute_reply":"2024-06-06T07:33:05.100273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# kappa_callback = KappaScoreCallback(validation_data=valid_gen)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:05.102291Z","iopub.execute_input":"2024-06-06T07:33:05.102674Z","iopub.status.idle":"2024-06-06T07:33:05.106844Z","shell.execute_reply.started":"2024-06-06T07:33:05.102639Z","shell.execute_reply":"2024-06-06T07:33:05.105895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import VGG16","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:05.108148Z","iopub.execute_input":"2024-06-06T07:33:05.108563Z","iopub.status.idle":"2024-06-06T07:33:05.11652Z","shell.execute_reply.started":"2024-06-06T07:33:05.108507Z","shell.execute_reply":"2024-06-06T07:33:05.115721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def vgg16_model(num_classes=None):\n    model = VGG16(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n    x = Flatten()(model.output)\n    output = Dense(num_classes, activation='softmax')(x)\n    model = Model(model.input, output)\n    return model\n\nvgg_conv = vgg16_model(6)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:05.117469Z","iopub.execute_input":"2024-06-06T07:33:05.118494Z","iopub.status.idle":"2024-06-06T07:33:08.196848Z","shell.execute_reply.started":"2024-06-06T07:33:05.118456Z","shell.execute_reply":"2024-06-06T07:33:08.196044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import cohen_kappa_score","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:08.198199Z","iopub.execute_input":"2024-06-06T07:33:08.198488Z","iopub.status.idle":"2024-06-06T07:33:08.202147Z","shell.execute_reply.started":"2024-06-06T07:33:08.198462Z","shell.execute_reply":"2024-06-06T07:33:08.201172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from tensorflow.keras.callbacks import Callback","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:08.203532Z","iopub.execute_input":"2024-06-06T07:33:08.203871Z","iopub.status.idle":"2024-06-06T07:33:08.213617Z","shell.execute_reply.started":"2024-06-06T07:33:08.203841Z","shell.execute_reply":"2024-06-06T07:33:08.212821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"class KappaScoreCallback(Callback):\n    def __init__(self, validation_data):\n        super().__init__()\n        self.validation_data = validation_data\n\n    def on_epoch_end(self, epoch, logs=None):\n        val_gen = self.validation_data\n        val_true = []\n        val_pred = []\n        for i in range(len(val_gen)):\n            x_val, y_val = val_gen[i]\n            val_true.extend(tf.argmax(y_val, axis=1).numpy())\n            val_pred.extend(tf.argmax(self.model.predict(x_val), axis=1).numpy())\n\n        kappa = cohen_kappa_score(val_true, val_pred)\n        print(f\"\\nEpoch {epoch + 1} - Cohen's Kappa: {kappa:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-04T09:52:14.671902Z","iopub.execute_input":"2024-06-04T09:52:14.672891Z","iopub.status.idle":"2024-06-04T09:52:14.6803Z","shell.execute_reply.started":"2024-06-04T09:52:14.672857Z","shell.execute_reply":"2024-06-04T09:52:14.679321Z"}}},{"cell_type":"code","source":"opt = Adam(learning_rate=0.001)\n#opt = SGD(learning_rate=0.001)\nvgg_conv.compile(loss='categorical_crossentropy',optimizer=opt,metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:08.214737Z","iopub.execute_input":"2024-06-06T07:33:08.215058Z","iopub.status.idle":"2024-06-06T07:33:08.228041Z","shell.execute_reply.started":"2024-06-06T07:33:08.215027Z","shell.execute_reply":"2024-06-06T07:33:08.227281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"nb_epochs = 50\nbatch_size=16\nnb_train_steps = train_df.shape[0]//batch_size\nnb_val_steps=val_df.shape[0]//batch_size\nprint(\"Number of training and validation steps: {} and {}\".format(nb_train_steps,nb_val_steps))","metadata":{"execution":{"iopub.status.busy":"2024-06-04T09:52:28.526342Z","iopub.execute_input":"2024-06-04T09:52:28.526712Z","iopub.status.idle":"2024-06-04T09:52:28.532443Z","shell.execute_reply.started":"2024-06-04T09:52:28.526685Z","shell.execute_reply":"2024-06-04T09:52:28.53159Z"}}},{"cell_type":"code","source":"# kappa_callback = KappaScoreCallback(validation_data=valid_gen)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:08.229044Z","iopub.execute_input":"2024-06-06T07:33:08.2293Z","iopub.status.idle":"2024-06-06T07:33:08.23529Z","shell.execute_reply.started":"2024-06-06T07:33:08.229277Z","shell.execute_reply":"2024-06-06T07:33:08.234353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = vgg_conv.fit(\n    train_gen,\n    steps_per_epoch=nb_train_steps,\n    epochs=nb_epochs,\n    validation_data=valid_gen,\n    validation_steps=nb_val_steps)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T07:33:08.236268Z","iopub.execute_input":"2024-06-06T07:33:08.236532Z","iopub.status.idle":"2024-06-06T08:53:07.074918Z","shell.execute_reply.started":"2024-06-06T07:33:08.23651Z","shell.execute_reply":"2024-06-06T08:53:07.074138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"history = inceptionv3_conv.fit(\n    train_gen,\n    steps_per_epoch=nb_train_steps,\n    epochs=nb_epochs,\n    validation_data=valid_gen,\n    validation_steps=nb_val_steps)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T09:53:28.776832Z","iopub.execute_input":"2024-06-04T09:53:28.77768Z","iopub.status.idle":"2024-06-04T10:47:52.20591Z","shell.execute_reply.started":"2024-06-04T09:53:28.777647Z","shell.execute_reply":"2024-06-04T10:47:52.205055Z"}}},{"cell_type":"code","source":"plt.figure(figsize=(14, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Train Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Model Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend(loc='upper left')\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Model Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend(loc='upper left')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T08:53:07.076076Z","iopub.execute_input":"2024-06-06T08:53:07.076357Z","iopub.status.idle":"2024-06-06T08:53:07.943179Z","shell.execute_reply.started":"2024-06-06T08:53:07.076332Z","shell.execute_reply":"2024-06-06T08:53:07.942276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import classification_report, confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2024-06-06T08:53:07.944508Z","iopub.execute_input":"2024-06-06T08:53:07.944849Z","iopub.status.idle":"2024-06-06T08:53:07.949267Z","shell.execute_reply.started":"2024-06-06T08:53:07.944818Z","shell.execute_reply":"2024-06-06T08:53:07.948358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"test_gen.reset()\nY_pred = inceptionv3_conv.predict(test_gen, steps=len(test_gen), verbose=1)\ny_pred = np.argmax(Y_pred, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-06-04T10:55:45.78656Z","iopub.execute_input":"2024-06-04T10:55:45.786875Z","iopub.status.idle":"2024-06-04T10:55:51.000409Z","shell.execute_reply.started":"2024-06-04T10:55:45.78685Z","shell.execute_reply":"2024-06-04T10:55:50.99953Z"}}},{"cell_type":"code","source":"test_gen.reset()\nY_pred = vgg_conv.predict(test_gen, steps=len(test_gen), verbose=1)\ny_pred = np.argmax(Y_pred, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T08:53:07.950514Z","iopub.execute_input":"2024-06-06T08:53:07.950836Z","iopub.status.idle":"2024-06-06T08:53:21.955363Z","shell.execute_reply.started":"2024-06-06T08:53:07.950805Z","shell.execute_reply":"2024-06-06T08:53:21.954635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_true = test_gen.classes","metadata":{"execution":{"iopub.status.busy":"2024-06-06T08:53:21.956652Z","iopub.execute_input":"2024-06-06T08:53:21.957011Z","iopub.status.idle":"2024-06-06T08:53:21.961161Z","shell.execute_reply.started":"2024-06-06T08:53:21.956979Z","shell.execute_reply":"2024-06-06T08:53:21.960258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_labels = list(test_gen.class_indices.keys())","metadata":{"execution":{"iopub.status.busy":"2024-06-06T08:53:21.96248Z","iopub.execute_input":"2024-06-06T08:53:21.962992Z","iopub.status.idle":"2024-06-06T08:53:21.974403Z","shell.execute_reply.started":"2024-06-06T08:53:21.96296Z","shell.execute_reply":"2024-06-06T08:53:21.973717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conf_matrix = confusion_matrix(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T08:53:21.975436Z","iopub.execute_input":"2024-06-06T08:53:21.97573Z","iopub.status.idle":"2024-06-06T08:53:21.988157Z","shell.execute_reply.started":"2024-06-06T08:53:21.975675Z","shell.execute_reply":"2024-06-06T08:53:21.987437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conf_matrix","metadata":{"execution":{"iopub.status.busy":"2024-06-06T08:53:21.989165Z","iopub.execute_input":"2024-06-06T08:53:21.989535Z","iopub.status.idle":"2024-06-06T08:53:21.99944Z","shell.execute_reply.started":"2024-06-06T08:53:21.98951Z","shell.execute_reply":"2024-06-06T08:53:21.998627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues', xticklabels=class_labels, yticklabels=class_labels)\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T08:53:22.000673Z","iopub.execute_input":"2024-06-06T08:53:22.000919Z","iopub.status.idle":"2024-06-06T08:53:22.465293Z","shell.execute_reply.started":"2024-06-06T08:53:22.000898Z","shell.execute_reply":"2024-06-06T08:53:22.464381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_report = classification_report(y_true, y_pred, target_names=class_labels)\nprint('Classification Report')\nprint(class_report)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T08:53:22.474818Z","iopub.execute_input":"2024-06-06T08:53:22.475104Z","iopub.status.idle":"2024-06-06T08:53:22.489227Z","shell.execute_reply.started":"2024-06-06T08:53:22.47508Z","shell.execute_reply":"2024-06-06T08:53:22.488226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import MobileNet\n\ndef mobilenet_model(num_classes=None):\n    base_model = MobileNet(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n    x = GlobalAveragePooling2D()(base_model.output)\n    output = Dense(num_classes, activation='softmax')(x)\n    model = Model(base_model.input, output)\n    return model\n\n = mobilenet_model(6)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T08:58:17.74211Z","iopub.execute_input":"2024-06-06T08:58:17.742489Z","iopub.status.idle":"2024-06-06T08:58:19.919183Z","shell.execute_reply.started":"2024-06-06T08:58:17.742458Z","shell.execute_reply":"2024-06-06T08:58:19.918272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opt = Adam(learning_rate=0.001)\nmobilenet_model.compile(loss='categorical_crossentropy',optimizer=opt,metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-06-06T09:00:57.958744Z","iopub.execute_input":"2024-06-06T09:00:57.959416Z","iopub.status.idle":"2024-06-06T09:00:57.968755Z","shell.execute_reply.started":"2024-06-06T09:00:57.959372Z","shell.execute_reply":"2024-06-06T09:00:57.967923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gen.reset()\nY_pred = mobilenet_model.predict(test_gen, steps=len(test_gen), verbose=1)\ny_pred = np.argmax(Y_pred, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T10:25:43.72587Z","iopub.execute_input":"2024-06-06T10:25:43.726338Z","iopub.status.idle":"2024-06-06T10:25:55.193186Z","shell.execute_reply.started":"2024-06-06T10:25:43.726301Z","shell.execute_reply":"2024-06-06T10:25:55.192382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(14, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Train Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Model Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend(loc='upper left')\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Model Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend(loc='upper left')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T10:26:20.360143Z","iopub.execute_input":"2024-06-06T10:26:20.360609Z","iopub.status.idle":"2024-06-06T10:26:21.091337Z","shell.execute_reply.started":"2024-06-06T10:26:20.360577Z","shell.execute_reply":"2024-06-06T10:26:21.090424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = mobilenet_model.fit(\n    train_gen,\n    steps_per_epoch=nb_train_steps,\n    epochs=nb_epochs,\n    validation_data=valid_gen,\n    validation_steps=nb_val_steps)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T09:00:59.590798Z","iopub.execute_input":"2024-06-06T09:00:59.59115Z","iopub.status.idle":"2024-06-06T10:19:07.534674Z","shell.execute_reply.started":"2024-06-06T09:00:59.591123Z","shell.execute_reply":"2024-06-06T10:19:07.533713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conf_matrix = confusion_matrix(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T10:26:44.144117Z","iopub.execute_input":"2024-06-06T10:26:44.145139Z","iopub.status.idle":"2024-06-06T10:26:44.151692Z","shell.execute_reply.started":"2024-06-06T10:26:44.145091Z","shell.execute_reply":"2024-06-06T10:26:44.150841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conf_matrix ","metadata":{"execution":{"iopub.status.busy":"2024-06-06T10:26:50.259518Z","iopub.execute_input":"2024-06-06T10:26:50.260195Z","iopub.status.idle":"2024-06-06T10:26:50.266194Z","shell.execute_reply.started":"2024-06-06T10:26:50.260164Z","shell.execute_reply":"2024-06-06T10:26:50.265228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues', xticklabels=class_labels, yticklabels=class_labels)\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T10:27:09.497289Z","iopub.execute_input":"2024-06-06T10:27:09.497658Z","iopub.status.idle":"2024-06-06T10:27:09.952643Z","shell.execute_reply.started":"2024-06-06T10:27:09.497629Z","shell.execute_reply":"2024-06-06T10:27:09.951628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_report = classification_report(y_true, y_pred, target_names=class_labels)\nprint('Classification Report')\nprint(class_report)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T10:27:34.634559Z","iopub.execute_input":"2024-06-06T10:27:34.635356Z","iopub.status.idle":"2024-06-06T10:27:34.650842Z","shell.execute_reply.started":"2024-06-06T10:27:34.635325Z","shell.execute_reply":"2024-06-06T10:27:34.649819Z"},"trusted":true},"execution_count":null,"outputs":[]}]}