{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":18647,"databundleVersionId":1126921,"sourceType":"competition"}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_gleason_score(score):\n    try:\n        parts = score.split('+')\n        return int(parts[0]) + int(parts[1])\n    except (ValueError, AttributeError):\n        return np.nan\n\ntrain_df['gleason_score_numeric'] = train_df['gleason_score'].apply(convert_gleason_score)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nplt.subplot(1, 2, 1)\nsns.countplot(x='isup_grade', data=train_df)\nplt.title('Distribution of ISUP Grade')\n\nfor p in plt.gca().patches:\n    plt.gca().annotate(f\"{p.get_height()}\", (p.get_x() + p.get_width() / 2., p.get_height()), ha='center', va='center', fontsize=10, color='black', xytext=(0, 5), textcoords='offset points')\n\nplt.subplot(1, 2, 2)\nsns.countplot(x='gleason_score_numeric', data=train_df)\nplt.title('Distribution of Gleason Score (Numeric)')\n\nfor p in plt.gca().patches:\n    plt.gca().annotate(f\"{p.get_height()}\", (p.get_x() + p.get_width() / 2., p.get_height()), ha='center', va='center', fontsize=10, color='black', xytext=(0, 5), textcoords='offset points')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_dir = '/kaggle/input/prostate-cancer-grade-assessment/train_images/'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_size = 256","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_ids = train_df['image_id'].iloc[:9].tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import openslide","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = train_df['image_id'].iloc[0]\nfull_image_path = image_dir + image_id + '.tiff'\n\ntry:\n\n    example = openslide.OpenSlide(full_image_path)\n\n    clipped_example = example.read_region((5000, 5000), 0, (image_size, image_size))\n\n    plt.imshow(clipped_example)\n    plt.title(f\"Image ID: {image_id}\")\n    plt.axis('off')\n\n    example.close()\nexcept Exception as e:\n    print(f\"Error loading image {image_id}: {e}\")\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"image_path\"] = [image_dir+image_id+\".tiff\" for image_id in train_df[\"image_id\"]]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='isup_grade')\nplt.title('Distribution of ISUP grades')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nplt.pie(train_df['isup_grade'].value_counts(), labels=train_df['isup_grade'].unique(), autopct='%1.1f%%', startangle=140)\nplt.title('Distribution of ISUP grades')\nplt.axis('equal')  \nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='data_provider')\nplt.title('Distribution of Data Providers')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nplt.pie(train_df['data_provider'].value_counts(), labels=train_df['data_provider'].unique(), autopct='%1.1f%%', startangle=140)\nplt.title('Distribution of Data Providers')\nplt.axis('equal')  \nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='gleason_score_numeric')\nplt.title('Distribution of Gleason Scores')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_matrix = train_df[['isup_grade', 'gleason_score_numeric']].corr()\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm')\nplt.title('Correlation Matrix')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from random import randint\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.layers import Conv2D, BatchNormalization, Activation, MaxPooling2D, GlobalAveragePooling2D, Dense, Dropout\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = shuffle(train_df)\ntraining_item_count = int(len(train_df) * 0.8)\nvalidation_df = train_df[training_item_count:]\ntrain_df = train_df[:training_item_count]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_single_sample(image_path, image_size=256, training=False, display=False):\n    image = openslide.OpenSlide(image_path)\n    mask_path = image_path.replace(\"train_images\", \"train_label_masks\").replace(\".tiff\", \"_mask.tiff\")\n    mask = openslide.OpenSlide(mask_path)\n    \n    stacked_image = []\n    groundtruth_per_image = []\n    \n    maximum_iteration = 0\n    selected_sample = False\n    while not selected_sample:\n        sampling_start_x = randint(image_size, image.dimensions[0] - image_size)\n        sampling_start_y = randint(image_size, image.dimensions[1] - image_size)\n\n        clipped_sample = image.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n        clipped_array = np.asarray(clipped_sample)\n        \n        if (not np.all(clipped_array == 255) and np.std(clipped_array) > 20) or maximum_iteration > 200:\n            if display:\n                plt.imshow(clipped_sample)\n                plt.show()\n                \n            sampled_image = clipped_array[:, :, :3]\n            \n            if training:\n                clipped_mask = mask.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n                groundtruth_per_image.append(np.mean(np.asarray(clipped_mask)[:, :, 0]))\n            \n            selected_sample = True\n        maximum_iteration += 1\n    \n    if training: \n        return np.array(sampled_image), np.array(groundtruth_per_image)\n    else:\n        return np.array(sampled_image)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_random_samples(image_path, image_size=256, display=False):\n    image = openslide.OpenSlide(image_path)\n    stacked_image = []\n    \n    selected_samples = 0\n    maximum_iteration = 0\n    while selected_samples < 3:\n        sampling_start_x = randint(image_size, image.dimensions[0] - image_size)\n        sampling_start_y = randint(image_size, image.dimensions[1] - image_size)\n\n        clipped_sample = image.read_region((sampling_start_x, sampling_start_y), 0, (256, 256))\n        clipped_array = np.asarray(clipped_sample)\n        \n        if (not np.all(clipped_array == 255) and np.std(clipped_array) > 20) or maximum_iteration > 200:\n            if display:\n                plt.imshow(clipped_sample)\n                plt.show()\n\n            stacked_image.append(clipped_array[:, :, :3])\n            selected_samples += 1\n        maximum_iteration += 1\n    return np.array(stacked_image)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_single_image_generator(image_path_list, batch_size=16):\n    while True:\n        for start in range(0, len(image_path_list), batch_size):\n            X_batch = []\n            Y_batch = []\n            end = min(start + batch_size, training_item_count)\n\n            image_info_list = [get_single_sample(image_path, training=True) for image_path in image_path_list[start:end]]\n            X_batch = np.array([image_info[0]/255. for image_info in image_info_list])\n            Y_batch = np.array([image_info[1] for image_info in image_info_list])\n            \n            yield X_batch, Y_batch ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_samples = get_random_samples(train_df.iloc[0].image_path, display=True)\nprint(\"Random samples shape:\", random_samples.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_path = train_df.iloc[0].image_path  \nsample_image, groundtruth = get_single_sample(image_path, training=True, display=True)\n\nprint(\"Sample Image Shape:\", sample_image.shape)\nprint(\"Ground Truth:\", groundtruth)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def branch(input_image):\n    x = Conv2D(64, (3, 3), padding='same')(input_image)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Conv2D(64, (3, 3), padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = MaxPooling2D(pool_size=(2, 2))(x)\n    \n    x = Conv2D(128, (3, 3), padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Conv2D(128, (3, 3), padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = MaxPooling2D(pool_size=(2, 2))(x)\n    \n    x = Conv2D(256, (3, 3), padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = Conv2D(256, (3, 3), padding='same')(x)\n    x = BatchNormalization()(x)\n    x = Activation('relu')(x)\n    x = GlobalAveragePooling2D()(x)\n    \n    x = Dense(256)(x)\n    x = Activation('relu')(x)\n    \n    return Dropout(0.3)(x)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_image = layers.Input(shape=(256, 256, 3))\noutput = branch(input_image)\nmodel = Model(inputs=input_image, outputs=output)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.layers import Input","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_image = Input(shape=(256, 256, 3))\n\ncore_branch = branch(input_image)\n\noutput = Dense(1, activation='linear')(core_branch)\n\nbranch_model = Model(inputs=input_image, outputs=output)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"branch_model.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.optimizers import SGD","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = SGD(learning_rate=0.01, momentum=0.9)\nbranch_model.compile(optimizer=optimizer, loss='mean_squared_error')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_paths = train_df[\"image_path\"].tolist()[:batch_size]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_image_paths = validation_df[\"image_path\"].tolist()\nvalidation_steps = len(validation_image_paths) // batch_size","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"steps_per_epoch = len(train_df) // batch_size\nepochs = 3","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_single_image_generator1(image_path_list, batch_size=16):\n    while True:\n        for start in range(0, len(image_path_list), batch_size):\n            X_batch = []\n            Y_batch = []\n            end = min(start + batch_size, len(image_path_list))\n\n            image_info_list = []\n            for image_path in image_path_list[start:end]:\n                try:\n                    image_info_list.append(get_single_sample(image_path, training=True))\n                except openslide.OpenSlideUnsupportedFormatError as e:\n                    print(f\"Ignoring unsupported image file: {image_path}\")\n                    continue\n                    \n            if not image_info_list:\n                continue\n\n            X_batch = np.array([image_info[0]/255. for image_info in image_info_list])\n            Y_batch = np.array([image_info[1] for image_info in image_info_list])\n            \n            yield X_batch, Y_batch","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = branch_model.fit(\n    custom_single_image_generator1(image_paths, batch_size=batch_size),\n    steps_per_epoch=steps_per_epoch,\n    epochs=epochs,\n    validation_data=custom_single_image_generator1(validation_image_paths, batch_size=batch_size),\n    validation_steps=validation_steps\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_model(model, data_generator, steps):\n    return model.evaluate(data_generator, steps=steps)\n\ndef plot_history(history):\n   \n    plt.plot(history.history['loss'], label='Training Loss')\n    plt.plot(history.history['val_loss'], label='Validation Loss')\n    plt.title('Model Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_generator = custom_single_image_generator1(validation_image_paths, batch_size=batch_size)\nvalidation_steps = len(validation_image_paths) // batch_size\nloss = evaluate_model(branch_model, validation_generator, validation_steps)\nprint(\"Validation Loss:\", loss)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_history(history)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming your model has been trained and you have a history object\n# containing training history\n\n# Generate predictions for the training set\ntrain_image_paths = train_df[\"image_path\"].tolist()\ntrain_predictions = generate_predictions(branch_model, train_image_paths)\n\n# Calculate training accuracy\ntrain_true_scores = train_df['gleason_score_numeric']\ntrain_accuracy = calculate_accuracy(train_true_scores, train_predictions)\nprint(f\"Training Accuracy: {train_accuracy:.2f}%\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you've already imported the necessary libraries and functions\n\n# Load the test data\ntest_df = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/test.csv')\n\n# Convert Gleason score to numeric\ntest_df['gleason_score_numeric'] = test_df['gleason_score'].apply(convert_gleason_score)\n\n# Add image paths\ntest_df[\"image_path\"] = [image_dir + image_id + \".tiff\" for image_id in test_df[\"image_id\"]]\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate predictions for the test data\ndef generate_predictions(model, image_paths):\n    predictions = []\n    for image_path in image_paths:\n        try:\n            sample_image = get_single_sample(image_path)\n            sample_image = sample_image[np.newaxis, ...] / 255.0  # Add batch dimension and normalize\n            prediction = model.predict(sample_image)\n            predictions.append(prediction[0][0])\n        except Exception as e:\n            print(f\"Error processing image {image_path}: {e}\")\n            predictions.append(np.nan)\n    return predictions\n\n# Generate predictions for the test set\ntest_image_paths = test_df[\"image_path\"].tolist()\npredictions = generate_predictions(branch_model, test_image_paths)\n\n# Add predictions to the test dataframe\ntest_df[\"predicted_gleason_score\"] = predictions\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate accuracy\ndef calculate_accuracy(true_scores, predicted_scores):\n    correct_predictions = 0\n    for true_score, pred_score in zip(true_scores, predicted_scores):\n        if np.isnan(true_score) or np.isnan(pred_score):\n            continue\n        if true_score == round(pred_score):\n            correct_predictions += 1\n    accuracy = correct_predictions / len(true_scores) * 100\n    return accuracy\n\ntrue_scores = test_df['gleason_score_numeric']\npredicted_scores = test_df['predicted_gleason_score']\n\naccuracy = calculate_accuracy(true_scores, predicted_scores)\nprint(f\"Accuracy on test set: {accuracy:.2f}%\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}