{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":37333,"databundleVersionId":3949526,"sourceType":"competition"},{"sourceId":3923586,"sourceType":"datasetVersion","datasetId":2329240},{"sourceId":4142384,"sourceType":"datasetVersion","datasetId":2446615}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Import library\nimport openslide\nfrom openslide import OpenSlide  \nimport pandas as pd\nimport tifffile as tiff\nimport matplotlib.pyplot as plt \nimport numpy as np\nimport tensorflow as tf\nimport os\nfrom PIL import Image\n\nfrom sklearn.model_selection import train_test_split\n\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import layers\nfrom tensorflow.python.keras.layers import Dense, Flatten\nfrom tensorflow.keras.optimizers import Adam\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:33:45.581028Z","iopub.execute_input":"2024-11-14T02:33:45.581774Z","iopub.status.idle":"2024-11-14T02:33:58.886201Z","shell.execute_reply.started":"2024-11-14T02:33:45.581732Z","shell.execute_reply":"2024-11-14T02:33:58.88543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Reading data\ninput_path = \"../input/mayo-clinic-strip-ai/\"\ntrain_df = pd.read_csv(input_path+\"train.csv\")\ntest_df = pd.read_csv(input_path+\"test.csv\")\nother_df = pd.read_csv(input_path+\"other.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:33:58.888335Z","iopub.execute_input":"2024-11-14T02:33:58.889078Z","iopub.status.idle":"2024-11-14T02:33:58.918818Z","shell.execute_reply.started":"2024-11-14T02:33:58.889029Z","shell.execute_reply":"2024-11-14T02:33:58.917976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:33:58.920225Z","iopub.execute_input":"2024-11-14T02:33:58.920542Z","iopub.status.idle":"2024-11-14T02:33:58.938779Z","shell.execute_reply.started":"2024-11-14T02:33:58.92051Z","shell.execute_reply":"2024-11-14T02:33:58.937781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_path= '/kaggle/input/stroke-blood-clot-origin-1k-scale-bg-crop/'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:33:58.939879Z","iopub.execute_input":"2024-11-14T02:33:58.94016Z","iopub.status.idle":"2024-11-14T02:33:58.94418Z","shell.execute_reply.started":"2024-11-14T02:33:58.940129Z","shell.execute_reply":"2024-11-14T02:33:58.943154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.shape)\nprint(train_df['image_id'].nunique())\nprint('Unique Values for: ')\nprint(train_df.nunique())\n\nprint('\\n')\n\nnu = train_df.nunique().reset_index()\nnu.columns = ['feature','nunique']\nplt.figure(figsize=(12,4))\nax = sns.barplot(x='feature', y='nunique', data=nu)\nax.bar_label(ax.containers[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:33:58.946221Z","iopub.execute_input":"2024-11-14T02:33:58.9465Z","iopub.status.idle":"2024-11-14T02:33:59.276511Z","shell.execute_reply.started":"2024-11-14T02:33:58.946469Z","shell.execute_reply":"2024-11-14T02:33:59.275638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[\"image_path\"] = train_df[\"image_id\"].apply(lambda x: file_path +\"train_images/\" + x + \".png\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:33:59.277886Z","iopub.execute_input":"2024-11-14T02:33:59.278258Z","iopub.status.idle":"2024-11-14T02:33:59.284436Z","shell.execute_reply.started":"2024-11-14T02:33:59.278213Z","shell.execute_reply":"2024-11-14T02:33:59.283548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:33:59.28575Z","iopub.execute_input":"2024-11-14T02:33:59.286113Z","iopub.status.idle":"2024-11-14T02:33:59.300618Z","shell.execute_reply.started":"2024-11-14T02:33:59.286071Z","shell.execute_reply":"2024-11-14T02:33:59.299682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    # Include both x and y\n    train, val = train_test_split(\n        train_df,\n        test_size=0.2,       # 20% for validation; adjust as needed\n        stratify=train_df['label'],  # Use this to maintain class distribution if 'label' exists\n        random_state=42      # For reproducibility\n    )\n    \n    \n    print(f\"train type: {type(train)}\")  \n    print(f\"val type: {type(val)}\")      \n    print(train.head())\n    IMG_SIZE= 256\n    \n    image_shape = (IMG_SIZE, IMG_SIZE)\n    \n    train_datagen=ImageDataGenerator(rescale=1./255,\n                               zoom_range=0.2,\n                               rotation_range=20,\n                               )\n    test_datagen = ImageDataGenerator(rescale=1./255)\n    \n    train_gen = train_datagen.flow_from_dataframe(\n                             train,\n                             #directory='image_path'\n                             x_col = 'image_path',\n                             y_col = 'label',\n                             target_size=image_shape,\n                             class_mode = 'sparse',\n                             color_mode = 'rgb',\n                             shuffle=True,\n                             batch_size=16,\n                             seed=19,\n                             )\n    val_gen = test_datagen.flow_from_dataframe(\n                             val,\n                             #directory='image_path',\n                             x_col = 'image_path',\n                             y_col = 'label',\n                             target_size=image_shape,\n                             class_mode = 'sparse',\n                             color_mode = 'rgb',\n                             shuffle=True,\n                             batch_size=16,\n                             seed=19,\n                             )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:33:59.301755Z","iopub.execute_input":"2024-11-14T02:33:59.30205Z","iopub.status.idle":"2024-11-14T02:34:03.040913Z","shell.execute_reply.started":"2024-11-14T02:33:59.302019Z","shell.execute_reply":"2024-11-14T02:34:03.040025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras import models\n\nfrom tensorflow.keras import layers, Model\n\nimport tensorflow as tf\nfrom tensorflow.keras.callbacks import ModelCheckpoint","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:34:16.228974Z","iopub.execute_input":"2024-11-14T02:34:16.229621Z","iopub.status.idle":"2024-11-14T02:34:16.234298Z","shell.execute_reply.started":"2024-11-14T02:34:16.229551Z","shell.execute_reply":"2024-11-14T02:34:16.233348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    checkpoint = ModelCheckpoint(\n    \n        filepath='inception_model.keras',  # Path to save the best model\n    \n        monitor='val_accuracy',   # Metric to monitor (e.g., validation accuracy)\n    \n        save_best_only=True,      # Save only the best model\n    \n        mode='max',                # 'max' for accuracy, 'min' for loss\n    \n        verbose=1                 # Print messages when a checkpoint is saved\n    \n    )\n    IMG_SIZE = 96  # Image input size\n    \n    \n    \n    # Load InceptionV3 as the base model\n    \n    inception_model = tf.keras.applications.InceptionV3(\n    \n        include_top=False,            # Exclude the top layers (fully connected)\n    \n        input_shape=(IMG_SIZE, IMG_SIZE, 3),\n    \n        pooling='avg',                # Global average pooling for a 1D output\n    \n        weights='imagenet'            # Load pretrained weights\n    \n    )\n    \n    \n    \n    # Freeze most of the base model layers, but unfreeze the last few\n    \n    for layer in inception_model.layers[:-2]:  # Unfreeze the last 2 layers\n    \n        layer.trainable = False\n    \n    # Build the model using the Functional API\n    \n    inputs = inception_model.input\n    \n    x = inception_model.output\n    \n    \n    \n    \n    \n    x = layers.Dense(256)(x)\n    \n    x = layers.BatchNormalization()(x)\n    \n    x = layers.Activation('relu')(x)\n    \n    x = layers.Dropout(0.5)(x)\n    \n    \n    \n    # Final output layer for binary classification\n    \n    outputs = layers.Dense(2, activation='softmax')(x)\n    \n    \n    \n    # Create the final model\n    \n    inception_model = Model(inputs=inputs, outputs=outputs)\n    \n    \n    \n    # Print the summary\n    \n    inception_model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:34:18.641619Z","iopub.execute_input":"2024-11-14T02:34:18.642013Z","iopub.status.idle":"2024-11-14T02:34:25.07819Z","shell.execute_reply.started":"2024-11-14T02:34:18.641975Z","shell.execute_reply":"2024-11-14T02:34:25.077189Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    inception_model.compile(optimizer=Adam(learning_rate=0.001),loss='sparse_categorical_crossentropy',metrics=['accuracy']) # Assumes from_logits=False by default\n    \n    reduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.2, verbose=1,mode='min',patience=3, min_lr=1E-5)\n\n    history = inception_model.fit(train_gen, epochs=20, shuffle=True, validation_data=val_gen, callbacks=[reduce_lr,checkpoint],verbose = 1)\n    \n    inception_model.save('inception_model.h5')\n    \n    print(\"Model training complete!\")\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:34:28.265134Z","iopub.execute_input":"2024-11-14T02:34:28.265547Z","iopub.status.idle":"2024-11-14T02:47:59.362834Z","shell.execute_reply.started":"2024-11-14T02:34:28.265506Z","shell.execute_reply":"2024-11-14T02:47:59.361909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_path ='/kaggle/input/mayoclinictest-imagesresized1024/'\ntest_df = pd.read_csv(input_path+\"test.csv\")\ntest_df[\"image_path\"] = test_df[\"image_id\"].apply(lambda x: test_path + \"test/\"+ x + \".png\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:48:25.732774Z","iopub.execute_input":"2024-11-14T02:48:25.733154Z","iopub.status.idle":"2024-11-14T02:48:25.742625Z","shell.execute_reply.started":"2024-11-14T02:48:25.733119Z","shell.execute_reply":"2024-11-14T02:48:25.741836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:48:27.307374Z","iopub.execute_input":"2024-11-14T02:48:27.308135Z","iopub.status.idle":"2024-11-14T02:48:27.318437Z","shell.execute_reply.started":"2024-11-14T02:48:27.308094Z","shell.execute_reply":"2024-11-14T02:48:27.317478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test_df.shape)\nprint(test_df['image_id'].nunique())\nprint('Unique Values for: ')\nprint(test_df.nunique())\n\nprint('\\n')\n\nnu = test_df.nunique().reset_index()\nnu.columns = ['feature','nunique']\nplt.figure(figsize=(12,4))\nax = sns.barplot(x='feature', y='nunique', data=nu)\nax.bar_label(ax.containers[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:48:31.307988Z","iopub.execute_input":"2024-11-14T02:48:31.3088Z","iopub.status.idle":"2024-11-14T02:48:31.617174Z","shell.execute_reply.started":"2024-11-14T02:48:31.308757Z","shell.execute_reply":"2024-11-14T02:48:31.616188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from glob import glob","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:48:36.294482Z","iopub.execute_input":"2024-11-14T02:48:36.295789Z","iopub.status.idle":"2024-11-14T02:48:36.300054Z","shell.execute_reply.started":"2024-11-14T02:48:36.295743Z","shell.execute_reply":"2024-11-14T02:48:36.298932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_images_mtdt = glob(test_path + \"test/\" + \"*\")\nprint(f\"Number of images in a testing set: {len(test_images_mtdt)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:49:50.559927Z","iopub.execute_input":"2024-11-14T02:49:50.560788Z","iopub.status.idle":"2024-11-14T02:49:50.570518Z","shell.execute_reply.started":"2024-11-14T02:49:50.560744Z","shell.execute_reply":"2024-11-14T02:49:50.569608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_images_mtdt[1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:55:51.233667Z","iopub.execute_input":"2024-11-14T02:55:51.234067Z","iopub.status.idle":"2024-11-14T02:55:51.24036Z","shell.execute_reply.started":"2024-11-14T02:55:51.23403Z","shell.execute_reply":"2024-11-14T02:55:51.239395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import defaultdict\nimport pandas as pd\nfrom PIL import Image\n\ndef get_images_info(images_mtdt, label_df):\n    img_prop = defaultdict(list)\n    for i, path in enumerate(images_mtdt):\n        img_path = images_mtdt[i]\n        image_id = img_path[-12:-4]\n        \n        # Open PNG file with PIL instead of OpenSlide\n        with Image.open(img_path) as img:\n            big_dim = 'none'\n            max_min_dim_ratio = 1.0\n            \n            # Get dimensions using PIL's size attribute\n            img_width, img_height = img.size\n            \n            if(img_width > img_height):\n                big_dim = 'width'\n                max_min_dim_ratio = round(img_width/img_height, 2)\n            elif(img_width < img_height):\n                big_dim = 'height'\n                max_min_dim_ratio = round(img_height/img_width, 2)\n                \n            img_prop['image_id'].append(image_id)\n            img_prop['width'].append(img_width)\n            img_prop['height'].append(img_height)\n            img_prop['big_dim'].append(big_dim)\n            img_prop['max_min_dim_ratio'].append(max_min_dim_ratio)\n            \n            split_size = round(max_min_dim_ratio)\n            img_prop['split_size'].append(split_size)\n            img_prop['path'].append(img_path)\n    \n    img_info = pd.DataFrame(img_prop)\n    img_info.sort_values(by='image_id', inplace=True)\n    img_info.reset_index(inplace=True, drop=True)\n    img_info = img_info.merge(label_df, on='image_id')\n    \n    return img_info","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:57:58.85003Z","iopub.execute_input":"2024-11-14T02:57:58.850405Z","iopub.status.idle":"2024-11-14T02:57:58.860765Z","shell.execute_reply.started":"2024-11-14T02:57:58.85037Z","shell.execute_reply":"2024-11-14T02:57:58.859777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE= 256","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:59:16.124343Z","iopub.execute_input":"2024-11-14T02:59:16.125142Z","iopub.status.idle":"2024-11-14T02:59:16.129225Z","shell.execute_reply.started":"2024-11-14T02:59:16.125097Z","shell.execute_reply":"2024-11-14T02:59:16.128296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport sys\nimport pandas as pd\nimport numpy as np\nimport math\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\n\nimport warnings\nimport random\nimport tensorflow as tf\nprint(\"TensorFlow version:\", tf.__version__)\n\nfrom tqdm import tqdm\nfrom PIL import Image\nfrom random import randrange\n\nfrom pathlib import Path\nfrom glob import glob\n\nfrom skimage.exposure import is_low_contrast\nfrom scipy.ndimage import zoom, rotate\nfrom skimage.io import imread, imsave\n\nfrom collections import defaultdict\nfrom openslide import OpenSlide","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:53:53.860596Z","iopub.execute_input":"2024-11-14T02:53:53.861305Z","iopub.status.idle":"2024-11-14T02:53:53.980859Z","shell.execute_reply.started":"2024-11-14T02:53:53.861258Z","shell.execute_reply":"2024-11-14T02:53:53.97996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_images_info = get_images_info(test_images_mtdt, test_df)\ntest_images_count = test_images_info['image_id'].nunique()\ntest_images = np.zeros((test_images_count, IMG_SIZE, IMG_SIZE, 3), dtype=np.uint8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:59:23.422366Z","iopub.execute_input":"2024-11-14T02:59:23.42276Z","iopub.status.idle":"2024-11-14T02:59:23.437462Z","shell.execute_reply.started":"2024-11-14T02:59:23.422723Z","shell.execute_reply":"2024-11-14T02:59:23.4367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_images = test_images/255.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:59:26.489671Z","iopub.execute_input":"2024-11-14T02:59:26.490063Z","iopub.status.idle":"2024-11-14T02:59:26.494779Z","shell.execute_reply.started":"2024-11-14T02:59:26.490027Z","shell.execute_reply":"2024-11-14T02:59:26.493903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_images_tensor = tf.convert_to_tensor(test_images)\nresult = inception_model.predict(test_images_tensor)\nresult = tf.nn.softmax(result)\nresult = np.round(result, 6)\nresult","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T02:59:28.314308Z","iopub.execute_input":"2024-11-14T02:59:28.315153Z","iopub.status.idle":"2024-11-14T02:59:39.988457Z","shell.execute_reply.started":"2024-11-14T02:59:28.315098Z","shell.execute_reply":"2024-11-14T02:59:39.987526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T03:03:08.574338Z","iopub.execute_input":"2024-11-14T03:03:08.574753Z","iopub.status.idle":"2024-11-14T03:03:08.585074Z","shell.execute_reply.started":"2024-11-14T03:03:08.574714Z","shell.execute_reply":"2024-11-14T03:03:08.584161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = pd.DataFrame(result, columns=['CE', 'LAA'])\nsubmission_df.insert(0, 'patient_id', test_df['patient_id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T03:03:21.901025Z","iopub.execute_input":"2024-11-14T03:03:21.901415Z","iopub.status.idle":"2024-11-14T03:03:21.907106Z","shell.execute_reply.started":"2024-11-14T03:03:21.901376Z","shell.execute_reply":"2024-11-14T03:03:21.906131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def maxprediction(data_frame):\n    data_frame = data_frame.assign(prob_diff = abs(data_frame['CE'] - data_frame['LAA']))\n    max_diff = data_frame.loc[data_frame['prob_diff'].idxmax()]\n    \n    return  max_diff[['CE', 'LAA']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T03:03:24.72932Z","iopub.execute_input":"2024-11-14T03:03:24.729997Z","iopub.status.idle":"2024-11-14T03:03:24.735127Z","shell.execute_reply.started":"2024-11-14T03:03:24.729958Z","shell.execute_reply":"2024-11-14T03:03:24.73402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = submission_df[['patient_id','CE', 'LAA']].groupby(['patient_id']).apply(maxprediction).reset_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T03:03:26.119905Z","iopub.execute_input":"2024-11-14T03:03:26.120764Z","iopub.status.idle":"2024-11-14T03:03:26.137488Z","shell.execute_reply.started":"2024-11-14T03:03:26.120723Z","shell.execute_reply":"2024-11-14T03:03:26.136407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index = False)\n!head submission.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-14T03:03:28.90752Z","iopub.execute_input":"2024-11-14T03:03:28.90822Z","iopub.status.idle":"2024-11-14T03:03:30.010912Z","shell.execute_reply.started":"2024-11-14T03:03:28.908182Z","shell.execute_reply":"2024-11-14T03:03:30.00976Z"}},"outputs":[],"execution_count":null}]}