{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-21T05:05:33.869895Z","iopub.execute_input":"2022-09-21T05:05:33.870384Z","iopub.status.idle":"2022-09-21T05:05:33.876241Z","shell.execute_reply.started":"2022-09-21T05:05:33.870347Z","shell.execute_reply":"2022-09-21T05:05:33.875256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from numpy.random import seed\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten, Activation\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\nfrom tensorflow.keras.optimizers import Adam\nimport os\nimport cv2\nfrom sklearn.utils import shuffle\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import train_test_split\nimport itertools\nimport shutil\nimport matplotlib.pyplot as plt\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:32:17.088689Z","iopub.execute_input":"2022-09-21T06:32:17.089384Z","iopub.status.idle":"2022-09-21T06:32:23.101876Z","shell.execute_reply.started":"2022-09-21T06:32:17.089346Z","shell.execute_reply":"2022-09-21T06:32:23.100934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = 96\nIMAGE_CHANNELS = 3","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:32:23.301417Z","iopub.execute_input":"2022-09-21T06:32:23.302202Z","iopub.status.idle":"2022-09-21T06:32:23.306807Z","shell.execute_reply.started":"2022-09-21T06:32:23.302173Z","shell.execute_reply":"2022-09-21T06:32:23.305615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(\"../input\")","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:32:25.279833Z","iopub.execute_input":"2022-09-21T06:32:25.280204Z","iopub.status.idle":"2022-09-21T06:32:25.291546Z","shell.execute_reply.started":"2022-09-21T06:32:25.280172Z","shell.execute_reply":"2022-09-21T06:32:25.290602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data = pd.read_csv(\"../input/mayo-clinic-strip-ai/train.csv\")\ndf_data1 = pd.read_csv(\"../input/mayo-clinic-strip-ai/test.csv\")\n# remove pictures that did not make it to the downscaled dataset:\ndf_data = df_data.drop(df_data.query(f\"image_id=='b894f4_0'\").index)\ndf_data = df_data.drop(df_data.query(f\"image_id=='6baf51_0'\").index)\n\nprint(df_data.shape)\nprint(df_data1.shape)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:32:27.613053Z","iopub.execute_input":"2022-09-21T06:32:27.613458Z","iopub.status.idle":"2022-09-21T06:32:27.65477Z","shell.execute_reply.started":"2022-09-21T06:32:27.613422Z","shell.execute_reply":"2022-09-21T06:32:27.653766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_data1.columns\ndf_data1['image_id']","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:32:56.757711Z","iopub.execute_input":"2022-09-21T06:32:56.758105Z","iopub.status.idle":"2022-09-21T06:32:56.770912Z","shell.execute_reply.started":"2022-09-21T06:32:56.758073Z","shell.execute_reply":"2022-09-21T06:32:56.765719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_PATH = \"../input/stroke-blood-clot-origin-1k-scale-bg-crop/train_images/\"\nTEST_IMAGE_PATH = \"../input/mayoclinictest-imagesresized1024/test\"\n\nprint(\n    f\"We train on a scaled down dataset containing {len(os.listdir(IMAGE_PATH))} images\"\n)\nprint(\n    f\"We will test on a downscaled test set containing {len(os.listdir(TEST_IMAGE_PATH))} images\"\n)\n# print(len(os.listdir('../input/test')))\ndf_data['path'] = IMAGE_PATH + df_data['image_id'] +\".png\"\ndf_data1['path'] = TEST_IMAGE_PATH + df_data1['image_id']+\".png\"","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:33.960099Z","iopub.execute_input":"2022-09-21T05:05:33.960777Z","iopub.status.idle":"2022-09-21T05:05:33.974432Z","shell.execute_reply.started":"2022-09-21T05:05:33.960738Z","shell.execute_reply":"2022-09-21T05:05:33.97346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for p in df_data1['path']:\n    print(p)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:33.975577Z","iopub.execute_input":"2022-09-21T05:05:33.97674Z","iopub.status.idle":"2022-09-21T05:05:33.98266Z","shell.execute_reply.started":"2022-09-21T05:05:33.976676Z","shell.execute_reply":"2022-09-21T05:05:33.981773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.head()\ndf_data1.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:33.98375Z","iopub.execute_input":"2022-09-21T05:05:33.984498Z","iopub.status.idle":"2022-09-21T05:05:34.001473Z","shell.execute_reply.started":"2022-09-21T05:05:33.984466Z","shell.execute_reply":"2022-09-21T05:05:34.000301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CE_count, LAA_count = df_data[\"label\"].value_counts()\nprint(f\"There are {CE_count} CE samples and {LAA_count} LAA samples\")\n\n# CE_count1, LAA_count1 = df_data1[\"label\"].value_counts()\n# print(f\"There are {CE_count1} CE samples and {LAA_count1} LAA samples\")","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:34.00311Z","iopub.execute_input":"2022-09-21T05:05:34.003541Z","iopub.status.idle":"2022-09-21T05:05:34.014517Z","shell.execute_reply.started":"2022-09-21T05:05:34.003499Z","shell.execute_reply":"2022-09-21T05:05:34.013661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# source: https://www.kaggle.com/gpreda/honey-bee-subspecies-classification\n\n\ndef draw_category_images(col_name, figure_cols, df, IMAGE_PATH):\n    categories = (df.groupby([col_name])[col_name].nunique()).index\n    f, ax = plt.subplots(\n        nrows=len(categories),\n        ncols=figure_cols,\n        figsize=(4 * figure_cols, 4 * len(categories)),\n    )  # adjust size here\n    # draw a number of images for each location\n    for i, cat in enumerate(categories):\n        sample = df[df[col_name] == cat].sample(\n            figure_cols\n        )  # figure_cols is also the sample size\n        for j in range(0, figure_cols):\n            file = IMAGE_PATH + sample.iloc[j][\"image_id\"] + \".png\"\n            im = cv2.imread(file)\n            ax[i, j].imshow(im, resample=True, cmap=\"gray\")\n            ax[i, j].set_title(cat, fontsize=16)\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:34.015601Z","iopub.execute_input":"2022-09-21T05:05:34.016549Z","iopub.status.idle":"2022-09-21T05:05:34.026956Z","shell.execute_reply.started":"2022-09-21T05:05:34.016514Z","shell.execute_reply":"2022-09-21T05:05:34.025812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"draw_category_images(\"label\", 4, df_data, IMAGE_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:34.061385Z","iopub.execute_input":"2022-09-21T05:05:34.061804Z","iopub.status.idle":"2022-09-21T05:05:37.138484Z","shell.execute_reply.started":"2022-09-21T05:05:34.061768Z","shell.execute_reply":"2022-09-21T05:05:37.137084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# take a random sample of class 0 with size equal to num samples in class 1\ndf_0 = df_data[df_data[\"label\"] == \"CE\"].sample(CE_count, random_state=101)\n# filter out class 1\ndf_1 = df_data[df_data[\"label\"] == \"LAA\"].sample(LAA_count, random_state=101)\n\n# concat the dataframes\ndf_data = pd.concat([df_0, df_1], axis=0).reset_index(drop=True)\n# shuffle\ndf_data = shuffle(df_data)\n\ndf_data[\"label\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:37.141199Z","iopub.execute_input":"2022-09-21T05:05:37.141664Z","iopub.status.idle":"2022-09-21T05:05:37.159371Z","shell.execute_reply.started":"2022-09-21T05:05:37.141627Z","shell.execute_reply":"2022-09-21T05:05:37.158061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_test_split\n\n# stratify=y creates a balanced validation set.\ny = df_data[\"label\"]\n\ndf_train, df_val = train_test_split(\n    df_data, test_size=0.10, random_state=101, stratify=y\n)\n\nprint(df_train.shape)\nprint(df_val.shape)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:37.160816Z","iopub.execute_input":"2022-09-21T05:05:37.161825Z","iopub.status.idle":"2022-09-21T05:05:37.170027Z","shell.execute_reply.started":"2022-09-21T05:05:37.16179Z","shell.execute_reply":"2022-09-21T05:05:37.169033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"label\"].value_counts()\n# df_train1[\"label\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:37.171731Z","iopub.execute_input":"2022-09-21T05:05:37.172442Z","iopub.status.idle":"2022-09-21T05:05:37.185567Z","shell.execute_reply.started":"2022-09-21T05:05:37.172393Z","shell.execute_reply":"2022-09-21T05:05:37.18444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_val[\"label\"].value_counts()\n# df_val1[\"label\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:37.188161Z","iopub.execute_input":"2022-09-21T05:05:37.188681Z","iopub.status.idle":"2022-09-21T05:05:37.199599Z","shell.execute_reply.started":"2022-09-21T05:05:37.188649Z","shell.execute_reply":"2022-09-21T05:05:37.198639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new directory\nbase_dir = \"base_dir\"\nos.makedirs(base_dir, exist_ok=True)\n\n\n# [CREATE FOLDERS INSIDE THE BASE DIRECTORY]\n\n# now we create 2 folders inside 'base_dir':\n\n# train_dir\n# a_CE\n# b_LAA\n\n# val_dir\n# a_CE\n# b_LAA\n\n# create a path to 'base_dir' to which we will join the names of the new folders\n# train_dir\ntrain_dir = os.path.join(base_dir, \"train_dir\")\nos.makedirs(train_dir, exist_ok=True)\n\n# val_dir\nval_dir = os.path.join(base_dir, \"val_dir\")\nos.makedirs(val_dir, exist_ok=True)\n\n# [CREATE FOLDERS INSIDE THE TRAIN AND VALIDATION FOLDERS]\n# Inside each folder we create seperate folders for each class\n\n# create new folders inside train_dir\nCE = os.path.join(train_dir, \"a_CE\")\nos.makedirs(CE, exist_ok=True)\nLAA = os.path.join(train_dir, \"b_LAA\")\nos.makedirs(LAA, exist_ok=True)\n\n\n# create new folders inside val_dir\nCE = os.path.join(val_dir, \"a_CE\")\nos.makedirs(CE, exist_ok=True)\nLAA = os.path.join(val_dir, \"b_LAA\")\nos.makedirs(LAA, exist_ok=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:37.200914Z","iopub.execute_input":"2022-09-21T05:05:37.201476Z","iopub.status.idle":"2022-09-21T05:05:37.212688Z","shell.execute_reply.started":"2022-09-21T05:05:37.201444Z","shell.execute_reply":"2022-09-21T05:05:37.211635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check that the folders have been created\nos.listdir(\"base_dir/train_dir\")\n# os.listdir(\"base_dir/test_dir\")","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:37.213943Z","iopub.execute_input":"2022-09-21T05:05:37.214784Z","iopub.status.idle":"2022-09-21T05:05:37.231079Z","shell.execute_reply.started":"2022-09-21T05:05:37.214737Z","shell.execute_reply":"2022-09-21T05:05:37.230149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the id as the index in df_data\ndf_data.set_index(\"image_id\", inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:37.232192Z","iopub.execute_input":"2022-09-21T05:05:37.233119Z","iopub.status.idle":"2022-09-21T05:05:37.241466Z","shell.execute_reply.started":"2022-09-21T05:05:37.23306Z","shell.execute_reply":"2022-09-21T05:05:37.240292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get a list of train and val images\ntrain_list = list(df_train[\"image_id\"])\nval_list = list(df_val[\"image_id\"])\n\n# Transfer the train images\nfor image in train_list:\n\n    # the id in the csv file does not have the .png extension therefore we add it here\n    fname = image + \".png\"\n    # get the label for a certain image\n    target = df_data.loc[image, \"label\"]\n\n    # these must match the folder names\n    if target == \"CE\":\n        label = \"a_CE\"\n    if target == \"LAA\":\n        label = \"b_LAA\"\n    # source path to image\n    src = os.path.join(IMAGE_PATH, fname)\n    # destination path to image\n    dst = os.path.join(train_dir, label, fname)\n    # copy the image from the source to the destination\n    shutil.copyfile(src, dst)\n\n# Transfer the val images\n\nfor image in val_list:\n\n    # the id in the csv file does not have the .png extension therefore we add it here\n    fname = image + \".png\"\n    # get the label for a certain image\n    target = df_data.loc[image, \"label\"]\n\n    # these must match the folder names\n    if target == \"CE\":\n        label = \"a_CE\"\n    if target == \"LAA\":\n        label = \"b_LAA\"\n\n    # source path to image\n    src = os.path.join(IMAGE_PATH, fname)\n    # destination path to image\n    dst = os.path.join(val_dir, label, fname)\n    # copy the image from the source to the destination\n    shutil.copyfile(src, dst)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:37.242916Z","iopub.execute_input":"2022-09-21T05:05:37.243547Z","iopub.status.idle":"2022-09-21T05:05:39.724641Z","shell.execute_reply.started":"2022-09-21T05:05:37.243513Z","shell.execute_reply":"2022-09-21T05:05:39.723726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check how many train images we have in each folder\nprint(len(os.listdir(\"base_dir/train_dir/a_CE\")))\nprint(len(os.listdir(\"base_dir/train_dir/b_LAA\")))","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:39.725808Z","iopub.execute_input":"2022-09-21T05:05:39.726565Z","iopub.status.idle":"2022-09-21T05:05:39.73371Z","shell.execute_reply.started":"2022-09-21T05:05:39.72653Z","shell.execute_reply":"2022-09-21T05:05:39.732555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check how many val images we have in each folder\n\nprint(len(os.listdir(\"base_dir/val_dir/a_CE\")))\nprint(len(os.listdir(\"base_dir/val_dir/b_LAA\")))","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:39.735141Z","iopub.execute_input":"2022-09-21T05:05:39.735755Z","iopub.status.idle":"2022-09-21T05:05:39.745842Z","shell.execute_reply.started":"2022-09-21T05:05:39.735691Z","shell.execute_reply":"2022-09-21T05:05:39.744909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = \"base_dir/train_dir\"\nvalid_path = \"base_dir/val_dir\"\ntest_path = \"../input/test\"\n\nnum_train_samples = len(df_train)\nnum_val_samples = len(df_val)\ntrain_batch_size = 5\nval_batch_size = 5\n\n\ntrain_steps = np.ceil(num_train_samples / train_batch_size)\nval_steps = np.ceil(num_val_samples / val_batch_size)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:39.747476Z","iopub.execute_input":"2022-09-21T05:05:39.748211Z","iopub.status.idle":"2022-09-21T05:05:39.757119Z","shell.execute_reply.started":"2022-09-21T05:05:39.748166Z","shell.execute_reply":"2022-09-21T05:05:39.75596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen = ImageDataGenerator(\n    rescale=1.0 / 255,\n    rotation_range=10,  # rotation\n    width_shift_range=0.2,  # horizontal shift\n    height_shift_range=0.2,  # vertical shift\n    zoom_range=0.2,  # zoom\n    horizontal_flip=True,  # horizontal flip\n    brightness_range=[0.2, 1.2],\n)  # brightness)\ntrain_gen = datagen.flow_from_directory(\n    train_path,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=train_batch_size,\n    color_mode=\"rgb\",  # for coloured images\n    class_mode=\"categorical\",\n)\n\nval_gen = datagen.flow_from_directory(\n    valid_path,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=val_batch_size,\n    color_mode=\"rgb\",  # for coloured images\n    class_mode=\"categorical\",\n)\n# Note: shuffle=False causes the test dataset to not be shuffled\ntest_gen = datagen.flow_from_directory(\n    valid_path,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=1,\n    class_mode=\"categorical\",\n    color_mode=\"rgb\",  # for coloured images\n    shuffle=False,\n)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:39.758613Z","iopub.execute_input":"2022-09-21T05:05:39.759213Z","iopub.status.idle":"2022-09-21T05:05:40.078513Z","shell.execute_reply.started":"2022-09-21T05:05:39.759164Z","shell.execute_reply":"2022-09-21T05:05:40.077408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check how many train images we have in each folder\n\nprint(len(os.listdir(\"base_dir/train_dir/a_CE\")))\nprint(len(os.listdir(\"base_dir/train_dir/b_LAA\")))\n# check how many val images we have in each folder\n\nprint(len(os.listdir(\"base_dir/val_dir/a_CE\")))\nprint(len(os.listdir(\"base_dir/val_dir/b_LAA\")))","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:40.08373Z","iopub.execute_input":"2022-09-21T05:05:40.084111Z","iopub.status.idle":"2022-09-21T05:05:40.091499Z","shell.execute_reply.started":"2022-09-21T05:05:40.084075Z","shell.execute_reply":"2022-09-21T05:05:40.090543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Input, Lambda, Dense, Flatten\nfrom tensorflow.keras.models import Model\n# from tensorflow.keras.applications.inception_v3 import InceptionV3\n\nfrom tensorflow.keras.applications import MobileNetV2\n# from tensorflow.keras.applications.inception_v3 import preprocess_input\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator,load_img\nimport numpy as np\nfrom glob import glob\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D\nfrom tensorflow.keras.layers import MaxPooling2D,GlobalAveragePooling2D\n#import matplotlib.pyplot as plt\n","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:40.092668Z","iopub.execute_input":"2022-09-21T05:05:40.093073Z","iopub.status.idle":"2022-09-21T05:05:40.103281Z","shell.execute_reply.started":"2022-09-21T05:05:40.093031Z","shell.execute_reply":"2022-09-21T05:05:40.102235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = [224, 224]\nmobilenetv2 = MobileNetV2(input_shape=IMAGE_SIZE+[3], weights='imagenet', include_top=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:40.10441Z","iopub.execute_input":"2022-09-21T05:05:40.105429Z","iopub.status.idle":"2022-09-21T05:05:41.298237Z","shell.execute_reply.started":"2022-09-21T05:05:40.105322Z","shell.execute_reply":"2022-09-21T05:05:41.296982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# don't train existing weights\nfor layer in mobilenetv2.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:41.299888Z","iopub.execute_input":"2022-09-21T05:05:41.300795Z","iopub.status.idle":"2022-09-21T05:05:41.312173Z","shell.execute_reply.started":"2022-09-21T05:05:41.300745Z","shell.execute_reply":"2022-09-21T05:05:41.310893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# useful for getting number of output classes\nfolders = glob('./base_dir/train_dir/*')\nlen(folders)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:41.313852Z","iopub.execute_input":"2022-09-21T05:05:41.314604Z","iopub.status.idle":"2022-09-21T05:05:41.329845Z","shell.execute_reply.started":"2022-09-21T05:05:41.314552Z","shell.execute_reply":"2022-09-21T05:05:41.328571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# our layers - you can add more if you want\n# base_model = MobileNetV2(input_shape = (224,224,3),weights = 'imagenet',include_top = True)\nx=mobilenetv2.output\n# x=GlobalAveragePooling2D()(x)\nx=Flatten()(x)\nx=Dense(224,activation='relu')(x)\n# x=Dense(4,activation='relu')(x)\nx=Dropout(0.2)(x)\n# x = Flatten()(mobilenetv2.output)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:41.331335Z","iopub.execute_input":"2022-09-21T05:05:41.331803Z","iopub.status.idle":"2022-09-21T05:05:41.407379Z","shell.execute_reply.started":"2022-09-21T05:05:41.331757Z","shell.execute_reply":"2022-09-21T05:05:41.406354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = Dense(2, activation='sigmoid')(x)\n# create a model object\nmodel = Model(inputs=mobilenetv2.input, outputs=prediction)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:41.409642Z","iopub.execute_input":"2022-09-21T05:05:41.410001Z","iopub.status.idle":"2022-09-21T05:05:41.43677Z","shell.execute_reply.started":"2022-09-21T05:05:41.409967Z","shell.execute_reply":"2022-09-21T05:05:41.435769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# view the structure of the model\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:41.438246Z","iopub.execute_input":"2022-09-21T05:05:41.438936Z","iopub.status.idle":"2022-09-21T05:05:41.466974Z","shell.execute_reply.started":"2022-09-21T05:05:41.4389Z","shell.execute_reply":"2022-09-21T05:05:41.466123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(Adam(learning_rate=0.001), loss=\"categorical_crossentropy\", metrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:41.468374Z","iopub.execute_input":"2022-09-21T05:05:41.468947Z","iopub.status.idle":"2022-09-21T05:05:41.480644Z","shell.execute_reply.started":"2022-09-21T05:05:41.468912Z","shell.execute_reply":"2022-09-21T05:05:41.479545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen = ImageDataGenerator(\n    rescale=1.0 / 255,\n    rotation_range=10,  # rotation\n    width_shift_range=0.2,  # horizontal shift\n    height_shift_range=0.2,  # vertical shift\n    zoom_range=0.2,  # zoom\n    horizontal_flip=True,  # horizontal flip\n    brightness_range=[0.2, 1.2],\n)  # brightness)\n\nIMAGE_SIZE = 224\ntraining_set = datagen.flow_from_directory(\n    train_path,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size= 32 ,\n    color_mode=\"rgb\",  # for coloured images\n    class_mode=\"categorical\",\n)\n\n\nval_gen = datagen.flow_from_directory(\n    valid_path,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=32,\n    color_mode=\"rgb\",  # for coloured images\n    class_mode=\"categorical\",\n)\n\n# Note: shuffle=False causes the test dataset to not be shuffled\ntest_set = datagen.flow_from_directory(\n    valid_path,\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=32,\n    class_mode=\"categorical\",\n    color_mode=\"rgb\",  # for coloured images\n    shuffle=False,\n)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:41.482181Z","iopub.execute_input":"2022-09-21T05:05:41.482557Z","iopub.status.idle":"2022-09-21T05:05:41.800081Z","shell.execute_reply.started":"2022-09-21T05:05:41.482526Z","shell.execute_reply":"2022-09-21T05:05:41.798989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.callbacks import ModelCheckpoint, EarlyStopping, CSVLogger\n\ncsvlog = CSVLogger('logs.csv', separator=\",\", append=False)\n\ncall_backs_list = [csvlog]","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:41.801444Z","iopub.execute_input":"2022-09-21T05:05:41.801791Z","iopub.status.idle":"2022-09-21T05:05:41.807181Z","shell.execute_reply.started":"2022-09-21T05:05:41.801759Z","shell.execute_reply":"2022-09-21T05:05:41.806226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r = model.fit(\n  training_set,\n  validation_data=test_set,\n  epochs=60,\n  steps_per_epoch=len(training_set),\n  validation_steps=len(test_set)\n)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T05:05:41.808861Z","iopub.execute_input":"2022-09-21T05:05:41.80953Z","iopub.status.idle":"2022-09-21T06:11:40.250305Z","shell.execute_reply.started":"2022-09-21T05:05:41.809497Z","shell.execute_reply":"2022-09-21T06:11:40.248771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot the loss\nimport matplotlib.pyplot as plt\nplt.plot(r.history['loss'], label='train loss')\nplt.plot(r.history['val_loss'], label='val loss')\nplt.legend()\nplt.show()\nplt.savefig('LossVal_loss')\n\n# plot the accuracy\nplt.plot(r.history['accuracy'], label='train acc')\nplt.plot(r.history['val_accuracy'], label='val acc')\nplt.legend()\nplt.show()\nplt.savefig('AccVal_acc')","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:40.252522Z","iopub.execute_input":"2022-09-21T06:11:40.252974Z","iopub.status.idle":"2022-09-21T06:11:40.587728Z","shell.execute_reply.started":"2022-09-21T06:11:40.2529Z","shell.execute_reply":"2022-09-21T06:11:40.586507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from keras.preprocessing.image import ImageDataGenerator\ntest_gen = ImageDataGenerator(rescale = 1./255,\n                                rotation_range = 30,\n                                shear_range=0.5,\n                                zoom_range=[0.5,0.7],\n                                horizontal_flip=True,\n                                vertical_flip=True\n                                )","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:40.589644Z","iopub.execute_input":"2022-09-21T06:11:40.59204Z","iopub.status.idle":"2022-09-21T06:11:40.598625Z","shell.execute_reply.started":"2022-09-21T06:11:40.591995Z","shell.execute_reply":"2022-09-21T06:11:40.597477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rescale=1. / 255,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:40.599928Z","iopub.execute_input":"2022-09-21T06:11:40.600245Z","iopub.status.idle":"2022-09-21T06:11:40.61022Z","shell.execute_reply.started":"2022-09-21T06:11:40.600217Z","shell.execute_reply":"2022-09-21T06:11:40.60912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights(\"model_saved.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:40.612568Z","iopub.execute_input":"2022-09-21T06:11:40.613251Z","iopub.status.idle":"2022-09-21T06:11:40.920683Z","shell.execute_reply.started":"2022-09-21T06:11:40.613216Z","shell.execute_reply":"2022-09-21T06:11:40.919477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import load_model\nfrom keras.preprocessing.image import load_img\nfrom keras.preprocessing.image import img_to_array","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:40.922507Z","iopub.execute_input":"2022-09-21T06:11:40.923029Z","iopub.status.idle":"2022-09-21T06:11:40.928844Z","shell.execute_reply.started":"2022-09-21T06:11:40.922991Z","shell.execute_reply":"2022-09-21T06:11:40.927541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rescale=1. / 255,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:40.930363Z","iopub.execute_input":"2022-09-21T06:11:40.931112Z","iopub.status.idle":"2022-09-21T06:11:40.94091Z","shell.execute_reply.started":"2022-09-21T06:11:40.931076Z","shell.execute_reply":"2022-09-21T06:11:40.939542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import load_model\n  \n#Model = load_model('model_saved.h5')\n#for i in range:  \nimage = load_img('../input/mayoclinictest-imagesresized1024/test/006388_0.png', target_size=(224,224))\nimg = np.array(image)\nimg = img / 255.0\nimg = img.reshape(1,224,224,3)\nlabel = model.predict(img)\nprint(\"Predicted Class (0 - CEE , 1- LAA): \", label[0][0])","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:40.942551Z","iopub.execute_input":"2022-09-21T06:11:40.942932Z","iopub.status.idle":"2022-09-21T06:11:42.29312Z","shell.execute_reply.started":"2022-09-21T06:11:40.942897Z","shell.execute_reply":"2022-09-21T06:11:42.291029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = load_img('../input/mayoclinictest-imagesresized1024/test/008e5c_0.png', target_size=(224,224))\nimg = np.array(image)\nimg = img / 255.0\nimg = img.reshape(1,224,224,3)\nlabel = model.predict(img)\nprint(\"Predicted Class (0 - CEE , 1- LAA): \", label[0][0])","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:42.294378Z","iopub.execute_input":"2022-09-21T06:11:42.294731Z","iopub.status.idle":"2022-09-21T06:11:42.439166Z","shell.execute_reply.started":"2022-09-21T06:11:42.294678Z","shell.execute_reply":"2022-09-21T06:11:42.437959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = load_img('../input/mayoclinictest-imagesresized1024/test/00c058_0.png', target_size=(224,224))\nimg = np.array(image)\nimg = img / 255.0\nimg = img.reshape(1,224,224,3)\nlabel = model.predict(img)\nprint(\"Predicted Class (0 - CEE , 1- LAA): \", label[0][0])\nprint(label)\n#import numpy as np\n#y_pred = np.argmax(label, axis=1)\n#print(y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:42.440746Z","iopub.execute_input":"2022-09-21T06:11:42.441979Z","iopub.status.idle":"2022-09-21T06:11:42.58274Z","shell.execute_reply.started":"2022-09-21T06:11:42.441929Z","shell.execute_reply":"2022-09-21T06:11:42.581528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = load_img(\"../input/mayoclinictest-imagesresized1024/test/01adc5_0.png\", target_size=(224,224))\nimg = np.array(image)\nimg = img / 255.0\nimg = img.reshape(1,224,224,3)\nlabel = model.predict(img)\nprint(\"Predicted Class (0 - CEE , 1- LAA): \", label[0][0])\nprint(label)\n#import numpy as np\n#y_pred = np.argmax(label, axis=1)\n#print(y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:42.584275Z","iopub.execute_input":"2022-09-21T06:11:42.584712Z","iopub.status.idle":"2022-09-21T06:11:42.756313Z","shell.execute_reply.started":"2022-09-21T06:11:42.584651Z","shell.execute_reply":"2022-09-21T06:11:42.755047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_dir =\"../input/mayoclinictest-imagesresized1024/test/\"","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:42.758051Z","iopub.execute_input":"2022-09-21T06:11:42.758522Z","iopub.status.idle":"2022-09-21T06:11:42.763855Z","shell.execute_reply.started":"2022-09-21T06:11:42.758475Z","shell.execute_reply":"2022-09-21T06:11:42.762513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gen = ImageDataGenerator(rescale = 1./255,\n                                rotation_range = 30,\n                                shear_range=0.5,\n                                zoom_range=[0.5,0.7],\n                                horizontal_flip=True,\n                                vertical_flip=True\n                                )\n#y_pred = model.predict(test_set)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:42.765464Z","iopub.execute_input":"2022-09-21T06:11:42.765996Z","iopub.status.idle":"2022-09-21T06:11:42.776268Z","shell.execute_reply.started":"2022-09-21T06:11:42.765937Z","shell.execute_reply":"2022-09-21T06:11:42.775057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_path = \"../input/mayoclinictest-imagesresized1024/test\"","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:42.777986Z","iopub.execute_input":"2022-09-21T06:11:42.778361Z","iopub.status.idle":"2022-09-21T06:11:42.788289Z","shell.execute_reply.started":"2022-09-21T06:11:42.778327Z","shell.execute_reply":"2022-09-21T06:11:42.787157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_gen = test_gen.flow_from_directory(\n    val_dir,\n    target_size=(1024,1024),\n    batch_size=4,\n    class_mode=\"binary\",\n    classes=['.']\n#     color_mode=\"rgb\",  # for coloured images\n#     shuffle=False,\n)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:42.789826Z","iopub.execute_input":"2022-09-21T06:11:42.790157Z","iopub.status.idle":"2022-09-21T06:11:42.905315Z","shell.execute_reply.started":"2022-09-21T06:11:42.790128Z","shell.execute_reply":"2022-09-21T06:11:42.904001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dir =\"../input/mayoclinictest-imagesresized1024/test/\"","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:42.90684Z","iopub.execute_input":"2022-09-21T06:11:42.907206Z","iopub.status.idle":"2022-09-21T06:11:42.91241Z","shell.execute_reply.started":"2022-09-21T06:11:42.907171Z","shell.execute_reply":"2022-09-21T06:11:42.911059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_gen = test_gen.flow_from_directory(test_dir,\n        target_size=(224,224),\n        batch_size= 4 ,shuffle=False,\n        class_mode= 'binary',classes=['.'])","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:42.921239Z","iopub.execute_input":"2022-09-21T06:11:42.921906Z","iopub.status.idle":"2022-09-21T06:11:43.029261Z","shell.execute_reply.started":"2022-09-21T06:11:42.921864Z","shell.execute_reply":"2022-09-21T06:11:43.028062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(test_data_gen)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:43.032783Z","iopub.execute_input":"2022-09-21T06:11:43.033147Z","iopub.status.idle":"2022-09-21T06:11:44.538497Z","shell.execute_reply.started":"2022-09-21T06:11:43.033115Z","shell.execute_reply":"2022-09-21T06:11:44.537537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:44.539762Z","iopub.execute_input":"2022-09-21T06:11:44.54035Z","iopub.status.idle":"2022-09-21T06:11:44.547367Z","shell.execute_reply.started":"2022-09-21T06:11:44.540317Z","shell.execute_reply":"2022-09-21T06:11:44.546238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_y=[]\nfor i in range(0,len(y_pred)):\n    pred_y.append(y_pred[i][0])\npred_y","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:44.548897Z","iopub.execute_input":"2022-09-21T06:11:44.549321Z","iopub.status.idle":"2022-09-21T06:11:44.560652Z","shell.execute_reply.started":"2022-09-21T06:11:44.549288Z","shell.execute_reply":"2022-09-21T06:11:44.559424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(df_data1[\"path\"])\nsubmission[\"CE\"] = pred_y\nsubmission[\"CE\"] = submission[\"CE\"].apply(lambda x : 0 if x<0 else x)\nsubmission[\"CE\"] = submission[\"CE\"].apply(lambda x : 1 if x>1 else x)\nsubmission[\"LAA\"] = 1- submission[\"CE\"]\n\nsubmission = submission.groupby(\"path\").mean()\nsubmission = submission[[\"CE\", \"LAA\"]].round(6).reset_index()\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:44.562249Z","iopub.execute_input":"2022-09-21T06:11:44.562625Z","iopub.status.idle":"2022-09-21T06:11:44.598553Z","shell.execute_reply.started":"2022-09-21T06:11:44.56257Z","shell.execute_reply":"2022-09-21T06:11:44.597572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['CE'],submission['LAA']","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:44.599808Z","iopub.execute_input":"2022-09-21T06:11:44.600569Z","iopub.status.idle":"2022-09-21T06:11:44.609818Z","shell.execute_reply.started":"2022-09-21T06:11:44.600531Z","shell.execute_reply":"2022-09-21T06:11:44.608479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:44.611063Z","iopub.execute_input":"2022-09-21T06:11:44.613746Z","iopub.status.idle":"2022-09-21T06:11:44.627328Z","shell.execute_reply.started":"2022-09-21T06:11:44.613681Z","shell.execute_reply":"2022-09-21T06:11:44.625725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-09-21T06:11:44.628765Z","iopub.execute_input":"2022-09-21T06:11:44.629853Z","iopub.status.idle":"2022-09-21T06:11:44.641913Z","shell.execute_reply.started":"2022-09-21T06:11:44.629807Z","shell.execute_reply":"2022-09-21T06:11:44.640573Z"},"trusted":true},"execution_count":null,"outputs":[]}]}