{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# [MAYO] Simple CNN (Eng/日本語）\n\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport gc\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Dropout, Flatten, Dense\nfrom tensorflow.keras.layers import GlobalMaxPooling2D\nimport openslide\nfrom openslide import OpenSlide\nimport cv2 ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-20T01:32:45.869957Z","iopub.execute_input":"2022-08-20T01:32:45.870323Z","iopub.status.idle":"2022-08-20T01:32:53.700873Z","shell.execute_reply.started":"2022-08-20T01:32:45.870297Z","shell.execute_reply":"2022-08-20T01:32:53.699599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Read Data\n\nLet's read Train data and Test data. I added the link to image data since image data is stored in separate folder.  \nLabel column (CE or LAA) is the target for prediction, hence this column was changed to 1 or 0.  ","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\ntest_df  = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:32:53.703251Z","iopub.execute_input":"2022-08-20T01:32:53.703934Z","iopub.status.idle":"2022-08-20T01:32:53.726093Z","shell.execute_reply.started":"2022-08-20T01:32:53.703899Z","shell.execute_reply":"2022-08-20T01:32:53.725306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:32:53.727575Z","iopub.execute_input":"2022-08-20T01:32:53.728004Z","iopub.status.idle":"2022-08-20T01:32:53.749757Z","shell.execute_reply.started":"2022-08-20T01:32:53.727978Z","shell.execute_reply":"2022-08-20T01:32:53.748793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"file_path\"] = train_df[\"image_id\"].apply(lambda x: \"../input/mayo-clinic-strip-ai/train/\" + x + \".tif\")\ntest_df[\"file_path\"]  = test_df[\"image_id\"].apply(lambda x: \"../input/mayo-clinic-strip-ai/test/\" + x + \".tif\")","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:32:58.442637Z","iopub.execute_input":"2022-08-20T01:32:58.442985Z","iopub.status.idle":"2022-08-20T01:32:58.456447Z","shell.execute_reply.started":"2022-08-20T01:32:58.442958Z","shell.execute_reply":"2022-08-20T01:32:58.455425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"target\"] = train_df[\"label\"].apply(lambda x : 1 if x==\"CE\" else 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:33:10.426304Z","iopub.execute_input":"2022-08-20T01:33:10.426875Z","iopub.status.idle":"2022-08-20T01:33:10.432605Z","shell.execute_reply.started":"2022-08-20T01:33:10.426849Z","shell.execute_reply":"2022-08-20T01:33:10.431145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:33:20.143457Z","iopub.execute_input":"2022-08-20T01:33:20.143782Z","iopub.status.idle":"2022-08-20T01:33:20.155634Z","shell.execute_reply.started":"2022-08-20T01:33:20.143757Z","shell.execute_reply":"2022-08-20T01:33:20.154993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Quick look at Image of CE and LAA\n\nLet7s take a look at some actual image. First 2 records are CE and the next 2 records are LAA.  \nHmmm... I don't see specific difference between CE and LAA...","metadata":{}},{"cell_type":"code","source":"%%time\nsample_train = train_df[:4]\n\nfor i in range(4):\n    slide = OpenSlide(sample_train.loc[i, \"file_path\"])\n    region = (0, 0)\n    size = (10000, 10000)\n    region = slide.read_region(region, 0, size)\n    plt.figure(figsize=(8, 8))\n    plt.imshow(region)\n    plt.show()  ","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:33:34.8098Z","iopub.execute_input":"2022-08-20T01:33:34.810155Z","iopub.status.idle":"2022-08-20T01:35:07.447311Z","shell.execute_reply.started":"2022-08-20T01:33:34.810131Z","shell.execute_reply":"2022-08-20T01:35:07.445869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Image data preprocessing for CNN\n\nImage pixel will be changed to no.array for CNN processing.  \nAs you see in above 2., it takes long time to read each image data by 10,000x10,000 pixel, thus in this notebook 5000x 5000 pixel is fed to CNN, which will lead to less Training data. In this situation, I would like to read as meaningful data as possible and hence image data reading is starting from (1000,1000)position from the very top-left of the image since it seems the top-left potion of each image tends to be blank.  \nAlso, image data is resized to 512x512 in order to avoid memory over error.  \n\n","metadata":{}},{"cell_type":"code","source":"%%time\ndef preprocess(image_path):\n    slide=OpenSlide(image_path)\n    region= (1000,1000)    \n    size  = (5000, 5000)\n    image = slide.read_region(region, 0, size)\n    image = tf.image.resize(image, (512, 512))\n    image = np.array(image)    \n    return image\n\nx_train=[]\nfor i in tqdm(train_df['file_path']):\n    x1=preprocess(i)\n    x_train.append(x1)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T01:35:07.449646Z","iopub.execute_input":"2022-08-20T01:35:07.450018Z","iopub.status.idle":"2022-08-20T02:22:42.444911Z","shell.execute_reply.started":"2022-08-20T01:35:07.449985Z","shell.execute_reply":"2022-08-20T02:22:42.443494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. CNN Modelling\n\nConvolutional Neural Network(CNN) is built using Conv2D() method with 3x3 kernel. There will be more room to improve the model by tuning the number of Layers or Filters or adding Pooling layer, etc. This is just a starter.\n\n","metadata":{}},{"cell_type":"code","source":"model = Sequential()\ninput_shape = (512, 512, 4)\n\nmodel.add(Conv2D(filters=32, kernel_size = (3,3), strides =2, padding = 'same', activation = 'relu', input_shape = input_shape))\nmodel.add(Conv2D(filters=64, kernel_size = (3,3), strides =2, padding = 'same', activation = 'relu'))\nmodel.add(Conv2D(filters=32, kernel_size = (3,3), strides =2, padding = 'same', activation = 'relu'))\nmodel.add(Flatten())\nmodel.add(Dense(128, activation = 'relu'))\nmodel.add(Dropout(0.25))\n\nmodel.add(Dense(1))\n\nmodel.compile(\n    loss = tf.keras.losses.MeanSquaredError(),    \n    metrics=[tf.keras.metrics.RootMeanSquaredError(name=\"rmse\"), \"mae\", \"mape\"],\n    optimizer = tf.keras.optimizers.Adam(1e-3))","metadata":{"execution":{"iopub.status.busy":"2022-08-20T02:22:42.447934Z","iopub.execute_input":"2022-08-20T02:22:42.449287Z","iopub.status.idle":"2022-08-20T02:22:42.714608Z","shell.execute_reply.started":"2022-08-20T02:22:42.449241Z","shell.execute_reply":"2022-08-20T02:22:42.712734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train=np.array(x_train)\ny_train=train_df['target']\n\nx_train,x_test,y_train,y_test=train_test_split(x_train,y_train,test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T02:22:42.715976Z","iopub.execute_input":"2022-08-20T02:22:42.716324Z","iopub.status.idle":"2022-08-20T02:22:47.82981Z","shell.execute_reply.started":"2022-08-20T02:22:42.716297Z","shell.execute_reply":"2022-08-20T02:22:47.828671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nimport math\nfrom tensorflow.keras.callbacks import LearningRateScheduler, EarlyStopping, Callback\n\ndef step_decay(epoch):\n    initial_lrate = 0.001\n    drop = 0.5\n    epochs_drop = 10.0\n    lrate = initial_lrate * math.pow(drop, math.floor((epoch)/epochs_drop))\n    return lrate\n\nlrate = LearningRateScheduler(step_decay)\nearstop = EarlyStopping(monitor = 'val_loss', min_delta = 0, patience = 5)\n\nhistory = model.fit(\n    x_train,\n    y_train,\n    epochs = 20,\n    batch_size=64,\n    validation_data = (x_test,y_test),\n    verbose = 1,\n    callbacks = [lrate, earstop]\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T02:22:47.832433Z","iopub.execute_input":"2022-08-20T02:22:47.832735Z","iopub.status.idle":"2022-08-20T02:30:10.181378Z","shell.execute_reply.started":"2022-08-20T02:22:47.832708Z","shell.execute_reply":"2022-08-20T02:30:10.18068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df, x_train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T02:30:10.183318Z","iopub.execute_input":"2022-08-20T02:30:10.183651Z","iopub.status.idle":"2022-08-20T02:30:10.655285Z","shell.execute_reply.started":"2022-08-20T02:30:10.183625Z","shell.execute_reply":"2022-08-20T02:30:10.654089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Predict and Submission\n\nThis competition is not binary classification of CE and LAA, and competition owner states that both probability does not have to sum to one(1). However, Knowing that, LAA probability is calculated by 1 - CE probability since I intend to build as simple model as possible.  \n> The submitted probabilities for a given image are not required to sum to one because they are rescaled prior to being scored (each row is divided by the row sum)\n\nFYI, if you sinply put cnn_pred into Sample_submission file, it is no problem on the screen but it will cause an error when you submit it to the competition because the actual test data in submission has multiple image data per each patient. Hence, groupby.mean() is used here to make 1record per each patient.\n\n","metadata":{}},{"cell_type":"code","source":"test1=[]\nfor i in test_df['file_path']:\n    x1=preprocess(i)\n    test1.append(x1)\ntest1=np.array(test1)\n\ncnn_pred=model.predict(test1)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T02:30:10.656315Z","iopub.execute_input":"2022-08-20T02:30:10.656881Z","iopub.status.idle":"2022-08-20T02:30:33.630104Z","shell.execute_reply.started":"2022-08-20T02:30:10.656854Z","shell.execute_reply":"2022-08-20T02:30:33.62917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame(test_df[\"patient_id\"].copy())\nsub[\"CE\"] = cnn_pred\nsub[\"CE\"] = sub[\"CE\"].apply(lambda x : 0 if x<0 else x)\nsub[\"CE\"] = sub[\"CE\"].apply(lambda x : 1 if x>1 else x)\nsub[\"LAA\"] = 1- sub[\"CE\"]\n\nsub = sub.groupby(\"patient_id\").mean()\nsub = sub[[\"CE\", \"LAA\"]].round(6).reset_index()\nsub","metadata":{"execution":{"iopub.status.busy":"2022-08-20T02:30:33.631387Z","iopub.execute_input":"2022-08-20T02:30:33.631619Z","iopub.status.idle":"2022-08-20T02:30:33.701331Z","shell.execute_reply.started":"2022-08-20T02:30:33.631597Z","shell.execute_reply":"2022-08-20T02:30:33.700639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index = False)\n!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-08-20T02:30:33.702408Z","iopub.execute_input":"2022-08-20T02:30:33.702754Z","iopub.status.idle":"2022-08-20T02:30:34.353553Z","shell.execute_reply.started":"2022-08-20T02:30:33.702731Z","shell.execute_reply":"2022-08-20T02:30:34.352252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}