{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tifffile as tiff\nimport pandas as pd \nimport cv2\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm \nimport os","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:58:08.858089Z","iopub.execute_input":"2022-08-12T02:58:08.859353Z","iopub.status.idle":"2022-08-12T02:58:09.206957Z","shell.execute_reply.started":"2022-08-12T02:58:08.859219Z","shell.execute_reply":"2022-08-12T02:58:09.20591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:58:09.208793Z","iopub.execute_input":"2022-08-12T02:58:09.209385Z","iopub.status.idle":"2022-08-12T02:58:09.21682Z","shell.execute_reply.started":"2022-08-12T02:58:09.209347Z","shell.execute_reply":"2022-08-12T02:58:09.215742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_dir = '../input/mayo-18-reduction-with-background-color-adjusted/nobackground_data/train'\n# os.makedirs('./train_512x512')\n# for index,row in tqdm(df.iterrows()):\n#     img_id = row['image_id']\n#     img_fn = '{}.png'.format(img_id)\n#     img_fp = os.path.join(train_dir, img_fn)\n#     img = cv2.imread(img_fp)\n#     img = cv2.resize(img, (512,512))\n#     cv2.imwrite('./train_512x512/' + str(img_id) + '.jpg', img)\n# # img = tiff.imread('../input/mayo-clinic-strip-ai/train/006388_0.tif')\n# # img = cv2.resize(img, (512,512))\n# # plt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:58:09.218515Z","iopub.execute_input":"2022-08-12T02:58:09.21892Z","iopub.status.idle":"2022-08-12T02:58:09.227871Z","shell.execute_reply.started":"2022-08-12T02:58:09.218884Z","shell.execute_reply":"2022-08-12T02:58:09.226543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create CSV file for training ","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\ndf","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:58:09.23057Z","iopub.execute_input":"2022-08-12T02:58:09.231051Z","iopub.status.idle":"2022-08-12T02:58:09.276937Z","shell.execute_reply.started":"2022-08-12T02:58:09.231017Z","shell.execute_reply":"2022-08-12T02:58:09.275849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import glob\n# img_path = '../input/jpg-images-strip-ai/train/*.jpg'\n# result = dict()\n# out_id = []\n# out_name = []\n# out_path = []\n# for img in glob.glob(img_path):\n#     img_name = os.path.basename(img)\n#     img_id = img_name.split('.')[0]\n#     out_id.append(img_id)\n#     out_name.append(img_name)\n#     out_path.append(img)\n# result['image_id'] = out_id\n# result['image_name'] = out_name\n# result['full_path'] = out_path\n# result = pd.DataFrame.from_dict(result)\n# print(result)\n# result.to_csv('train1.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:58:09.277976Z","iopub.execute_input":"2022-08-12T02:58:09.278386Z","iopub.status.idle":"2022-08-12T02:58:09.285255Z","shell.execute_reply.started":"2022-08-12T02:58:09.27834Z","shell.execute_reply":"2022-08-12T02:58:09.284138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1 = pd.read_csv('./train1.csv')\n# df_train = pd.merge(df,df1,on=['image_id'])\n# df_train.to_csv('train_clean_data.csv', index=False)\n# print(df_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:58:09.287202Z","iopub.execute_input":"2022-08-12T02:58:09.288348Z","iopub.status.idle":"2022-08-12T02:58:09.294074Z","shell.execute_reply.started":"2022-08-12T02:58:09.288312Z","shell.execute_reply":"2022-08-12T02:58:09.293116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Custom DataGenerators","metadata":{}},{"cell_type":"code","source":"# df1 = pd.read_csv('../input/strip-ai-datasetssss/train_images/train_clean_train_data.csv')\n# df2 = pd.read_csv('../input/strip-ai-datasetssss/train_images/train_clean_val_data.csv')\n\n# data_dir_train = '../input/strip-ai-datasetssss/train_images/train/'\n# df1['Ids'] = df1['image_name'].apply(lambda x: f'{data_dir_train}{x}')\n# print(df1.head())\n# df1.to_csv('train_df.csv',index=False)\n# df1\n# data_dir_val = '../input/strip-ai-datasetssss/train_images/test/'\n# df2['Ids'] = df2['image_name'].apply(lambda x: f'{data_dir_val}{x}')\n# print(df2.head())\n# df2.to_csv('val_df.csv',index=False)\n# df2","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:58:09.295491Z","iopub.execute_input":"2022-08-12T02:58:09.297272Z","iopub.status.idle":"2022-08-12T02:58:09.306401Z","shell.execute_reply.started":"2022-08-12T02:58:09.297236Z","shell.execute_reply":"2022-08-12T02:58:09.30531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport numpy as np\nIMG_SIZE = 512\ndef load_data(db):\n    if db == \"train\":\n        \n        df_train = pd.read_csv('../input/csv-training-strip-ai-data/train_df.csv')\n        \n        df_val = pd.read_csv('../input/csv-training-strip-ai-data/val_df.csv')\n        \n        datagen = ImageDataGenerator(\n#                 rotation_range=30,\n#                 width_shift_range=0.5,\n#                 height_shift_range=0.5,\n#                 zoom_range=0.45,\n#                 shear_range=0.45,\n            )\n        train_data = datagen.flow_from_dataframe(\n                dataframe=df_train,\n                x_col='Ids',\n                y_col='label',\n                target_size=(IMG_SIZE, IMG_SIZE),\n                shuffle=True,\n                batch_size=64,\n                class_mode='categorical',\n                color_mode='rgb'\n            )\n        val_datagen = ImageDataGenerator()\n        val_data = val_datagen.flow_from_dataframe(\n                dataframe=df_val,\n                x_col='Ids',\n                y_col='label',\n                target_size=(IMG_SIZE, IMG_SIZE),\n                shuffle=False,\n                batch_size=32,\n                class_mode='categorical',\n                color_mode='rgb'\n\n            )\n        return train_data, val_data\n    else:\n        print('Can not load data')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:58:09.307977Z","iopub.execute_input":"2022-08-12T02:58:09.308961Z","iopub.status.idle":"2022-08-12T02:58:09.838044Z","shell.execute_reply.started":"2022-08-12T02:58:09.308924Z","shell.execute_reply":"2022-08-12T02:58:09.837119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training ","metadata":{}},{"cell_type":"code","source":"os.makedirs('models')\nos.makedirs('checkpoint')\nos.makedirs('logs')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:58:09.839497Z","iopub.execute_input":"2022-08-12T02:58:09.840294Z","iopub.status.idle":"2022-08-12T02:58:09.846407Z","shell.execute_reply.started":"2022-08-12T02:58:09.840255Z","shell.execute_reply":"2022-08-12T02:58:09.845249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB7\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\nfrom tensorflow.keras import optimizers\nfrom tensorflow.keras import callbacks\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import models\nimport time","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:58:09.850696Z","iopub.execute_input":"2022-08-12T02:58:09.851104Z","iopub.status.idle":"2022-08-12T02:58:15.170969Z","shell.execute_reply.started":"2022-08-12T02:58:09.851048Z","shell.execute_reply":"2022-08-12T02:58:15.169925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef train(train_generator, val_generator, LR):\n    checkpoint_file_path = 'epoch-{epoch:02d}-val_acc-{val_accuracy:.4f}.h5'\n    MODELS = './models'\n    CPS = './checkpoint'\n    LOG = './logs'\n    IMG_SIZE = 512\n    base_model = EfficientNetB7(weights=None, include_top=False, input_tensor=layers.Input(shape=(IMG_SIZE,IMG_SIZE,3)))\n    base_model.load_weights('../input/pretrained-model-1/efficientnetb7_notop.h5')\n    model = base_model.output\n    model = layers.GlobalAveragePooling2D()(model)\n    model = layers.Flatten()(model)\n    model = layers.Dense(2048, activation='relu')(model)\n    model = layers.Dense(2, activation='softmax')(model)\n    for layer in base_model.layers:\n        layer.trainable = False\n    model = models.Model(base_model.input, model)\n#     model.summary()\n    print(model.count_params(), model.inputs, model.outputs)\n    \n    checkpoint = callbacks.ModelCheckpoint(\n        filepath=os.path.join(CPS, checkpoint_file_path),\n        save_best_only=True,\n        save_weights_only=False,\n        monitor='val_accuracy',\n        mode='auto')\n\n    early = callbacks.EarlyStopping(\n        monitor=\"val_accuracy\", mode=\"min\", patience=4, verbose=1,restore_best_weights=True)\n    redonplat = callbacks.ReduceLROnPlateau(\n        monitor=\"val_accuracy\", factor=0.1, mode=\"min\", patience=3, verbose=1\n    )\n    \n    csv_logger = callbacks.CSVLogger(\n        os.path.join(LOG, 'Strip_AI_log_{}_{}.csv'.format(\n            IMG_SIZE, time.time()\n        )),\n        append=False, separator=','\n    )\n\n    callbacks_list = [\n        checkpoint,\n        early,\n        redonplat,\n        csv_logger,\n    ]\n\n    optim = optimizers.Adam(learning_rate=LR)\n    model.compile(loss='binary_crossentropy', optimizer=optim,\n                  metrics='accuracy')\n\n    history = model.fit(\n        train_generator,\n        steps_per_epoch=train_generator.samples // train_generator.batch_size,\n        epochs=50,\n        callbacks=callbacks_list,\n        validation_data=val_generator,\n        validation_steps=val_generator.samples // val_generator.batch_size\n    )\n    \n    model.save(os.path.join(MODELS,'Strip_AI.h5'))\n    return history, LR\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:58:15.172671Z","iopub.execute_input":"2022-08-12T02:58:15.173439Z","iopub.status.idle":"2022-08-12T02:58:15.188667Z","shell.execute_reply.started":"2022-08-12T02:58:15.173325Z","shell.execute_reply":"2022-08-12T02:58:15.185833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data, val_data = load_data(\"train\")\ntrain(train_generator=train_data, val_generator=val_data, LR=1e-4)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:58:15.189881Z","iopub.execute_input":"2022-08-12T02:58:15.19078Z","iopub.status.idle":"2022-08-12T03:06:19.66046Z","shell.execute_reply.started":"2022-08-12T02:58:15.190743Z","shell.execute_reply":"2022-08-12T03:06:19.657106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport cv2 \nfrom tensorflow.keras.models import load_model\nfrom tensorflow.keras.preprocessing.image import img_to_array\n\nsubmission = pd.read_csv('../input/mayo-clinic-strip-ai/sample_submission.csv')\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:06:19.663841Z","iopub.status.idle":"2022-08-12T03:06:19.664621Z","shell.execute_reply.started":"2022-08-12T03:06:19.664351Z","shell.execute_reply":"2022-08-12T03:06:19.664378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = load_model('./models/Strip_AI.h5')\n# model = load_model('./checkpoint/epoch-04-val_acc-0.7500.h5')\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:06:19.666157Z","iopub.status.idle":"2022-08-12T03:06:19.666928Z","shell.execute_reply.started":"2022-08-12T03:06:19.666663Z","shell.execute_reply":"2022-08-12T03:06:19.666687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport os\ntest_path = '../input/jpg-images-strip-ai/test'\ndf_test = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\nout_pattern = []\nout_ce = []\nout_laa = []\nresult = dict()\n\nfor index, row in df_test.iterrows():\n    ids = row['image_id']\n    pattern = row['patient_id']\n    print(pattern)\n    img_fn = '{}.jpg'.format(ids)\n    img_fp = os.path.join(test_path, img_fn)\n    image = cv2.imread(img_fp)[:,:,::-1]\n    img = cv2.resize(image,(512,512))\n    img = img_to_array(img)\n#     img = img / 255\n#     img = preprocess_input(img)\n    img = np.expand_dims(img, axis=0)\n    pred = model.predict(img)\n    print(pred)\n    out_pattern.append(pattern)\n    out_ce.append(pred[:,0])\n    out_laa.append(pred[:,1])\n\nresult['patient_id'] = out_pattern\n# result['CE'] = out_ce\n# result['LAA'] = out_laa\nresult['CE'] = list(map(float, out_ce))\nresult['LAA'] = list(map(float, out_laa))\nresult = pd.DataFrame.from_dict(result)\n# result['CE'] = list(map(float, result['CE']))\n# result['LAA'] = list(map(float, result['LAA']))\nresult.to_csv('submission.csv',mode='a', index=False)\nresult","metadata":{"execution":{"iopub.status.busy":"2022-08-12T03:06:19.668368Z","iopub.status.idle":"2022-08-12T03:06:19.669185Z","shell.execute_reply.started":"2022-08-12T03:06:19.668866Z","shell.execute_reply":"2022-08-12T03:06:19.668901Z"},"trusted":true},"execution_count":null,"outputs":[]}]}