{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport shutil\n\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n\ntrain_csv = \"/kaggle/input/mayo-clinic-strip-ai/train.csv\"\nprocessed_files = \"/kaggle/input/explore-traindata/processed\"\nprocessed_data_path = \"/kaggle/working/processed_data.csv\"\nprocessed_data_path1 = \"/kaggle/input/mayo-v1/processed_data.csv\"\n\nif os.path.isfile(processed_data_path1):\n    shutil.copyfile(processed_data_path1, processed_data_path)\n# !ls /kaggle/input/explore/processed \n\nimport pandas as pd\ndef process_data():\n    df = pd.read_csv(train_csv)\n    print(df)\n    df.drop(['center_id','patient_id','image_num'], inplace=True, axis=1)\n    print(df)\n\n    processed_data = []\n    for i in df.values.tolist():\n        [imageid,label] = i\n    #     print(imageid,label)\n        imageid_path = os.path.join(processed_files,imageid)\n        files = [[os.path.join(imageid_path,f),label] for f in os.listdir(imageid_path) if os.path.isfile(os.path.join(imageid_path,f))]\n        processed_data = processed_data+files\n\n    print(processed_data[:20],\"FINISH\")\n    np.savetxt(\"processed_data.csv\", \n           processed_data,\n           delimiter =\", \", \n           fmt ='% s')\n\nDO_PROCESS_DATA = False\nif DO_PROCESS_DATA:\n    process_data()\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-19T09:21:02.096174Z","iopub.execute_input":"2022-08-19T09:21:02.096796Z","iopub.status.idle":"2022-08-19T09:26:39.936931Z","shell.execute_reply.started":"2022-08-19T09:21:02.096598Z","shell.execute_reply":"2022-08-19T09:26:39.935442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.path.isfile(processed_data_path))\ndf = pd.read_csv(processed_data_path,usecols=[0,1], header=0, index_col=False, names=['img', 'label'])\n\ndf = df.head(df.shape[0]//3)\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:53:01.524336Z","iopub.execute_input":"2022-08-19T09:53:01.524838Z","iopub.status.idle":"2022-08-19T09:53:01.930002Z","shell.execute_reply.started":"2022-08-19T09:53:01.524806Z","shell.execute_reply":"2022-08-19T09:53:01.928589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom tensorflow.keras.applications import EfficientNetB3,EfficientNetB2\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Conv2D\nfrom tensorflow.keras.regularizers import l2\n\n# ensure consistency across runs\n# from numpy.random import seed\n# seed(1)\n# from tensorflow import set_random_seed\n# set_random_seed(2)\n\n\n\n\n\ndef train_validate_test_split(df, train_percent=.6, validate_percent=.2, seed=None):\n    np.random.seed(seed)\n    perm = np.random.permutation(df.index)\n    m = len(df.index)\n    train_end = int(train_percent * m)\n    validate_end = int(validate_percent * m) + train_end\n    train = df.iloc[perm[:train_end]]\n    validate = df.iloc[perm[train_end:validate_end]]\n    test = df.iloc[perm[validate_end:]]\n    return train, validate, test\n\n\ntrain, validate, test = train_validate_test_split(df)\nprint(train, validate, test )\n\n\n\ntrain_datagen = ImageDataGenerator(rescale=1./255)\ntrain_set = train_datagen.flow_from_dataframe(train,\n                                              directory=None,\n                                              x_col=\"img\",\n                                              y_col=\"label\",\n                                              target_size=(300, 300),\n                                              batch_size=32,\n                                              class_mode='categorical',\n                                              )\n\nvalidate_datagen = ImageDataGenerator(rescale=1./255)\nvalidate_set = validate_datagen.flow_from_dataframe(validate,\n                                              directory=None,\n                                              x_col=\"img\",\n                                              y_col=\"label\",\n                                              target_size=(300, 300),\n                                              batch_size=32,\n                                              class_mode='categorical',\n                                              )\n\nSTEP_SIZE_TRAIN=train_set.n//train_set.batch_size\nSTEP_SIZE_VALID=validate_set.n//validate_set.batch_size\nprint(STEP_SIZE_TRAIN,STEP_SIZE_VALID)\n# STEP_SIZE_TEST=test_generator.n//test_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:53:04.592528Z","iopub.execute_input":"2022-08-19T09:53:04.59326Z","iopub.status.idle":"2022-08-19T09:54:16.49609Z","shell.execute_reply.started":"2022-08-19T09:53:04.593215Z","shell.execute_reply":"2022-08-19T09:54:16.494489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.optimizers import SGD,Adam\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras import layers\nprint(train_set.class_indices,validate_set.class_indices)\n\n\nefficientnetb2 = EfficientNetB2(\n#         weights='imagenet',\n        input_shape=(300,300,3),\n        include_top=False)\n\nefficientnetb3 = EfficientNetB3(\n#         weights='imagenet',\n        input_shape=(300,300,3),\n        include_top=False)\n\n\ndef build_model():\n    model = Sequential()\n    model.add(efficientnetb2)\n    model.add(Conv2D(32, (3,3), kernel_regularizer=l2(0.01), bias_regularizer=l2(0.01)))\n    model.add(layers.GlobalAveragePooling2D())\n    model.add(layers.Dropout(0.6))\n#     model.add(layers.BatchNormalization())\n    model.add(layers.Dense(2, activation='softmax'))\n    \n    model.compile(\n        loss='categorical_crossentropy',\n        optimizer=Adam(lr=1e-4,decay=1e-6),\n        metrics=['accuracy']\n    )\n    \n    return model\n\n\n# model = Sequential([\n#   base_model,\n#   layers.GlobalAveragePooling2D(),\n#   layers.Dense(2, activation='softmax'),\n# ])\n\n\nmodel = build_model()\nmodel.summary()\n\n\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-19T09:54:16.498831Z","iopub.execute_input":"2022-08-19T09:54:16.500073Z","iopub.status.idle":"2022-08-19T09:54:32.465438Z","shell.execute_reply.started":"2022-08-19T09:54:16.500027Z","shell.execute_reply":"2022-08-19T09:54:32.462499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\nfilepath=\"LPT-{epoch:02d}-{loss:.4f}.h5\"\ncheckpoint = ModelCheckpoint(filepath, monitor='loss', verbose=1, save_best_only=True, mode='min')\n\n\nmodel.fit_generator(generator=train_set,\n                    steps_per_epoch=STEP_SIZE_TRAIN,\n                    validation_data=validate_set,\n                    validation_steps=STEP_SIZE_VALID,\n                    callbacks=[checkpoint],\n                    epochs=9)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T10:02:49.513059Z","iopub.execute_input":"2022-08-19T10:02:49.513942Z"},"trusted":true},"execution_count":null,"outputs":[]}]}