{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ***1- Import Libraries***","metadata":{}},{"cell_type":"code","source":"import time\nimport os, cv2\nimport warnings\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport seaborn as sns\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nwarnings.filterwarnings('ignore')\n\nfrom tensorflow.keras.optimizers import  Adam\nfrom sklearn.preprocessing import LabelEncoder\nfrom tensorflow.keras.models import  Sequential\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.losses import binary_crossentropy\nfrom tensorflow.keras.applications import EfficientNetB3, Xception\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, BatchNormalization, Dropout","metadata":{"execution":{"iopub.status.busy":"2024-01-04T12:50:02.381088Z","iopub.execute_input":"2024-01-04T12:50:02.381803Z","iopub.status.idle":"2024-01-04T12:50:02.392935Z","shell.execute_reply.started":"2024-01-04T12:50:02.381767Z","shell.execute_reply":"2024-01-04T12:50:02.391844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ***2- Look at the data***","metadata":{}},{"cell_type":"code","source":"Main_Path = \"/kaggle/input/UBC-OCEAN/\"\n\ntrain, test = pd.read_csv(Main_Path + 'train.csv'), pd.read_csv(Main_Path + 'test.csv')\n\ndisplay(\n    train,\n    \n    test)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T12:50:02.395226Z","iopub.execute_input":"2024-01-04T12:50:02.39579Z","iopub.status.idle":"2024-01-04T12:50:02.442146Z","shell.execute_reply.started":"2024-01-04T12:50:02.395748Z","shell.execute_reply":"2024-01-04T12:50:02.441191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_img, test_img = os.listdir(Main_Path + \"train_images\"), os.listdir(Main_Path + \"test_images\")\n\ndisplay(\n    len(test_img), test_img[:],\n    \"--------------------------\",\n    len(train_img), train_img[:5])","metadata":{"execution":{"iopub.status.busy":"2024-01-04T12:50:02.44353Z","iopub.execute_input":"2024-01-04T12:50:02.444054Z","iopub.status.idle":"2024-01-04T12:50:02.461688Z","shell.execute_reply.started":"2024-01-04T12:50:02.444001Z","shell.execute_reply":"2024-01-04T12:50:02.46042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_thumb, test_thumb = os.listdir(Main_Path + \"train_thumbnails\"), os.listdir(Main_Path + \"test_thumbnails\")\n\ndisplay(\n    len(test_thumb), test_thumb[:],\n    \"------------------------------\",\n    len(train_thumb), train_thumb[:5])","metadata":{"execution":{"iopub.status.busy":"2024-01-04T12:50:02.464213Z","iopub.execute_input":"2024-01-04T12:50:02.465323Z","iopub.status.idle":"2024-01-04T12:50:02.483319Z","shell.execute_reply.started":"2024-01-04T12:50:02.465282Z","shell.execute_reply":"2024-01-04T12:50:02.482087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ***3- Some Feature Manipulation***","metadata":{}},{"cell_type":"code","source":"tma = train[train['is_tma'] == True]\n\nno_tma = train[train['is_tma'] == False]\n\ndisplay(\n    len(tma),\n    '--------',\n    len(no_tma)\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T12:50:02.484926Z","iopub.execute_input":"2024-01-04T12:50:02.485377Z","iopub.status.idle":"2024-01-04T12:50:02.50152Z","shell.execute_reply.started":"2024-01-04T12:50:02.485337Z","shell.execute_reply":"2024-01-04T12:50:02.500246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tma['image_id'] = [f\"{i}.png\" for i in tma['image_id']]\n\nno_tma['image_id'] = [f\"{i}_thumbnail.png\" for i in no_tma['image_id']]\n\ndisplay(\n    tma.shape,\n    \n    no_tma.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T12:50:02.503424Z","iopub.execute_input":"2024-01-04T12:50:02.503747Z","iopub.status.idle":"2024-01-04T12:50:02.517465Z","shell.execute_reply.started":"2024-01-04T12:50:02.503719Z","shell.execute_reply":"2024-01-04T12:50:02.515884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_data = []\nimage_label = []\n\nfor img , label in zip(no_tma['image_id'], no_tma['label']):\n    \n    image = Image.open(\"/kaggle/input/UBC-OCEAN/train_thumbnails/\"+img)\n    image = image.resize((600, 600))\n    image = image.convert(\"RGB\")\n    image = np.array(image)\n    image_data.append(image)\n    image_label.append(label)\n    \nfor img , label in zip(tma['image_id'], tma['label']):\n    \n    image = Image.open(\"/kaggle/input/UBC-OCEAN/train_images/\"+img)\n    image = image.resize((600, 600))\n    image = image.convert(\"RGB\")\n    image = np.array(image)\n    image_data.append(image)\n    image_label.append(label)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T12:50:02.519024Z","iopub.execute_input":"2024-01-04T12:50:02.519369Z","iopub.status.idle":"2024-01-04T12:52:36.03383Z","shell.execute_reply.started":"2024-01-04T12:50:02.51934Z","shell.execute_reply":"2024-01-04T12:52:36.032131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(\n    len(image_data),\n    \n    len(image_label)\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T12:52:36.037387Z","iopub.execute_input":"2024-01-04T12:52:36.038127Z","iopub.status.idle":"2024-01-04T12:52:36.047415Z","shell.execute_reply.started":"2024-01-04T12:52:36.038074Z","shell.execute_reply":"2024-01-04T12:52:36.045991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder()\nimage_label = le.fit_transform(image_label)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T12:52:36.049582Z","iopub.execute_input":"2024-01-04T12:52:36.050112Z","iopub.status.idle":"2024-01-04T12:52:36.058228Z","shell.execute_reply.started":"2024-01-04T12:52:36.050048Z","shell.execute_reply":"2024-01-04T12:52:36.056732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.array(image_data)\ny = np.array(image_label)\n\ndisplay(\n    X.shape,\n    \n    y.shape)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T12:52:36.059939Z","iopub.execute_input":"2024-01-04T12:52:36.060435Z","iopub.status.idle":"2024-01-04T12:52:36.577398Z","shell.execute_reply.started":"2024-01-04T12:52:36.060394Z","shell.execute_reply":"2024-01-04T12:52:36.576491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.10, shuffle=True)\n\ndisplay(\n    X_train.shape,\n    X_test.shape,\n    \n    y_train.shape,\n    y_test.shape\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T12:52:36.579087Z","iopub.execute_input":"2024-01-04T12:52:36.579467Z","iopub.status.idle":"2024-01-04T12:52:36.923335Z","shell.execute_reply.started":"2024-01-04T12:52:36.579433Z","shell.execute_reply":"2024-01-04T12:52:36.922117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_tr_scaled = X_train/255\nX_te_scaled = X_test/255\n\ndisplay(\n    X_tr_scaled.shape,\n    \n    X_te_scaled.shape\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T12:52:36.924827Z","iopub.execute_input":"2024-01-04T12:52:36.925963Z","iopub.status.idle":"2024-01-04T12:52:40.037564Z","shell.execute_reply.started":"2024-01-04T12:52:36.925914Z","shell.execute_reply":"2024-01-04T12:52:40.03657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the CNN model\nnum_classes = 5\n\n# Convert the labels to one-hot encoded format\ny_train = to_categorical(y_train, num_classes)\ny_test = to_categorical(y_test, num_classes)\n\n\nmodel = Sequential([\n    \n    Conv2D(32, (3, 3), activation='relu', input_shape=(600, 600, 3)),\n    MaxPooling2D(2, 2),\n    \n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D(2, 2),\n    \n    Conv2D(128, (3, 3), activation='relu'),\n    MaxPooling2D(2, 2),\n    \n    Flatten(),\n    #Dense(512, activation='relu'),\n    \n    #Dropout(0.5),\n    Dense(num_classes, activation='softmax')\n])\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Train the model\nhistory = model.fit(X_tr_scaled, y_train, epochs=3, validation_data=(X_te_scaled, y_test))","metadata":{"execution":{"iopub.status.busy":"2024-01-04T12:52:40.038996Z","iopub.execute_input":"2024-01-04T12:52:40.039809Z","iopub.status.idle":"2024-01-04T13:03:15.257362Z","shell.execute_reply.started":"2024-01-04T12:52:40.039766Z","shell.execute_reply":"2024-01-04T13:03:15.252937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, accuracy = model.evaluate(X_te_scaled, y_test)\n\ndisplay(\n    loss,\n    \n    accuracy\n)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T13:03:15.264306Z","iopub.execute_input":"2024-01-04T13:03:15.267474Z","iopub.status.idle":"2024-01-04T13:03:37.748236Z","shell.execute_reply.started":"2024-01-04T13:03:15.267357Z","shell.execute_reply":"2024-01-04T13:03:37.746618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_predict = model.predict(X_te_scaled)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T13:03:37.750178Z","iopub.execute_input":"2024-01-04T13:03:37.750572Z","iopub.status.idle":"2024-01-04T13:03:49.481832Z","shell.execute_reply.started":"2024-01-04T13:03:37.750538Z","shell.execute_reply":"2024-01-04T13:03:49.480509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_label = [np.argmax(i) for i in y_predict]\n\ny_pred_label[:10]","metadata":{"execution":{"iopub.status.busy":"2024-01-04T13:03:49.484184Z","iopub.execute_input":"2024-01-04T13:03:49.485122Z","iopub.status.idle":"2024-01-04T13:03:49.494349Z","shell.execute_reply.started":"2024-01-04T13:03:49.485074Z","shell.execute_reply":"2024-01-04T13:03:49.493057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_img = []\n\nfor img in test_thumb:\n    \n    image = Image.open(\"/kaggle/input/UBC-OCEAN/test_thumbnails/\"+img)\n    image = image.resize((600, 600))\n    image = image.convert(\"RGB\")\n    image = np.array(image)\n    test_img.append(image)\n                           \npred_test = np.array(test_img)\n\n# Scaling  The test Images\npredict_scaled = pred_test/255","metadata":{"execution":{"iopub.status.busy":"2024-01-04T13:03:49.495949Z","iopub.execute_input":"2024-01-04T13:03:49.496321Z","iopub.status.idle":"2024-01-04T13:03:49.923793Z","shell.execute_reply.started":"2024-01-04T13:03:49.496291Z","shell.execute_reply":"2024-01-04T13:03:49.922366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(pred_test)\npredictions = [np.argmax(i) for i in predictions]\npredictions","metadata":{"execution":{"iopub.status.busy":"2024-01-04T13:03:49.925766Z","iopub.execute_input":"2024-01-04T13:03:49.926317Z","iopub.status.idle":"2024-01-04T13:03:50.28366Z","shell.execute_reply.started":"2024-01-04T13:03:49.926265Z","shell.execute_reply":"2024-01-04T13:03:50.282604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_labels = ['HGSC', 'LGSC', 'EC', 'MC', 'CC']\n\nsubmission = pd.DataFrame({\n    'image_id': test['image_id'],\n    'label':  [class_labels[i] for i in predictions]\n})\n\nsubmission.to_csv('submission.csv', index_label=False)\n\nsubmission","metadata":{"execution":{"iopub.status.busy":"2024-01-04T13:03:50.287413Z","iopub.execute_input":"2024-01-04T13:03:50.287804Z","iopub.status.idle":"2024-01-04T13:03:50.311209Z","shell.execute_reply.started":"2024-01-04T13:03:50.287771Z","shell.execute_reply":"2024-01-04T13:03:50.310353Z"},"trusted":true},"execution_count":null,"outputs":[]}]}