{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport progressbar\nimport os\nfrom sklearn.model_selection import train_test_split\nimport time\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.metrics import f1_score","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-07T18:33:18.936561Z","iopub.execute_input":"2023-07-07T18:33:18.937257Z","iopub.status.idle":"2023-07-07T18:33:18.943951Z","shell.execute_reply.started":"2023-07-07T18:33:18.937223Z","shell.execute_reply":"2023-07-07T18:33:18.942656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_embeddings = np.load('/kaggle/input/t5embeds/train_embeds.npy')\nlabels_y = np.load(\"/kaggle/input/xgbdata/Y_1499.npy\")","metadata":{"execution":{"iopub.status.busy":"2023-07-07T18:35:31.796787Z","iopub.execute_input":"2023-07-07T18:35:31.797173Z","iopub.status.idle":"2023-07-07T18:35:41.16812Z","shell.execute_reply.started":"2023-07-07T18:35:31.797143Z","shell.execute_reply":"2023-07-07T18:35:41.167133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_y.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-27T13:31:27.424989Z","iopub.execute_input":"2023-06-27T13:31:27.425349Z","iopub.status.idle":"2023-06-27T13:31:27.431413Z","shell.execute_reply.started":"2023-06-27T13:31:27.425319Z","shell.execute_reply":"2023-06-27T13:31:27.430583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_trn, X_tst, y_trn, y_tst = train_test_split( train_embeddings, labels_y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-07-07T18:35:42.776198Z","iopub.execute_input":"2023-07-07T18:35:42.776547Z","iopub.status.idle":"2023-07-07T18:35:46.392251Z","shell.execute_reply.started":"2023-07-07T18:35:42.776521Z","shell.execute_reply":"2023-07-07T18:35:46.391254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'X_trn: {X_trn.shape}, X_tst: {X_tst.shape}, Y_trn: {y_trn.shape}, Y_tst: {y_tst.shape}')","metadata":{"execution":{"iopub.status.busy":"2023-07-07T18:34:43.412451Z","iopub.status.idle":"2023-07-07T18:34:43.413188Z","shell.execute_reply.started":"2023-07-07T18:34:43.412916Z","shell.execute_reply":"2023-07-07T18:34:43.412942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\",sep=\"\\t\")\nprint(train_terms.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-07T18:34:43.414516Z","iopub.status.idle":"2023-07-07T18:34:43.41521Z","shell.execute_reply.started":"2023-07-07T18:34:43.414958Z","shell.execute_reply":"2023-07-07T18:34:43.414982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:41:31.109475Z","iopub.execute_input":"2023-06-21T09:41:31.109836Z","iopub.status.idle":"2023-06-21T09:41:31.12884Z","shell.execute_reply.started":"2023-06-21T09:41:31.109808Z","shell.execute_reply":"2023-06-21T09:41:31.1277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_protein_ids = np.load('/kaggle/input/t5embeds/train_ids.npy')\nprint(train_protein_ids.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:41:32.495518Z","iopub.execute_input":"2023-06-21T09:41:32.495872Z","iopub.status.idle":"2023-06-21T09:41:32.544336Z","shell.execute_reply.started":"2023-06-21T09:41:32.495843Z","shell.execute_reply":"2023-06-21T09:41:32.543231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"column_num = train_embeddings.shape[1]\ntrain_df = pd.DataFrame(train_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\nprint(train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:41:35.255914Z","iopub.execute_input":"2023-06-21T09:41:35.256326Z","iopub.status.idle":"2023-06-21T09:41:46.021459Z","shell.execute_reply.started":"2023-06-21T09:41:35.256294Z","shell.execute_reply":"2023-06-21T09:41:46.020445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:41:46.023508Z","iopub.execute_input":"2023-06-21T09:41:46.023891Z","iopub.status.idle":"2023-06-21T09:41:46.030057Z","shell.execute_reply.started":"2023-06-21T09:41:46.023857Z","shell.execute_reply":"2023-06-21T09:41:46.029012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_of_labels = 1499\nlabels_to_consider = train_terms['term'].value_counts().index[:num_of_labels].tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms_updated = train_terms.loc[train_terms['term'].isin(labels_to_consider)]","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:43:12.955899Z","iopub.execute_input":"2023-06-21T09:43:12.956433Z","iopub.status.idle":"2023-06-21T09:43:13.880082Z","shell.execute_reply.started":"2023-06-21T09:43:12.956396Z","shell.execute_reply":"2023-06-21T09:43:13.879143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms_updated.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-21T09:43:28.641655Z","iopub.execute_input":"2023-06-21T09:43:28.64201Z","iopub.status.idle":"2023-06-21T09:43:28.64891Z","shell.execute_reply.started":"2023-06-21T09:43:28.641974Z","shell.execute_reply":"2023-06-21T09:43:28.647811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_SHAPE = [X_trn.shape[1]]\nBATCH_SIZE = 5120\n\nmodel = tf.keras.Sequential([\n    tf.keras.layers.BatchNormalization(input_shape=INPUT_SHAPE),    \n    tf.keras.layers.Dense(units=512, activation='relu'),\n    tf.keras.layers.Dense(units=1024, activation='relu'),\n    tf.keras.layers.Dense(units=1256, activation='relu'),\n    tf.keras.layers.Dense(units=1340, activation='relu'),\n    tf.keras.layers.Dense(units=1420, activation='relu'),\n    tf.keras.layers.Dense(units=y_trn.shape[1],activation='sigmoid')\n])\n\ncheckpoint_path = \"training/cp.ckpt\"\ncheckpoint_dir = os.path.dirname(checkpoint_path)\n\n\ncp_callback = tf.keras.callbacks.ModelCheckpoint(filepath=checkpoint_path,\n                                                 save_weights_only=True,\n                                                 verbose=1,save_freq='epoch')\n\n# Compile model\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy', tf.keras.metrics.AUC()]\n)\n\nhistory = model.fit(\n    X_trn, y_trn,\n    batch_size=BATCH_SIZE,\n    epochs=50,callbacks=[cp_callback]\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-07T18:37:38.746855Z","iopub.execute_input":"2023-07-07T18:37:38.747564Z","iopub.status.idle":"2023-07-07T18:39:51.69672Z","shell.execute_reply.started":"2023-07-07T18:37:38.747531Z","shell.execute_reply":"2023-07-07T18:39:51.695709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the threshold\nthreshold = 0.5\n# model.fit(X_trn,y_trn)\npredictions = model.predict(X_tst)\nval_preds = np.where(predictions > threshold, 1, 0)\nval_labels = y_tst\ntp = np.sum((val_preds == 1) & (val_labels == 1))\nfp = np.sum((val_preds == 1) & (val_labels == 0))\nfn = np.sum((val_preds == 0) & (val_labels == 1))\n\nprecision = tp / (tp + fp)\nrecall = tp / (tp + fn)\nf1_score = 2 * (precision * recall) / (precision + recall)\nhamming_loss = np.mean(val_preds != val_labels)\n\nprint(\"Hamming loss:\", hamming_loss)\nprint(\"F-max score:\", f1_score)","metadata":{"execution":{"iopub.status.busy":"2023-07-07T18:45:05.339026Z","iopub.execute_input":"2023-07-07T18:45:05.340209Z","iopub.status.idle":"2023-07-07T18:45:09.660165Z","shell.execute_reply.started":"2023-07-07T18:45:05.340166Z","shell.execute_reply":"2023-07-07T18:45:09.658387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Use cells below if using custom pre trained weights","metadata":{}},{"cell_type":"code","source":"# Testing\nINPUT_SHAPE = [X_trn.shape[1]]\nBATCH_SIZE = 5120\n\nmodel = tf.keras.Sequential([\n    tf.keras.layers.BatchNormalization(input_shape=INPUT_SHAPE),    \n    tf.keras.layers.Dense(units=512, activation='relu'),\n    tf.keras.layers.Dense(units=1024, activation='relu'),\n    tf.keras.layers.Dense(units=1256, activation='relu'),\n    tf.keras.layers.Dense(units=1340, activation='relu'),\n    tf.keras.layers.Dense(units=1420, activation='relu'),\n    tf.keras.layers.Dense(units=y_trn.shape[1],activation='sigmoid')\n])\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy', tf.keras.metrics.AUC()]\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-27T13:31:37.701919Z","iopub.execute_input":"2023-06-27T13:31:37.702308Z","iopub.status.idle":"2023-06-27T13:31:40.592634Z","shell.execute_reply.started":"2023-06-27T13:31:37.702254Z","shell.execute_reply":"2023-06-27T13:31:40.591663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"latest = tf.train.latest_checkpoint(\"/kaggle/input/checkpt\")\nlatest","metadata":{"execution":{"iopub.status.busy":"2023-06-27T13:31:43.199744Z","iopub.execute_input":"2023-06-27T13:31:43.200092Z","iopub.status.idle":"2023-06-27T13:31:43.215434Z","shell.execute_reply.started":"2023-06-27T13:31:43.200064Z","shell.execute_reply":"2023-06-27T13:31:43.214441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights(latest)\ndf_models_stat = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2023-06-27T13:31:44.179051Z","iopub.execute_input":"2023-06-27T13:31:44.179723Z","iopub.status.idle":"2023-06-27T13:31:44.645863Z","shell.execute_reply.started":"2023-06-27T13:31:44.17969Z","shell.execute_reply":"2023-06-27T13:31:44.644941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t0 = time.time()\nmodel.fit(X_trn,y_trn)\nY_pred_test = model.predict(X_tst)\ntt = time.time() - t0\nprint(\"MLP hehe\", tt)\nl = []\nfor i in range(y_trn.shape[1]):\n    if len(np.unique(X_tst) ) > 1:\n        s = roc_auc_score(y_tst[:,i], Y_pred_test[:,i]);\n    else:\n        s = 0.5\n    l.append(s)        \n    if i %10 == 0:\n        print(i, s)\ndf_models_stat.loc[\"MLP\",'RocAuc Mean Test'] = np.mean(l)\ndf_models_stat.loc[\"MLP\",'Time'] = np.round(tt,1)\ndf_models_stat.loc[\"MLP\",'Test Size'] = len(X_tst)\ndf_models_stat","metadata":{"execution":{"iopub.status.busy":"2023-06-27T13:31:53.173614Z","iopub.execute_input":"2023-06-27T13:31:53.173977Z","iopub.status.idle":"2023-06-27T13:51:02.42853Z","shell.execute_reply.started":"2023-06-27T13:31:53.173949Z","shell.execute_reply":"2023-06-27T13:51:02.427396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_model_stat.loc[\"MLP\",\"RocAuc Mean Test\"].plot(title=\"ROCAUC\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, ['binary_accuracy']].plot(title=\"Accuracy\")","metadata":{"execution":{"iopub.status.busy":"2023-07-07T18:40:38.719985Z","iopub.execute_input":"2023-07-07T18:40:38.721028Z","iopub.status.idle":"2023-07-07T18:40:39.72661Z","shell.execute_reply.started":"2023-07-07T18:40:38.720971Z","shell.execute_reply":"2023-07-07T18:40:39.725655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_embeddings = np.load('/kaggle/input/t5embeds/test_embeds.npy')\n\ncolumn_num = test_embeddings.shape[1]\ntest_df = pd.DataFrame(test_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\nprint(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T20:05:58.479172Z","iopub.execute_input":"2023-06-21T20:05:58.479482Z","iopub.status.idle":"2023-06-21T20:06:09.983092Z","shell.execute_reply.started":"2023-06-21T20:05:58.479455Z","shell.execute_reply":"2023-06-21T20:06:09.982004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T20:06:09.984408Z","iopub.execute_input":"2023-06-21T20:06:09.985152Z","iopub.status.idle":"2023-06-21T20:06:10.04113Z","shell.execute_reply.started":"2023-06-21T20:06:09.985118Z","shell.execute_reply":"2023-06-21T20:06:10.040227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test_df)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T20:06:10.042802Z","iopub.execute_input":"2023-06-21T20:06:10.044507Z","iopub.status.idle":"2023-06-21T20:06:26.904404Z","shell.execute_reply.started":"2023-06-21T20:06:10.044476Z","shell.execute_reply":"2023-06-21T20:06:26.903423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save(\"pred\", predictions)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T20:06:26.907634Z","iopub.execute_input":"2023-06-21T20:06:26.908087Z","iopub.status.idle":"2023-06-21T20:06:28.675235Z","shell.execute_reply.started":"2023-06-21T20:06:26.908051Z","shell.execute_reply":"2023-06-21T20:06:28.673992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_small = predictions[:35000]","metadata":{"execution":{"iopub.status.busy":"2023-06-21T19:55:18.70266Z","iopub.execute_input":"2023-06-21T19:55:18.703338Z","iopub.status.idle":"2023-06-21T19:55:18.707926Z","shell.execute_reply.started":"2023-06-21T19:55:18.703306Z","shell.execute_reply":"2023-06-21T19:55:18.706876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_to_consider_repeated = [item for _ in range(35000) for item in labels_y]\nlabels_to_consider_repeated[:10]","metadata":{"execution":{"iopub.status.busy":"2023-06-22T10:36:53.788163Z","iopub.execute_input":"2023-06-22T10:36:53.788561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])\ntest_protein_ids = np.load('/kaggle/input/t5embeds/test_ids.npy')\nl = []\nfor k in list(test_protein_ids[:35000]):\n    l += [ k] * predictions_small.shape[1]   \n\ndf_submission['Protein Id'] = l\ndf_submission['GO Term Id'] = labels_to_consider_repeated\ndf_submission['Prediction'] = predictions_small.ravel()\ndf_submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-19T15:31:42.05318Z","iopub.execute_input":"2023-06-19T15:31:42.053529Z","iopub.status.idle":"2023-06-19T15:32:04.44924Z","shell.execute_reply.started":"2023-06-19T15:31:42.053497Z","shell.execute_reply":"2023-06-19T15:32:04.448252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{},"execution_count":null,"outputs":[]}]}