{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2023-07-21T20:31:43.982495Z","iopub.execute_input":"2023-07-21T20:31:43.982902Z","iopub.status.idle":"2023-07-21T20:31:44.047244Z","shell.execute_reply.started":"2023-07-21T20:31:43.982872Z","shell.execute_reply":"2023-07-21T20:31:44.045932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import required Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport seaborn as sns\n\nimport progressbar\n\nplt.style.use(\"ggplot\")","metadata":{"execution":{"iopub.status.busy":"2023-07-21T20:31:46.634294Z","iopub.execute_input":"2023-07-21T20:31:46.634697Z","iopub.status.idle":"2023-07-21T20:31:57.667991Z","shell.execute_reply.started":"2023-07-21T20:31:46.634664Z","shell.execute_reply":"2023-07-21T20:31:57.666765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import our Data","metadata":{"execution":{"iopub.status.busy":"2023-07-20T23:58:57.65409Z","iopub.execute_input":"2023-07-20T23:58:57.654632Z","iopub.status.idle":"2023-07-20T23:58:57.662906Z","shell.execute_reply.started":"2023-07-20T23:58:57.654576Z","shell.execute_reply":"2023-07-20T23:58:57.661462Z"}}},{"cell_type":"code","source":"train_terms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\", sep=\"\\t\")\n\ntrain_terms.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-21T20:32:07.089786Z","iopub.execute_input":"2023-07-21T20:32:07.090941Z","iopub.status.idle":"2023-07-21T20:32:11.400892Z","shell.execute_reply.started":"2023-07-21T20:32:07.090887Z","shell.execute_reply":"2023-07-21T20:32:11.399595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Our training dataset is about contains 3 columns **id** , **term** and  **aspect** and of a shape 5363863 records and 3 columns","metadata":{}},{"cell_type":"code","source":"train_terms.info()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T14:19:42.2027Z","iopub.execute_input":"2023-07-21T14:19:42.204347Z","iopub.status.idle":"2023-07-21T14:19:42.244986Z","shell.execute_reply.started":"2023-07-21T14:19:42.204272Z","shell.execute_reply":"2023-07-21T14:19:42.243177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:44:36.667952Z","iopub.execute_input":"2023-07-21T11:44:36.669012Z","iopub.status.idle":"2023-07-21T11:44:36.691022Z","shell.execute_reply.started":"2023-07-21T11:44:36.668964Z","shell.execute_reply":"2023-07-21T11:44:36.68881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the protein ids of the protein embeddings in the training_dataset form 't5embeds/train_ids.npy'\n\ntrain_protein_ids = np.load(\"/kaggle/input/t5embeds/train_ids.npy\")\n\ntrain_protein_ids.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-21T20:32:14.125955Z","iopub.execute_input":"2023-07-21T20:32:14.126374Z","iopub.status.idle":"2023-07-21T20:32:14.197957Z","shell.execute_reply.started":"2023-07-21T20:32:14.126343Z","shell.execute_reply":"2023-07-21T20:32:14.196699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(train_protein_ids)\n\nprint(train_protein_ids[:10])","metadata":{"execution":{"iopub.status.busy":"2023-07-21T14:19:49.19182Z","iopub.execute_input":"2023-07-21T14:19:49.192513Z","iopub.status.idle":"2023-07-21T14:19:49.199159Z","shell.execute_reply.started":"2023-07-21T14:19:49.192467Z","shell.execute_reply":"2023-07-21T14:19:49.197919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load the training protein embeddings also as a npy file from \"t5embeds/train_embeds.npy\"\n\ntrain_embeddings = np.load(\"/kaggle/input/t5embeds/train_embeds.npy\")\n\n# convert our ndarrays into pandas datafame\ncolumn_num = train_embeddings.shape[1]\ntrain_df = pd.DataFrame(train_embeddings, columns=[\"Column_\" + str(i)for i  in range(1,column_num+1)])\n\n# train_embeddings[:5]\n\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-21T20:32:18.791767Z","iopub.execute_input":"2023-07-21T20:32:18.792216Z","iopub.status.idle":"2023-07-21T20:32:31.164462Z","shell.execute_reply.started":"2023-07-21T20:32:18.792182Z","shell.execute_reply":"2023-07-21T20:32:31.163198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:53:57.433025Z","iopub.execute_input":"2023-07-21T11:53:57.433436Z","iopub.status.idle":"2023-07-21T11:53:57.463389Z","shell.execute_reply.started":"2023-07-21T11:53:57.433403Z","shell.execute_reply":"2023-07-21T11:53:57.462176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"# subdf has only 100 row \nsub_df = train_terms[\"term\"].value_counts().iloc[:100]\n\nfigure, axis = plt.subplots(1,1,figsize=(12,6))\n\nbarrplot =sns.barplot(ax=axis, x=np.array(sub_df.index),y=sub_df.values)\nbarrplot.set_xticklabels(barrplot.get_xticklabels(), rotation=90,size=6)\naxis.set_title(\"Top 100 frequent GO term IDs\")\nbarrplot.set_xlabel(\"GO term IDs\", fontsize=12)\nbarrplot.set_ylabel(\"Count\", fontsize=12)\n\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-21T14:24:09.433289Z","iopub.execute_input":"2023-07-21T14:24:09.433988Z","iopub.status.idle":"2023-07-21T14:24:12.081932Z","shell.execute_reply.started":"2023-07-21T14:24:09.433936Z","shell.execute_reply":"2023-07-21T14:24:12.078691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the limit for label\n\nnum_labels = 1500\nlabels = train_terms[\"term\"].value_counts().index[:1500].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T20:32:37.839914Z","iopub.execute_input":"2023-07-21T20:32:37.840323Z","iopub.status.idle":"2023-07-21T20:32:38.904535Z","shell.execute_reply.started":"2023-07-21T20:32:37.840291Z","shell.execute_reply":"2023-07-21T20:32:38.903414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fetch the train_terms data for the relevant labels only\ntrain_terms_updated = train_terms.loc[train_terms[\"term\"].isin(labels)]","metadata":{"execution":{"iopub.status.busy":"2023-07-21T20:32:43.167177Z","iopub.execute_input":"2023-07-21T20:32:43.167911Z","iopub.status.idle":"2023-07-21T20:32:44.032971Z","shell.execute_reply.started":"2023-07-21T20:32:43.167873Z","shell.execute_reply":"2023-07-21T20:32:44.031322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pie chart visualize the per of each class of the aspects \n\ndf_3 = train_terms_updated[\"aspect\"].value_counts()\npalette = sns.color_palette(\"bright\")\nplt.pie(df_3.values, labels=np.array(df_3.index), colors=palette,autopct=\"%0.f%%\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T20:32:45.602419Z","iopub.execute_input":"2023-07-21T20:32:45.602802Z","iopub.status.idle":"2023-07-21T20:32:46.532734Z","shell.execute_reply.started":"2023-07-21T20:32:45.602772Z","shell.execute_reply":"2023-07-21T20:32:46.530952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"63% of GO term Ids have BPO (Biological Process Ontology) as their aspect,and the 25% for the cco (Cellular component Ontology) ,and last 12% for MF (Molecular function Ontology)","metadata":{}},{"cell_type":"code","source":"#Setup progress bar settings\n\n\nbar = progressbar.ProgressBar(maxval=num_labels, \\\n                             widgets=[progressbar.Bar(\"=\",\"[\",\"]\"),\" \",progressbar.Percentage()])\n\n# empty df of required size for storing the labels,\n\ntrain_size = train_protein_ids.shape[0]\ntrain_labels = np.zeros((train_size, num_labels))\n\n\n# Convert from numpy to pandas series for better handling\nseries_train_protein_ids = pd.Series(train_protein_ids)\n\n\n# Loop through each label\nfor i in range(num_labels):\n    # For each label, fetch the corresponding train_terms data\n    n_train_terms = train_terms_updated[train_terms_updated['term'] ==  labels[i]]\n\n    label_related_proteins = n_train_terms['EntryID'].unique()\n\n\n    # Replace the ith column of train_Y with with that pandas series.\n    train_labels[:,i] =  series_train_protein_ids.isin(label_related_proteins).astype(float)\n    \n    bar.update(i+1)\n    \nbar.finish()\n\n# Convert train_label numpy into pd df\nlabels_df = pd.DataFrame(data=train_labels, columns=labels)\n\nprint(labels_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T20:32:49.026886Z","iopub.execute_input":"2023-07-21T20:32:49.027336Z","iopub.status.idle":"2023-07-21T20:53:17.095668Z","shell.execute_reply.started":"2023-07-21T20:32:49.027298Z","shell.execute_reply":"2023-07-21T20:53:17.094211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The final df label_df is composed of 1500 columns and 142246 entries. We can see all 1500 dim of the dataset by printing","metadata":{}},{"cell_type":"code","source":"labels_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T14:50:03.883255Z","iopub.execute_input":"2023-07-21T14:50:03.883736Z","iopub.status.idle":"2023-07-21T14:50:03.927955Z","shell.execute_reply.started":"2023-07-21T14:50:03.883702Z","shell.execute_reply":"2023-07-21T14:50:03.927057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training our model using the train_df\nwe are gonna use **Tensorflow**","metadata":{}},{"cell_type":"code","source":"# initialize our model and add the required layers to it with certain parameters \nmodel = tf.keras.Sequential([\ntf.keras.layers.BatchNormalization(input_shape=[train_df.shape[1]]),\n    tf.keras.layers.Dense(units=512,activation=\"relu\"),\n    tf.keras.layers.Dense(units=512,activation=\"relu\"),\n    tf.keras.layers.Dense(units=512,activation=\"relu\"),\n    tf.keras.layers.Dense(units=num_labels,activation=\"sigmoid\")\n\n])\n\n# Compile our model\n\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n             loss=\"binary_crossentropy\",\n             metrics=[\"binary_accuracy\",tf.keras.metrics.AUC()]\n             )\n\n# fit the model on our training data\nhistory = model.fit(train_df,labels_df,batch_size=5000,epochs=5)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T20:54:33.772454Z","iopub.execute_input":"2023-07-21T20:54:33.772895Z","iopub.status.idle":"2023-07-21T20:56:55.95625Z","shell.execute_reply.started":"2023-07-21T20:54:33.772863Z","shell.execute_reply":"2023-07-21T20:56:55.955091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize and plot the loss and the accuracy for each epoch","metadata":{}},{"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, [\"binary_accuracy\"]].plot(title=\"Accuracy\")","metadata":{"execution":{"iopub.status.busy":"2023-07-21T21:02:02.737507Z","iopub.execute_input":"2023-07-21T21:02:02.737931Z","iopub.status.idle":"2023-07-21T21:02:03.588341Z","shell.execute_reply.started":"2023-07-21T21:02:02.737899Z","shell.execute_reply":"2023-07-21T21:02:03.587022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predictions **Submission** ","metadata":{}},{"cell_type":"code","source":"test_embeddings = np.load(\"/kaggle/input/t5embeds/test_embeds.npy\")\n\n# convert from npy array to pd dataframe\ncolumn_num = test_embeddings.shape[1]\ntest_df = pd.DataFrame(test_embeddings, columns=[\"Column_\" + str(i) for i in range(1,column_num+1)])\n\n\ntest_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-21T21:02:06.563298Z","iopub.execute_input":"2023-07-21T21:02:06.563691Z","iopub.status.idle":"2023-07-21T21:02:18.266485Z","shell.execute_reply.started":"2023-07-21T21:02:06.563661Z","shell.execute_reply":"2023-07-21T21:02:18.2653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict using our model\npredictions = model.predict(test_df)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T21:02:22.250443Z","iopub.execute_input":"2023-07-21T21:02:22.250916Z","iopub.status.idle":"2023-07-21T21:02:49.974123Z","shell.execute_reply.started":"2023-07-21T21:02:22.250874Z","shell.execute_reply":"2023-07-21T21:02:49.972675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submissions = pd.DataFrame(columns=[\"Protein ID\",\"GO Term ID\",\"Prediction\"])\n\ntest_protein_ids = np.load(\"/kaggle/input/t5embeds/test_ids.npy\")\n\nlistt = []\n\nfor j in list(test_protein_ids):\n    listt += [ j] * predictions.shape[1]\n    \n\nsubmissions[\"Protein ID\"] = listt\nsubmissions[\"GO Term Id\"] = labels * predictions.shape[0]\nsubmissions[\"Prediction\"] = predictions.ravel()\n\nsubmissions.to_csv(\"submission.tsv\",header=False,index=False,sep=\"\\t\")\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-21T21:02:53.345527Z","iopub.execute_input":"2023-07-21T21:02:53.345949Z","iopub.status.idle":"2023-07-21T21:21:45.586809Z","shell.execute_reply.started":"2023-07-21T21:02:53.345916Z","shell.execute_reply":"2023-07-21T21:21:45.585604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submissions.info()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T15:25:22.465895Z","iopub.execute_input":"2023-07-21T15:25:22.466409Z","iopub.status.idle":"2023-07-21T15:25:22.483057Z","shell.execute_reply.started":"2023-07-21T15:25:22.466367Z","shell.execute_reply":"2023-07-21T15:25:22.48179Z"},"trusted":true},"execution_count":null,"outputs":[]}]}