{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"> Code Help from Starter Notebook","metadata":{}},{"cell_type":"code","source":"!pip install seaborn","metadata":{"execution":{"iopub.status.busy":"2023-07-06T03:26:11.765718Z","iopub.execute_input":"2023-07-06T03:26:11.766055Z","iopub.status.idle":"2023-07-06T03:26:17.695988Z","shell.execute_reply.started":"2023-07-06T03:26:11.76603Z","shell.execute_reply":"2023-07-06T03:26:17.695082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install progressbar","metadata":{"execution":{"iopub.status.busy":"2023-07-06T03:26:17.697867Z","iopub.execute_input":"2023-07-06T03:26:17.698153Z","iopub.status.idle":"2023-07-06T03:26:23.531606Z","shell.execute_reply.started":"2023-07-06T03:26:17.698124Z","shell.execute_reply":"2023-07-06T03:26:23.530711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport progressbar\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \nterms = \"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\"","metadata":{"execution":{"iopub.status.busy":"2023-07-06T03:26:23.53306Z","iopub.execute_input":"2023-07-06T03:26:23.533414Z","iopub.status.idle":"2023-07-06T03:26:25.702122Z","shell.execute_reply.started":"2023-07-06T03:26:23.53338Z","shell.execute_reply":"2023-07-06T03:26:25.701269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms = pd.read_csv(terms, sep=\"\\t\")\nprint(train_terms.shape)\n\ntrain_terms.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T03:26:25.703426Z","iopub.execute_input":"2023-07-06T03:26:25.703867Z","iopub.status.idle":"2023-07-06T03:26:29.162153Z","shell.execute_reply.started":"2023-07-06T03:26:25.703835Z","shell.execute_reply":"2023-07-06T03:26:29.161358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_protein_ids = np.load('/kaggle/input/t5-embeds/T5_embeds/train_ids.npy')\nprint(train_protein_ids.shape)\n\ntrain_protein_ids[:5]","metadata":{"execution":{"iopub.status.busy":"2023-07-06T03:26:29.165102Z","iopub.execute_input":"2023-07-06T03:26:29.165424Z","iopub.status.idle":"2023-07-06T03:26:29.213961Z","shell.execute_reply.started":"2023-07-06T03:26:29.165396Z","shell.execute_reply":"2023-07-06T03:26:29.213077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_embeddings = np.load('/kaggle/input/t5-embeds/T5_embeds/train_embeds.npy')\n\n# Now lets convert embeddings numpy array(train_embeddings) into pandas dataframe.\ncolumn_num = train_embeddings.shape[1]\ntrain_df = pd.DataFrame(train_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\nprint(train_df.shape)\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T03:26:29.215069Z","iopub.execute_input":"2023-07-06T03:26:29.215337Z","iopub.status.idle":"2023-07-06T03:26:38.254868Z","shell.execute_reply.started":"2023-07-06T03:26:29.215306Z","shell.execute_reply":"2023-07-06T03:26:38.253926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the limit for label\nnum_of_labels = 1500\n\n# Take value counts in descending order and fetch first 1500 `GO term ID` as labels\nlabels = train_terms['term'].value_counts().index[:num_of_labels].tolist()\n\n# Fetch the train_terms data for the relevant labels only\ntrain_terms_updated = train_terms.loc[train_terms['term'].isin(labels)]","metadata":{"execution":{"iopub.status.busy":"2023-07-06T03:26:38.256058Z","iopub.execute_input":"2023-07-06T03:26:38.256344Z","iopub.status.idle":"2023-07-06T03:26:39.217525Z","shell.execute_reply.started":"2023-07-06T03:26:38.256319Z","shell.execute_reply":"2023-07-06T03:26:39.216695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pie_df = train_terms_updated['aspect'].value_counts()\npalette_color = sns.color_palette('bright')\nplt.pie(pie_df.values, labels=np.array(pie_df.index), colors=palette_color, autopct='%.0f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T03:26:39.21868Z","iopub.execute_input":"2023-07-06T03:26:39.21896Z","iopub.status.idle":"2023-07-06T03:26:39.678496Z","shell.execute_reply.started":"2023-07-06T03:26:39.218935Z","shell.execute_reply":"2023-07-06T03:26:39.676518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setup progressbar settings.\n# This is strictly for aesthetic.\nbar = progressbar.ProgressBar(maxval=num_of_labels, \\\n    widgets=[progressbar.Bar('=', '[', ']'), ' ', progressbar.Percentage()])\n\n# Create an empty dataframe of required size for storing the labels,\n# i.e, train_size x num_of_labels (142246 x 1500)\ntrain_size = train_protein_ids.shape[0] # len(X)\ntrain_labels = np.zeros((train_size ,num_of_labels))\n\n# Convert from numpy to pandas series for better handling\nseries_train_protein_ids = pd.Series(train_protein_ids)\n\nbar.start()\n# Loop through each label\nfor i in range(num_of_labels):\n    # For each label, fetch the corresponding train_terms data\n    n_train_terms = train_terms_updated[train_terms_updated['term'] ==  labels[i]]\n    \n    # Fetch all the unique EntryId aka proteins related to the current label(GO term ID)\n    label_related_proteins = n_train_terms['EntryID'].unique()\n    \n    # In the series_train_protein_ids pandas series, if a protein is related\n    # to the current label, then mark it as 1, else 0.\n    # Replace the ith column of train_Y with with that pandas series.\n    train_labels[:,i] =  series_train_protein_ids.isin(label_related_proteins).astype(float)\n    \n    # Progress bar percentage increase\n    bar.update(i+1)\n\n# Notify the end of progress bar \nbar.finish()\n\n# Convert train_Y numpy into pandas dataframe\nlabels_df = pd.DataFrame(data = train_labels, columns = labels)\nprint(labels_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T03:26:39.681169Z","iopub.execute_input":"2023-07-06T03:26:39.681881Z","iopub.status.idle":"2023-07-06T03:35:26.743938Z","shell.execute_reply.started":"2023-07-06T03:26:39.681803Z","shell.execute_reply":"2023-07-06T03:35:26.743102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T03:35:26.74509Z","iopub.execute_input":"2023-07-06T03:35:26.745375Z","iopub.status.idle":"2023-07-06T03:35:26.770089Z","shell.execute_reply.started":"2023-07-06T03:35:26.745349Z","shell.execute_reply":"2023-07-06T03:35:26.768977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-06T03:35:26.77127Z","iopub.execute_input":"2023-07-06T03:35:26.771553Z","iopub.status.idle":"2023-07-06T03:35:26.783085Z","shell.execute_reply.started":"2023-07-06T03:35:26.771527Z","shell.execute_reply":"2023-07-06T03:35:26.782292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T03:35:26.784223Z","iopub.execute_input":"2023-07-06T03:35:26.784558Z","iopub.status.idle":"2023-07-06T03:35:26.826335Z","shell.execute_reply.started":"2023-07-06T03:35:26.784531Z","shell.execute_reply":"2023-07-06T03:35:26.825603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-07-06T03:35:26.827309Z","iopub.execute_input":"2023-07-06T03:35:26.8277Z","iopub.status.idle":"2023-07-06T03:36:05.813707Z","shell.execute_reply.started":"2023-07-06T03:35:26.827673Z","shell.execute_reply":"2023-07-06T03:36:05.81271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train, x_valid, y_train, y_valid = train_test_split(train_df, labels_df, shuffle=True, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T04:47:33.64379Z","iopub.execute_input":"2023-07-06T04:47:33.645053Z","iopub.status.idle":"2023-07-06T04:47:34.797952Z","shell.execute_reply.started":"2023-07-06T04:47:33.645003Z","shell.execute_reply":"2023-07-06T04:47:34.796486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_SHAPE = [train_df.shape[1]]\nBATCH_SIZE = 6000\n\nmodel = tf.keras.Sequential([\n    tf.keras.layers.BatchNormalization(input_shape=INPUT_SHAPE),    \n    tf.keras.layers.Dense(units=1320, activation='relu'),\n    tf.keras.layers.Dropout(0.455, input_shape=(2,)),\n    tf.keras.layers.Dense(units=1320, activation='relu'),\n    tf.keras.layers.Dropout(0.455, input_shape=(2,)),\n    tf.keras.layers.Dense(units=1320, activation='relu'),\n    tf.keras.layers.Dropout(0.455, input_shape=(2,)),\n    tf.keras.layers.Dense(units=1320, activation='relu'),\n    tf.keras.layers.Dropout(0.455, input_shape=(2,)),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(units=num_of_labels,activation='sigmoid')\n])\n\n\n# Compile model\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.003),\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy', tf.keras.metrics.AUC()],\n)\n\nhistory = model.fit(\n    train_df, labels_df,\n    batch_size=BATCH_SIZE,\n    epochs=25,\n    validation_data=(x_valid, y_valid)\n)\n\n#!mkdir -p saved_model\n#model.save('saved_model/model')","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:02:32.894125Z","iopub.execute_input":"2023-07-06T05:02:32.895398Z","iopub.status.idle":"2023-07-06T05:11:34.112367Z","shell.execute_reply.started":"2023-07-06T05:02:32.895347Z","shell.execute_reply":"2023-07-06T05:11:34.110996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"0.01","metadata":{}},{"cell_type":"markdown","source":"0.3187\n\nlr = 0.008\n\nBATCH_SIZE = 665\n\nmodel = tf.keras.Sequential([\n    tf.keras.layers.BatchNormalization(input_shape=INPUT_SHAPE),    \n    tf.keras.layers.Dense(units=60, activation='relu'),\n    \n    tf.keras.layers.Dense(units=num_of_labels, activation='sigmoid')\n])","metadata":{}},{"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, ['binary_accuracy']].plot(title=\"Accuracy\")","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:15:45.529311Z","iopub.execute_input":"2023-07-06T05:15:45.530304Z","iopub.status.idle":"2023-07-06T05:15:46.048927Z","shell.execute_reply.started":"2023-07-06T05:15:45.53026Z","shell.execute_reply":"2023-07-06T05:15:46.047412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_embeddings = np.load('/kaggle/input/t5-embeds/T5_embeds/test_embeds.npy')\n\n# Convert test_embeddings to dataframe\ncolumn_num = test_embeddings.shape[1]\ntest_df = pd.DataFrame(test_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\nprint(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:15:50.501715Z","iopub.execute_input":"2023-07-06T05:15:50.502598Z","iopub.status.idle":"2023-07-06T05:16:00.469568Z","shell.execute_reply.started":"2023-07-06T05:15:50.502551Z","shell.execute_reply":"2023-07-06T05:16:00.468408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions =  model.predict(test_df)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:16:00.47122Z","iopub.execute_input":"2023-07-06T05:16:00.471523Z","iopub.status.idle":"2023-07-06T05:16:32.628457Z","shell.execute_reply.started":"2023-07-06T05:16:00.471495Z","shell.execute_reply":"2023-07-06T05:16:32.627011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])\ntest_protein_ids = np.load('/kaggle/input/t5-embeds/T5_embeds/test_ids.npy')\nl = []\nfor k in list(test_protein_ids):\n    l += [ k] * predictions.shape[1]   \n\ndf_submission['Protein Id'] = l\ndf_submission['GO Term Id'] = labels * predictions.shape[0]\ndf_submission['Prediction'] = predictions.ravel()\nprint(df_submission.head())\nprint(df_submission.shape)\n#df_submission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:16:32.630897Z","iopub.execute_input":"2023-07-06T05:16:32.631261Z","iopub.status.idle":"2023-07-06T05:17:21.482938Z","shell.execute_reply.started":"2023-07-06T05:16:32.631229Z","shell.execute_reply":"2023-07-06T05:17:21.481669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-07-06T05:17:21.484753Z","iopub.execute_input":"2023-07-06T05:17:21.485097Z","iopub.status.idle":"2023-07-06T05:26:58.655349Z","shell.execute_reply.started":"2023-07-06T05:17:21.485066Z","shell.execute_reply":"2023-07-06T05:26:58.653714Z"},"trusted":true},"execution_count":null,"outputs":[]}]}