{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.neighbors import KNeighborsClassifier\nimport progressbar","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-15T06:43:56.271583Z","iopub.execute_input":"2023-06-15T06:43:56.272037Z","iopub.status.idle":"2023-06-15T06:44:08.853036Z","shell.execute_reply.started":"2023-06-15T06:43:56.272006Z","shell.execute_reply":"2023-06-15T06:44:08.851945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\",sep=\"\\t\")\nprint(train_terms.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T06:44:08.855501Z","iopub.execute_input":"2023-06-15T06:44:08.85625Z","iopub.status.idle":"2023-06-15T06:44:12.884959Z","shell.execute_reply.started":"2023-06-15T06:44:08.856213Z","shell.execute_reply":"2023-06-15T06:44:12.883768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T06:44:12.886654Z","iopub.execute_input":"2023-06-15T06:44:12.887055Z","iopub.status.idle":"2023-06-15T06:44:12.923115Z","shell.execute_reply.started":"2023-06-15T06:44:12.887026Z","shell.execute_reply":"2023-06-15T06:44:12.921749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select first 1500 values for plotting\nplot_df = train_terms['term'].value_counts().iloc[:100]\n\nfigure, axis = plt.subplots(1, 1, figsize=(12, 6))\n\nbp = sns.barplot(ax=axis, x=np.array(plot_df.index), y=plot_df.values)\nbp.set_xticklabels(bp.get_xticklabels(), rotation=90, size = 6)\naxis.set_title('Top 100 frequent GO term IDs')\nbp.set_xlabel(\"GO term IDs\", fontsize = 12)\nbp.set_ylabel(\"Count\", fontsize = 12)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T06:44:12.925706Z","iopub.execute_input":"2023-06-15T06:44:12.926462Z","iopub.status.idle":"2023-06-15T06:44:15.347653Z","shell.execute_reply.started":"2023-06-15T06:44:12.926417Z","shell.execute_reply":"2023-06-15T06:44:15.346349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the limit for label\nnum_of_labels = 1500\n\n# Take value counts in descending order and fetch first 1500 `GO term ID` as labels\nlabels = train_terms['term'].value_counts().index[:num_of_labels].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T06:44:15.349612Z","iopub.execute_input":"2023-06-15T06:44:15.350105Z","iopub.status.idle":"2023-06-15T06:44:16.386479Z","shell.execute_reply.started":"2023-06-15T06:44:15.350059Z","shell.execute_reply":"2023-06-15T06:44:16.385128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fetch the train_terms data for the relevant labels only\ntrain_terms_updated = train_terms.loc[train_terms['term'].isin(labels)]","metadata":{"execution":{"iopub.status.busy":"2023-06-15T06:44:16.388154Z","iopub.execute_input":"2023-06-15T06:44:16.388613Z","iopub.status.idle":"2023-06-15T06:44:17.203383Z","shell.execute_reply.started":"2023-06-15T06:44:16.388569Z","shell.execute_reply":"2023-06-15T06:44:17.202204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pie_df = train_terms_updated['aspect'].value_counts()\npalette_color = sns.color_palette('bright')\nplt.pie(pie_df.values, labels=np.array(pie_df.index), colors=palette_color, autopct='%.0f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T06:44:17.205062Z","iopub.execute_input":"2023-06-15T06:44:17.205412Z","iopub.status.idle":"2023-06-15T06:44:18.071557Z","shell.execute_reply.started":"2023-06-15T06:44:17.205384Z","shell.execute_reply":"2023-06-15T06:44:18.069871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_protein_ids = np.load('/kaggle/input/t5embeds/train_ids.npy')\nprint(train_protein_ids.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T06:44:18.07433Z","iopub.execute_input":"2023-06-15T06:44:18.075456Z","iopub.status.idle":"2023-06-15T06:44:18.155583Z","shell.execute_reply.started":"2023-06-15T06:44:18.075394Z","shell.execute_reply":"2023-06-15T06:44:18.15434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_embeddings = np.load('/kaggle/input/t5embeds/train_embeds.npy')\n\n# Now lets convert embeddings numpy array(train_embeddings) into pandas dataframe.\ncolumn_num = train_embeddings.shape[1]\ntrain_df = pd.DataFrame(train_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\nprint(train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T06:44:18.157435Z","iopub.execute_input":"2023-06-15T06:44:18.157836Z","iopub.status.idle":"2023-06-15T06:44:31.479824Z","shell.execute_reply.started":"2023-06-15T06:44:18.157803Z","shell.execute_reply":"2023-06-15T06:44:31.478617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setup progressbar settings.\n# This is strictly for aesthetic.\nbar = progressbar.ProgressBar(maxval=num_of_labels, \\\n    widgets=[progressbar.Bar('=', '[', ']'), ' ', progressbar.Percentage()])\n\n# Create an empty dataframe of required size for storing the labels,\n# i.e, train_size x num_of_labels (142246 x 1500)\ntrain_size = train_protein_ids.shape[0] # len(X)\ntrain_labels = np.zeros((train_size ,num_of_labels))\n\n# Convert from numpy to pandas series for better handling\nseries_train_protein_ids = pd.Series(train_protein_ids)\n\n# Loop through each label\nfor i in range(num_of_labels):\n    # For each label, fetch the corresponding train_terms data\n    n_train_terms = train_terms_updated[train_terms_updated['term'] ==  labels[i]]\n    \n    # Fetch all the unique EntryId aka proteins related to the current label(GO term ID)\n    label_related_proteins = n_train_terms['EntryID'].unique()\n    \n    # In the series_train_protein_ids pandas series, if a protein is related\n    # to the current label, then mark it as 1, else 0.\n    # Replace the ith column of train_Y with with that pandas series.\n    train_labels[:,i] =  series_train_protein_ids.isin(label_related_proteins).astype(float)\n    \n    # Progress bar percentage increase\n    bar.update(i+1)\n\n# Notify the end of progress bar \nbar.finish()\n\n# Convert train_Y numpy into pandas dataframe\nlabels_df = pd.DataFrame(data = train_labels, columns = labels)\nprint(labels_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T06:44:31.48395Z","iopub.execute_input":"2023-06-15T06:44:31.484607Z","iopub.status.idle":"2023-06-15T07:04:47.329715Z","shell.execute_reply.started":"2023-06-15T06:44:31.484565Z","shell.execute_reply":"2023-06-15T07:04:47.328547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T07:04:47.331355Z","iopub.execute_input":"2023-06-15T07:04:47.332445Z","iopub.status.idle":"2023-06-15T07:04:47.366231Z","shell.execute_reply.started":"2023-06-15T07:04:47.332401Z","shell.execute_reply":"2023-06-15T07:04:47.364883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_embeddings = np.load('/kaggle/input/t5embeds/test_embeds.npy')\n\ncolumn_num = test_embeddings.shape[1]\ntest_df = pd.DataFrame(test_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\nprint(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T07:19:32.863755Z","iopub.execute_input":"2023-06-15T07:19:32.864348Z","iopub.status.idle":"2023-06-15T07:19:33.468262Z","shell.execute_reply.started":"2023-06-15T07:19:32.864313Z","shell.execute_reply":"2023-06-15T07:19:33.466795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"knn_clf=KNeighborsClassifier()\nknn_clf.fit(train_df,labels_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_raw_outputs_with_verbose(classifier, X):\n    n_samples = len(X)\n    raw_outputs = []\n\n    print(\"Getting raw outputs:\")\n    for i, x in enumerate(X):\n        progress = (i + 1) / n_samples * 100\n        print(f\"\\rProgress: {progress:.2f}%\", end=\"\")\n        raw_output = classifier.predict_proba([x])[0]\n        raw_outputs.append(raw_output)\n\n    print(\"\\nRaw outputs obtained.\")\n    return raw_outputs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predictions =  knn_clf.predict(test_df)\nraw_outputs = get_raw_outputs_with_verbose(test_df)\nraw_outputs.to_csv('prob_pred.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T06:19:56.583075Z","iopub.execute_input":"2023-06-15T06:19:56.583501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = pd.read_csv('/kaggle/input/predictions/predictions.csv', on_bad_lines='skip')","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:07:35.781649Z","iopub.execute_input":"2023-06-15T05:07:35.782387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ypred_num = predictions.shape[1]\nypred_df = pd.DataFrame(predictions, columns = [\"Column_\" + str(i) for i in range(1, ypred_num+1)])\nprint(ypred_df.shape)\n#print(ypred_df)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:10:22.481085Z","iopub.execute_input":"2023-06-15T05:10:22.481508Z","iopub.status.idle":"2023-06-15T05:10:22.489743Z","shell.execute_reply.started":"2023-06-15T05:10:22.481481Z","shell.execute_reply":"2023-06-15T05:10:22.488538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ypred_df.to_csv('predictions.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T20:25:32.895322Z","iopub.execute_input":"2023-06-14T20:25:32.895787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-14T20:24:42.229943Z","iopub.execute_input":"2023-06-14T20:24:42.230344Z","iopub.status.idle":"2023-06-14T20:24:42.238478Z","shell.execute_reply.started":"2023-06-14T20:24:42.230315Z","shell.execute_reply":"2023-06-14T20:24:42.237222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_small = predictions.iloc[0:1000,0:10000]\npredictions_small.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:40:12.072268Z","iopub.execute_input":"2023-06-15T05:40:12.072679Z","iopub.status.idle":"2023-06-15T05:40:12.081521Z","shell.execute_reply.started":"2023-06-15T05:40:12.072647Z","shell.execute_reply":"2023-06-15T05:40:12.080246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_protein_ids = np.load('/kaggle/input/t5embeds/test_ids.npy')\ntest_protein_ids","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:10:37.713836Z","iopub.execute_input":"2023-06-15T05:10:37.714286Z","iopub.status.idle":"2023-06-15T05:10:37.768653Z","shell.execute_reply.started":"2023-06-15T05:10:37.71424Z","shell.execute_reply":"2023-06-15T05:10:37.767553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])\ndf_submission","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:39:18.115528Z","iopub.execute_input":"2023-06-15T05:39:18.115945Z","iopub.status.idle":"2023-06-15T05:39:18.126445Z","shell.execute_reply.started":"2023-06-15T05:39:18.115909Z","shell.execute_reply":"2023-06-15T05:39:18.125399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_small.shape[1]","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:40:24.260976Z","iopub.execute_input":"2023-06-15T05:40:24.261377Z","iopub.status.idle":"2023-06-15T05:40:24.268846Z","shell.execute_reply.started":"2023-06-15T05:40:24.261346Z","shell.execute_reply":"2023-06-15T05:40:24.267779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l = []\nfor k in list(test_protein_ids):\n    l += [k] * predictions_small.shape[1]   ","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:40:27.046836Z","iopub.execute_input":"2023-06-15T05:40:27.047221Z","iopub.status.idle":"2023-06-15T05:40:30.812207Z","shell.execute_reply.started":"2023-06-15T05:40:27.047192Z","shell.execute_reply":"2023-06-15T05:40:30.811103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission['Protein Id'] = l[0:10000]\ndf_submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:42:13.310182Z","iopub.execute_input":"2023-06-15T05:42:13.310626Z","iopub.status.idle":"2023-06-15T05:42:13.324189Z","shell.execute_reply.started":"2023-06-15T05:42:13.310588Z","shell.execute_reply":"2023-06-15T05:42:13.323304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = labels[0:10000]\nlen(labels)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-15T05:42:26.442792Z","iopub.execute_input":"2023-06-15T05:42:26.444109Z","iopub.status.idle":"2023-06-15T05:42:26.451433Z","shell.execute_reply.started":"2023-06-15T05:42:26.444059Z","shell.execute_reply":"2023-06-15T05:42:26.45019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission['GO Term Id'] = pd.Series(labels * predictions_small.shape[0])\ndf_submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:42:31.045598Z","iopub.execute_input":"2023-06-15T05:42:31.046362Z","iopub.status.idle":"2023-06-15T05:42:31.08666Z","shell.execute_reply.started":"2023-06-15T05:42:31.04632Z","shell.execute_reply":"2023-06-15T05:42:31.085848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_small = np.array(predictions_small)","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:42:38.930321Z","iopub.execute_input":"2023-06-15T05:42:38.931013Z","iopub.status.idle":"2023-06-15T05:42:38.971758Z","shell.execute_reply.started":"2023-06-15T05:42:38.930977Z","shell.execute_reply":"2023-06-15T05:42:38.970646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission['Prediction'] = pd.Series(predictions_small.ravel())","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:42:48.53238Z","iopub.execute_input":"2023-06-15T05:42:48.532766Z","iopub.status.idle":"2023-06-15T05:42:48.549031Z","shell.execute_reply.started":"2023-06-15T05:42:48.532734Z","shell.execute_reply":"2023-06-15T05:42:48.548125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = df_submission[0:10000]","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:43:03.877133Z","iopub.execute_input":"2023-06-15T05:43:03.877538Z","iopub.status.idle":"2023-06-15T05:43:03.882807Z","shell.execute_reply.started":"2023-06-15T05:43:03.877505Z","shell.execute_reply":"2023-06-15T05:43:03.881479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:43:06.590051Z","iopub.execute_input":"2023-06-15T05:43:06.590423Z","iopub.status.idle":"2023-06-15T05:43:06.604613Z","shell.execute_reply.started":"2023-06-15T05:43:06.590394Z","shell.execute_reply":"2023-06-15T05:43:06.603212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")\ndf_submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:43:18.19058Z","iopub.execute_input":"2023-06-15T05:43:18.191232Z","iopub.status.idle":"2023-06-15T05:43:18.225602Z","shell.execute_reply.started":"2023-06-15T05:43:18.191196Z","shell.execute_reply":"2023-06-15T05:43:18.22441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/working/submission.tsv\",sep=\"\\t\", header = None)\n\nsub.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:43:34.061711Z","iopub.execute_input":"2023-06-15T05:43:34.064683Z","iopub.status.idle":"2023-06-15T05:43:34.079616Z","shell.execute_reply.started":"2023-06-15T05:43:34.064641Z","shell.execute_reply":"2023-06-15T05:43:34.078657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:18:16.839533Z","iopub.execute_input":"2023-06-15T05:18:16.839941Z","iopub.status.idle":"2023-06-15T05:18:16.846842Z","shell.execute_reply.started":"2023-06-15T05:18:16.83991Z","shell.execute_reply":"2023-06-15T05:18:16.845639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import classification_report, confusion_matrix, accuracy_score\n# result = confusion_matrix(y_test, ypred)\n# print(“Confusion Matrix:”)\n# print(result)\n# result1 = classification_report(y_test, ypred)\n# print(“Classification Report:”,)\n# print (result1)\n# result2 = accuracy_score(y_test,ypred)\n# print(“Accuracy:”,result2)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T14:33:42.974767Z","iopub.execute_input":"2023-05-27T14:33:42.975464Z","iopub.status.idle":"2023-05-27T14:33:42.989335Z","shell.execute_reply.started":"2023-05-27T14:33:42.975432Z","shell.execute_reply":"2023-05-27T14:33:42.988135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Reference: https://www.kaggle.com/code/alexandervc/baseline-multilabel-to-multitarget-binary\n\n# df_submission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])\n# test_protein_ids = np.load('/kaggle/input/t5embeds/test_ids.npy')\nl = []\nfor k in list(test_protein_ids):\n    l += [ k] * predictions_small.shape[1]   \n\ndf_submission['Protein Id'] = l[0:500]\ndf_submission['GO Term Id'] = labels * predictions_small.shape[0]\ndf_submission['Prediction'] = predictions_small.ravel()\ndf_submission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:19:34.737601Z","iopub.execute_input":"2023-06-15T05:19:34.738024Z","iopub.status.idle":"2023-06-15T05:19:36.52522Z","shell.execute_reply.started":"2023-06-15T05:19:34.737989Z","shell.execute_reply":"2023-06-15T05:19:36.523506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission","metadata":{"execution":{"iopub.status.busy":"2023-06-15T05:13:15.307754Z","iopub.execute_input":"2023-06-15T05:13:15.308168Z","iopub.status.idle":"2023-06-15T05:13:15.322043Z","shell.execute_reply.started":"2023-06-15T05:13:15.308134Z","shell.execute_reply":"2023-06-15T05:13:15.320703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-06-07T04:48:31.247656Z","iopub.execute_input":"2023-06-07T04:48:31.248063Z","iopub.status.idle":"2023-06-07T04:48:31.253051Z","shell.execute_reply.started":"2023-06-07T04:48:31.248035Z","shell.execute_reply":"2023-06-07T04:48:31.252173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}