{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-17T03:55:49.713214Z","iopub.execute_input":"2023-07-17T03:55:49.714342Z","iopub.status.idle":"2023-07-17T03:55:49.770955Z","shell.execute_reply.started":"2023-07-17T03:55:49.714291Z","shell.execute_reply":"2023-07-17T03:55:49.76997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## setting the environment","metadata":{}},{"cell_type":"code","source":"!pip install goatools","metadata":{"execution":{"iopub.status.busy":"2023-07-17T03:57:40.430655Z","iopub.execute_input":"2023-07-17T03:57:40.431045Z","iopub.status.idle":"2023-07-17T03:58:26.959972Z","shell.execute_reply.started":"2023-07-17T03:57:40.431011Z","shell.execute_reply":"2023-07-17T03:58:26.958899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport requests\n\n# Required for progressbar widget\nimport progressbar\nfrom goatools.obo_parser import GODag","metadata":{"execution":{"iopub.status.busy":"2023-07-17T03:58:57.679129Z","iopub.execute_input":"2023-07-17T03:58:57.679643Z","iopub.status.idle":"2023-07-17T03:58:57.927948Z","shell.execute_reply.started":"2023-07-17T03:58:57.679604Z","shell.execute_reply":"2023-07-17T03:58:57.926541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## loading the row input data & T5-embedded data","metadata":{}},{"cell_type":"code","source":"train_terms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\",sep=\"\\t\")\nprint(train_terms.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T03:59:00.292124Z","iopub.execute_input":"2023-07-17T03:59:00.292565Z","iopub.status.idle":"2023-07-17T03:59:03.700325Z","shell.execute_reply.started":"2023-07-17T03:59:00.29253Z","shell.execute_reply":"2023-07-17T03:59:03.699327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_protein_ids = np.load('/kaggle/input/t5embeds/train_ids.npy')\nprint(train_protein_ids.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T03:59:03.70198Z","iopub.execute_input":"2023-07-17T03:59:03.702355Z","iopub.status.idle":"2023-07-17T03:59:03.759408Z","shell.execute_reply.started":"2023-07-17T03:59:03.702324Z","shell.execute_reply":"2023-07-17T03:59:03.758236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_embeddings = np.load('/kaggle/input/t5embeds/train_embeds.npy')\n\n# Now lets convert embeddings numpy array(train_embeddings) into pandas dataframe.\ncolumn_num = train_embeddings.shape[1]\ntrain_df = pd.DataFrame(train_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\nprint(train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T03:59:03.760889Z","iopub.execute_input":"2023-07-17T03:59:03.761642Z","iopub.status.idle":"2023-07-17T03:59:15.633256Z","shell.execute_reply.started":"2023-07-17T03:59:03.7616Z","shell.execute_reply":"2023-07-17T03:59:15.632276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## check the content of the training data","metadata":{}},{"cell_type":"code","source":"# Select first 1500 values for plotting\nplot_df = train_terms['term'].value_counts().iloc[:100]\n\nfigure, axis = plt.subplots(1, 1, figsize=(12, 6))\n\nbp = sns.barplot(ax=axis, x=np.array(plot_df.index), y=plot_df.values)\nbp.set_xticklabels(bp.get_xticklabels(), rotation=90, size = 6)\naxis.set_title('Top 100 frequent GO term IDs')\nbp.set_xlabel(\"GO term IDs\", fontsize = 12)\nbp.set_ylabel(\"Count\", fontsize = 12)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T03:59:17.101108Z","iopub.execute_input":"2023-07-17T03:59:17.101545Z","iopub.status.idle":"2023-07-17T03:59:18.779984Z","shell.execute_reply.started":"2023-07-17T03:59:17.101507Z","shell.execute_reply":"2023-07-17T03:59:18.778935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_all_df = train_terms['term'].value_counts()\nplot_all_data = np.array(plot_all_df)\nplt.plot(plot_all_data)\nplt.title(\"Frequency vs GO-term Rank\")\nplt.xlabel(\"GO-term rank\")\nplt.ylabel(\"Frequency\")\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T03:59:19.965959Z","iopub.execute_input":"2023-07-17T03:59:19.966358Z","iopub.status.idle":"2023-07-17T03:59:20.788971Z","shell.execute_reply.started":"2023-07-17T03:59:19.966326Z","shell.execute_reply":"2023-07-17T03:59:20.787857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"only some GO terms frequently appears out of over 30,000 terms\\\nlets have a closer look","metadata":{}},{"cell_type":"code","source":"plt.plot(plot_all_data)\nplt.xlim(0,1000)\nplt.ylim(0,5000)\n#plt.yscale(\"log\")\nplt.grid()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-07-17T03:59:22.075107Z","iopub.execute_input":"2023-07-17T03:59:22.075559Z","iopub.status.idle":"2023-07-17T03:59:22.269938Z","shell.execute_reply.started":"2023-07-17T03:59:22.075525Z","shell.execute_reply":"2023-07-17T03:59:22.268881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"100th term only appears less than 1000 times.","metadata":{}},{"cell_type":"code","source":"cum = np.cumsum(plot_all_data) / np.sum(plot_all_data)\nplt.plot(cum)\nplt.xlim(0,2000)\nplt.ylim(0,1)\n#plt.yscale(\"log\")\nplt.title(\"cumulative ratio\")\nplt.xlabel(\"tern rank\")\nplt.grid()\nprint(cum[500], cum[1000])","metadata":{"execution":{"iopub.status.busy":"2023-07-17T04:36:48.796104Z","iopub.execute_input":"2023-07-17T04:36:48.796717Z","iopub.status.idle":"2023-07-17T04:36:49.109516Z","shell.execute_reply.started":"2023-07-17T04:36:48.796668Z","shell.execute_reply":"2023-07-17T04:36:49.10833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1-500th terms occupy around 67% \\\n1-1000th terms occupy around 77%\\\nnot significant difference","metadata":{}},{"cell_type":"code","source":"\n# slopes = np.diff(plot_all_df) / np.diff(range(len(plot_all_df)))\n\n# # 最大傾きのインデックスを見つける\n# elbow_index = np.argmax(slopes)\n\n# # エルボーの累積度数\n# elbow_cumulative_count = cumulative_counts[elbow_index]","metadata":{"execution":{"iopub.status.busy":"2023-07-17T04:00:08.619917Z","iopub.execute_input":"2023-07-17T04:00:08.620345Z","iopub.status.idle":"2023-07-17T04:00:08.629633Z","shell.execute_reply.started":"2023-07-17T04:00:08.620314Z","shell.execute_reply":"2023-07-17T04:00:08.628547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# elbow_index","metadata":{"execution":{"iopub.status.busy":"2023-07-17T04:00:39.295121Z","iopub.execute_input":"2023-07-17T04:00:39.295518Z","iopub.status.idle":"2023-07-17T04:00:39.302819Z","shell.execute_reply.started":"2023-07-17T04:00:39.295487Z","shell.execute_reply":"2023-07-17T04:00:39.301522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## exert the terms which freqntly appears & analysis","metadata":{}},{"cell_type":"code","source":"# Set the limit for label\nnum_of_labels = 1500\n\n# Take value counts in descending order and fetch first 1500 `GO term ID` as labels\nlabels_1500 = train_terms['term'].value_counts().index[:num_of_labels].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T04:38:07.433861Z","iopub.execute_input":"2023-07-17T04:38:07.434398Z","iopub.status.idle":"2023-07-17T04:38:07.975418Z","shell.execute_reply.started":"2023-07-17T04:38:07.434358Z","shell.execute_reply":"2023-07-17T04:38:07.974383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fetch the train_terms data for the relevant labels only\ntrain_terms_updated_1500 = train_terms.loc[train_terms['term'].isin(labels_1500)]\ntrain_terms_updated_1500.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-17T04:38:10.179005Z","iopub.execute_input":"2023-07-17T04:38:10.179378Z","iopub.status.idle":"2023-07-17T04:38:10.84091Z","shell.execute_reply.started":"2023-07-17T04:38:10.179349Z","shell.execute_reply":"2023-07-17T04:38:10.83986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_1500[:10]","metadata":{"execution":{"iopub.status.busy":"2023-07-17T04:38:11.512929Z","iopub.execute_input":"2023-07-17T04:38:11.513321Z","iopub.status.idle":"2023-07-17T04:38:11.521829Z","shell.execute_reply.started":"2023-07-17T04:38:11.513291Z","shell.execute_reply":"2023-07-17T04:38:11.520659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"have a look at the top-ranked terms","metadata":{}},{"cell_type":"code","source":"top_20_terms = labels_1500[:20]\ntop_20_content = {}\nfor i, term in enumerate(top_20_terms):\n    url = f\"https://www.ebi.ac.uk/QuickGO/services/ontology/go/terms/{term}\"\n    response = requests.get(url)\n    data = response.json()\n    freq = plot_all_df[i]\n    top_20_content[term] = [data[\"results\"][0][\"name\"], data[\"results\"][0][\"aspect\"], freq] \n\n# print(data[\"results\"][0][\"name\"]) # GO termの名前\n# print(data[\"results\"][0][\"definition\"][\"text\"]) # GO termの定義","metadata":{"execution":{"iopub.status.busy":"2023-07-17T04:03:33.171877Z","iopub.execute_input":"2023-07-17T04:03:33.172273Z","iopub.status.idle":"2023-07-17T04:03:55.455844Z","shell.execute_reply.started":"2023-07-17T04:03:33.172242Z","shell.execute_reply":"2023-07-17T04:03:55.454684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_20_content","metadata":{"execution":{"iopub.status.busy":"2023-07-17T04:03:55.459959Z","iopub.execute_input":"2023-07-17T04:03:55.460455Z","iopub.status.idle":"2023-07-17T04:03:55.46893Z","shell.execute_reply.started":"2023-07-17T04:03:55.460421Z","shell.execute_reply":"2023-07-17T04:03:55.467692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"seems like CC + BP + MF >>  num_of_sequences\\\none sequence can belong to several subdomain at the same time\\\nAlso, the destribution of CC, BP and MF are different from the number of each terms(will be discussed later)","metadata":{}},{"cell_type":"markdown","source":"The go term takes the hieralchic relationship.\\\nSo, if a term is assigned to one sequence id, its parent terms shold also be assigned to the same sequence.\\\nCheck wheter the input data is as such ( in other word, does not miss the hieralcal relationship)","metadata":{}},{"cell_type":"code","source":"go_dag = GODag(\"/kaggle/input/cafa-5-protein-function-prediction/Train/go-basic.obo\")\n\n# 親のGO termを取得する関数\ndef get_parent_go_terms(go_term):\n    go_term_obj = go_dag[go_term]\n    parents = go_term_obj.get_all_parents()\n    for parent in parents:\n        #print(go_dag[parent].id)\n        pass\n    return [go_dag[parent].id for parent in parents]","metadata":{"execution":{"iopub.status.busy":"2023-07-17T04:43:23.176958Z","iopub.execute_input":"2023-07-17T04:43:23.177411Z","iopub.status.idle":"2023-07-17T04:43:25.825053Z","shell.execute_reply.started":"2023-07-17T04:43:23.17738Z","shell.execute_reply":"2023-07-17T04:43:25.823776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 例としてのGO term\ngo_term = \"GO:0110165\"\nprint(get_parent_go_terms(go_term))","metadata":{"execution":{"iopub.status.busy":"2023-07-17T04:43:27.959534Z","iopub.execute_input":"2023-07-17T04:43:27.959947Z","iopub.status.idle":"2023-07-17T04:43:27.965213Z","shell.execute_reply.started":"2023-07-17T04:43:27.959912Z","shell.execute_reply":"2023-07-17T04:43:27.964023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_grouped = train_terms.groupby('EntryID')['term'].apply(set).reset_index()\ndf_grouped.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T04:43:28.501638Z","iopub.execute_input":"2023-07-17T04:43:28.502023Z","iopub.status.idle":"2023-07-17T04:43:33.82944Z","shell.execute_reply.started":"2023-07-17T04:43:28.501991Z","shell.execute_reply":"2023-07-17T04:43:33.828639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in df_grouped.iterrows():\n    term_set = row[\"term\"]\n    id = row[\"EntryID\"]\n    for term in term_set:\n        parents = get_parent_go_terms(term)\n        for parent in parents:\n            if parent not in term_set:\n                print(f\"{parent} not in parent_list in {id}\")\n#             else:\n#                 print(f\"{parent} in parent_list in {id} !!!!!!!!!!!!!\")\n#     if index == 10:\n#         break\nelse:\n    print(\"every search finished\")","metadata":{"execution":{"iopub.status.busy":"2023-07-17T05:18:54.713789Z","iopub.execute_input":"2023-07-17T05:18:54.714378Z","iopub.status.idle":"2023-07-17T05:20:06.611909Z","shell.execute_reply.started":"2023-07-17T05:18:54.714325Z","shell.execute_reply":"2023-07-17T05:20:06.610824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No output\\\nIt means that the input data represent the hieralchal feature perfectly","metadata":{}},{"cell_type":"markdown","source":"Lets also have a look at the best public notebook submission.","metadata":{}},{"cell_type":"code","source":"sub_053818_df = pd.read_csv(\"/kaggle/input/cafa-5-053818-pred/submission (3).tsv\",sep=\"\\t\" , header=None,\\\n                           names=[\"ID\", \"term\", \"Conf\"])\nsub_053818_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T05:06:20.321962Z","iopub.execute_input":"2023-07-17T05:06:20.32239Z","iopub.status.idle":"2023-07-17T05:06:25.26469Z","shell.execute_reply.started":"2023-07-17T05:06:20.322352Z","shell.execute_reply":"2023-07-17T05:06:25.263529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_053818_df_grouped = sub_053818_df.groupby('ID')['term'].apply(set).reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T05:08:14.763285Z","iopub.execute_input":"2023-07-17T05:08:14.763663Z","iopub.status.idle":"2023-07-17T05:08:28.717471Z","shell.execute_reply.started":"2023-07-17T05:08:14.763629Z","shell.execute_reply":"2023-07-17T05:08:28.716469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_053818_df_grouped.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-17T05:10:09.182507Z","iopub.execute_input":"2023-07-17T05:10:09.183236Z","iopub.status.idle":"2023-07-17T05:10:09.191212Z","shell.execute_reply.started":"2023-07-17T05:10:09.183187Z","shell.execute_reply":"2023-07-17T05:10:09.190081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_053818_df_grouped.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T05:08:52.394333Z","iopub.execute_input":"2023-07-17T05:08:52.39483Z","iopub.status.idle":"2023-07-17T05:08:52.412018Z","shell.execute_reply.started":"2023-07-17T05:08:52.394788Z","shell.execute_reply":"2023-07-17T05:08:52.411242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in sub_053818_df_grouped.iterrows():\n    term_set = row[\"term\"]\n    id = row[\"ID\"]\n    for term in term_set:\n        parents = get_parent_go_terms(term)\n        for parent in parents:\n            if parent not in term_set:\n                print(f\"{parent} not in parent_list in {id}\")\n            else:\n                print(f\"{parent} in parent_list in {id} !!!!!!!!!!!!!\")\n    if index == 1:\n        break\nelse:\n    print(\"every search finished\")","metadata":{"execution":{"iopub.status.busy":"2023-07-17T05:57:52.779984Z","iopub.execute_input":"2023-07-17T05:57:52.781162Z","iopub.status.idle":"2023-07-17T05:57:52.792271Z","shell.execute_reply.started":"2023-07-17T05:57:52.781115Z","shell.execute_reply":"2023-07-17T05:57:52.791232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## take top 1000, 500, 100 etc terms","metadata":{}},{"cell_type":"code","source":"labels_1000 = train_terms['term'].value_counts().index[:1000].tolist()\n# Fetch the train_terms data for the relevant labels only\ntrain_terms_updated_1000 = train_terms.loc[train_terms['term'].isin(labels_1000)]\ntrain_terms_updated_1000.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_500 = train_terms['term'].value_counts().index[:500].tolist()\n# Fetch the train_terms data for the relevant labels only\ntrain_terms_updated_500 = train_terms.loc[train_terms['term'].isin(labels_500)]\ntrain_terms_updated_500.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_500 = train_terms['term'].value_counts().index[:500].tolist()\n# Fetch the train_terms data for the relevant labels only\ntrain_terms_updated_500 = train_terms.loc[train_terms['term'].isin(labels_500)]\ntrain_terms_updated_500.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_500 = train_terms['term'].value_counts().index[:500].tolist()\n# Fetch the train_terms data for the relevant labels only\ntrain_terms_updated_500 = train_terms.loc[train_terms['term'].isin(labels_500)]\ntrain_terms_updated_500.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_100 = train_terms['term'].value_counts().index[:100].tolist()\n# Fetch the train_terms data for the relevant labels only\ntrain_terms_updated_100 = train_terms.loc[train_terms['term'].isin(labels_100)]\ntrain_terms_updated_100.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pie_df = train_terms['aspect'].value_counts()\npalette_color = sns.color_palette('bright')\nplt.pie(pie_df.values, labels=np.array(pie_df.index), colors=palette_color, autopct='%.0f%%')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pie_df = train_terms_updated_1000['aspect'].value_counts()\npalette_color = sns.color_palette('bright')\nplt.pie(pie_df.values, labels=np.array(pie_df.index), colors=palette_color, autopct='%.0f%%')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pie_df = train_terms_updated_500['aspect'].value_counts()\npalette_color = sns.color_palette('bright')\nplt.pie(pie_df.values, labels=np.array(pie_df.index), colors=palette_color, autopct='%.0f%%')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pie_df = train_terms_updated_100['aspect'].value_counts()\npalette_color = sns.color_palette('bright')\nplt.pie(pie_df.values, labels=np.array(pie_df.index), colors=palette_color, autopct='%.0f%%')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nnum_of_labels = len(labels_1000)\n\n# Setup progressbar settings.\n# This is strictly for aesthetic.\nbar = progressbar.ProgressBar(maxval=num_of_labels, \\\n    widgets=[progressbar.Bar('=', '[', ']'), ' ', progressbar.Percentage()])\n\n# Create an empty dataframe of required size for storing the labels,\n# i.e, train_size x num_of_labels (142246 x 1500)\ntrain_size = train_protein_ids.shape[0] # len(X)\ntrain_labels = np.zeros((train_size ,num_of_labels))\n\n# Convert from numpy to pandas series for better handling\nseries_train_protein_ids = pd.Series(train_protein_ids)\n\n\n\n\n# Loop through each label\nfor i in range(num_of_labels):\n    # For each label, fetch the corresponding train_terms data\n    n_train_terms = train_terms_updated_1000[train_terms_updated_1000['term'] ==  labels_1000[i]]\n    \n    # Fetch all the unique EntryId aka proteins related to the current label(GO term ID)\n    label_related_proteins = n_train_terms['EntryID'].unique()\n    \n    # In the series_train_protein_ids pandas series, if a protein is related\n    # to the current label, then mark it as 1, else 0.\n    # Replace the ith column of train_Y with with that pandas series.\n    train_labels[:,i] =  series_train_protein_ids.isin(label_related_proteins).astype(float)\n    \n    # Progress bar percentage increase\n    bar.update(i+1)\n\n# Notify the end of progress bar \nbar.finish()\n\n# Convert train_Y numpy into pandas dataframe\nlabels_df = pd.DataFrame(data = train_labels, columns = labels_1000)\nprint(labels_df.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_df = pd.DataFrame(data = train_labels, columns = labels_1000)\nprint(labels_df.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms_updated_1000[\"term\"].unique().shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_1000[:3]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_df_ordered = labels_df[labels_1000]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_df_ordered","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_targets_top1000 = labels_df_ordered.values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/working/","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save(\"train_targets_top1000\", train_targets_top1000)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_top500 = np.load(\"/kaggle/input/train-targets-top500/train_targets_top500.npy\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_top500_df = pd.DataFrame(train_top500)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_top500_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}