{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-20T09:10:14.278239Z","iopub.execute_input":"2023-09-20T09:10:14.278625Z","iopub.status.idle":"2023-09-20T09:10:14.34506Z","shell.execute_reply.started":"2023-09-20T09:10:14.278597Z","shell.execute_reply":"2023-09-20T09:10:14.343923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\", sep='\\t')\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:10:14.347371Z","iopub.execute_input":"2023-09-20T09:10:14.347722Z","iopub.status.idle":"2023-09-20T09:10:19.503382Z","shell.execute_reply.started":"2023-09-20T09:10:14.347693Z","shell.execute_reply":"2023-09-20T09:10:19.501904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids_data = train_data[train_data['EntryID'] == 'Q06628']","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:10:19.505031Z","iopub.execute_input":"2023-09-20T09:10:19.505511Z","iopub.status.idle":"2023-09-20T09:10:20.502611Z","shell.execute_reply.started":"2023-09-20T09:10:19.505471Z","shell.execute_reply":"2023-09-20T09:10:20.501211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids_data","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:10:20.506323Z","iopub.execute_input":"2023-09-20T09:10:20.50724Z","iopub.status.idle":"2023-09-20T09:10:20.530927Z","shell.execute_reply.started":"2023-09-20T09:10:20.507172Z","shell.execute_reply":"2023-09-20T09:10:20.529499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data.shape)\ntrain_data.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:10:20.533435Z","iopub.execute_input":"2023-09-20T09:10:20.534302Z","iopub.status.idle":"2023-09-20T09:10:22.186155Z","shell.execute_reply.started":"2023-09-20T09:10:20.534242Z","shell.execute_reply":"2023-09-20T09:10:22.184663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"terms = train_data.groupby(['aspect', 'term'])['term'].count().reset_index(name='frequency')\nterms","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:10:22.187621Z","iopub.execute_input":"2023-09-20T09:10:22.187972Z","iopub.status.idle":"2023-09-20T09:10:24.289765Z","shell.execute_reply.started":"2023-09-20T09:10:22.187943Z","shell.execute_reply":"2023-09-20T09:10:24.288278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_aspect = terms.groupby('aspect')['term'].nunique()\nprint(unique_aspect)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:10:24.291869Z","iopub.execute_input":"2023-09-20T09:10:24.292713Z","iopub.status.idle":"2023-09-20T09:10:24.322978Z","shell.execute_reply.started":"2023-09-20T09:10:24.292666Z","shell.execute_reply":"2023-09-20T09:10:24.321454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_term = terms['term'].nunique()\nprint(unique_term)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:10:24.324758Z","iopub.execute_input":"2023-09-20T09:10:24.32511Z","iopub.status.idle":"2023-09-20T09:10:24.345535Z","shell.execute_reply.started":"2023-09-20T09:10:24.325083Z","shell.execute_reply":"2023-09-20T09:10:24.344622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fractions = (terms.groupby('aspect')['term'].nunique() / terms['term'].nunique() * 2000).apply(round)\nprint(fractions)\n\nselected_terms = set()\nfor aspect, number in fractions.items():\n    selection = terms.loc[(terms.aspect == aspect)]\n    selection = selection.nlargest(number, columns='frequency', keep='first')\n    selected_terms.update(selection.term.to_list())\n    \n#selected_terms","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:10:24.346764Z","iopub.execute_input":"2023-09-20T09:10:24.348018Z","iopub.status.idle":"2023-09-20T09:10:24.414706Z","shell.execute_reply.started":"2023-09-20T09:10:24.347984Z","shell.execute_reply":"2023-09-20T09:10:24.413396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\ntqdm.pandas()\n\ndef assign_labels(annotations, selected_terms=selected_terms):\n    \n    intersection = selected_terms.intersection(annotations)\n    labels = np.isin(np.array(list(selected_terms)), np.array(list(intersection)))\n    \n    return list(labels.astype('int'))\n\nannotations = train_data.groupby('EntryID')['term'].apply(set)\nlabels = annotations.progress_apply(assign_labels)\n\nlabels.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:10:24.418456Z","iopub.execute_input":"2023-09-20T09:10:24.418833Z","iopub.status.idle":"2023-09-20T09:14:51.409366Z","shell.execute_reply.started":"2023-09-20T09:10:24.418803Z","shell.execute_reply":"2023-09-20T09:14:51.407896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(labels.shape)\nlabels[:5]","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:14:51.410932Z","iopub.execute_input":"2023-09-20T09:14:51.411272Z","iopub.status.idle":"2023-09-20T09:14:51.42716Z","shell.execute_reply.started":"2023-09-20T09:14:51.411231Z","shell.execute_reply":"2023-09-20T09:14:51.426159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids = np.load('/kaggle/input/t5embeds/train_ids.npy')\n\nx = np.load('/kaggle/input/t5embeds/train_embeds.npy')\ny = np.array(labels[train_ids].to_list())","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:14:51.428209Z","iopub.execute_input":"2023-09-20T09:14:51.428537Z","iopub.status.idle":"2023-09-20T09:15:29.17718Z","shell.execute_reply.started":"2023-09-20T09:14:51.428511Z","shell.execute_reply":"2023-09-20T09:15:29.175697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids[:5]","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:17:23.517966Z","iopub.execute_input":"2023-09-20T09:17:23.518498Z","iopub.status.idle":"2023-09-20T09:17:23.528002Z","shell.execute_reply.started":"2023-09-20T09:17:23.518459Z","shell.execute_reply":"2023-09-20T09:17:23.526543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tids =np.array(['P20536', 'A0A0B4J1F4', 'P54366', 'A0A009IHW8'])\ntids","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:19:15.797171Z","iopub.execute_input":"2023-09-20T09:19:15.797678Z","iopub.status.idle":"2023-09-20T09:19:15.807503Z","shell.execute_reply.started":"2023-09-20T09:19:15.797646Z","shell.execute_reply":"2023-09-20T09:19:15.80576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lbls=labels[tids]\nlbls","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:20:00.810263Z","iopub.execute_input":"2023-09-20T09:20:00.810756Z","iopub.status.idle":"2023-09-20T09:20:00.826959Z","shell.execute_reply.started":"2023-09-20T09:20:00.810723Z","shell.execute_reply":"2023-09-20T09:20:00.825405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x.shape)\nprint(y.shape)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:15:29.178723Z","iopub.execute_input":"2023-09-20T09:15:29.179161Z","iopub.status.idle":"2023-09-20T09:15:29.186549Z","shell.execute_reply.started":"2023-09-20T09:15:29.179121Z","shell.execute_reply":"2023-09-20T09:15:29.185177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x[:5])\nprint(y[:5])","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:15:29.188609Z","iopub.execute_input":"2023-09-20T09:15:29.18905Z","iopub.status.idle":"2023-09-20T09:15:29.208333Z","shell.execute_reply.started":"2023-09-20T09:15:29.189016Z","shell.execute_reply":"2023-09-20T09:15:29.20687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f = open(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\", \"r\")\nline1 = f.readline()\nline2 = f.readline()\nline3 = f.readline()\nprint(line1)\nprint(line2)\nprint(line3)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:15:29.21003Z","iopub.execute_input":"2023-09-20T09:15:29.210462Z","iopub.status.idle":"2023-09-20T09:15:29.237507Z","shell.execute_reply.started":"2023-09-20T09:15:29.210428Z","shell.execute_reply":"2023-09-20T09:15:29.236281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv\", sep='\\t')\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:15:29.239523Z","iopub.execute_input":"2023-09-20T09:15:29.240256Z","iopub.status.idle":"2023-09-20T09:15:29.406371Z","shell.execute_reply.started":"2023-09-20T09:15:29.240193Z","shell.execute_reply":"2023-09-20T09:15:29.405299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset-taxon-list.tsv\", sep='\\t', encoding=\"ISO-8859-1\")\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:15:29.40818Z","iopub.execute_input":"2023-09-20T09:15:29.408585Z","iopub.status.idle":"2023-09-20T09:15:29.439783Z","shell.execute_reply.started":"2023-09-20T09:15:29.408554Z","shell.execute_reply":"2023-09-20T09:15:29.438742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f = open(\"/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta\", \"r\")\nline1 = f.readline()\nline2 = f.readline()\nline3 = f.readline()\nprint(line1)\nprint(line2)\nprint(line3)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:15:29.441881Z","iopub.execute_input":"2023-09-20T09:15:29.442321Z","iopub.status.idle":"2023-09-20T09:15:29.458777Z","shell.execute_reply.started":"2023-09-20T09:15:29.442285Z","shell.execute_reply":"2023-09-20T09:15:29.456689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install obonet","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:15:29.460732Z","iopub.execute_input":"2023-09-20T09:15:29.46131Z","iopub.status.idle":"2023-09-20T09:15:45.188042Z","shell.execute_reply.started":"2023-09-20T09:15:29.461259Z","shell.execute_reply":"2023-09-20T09:15:45.186311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import obonet\ngraph = obonet.read_obo(\"/kaggle/input/cafa-5-protein-function-prediction/Train/go-basic.obo\")\ngraph.nodes['GO:0000324']","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:15:45.190921Z","iopub.execute_input":"2023-09-20T09:15:45.191421Z","iopub.status.idle":"2023-09-20T09:16:10.562544Z","shell.execute_reply.started":"2023-09-20T09:15:45.191375Z","shell.execute_reply":"2023-09-20T09:16:10.561264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/sample_submission.tsv\", sep='\\t')\nsample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T09:16:10.56432Z","iopub.execute_input":"2023-09-20T09:16:10.564704Z","iopub.status.idle":"2023-09-20T09:16:10.946797Z","shell.execute_reply.started":"2023-09-20T09:16:10.564673Z","shell.execute_reply":"2023-09-20T09:16:10.94531Z"},"trusted":true},"execution_count":null,"outputs":[]}]}