{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-18T16:38:04.3978Z","iopub.execute_input":"2023-06-18T16:38:04.39815Z","iopub.status.idle":"2023-06-18T16:38:04.417281Z","shell.execute_reply.started":"2023-06-18T16:38:04.398127Z","shell.execute_reply":"2023-06-18T16:38:04.416338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:38:05.087914Z","iopub.execute_input":"2023-06-18T16:38:05.088296Z","iopub.status.idle":"2023-06-18T16:38:05.093678Z","shell.execute_reply.started":"2023-06-18T16:38:05.088271Z","shell.execute_reply":"2023-06-18T16:38:05.092364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Read and parse files","metadata":{}},{"cell_type":"code","source":"protein_seqs = pd.read_table(\"/kaggle/input/ProteinKG25/ProteinKG25/ProteinKG25/protein_seq.txt\", header = None)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:38:06.261523Z","iopub.execute_input":"2023-06-18T16:38:06.261991Z","iopub.status.idle":"2023-06-18T16:38:08.700567Z","shell.execute_reply.started":"2023-06-18T16:38:06.26196Z","shell.execute_reply":"2023-06-18T16:38:08.699575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cafa_train = pd.read_table(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\", sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:38:08.7022Z","iopub.execute_input":"2023-06-18T16:38:08.702527Z","iopub.status.idle":"2023-06-18T16:38:10.628238Z","shell.execute_reply.started":"2023-06-18T16:38:08.702506Z","shell.execute_reply":"2023-06-18T16:38:10.627294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"go2id = pd.read_table(\"/kaggle/input/ProteinKG25/ProteinKG25/ProteinKG25/go2id.txt\", header=None, sep = \" \")\ngo_type = pd.read_table(\"/kaggle/input/ProteinKG25/ProteinKG25/ProteinKG25/go_type.txt\", header=None, sep = \" \")\nprotein2id = pd.read_table(\"/kaggle/input/ProteinKG25/ProteinKG25/ProteinKG25/protein2id.txt\", header = None, sep = \" \")","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:38:10.629498Z","iopub.execute_input":"2023-06-18T16:38:10.629757Z","iopub.status.idle":"2023-06-18T16:38:11.033763Z","shell.execute_reply.started":"2023-06-18T16:38:10.629728Z","shell.execute_reply":"2023-06-18T16:38:11.032261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"go2id = go2id.merge(go_type, how=\"left\", left_on=1, right_on=go_type.index)\ngo2id.columns = [\"term\", \"goID\", \"aspect\"]\ngo2id.replace({\"Function\": \"MFO\", \"Process\": \"BPO\", \"Component\": \"CCO\"}, inplace=True)\nprotein2id.columns = [\"EntryID\", \"protID\"]","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:38:11.036211Z","iopub.execute_input":"2023-06-18T16:38:11.036577Z","iopub.status.idle":"2023-06-18T16:38:11.086837Z","shell.execute_reply.started":"2023-06-18T16:38:11.036548Z","shell.execute_reply":"2023-06-18T16:38:11.08519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Merge train, text and valid","metadata":{}},{"cell_type":"code","source":"files = [\"/kaggle/input/ProteinKG25/ProteinKG25/ProteinKG25/protein_go_test_triplet.txt\",\n        \"/kaggle/input/ProteinKG25/ProteinKG25/ProteinKG25/protein_go_valid_triplet.txt\",\n        \"/kaggle/input/ProteinKG25/ProteinKG25/ProteinKG25/protein_go_train_triplet.txt\"]\nprot_go_triplets = pd.DataFrame()\nfor i in files:\n    print(i)\n    try:\n        prot_go_triplets = pd.concat([prot_go_triplets, pd.read_table(i, header = None, sep = \" \")], axis = 0)\n    except:\n        prot_go_triplets = pd.concat([prot_go_triplets, pd.read_table(i, header = None, sep = \" \", skiprows = 1)], axis = 0)\n\nprot_go_triplets.columns = [\"protID\", \"relation\", \"goID\"]","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:38:11.088274Z","iopub.execute_input":"2023-06-18T16:38:11.088616Z","iopub.status.idle":"2023-06-18T16:38:13.919529Z","shell.execute_reply.started":"2023-06-18T16:38:11.088587Z","shell.execute_reply":"2023-06-18T16:38:13.918607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Annotate","metadata":{}},{"cell_type":"code","source":"data_terms = prot_go_triplets.merge(go2id, how=\"left\", on=\"goID\")\ndata_terms = data_terms.merge(protein2id, how=\"left\", on=\"protID\")","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:38:13.92086Z","iopub.execute_input":"2023-06-18T16:38:13.92148Z","iopub.status.idle":"2023-06-18T16:38:19.716629Z","shell.execute_reply.started":"2023-06-18T16:38:13.921445Z","shell.execute_reply":"2023-06-18T16:38:19.715509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_terms = data_terms[[\"EntryID\", \"term\", \"aspect\"]]","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:38:19.718227Z","iopub.execute_input":"2023-06-18T16:38:19.718618Z","iopub.status.idle":"2023-06-18T16:38:21.641234Z","shell.execute_reply.started":"2023-06-18T16:38:19.718583Z","shell.execute_reply":"2023-06-18T16:38:21.639821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Get terms in CAFA but not in ProteinKG25","metadata":{}},{"cell_type":"code","source":"set_1 = cafa_train[~cafa_train[\"term\"].isin(data_terms[\"term\"])]\nlen(set_1[\"EntryID\"].unique())","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:38:21.643559Z","iopub.execute_input":"2023-06-18T16:38:21.643901Z","iopub.status.idle":"2023-06-18T16:38:22.701271Z","shell.execute_reply.started":"2023-06-18T16:38:21.643847Z","shell.execute_reply":"2023-06-18T16:38:22.700341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(set_1[\"term\"].unique()))\nprint(len(cafa_train[\"term\"].unique()))","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:39:14.25562Z","iopub.execute_input":"2023-06-18T16:39:14.255977Z","iopub.status.idle":"2023-06-18T16:39:14.71496Z","shell.execute_reply.started":"2023-06-18T16:39:14.255954Z","shell.execute_reply":"2023-06-18T16:39:14.713915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(cafa_train[\"EntryID\"].unique())","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:35:28.215072Z","iopub.execute_input":"2023-06-18T15:35:28.215493Z","iopub.status.idle":"2023-06-18T15:35:28.559895Z","shell.execute_reply.started":"2023-06-18T15:35:28.215461Z","shell.execute_reply":"2023-06-18T15:35:28.558954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create SeqRocords to write in the FASTA file","metadata":{}},{"cell_type":"code","source":"protein2id = protein2id.merge(protein_seqs, how=\"left\", left_on=\"protID\", right_index=True)\nprotein2id.rename({0: \"seq\"}, inplace=True, axis=1)\nprotein2id.index = protein2id[\"EntryID\"]","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:35:46.00081Z","iopub.execute_input":"2023-06-18T15:35:46.001265Z","iopub.status.idle":"2023-06-18T15:35:46.120035Z","shell.execute_reply.started":"2023-06-18T15:35:46.001235Z","shell.execute_reply":"2023-06-18T15:35:46.118774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from Bio import SeqIO\nfrom Bio.Seq import Seq\nfrom Bio.SeqRecord import SeqRecord\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:35:46.775705Z","iopub.execute_input":"2023-06-18T15:35:46.777152Z","iopub.status.idle":"2023-06-18T15:35:46.901896Z","shell.execute_reply.started":"2023-06-18T15:35:46.777091Z","shell.execute_reply":"2023-06-18T15:35:46.901159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seq_list = []\nfor prot in tqdm(data_terms[\"EntryID\"].unique()):\n    seq = Seq(protein2id.loc[prot, \"seq\"])\n    seq_list.append(SeqRecord(seq=seq, id=prot))","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:35:48.925042Z","iopub.execute_input":"2023-06-18T15:35:48.925562Z","iopub.status.idle":"2023-06-18T15:36:03.155003Z","shell.execute_reply.started":"2023-06-18T15:35:48.925532Z","shell.execute_reply":"2023-06-18T15:36:03.153646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cafa_train_seqs = list(SeqIO.parse(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\", \"fasta\"))\nseq_list_set_1 = []\nfor prot in tqdm(cafa_train_seqs):\n    if prot.id in set_1[\"EntryID\"].to_list():\n        seq_list_set_1.append(prot)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:36:58.248622Z","iopub.execute_input":"2023-06-18T15:36:58.249719Z","iopub.status.idle":"2023-06-18T16:26:19.480574Z","shell.execute_reply.started":"2023-06-18T15:36:58.249681Z","shell.execute_reply":"2023-06-18T16:26:19.47957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(seq_list_set_1))","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:26:19.482199Z","iopub.execute_input":"2023-06-18T16:26:19.48243Z","iopub.status.idle":"2023-06-18T16:26:19.488765Z","shell.execute_reply.started":"2023-06-18T16:26:19.482411Z","shell.execute_reply":"2023-06-18T16:26:19.487664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Write to files","metadata":{}},{"cell_type":"code","source":"SeqIO.write(seq_list, \"proteinKG25_sequences.fasta\", \"fasta\")\ndata_terms.to_csv(\"proteinKG25_terms.tsv\", sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:26:19.489905Z","iopub.execute_input":"2023-06-18T16:26:19.49017Z","iopub.status.idle":"2023-06-18T16:26:50.794308Z","shell.execute_reply.started":"2023-06-18T16:26:19.490148Z","shell.execute_reply":"2023-06-18T16:26:50.793005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SeqIO.write(seq_list_set_1, \"cafa5_not_KG25_sequences.fasta\", \"fasta\")\nset_1.to_csv(\"cafa5_not_KG25_terms.tsv\", sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-06-18T16:26:50.796308Z","iopub.execute_input":"2023-06-18T16:26:50.796593Z","iopub.status.idle":"2023-06-18T16:26:53.507998Z","shell.execute_reply.started":"2023-06-18T16:26:50.796572Z","shell.execute_reply":"2023-06-18T16:26:53.506623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}