{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#This file is used to run the predictions and create Submission file\nresults_path = '/kaggle/input/protein-galvatron-v3-early-stopping'\nfunction = 'BPO' #BPO, CCO, None\n\nif function == 'BPO':\n    function_path = '/kaggle/input/vectorize-go-labels'\nif function == 'CCO':\n    function_path = '/kaggle/input/vectorize-go-labels-cco'\nif function == 'MFO':\n    function_path = '/kaggle/input/vectorize-go-labels-mfo'\n\n    \nMAIN_DIR = \"/kaggle/input/cafa-5-protein-function-prediction\"\nclass config:\n    train_sequences_path = MAIN_DIR  + \"/Train/train_sequences.fasta\"\n    train_labels_path = MAIN_DIR + \"/Train/train_terms.tsv\"\n    test_sequences_path = MAIN_DIR + \"/Test (Targets)/testsuperset.fasta\"\n    \n    #Protein Sequence and GO terms hparams\n    max_seq_length = 1000 #max length of Go sequence\n    #GO_len = 2000 #max number of top GO Term labels for each tree\n    \n# UTILITARIES\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport time\n\n\nimport os\nimport gc\n\n        \nimport polars\n\n\n#load predictions\nvalidate_predictions = np.concatenate([np.load(results_path+'/test_predictions_0.npy'),\n                np.load(results_path+'/test_predictions_1.npy')])\n\nvalidate_ids = np.concatenate([np.load(results_path+'/test_ids_0.npy',allow_pickle=True).astype(str),\n                np.load(results_path+'/test_ids_1.npy',allow_pickle=True).astype(str)])\nconfig.GO_len = validate_predictions.shape[1]\n\n#read the sequences\nfrom Bio import SeqIO\n\nfasta_train = SeqIO.parse(config.train_sequences_path, \"fasta\")\n\n\n# string_list = []\n# id_list = []\n# sequence_len = []\n\n# for item in tqdm(fasta_train):\n#     seq = str(item.seq)\n#     item_id = item.id \n#     if len(seq) < config.max_seq_length:\n#         string_list.append({'seq':seq,'id':item_id})\n#         id_list.append(item_id)\n#         sequence_len.append(len(seq))\n\n#load the one hot vector embeddings of go terms\nlabel_vectors = np.load(function_path+'/label_vectors.npy')[:,:config.GO_len]\ntraining_ids = np.load(function_path+\"/training_ids.npy\", allow_pickle=True)\nlabel_encoder = np.load(function_path+'/go_terms.npy', \n                        allow_pickle=True).astype(str)[:config.GO_len]\n\n\nlabel_encoder = polars.DataFrame([label_encoder, range(len(label_encoder))], schema= ['term', 'index'])\n\n#get the mapping of GO_terms to vectors\n\nlabels = polars.read_csv(MAIN_DIR+\"/Train/train_terms.tsv\", separator = \"\\t\")\nsorted_terms = labels.groupby('term').count().sort(['count'], descending = True)\n\nlabel_encoder = label_encoder.join(sorted_terms, on = 'term', how = 'left')\n\n# label_encoder = label_encoder.with_columns(polars.Series(name=\"index\", values= range(len(label_encoder))))    \n    \nlabel_encoder = label_encoder.join(labels.drop(\"EntryID\").unique(), on = 'term', how = 'left')\n\n#load ia_weights\nia_weights = polars.read_csv(MAIN_DIR+\"/IA.txt\", separator = \"\\t\", has_header = False, new_columns = ['term',\"IA\"])\nlabel_encoder = label_encoder.join(ia_weights, on = 'term', how = 'left')\nia_weights = label_encoder['IA'].to_numpy()\n#############################################\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-25T13:51:20.602735Z","iopub.execute_input":"2023-07-25T13:51:20.603429Z","iopub.status.idle":"2023-07-25T13:51:26.079548Z","shell.execute_reply.started":"2023-07-25T13:51:20.603397Z","shell.execute_reply":"2023-07-25T13:51:26.078692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!wget http://github.com/bbuchfink/diamond/releases/download/v2.1.6/diamond-linux64.tar.gz\n!tar xzf diamond-linux64.tar.gz\n!rm diamond-linux64.tar.gz\n","metadata":{"execution":{"iopub.status.busy":"2023-07-25T14:24:03.15247Z","iopub.execute_input":"2023-07-25T14:24:03.152873Z","iopub.status.idle":"2023-07-25T14:24:05.2873Z","shell.execute_reply.started":"2023-07-25T14:24:03.152842Z","shell.execute_reply":"2023-07-25T14:24:05.286269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"db_name='train_db'\noutfile_name='matches.tsv'\nk=16 \n\nfrom subprocess import Popen, PIPE\np = Popen(['./diamond', 'makedb', \n           '--in', '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta',\n            '-d', db_name], stdin=PIPE, stdout=PIPE)\nstdout, stderr = p.communicate()","metadata":{"execution":{"iopub.status.busy":"2023-07-25T14:40:20.481866Z","iopub.execute_input":"2023-07-25T14:40:20.482382Z","iopub.status.idle":"2023-07-25T14:40:23.753953Z","shell.execute_reply.started":"2023-07-25T14:40:20.482343Z","shell.execute_reply":"2023-07-25T14:40:23.752885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\ntime0 = time.time() \np = Popen(['./diamond', 'blastp', '-d', db_name,\n           '-q', '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta',\n            '-o', outfile_name, '--max-target-seqs', str(k), '--quiet','--ultra-sensitive'], stdin=PIPE, stdout=PIPE)\nstdout, stderr = p.communicate()\nprint(f'Execution time: {time.time()-time0}s')","metadata":{"execution":{"iopub.status.busy":"2023-07-25T14:40:48.647732Z","iopub.execute_input":"2023-07-25T14:40:48.648125Z","iopub.status.idle":"2023-07-25T14:45:57.309077Z","shell.execute_reply.started":"2023-07-25T14:40:48.6481Z","shell.execute_reply":"2023-07-25T14:45:57.308279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\ntime0 = time.time() \np = Popen(['./diamond', 'blastp', '-d', db_name,\n           '-q', '/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta',\n            '-o', 'test_matches.tsv', '--max-target-seqs', str(k), '--quiet','--ultra-sensitive'], stdin=PIPE, stdout=PIPE)\nstdout, stderr = p.communicate()\nprint(f'Execution time: {time.time()-time0}s')","metadata":{"execution":{"iopub.status.busy":"2023-07-25T15:31:57.011933Z","iopub.execute_input":"2023-07-25T15:31:57.012336Z","iopub.status.idle":"2023-07-25T15:35:44.458613Z","shell.execute_reply.started":"2023-07-25T15:31:57.012298Z","shell.execute_reply":"2023-07-25T15:35:44.457421Z"},"trusted":true},"execution_count":null,"outputs":[]}]}