{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":116062,"databundleVersionId":14084779,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":41875,"databundleVersionId":5521661,"sourceType":"competition"}],"dockerImageVersionId":30474,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## DIAMOND","metadata":{}},{"cell_type":"markdown","source":"DIAMOND is a sequence aligner for protein and translated DNA searches, designed for high performance analysis of big sequence data. ","metadata":{}},{"cell_type":"markdown","source":"https://github.com/bbuchfink/diamond","metadata":{}},{"cell_type":"markdown","source":"### Download diamond","metadata":{}},{"cell_type":"code","source":"!wget http://github.com/bbuchfink/diamond/releases/download/v2.1.15/diamond-linux64.tar.gz\n!tar xzf diamond-linux64.tar.gz\n!rm diamond-linux64.tar.gz","metadata":{"execution":{"iopub.status.busy":"2023-05-03T11:44:05.806246Z","iopub.execute_input":"2023-05-03T11:44:05.806624Z","iopub.status.idle":"2023-05-03T11:44:12.641272Z","shell.execute_reply.started":"2023-05-03T11:44:05.806587Z","shell.execute_reply":"2023-05-03T11:44:12.640043Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from subprocess import Popen, PIPE","metadata":{"execution":{"iopub.status.busy":"2023-05-03T11:44:12.646537Z","iopub.execute_input":"2023-05-03T11:44:12.646959Z","iopub.status.idle":"2023-05-03T11:44:12.654081Z","shell.execute_reply.started":"2023-05-03T11:44:12.646913Z","shell.execute_reply":"2023-05-03T11:44:12.652968Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"db_name='train_db'\noutfile_name='matches.tsv'\nk=16 ","metadata":{"execution":{"iopub.status.busy":"2023-05-03T11:44:12.657159Z","iopub.execute_input":"2023-05-03T11:44:12.657833Z","iopub.status.idle":"2023-05-03T11:44:12.696671Z","shell.execute_reply.started":"2023-05-03T11:44:12.657793Z","shell.execute_reply":"2023-05-03T11:44:12.695656Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Create a database file from training data","metadata":{}},{"cell_type":"code","source":"p = Popen(['./diamond', 'makedb', \n           '--in', '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta',\n            '-d', db_name], stdin=PIPE, stdout=PIPE)\nstdout, stderr = p.communicate()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T11:44:12.698682Z","iopub.execute_input":"2023-05-03T11:44:12.699759Z","iopub.status.idle":"2023-05-03T11:44:17.69176Z","shell.execute_reply.started":"2023-05-03T11:44:12.69972Z","shell.execute_reply":"2023-05-03T11:44:17.690948Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Run a blastp-like search for test set","metadata":{}},{"cell_type":"code","source":"import time","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"time0 = time.time() \np = Popen(['./diamond', 'blastp', '-d', db_name,\n           '-q', '/kaggle/input/cafa-6-protein-function-prediction/Test/testsuperset.fasta',\n            '-o', outfile_name, '--max-target-seqs', str(k), '--quiet'], stdin=PIPE, stdout=PIPE)\nstdout, stderr = p.communicate()\nprint(f'Execution time: {time.time()-time0}s')","metadata":{"execution":{"iopub.status.busy":"2023-05-03T11:44:17.692655Z","iopub.execute_input":"2023-05-03T11:44:17.692996Z","iopub.status.idle":"2023-05-03T11:48:21.098015Z","shell.execute_reply.started":"2023-05-03T11:44:17.692951Z","shell.execute_reply":"2023-05-03T11:48:21.097029Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Tuning of alignment options is welcome (https://github.com/bbuchfink/diamond/wiki/3.-Command-line-options)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-05-03T11:48:21.098974Z","iopub.execute_input":"2023-05-03T11:48:21.09925Z","iopub.status.idle":"2023-05-03T11:48:21.103773Z","shell.execute_reply.started":"2023-05-03T11:48:21.099226Z","shell.execute_reply":"2023-05-03T11:48:21.102938Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"matches=pd.read_csv(outfile_name, sep='\\t', header=None, \n                    names=['qseqid', 'sseqid', 'pident', 'length', 'mismatch', \n                           'gapopen', 'qstart', 'qend', 'sstart','send', 'evalue', 'bitscore'])\nmatches['qseqid']=matches['qseqid'].apply(lambda x: x.split('\\\\t')[0])\nmatches.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T11:48:21.104707Z","iopub.execute_input":"2023-05-03T11:48:21.105003Z","iopub.status.idle":"2023-05-03T11:48:23.277645Z","shell.execute_reply.started":"2023-05-03T11:48:21.104979Z","shell.execute_reply":"2023-05-03T11:48:23.276586Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Get GO terms from similar sequences","metadata":{}},{"cell_type":"markdown","source":"For each target sequence of each query I find its GO terms. Then for each query I define probability of a certain term as a fraction of target sequences with this term.","metadata":{}},{"cell_type":"markdown","source":"Taking into account other alignment parameters is welcome","metadata":{}},{"cell_type":"code","source":"train_terms=pd.read_csv('/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv', sep='\\t')\narrayofgos=train_terms.groupby('EntryID').term.apply(lambda x: np.array(x))\nmatches['terms']=arrayofgos[matches.sseqid.values].values\nmatches['ntargets']=matches.groupby(['qseqid']).qseqid.count()[matches['qseqid'].values].values","metadata":{"execution":{"iopub.status.busy":"2023-05-03T11:51:06.56415Z","iopub.execute_input":"2023-05-03T11:51:06.564521Z","iopub.status.idle":"2023-05-03T11:51:14.377232Z","shell.execute_reply.started":"2023-05-03T11:51:06.564492Z","shell.execute_reply":"2023-05-03T11:51:14.376288Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df=matches[['qseqid','terms', 'ntargets']].explode('terms').reset_index()\ntest_df['ntargets']=1/test_df['ntargets']\ntest_df=test_df.groupby(['qseqid', 'terms']).sum().round(3).reset_index()\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T12:00:29.22776Z","iopub.execute_input":"2023-05-03T12:00:29.22868Z","iopub.status.idle":"2023-05-03T12:01:05.014007Z","shell.execute_reply.started":"2023-05-03T12:00:29.228638Z","shell.execute_reply":"2023-05-03T12:01:05.012868Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Save data","metadata":{}},{"cell_type":"code","source":"test_df[['qseqid','terms','ntargets']].to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-05-03T12:01:20.418274Z","iopub.execute_input":"2023-05-03T12:01:20.418711Z","iopub.status.idle":"2023-05-03T12:02:15.474567Z","shell.execute_reply.started":"2023-05-03T12:01:20.418679Z","shell.execute_reply":"2023-05-03T12:02:15.473503Z"},"trusted":true},"outputs":[],"execution_count":null}]}