{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nfrom IPython.display import clear_output\nimport re\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-10T14:05:13.512036Z","iopub.execute_input":"2023-05-10T14:05:13.51267Z","iopub.status.idle":"2023-05-10T14:05:13.562459Z","shell.execute_reply.started":"2023-05-10T14:05:13.512636Z","shell.execute_reply":"2023-05-10T14:05:13.561248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!wget http://github.com/bbuchfink/diamond/releases/download/v2.1.6/diamond-linux64.tar.gz\n!tar xzf diamond-linux64.tar.gz\n!rm diamond-linux64.tar.gz","metadata":{"execution":{"iopub.status.busy":"2023-05-10T14:05:42.69388Z","iopub.execute_input":"2023-05-10T14:05:42.69427Z","iopub.status.idle":"2023-05-10T14:05:54.672031Z","shell.execute_reply.started":"2023-05-10T14:05:42.694239Z","shell.execute_reply":"2023-05-10T14:05:54.670685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"```\n!git clone https://github.com/bio-ontology-research-group/deepgo\n!python deepgo/predict_all.py -i /kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta\n```","metadata":{"execution":{"iopub.status.busy":"2023-05-02T16:56:04.228504Z","iopub.execute_input":"2023-05-02T16:56:04.228983Z","iopub.status.idle":"2023-05-02T17:00:24.252716Z","shell.execute_reply.started":"2023-05-02T16:56:04.228942Z","shell.execute_reply":"2023-05-02T17:00:24.251511Z"}}},{"cell_type":"code","source":"# from https://github.com/bio-ontology-research-group/deepgo/blob/master/predict_all.py\n\nfrom keras.models import load_model, model_from_json\nfrom subprocess import Popen, PIPE\nimport time\nimport json\n\ndef is_ok(seq):\n    for c in set(['U', 'O', 'B', 'Z', 'J', 'X', '*']):\n        seq=seq.replace(c,'G')\n    return seq\n\n\ndef read_fasta(filename, chunk_size):\n    seqs = list()\n    info = list()\n    seq = ''\n    inf = ''\n    with open(filename,'r') as f:\n        for line in f:\n            line = line.strip()\n            if line.startswith('>'):\n                if seq != '':\n                    seq=is_ok(seq)\n                    seqs.append(seq)\n                    info.append(inf)\n                    if len(info) == chunk_size:\n                        yield (info, seqs)\n                        seqs = list()\n                        info = list()   \n                    seq = ''\n                inf = line[1:].split()[0]\n            else:\n                seq += line\n        seqs.append(seq)\n        info.append(inf)\n    yield (info, seqs)\n\ndef get_data(sequences, prot_ids):\n    n = len(sequences)\n    data = np.zeros((n, 1000), dtype=np.float32)\n    embeds = np.zeros((n, 256), dtype=np.float32)\n\n    if prot_ids is None:\n        p = Popen(['./diamond', 'blastp', '-d', '/kaggle/input/deepgo-files/embeddings',\n                   '--max-target-seqs', '1', '--min-score', '60',\n                   '--outfmt', '6', 'qseqid', 'sseqid'], stdin=PIPE, stdout=PIPE)\n        for i in range(n):\n            p.stdin.write(bytes('>' + str(i) + '\\n' + sequences[i] + '\\n', encoding='utf-8'))\n        p.stdin.close()\n\n        prot_ids = {}\n        if p.wait() == 0:\n            for line in p.stdout:\n                it = line.decode('utf-8').strip().split('\\t')\n                if len(it) == 2:\n                    prot_ids[it[1]] = int(it[0])\n\n    prots = embed_df[embed_df['accessions'].isin(list(prot_ids.keys()))]\n    for i, row in prots.iterrows():\n        embeds[prot_ids[row['accessions']], :] = row['embeddings']\n\n    for i in range(len(sequences)):\n        seq = sequences[i]\n        for j in range(min(MAXLEN, len(seq)) - gram_len + 1):\n            data[i, j] = vocab[seq[j: (j + gram_len)]]\n    return [data, embeds]\n\n\ndef predict(data, model, model_name, functions, threshold, batch_size):\n    n = data[0].shape[0]\n    result = list()\n    for i in range(n):\n        result.append(list())\n    predictions = model.predict(\n        data, batch_size=batch_size, verbose=1)\n    for i in range(n):\n        pred = (predictions[i] >= threshold).astype('int32')\n        for j in range(len(functions)):\n            if pred[j] == 1:\n                result[i].append(model_name + '_' + functions[j] + '|' + '%.2f' % predictions[i][j])\n    return result\n\n\ndef predict_functions(sequences, prot_ids, batch_size, threshold):\n    print('Predictions started')\n    start_time = time.time()\n    data = get_data(sequences, prot_ids)\n    result = list()\n    n = len(sequences)\n    for i in range(n):\n        result.append([])\n    for i in range(len(models)):\n        model, functions = models[i]\n        print('Running predictions for model %s' % funcs[i])\n        res = predict(data, model, funcs[i], functions, threshold, batch_size)\n        for j in range(n):\n            result[j] += res[j]\n    print(('Predictions time: {}'.format(time.time() - start_time)))\n    return result\n","metadata":{"execution":{"iopub.status.busy":"2023-05-10T16:11:45.005043Z","iopub.execute_input":"2023-05-10T16:11:45.00548Z","iopub.status.idle":"2023-05-10T16:11:45.027685Z","shell.execute_reply.started":"2023-05-10T16:11:45.005445Z","shell.execute_reply":"2023-05-10T16:11:45.026599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#function to provide compatibility between latest tensorflow version and model configuration file from tf==0.12.0\ndef make_compat(s,l=[('\"input_dtype\": [^\\s\\}]+,*',''),\n                            ('\"W_constraint\": [^\\s\\}]+,*',''),\n                            ('\"W_regularizer\": [^\\s\\}]+,*',''),\n                            ('\"b_constraint\": [^\\s\\}]+,*',''),\n                            ('\"b_regularizer\": [^\\s\\}]+,*',''),\n                            ('\"border_mode\": [^\\s\\}]+,*',''),\n                            ('\"dropout\": [^\\s\\}]+,*',''),\n                            ('\"init\": [^\\s\\}]+,*',''),\n                            ('\"subsample_length\": [^\\s\\}]+,*',''),\n                            ('\"input_length\": [^\\s\\}]+,*',''),\n                            ('nb_filter','filters'),\n                            ('filter_length','kernel_size'),\n                            ('bias','use_bias'),\n                            ('stride','strides'),\n                            ('pool_length','pool_size'),\n                            ('\\, *\\}','}')\n                           ]):\n    for a,b in l:\n        s=re.sub(a,b,s)\n    s=json.loads(s)\n    for arg in s[\"config\"][\"layers\"]:\n        if arg[\"class_name\"]=='Merge':\n            if arg[\"config\"]['mode']=='max':\n                arg[\"class_name\"]=\"Maximum\"\n                arg[\"config\"]={'name': arg[\"config\"]['name']}\n            elif arg[\"config\"]['mode']=='concat':\n                arg[\"class_name\"]=\"Concatenate\"\n                arg[\"config\"]={'name': arg[\"config\"]['name'],'axis': arg[\"config\"]['concat_axis']}\n        elif arg[\"class_name\"]=='Dense':\n            arg[\"config\"]['units']=arg[\"config\"].pop('output_dim')\n        \n    return json.dumps(s)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-10T16:09:06.078361Z","iopub.execute_input":"2023-05-10T16:09:06.079366Z","iopub.status.idle":"2023-05-10T16:09:06.088864Z","shell.execute_reply.started":"2023-05-10T16:09:06.079326Z","shell.execute_reply":"2023-05-10T16:09:06.088033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"in_file='/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta'\nchunk_size=1000\nout_file='results.tsv'\nthreshold=0.3\nbatch_size=1\n\nprot_ids = None\nmapping = None\n\nMAXLEN = 1002\n\nfuncs = ['cc', 'mf', 'bp']","metadata":{"execution":{"iopub.status.busy":"2023-05-10T16:32:02.253932Z","iopub.execute_input":"2023-05-10T16:32:02.255094Z","iopub.status.idle":"2023-05-10T16:32:02.261282Z","shell.execute_reply.started":"2023-05-10T16:32:02.255051Z","shell.execute_reply":"2023-05-10T16:32:02.260112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = list()\n\n# init models\n\nfor onto in funcs:\n    \n    #model = load_model('/kaggle/input/deepgo-files/model_%s.h5' % onto)\n    with open('/kaggle/input/deepgo-files/model_%s_config.json' % onto) as f:\n        model=model_from_json(make_compat(f.read()))\n    model.load_weights('/kaggle/input/deepgo-files/model_%s_weights.h5' % onto)\n    \n    df = pd.read_pickle('/kaggle/input/deepgo-files/%s.pkl' % onto)\n    functions = df['functions']\n    models.append((model, functions))\n    print('Model %s initialized.' % onto)\n    \nngram_df = pd.read_pickle('/kaggle/input/deepgo-files/ngrams.pkl')\nembed_df = pd.read_pickle('/kaggle/input/deepgo-files/graph_new_embeddings.pkl')\nvocab = {}\n\nfor key, gram in enumerate(ngram_df['ngrams']):\n    vocab[gram] = key + 1\n    gram_len = len(ngram_df['ngrams'][0])\n\nprint(('Gram length:', gram_len))\nprint(('Vocabulary size:', len(vocab)))\n   \n","metadata":{"execution":{"iopub.status.busy":"2023-05-10T16:01:30.572364Z","iopub.execute_input":"2023-05-10T16:01:30.57281Z","iopub.status.idle":"2023-05-10T16:06:23.023202Z","shell.execute_reply.started":"2023-05-10T16:01:30.572775Z","shell.execute_reply":"2023-05-10T16:06:23.02192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for ids, sequences in read_fasta(in_file, chunk_size):\n    results = predict_functions(sequences, prot_ids, batch_size, threshold)\n    \n    df=pd.DataFrame(index=ids, data=np.array(results))\n    df=df.explode(0).reset_index()\n    df['term']=df[0].apply(lambda x: x[3:].split('|')[0])\n    df['pred']=df[0].apply(lambda x: float(x.split('|')[1]))\n    df[['index','term','pred']].to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\",mode='a')\n    \n    clear_output()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-10T16:32:09.876014Z","iopub.execute_input":"2023-05-10T16:32:09.876806Z","iopub.status.idle":"2023-05-10T16:33:07.475945Z","shell.execute_reply.started":"2023-05-10T16:32:09.876768Z","shell.execute_reply":"2023-05-10T16:33:07.474226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('submission.tsv') as f:\n    for i in range(10):\n        print(f.readline().strip())","metadata":{"execution":{"iopub.status.busy":"2023-05-10T16:30:29.776068Z","iopub.execute_input":"2023-05-10T16:30:29.776515Z","iopub.status.idle":"2023-05-10T16:30:29.783204Z","shell.execute_reply.started":"2023-05-10T16:30:29.77648Z","shell.execute_reply":"2023-05-10T16:30:29.782119Z"},"trusted":true},"execution_count":null,"outputs":[]}]}