{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install obonet","metadata":{"execution":{"iopub.status.busy":"2023-06-25T20:40:55.368981Z","iopub.execute_input":"2023-06-25T20:40:55.369453Z","iopub.status.idle":"2023-06-25T20:41:09.593645Z","shell.execute_reply.started":"2023-06-25T20:40:55.369419Z","shell.execute_reply":"2023-06-25T20:41:09.591689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport obonet\nimport contextlib\nimport random\nimport os\nimport seaborn as sns\nfrom tqdm.notebook import tqdm\nfrom scipy.sparse import (\n    csr_matrix,\n    csc_matrix\n)\nfrom functools import reduce\nfrom collections import Counter","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-25T20:41:09.596752Z","iopub.execute_input":"2023-06-25T20:41:09.597217Z","iopub.status.idle":"2023-06-25T20:41:10.614102Z","shell.execute_reply.started":"2023-06-25T20:41:09.597171Z","shell.execute_reply":"2023-06-25T20:41:10.612661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_DIR = '/kaggle/input/cafa-5-protein-function-prediction/'\n\nTRAIN_FILENAME = ROOT_DIR+'Train/train_sequences.fasta'\nTRAIN_TERMS_FILENAME = ROOT_DIR + 'Train/train_terms.tsv'\nTEST_FILENAME = ROOT_DIR+'Test (Targets)/testsuperset.fasta'\n\nNGRAM_SIZE = 8\n\nLEVEL = 0","metadata":{"execution":{"iopub.status.busy":"2023-06-25T20:41:10.615965Z","iopub.execute_input":"2023-06-25T20:41:10.616723Z","iopub.status.idle":"2023-06-25T20:41:10.623654Z","shell.execute_reply.started":"2023-06-25T20:41:10.61668Z","shell.execute_reply":"2023-06-25T20:41:10.622372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nprotein_2_terms = {}\n\ndef collect_terms(row):\n    try:\n        protein_2_terms[row['EntryID']].append(row['term'])\n    except KeyError:\n        protein_2_terms[row['EntryID']] = [row['term']]\n\npd.read_csv(TRAIN_TERMS_FILENAME,sep = '\\t').apply(collect_terms, axis = 1);","metadata":{"execution":{"iopub.status.busy":"2023-06-25T20:41:10.627162Z","iopub.execute_input":"2023-06-25T20:41:10.627618Z","iopub.status.idle":"2023-06-25T20:42:48.178755Z","shell.execute_reply.started":"2023-06-25T20:41:10.62758Z","shell.execute_reply":"2023-06-25T20:42:48.177405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_fasta(filename, report_interval = 1): # yield (protein, desc, seqdata)\n    protein = ''\n    print('Parse fasta:' + filename)\n    t = tqdm(total = os.path.getsize(filename))\n    current_pos = 0\n    report_index = 0\n    with open(filename) as file:\n        line = file.readline()\n        while line:\n            report_index += 1\n            if report_index == report_interval:\n                new_pos = file.tell()\n                t.update(new_pos - current_pos)\n                current_pos = new_pos\n                report_index = 0\n            if not line.startswith(';'):\n                if line.startswith('>'):\n                    if len(protein) > 0:\n                        yield protein, desc, seqdata\n                    protein, desc = line.rstrip('\\n').lstrip('>').replace('\\t', ' ').split(maxsplit = 1)\n                    seqdata = ''\n                else:\n                    seqdata = seqdata + line.rstrip('\\n')  \n                line = file.readline()\n        if len(protein) > 0:\n            yield protein, desc, seqdata  \n        t.update(file.tell() - current_pos)  ","metadata":{"execution":{"iopub.status.busy":"2023-06-25T20:42:48.18035Z","iopub.execute_input":"2023-06-25T20:42:48.180728Z","iopub.status.idle":"2023-06-25T20:42:48.193569Z","shell.execute_reply.started":"2023-06-25T20:42:48.180695Z","shell.execute_reply":"2023-06-25T20:42:48.191778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seq_2_ngram_stat(seqdata, ngram_size):\n    ngram_stat = {}\n    for offset in range(len(seqdata) - ngram_size + 1):\n        seqitem = seqdata[offset: offset + ngram_size]\n        ngram_stat[seqitem] = ngram_stat.get(seqitem, 0) + 1\n    return ngram_stat ","metadata":{"execution":{"iopub.status.busy":"2023-06-25T20:42:48.195433Z","iopub.execute_input":"2023-06-25T20:42:48.195961Z","iopub.status.idle":"2023-06-25T20:42:48.216019Z","shell.execute_reply.started":"2023-06-25T20:42:48.195894Z","shell.execute_reply":"2023-06-25T20:42:48.214273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_train(filename, ngram_size): # sparse dataframe\n    ngram_2_protein_with_counts = {}\n    for protein, _, seqdata in read_fasta(filename, report_interval = 10000):\n        protein_len = len(seqdata) - ngram_size + 1\n        if protein_len > 0:\n            for ngram, count in seq_2_ngram_stat(seqdata, ngram_size).items():\n                try:\n                    ngram_2_protein_with_counts[ngram].append(\n                        (protein, count, protein_len)\n                    )\n                except KeyError:\n                    ngram_2_protein_with_counts[ngram] = [\n                        (protein, count, protein_len)\n                    ]\n    return ngram_2_protein_with_counts","metadata":{"execution":{"iopub.status.busy":"2023-06-25T20:42:48.219147Z","iopub.execute_input":"2023-06-25T20:42:48.219724Z","iopub.status.idle":"2023-06-25T20:42:48.23151Z","shell.execute_reply.started":"2023-06-25T20:42:48.219684Z","shell.execute_reply":"2023-06-25T20:42:48.229893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nngram_2_protein_with_counts = load_train(TRAIN_FILENAME, NGRAM_SIZE)","metadata":{"execution":{"iopub.status.busy":"2023-06-25T20:42:48.233809Z","iopub.execute_input":"2023-06-25T20:42:48.234233Z","iopub.status.idle":"2023-06-25T20:46:42.383608Z","shell.execute_reply.started":"2023-06-25T20:42:48.2342Z","shell.execute_reply":"2023-06-25T20:46:42.382325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ngram_2_term_with_counts(ngram):\n    term_with_counts = {}\n    for protein, count, protein_len in ngram_2_protein_with_counts.get(ngram, []):\n        for term in protein_2_terms[protein]:\n            counts = term_with_counts.get(term, (0,0))\n            counts = (counts[0] + 1, counts[1] + count / protein_len)\n            term_with_counts[term] = counts\n    return term_with_counts","metadata":{"execution":{"iopub.status.busy":"2023-06-25T20:46:42.38606Z","iopub.execute_input":"2023-06-25T20:46:42.386639Z","iopub.status.idle":"2023-06-25T20:46:42.396696Z","shell.execute_reply.started":"2023-06-25T20:46:42.386587Z","shell.execute_reply":"2023-06-25T20:46:42.394952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_biofuncs_for(test_protein, test_seqdata):\n    term_2_res = {}               \n    test_len = len(test_seqdata) - NGRAM_SIZE + 1\n    if test_len <= 0:\n        return {}\n    for test_ngram, test_count in seq_2_ngram_stat(test_seqdata, NGRAM_SIZE).items():\n        for term, counts in ngram_2_term_with_counts(test_ngram).items():\n            term_2_res[term] = term_2_res.get(term, 0) + counts[1] * test_count / test_len\n    return term_2_res   ","metadata":{"execution":{"iopub.status.busy":"2023-06-25T20:46:42.402796Z","iopub.execute_input":"2023-06-25T20:46:42.403196Z","iopub.status.idle":"2023-06-25T20:46:42.413454Z","shell.execute_reply.started":"2023-06-25T20:46:42.403165Z","shell.execute_reply":"2023-06-25T20:46:42.412126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with contextlib.suppress(FileNotFoundError):\n    os.remove('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-25T20:46:42.415127Z","iopub.execute_input":"2023-06-25T20:46:42.41561Z","iopub.status.idle":"2023-06-25T20:46:42.425916Z","shell.execute_reply.started":"2023-06-25T20:46:42.415568Z","shell.execute_reply":"2023-06-25T20:46:42.424772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def flush_buffer(buffer):\n    pd.DataFrame(buffer, columns = ['protein', 'term', 'weight']).to_csv(\n            'submission.tsv', mode = 'a', header=False, index = False, sep = '\\t'\n    ) \n    buffer.clear()    ","metadata":{"execution":{"iopub.status.busy":"2023-06-25T20:46:42.428365Z","iopub.execute_input":"2023-06-25T20:46:42.429037Z","iopub.status.idle":"2023-06-25T20:46:42.441397Z","shell.execute_reply.started":"2023-06-25T20:46:42.428951Z","shell.execute_reply":"2023-06-25T20:46:42.439898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nFLUSH_BUFFER_LEN = 1000\nbuffer = []\n\nfor protein, _, seqdata in read_fasta(TEST_FILENAME, report_interval = 100):\n    for term, res in get_biofuncs_for(protein, seqdata).items():\n        buffer.append([protein, term, res])\n    if len(buffer) > FLUSH_BUFFER_LEN:\n        flush_buffer(buffer)\nif len(buffer) > 0:\n    flush_buffer(buffer)","metadata":{"execution":{"iopub.status.busy":"2023-06-25T20:46:42.442926Z","iopub.execute_input":"2023-06-25T20:46:42.444407Z"},"trusted":true},"execution_count":null,"outputs":[]}]}