{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Let's use BLAST to give us an edge","metadata":{}},{"cell_type":"markdown","source":"BLAST stands for Basic Local Alignment Search Tool. It is a standard tool biologists and bioinformaticians use to infer functional and evolutionary relationships between sequences. I have created a k-nn classifier based on BLAST and shared it via github: \nhttps://github.com/SamusRam/ProFun\n\nThis notebook shows how to use the init version of [my GitHub library](https://github.com/SamusRam/ProFun) to improve on top of [the currently best public notebook](https://www.kaggle.com/code/mtinti/sprof-predictions) by [@MT](https://www.kaggle.com/mtinti).\n\nI computed the predictions offline on a server with 56 CPUs in ~3 hours.","metadata":{}},{"cell_type":"code","source":"!wget https://ftp.ncbi.nlm.nih.gov/blast/executables/LATEST/ncbi-blast-2.14.0+-x64-linux.tar.gz","metadata":{"execution":{"iopub.status.busy":"2023-06-16T11:27:26.47714Z","iopub.execute_input":"2023-06-16T11:27:26.477894Z","iopub.status.idle":"2023-06-16T11:27:30.715518Z","shell.execute_reply.started":"2023-06-16T11:27:26.477847Z","shell.execute_reply":"2023-06-16T11:27:30.713523Z"},"_kg_hide-input":false,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar zxvpf ncbi-blast-2.14.0+-x64-linux.tar.gz","metadata":{"execution":{"iopub.status.busy":"2023-06-16T11:27:48.969378Z","iopub.execute_input":"2023-06-16T11:27:48.969859Z","iopub.status.idle":"2023-06-16T11:27:58.563991Z","shell.execute_reply.started":"2023-06-16T11:27:48.96981Z","shell.execute_reply":"2023-06-16T11:27:58.562266Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp /kaggle/working/ncbi-blast-2.14.0+/bin/* /opt/conda/bin","metadata":{"execution":{"iopub.status.busy":"2023-06-16T11:28:05.327593Z","iopub.execute_input":"2023-06-16T11:28:05.328149Z","iopub.status.idle":"2023-06-16T11:28:08.462905Z","shell.execute_reply.started":"2023-06-16T11:28:05.328099Z","shell.execute_reply":"2023-06-16T11:28:08.461154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install git+https://github.com/SamusRam/ProFun.git","metadata":{"execution":{"iopub.status.busy":"2023-06-16T11:28:08.46554Z","iopub.execute_input":"2023-06-16T11:28:08.466047Z","iopub.status.idle":"2023-06-16T11:28:25.509515Z","shell.execute_reply.started":"2023-06-16T11:28:08.466Z","shell.execute_reply":"2023-06-16T11:28:25.508279Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\nfrom tqdm.auto import tqdm, trange\nfrom Bio import SeqIO\nimport numpy as np\n\nfrom profun.models import BlastMatching, BlastConfig\nfrom profun.utils.project_info import ExperimentInfo","metadata":{"execution":{"iopub.status.busy":"2023-06-16T11:29:24.231156Z","iopub.execute_input":"2023-06-16T11:29:24.231665Z","iopub.status.idle":"2023-06-16T11:29:24.239567Z","shell.execute_reply.started":"2023-06-16T11:29:24.231608Z","shell.execute_reply":"2023-06-16T11:29:24.23811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Obtaining train data","metadata":{}},{"cell_type":"code","source":"data_root = Path('/kaggle/input/cafa-5-protein-function-prediction/')\ntrain_terms = pd.read_csv(data_root/\"Train/train_terms.tsv\",sep=\"\\t\")\n\nids = []\nseqs = []\nwith open(data_root/\"Train/train_sequences.fasta\") as handle:\n    for record in SeqIO.parse(handle, \"fasta\"):\n        ids.append(record.id)\n        seqs.append(str(record.seq))\ntrain_seqs_df = pd.DataFrame({'EntryID': ids, 'Seq': seqs})\ntrain_df_long = train_terms.merge(train_seqs_df, on='EntryID')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-16T11:29:25.16862Z","iopub.execute_input":"2023-06-16T11:29:25.169112Z","iopub.status.idle":"2023-06-16T11:29:32.164438Z","shell.execute_reply.started":"2023-06-16T11:29:25.169068Z","shell.execute_reply":"2023-06-16T11:29:32.163036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Init model","metadata":{"execution":{"iopub.status.busy":"2023-06-11T20:25:23.042425Z","iopub.execute_input":"2023-06-11T20:25:23.042957Z","iopub.status.idle":"2023-06-11T20:25:28.55935Z","shell.execute_reply.started":"2023-06-11T20:25:23.042919Z","shell.execute_reply":"2023-06-11T20:25:28.558633Z"}}},{"cell_type":"code","source":"experiment_info = ExperimentInfo(validation_schema='public_lb', \n                                 model_type='blast', model_version='1nn')\n\nconfig = BlastConfig(experiment_info=experiment_info, \n                     id_col_name='EntryID', \n                     target_col_name='term', \n                     seq_col_name='Seq', \n                     class_names=list(train_df_long['term'].unique()), \n                     optimize_hyperparams=False, \n                     n_calls_hyperparams_opt=None,\n                    hyperparam_dimensions=None,\n                    per_class_optimization=None,\n                    class_weights=None,\n                    n_neighbours=5,\n                    e_threshold=0.0001,\n                     n_jobs=100,\n                     pred_batch_size=10\n                    )\n\nblast_model = BlastMatching(config)","metadata":{"execution":{"iopub.status.busy":"2023-06-16T11:29:32.166836Z","iopub.execute_input":"2023-06-16T11:29:32.167214Z","iopub.status.idle":"2023-06-16T11:29:32.767309Z","shell.execute_reply.started":"2023-06-16T11:29:32.167175Z","shell.execute_reply":"2023-06-16T11:29:32.766054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train model","metadata":{}},{"cell_type":"code","source":"blast_model.fit(train_df_long)","metadata":{"execution":{"iopub.status.busy":"2023-06-16T11:28:31.942826Z","iopub.execute_input":"2023-06-16T11:28:31.943333Z","iopub.status.idle":"2023-06-16T11:28:37.326441Z","shell.execute_reply.started":"2023-06-16T11:28:31.943281Z","shell.execute_reply":"2023-06-16T11:28:37.32487Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predict on test\nIt's an illustration","metadata":{}},{"cell_type":"code","source":"ids = []\nseqs = []\nwith open(data_root/\"Test (Targets)/testsuperset.fasta\") as handle:\n    for record in SeqIO.parse(handle, \"fasta\"):\n        ids.append(record.id)\n        seqs.append(str(record.seq))\ntest_seqs_df = pd.DataFrame({'EntryID': ids, 'Seq': seqs})\ntest_pred_df = blast_model.predict_proba(test_seqs_df.sample(42).drop_duplicates('EntryID'), return_long_df=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-16T11:12:02.589853Z","iopub.execute_input":"2023-06-16T11:12:02.590293Z","iopub.status.idle":"2023-06-16T11:12:49.810495Z","shell.execute_reply.started":"2023-06-16T11:12:02.590247Z","shell.execute_reply":"2023-06-16T11:12:49.809029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Combining with the best public result","metadata":{}},{"cell_type":"code","source":"test_pred_df_blast = pd.read_csv('/kaggle/input/proteinet-best/blast_submission.tsv',\n    sep='\\t', header=None).drop(0, axis=1)\n\nsubmission_best_public = pd.read_csv('/kaggle/input/sprof-predictions/submission.tsv',\n    sep='\\t', header=None, names=['Id', 'GO term', 'Confidence'])","metadata":{"execution":{"iopub.status.busy":"2023-06-16T11:16:25.162728Z","iopub.execute_input":"2023-06-16T11:16:25.16324Z","iopub.status.idle":"2023-06-16T11:16:37.012837Z","shell.execute_reply.started":"2023-06-16T11:16:25.163176Z","shell.execute_reply":"2023-06-16T11:16:37.011276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submissions_merged = submission_best_public.merge(test_pred_df_blast, left_on=['Id', 'GO term'], \n                                                  right_on=[1, 2], how='outer')\nsubmissions_merged.drop([1, 2], axis=1, inplace=True)\nsubmissions_merged['confidence_combined'] = submissions_merged.apply(lambda row: row['Confidence'] if not np.isnan(row['Confidence']) else row[3], axis=1)\nsubmissions_merged[['Id', 'GO term', 'confidence_combined']].to_csv('submission.tsv',\n    sep='\\t', header=False, index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-16T11:16:37.015209Z","iopub.execute_input":"2023-06-16T11:16:37.016447Z","iopub.status.idle":"2023-06-16T11:25:25.309603Z","shell.execute_reply.started":"2023-06-16T11:16:37.016388Z","shell.execute_reply":"2023-06-16T11:25:25.307643Z"},"trusted":true},"execution_count":null,"outputs":[]}]}