{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Let's use Foldseek to give us an edge\n\nI've added a Foldseek-based model to [my libary ProFun](https://github.com/SamusRam/ProFun).","metadata":{}},{"cell_type":"markdown","source":"![](https://raw.githubusercontent.com/steineggerlab/foldseek/master/.github/foldseek.png)\n\n\nFoldseek is a recent tool from the dream teams of amazing [Johannes Söding](https://www.mpinat.mpg.de/642011/cv_soeding) and [Martin Steinegger](https://steineggerlab.com/en/). It \"enables fast and sensitive comparisons of large structure sets\" ([from the Foldseek official github](https://github.com/steineggerlab/foldseek)).\n\"Foldseek aligns the structure of a query protein against a database by describing tertiary amino acid interactions within proteins as sequences over a structural alphabet.\" [[van Kempen, Michel, et al. \"Fast and accurate protein structure search with Foldseek.\" Nature Biotechnology (2023): 1-4.](https://www.nature.com/articles/s41587-023-01773-0)]\n\n\nI have created a k-nn classifier based on Foldseek and shared it via github: \nhttps://github.com/SamusRam/ProFun\n\nThis notebook shows how to use the init version of [my GitHub library](https://github.com/SamusRam/ProFun) to improve on top of [the currently best public notebook](https://www.kaggle.com/code/kirilldubovik/cafa5-tuning-merge-datasets) by [@Kirill Dubovik](https://www.kaggle.com/kirilldubovik), which is a version of the nice [notebook](https://www.kaggle.com/code/mtinti/merge-datasets) by [@MT](https://www.kaggle.com/mtinti).","metadata":{}},{"cell_type":"code","source":"!conda install -c conda-forge -c bioconda foldseek -y","metadata":{"execution":{"iopub.status.busy":"2023-06-29T13:40:49.200527Z","iopub.execute_input":"2023-06-29T13:40:49.201012Z","iopub.status.idle":"2023-06-29T13:42:16.000716Z","shell.execute_reply.started":"2023-06-29T13:40:49.200967Z","shell.execute_reply":"2023-06-29T13:42:15.999143Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install git+https://github.com/SamusRam/ProFun.git","metadata":{"execution":{"iopub.status.busy":"2023-06-29T13:47:32.543506Z","iopub.execute_input":"2023-06-29T13:47:32.543999Z","iopub.status.idle":"2023-06-29T13:47:51.253604Z","shell.execute_reply.started":"2023-06-29T13:47:32.543949Z","shell.execute_reply":"2023-06-29T13:47:51.25205Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\nfrom tqdm.auto import tqdm, trange\nfrom Bio import SeqIO\nimport numpy as np\n\nfrom profun.models import FoldseekMatching, FoldseekConfig\nfrom profun.utils.project_info import ExperimentInfo","metadata":{"execution":{"iopub.status.busy":"2023-06-29T13:47:51.256279Z","iopub.execute_input":"2023-06-29T13:47:51.256769Z","iopub.status.idle":"2023-06-29T13:47:52.379466Z","shell.execute_reply.started":"2023-06-29T13:47:51.256715Z","shell.execute_reply":"2023-06-29T13:47:52.378297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Obtaining train data","metadata":{}},{"cell_type":"code","source":"data_root = Path('/kaggle/input/cafa-5-protein-function-prediction/')\ntrain_terms = pd.read_csv(data_root/\"Train/train_terms.tsv\",sep=\"\\t\")\n\nids = []\nseqs = []\nwith open(data_root/\"Train/train_sequences.fasta\") as handle:\n    for record in SeqIO.parse(handle, \"fasta\"):\n        ids.append(record.id)\n        seqs.append(str(record.seq))\ntrain_seqs_df = pd.DataFrame({'EntryID': ids, 'Seq': seqs})\ntrain_df_long = train_terms.merge(train_seqs_df, on='EntryID')\ntrain_df_long_sample = train_df_long.sample(200)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-29T13:47:52.381806Z","iopub.execute_input":"2023-06-29T13:47:52.383206Z","iopub.status.idle":"2023-06-29T13:48:00.239163Z","shell.execute_reply.started":"2023-06-29T13:47:52.383165Z","shell.execute_reply":"2023-06-29T13:48:00.237741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Init model","metadata":{"execution":{"iopub.status.busy":"2023-06-11T20:25:23.042425Z","iopub.execute_input":"2023-06-11T20:25:23.042957Z","iopub.status.idle":"2023-06-11T20:25:28.55935Z","shell.execute_reply.started":"2023-06-11T20:25:23.042919Z","shell.execute_reply":"2023-06-11T20:25:28.558633Z"}}},{"cell_type":"code","source":"experiment_info = ExperimentInfo(validation_schema='public_lb', \n                                 model_type='foldseek', model_version='5nn')\n\nconfig = FoldseekConfig(experiment_info=experiment_info, \n                        id_col_name='EntryID', \n                        target_col_name='term',\n                        seq_col_name='Seq',\n                        class_names=list(train_df_long_sample['term'].unique()), \n                        optimize_hyperparams=False, \n                        n_calls_hyperparams_opt=None,\n                        hyperparam_dimensions=None,\n                        per_class_optimization=None,\n                        class_weights=None,\n                        n_neighbours=5,\n                        e_threshold=0.0001,\n                        n_jobs=1,\n                        pred_batch_size=10,\n                        local_pdb_storage_path=None #then it stores structures into the working dir\n                    )\n\nmodel = FoldseekMatching(config)","metadata":{"execution":{"iopub.status.busy":"2023-06-29T13:48:00.241566Z","iopub.execute_input":"2023-06-29T13:48:00.241928Z","iopub.status.idle":"2023-06-29T13:48:00.251556Z","shell.execute_reply.started":"2023-06-29T13:48:00.241892Z","shell.execute_reply":"2023-06-29T13:48:00.250217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train model\n\nDuring the training the predicted AlphaFold2 structures are automatically downloaded from [https://alphafold.ebi.ac.uk/https://alphafold.ebi.ac.uk/](https://alphafold.ebi.ac.uk/https://alphafold.ebi.ac.uk/). Any proteins missing from the DB of predicted structures are omitted.","metadata":{}},{"cell_type":"code","source":"model.fit(train_df_long_sample)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-29T13:48:00.252881Z","iopub.execute_input":"2023-06-29T13:48:00.253269Z","iopub.status.idle":"2023-06-29T13:51:07.584015Z","shell.execute_reply.started":"2023-06-29T13:48:00.253232Z","shell.execute_reply":"2023-06-29T13:51:07.580693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predict on test\nIt's an illustration. I computed the predictions for the whole test set offline. Unfortunately, on Kaggle I experience MMseq2 error reported [here](https://github.com/soedinglab/metaeuk/issues/48) (there's an OpenMP-related [check in MMSeq2](https://github.com/soedinglab/metaeuk/blob/1da320a9daa75dce5539442b5674f69951a2fe4f/lib/mmseqs/src/commons/CommandCaller.cpp#L17). If you experience the same error on your local machine, please refer to [this thread](https://github.com/soedinglab/metaeuk/issues/48)).","metadata":{}},{"cell_type":"code","source":"ids = []\nseqs = []\nwith open(data_root/\"Test (Targets)/testsuperset.fasta\") as handle:\n    for record in SeqIO.parse(handle, \"fasta\"):\n        ids.append(record.id)\n        seqs.append(str(record.seq))\ntest_seqs_df = pd.DataFrame({'EntryID': ids, 'Seq': seqs})\n# test_pred_df = model.predict_proba(test_seqs_df.sample(42).drop_duplicates('EntryID'), return_long_df=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-29T14:11:50.920918Z","iopub.execute_input":"2023-06-29T14:11:50.922173Z","iopub.status.idle":"2023-06-29T14:11:53.05568Z","shell.execute_reply.started":"2023-06-29T14:11:50.922119Z","shell.execute_reply":"2023-06-29T14:11:53.054547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Combining with the best public result","metadata":{}},{"cell_type":"code","source":"test_pred_df_foldseek = pd.read_csv('/kaggle/input/foldseek-cafa/foldseek_submission.tsv',\n    sep='\\t', header=None, names=[1, 2, 3])\ntest_pred_df_foldseek = test_pred_df_foldseek[test_pred_df_foldseek[3] > 0.6]\n\nsubmission_best_public = pd.read_csv('/kaggle/input/cafa5-tuning-merge-datasets/submission.tsv',\n    sep='\\t', header=None, names=['Id', 'GO term', 'Confidence'])\n\nsubmissions_merged = submission_best_public.merge(test_pred_df_foldseek, left_on=['Id', 'GO term'], \n                                                  right_on=[1, 2], how='outer')\nsubmissions_merged.drop([1, 2], axis=1, inplace=True)\nsubmissions_merged['confidence_combined'] = submissions_merged.apply(lambda row: row['Confidence'] if not np.isnan(row['Confidence']) else row[3], axis=1)\nsubmissions_merged[['Id', 'GO term', 'confidence_combined']].to_csv('submission.tsv',\n    sep='\\t', header=False, index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-29T14:11:53.41548Z","iopub.execute_input":"2023-06-29T14:11:53.41616Z"},"trusted":true},"execution_count":null,"outputs":[]}]}