{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Combining datasets\n\n### Here we work towards a method to combine datasets and outputs predictions\n\n#### We will use: SPROF Predictions, ProFun (https://github.com/SamusRam/ProFun) Predictions and QuickGo Annotations","metadata":{}},{"cell_type":"code","source":"#Create a dictionary to assign each go term to the roots (CCO, MFO, BPO)\n\nimport re\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport joblib\nimport pickle\nimport statistics\nfrom tqdm import tqdm\nfrom Bio import SeqIO\nimport gc\n\ndef extract_go_terms_and_branches(file_path):\n    with open(file_path, 'r') as file:\n        content = file.read()\n        # Match each stanza with [Term] in the OBO file\n        stanzas = re.findall(r'\\[Term\\][\\s\\S]*?(?=\\n\\[|$)', content)\n\n    go_terms_dict = {}\n    for stanza in stanzas:\n        # Extract the GO term ID\n        go_id = re.search(r'^id: (GO:\\d+)', stanza, re.MULTILINE)\n        if go_id:\n            go_id = go_id.group(1)\n\n        # Extract the namespace (branch)\n        namespace = re.search(r'^namespace: (\\w+)', stanza, re.MULTILINE)\n        if namespace:\n            namespace = namespace.group(1)\n\n        if go_id and namespace:\n            # Map the branch abbreviation to the corresponding BPO, CCO, or MFO\n            branch_abbr = {'biological_process': 'BPO', 'cellular_component': 'CCO', 'molecular_function': 'MFO'}\n            go_terms_dict[go_id] = branch_abbr[namespace]\n\n    return go_terms_dict\n\nfile_path = '/kaggle/input/cafa-5-protein-function-prediction/Train/go-basic.obo'\ngo_terms_dict = extract_go_terms_and_branches(file_path)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T12:33:14.644082Z","iopub.execute_input":"2023-06-28T12:33:14.644815Z","iopub.status.idle":"2023-06-28T12:33:17.173847Z","shell.execute_reply.started":"2023-06-28T12:33:14.644764Z","shell.execute_reply":"2023-06-28T12:33:17.172838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a class to manage predictions for proteins.\n# The class keeps track of the highest score for each GO (Gene Ontology) term prediction.\n# Note: This assumes scores are comparable, which might not be the case.\n# A ranking-based selection could be more suitable.\n# Each branch outputs a maximum of 35 predictions for each protein after sorting predictions from highest to lowest.\n# There is an option to add a bonus to the score if the term is predicted by multiple methods.\n\nclass ProteinPredictions:\n    # Initialize an empty dictionary to store the predictions\n    def __init__(self):\n        self.predictions = {}\n\n    # Add a prediction to the storage, with optional bonus\n    # Arguments:\n    #   - protein: Identifier for the protein\n    #   - go_term: GO term that is being predicted\n    #   - score: Confidence score of the prediction\n    #   - branch: Branch of the Gene Ontology (e.g., 'CCO', 'MFO', 'BPO')\n    #   - bonus: Optional bonus to be added to the score\n    def add_prediction(self, protein, go_term, score, branch):\n        # If the protein is not already in the storage, initialize its structure\n        if protein not in self.predictions:\n            self.predictions[protein] = {'CCO': {}, 'MFO': {}, 'BPO': {}}        \n        # Convert the score to a float for comparison and calculation\n        score = float(score)\n        # If this GO term has already been predicted for this protein and branch,\n        # add the bonus to the score. Keep the highest score.\n        if go_term in self.predictions[protein][branch]:\n            self.predictions[protein][branch][go_term].append(score)\n        # If this GO term has not been predicted yet, store it with the score\n        else:\n            self.predictions[protein][branch][go_term] = [score]\n            \n    def check_prediction(self, protein, go_term, branch):\n        # Let's replace the prediction with the average value\n        if type(self.predictions[protein][branch][go_term]) is list:\n            self.predictions[protein][branch][go_term] = statistics.mean(self.predictions[protein][branch][go_term])    \n            \n    # Export the stored predictions to a file\n    # Arguments:\n    #   - output_file: File name for the exported predictions\n    #   - top: Number of top predictions to export for each protein and branch\n    def get_predictions(self, output_file='submission.tsv', top=45):\n        # Open the output file\n        with open(output_file, 'w') as f:\n            # Iterate through each protein and its branches\n            for protein, branches in self.predictions.items():\n                # For each branch, sort the GO terms by score in descending order and select the top ones\n                for branch, go_terms in branches.items():\n                    # Sort go_terms by score in descending order and take the top ones\n                    top_go_terms = sorted(go_terms.items(), key=lambda x: x[1], reverse=True)[:top]\n                    # Write each of the top predictions to the file\n                    for go_term, score in top_go_terms:\n                        f.write(f\"{protein}\\t{go_term}\\t{score:.3f}\\n\")\n","metadata":{"execution":{"iopub.status.busy":"2023-06-28T12:33:17.175572Z","iopub.execute_input":"2023-06-28T12:33:17.176826Z","iopub.status.idle":"2023-06-28T12:33:17.194617Z","shell.execute_reply.started":"2023-06-28T12:33:17.176767Z","shell.execute_reply":"2023-06-28T12:33:17.193127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_predictions = ProteinPredictions()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T12:33:17.196195Z","iopub.execute_input":"2023-06-28T12:33:17.196612Z","iopub.status.idle":"2023-06-28T12:33:17.21523Z","shell.execute_reply.started":"2023-06-28T12:33:17.196564Z","shell.execute_reply":"2023-06-28T12:33:17.213955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for l in tqdm(open('/kaggle/input/quick-go-2022-03-02/quickgo.tsv')):\n    item_list = l.split('\\t')\n    temp_id = item_list[1]\n    go=item_list[2].strip()\n    score = float(1)\n    if go in go_terms_dict:\n        root = go_terms_dict[go]\n        #branch = item_list[3].strip()\n        protein_predictions.add_prediction(temp_id, go, score, root)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-28T12:33:17.219138Z","iopub.execute_input":"2023-06-28T12:33:17.22015Z","iopub.status.idle":"2023-06-28T12:33:31.404602Z","shell.execute_reply.started":"2023-06-28T12:33:17.220097Z","shell.execute_reply":"2023-06-28T12:33:31.403314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for l in tqdm(open('/kaggle/input/proteinet-best/blast_submission.tsv')):\n    item_list = l.split('\\t')\n    temp_id = item_list[1]\n    go=item_list[2]\n    score = float(item_list[3].strip())\n    if go in go_terms_dict:\n        root = go_terms_dict[go]\n        #branch = item_list[3].strip()\n        protein_predictions.add_prediction(temp_id, go, score, root)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T12:33:31.405931Z","iopub.execute_input":"2023-06-28T12:33:31.406273Z","iopub.status.idle":"2023-06-28T12:34:26.04259Z","shell.execute_reply.started":"2023-06-28T12:33:31.406241Z","shell.execute_reply":"2023-06-28T12:34:26.041394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for l in tqdm(open('/kaggle/input/sprof-predictions/submission.tsv')):\n    item_list = l.split('\\t')\n    temp_id = item_list[0]\n    go=item_list[1]\n    score = float(item_list[2].strip())\n    if go in go_terms_dict:\n        root = go_terms_dict[go]\n        #branch = item_list[3].strip()\n        protein_predictions.add_prediction(temp_id, go, score, root)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T12:34:26.044258Z","iopub.execute_input":"2023-06-28T12:34:26.0447Z","iopub.status.idle":"2023-06-28T12:34:37.594033Z","shell.execute_reply.started":"2023-06-28T12:34:26.044653Z","shell.execute_reply":"2023-06-28T12:34:37.592761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for l in tqdm(open('/kaggle/input/quick-go-2022-03-02/quickgo.tsv')):\n    item_list = l.split('\\t')\n    temp_id = item_list[1]\n    go=item_list[2].strip()\n    if go in go_terms_dict:\n        root = go_terms_dict[go]\n        #branch = item_list[3].strip()\n        protein_predictions.check_prediction(temp_id, go, root)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T12:34:37.595463Z","iopub.execute_input":"2023-06-28T12:34:37.595835Z","iopub.status.idle":"2023-06-28T12:35:38.412925Z","shell.execute_reply.started":"2023-06-28T12:34:37.595797Z","shell.execute_reply":"2023-06-28T12:35:38.411745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for l in tqdm(open('/kaggle/input/proteinet-best/blast_submission.tsv')):\n    item_list = l.split('\\t')\n    temp_id = item_list[1]\n    go=item_list[2]\n    if go in go_terms_dict:\n        root = go_terms_dict[go]\n        #branch = item_list[3].strip()\n        protein_predictions.check_prediction(temp_id, go, root)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T12:35:38.414558Z","iopub.execute_input":"2023-06-28T12:35:38.415047Z","iopub.status.idle":"2023-06-28T12:41:25.493194Z","shell.execute_reply.started":"2023-06-28T12:35:38.415009Z","shell.execute_reply":"2023-06-28T12:41:25.491787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for l in tqdm(open('/kaggle/input/sprof-predictions/submission.tsv')):\n    item_list = l.split('\\t')\n    temp_id = item_list[0]\n    go=item_list[1]\n    if go in go_terms_dict:\n        root = go_terms_dict[go]\n        #branch = item_list[3].strip()\n        protein_predictions.check_prediction(temp_id, go, root)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T12:41:25.494668Z","iopub.execute_input":"2023-06-28T12:41:25.495055Z","iopub.status.idle":"2023-06-28T12:42:19.529793Z","shell.execute_reply.started":"2023-06-28T12:41:25.495015Z","shell.execute_reply":"2023-06-28T12:42:19.52812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_predictions.get_predictions()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T12:42:19.53338Z","iopub.execute_input":"2023-06-28T12:42:19.534466Z","iopub.status.idle":"2023-06-28T12:42:32.919829Z","shell.execute_reply.started":"2023-06-28T12:42:19.534418Z","shell.execute_reply":"2023-06-28T12:42:32.918567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head -n 200 'submission.tsv'","metadata":{"execution":{"iopub.status.busy":"2023-06-28T12:42:32.921244Z","iopub.execute_input":"2023-06-28T12:42:32.921578Z","iopub.status.idle":"2023-06-28T12:42:34.05896Z","shell.execute_reply.started":"2023-06-28T12:42:32.921546Z","shell.execute_reply":"2023-06-28T12:42:34.057714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}