{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## CAFA 5 Calculate Sequence Similarity","metadata":{}},{"cell_type":"markdown","source":"Amino acid sequences that are highly homologous often have similarities in structure and function to the proteins encoded by the sequences, and are therefore presumed to have similar functions.\n\nAmino acid sequences greatly affect protein structure and function. The structure of proteins is directly determined by the sequence of amino acids. Also, protein function is structurally determined by its amino acid sequence.","metadata":{}},{"cell_type":"code","source":"!pip install obonet","metadata":{"execution":{"iopub.status.busy":"2023-04-28T08:02:16.490984Z","iopub.execute_input":"2023-04-28T08:02:16.491855Z","iopub.status.idle":"2023-04-28T08:02:30.924827Z","shell.execute_reply.started":"2023-04-28T08:02:16.491795Z","shell.execute_reply":"2023-04-28T08:02:30.923311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom Bio import SeqIO\nfrom Bio import pairwise2\nfrom Bio.Seq import Seq\nimport obonet\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-04-28T08:02:30.927165Z","iopub.execute_input":"2023-04-28T08:02:30.927594Z","iopub.status.idle":"2023-04-28T08:02:31.454029Z","shell.execute_reply.started":"2023-04-28T08:02:30.927532Z","shell.execute_reply":"2023-04-28T08:02:31.452852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EntryID and its sequence info","metadata":{}},{"cell_type":"code","source":"fasta_file = \"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\"\nrecords = SeqIO.parse(fasta_file, \"fasta\")\ndf2 = pd.DataFrame(columns=['sequence', 'EntryID'])\nfor record in records:\n    df2 = df2.append({'sequence': str(record.seq), 'EntryID': record.id}, ignore_index=True)\ndisplay(df2[0:3])","metadata":{"execution":{"iopub.status.busy":"2023-04-28T08:02:31.455363Z","iopub.execute_input":"2023-04-28T08:02:31.455765Z","iopub.status.idle":"2023-04-28T08:16:56.119129Z","shell.execute_reply.started":"2023-04-28T08:02:31.455728Z","shell.execute_reply":"2023-04-28T08:16:56.117802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# How to calculate similarity value between two sequences","metadata":{}},{"cell_type":"code","source":"seq1 = Seq(df2.iloc[0,0])\nseq2 = Seq(df2.iloc[1,0])\n\ndef seq_similarity(seq1,seq2):\n    \n    alignments = pairwise2.align.globalxx(seq1, seq2)\n    best_alignment = alignments[0]\n    aligned_seq1 = best_alignment[0]\n    aligned_seq2 = best_alignment[1]\n    num_matches = 0\n    num_mismatches = 0\n    for i in range(len(aligned_seq1)):\n        if aligned_seq1[i] == aligned_seq2[i]:\n            num_matches += 1\n        else:\n            num_mismatches += 1\n    similarity = num_matches / (num_matches + num_mismatches)\n    \n    return similarity\n\nsimilarity=seq_similarity(seq1,seq2)\nprint(len(seq1))\nprint(seq1)\nprint()\nprint(len(seq2))\nprint(seq2)\nprint()\nprint(\"Similarity：\", similarity)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T08:16:56.123054Z","iopub.execute_input":"2023-04-28T08:16:56.123437Z","iopub.status.idle":"2023-04-28T08:16:56.161069Z","shell.execute_reply.started":"2023-04-28T08:16:56.123403Z","shell.execute_reply":"2023-04-28T08:16:56.15974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}