{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install obonet","metadata":{"execution":{"iopub.status.busy":"2023-05-05T18:52:05.409315Z","iopub.execute_input":"2023-05-05T18:52:05.409718Z","iopub.status.idle":"2023-05-05T18:52:18.476147Z","shell.execute_reply.started":"2023-05-05T18:52:05.409688Z","shell.execute_reply":"2023-05-05T18:52:18.474748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyvis","metadata":{"execution":{"iopub.status.busy":"2023-05-05T18:52:18.478518Z","iopub.execute_input":"2023-05-05T18:52:18.478923Z","iopub.status.idle":"2023-05-05T18:52:29.754672Z","shell.execute_reply.started":"2023-05-05T18:52:18.478885Z","shell.execute_reply":"2023-05-05T18:52:29.753315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport json\nimport time\nfrom PIL import Image\nfrom typing import Dict\nfrom collections import Counter\nimport random\n# import cv2\nimport obonet\nimport networkx\nimport pandas as pd\nimport numpy as np\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatch\nfrom Bio import SeqIO\nfrom pyvis.network import Network","metadata":{"execution":{"iopub.status.busy":"2023-05-05T18:53:08.263458Z","iopub.execute_input":"2023-05-05T18:53:08.264266Z","iopub.status.idle":"2023-05-05T18:53:09.365809Z","shell.execute_reply.started":"2023-05-05T18:53:08.264216Z","shell.execute_reply":"2023-05-05T18:53:09.364725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CAFA5 = '/kaggle/input/cafa-5-protein-function-prediction/'\nTRAIN = '/kaggle/input/cafa-5-protein-function-prediction/Train/'","metadata":{"execution":{"iopub.status.busy":"2023-05-05T18:57:35.020318Z","iopub.execute_input":"2023-05-05T18:57:35.020728Z","iopub.status.idle":"2023-05-05T18:57:35.02634Z","shell.execute_reply.started":"2023-05-05T18:57:35.0207Z","shell.execute_reply":"2023-05-05T18:57:35.025139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    gobasic = TRAIN + 'go-basic.obo'\n    sequences = TRAIN + 'train_sequences.fasta' \n    terms = TRAIN + 'train_terms.tsv'\n    taxonomy = TRAIN + 'train_taxonomy.tsv'\n    ia = CAFA5 + 'IA.txt'\n    \nclass color:\n    PURPLE = '\\033[95m'\n    CYAN = '\\033[96m'\n    DARKCYAN = '\\033[36m'\n    BLUE = '\\033[94m'\n    GREEN = '\\033[92m'\n    YELLOW = '\\033[93m'\n    RED = '\\033[91m'\n    BOLD = '\\033[1m'\n    UNDERLINE = '\\033[4m'\n    END = '\\033[0m'\n    \n#obonet package to load this file\ngraph = obonet.read_obo(CFG.gobasic)\nprint(f\"Number of nodes: {len(graph)}\")\nprint(f\"Number of edges: {graph.number_of_edges()}\")\n\nterm = \"GO:0034655\"\n\n# nucleobase-containing compound catabolic process node properties\n# print(graph.nodes[term])\n# print(type(graph))  # --> <class 'networkx.classes.multidigraph.MultiDiGraph'>\n# print(type(graph.nodes[term]))  # --> <class 'dict'>\n\n# analyze  protein sequences from sequences file with Biopython package\nprint(\"Sequence example:\\n\\n\", next(iter(SeqIO.parse(CFG.sequences, \"fasta\"))))\n#  count the number of sequences\nsequences = SeqIO.parse(CFG.sequences, 'fasta')\nnum_sequences = sum(1 for seq in sequences)\nprint(\"Number of sequences:\", num_sequences)","metadata":{"execution":{"iopub.status.busy":"2023-05-05T19:02:22.453892Z","iopub.execute_input":"2023-05-05T19:02:22.454487Z","iopub.status.idle":"2023-05-05T19:02:32.879247Z","shell.execute_reply.started":"2023-05-05T19:02:22.454441Z","shell.execute_reply":"2023-05-05T19:02:32.877994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot the length distribution of the protein sequences\nsequences = SeqIO.parse(CFG.sequences, 'fasta')\n# get the length of each sequence\n\nlengths = [len(seq) for seq in sequences]\nfig = px.histogram(x=lengths, nbins=1000, color_discrete_sequence=['goldenrod'])\nfig.update_layout(\n    title={\n        'text'   : \"Distribution of protein sequence lengths\",\n        'y'      : 0.95,\n        'x'      : 0.5, \n        'xanchor': 'center',\n        'yanchor': 'top'\n      }, \n    xaxis_title=\"Sequence length\", yaxis_title=\"Count\"\n)\nfig.show() \nnp.percentile(lengths, 99)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-05T19:08:36.257832Z","iopub.execute_input":"2023-05-05T19:08:36.2582Z","iopub.status.idle":"2023-05-05T19:08:37.854328Z","shell.execute_reply.started":"2023-05-05T19:08:36.258171Z","shell.execute_reply":"2023-05-05T19:08:37.853177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate the amino acid composition of each protein sequence\nrecords = SeqIO.parse(CFG.sequences, 'fasta')\n\n# create a list of all amino acids in the sequences\naa_list = [aa for record in records for aa in record.seq]\n# count the frequency of each amino acid\naa_count = Counter(aa_list)\nfig = px.bar(\n    x=list(aa_count.values()), y=list(aa_count.keys()),\n    color_discrete_sequence=['darkslateblue'],\n    orientation='h', height=700\n)\nfig.update_layout(\n    title={\n        'text': \"Amino Acid Composition\",\n        'y': 0.95,\n        'x': 0.5,\n        'xanchor': 'center',\n        'yanchor': 'top'\n    },\n    xaxis_title=\"Frequency\", yaxis_title=\"Amino Acid\"\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-05T19:10:23.00959Z","iopub.execute_input":"2023-05-05T19:10:23.009943Z","iopub.status.idle":"2023-05-05T19:10:33.435515Z","shell.execute_reply.started":"2023-05-05T19:10:23.009912Z","shell.execute_reply":"2023-05-05T19:10:33.434288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the train terms dataframe\ntrain_terms = pd.read_csv(CFG.terms, sep=\"\\t\")\ntrain_terms.head()\ntrain_terms.describe()\n# TODO: Join train_terms_df.EntryID on train_taxonomy_df.EntryID\n# Information accretion\nlimit = 10\n\nwith open(CFG.ia) as f:\n    ia_weights = [x.replace(\"\\n\", \"\").split(\"\\t\") for x in f.readlines()]\n\nia_weights[:limit]","metadata":{"execution":{"iopub.status.busy":"2023-05-05T19:10:33.437367Z","iopub.execute_input":"2023-05-05T19:10:33.43779Z","iopub.status.idle":"2023-05-05T19:10:39.314374Z","shell.execute_reply.started":"2023-05-05T19:10:33.437758Z","shell.execute_reply":"2023-05-05T19:10:39.313157Z"},"trusted":true},"execution_count":null,"outputs":[]}]}