{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install obonet -q\n!pip install pyvis -q","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-21T22:42:55.782353Z","iopub.execute_input":"2023-05-21T22:42:55.783897Z","iopub.status.idle":"2023-05-21T22:43:21.890394Z","shell.execute_reply.started":"2023-05-21T22:42:55.783837Z","shell.execute_reply":"2023-05-21T22:43:21.88869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport json\nfrom PIL import Image\nfrom typing import Dict\nfrom collections import Counter\n\nimport random\nimport cv2\nimport obonet\nimport networkx\nimport pandas as pd\nimport numpy as np\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatch\nfrom Bio import SeqIO\nfrom pyvis.network import Network","metadata":{"execution":{"iopub.status.busy":"2023-05-21T22:43:21.893776Z","iopub.execute_input":"2023-05-21T22:43:21.894171Z","iopub.status.idle":"2023-05-21T22:43:21.902517Z","shell.execute_reply.started":"2023-05-21T22:43:21.89413Z","shell.execute_reply":"2023-05-21T22:43:21.901489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    train_go_obo_path: str = \"/kaggle/input/cafa-5-protein-function-prediction/Train/go-basic.obo\"\n    train_seq_fasta_path: str = \"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\"\n    train_terms_path: str = \"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\"\n    train_taxonomy_path: str = \"/kaggle/input/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv\"\n    train_ia_path: str = \"/kaggle/input/cafa-5-protein-function-prediction/IA.txt\"","metadata":{"execution":{"iopub.status.busy":"2023-05-21T22:43:21.903507Z","iopub.execute_input":"2023-05-21T22:43:21.904162Z","iopub.status.idle":"2023-05-21T22:43:21.921565Z","shell.execute_reply.started":"2023-05-21T22:43:21.904132Z","shell.execute_reply":"2023-05-21T22:43:21.920593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class color:\n   PURPLE = '\\033[95m'\n   CYAN = '\\033[96m'\n   DARKCYAN = '\\033[36m'\n   BLUE = '\\033[94m'\n   GREEN = '\\033[92m'\n   YELLOW = '\\033[93m'\n   RED = '\\033[91m'\n   BOLD = '\\033[1m'\n   UNDERLINE = '\\033[4m'\n   END = '\\033[0m'","metadata":{"execution":{"iopub.status.busy":"2023-05-21T22:43:21.924607Z","iopub.execute_input":"2023-05-21T22:43:21.925029Z","iopub.status.idle":"2023-05-21T22:43:21.933577Z","shell.execute_reply.started":"2023-05-21T22:43:21.924959Z","shell.execute_reply":"2023-05-21T22:43:21.932423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_dag(graph, term, radius=1):\n    # create smaller subgraph\n    # radius - include all neighbors of distance<=radius from n (increse it to add further parent's branches).\n    ng_graph = networkx.ego_graph(graph, term, radius=radius)\n\n    for n in ng_graph.nodes(data=True):\n        # concatenate label of the node with its attribute\n        n[1][\"label\"] = n[0] + \" \" +n[1][\"name\"]\n\n    nt = Network(directed=True, notebook=True, cdn_resources=\"in_line\")\n    nt.from_nx(ng_graph)\n    return nt.show(\"network.html\")","metadata":{"execution":{"iopub.status.busy":"2023-05-21T22:43:21.93525Z","iopub.execute_input":"2023-05-21T22:43:21.935624Z","iopub.status.idle":"2023-05-21T22:43:21.945683Z","shell.execute_reply.started":"2023-05-21T22:43:21.935592Z","shell.execute_reply":"2023-05-21T22:43:21.944655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"graph = obonet.read_obo(CFG.train_go_obo_path)","metadata":{"execution":{"iopub.status.busy":"2023-05-21T22:43:21.94701Z","iopub.execute_input":"2023-05-21T22:43:21.947828Z","iopub.status.idle":"2023-05-21T22:43:41.295317Z","shell.execute_reply.started":"2023-05-21T22:43:21.947794Z","shell.execute_reply":"2023-05-21T22:43:41.294109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Number of nodes: {len(graph)}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-21T22:43:41.296915Z","iopub.execute_input":"2023-05-21T22:43:41.297383Z","iopub.status.idle":"2023-05-21T22:43:41.303293Z","shell.execute_reply.started":"2023-05-21T22:43:41.297349Z","shell.execute_reply":"2023-05-21T22:43:41.302151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Number of edges: {graph.number_of_edges()}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-21T22:43:41.304759Z","iopub.execute_input":"2023-05-21T22:43:41.305102Z","iopub.status.idle":"2023-05-21T22:43:41.464891Z","shell.execute_reply.started":"2023-05-21T22:43:41.305073Z","shell.execute_reply":"2023-05-21T22:43:41.463503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequences = SeqIO.parse(CFG.train_seq_fasta_path, \"fasta\")\nnum_sequences = sum(1 for seq in sequences)\nprint(num_sequences)","metadata":{"execution":{"iopub.status.busy":"2023-05-21T22:43:41.466507Z","iopub.execute_input":"2023-05-21T22:43:41.46777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequences = SeqIO.parse(CFG.train_seq_fasta_path, \"fasta\")\n\n# get the length of each sequence\nlengths = [len(seq) for seq in sequences]\n\nfig = px.histogram(x=lengths, nbins=1000, color_discrete_sequence=['goldenrod'])\nfig.update_layout(\n    title={\n        'text': \"Distribution of protein sequence lengths\",\n        'y':0.95,\n        'x':0.5,\n        'xanchor': 'center',\n        'yanchor': 'top'\n    },\n    xaxis_title=\"Sequence length\", yaxis_title=\"Count\"\n)\n\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"records = SeqIO.parse(CFG.train_seq_fasta_path, \"fasta\")\n\n# create a list of all amino acids in the sequences\naa_list = [aa for record in records for aa in record.seq]\n\n# count the frequency of each amino acid\naa_count = Counter(aa_list)\n\nfig = px.bar(\n    x=list(aa_count.values()), y=list(aa_count.keys()),\n    color_discrete_sequence=['darkslateblue'],\n    orientation='h', height=700\n)\nfig.update_layout(\n    title={\n        'text': \"Amino Acid Composition\",\n        'y':0.95,\n        'x':0.5,\n        'xanchor': 'center',\n        'yanchor': 'top'\n    },\n    xaxis_title=\"Frequency\", yaxis_title=\"Amino Acid\"\n)\nfig.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"records = SeqIO.parse(CFG.train_seq_fasta_path, \"fasta\")\n\n# create a list of all amino acids in the sequences\naa_list = [aa for record in records for aa in record.seq]\n\n# count the frequency of each amino acid\naa_count = Counter(aa_list)\n\nfig = px.bar(\n    x=list(aa_count.values()), y=list(aa_count.keys()),\n    color_discrete_sequence=['darkslateblue'],\n    orientation='h', height=700\n)\nfig.update_layout(\n    title={\n        'text': \"Amino Acid Composition\",\n        'y':0.95,\n        'x':0.5,\n        'xanchor': 'center',\n        'yanchor': 'top'\n    },\n    xaxis_title=\"Frequency\", yaxis_title=\"Amino Acid\"\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T22:56:16.426859Z","iopub.execute_input":"2023-05-21T22:56:16.427305Z","iopub.status.idle":"2023-05-21T22:56:27.509108Z","shell.execute_reply.started":"2023-05-21T22:56:16.427273Z","shell.execute_reply":"2023-05-21T22:56:27.507904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms_df = pd.read_csv(CFG.train_terms_path, sep=\"\\t\")\ntrain_terms_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T22:59:38.050896Z","iopub.execute_input":"2023-05-21T22:59:38.051405Z","iopub.status.idle":"2023-05-21T22:59:42.390882Z","shell.execute_reply.started":"2023-05-21T22:59:38.051373Z","shell.execute_reply":"2023-05-21T22:59:42.38968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T23:01:36.303617Z","iopub.execute_input":"2023-05-21T23:01:36.30504Z","iopub.status.idle":"2023-05-21T23:01:40.875874Z","shell.execute_reply.started":"2023-05-21T23:01:36.304991Z","shell.execute_reply":"2023-05-21T23:01:40.874689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aspect_counts = train_terms_df.aspect.value_counts()\n\nfig = px.pie(values=aspect_counts.values, names=aspect_counts.index)\nfig.update_traces(textposition='inside', textfont_size=14)\nfig.update_layout(\n    title={\n        'text': \"Pie distribution of aspect values\",\n        'y':0.95,\n        'x':0.5,\n        'xanchor': 'center',\n        'yanchor': 'top'\n    },\n    legend_title_text='Aspect:'\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T23:02:27.328579Z","iopub.execute_input":"2023-05-21T23:02:27.329697Z","iopub.status.idle":"2023-05-21T23:02:28.335567Z","shell.execute_reply.started":"2023-05-21T23:02:27.329621Z","shell.execute_reply":"2023-05-21T23:02:28.334302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_taxonomy_df = pd.read_csv(CFG.train_taxonomy_path, sep=\"\\t\")\ntrain_taxonomy_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T23:05:11.128379Z","iopub.execute_input":"2023-05-21T23:05:11.128899Z","iopub.status.idle":"2023-05-21T23:05:11.277775Z","shell.execute_reply.started":"2023-05-21T23:05:11.128865Z","shell.execute_reply":"2023-05-21T23:05:11.27634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_taxonomy_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T23:05:14.769479Z","iopub.execute_input":"2023-05-21T23:05:14.769966Z","iopub.status.idle":"2023-05-21T23:05:14.797688Z","shell.execute_reply.started":"2023-05-21T23:05:14.769928Z","shell.execute_reply":"2023-05-21T23:05:14.796252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_taxonomy_df)\n# matches with the num of unique enteries in train_terms.fasta","metadata":{"execution":{"iopub.status.busy":"2023-05-21T23:10:37.205331Z","iopub.execute_input":"2023-05-21T23:10:37.205906Z","iopub.status.idle":"2023-05-21T23:10:37.212863Z","shell.execute_reply.started":"2023-05-21T23:10:37.205863Z","shell.execute_reply":"2023-05-21T23:10:37.211988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df = pd.merge(train_terms_df,train_taxonomy_df,on='EntryID')\nmerged_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-21T23:09:45.734339Z","iopub.execute_input":"2023-05-21T23:09:45.73508Z","iopub.status.idle":"2023-05-21T23:09:47.736865Z","shell.execute_reply.started":"2023-05-21T23:09:45.735044Z","shell.execute_reply":"2023-05-21T23:09:47.735654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"limit = 10\n\nwith open(CFG.train_ia_path) as f:\n    ia_weights = [x.replace(\"\\n\", \"\").split(\"\\t\") for x in f.readlines()]\n\nia_weights[:limit]","metadata":{},"execution_count":null,"outputs":[]}]}