{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Exploratory Data Analysis - CAFA 5 Protein Function Analysis Problem","metadata":{}},{"cell_type":"code","source":"!pip install networkx\n!pip install biopython\n!pip install obonet\n!pip install pyvis","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-05-21T07:25:47.472911Z","iopub.execute_input":"2023-05-21T07:25:47.47423Z","iopub.status.idle":"2023-05-21T07:26:39.958779Z","shell.execute_reply.started":"2023-05-21T07:25:47.47415Z","shell.execute_reply":"2023-05-21T07:26:39.957179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import Dependecies","metadata":{}},{"cell_type":"code","source":"from pyvis.network import Network\nfrom random import choice\nfrom Bio import SeqIO\nfrom matplotlib import pyplot as plt\nfrom collections import Counter\n\nimport networkx\nimport numpy as np\nimport pandas as pd\nimport obonet","metadata":{"ExecuteTime":{"end_time":"2023-05-21T04:47:00.788539Z","start_time":"2023-05-21T04:46:58.685333Z"},"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-05-21T07:26:39.962026Z","iopub.execute_input":"2023-05-21T07:26:39.962541Z","iopub.status.idle":"2023-05-21T07:26:40.577731Z","shell.execute_reply.started":"2023-05-21T07:26:39.962491Z","shell.execute_reply":"2023-05-21T07:26:40.576619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"markdown","source":"Here We store the paths to all the files in a ```class Config``` and define ```plot_dag``` for plotting the graph","metadata":{}},{"cell_type":"code","source":"class Config:\n    go_basic_obo_path = \"/kaggle/input/cafa-5-protein-function-prediction/Train/go-basic.obo\"\n    train_sequences_fasta_path = \"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\"\n    train_terms_tsv_path = \"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\"\n    train_taxonomy_tsv_path = \"/kaggle/input/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv\"\n    ia_txt_path = \"/kaggle/input/cafa-5-protein-function-prediction/IA.txt\"\n\ndef plot_dag(graph, term, radius=1, filename=\"network.html\"):\n    # create a smaller neighbors' graph\n    ng_graph = networkx.ego_graph(graph, term, radius)\n\n    # Add the name of the term with its GO code to make thier labels\n    for node in ng_graph.nodes(data=True):\n        node[1][\"label\"] = node[0] + \" \" + node[1][\"name\"]\n\n    nt = Network(directed=True, notebook=True, cdn_resources=\"in_line\")\n    nt.from_nx(ng_graph)\n\n    return nt.show(filename)","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T04:51:19.700939Z","start_time":"2023-05-21T04:51:19.675928Z"},"jupyter":{"outputs_hidden":false},"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-05-21T07:26:40.578963Z","iopub.execute_input":"2023-05-21T07:26:40.579308Z","iopub.status.idle":"2023-05-21T07:26:40.587299Z","shell.execute_reply.started":"2023-05-21T07:26:40.579279Z","shell.execute_reply":"2023-05-21T07:26:40.586348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the GO Data","metadata":{}},{"cell_type":"markdown","source":"Load the ```go_basic.obo``` file as a graph using obonet","metadata":{}},{"cell_type":"code","source":"graph = obonet.read_obo(Config.go_basic_obo_path)\nnum_of_terms = len(graph.nodes)","metadata":{"ExecuteTime":{"end_time":"2023-05-21T04:51:39.279974Z","start_time":"2023-05-21T04:51:33.031008Z"},"execution":{"iopub.status.busy":"2023-05-21T07:26:40.588666Z","iopub.execute_input":"2023-05-21T07:26:40.589037Z","iopub.status.idle":"2023-05-21T07:26:59.943065Z","shell.execute_reply.started":"2023-05-21T07:26:40.589008Z","shell.execute_reply":"2023-05-21T07:26:59.942158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"there are {num_of_terms} GO-terms associated with {len(graph.edges)} connections between them\")","metadata":{"ExecuteTime":{"end_time":"2023-05-21T04:55:21.220887Z","start_time":"2023-05-21T04:55:21.217113Z"},"execution":{"iopub.status.busy":"2023-05-21T07:26:59.945834Z","iopub.execute_input":"2023-05-21T07:26:59.94612Z","iopub.status.idle":"2023-05-21T07:26:59.984373Z","shell.execute_reply.started":"2023-05-21T07:26:59.946096Z","shell.execute_reply":"2023-05-21T07:26:59.983292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Chose a random term and see its plot, We can also take some specified node to plot the graph","metadata":{}},{"cell_type":"code","source":"# chose some random GO term and plot its graph\nterm = str(choice(list(graph.nodes.keys())))","metadata":{"ExecuteTime":{"end_time":"2023-05-21T04:55:34.040913Z","start_time":"2023-05-21T04:55:34.034843Z"},"execution":{"iopub.status.busy":"2023-05-21T07:26:59.985789Z","iopub.execute_input":"2023-05-21T07:26:59.98621Z","iopub.status.idle":"2023-05-21T07:26:59.999225Z","shell.execute_reply.started":"2023-05-21T07:26:59.98617Z","shell.execute_reply":"2023-05-21T07:26:59.998384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_dag(graph=graph, term=term)","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T04:55:34.8164Z","start_time":"2023-05-21T04:55:34.70162Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:00.000622Z","iopub.execute_input":"2023-05-21T07:27:00.001488Z","iopub.status.idle":"2023-05-21T07:27:00.295609Z","shell.execute_reply.started":"2023-05-21T07:27:00.001456Z","shell.execute_reply":"2023-05-21T07:27:00.294487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_dag(graph = graph, term = term, radius = 1000)","metadata":{"ExecuteTime":{"end_time":"2023-05-21T04:55:38.211491Z","start_time":"2023-05-21T04:55:38.096528Z"},"execution":{"iopub.status.busy":"2023-05-21T07:27:00.296821Z","iopub.execute_input":"2023-05-21T07:27:00.297585Z","iopub.status.idle":"2023-05-21T07:27:00.577959Z","shell.execute_reply.started":"2023-05-21T07:27:00.297553Z","shell.execute_reply":"2023-05-21T07:27:00.576934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the Training Sequences Data","metadata":{}},{"cell_type":"code","source":"sequences = SeqIO.parse(Config.train_sequences_fasta_path, \"fasta\")\nnum_of_proteins = sum(1 for seq in sequences)\nprint(f\"There are {num_of_proteins} protein Sequences given in the file\")","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T04:57:02.034699Z","start_time":"2023-05-21T04:57:00.991087Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:00.579245Z","iopub.execute_input":"2023-05-21T07:27:00.579549Z","iopub.status.idle":"2023-05-21T07:27:03.154688Z","shell.execute_reply.started":"2023-05-21T07:27:00.579523Z","shell.execute_reply":"2023-05-21T07:27:03.153617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Analysis based on length of protein sequences.","metadata":{}},{"cell_type":"code","source":"sequences = SeqIO.parse(Config.train_sequences_fasta_path,\"fasta\")\nlengths = [len(seq) for seq in sequences]\n\nplt.hist(lengths, bins=\"fd\")\nplt.title(\"Distribution of lengths of the protein sequences\")\nplt.xlabel(\"Length of the protein sequence\")\nplt.ylabel(\"Number of protein sequences\")\nplt.show()","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T04:58:08.252664Z","start_time":"2023-05-21T04:58:05.279843Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:03.156203Z","iopub.execute_input":"2023-05-21T07:27:03.156622Z","iopub.status.idle":"2023-05-21T07:27:10.076383Z","shell.execute_reply.started":"2023-05-21T07:27:03.156584Z","shell.execute_reply":"2023-05-21T07:27:10.075551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(np.log10(lengths), bins=\"fd\")\nplt.title(\"Distribution of lengths of the protein sequences\")\nplt.xlabel(\"log$_1$$_0$(Length of the protein sequence)\")\nplt.ylabel(\"Number of protein sequences\")\nplt.show()","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T04:58:23.324566Z","start_time":"2023-05-21T04:58:22.901008Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:10.07746Z","iopub.execute_input":"2023-05-21T07:27:10.077955Z","iopub.status.idle":"2023-05-21T07:27:10.841031Z","shell.execute_reply.started":"2023-05-21T07:27:10.077927Z","shell.execute_reply":"2023-05-21T07:27:10.839859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The length of a protein appears to follow a log-normal distribution.\nThe number of sequences with length more than 5K are nearly negligible although the length goes upto 35K.","metadata":{}},{"cell_type":"markdown","source":"### Now checking the distribution of various amino acids in the given protein sequences","metadata":{}},{"cell_type":"code","source":"records = SeqIO.parse(Config.train_sequences_fasta_path,\"fasta\")\namino_acid_counts = Counter()\n\nfor record in records:\n    seq = str(record.seq)\n    amino_acid_counts.update(seq)\n","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:02:17.708497Z","start_time":"2023-05-21T05:02:12.459934Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:10.84256Z","iopub.execute_input":"2023-05-21T07:27:10.842977Z","iopub.status.idle":"2023-05-21T07:27:17.717563Z","shell.execute_reply.started":"2023-05-21T07:27:10.842941Z","shell.execute_reply":"2023-05-21T07:27:17.716659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = []\nsizes = []\nfor key, value in amino_acid_counts.items():\n    labels.append(key)\n    sizes.append(value)\n\nplt.barh(labels,sizes)\nplt.title(\"Count of Different Amino Acids found in Training Set\")\nplt.xlabel(\"Count\")\nplt.ylabel(\"Amino Acid\")\nplt.show()","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:02:17.892911Z","start_time":"2023-05-21T05:02:17.720073Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:17.719015Z","iopub.execute_input":"2023-05-21T07:27:17.719514Z","iopub.status.idle":"2023-05-21T07:27:18.125435Z","shell.execute_reply.started":"2023-05-21T07:27:17.719478Z","shell.execute_reply":"2023-05-21T07:27:18.12424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The most common amino acids in this dataset are leucine (L), serine (S), alanine (A), and glycine (G). These amino acids are known to be abundant in proteins and play important roles in protein structure and function.\n\n* The least common amino acids in this dataset are cysteine (C), methionine (M), tryptophan (W), and histidine (H). These amino acids are typically less abundant in proteins, but they can be important for specific functions, such as catalysis, metal binding, or protein-protein interactions.\n\n* The presence of the amino acid selenocysteine (U) in the dataset suggests that some of the proteins may be selenoproteins, which contain selenium in the form of selenocysteine instead of cysteine.\n\n* The presence of ambiguous amino acids (X, B, Z) and rare amino acids (O, U) in the dataset suggests that some of the sequences may be incomplete or contain errors.","metadata":{}},{"cell_type":"markdown","source":"## Load Labels Data","metadata":{}},{"cell_type":"code","source":"train_terms_df = pd.read_csv(Config.train_terms_tsv_path,sep=\"\\t\")\ntrain_terms_df.head()","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:02:29.735286Z","start_time":"2023-05-21T05:02:28.189832Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:18.130344Z","iopub.execute_input":"2023-05-21T07:27:18.130729Z","iopub.status.idle":"2023-05-21T07:27:22.007528Z","shell.execute_reply.started":"2023-05-21T07:27:18.130698Z","shell.execute_reply":"2023-05-21T07:27:22.006502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms_df.describe()","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:02:32.112321Z","start_time":"2023-05-21T05:02:30.823551Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:22.008733Z","iopub.execute_input":"2023-05-21T07:27:22.00904Z","iopub.status.idle":"2023-05-21T07:27:26.336705Z","shell.execute_reply.started":"2023-05-21T07:27:22.009014Z","shell.execute_reply":"2023-05-21T07:27:26.335594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So, there are 142246 proteins annotated by 31466 different GO terms out of 43248.","metadata":{}},{"cell_type":"markdown","source":"### pie chart describing the distribution of the different subontologies among different proteins.","metadata":{}},{"cell_type":"code","source":"aspect_counts = train_terms_df.aspect.value_counts()\nplt.pie(aspect_counts.values,labels=aspect_counts.index)\nplt.title(\"Distribution of GO-terms along subontologies\")\nplt.show()","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:03:01.815603Z","start_time":"2023-05-21T05:03:01.764262Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:26.33846Z","iopub.execute_input":"2023-05-21T07:27:26.33886Z","iopub.status.idle":"2023-05-21T07:27:27.315841Z","shell.execute_reply.started":"2023-05-21T07:27:26.338824Z","shell.execute_reply":"2023-05-21T07:27:27.314597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Taxonomy Data","metadata":{}},{"cell_type":"code","source":"taxonomy_df = pd.read_csv(Config.train_taxonomy_tsv_path,sep=\"\\t\")\ntaxonomy_df.head()","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:03:26.517244Z","start_time":"2023-05-21T05:03:26.457784Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:27.317887Z","iopub.execute_input":"2023-05-21T07:27:27.318439Z","iopub.status.idle":"2023-05-21T07:27:27.487747Z","shell.execute_reply.started":"2023-05-21T07:27:27.31839Z","shell.execute_reply":"2023-05-21T07:27:27.486919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"taxonomy_df.taxonomyID.value_counts()","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:03:27.115052Z","start_time":"2023-05-21T05:03:27.107652Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:27.489216Z","iopub.execute_input":"2023-05-21T07:27:27.489834Z","iopub.status.idle":"2023-05-21T07:27:27.503177Z","shell.execute_reply.started":"2023-05-21T07:27:27.489804Z","shell.execute_reply":"2023-05-21T07:27:27.501976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"We have about\", len(taxonomy_df.taxonomyID.unique()), \"different Taxonomy IDs\")","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:04:37.599651Z","start_time":"2023-05-21T05:04:37.592687Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:27.5051Z","iopub.execute_input":"2023-05-21T07:27:27.505848Z","iopub.status.idle":"2023-05-21T07:27:27.515598Z","shell.execute_reply.started":"2023-05-21T07:27:27.505784Z","shell.execute_reply":"2023-05-21T07:27:27.514875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Merge the train labels and the taxonomy data frame.","metadata":{}},{"cell_type":"code","source":"training_df = pd.merge(train_terms_df,taxonomy_df, on= \"EntryID\")\ntraining_df.head()","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:04:57.470893Z","start_time":"2023-05-21T05:04:56.862198Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:27.516902Z","iopub.execute_input":"2023-05-21T07:27:27.517257Z","iopub.status.idle":"2023-05-21T07:27:29.454279Z","shell.execute_reply.started":"2023-05-21T07:27:27.517229Z","shell.execute_reply":"2023-05-21T07:27:29.453217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Define prot_vs_terms_analysis to run the proteins-per-term and terms-per-protein analysis","metadata":{}},{"cell_type":"code","source":"def prot_vs_terms_analysis(df):\n    groupby_entryid = df.groupby(\"EntryID\")\n    groupby_term = df.groupby(\"term\")\n\n    terms_per_protein = []\n    proteins_per_term = []\n\n    for prots in groupby_entryid[\"term\"].unique():\n        terms_per_protein.append(len(prots))\n\n    for terms in groupby_term[\"EntryID\"].unique():\n        proteins_per_term.append(len(terms))\n\n    terms_per_protein = pd.Series(terms_per_protein)\n    proteins_per_term = pd.Series(proteins_per_term)\n\n    print(\"terms_per_protein.describe()\\n\")\n    print(terms_per_protein.describe())\n    print(\"Median : {}\".format(terms_per_protein.median()))\n\n    plt.hist(terms_per_protein,bins=\"fd\")\n    plt.title(\"Distribution of Number of Terms Annotating a protein\")\n    plt.xlabel(\"# of terms per unit protein\")\n    plt.ylabel(\"frequency\")\n    plt.show()\n\n    plt.hist(np.log10(terms_per_protein),bins=\"fd\")\n    plt.title(\"Distribution of Number of Terms Annotating a protein\")\n    plt.xlabel(\"log$_1$$_0$(# of terms per unit protein)\")\n    plt.ylabel(\"frequency\")\n    plt.show()\n\n    print(\"proteins_per_term.describe()\\n\")\n    print(proteins_per_term.describe())\n    print(\"Median : {}\".format(proteins_per_term.median()))\n\n    plt.hist(np.log10(proteins_per_term),bins=\"fd\")\n    plt.title(\"Distribution of Number of proteins Annotated by a Term\")\n    plt.xlabel(\"log$_1$$_0$(# of proteins per term)\")\n    plt.ylabel(\"frequency\")\n    plt.show()","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:17:55.511508Z","start_time":"2023-05-21T05:17:55.500924Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:29.455938Z","iopub.execute_input":"2023-05-21T07:27:29.456385Z","iopub.status.idle":"2023-05-21T07:27:29.467818Z","shell.execute_reply.started":"2023-05-21T07:27:29.456346Z","shell.execute_reply":"2023-05-21T07:27:29.466809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Perform proteins-per-term and terms-per-protein Analysis","metadata":{}},{"cell_type":"code","source":"prot_vs_terms_analysis(training_df)","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:18:04.482033Z","start_time":"2023-05-21T05:17:56.414876Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:29.469379Z","iopub.execute_input":"2023-05-21T07:27:29.469883Z","iopub.status.idle":"2023-05-21T07:27:49.329551Z","shell.execute_reply.started":"2023-05-21T07:27:29.469845Z","shell.execute_reply":"2023-05-21T07:27:49.327966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Number of GO-terms associated to a Protein**\nThere are about 24(median) terms associated with a protein, as low as 2 terms and high as 815 terms with the average of 37.\nThe distribution is heavily skewed to the right, and the number of proteins having more than 150-200 terms associated to them are negligible.\n\n**Number of Protiens annotated by a GO-term**\nThere are about 8(median) proteins annotated by a GO-term but the data is very k=skewed, even so that the graph cannot be plotted linearly and has to be plotted on log-scale.\nThere are as low as 1 protein and high as 92192 with the average of 170.\nEven the 50th percentile is just 8 and the 75th percentile is 35!\nThe distribution is heavily skewed to the right, and the number of proteins having more than 10^(3.5) = 3000 (approx) terms associated to them are negligible.\n","metadata":{}},{"cell_type":"markdown","source":"## Dividing the dataframe by subontologies","metadata":{}},{"cell_type":"code","source":"bpo_training_df = training_df[training_df[\"aspect\"] == \"BPO\"].drop([\"aspect\"], axis=1)\ncco_training_df = training_df[training_df[\"aspect\"] == \"CCO\"].drop([\"aspect\"], axis=1)\nmfo_training_df = training_df[training_df[\"aspect\"] == \"MFO\"].drop([\"aspect\"], axis=1)","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:25:34.081688Z","start_time":"2023-05-21T05:25:32.963578Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:49.331269Z","iopub.execute_input":"2023-05-21T07:27:49.332194Z","iopub.status.idle":"2023-05-21T07:27:53.184764Z","shell.execute_reply.started":"2023-05-21T07:27:49.332138Z","shell.execute_reply":"2023-05-21T07:27:53.183661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### performing the protein vs terms analysis as above with the above Data","metadata":{}},{"cell_type":"code","source":"pd.DataFrame({\"#Prots\":[len(bpo_training_df.EntryID.unique()),\n                        len(cco_training_df.EntryID.unique()),\n                        len(mfo_training_df.EntryID.unique())],\n              \"#Terms\":[len(bpo_training_df.term.unique()),\n                        len(cco_training_df.term.unique()),\n                        len(mfo_training_df.term.unique())],\n              \"aspect\": [\"BPO\", \"CCO\", \"MFO\"]}).set_index('aspect')","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:25:45.051486Z","start_time":"2023-05-21T05:25:44.688575Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:53.186967Z","iopub.execute_input":"2023-05-21T07:27:53.187659Z","iopub.status.idle":"2023-05-21T07:27:54.228609Z","shell.execute_reply.started":"2023-05-21T07:27:53.187611Z","shell.execute_reply":"2023-05-21T07:27:54.227633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Biological Processes Ontology","metadata":{}},{"cell_type":"code","source":"prot_vs_terms_analysis(bpo_training_df)","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:27:56.813017Z","start_time":"2023-05-21T05:27:51.274102Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:27:54.2298Z","iopub.execute_input":"2023-05-21T07:27:54.230193Z","iopub.status.idle":"2023-05-21T07:28:08.150755Z","shell.execute_reply.started":"2023-05-21T07:27:54.230166Z","shell.execute_reply":"2023-05-21T07:28:08.149679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cellular Components Ontology","metadata":{}},{"cell_type":"code","source":"prot_vs_terms_analysis(cco_training_df)","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:28:05.339348Z","start_time":"2023-05-21T05:28:01.439448Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:28:08.152009Z","iopub.execute_input":"2023-05-21T07:28:08.152347Z","iopub.status.idle":"2023-05-21T07:28:17.79945Z","shell.execute_reply.started":"2023-05-21T07:28:08.152319Z","shell.execute_reply":"2023-05-21T07:28:17.798341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Molecular Functions Ontology","metadata":{}},{"cell_type":"code","source":"prot_vs_terms_analysis(mfo_training_df)","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:28:11.976176Z","start_time":"2023-05-21T05:28:08.007439Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:28:17.800692Z","iopub.execute_input":"2023-05-21T07:28:17.801026Z","iopub.status.idle":"2023-05-21T07:28:26.28071Z","shell.execute_reply.started":"2023-05-21T07:28:17.800987Z","shell.execute_reply":"2023-05-21T07:28:26.279675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the Information Accredition Weights of the data","metadata":{}},{"cell_type":"code","source":"with open(Config.ia_txt_path,\"r\") as f:\n    ia = [x.replace(\"\\n\",\"\").split(\"\\t\") for x in f.readlines()]\n\nia = pd.DataFrame(ia, columns=[\"term\",\"WeightAssigned\"])\nia[\"WeightAssigned\"] = ia[\"WeightAssigned\"].astype(np.float64)","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:28:36.948681Z","start_time":"2023-05-21T05:28:36.885539Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:28:26.282007Z","iopub.execute_input":"2023-05-21T07:28:26.282354Z","iopub.status.idle":"2023-05-21T07:28:26.376937Z","shell.execute_reply.started":"2023-05-21T07:28:26.282327Z","shell.execute_reply":"2023-05-21T07:28:26.376067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### plot of distribution of weights assigned to every GO-term.","metadata":{}},{"cell_type":"code","source":"ia[\"WeightAssigned\"].describe()","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:28:38.2361Z","start_time":"2023-05-21T05:28:38.226852Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:28:26.378328Z","iopub.execute_input":"2023-05-21T07:28:26.378669Z","iopub.status.idle":"2023-05-21T07:28:26.390889Z","shell.execute_reply.started":"2023-05-21T07:28:26.378633Z","shell.execute_reply":"2023-05-21T07:28:26.389835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(np.log10(ia[\"WeightAssigned\"][ia[\"WeightAssigned\"] != 0]),bins=\"fd\")\nplt.title(\"Distribution of Weights Assigned to different GO-terms\")\nplt.xlabel(\"log$_1$$_0$ Weights Assigned\")\nplt.ylabel(\"frequency\")\nplt.show()","metadata":{"collapsed":false,"ExecuteTime":{"end_time":"2023-05-21T05:28:39.153523Z","start_time":"2023-05-21T05:28:38.968146Z"},"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-05-21T07:28:26.392197Z","iopub.execute_input":"2023-05-21T07:28:26.392513Z","iopub.status.idle":"2023-05-21T07:28:27.183392Z","shell.execute_reply.started":"2023-05-21T07:28:26.392487Z","shell.execute_reply":"2023-05-21T07:28:27.182332Z"},"trusted":true},"execution_count":null,"outputs":[]}]}