{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nimport statistics","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-22T17:46:01.903785Z","iopub.execute_input":"2023-05-22T17:46:01.904232Z","iopub.status.idle":"2023-05-22T17:46:01.934689Z","shell.execute_reply.started":"2023-05-22T17:46:01.904194Z","shell.execute_reply":"2023-05-22T17:46:01.93381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_tax = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv\", sep = \"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:22:39.95163Z","iopub.execute_input":"2023-05-22T17:22:39.952567Z","iopub.status.idle":"2023-05-22T17:22:40.045942Z","shell.execute_reply.started":"2023-05-22T17:22:39.952496Z","shell.execute_reply":"2023-05-22T17:22:40.044491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_tax.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:22:40.048762Z","iopub.execute_input":"2023-05-22T17:22:40.049402Z","iopub.status.idle":"2023-05-22T17:22:40.061679Z","shell.execute_reply.started":"2023-05-22T17:22:40.049341Z","shell.execute_reply":"2023-05-22T17:22:40.060337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\", sep = \"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:22:40.063221Z","iopub.execute_input":"2023-05-22T17:22:40.063645Z","iopub.status.idle":"2023-05-22T17:22:42.849212Z","shell.execute_reply.started":"2023-05-22T17:22:40.063594Z","shell.execute_reply":"2023-05-22T17:22:42.847998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:22:42.852796Z","iopub.execute_input":"2023-05-22T17:22:42.853401Z","iopub.status.idle":"2023-05-22T17:22:42.86815Z","shell.execute_reply.started":"2023-05-22T17:22:42.853348Z","shell.execute_reply":"2023-05-22T17:22:42.866499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Distribution of GO terms","metadata":{}},{"cell_type":"markdown","source":"The terms have a long tailed distribution.","metadata":{}},{"cell_type":"code","source":"freq_df = pd.DataFrame(train_terms[\"term\"].value_counts())\nfreq_df = freq_df.reset_index()\nfreq_df.columns = ['Category', 'Count']\n\nfreq_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:45:09.482755Z","iopub.execute_input":"2023-05-22T17:45:09.483424Z","iopub.status.idle":"2023-05-22T17:45:10.453563Z","shell.execute_reply.started":"2023-05-22T17:45:09.483378Z","shell.execute_reply":"2023-05-22T17:45:10.452202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(freq_df[\"Count\"] < 10)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:24:47.49471Z","iopub.execute_input":"2023-05-22T17:24:47.495625Z","iopub.status.idle":"2023-05-22T17:24:47.508897Z","shell.execute_reply.started":"2023-05-22T17:24:47.495574Z","shell.execute_reply":"2023-05-22T17:24:47.507363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The median number of proteins corresponding to a function is 8, while the maximum is 92k. This suggests it's a long tailed distribution","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:22:43.896054Z","iopub.execute_input":"2023-05-22T17:22:43.896418Z","iopub.status.idle":"2023-05-22T17:22:43.903794Z","shell.execute_reply.started":"2023-05-22T17:22:43.896376Z","shell.execute_reply":"2023-05-22T17:22:43.902262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_freq_df = freq_df[0:1000]\ntop_freq_df","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:44:59.725166Z","iopub.execute_input":"2023-05-22T17:44:59.725774Z","iopub.status.idle":"2023-05-22T17:44:59.74424Z","shell.execute_reply.started":"2023-05-22T17:44:59.725731Z","shell.execute_reply":"2023-05-22T17:44:59.743046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.bar(top_freq_df[\"Category\"], top_freq_df[\"Count\"])\nplt.xlabel(\"Term\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:26:29.920725Z","iopub.execute_input":"2023-05-22T17:26:29.92181Z","iopub.status.idle":"2023-05-22T17:26:43.215242Z","shell.execute_reply.started":"2023-05-22T17:26:29.92176Z","shell.execute_reply":"2023-05-22T17:26:43.213808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Aspect wise distribution","metadata":{}},{"cell_type":"markdown","source":"Most annotations pertain to Biological Process, followed by Cellular Component, and then Molecular Function.","metadata":{}},{"cell_type":"code","source":"aspectdf = pd.DataFrame(train_terms[\"aspect\"].value_counts())\naspectdf = aspectdf.reset_index()\naspectdf.columns = [\"Category\", \"Count\"]\n\nplt.bar(aspectdf[\"Category\"], aspectdf[\"Count\"] )\nplt.xlabel(\"Subontology\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:30:50.474823Z","iopub.execute_input":"2023-05-22T17:30:50.475678Z","iopub.status.idle":"2023-05-22T17:30:51.589771Z","shell.execute_reply.started":"2023-05-22T17:30:50.475632Z","shell.execute_reply":"2023-05-22T17:30:51.588354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bpo_df = train_terms[train_terms[\"aspect\"] == \"BPO\"]\nbpo_counts = pd.DataFrame(bpo_df[\"term\"].value_counts())\nbpo_counts = bpo_counts.reset_index()\nbpo_counts.columns = [\"Category\", \"Count\"]\nbpo_counts","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:53:16.985072Z","iopub.execute_input":"2023-05-22T17:53:16.986185Z","iopub.status.idle":"2023-05-22T17:53:18.776958Z","shell.execute_reply.started":"2023-05-22T17:53:16.986102Z","shell.execute_reply":"2023-05-22T17:53:18.775606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bpo_counts.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:55:19.096946Z","iopub.execute_input":"2023-05-22T17:55:19.097407Z","iopub.status.idle":"2023-05-22T17:55:19.120071Z","shell.execute_reply.started":"2023-05-22T17:55:19.097367Z","shell.execute_reply":"2023-05-22T17:55:19.118777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The BPO terms appear to follow a similarly long tailed distribution.","metadata":{}},{"cell_type":"code","source":"plt.bar(bpo_counts[\"Category\"], bpo_counts[\"Count\"])\nplt.show","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cco_df = train_terms[train_terms[\"aspect\"] == \"CCO\"]\ncco_counts = pd.DataFrame(cco_df[\"term\"].value_counts())\ncco_counts = cco_counts.reset_index()\ncco_counts.columns = [\"Category\", \"Count\"]\ncco_counts.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:55:07.80054Z","iopub.execute_input":"2023-05-22T17:55:07.801073Z","iopub.status.idle":"2023-05-22T17:55:09.109265Z","shell.execute_reply.started":"2023-05-22T17:55:07.801026Z","shell.execute_reply":"2023-05-22T17:55:09.107752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mfo_df = train_terms[train_terms[\"aspect\"] == \"MFO\"]\nmfo_counts = pd.DataFrame(mfo_df[\"term\"].value_counts())\nmfo_counts = mfo_counts.reset_index()\nmfo_counts.columns = [\"Category\", \"Count\"]\nmfo_counts.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:56:41.622388Z","iopub.execute_input":"2023-05-22T17:56:41.623712Z","iopub.status.idle":"2023-05-22T17:56:42.807975Z","shell.execute_reply.started":"2023-05-22T17:56:41.623654Z","shell.execute_reply":"2023-05-22T17:56:42.8066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can observe that BP and CC terms are similarly distributed while distribution of MF terms is even more skewed.","metadata":{}},{"cell_type":"markdown","source":"Protein wise","metadata":{}},{"cell_type":"code","source":"protein_df = pd.DataFrame(train_terms[\"EntryID\"].value_counts())\nprotein_df = protein_df.reset_index()\nprotein_df.columns = [\"Category\", \"Count\"] \nprotein_df","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:40:08.953488Z","iopub.execute_input":"2023-05-22T17:40:08.954507Z","iopub.status.idle":"2023-05-22T17:40:09.847003Z","shell.execute_reply.started":"2023-05-22T17:40:08.954459Z","shell.execute_reply":"2023-05-22T17:40:09.845411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:43:22.656072Z","iopub.execute_input":"2023-05-22T17:43:22.656502Z","iopub.status.idle":"2023-05-22T17:43:22.687776Z","shell.execute_reply.started":"2023-05-22T17:43:22.656462Z","shell.execute_reply":"2023-05-22T17:43:22.686577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_prot_df = protein_df[0:1000]\ntop_prot_df","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:41:43.810186Z","iopub.execute_input":"2023-05-22T17:41:43.810643Z","iopub.status.idle":"2023-05-22T17:41:43.82585Z","shell.execute_reply.started":"2023-05-22T17:41:43.810601Z","shell.execute_reply":"2023-05-22T17:41:43.824463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can observe that proteins have a relatively even distribution compared to GO terms.","metadata":{}},{"cell_type":"code","source":"plt.bar(top_prot_df[\"Category\"], top_prot_df[\"Count\"] )\nplt.xlabel(\"Protein\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:41:46.39555Z","iopub.execute_input":"2023-05-22T17:41:46.396002Z","iopub.status.idle":"2023-05-22T17:41:58.08075Z","shell.execute_reply.started":"2023-05-22T17:41:46.39596Z","shell.execute_reply":"2023-05-22T17:41:58.079475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Around 16% of proteins have less than 10 GO terms annotated","metadata":{}},{"cell_type":"code","source":"sum(protein_df[\"Count\"] < 10)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T17:42:09.480114Z","iopub.execute_input":"2023-05-22T17:42:09.480569Z","iopub.status.idle":"2023-05-22T17:42:09.510584Z","shell.execute_reply.started":"2023-05-22T17:42:09.480515Z","shell.execute_reply":"2023-05-22T17:42:09.509272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA of sequences","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}