{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Introduction\n\n* Background information\n    * The fact that a protein has a GO term can affect what another GO term the protein will have.\n    * For example, a protein that is known to be involved in cell signaling will likely have an ontology that reflects this   function. This ontology may include terms such as \"signal transduction\" or \"cellular communication.\" - Bard AI Answer :)\n\n\n* Based on background, I try to extract topN combination of 2 terms about frequent GE term\n","metadata":{}},{"cell_type":"markdown","source":"## TL;DR\n\n1. This kernel is about stuyding about coincidece between GO terms in one protein **(for using in retrieval process as supporting metric for thresholing)**\n2. **The more a GO term get higher weight, the more a GO term is rare**\n3. **But, we will be able to use the coincidence information about sufficient samples**\n4. **If we try to predict the high weight term (rare term), we should make full use of the protein sequence information**","metadata":{}},{"cell_type":"markdown","source":"## Setup","metadata":{}},{"cell_type":"code","source":"!pip install obonet","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:36:59.694783Z","iopub.execute_input":"2023-05-23T03:36:59.695414Z","iopub.status.idle":"2023-05-23T03:37:15.836109Z","shell.execute_reply.started":"2023-05-23T03:36:59.695368Z","shell.execute_reply":"2023-05-23T03:37:15.834549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport pickle\n\nfrom Bio import SeqIO\nfrom Bio.SeqUtils.ProtParam import ProteinAnalysis\nfrom collections import Counter\nimport obonet\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-23T03:37:15.839031Z","iopub.execute_input":"2023-05-23T03:37:15.839545Z","iopub.status.idle":"2023-05-23T03:37:17.259789Z","shell.execute_reply.started":"2023-05-23T03:37:15.839499Z","shell.execute_reply":"2023-05-23T03:37:17.258625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pickleIO(obj, src, op=\"r\"):\n    if op==\"w\":\n        with open(src, op + \"b\") as f:\n            pickle.dump(obj, f)\n    elif op==\"r\":\n        with open(src, op + \"b\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    else:\n        print(\"unknown operation\")\n        return obj","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:37:17.261366Z","iopub.execute_input":"2023-05-23T03:37:17.261793Z","iopub.status.idle":"2023-05-23T03:37:17.269443Z","shell.execute_reply.started":"2023-05-23T03:37:17.261753Z","shell.execute_reply":"2023-05-23T03:37:17.268311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set amino acid mapper\naa_map = {'VAL': 'V', 'PRO': 'P', 'ASN': 'N', 'GLU': 'E', 'ASP': 'D', 'ALA': 'A', 'THR': 'T', 'SER': 'S',\n          'LEU': 'L', 'LYS': 'K', 'GLY': 'G', 'GLN': 'Q', 'ILE': 'I', 'PHE': 'F', 'CYS': 'C', 'TRP': 'W',\n          'ARG': 'R', 'TYR': 'Y', 'HIS': 'H', 'MET': 'M'}\naa_map[\"X\"] = \"X\"\naa_map_encoder = {x:y for x,y in zip(list(aa_map.values()),np.arange(21))}\n\n# set amino acid group mapper\naa_groups = {\n    # Electrically Charged Side Chains - positive\n    \"AAG0\": [\"R\", \"H\", \"K\"],\n    # Electrically Charged Side Chains - negative\n    \"AAG1\": [\"D\", \"E\"],\n    # Polar Uncharged Side Chains\n    \"AAG2\": [\"S\", \"T\", \"N\", \"Q\"],\n    # Hydrophobic Side Chains\n    \"AAG3\": [\"A\", \"V\", \"I\", \"L\", \"M\", \"F\", \"Y\", \"W\"],\n    # Not including any group\n    \"AAG4\": [\"P\", \"G\", \"C\", \"X\"],\n}\naa_groups_encoder = {}\nvalue = 0\nfor i in aa_groups.values():\n    for j in i: aa_groups_encoder[j] = value\n    value += 1\ndef get_amino_acids_group_percent(seq):\n    counter = Counter([aa_groups_encoder[i] for i in seq])\n    norm = sum(counter.values())\n    return {f\"AAG{k}\": v / norm for k, v in counter.items()}\n\n# get amino acid properties\n# https://www.kaggle.com/datasets/alejopaullier/aminoacids-physical-and-chemical-properties\naa_props = pd.read_csv(\"/kaggle/input/aminoacids-physical-and-chemical-properties/aminoacids.csv\").set_index('Letter')\n# set property variable for analysis\nPROPS = ['Molecular Weight', 'Residue Weight', 'pKa1', 'pKb2', 'pKx3', 'pl4', \n         'H', 'VSC', 'P1', 'P2', 'SASA', 'NCISC', 'carbon', 'hydrogen', 'oxygen']\n# remove pKx3 which includes na values\nPROPS.remove(\"pKx3\")\n# remove special case amino acid\naa_props = aa_props.drop([\"O\", \"U\"])\n# impute X amino acid property values with mean of other amino acids\nvalue = aa_props.mean()\nfor i in [\"Name\", \"Abbr\", \"Molecular Formula\", \"Residue Formula\"]:\n    value[i] = \"X\"\naa_props.loc[\"X\"] = value\n# shape check\nprint('Amino Acid properties dataframe. Shape:', aa_props.shape )\n# validation check\nfor i in aa_props.index:\n    if i not in aa_map.values():\n        print(i)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:37:17.273752Z","iopub.execute_input":"2023-05-23T03:37:17.274629Z","iopub.status.idle":"2023-05-23T03:37:17.353775Z","shell.execute_reply.started":"2023-05-23T03:37:17.274581Z","shell.execute_reply":"2023-05-23T03:37:17.352377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aa_props","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:37:17.355508Z","iopub.execute_input":"2023-05-23T03:37:17.356647Z","iopub.status.idle":"2023-05-23T03:37:17.430126Z","shell.execute_reply.started":"2023-05-23T03:37:17.356599Z","shell.execute_reply":"2023-05-23T03:37:17.428972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(aa_map_encoder)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:37:17.431782Z","iopub.execute_input":"2023-05-23T03:37:17.432275Z","iopub.status.idle":"2023-05-23T03:37:17.438396Z","shell.execute_reply.started":"2023-05-23T03:37:17.43222Z","shell.execute_reply":"2023-05-23T03:37:17.436949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(aa_groups_encoder)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:37:17.439978Z","iopub.execute_input":"2023-05-23T03:37:17.440388Z","iopub.status.idle":"2023-05-23T03:37:17.451358Z","shell.execute_reply.started":"2023-05-23T03:37:17.440356Z","shell.execute_reply":"2023-05-23T03:37:17.450263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GO annotation data (target)","metadata":{}},{"cell_type":"code","source":"df_term = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\", sep=\"\\t\")\ndf_term = df_term.set_index(\"EntryID\")","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:37:17.452734Z","iopub.execute_input":"2023-05-23T03:37:17.453366Z","iopub.status.idle":"2023-05-23T03:37:21.890824Z","shell.execute_reply.started":"2023-05-23T03:37:17.453332Z","shell.execute_reply":"2023-05-23T03:37:21.889595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_term.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:37:21.89234Z","iopub.execute_input":"2023-05-23T03:37:21.892778Z","iopub.status.idle":"2023-05-23T03:37:21.914574Z","shell.execute_reply.started":"2023-05-23T03:37:21.892739Z","shell.execute_reply":"2023-05-23T03:37:21.912947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_term.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:37:21.918184Z","iopub.execute_input":"2023-05-23T03:37:21.9186Z","iopub.status.idle":"2023-05-23T03:37:21.930893Z","shell.execute_reply.started":"2023-05-23T03:37:21.918565Z","shell.execute_reply":"2023-05-23T03:37:21.929731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### GO Term coincidence counting","metadata":{}},{"cell_type":"code","source":"from collections import defaultdict\nfrom itertools import permutations\nfrom itertools import combinations","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:37:21.932389Z","iopub.execute_input":"2023-05-23T03:37:21.932759Z","iopub.status.idle":"2023-05-23T03:37:21.943481Z","shell.execute_reply.started":"2023-05-23T03:37:21.932729Z","shell.execute_reply":"2023-05-23T03:37:21.941862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_term_group = df_term.groupby(\"EntryID\")[\"term\"].apply(lambda x: sorted(x))\ndf_term_group.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:37:21.945116Z","iopub.execute_input":"2023-05-23T03:37:21.94579Z","iopub.status.idle":"2023-05-23T03:37:30.19767Z","shell.execute_reply.started":"2023-05-23T03:37:21.945749Z","shell.execute_reply":"2023-05-23T03:37:30.196225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# combination with 2 term\ncounter = Counter()\nfor i in tqdm(df_term_group):\n    for j in combinations(i, 2):\n        counter[j] += 1\n#     break\n\ndisplay(counter.most_common(20))\npickleIO(counter, \"./term_comb2.pkl\", \"w\")\n\ncounter_norm = Counter({k: v / len(df_term_group) for k, v in counter.items()})\ndisplay(counter_norm.most_common(20))\npickleIO(counter_norm, \"./term_comb2_norm.pkl\", \"w\")","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:37:30.198978Z","iopub.execute_input":"2023-05-23T03:37:30.199359Z","iopub.status.idle":"2023-05-23T03:42:31.646316Z","shell.execute_reply.started":"2023-05-23T03:37:30.19932Z","shell.execute_reply":"2023-05-23T03:42:31.645398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"term_topN = df_term[\"term\"].value_counts()\ndisplay(term_topN)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:42:31.647643Z","iopub.execute_input":"2023-05-23T03:42:31.648161Z","iopub.status.idle":"2023-05-23T03:42:32.657513Z","shell.execute_reply.started":"2023-05-23T03:42:31.64813Z","shell.execute_reply":"2023-05-23T03:42:32.656318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# top 5 GO term's best coincidence\ntopN_count = dict.fromkeys(term_topN.index[:5].values)\nfor idx, value in zip(term_topN.index[:5], term_topN.values[:5]):\n    topN_count[idx] = [value, pd.DataFrame()]\n    tmp_output = {\n        \"comb\": [],\n        \"count\": [],\n    }\n    for k, v in tqdm(counter.items()):\n        if idx in k:\n            tmp_output[\"comb\"].append(k)\n            tmp_output[\"count\"].append(v)\n    topN_count[idx][1][\"comb\"] = tmp_output[\"comb\"]\n    topN_count[idx][1][\"count\"] = tmp_output[\"count\"]\n    topN_count[idx][1] = topN_count[idx][1].sort_values(\"count\", ascending=False)\n    topN_count[idx][1][\"count_norm\"] = topN_count[idx][1][\"count\"] / len(df_term_group)\n    display(topN_count[idx][1].head())","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:42:32.658704Z","iopub.execute_input":"2023-05-23T03:42:32.659042Z","iopub.status.idle":"2023-05-23T03:43:30.981565Z","shell.execute_reply.started":"2023-05-23T03:42:32.659014Z","shell.execute_reply":"2023-05-23T03:43:30.980212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualization (Ratio is total samples)\n\n* We can find that there is more chance to be some specific terms when any term appears","metadata":{}},{"cell_type":"code","source":"# Visualization\nfor k, v in topN_count.items():\n    df_tmp = v[1].head(10)\n    sns.barplot(x=df_tmp[\"count_norm\"].values, y=df_tmp[\"comb\"].astype(\"str\").values)\n    plt.title(f\"{k} term combination ratio (frequency : {v[0]} / {len(df_term_group)})\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:43:30.983307Z","iopub.execute_input":"2023-05-23T03:43:30.983786Z","iopub.status.idle":"2023-05-23T03:43:32.664722Z","shell.execute_reply.started":"2023-05-23T03:43:30.983734Z","shell.execute_reply":"2023-05-23T03:43:32.663327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualization (Ratio is only about the GO Term)","metadata":{}},{"cell_type":"code","source":"# top 5 GO term's best coincidence\ntopN_count = dict.fromkeys(term_topN.index[:5].values)\nfor idx, value in zip(term_topN.index[:5], term_topN.values[:5]):\n    topN_count[idx] = [value, pd.DataFrame()]\n    tmp_output = {\n        \"comb\": [],\n        \"count\": [],\n    }\n    for k, v in tqdm(counter.items()):\n        if idx in k:\n            tmp_output[\"comb\"].append(k)\n            tmp_output[\"count\"].append(v)\n    topN_count[idx][1][\"comb\"] = tmp_output[\"comb\"]\n    topN_count[idx][1][\"count\"] = tmp_output[\"count\"]\n    topN_count[idx][1] = topN_count[idx][1].sort_values(\"count\", ascending=False)\n    topN_count[idx][1][\"count_norm\"] = topN_count[idx][1][\"count\"] / df_term_group.apply(lambda x: idx in x).sum()\n    display(topN_count[idx][1].head())","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:49:50.638397Z","iopub.execute_input":"2023-05-23T03:49:50.638783Z","iopub.status.idle":"2023-05-23T03:50:46.555705Z","shell.execute_reply.started":"2023-05-23T03:49:50.638754Z","shell.execute_reply":"2023-05-23T03:50:46.554215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualization\nfor k, v in topN_count.items():\n    df_tmp = v[1].head(10)\n    sns.barplot(x=df_tmp[\"count_norm\"].values, y=df_tmp[\"comb\"].astype(\"str\").values)\n    plt.title(f\"{k} term combination ratio (frequency : {v[0]} / {len(df_term_group)})\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:50:46.557763Z","iopub.execute_input":"2023-05-23T03:50:46.558139Z","iopub.status.idle":"2023-05-23T03:50:49.471747Z","shell.execute_reply.started":"2023-05-23T03:50:46.558105Z","shell.execute_reply":"2023-05-23T03:50:49.470686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* We can find there are some perfect coincidence combinations\n     * Ex. (GO:005575, GO:0110165)","metadata":{}},{"cell_type":"markdown","source":"### How about high weight GO term?\n\n* We can find that the terms with high weight is very rare in train set\n    * They already informed us that the terms with high depth which is hard to predicting get high score ","metadata":{}},{"cell_type":"code","source":"with open(\"/kaggle/input/cafa-5-protein-function-prediction/IA.txt\", \"r\") as f:\n    tmp = f.readlines()\n    tmp = [i.rstrip(\"\\n\").split(\"\\t\") for i in tmp]\n    res1, res2 = map(list, zip(*tmp))\n    term_weight = pd.Series({k: v for k, v in zip(res1, res2)}).astype(\"float32\")\nterm_weight = term_weight[term_weight.index.isin(df_term[\"term\"].unique())]\nterm_weight.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:18:30.681624Z","iopub.execute_input":"2023-05-23T03:18:30.682492Z","iopub.status.idle":"2023-05-23T03:18:32.219648Z","shell.execute_reply.started":"2023-05-23T03:18:30.682445Z","shell.execute_reply":"2023-05-23T03:18:32.21834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"term_weight.sort_values(ascending=False).head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:18:32.221122Z","iopub.execute_input":"2023-05-23T03:18:32.221576Z","iopub.status.idle":"2023-05-23T03:18:32.236542Z","shell.execute_reply.started":"2023-05-23T03:18:32.221531Z","shell.execute_reply":"2023-05-23T03:18:32.235199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"term_topN = df_term[\"term\"].value_counts().loc[term_weight.sort_values(ascending=False).index[:5]]\ndisplay(term_topN)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:18:32.238229Z","iopub.execute_input":"2023-05-23T03:18:32.238983Z","iopub.status.idle":"2023-05-23T03:18:33.237126Z","shell.execute_reply.started":"2023-05-23T03:18:32.238949Z","shell.execute_reply":"2023-05-23T03:18:33.236054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# top 5 GO term's best coincidence\ntopN_count = dict.fromkeys(term_topN.index[:5].values)\nfor idx, value in zip(term_topN.index[:5], term_topN.values[:5]):\n    topN_count[idx] = [value, pd.DataFrame()]\n    tmp_output = {\n        \"comb\": [],\n        \"count\": [],\n    }\n    for k, v in tqdm(counter.items()):\n        if idx in k:\n            tmp_output[\"comb\"].append(k)\n            tmp_output[\"count\"].append(v)\n    topN_count[idx][1][\"comb\"] = tmp_output[\"comb\"]\n    topN_count[idx][1][\"count\"] = tmp_output[\"count\"]\n    topN_count[idx][1] = topN_count[idx][1].sort_values(\"count\", ascending=False)\n    topN_count[idx][1][\"count_norm\"] = topN_count[idx][1][\"count\"] / len(df_term_group)\n    display(topN_count[idx][1].head())","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:18:33.238465Z","iopub.execute_input":"2023-05-23T03:18:33.238861Z","iopub.status.idle":"2023-05-23T03:19:26.524498Z","shell.execute_reply.started":"2023-05-23T03:18:33.238832Z","shell.execute_reply":"2023-05-23T03:19:26.523473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualization\nfor k, v in topN_count.items():\n    df_tmp = v[1].head(10)\n    sns.barplot(x=df_tmp[\"count_norm\"].values, y=df_tmp[\"comb\"].astype(\"str\").values)\n    plt.title(f\"{k} term combination ratio (frequency : {v[0]} / {len(df_term_group)})\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:19:26.525962Z","iopub.execute_input":"2023-05-23T03:19:26.526521Z","iopub.status.idle":"2023-05-23T03:19:27.881039Z","shell.execute_reply.started":"2023-05-23T03:19:26.526479Z","shell.execute_reply":"2023-05-23T03:19:27.879502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's find more about this phenomenon","metadata":{}},{"cell_type":"code","source":"term_weight.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:19:27.882769Z","iopub.execute_input":"2023-05-23T03:19:27.883314Z","iopub.status.idle":"2023-05-23T03:19:27.899393Z","shell.execute_reply.started":"2023-05-23T03:19:27.883243Z","shell.execute_reply":"2023-05-23T03:19:27.898259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(term_weight, color=\"green\")\nplt.title(\"GO Term Weight distribution\")\nplt.show()\n\nsns.histplot(term_weight[term_weight != 0], color=\"orange\")\nplt.title(\"GO Term Weight distribution (Removing 0)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:19:27.900944Z","iopub.execute_input":"2023-05-23T03:19:27.901256Z","iopub.status.idle":"2023-05-23T03:19:28.698783Z","shell.execute_reply.started":"2023-05-23T03:19:27.901229Z","shell.execute_reply":"2023-05-23T03:19:28.697684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x=term_weight.sort_index().values, y=df_term[\"term\"].value_counts().sort_index().values, color=\"green\")\nplt.title(\"GO Term Weight & Frequency scatter plot\")\nplt.show()\n\nmask = term_weight.sort_index().values != 0\nsns.scatterplot(x=term_weight.sort_index().values[mask], y=df_term[\"term\"].value_counts().sort_index().values[mask], color=\"orange\")\nplt.title(\"GO Term Weight & Frequency scatter plot (Removing 0)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:19:28.700104Z","iopub.execute_input":"2023-05-23T03:19:28.700467Z","iopub.status.idle":"2023-05-23T03:19:31.57729Z","shell.execute_reply.started":"2023-05-23T03:19:28.700437Z","shell.execute_reply":"2023-05-23T03:19:31.576139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* We can find that the more term gets high value the more the terms are rare","metadata":{}},{"cell_type":"code","source":"df_count = df_term[\"term\"].value_counts()\ndf_weight = term_weight.sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:22:28.73398Z","iopub.execute_input":"2023-05-23T03:22:28.734528Z","iopub.status.idle":"2023-05-23T03:22:29.74967Z","shell.execute_reply.started":"2023-05-23T03:22:28.734477Z","shell.execute_reply":"2023-05-23T03:22:29.748387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_samples = 3\nthreshold = 0.95\nratio_list = []\nfor i in range(16):\n    tmp = (df_count.loc[df_weight[df_weight >= i].index] <= n_samples).mean()\n    print(f\"GO Term ratio less than {n_samples} is {round(tmp, 3)}\")\n    if tmp >= threshold:\n        print(f\"Minimum value of larger than threshold {threshold} is {i}\")\n    ratio_list.append(tmp)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:22:29.751518Z","iopub.execute_input":"2023-05-23T03:22:29.752266Z","iopub.status.idle":"2023-05-23T03:22:29.851387Z","shell.execute_reply.started":"2023-05-23T03:22:29.75223Z","shell.execute_reply":"2023-05-23T03:22:29.850132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(x=range(16), y=ratio_list, marker=\"o\")\nplt.title(f\"GO Term ratio the frequency of term is less than equal {n_samples}\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T03:22:30.000093Z","iopub.execute_input":"2023-05-23T03:22:30.000501Z","iopub.status.idle":"2023-05-23T03:22:30.300021Z","shell.execute_reply.started":"2023-05-23T03:22:30.000469Z","shell.execute_reply":"2023-05-23T03:22:30.298852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* We can find there are only samples of which frequency is less than equal 3, if GO term weight is larger than equal 14\n    * In other words, this imply hard to use interaction information between GO terms when we predict high weight term","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}