{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Create training and validation split that matches the evaluation set-up\n\n\n## Motivation\n\n[Public notebooks](https://www.kaggle.com/code/sergeifironov/validate-ridge/notebook) typically split the proteins randomly between the train and validation set.\n\nHowever, as it has been pointed out in this [discussion](https://www.kaggle.com/competitions/cafa-5-protein-function-prediction/discussion/409329), it may not be the ideal strategy.\n\n\n## Alternative approaches\n\nI'm aware of these 2 noteable alternatives:\n\n* [Use simlarity measure between proteins](https://www.kaggle.com/code/alexandervc/cafa5-23-groups-and-folds-diamond-igraph)\n* [Use annotation date](https://www.kaggle.com/code/liudacheldieva/p38666-date-of-annotation/notebook)\n\nI'd love to explore both of these ideas, but as I'm late for the party and have zero domain knowledge, so I tried to think about a simpler alternative.\n\n## Current proposal: simulate the evaluation phase\n\nIf I understood this passage correctly, in order to mimick the evalution phase as closely as possible, we need to make sure that all labels of a given protein in a given subontology land either in training or validation set, but not both.\n\nThe Evaluation section [says](https://www.kaggle.com/competitions/cafa-5-protein-function-prediction/overview/evaluation)\n\n    Submissions will be evaluated on proteins (test set) that did not have experimentally determined functional annotations in at least one subontology of the Gene Ontology (GO) before the submission deadline and have accumulated experimentally-validated functional annotations in that subontology between the submission deadline and the time of evaluation. For example, a protein that had no experimental terms in say Molecular Function (MF) subontology of GO and has accumulated experimental annotations in MF after the submission deadline will be included in the test set for evaluating the MF term predictions. The same holds for the Biological Process (BP) or Cellular Component (CC) subontologies of GO. The proteins that qualify will create three different test sets, one for each subontology of GO. The same protein can appear in more than one test set if it accumulates experimentally-validated annotations in more than a single subontology.\n    \n    \n","metadata":{}},{"cell_type":"code","source":"from typing import Dict, List, Union\n\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom pathlib import Path\nimport gc\n\nfrom Bio import SeqIO\nfrom tqdm import tqdm\n\n%config Completer.use_jedi = False # solves potential issue with slow autocomplete in Jupyter 6\n\n\nRANDOM_SEEED = 232145\nDIR_COMPET = Path('/kaggle/input/cafa-5-protein-function-prediction')\n\ndef read_ia_weights(path: Union[Path, str]) -> Dict[str, float]:\n    with open(path, 'r') as f:\n        weights = {}\n        while True:\n            line = f.readline()\n            if not line:\n                break\n            go_term, weight = line.split()\n            weights[go_term] = float(weight)\n    return weights","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-29T08:53:54.492675Z","iopub.execute_input":"2023-07-29T08:53:54.49307Z","iopub.status.idle":"2023-07-29T08:53:54.5034Z","shell.execute_reply.started":"2023-07-29T08:53:54.493039Z","shell.execute_reply":"2023-07-29T08:53:54.502458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(DIR_COMPET / 'Train/train_terms.tsv', sep='\\t')\ndf_train.rename(columns={'EntryID': 'protein', 'term': 'label', 'aspect': 'ontology'}, inplace=True)\ndf_tax = pd.read_csv(DIR_COMPET / 'Train/train_taxonomy.tsv', sep='\\t')\n\n\ndf_submis_sample = pd.read_csv(DIR_COMPET / 'sample_submission.tsv', sep='\\t', header=None)\ndf_submis_sample.columns = ['protein', 'label', 'prob']\nprint(df_submis_sample.protein.nunique())\n\n\nweights_ia = read_ia_weights(DIR_COMPET / 'IA.txt')","metadata":{"execution":{"iopub.status.busy":"2023-07-29T08:35:06.686269Z","iopub.execute_input":"2023-07-29T08:35:06.687181Z","iopub.status.idle":"2023-07-29T08:35:10.838065Z","shell.execute_reply.started":"2023-07-29T08:35:06.687147Z","shell.execute_reply":"2023-07-29T08:35:10.836887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_sequences = list(SeqIO.parse(DIR_COMPET / \"Test (Targets)/testsuperset.fasta\", \"fasta\"))\nlen(test_sequences)","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:09:06.034916Z","iopub.execute_input":"2023-07-29T10:09:06.035354Z","iopub.status.idle":"2023-07-29T10:09:08.175539Z","shell.execute_reply.started":"2023-07-29T10:09:06.035318Z","shell.execute_reply":"2023-07-29T10:09:08.174264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(df_submis_sample.protein) - set(getattr(x, 'id') for x in test_sequences)","metadata":{"execution":{"iopub.status.busy":"2023-07-29T08:35:13.94515Z","iopub.execute_input":"2023-07-29T08:35:13.945499Z","iopub.status.idle":"2023-07-29T08:35:14.078463Z","shell.execute_reply.started":"2023-07-29T08:35:13.945469Z","shell.execute_reply":"2023-07-29T08:35:14.077239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.label.nunique(), df_train.protein.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T08:35:14.079762Z","iopub.execute_input":"2023-07-29T08:35:14.080098Z","iopub.status.idle":"2023-07-29T08:35:14.876989Z","shell.execute_reply.started":"2023-07-29T08:35:14.080067Z","shell.execute_reply":"2023-07-29T08:35:14.875804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check how the proteins are distributed among the training and test samples","metadata":{}},{"cell_type":"code","source":"df_val_check = df_train.drop_duplicates(\"protein\").merge(df_submis_sample.drop_duplicates('protein'), \n                                                         how='outer', \n                                                         on='protein',\n                                                        suffixes=('_train','_test'))\nassert not df_val_check.protein.isnull().any()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T08:35:14.878473Z","iopub.execute_input":"2023-07-29T08:35:14.879046Z","iopub.status.idle":"2023-07-29T08:35:15.44334Z","shell.execute_reply.started":"2023-07-29T08:35:14.879014Z","shell.execute_reply":"2023-07-29T08:35:15.442154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_val_check.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-29T08:35:15.444934Z","iopub.execute_input":"2023-07-29T08:35:15.445631Z","iopub.status.idle":"2023-07-29T08:35:15.452605Z","shell.execute_reply.started":"2023-07-29T08:35:15.445596Z","shell.execute_reply":"2023-07-29T08:35:15.451413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_only_train = df_val_check.prob.isnull().sum()\nn_inters = df_val_check[(df_val_check.prob.notnull()) & (df_val_check.label_train.notnull())].shape[0]\nn_only_test = df_val_check[(df_val_check.prob.notnull()) & (df_val_check.label_train.isnull())].shape[0]\n\n\nprint(f\"Number of proteins that are present only in the training set: {n_only_train}\")\nprint(f\"Number of proteins that are present both in training set and test set: {n_inters}\")\nprint(f\"Number of proteins that are present only in the test set: {n_only_test}\")","metadata":{"execution":{"iopub.status.busy":"2023-07-29T08:37:49.550503Z","iopub.execute_input":"2023-07-29T08:37:49.551Z","iopub.status.idle":"2023-07-29T08:37:49.620111Z","shell.execute_reply.started":"2023-07-29T08:37:49.550958Z","shell.execute_reply":"2023-07-29T08:37:49.618785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_inters / df_val_check.protein.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:53:59.213125Z","iopub.execute_input":"2023-07-29T09:53:59.21368Z","iopub.status.idle":"2023-07-29T09:53:59.280297Z","shell.execute_reply.started":"2023-07-29T09:53:59.213633Z","shell.execute_reply":"2023-07-29T09:53:59.279433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_only_train / df_val_check.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-07-29T08:37:50.500497Z","iopub.execute_input":"2023-07-29T08:37:50.501336Z","iopub.status.idle":"2023-07-29T08:37:50.509397Z","shell.execute_reply.started":"2023-07-29T08:37:50.501292Z","shell.execute_reply":"2023-07-29T08:37:50.508262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_inters / df_val_check.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-07-29T08:37:50.686327Z","iopub.execute_input":"2023-07-29T08:37:50.687138Z","iopub.status.idle":"2023-07-29T08:37:50.694491Z","shell.execute_reply.started":"2023-07-29T08:37:50.687094Z","shell.execute_reply":"2023-07-29T08:37:50.693395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_only_test / df_val_check.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-07-29T08:37:50.852469Z","iopub.execute_input":"2023-07-29T08:37:50.853147Z","iopub.status.idle":"2023-07-29T08:37:50.86038Z","shell.execute_reply.started":"2023-07-29T08:37:50.853103Z","shell.execute_reply":"2023-07-29T08:37:50.859178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert  n_only_train + n_inters + n_only_test == df_val_check.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-07-29T08:38:48.49532Z","iopub.execute_input":"2023-07-29T08:38:48.49571Z","iopub.status.idle":"2023-07-29T08:38:48.500405Z","shell.execute_reply.started":"2023-07-29T08:38:48.495682Z","shell.execute_reply":"2023-07-29T08:38:48.499469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check how many labels does each protein in the training set have in each subontology\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-24T20:13:09.961024Z","iopub.execute_input":"2023-07-24T20:13:09.961405Z","iopub.status.idle":"2023-07-24T20:13:09.967278Z","shell.execute_reply.started":"2023-07-24T20:13:09.961376Z","shell.execute_reply":"2023-07-24T20:13:09.965585Z"}}},{"cell_type":"markdown","source":"    45% of proteins had at least one label in only 1 subontology - each protein can go either to train or validation  => protein is present in either train or test\n    \n    25% of proteins had at least one label in 2 subontologies - one subontology goes to train, another to validation => protein is present in both train and test\n    \n    30% of proteins had at least one label in 3 subontologies - 1 or 2 subontologies go to train, the other(s) to validation => protein is present in both train and test","metadata":{}},{"cell_type":"code","source":"df_counts_long = df_train.groupby(['protein', 'ontology']).size()\ndf_counts = df_counts_long.unstack().fillna(0)\n\nn_ontol = (df_counts > 0).sum(axis=1)\nn_ontol.value_counts(normalize=True).sort_index().round(2)","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:25:33.362271Z","iopub.execute_input":"2023-07-29T09:25:33.362664Z","iopub.status.idle":"2023-07-29T09:25:34.839098Z","shell.execute_reply.started":"2023-07-29T09:25:33.362633Z","shell.execute_reply":"2023-07-29T09:25:34.838011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_ontol.value_counts().sort_index()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:25:35.168806Z","iopub.execute_input":"2023-07-29T09:25:35.169285Z","iopub.status.idle":"2023-07-29T09:25:35.178569Z","shell.execute_reply.started":"2023-07-29T09:25:35.169249Z","shell.execute_reply":"2023-07-29T09:25:35.177622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_counts.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T08:40:13.505367Z","iopub.execute_input":"2023-07-29T08:40:13.505871Z","iopub.status.idle":"2023-07-29T08:40:13.526314Z","shell.execute_reply.started":"2023-07-29T08:40:13.505827Z","shell.execute_reply":"2023-07-29T08:40:13.524842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.set_index(['protein', 'ontology'])","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:30:52.452857Z","iopub.execute_input":"2023-07-29T09:30:52.453398Z","iopub.status.idle":"2023-07-29T09:30:54.001905Z","shell.execute_reply.started":"2023-07-29T09:30:52.453339Z","shell.execute_reply":"2023-07-29T09:30:54.000741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split into train and validation set","metadata":{}},{"cell_type":"markdown","source":"## 1. Protein with labels in only one subontology\n\n    Assign <TRAIN_PROP> to the training set","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nTRAIN_PROP = 0.6","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:57:55.79749Z","iopub.execute_input":"2023-07-29T09:57:55.797883Z","iopub.status.idle":"2023-07-29T09:57:55.802747Z","shell.execute_reply.started":"2023-07-29T09:57:55.797845Z","shell.execute_reply":"2023-07-29T09:57:55.801738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx_train, _ = train_test_split(n_ontol[n_ontol==1].index, train_size=TRAIN_PROP, random_state=RANDOM_SEEED)\nlen(idx_train)","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:57:55.940248Z","iopub.execute_input":"2023-07-29T09:57:55.940865Z","iopub.status.idle":"2023-07-29T09:57:55.955871Z","shell.execute_reply.started":"2023-07-29T09:57:55.940831Z","shell.execute_reply":"2023-07-29T09:57:55.955069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(n_ontol[n_ontol==1].index) * TRAIN_PROP","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:57:57.390138Z","iopub.execute_input":"2023-07-29T09:57:57.390715Z","iopub.status.idle":"2023-07-29T09:57:57.401174Z","shell.execute_reply.started":"2023-07-29T09:57:57.390685Z","shell.execute_reply":"2023-07-29T09:57:57.400176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(n_ontol==1).mean()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:58:10.066628Z","iopub.execute_input":"2023-07-29T09:58:10.067012Z","iopub.status.idle":"2023-07-29T09:58:10.075089Z","shell.execute_reply.started":"2023-07-29T09:58:10.066984Z","shell.execute_reply":"2023-07-29T09:58:10.073897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx_train","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:58:10.193962Z","iopub.execute_input":"2023-07-29T09:58:10.194342Z","iopub.status.idle":"2023-07-29T09:58:10.202957Z","shell.execute_reply.started":"2023-07-29T09:58:10.194313Z","shell.execute_reply":"2023-07-29T09:58:10.201645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Protein with labels in 2 subontologies\n\n    For each protein, assign the most numerous ontology to train ","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:46:51.202977Z","iopub.execute_input":"2023-07-29T09:46:51.203371Z","iopub.status.idle":"2023-07-29T09:46:51.209171Z","shell.execute_reply.started":"2023-07-29T09:46:51.203342Z","shell.execute_reply":"2023-07-29T09:46:51.20779Z"}}},{"cell_type":"code","source":"len(n_ontol[n_ontol==2].index)","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:58:10.427339Z","iopub.execute_input":"2023-07-29T09:58:10.427697Z","iopub.status.idle":"2023-07-29T09:58:10.437971Z","shell.execute_reply.started":"2023-07-29T09:58:10.427668Z","shell.execute_reply":"2023-07-29T09:58:10.436736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_counts_long.loc[n_ontol[n_ontol==2].index].reset_index().rename(columns={0: 'n_labels'}).sort_values(['protein','n_labels'], ascending=False)","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:58:10.563562Z","iopub.execute_input":"2023-07-29T09:58:10.564167Z","iopub.status.idle":"2023-07-29T09:58:10.696789Z","shell.execute_reply.started":"2023-07-29T09:58:10.564136Z","shell.execute_reply":"2023-07-29T09:58:10.69578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx_train_2 = (df_counts_long.loc[n_ontol[n_ontol==2].index].reset_index().\n               rename(columns={0: 'n_labels'}).\n               sort_values(['protein','n_labels'], ascending=False).\n               drop_duplicates('protein', keep='first').\n               set_index(['protein','ontology']).index\n              )","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:58:10.698569Z","iopub.execute_input":"2023-07-29T09:58:10.698901Z","iopub.status.idle":"2023-07-29T09:58:10.877722Z","shell.execute_reply.started":"2023-07-29T09:58:10.698871Z","shell.execute_reply":"2023-07-29T09:58:10.876487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Protein with labels in all 3 subontologies\n\n    For each protein, N_ONTOL_IN_TRAIN ontologies to train","metadata":{}},{"cell_type":"code","source":"N_ONTOL_IN_TRAIN = 2","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:58:11.749311Z","iopub.execute_input":"2023-07-29T09:58:11.749701Z","iopub.status.idle":"2023-07-29T09:58:11.75396Z","shell.execute_reply.started":"2023-07-29T09:58:11.749669Z","shell.execute_reply":"2023-07-29T09:58:11.752945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prot_prev = None\nidx_train_3 = []\nfor row in (df_counts_long.loc[n_ontol[n_ontol==3].index].reset_index().\n               rename(columns={0: 'n_labels'}).\n               sort_values(['protein','n_labels'], ascending=False)\n           ).itertuples():\n    if row.protein != prot_prev:    \n        prot_prev = row.protein\n        n_ontol_processed = 0\n    if n_ontol_processed == N_ONTOL_IN_TRAIN:\n        continue\n    idx_train_3.append((row.protein, row.ontology))\n    n_ontol_processed += 1\n\nidx_train_3 = pd.Index(idx_train_3)\nassert df_counts_long.loc[idx_train_3].unstack().shape[0] == len(n_ontol[n_ontol==3].index)\nassert (df_counts_long.loc[idx_train_3].unstack() > 0).sum(axis=1).value_counts().index.item() == N_ONTOL_IN_TRAIN","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:58:11.881115Z","iopub.execute_input":"2023-07-29T09:58:11.881503Z","iopub.status.idle":"2023-07-29T09:58:12.501415Z","shell.execute_reply.started":"2023-07-29T09:58:11.881471Z","shell.execute_reply":"2023-07-29T09:58:12.500048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_split = pd.DataFrame(df_counts_long).rename(columns={0: 'n_labels'}).assign(is_train = False)\nassert df_split.shape[0] == df_counts_long.shape[0]\n\ndf_split.loc[idx_train, 'is_train'] = True\ndf_split.loc[idx_train_2, 'is_train'] = True\ndf_split.loc[idx_train_3, 'is_train'] = True\ndf_split","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:58:12.86179Z","iopub.execute_input":"2023-07-29T09:58:12.862325Z","iopub.status.idle":"2023-07-29T09:58:13.055259Z","shell.execute_reply.started":"2023-07-29T09:58:12.86228Z","shell.execute_reply":"2023-07-29T09:58:13.05394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_split[df_split.is_train]","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:58:13.12501Z","iopub.execute_input":"2023-07-29T09:58:13.125415Z","iopub.status.idle":"2023-07-29T09:58:13.145663Z","shell.execute_reply.started":"2023-07-29T09:58:13.125381Z","shell.execute_reply":"2023-07-29T09:58:13.144756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_split[~df_split.is_train]","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:58:13.272065Z","iopub.execute_input":"2023-07-29T09:58:13.272448Z","iopub.status.idle":"2023-07-29T09:58:13.293414Z","shell.execute_reply.started":"2023-07-29T09:58:13.272415Z","shell.execute_reply":"2023-07-29T09:58:13.292384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_split = df_train.loc[df_split[df_split.is_train].index].reset_index()\ndf_valid_split = df_train.loc[df_split[~df_split.is_train].index].reset_index()\nassert df_train_split.shape[0] + df_valid_split.shape[0] == df_train.shape[0]\nassert len(df_train_split) + len(df_valid_split) == len(df_train)\ndf_train_split.protein.nunique(), df_valid_split.protein.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:58:14.242559Z","iopub.execute_input":"2023-07-29T09:58:14.242932Z","iopub.status.idle":"2023-07-29T09:58:32.140947Z","shell.execute_reply.started":"2023-07-29T09:58:14.242902Z","shell.execute_reply":"2023-07-29T09:58:32.140077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:37:10.937432Z","iopub.execute_input":"2023-07-29T10:37:10.937838Z","iopub.status.idle":"2023-07-29T10:37:10.95232Z","shell.execute_reply.started":"2023-07-29T10:37:10.937807Z","shell.execute_reply":"2023-07-29T10:37:10.951489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Percent of proteins in the traning set: {df_train_split.protein.nunique() / len(set(df_train.index.get_level_values(0))): .1%}\")","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:38:09.235574Z","iopub.execute_input":"2023-07-29T10:38:09.235956Z","iopub.status.idle":"2023-07-29T10:38:10.172392Z","shell.execute_reply.started":"2023-07-29T10:38:09.235927Z","shell.execute_reply":"2023-07-29T10:38:10.171046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Percent of labels in the traning set: {len(df_train_split) / len(df_train): .1%}\")","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:37:53.856356Z","iopub.execute_input":"2023-07-29T10:37:53.856773Z","iopub.status.idle":"2023-07-29T10:37:53.862139Z","shell.execute_reply.started":"2023-07-29T10:37:53.856739Z","shell.execute_reply":"2023-07-29T10:37:53.861003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Verify the logic","metadata":{}},{"cell_type":"code","source":"df_counts_train = df_train_split.groupby(['protein', 'ontology']).size().unstack().fillna(0)\ndf_counts_valid = df_valid_split.groupby(['protein', 'ontology']).size().unstack().fillna(0)\n\nassert not df_train_split.duplicated().any()\nassert not df_valid_split.duplicated().any()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:59:00.014728Z","iopub.execute_input":"2023-07-29T09:59:00.0151Z","iopub.status.idle":"2023-07-29T09:59:01.317282Z","shell.execute_reply.started":"2023-07-29T09:59:00.015071Z","shell.execute_reply":"2023-07-29T09:59:01.31636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_counts_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:13:42.212108Z","iopub.execute_input":"2023-07-29T10:13:42.212489Z","iopub.status.idle":"2023-07-29T10:13:42.224594Z","shell.execute_reply.started":"2023-07-29T10:13:42.212458Z","shell.execute_reply":"2023-07-29T10:13:42.223334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_counts_valid.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:13:42.322177Z","iopub.execute_input":"2023-07-29T10:13:42.323222Z","iopub.status.idle":"2023-07-29T10:13:42.33607Z","shell.execute_reply.started":"2023-07-29T10:13:42.323162Z","shell.execute_reply":"2023-07-29T10:13:42.334948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_counts","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:13:44.251276Z","iopub.execute_input":"2023-07-29T10:13:44.251952Z","iopub.status.idle":"2023-07-29T10:13:44.265966Z","shell.execute_reply.started":"2023-07-29T10:13:44.251917Z","shell.execute_reply":"2023-07-29T10:13:44.264794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_counts.loc[n_ontol[n_ontol==3].index]","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:11:44.237593Z","iopub.execute_input":"2023-07-29T10:11:44.238058Z","iopub.status.idle":"2023-07-29T10:11:44.284184Z","shell.execute_reply.started":"2023-07-29T10:11:44.238015Z","shell.execute_reply":"2023-07-29T10:11:44.283132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_counts_train.loc['X5JB51']","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:11:52.961279Z","iopub.execute_input":"2023-07-29T10:11:52.961627Z","iopub.status.idle":"2023-07-29T10:11:52.969214Z","shell.execute_reply.started":"2023-07-29T10:11:52.961599Z","shell.execute_reply":"2023-07-29T10:11:52.968114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_counts_valid.loc['X5JB51']","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:11:59.480738Z","iopub.execute_input":"2023-07-29T10:11:59.48112Z","iopub.status.idle":"2023-07-29T10:11:59.489557Z","shell.execute_reply.started":"2023-07-29T10:11:59.481088Z","shell.execute_reply":"2023-07-29T10:11:59.488259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_counts_train.loc['A0A009IHW8']","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:59:46.305275Z","iopub.execute_input":"2023-07-29T09:59:46.305638Z","iopub.status.idle":"2023-07-29T09:59:46.31443Z","shell.execute_reply.started":"2023-07-29T09:59:46.30561Z","shell.execute_reply":"2023-07-29T09:59:46.313212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_counts_valid.loc['A0A009IHW8']","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:59:53.625825Z","iopub.execute_input":"2023-07-29T09:59:53.626242Z","iopub.status.idle":"2023-07-29T09:59:53.634461Z","shell.execute_reply.started":"2023-07-29T09:59:53.626206Z","shell.execute_reply":"2023-07-29T09:59:53.633459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save the result","metadata":{"execution":{"iopub.status.busy":"2023-07-29T09:44:22.229991Z","iopub.execute_input":"2023-07-29T09:44:22.230384Z","iopub.status.idle":"2023-07-29T09:44:22.234767Z","shell.execute_reply.started":"2023-07-29T09:44:22.230348Z","shell.execute_reply":"2023-07-29T09:44:22.233711Z"}}},{"cell_type":"code","source":"MAP_NAME = {'protein': 'EntryID', 'label': 'term', 'ontology': 'aspect'}\nd  = pd.read_csv(DIR_COMPET / 'Train/train_terms.tsv', sep='\\t', nrows=1)\nCOLS = d.columns\nd","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:42:58.779184Z","iopub.execute_input":"2023-07-29T10:42:58.779612Z","iopub.status.idle":"2023-07-29T10:42:58.795057Z","shell.execute_reply.started":"2023-07-29T10:42:58.779578Z","shell.execute_reply":"2023-07-29T10:42:58.793998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_split[:1].rename(columns=MAP_NAME)[COLS]","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:43:13.490814Z","iopub.execute_input":"2023-07-29T10:43:13.49126Z","iopub.status.idle":"2023-07-29T10:43:13.506398Z","shell.execute_reply.started":"2023-07-29T10:43:13.49122Z","shell.execute_reply":"2023-07-29T10:43:13.50498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_valid_split[:1].rename(columns=MAP_NAME)[COLS]","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:43:17.618788Z","iopub.execute_input":"2023-07-29T10:43:17.619244Z","iopub.status.idle":"2023-07-29T10:43:17.636676Z","shell.execute_reply.started":"2023-07-29T10:43:17.619204Z","shell.execute_reply":"2023-07-29T10:43:17.634898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_split.rename(columns=MAP_NAME)[COLS].to_csv('train_split.tsv', sep='\\t', index=False)\ndf_valid_split.rename(columns=MAP_NAME)[COLS].to_csv('valid_split.tsv', sep='\\t', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-07-29T10:43:35.054136Z","iopub.execute_input":"2023-07-29T10:43:35.054566Z","iopub.status.idle":"2023-07-29T10:43:47.43474Z","shell.execute_reply.started":"2023-07-29T10:43:35.05453Z","shell.execute_reply":"2023-07-29T10:43:47.433431Z"},"trusted":true},"execution_count":null,"outputs":[]}]}