{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ? \n\nSimple Baseline to start with : Covert MultiLabel to MultiTarge  + Embeddings + Ridge \n\n    Features - precalculated embeddings for protein sequences. Thanks to Grandmaster Sergei Fironov for sharing protein emebedding calculated by T5 protein language model from the Rost Lab. \n    \n    Targets - multi-label is converted to mult-target (binary classification) task - i.e. for each sample we are preciting the probability that this label is assigned to that sample. In total there can be 40 000 labels - that is too much, so we choose only N the most frequent ones. \n    \n    After that - use any ML-model you like to make predictions. Start with Ridge as the he most simple and fast one. \n    ","metadata":{}},{"cell_type":"markdown","source":"Thanks to all  authors of the public notebooks and datasets which are quite helpful (please upvote them) and especially those ones:\n\nLEONID KULYK: https://www.kaggle.com/code/leonidkulyk/eda-cafa5-pfp-interactive-dags-plotly\n\nMARÍLIA PRATA: https://www.kaggle.com/code/mpwolke/cafa-5-protein-prediction\n\nDAREK KŁECZEK:  https://www.kaggle.com/code/thedrcat/cafa-eda\n\nD_KHATRI:  https://www.kaggle.com/code/dhruvkhatri/naive-submission-afa\n\n\nAnd special thanks again : to Grandmaster Sergei Fironov for sharing protein emebedding calculated by T5 protein language model from the Rost Lab:  https://www.kaggle.com/datasets/sergeifironov/t5embeds\n\n","metadata":{}},{"cell_type":"markdown","source":"# Key param(s)\n\n","metadata":{}},{"cell_type":"code","source":"n_labels_to_consider = 1499 # We will choose only top frequent labels (in train) and predict only them. ","metadata":{"execution":{"iopub.status.busy":"2023-04-30T16:18:42.795155Z","iopub.execute_input":"2023-04-30T16:18:42.795551Z","iopub.status.idle":"2023-04-30T16:18:42.830941Z","shell.execute_reply.started":"2023-04-30T16:18:42.795516Z","shell.execute_reply":"2023-04-30T16:18:42.829463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport time\nt0start = time.time() \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-30T16:18:42.833485Z","iopub.execute_input":"2023-04-30T16:18:42.83386Z","iopub.status.idle":"2023-04-30T16:18:42.861729Z","shell.execute_reply.started":"2023-04-30T16:18:42.833823Z","shell.execute_reply":"2023-04-30T16:18:42.860632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare multi-target Y  ( transition from multi-label task to multi target task - binary classifiction ). ","metadata":{}},{"cell_type":"markdown","source":"## Load train labels and select the most frequent ones","metadata":{}},{"cell_type":"code","source":"%%time\ntrainTerms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\",sep=\"\\t\")\nprint(trainTerms.shape)\ndisplay(trainTerms.head(2))\nvec_freqCount = (trainTerms['term'].value_counts())\nprint(vec_freqCount )\n\nprint()\nlabels_to_consider = list(vec_freqCount.index[:n_labels_to_consider] )\nprint('n_labels_to_consider:', len(labels_to_consider), 'First 10:', labels_to_consider[:10] ) \n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load protein Ids in train","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/train_ids.npy'\nvec_train_protein_ids = np.load(fn)\nprint(vec_train_protein_ids.shape)\nvec_train_protein_ids","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Y ","metadata":{}},{"cell_type":"code","source":"%%time \ntrain_size = 142246 # len(X)\nY = np.zeros( (train_size ,n_labels_to_consider) )\nprint(Y.shape)\n\nseries_train_protein_ids = pd.Series(vec_train_protein_ids ) # \n\ntrainTerms_smaller = trainTerms[ trainTerms['term'].isin( labels_to_consider ) ] # to speed-up the next step \nprint( trainTerms_smaller.shape)\n\nfor i in range(Y.shape[1]):\n    m = trainTerms_smaller['term'] ==  labels_to_consider[i]\n#     m.sum()\n    Y[:,i] =  series_train_protein_ids.isin(  set(trainTerms_smaller[m]['EntryID'] ) ).astype(float )\n    if (i % 10) == 0: \n        print(i, m.sum())\nY \n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n# save for possible future reuse \nfn4saveY = 'Y_'+str(Y.shape[1])\nprint(fn4saveY)\nnp.save( fn4saveY , Y) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfn4save_labels = 'Y_'+str(Y.shape[1]) + '_labels'\nnp.save(fn4save_labels, labels_to_consider )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print( list(np.load(fn4save_labels +'.npy' ))[:10] )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n# Someone may prefer  Y as dataframe \nif 1:\n    df_Y = pd.DataFrame(data = Y, columns = labels_to_consider)\n    display(df_Y.head(2))\n#     print( df.info().sum() )\n    print('memory_usage:', df_Y.memory_usage(index=True).sum() )\n    display(df_Y.describe() )    \n    fn4save =  'df_Y_'+str(Y.shape[1]) + '.csv'\n    df_Y.to_csv(fn4save)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load train features - precalculated embeddings for the proteins","metadata":{}},{"cell_type":"code","source":"%%time\n\n# fn = '/kaggle/input/protein-embeddings-1/reduced_embeddings_file.npy'\n# fn = '/kaggle/input/protein-embeddings-1/embed_protbert_train_clip_1200_first_70000_prot.csv'\nfn = '/kaggle/input/uniprot-description-only/concatenated_array_train_v3.npy'\n# fn = '/kaggle/input/t5embeds/test_embeds.npy'\n\nprint(fn)\nif '.csv' in fn:\n    df = pd.read_csv(fn, index_col = 0)\n    X = df.values\nelif '.npy' in fn:\n    X = np.load(fn)\nprint(X.shape)\nX","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load protein Ids ","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/train_ids.npy'\nvec_train_protein_ids = np.load(fn)\nprint(vec_train_protein_ids.shape)\nvec_train_protein_ids","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sanity check \n\nIds from the train data are the same as from the train labels data","metadata":{}},{"cell_type":"code","source":"s = set(vec_train_protein_ids) &set (trainTerms['EntryID'] )\nprint( len(s), len( X ) )  # get same numbers ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare Train-Test split ","metadata":{"execution":{"iopub.status.busy":"2023-04-24T10:45:13.246257Z","iopub.execute_input":"2023-04-24T10:45:13.247139Z","iopub.status.idle":"2023-04-24T10:45:23.612468Z","shell.execute_reply.started":"2023-04-24T10:45:13.247095Z","shell.execute_reply":"2023-04-24T10:45:23.611491Z"}}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nIX = np.arange(len(X))\nIX_train, IX_test, _,_ = train_test_split( IX, IX, train_size=0.1, random_state=42)\nprint(len(IX_train), len(IX_test),  IX_train[:10], IX_test[:10] )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Ridge\n\nmodel = Ridge(alpha=100.0)\nstr_model_id = 'Ridge1'\n\ndf_models_stat = pd.DataFrame()\nmodel\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nimport time\nfrom sklearn.metrics import roc_auc_score\n\nt0 = time.time()\nmodel.fit(X[IX_train,:],Y[IX_train,:])\nY_pred_test = model.predict(X[IX_test,:])\ntt = time.time() - t0\nprint(str_model_id, tt)\nl = []\nfor i in range(Y.shape[1]):\n    if len(np.unique(Y[IX_test,i]) ) > 1:\n        s = roc_auc_score(Y[IX_test,i], Y_pred_test[:,i]);\n    else:\n        s = 0.5\n    l.append(s)        \n    if i %10 == 0:\n        print(i, s)\ndf_models_stat.loc[str_model_id,'RocAuc Mean Test'] = np.mean(l)\ndf_models_stat.loc[str_model_id,'Time'] = np.round(tt,1)\ndf_models_stat.loc[str_model_id,'Test Size'] = len(IX_test)\ndf_models_stat\n\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scores statistics over targets ","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.hist(l)\nplt.show()\npd.Series(l).describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Retrain model on the the full sample ","metadata":{}},{"cell_type":"code","source":"%%time\nmodel.fit(X,Y)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission preparations Step 1 - load features and calculate predictions ","metadata":{}},{"cell_type":"markdown","source":"## Load features for submission","metadata":{}},{"cell_type":"code","source":"%%time\n# fn = '/kaggle/input/protein-embeddings-1/reduced_embeddings_file.npy'\n# fn = '/kaggle/input/protein-embeddings-1/embed_protbert_train_clip_1200_first_70000_prot.csv'\n# fn = '/kaggle/input/t5embeds/train_embeds.npy'\nfn = '/kaggle/input/uniprot-description-only/concatenated_array_test_v3.npy'\nprint(fn)\nX_submit = np.load(fn)\nprint(X_submit.shape)\n# X_submit","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Calculate prediction for submission","metadata":{}},{"cell_type":"code","source":"%%time\nY_submit =  model.predict(X_submit)\nprint(Y_submit.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission preparations Step 2 - prepare submision in desired format  ","metadata":{}},{"cell_type":"code","source":"%%time \ndf_finalSubmission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load protein ids for the submission","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/test_ids.npy'\nvec_test_protein_ids = np.load(fn)\nprint(vec_test_protein_ids.shape)\nvec_test_protein_ids","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## \"Melt\" protein ids ","metadata":{}},{"cell_type":"code","source":"%%time \nl = []\nfor k in list(vec_test_protein_ids):\n    l += [ k] * Y_submit.shape[1]\nprint(len(l), l[:20])    \n\ndf_finalSubmission['Protein Id'] = l","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time \n# df_finalSubmission.head(3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## \"Melt\" Labels (Gene ontology terms )","metadata":{"execution":{"iopub.status.busy":"2023-04-24T12:31:15.267721Z","iopub.execute_input":"2023-04-24T12:31:15.268181Z","iopub.status.idle":"2023-04-24T12:31:15.613742Z","shell.execute_reply.started":"2023-04-24T12:31:15.268143Z","shell.execute_reply":"2023-04-24T12:31:15.612514Z"}}},{"cell_type":"code","source":"df_finalSubmission['GO Term Id'] = labels_to_consider * Y_submit.shape[0]\n# df_finalSubmission.head(3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Assign predictions ","metadata":{}},{"cell_type":"code","source":"df_finalSubmission['Prediction'] = Y_submit.ravel()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save ","metadata":{}},{"cell_type":"code","source":"%%time \ndf_finalSubmission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show some info ","metadata":{}},{"cell_type":"code","source":"%%time \ndf_finalSubmission.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \ndf_finalSubmission.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nplt.figure(figsize = (15,4))\nplt.hist(df_finalSubmission['Prediction'].values, bins = 1000 )\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time \n# def GetDfFromFasta(fastaPath):    \n#     input_file = fastaPath\n#     fasta_sequences = SeqIO.parse(open(input_file),'fasta')\n#     nameList = [] \n#     sequenceList = []\n#     for fasta in fasta_sequences:\n#         name, sequence = fasta.id, str(fasta.seq)\n#         nameList.append(name)\n#         sequenceList.append(sequence)\n#     return pd.DataFrame(list(zip(nameList, sequenceList)), columns={\"Id\",\"seq\"})\n# trainTerms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\",sep=\"\\t\")\n# freqCount = dict(trainTerms['term'].value_counts())\n# print( str(freqCount.items())[:100] )\n\n# maxCount = len(trainTerms)\n# freqCount = {k:(v/maxCount) for k, v in freqCount.items()}\n# print( str(freqCount.items())[:100] )\n\n# freqTop  = list(freqCount.items())[0:10]\n# print(freqTop)\n\n# # %%time\n# submission = GetDfFromFasta(\"/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta\")\n# print(type( submission ))\n# display( submission.head(3) ) \n\n# %%time \n# seqId = []\n# goTerm = [] \n# confidence = [] \n# # Add only first 10 most frequent GO terms?!\n# # Also how is relative frequency related to node/branch level\n# freqTop  = list(freqCount.items())[0:10]\n# for indx, row in tqdm(submission.iterrows(),total = submission.shape[0], position=0):\n#     for gofq in freqTop:\n#         seqId.append(row['seq'])\n#         goTerm.append(gofq[0])\n#         confidence.append(gofq[1])\n        \n# finalSubmission = pd.DataFrame(list(zip(seqId,goTerm,confidence)))\n# #submission[\"confidence\"] = submission[\"confidence\"]/max(submission[\"confidence\"])\n# # finalSubmission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}