{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ? \n\nSimple Baseline to start with : Convert MultiLabel to MultiTarge  + Embeddings + Ridge \n\n    Features - precalculated embeddings for protein sequences. Thanks to Grandmaster Sergei Fironov for sharing protein emebedding calculated by T5 protein language model from the Rost Lab. \n    \n    Targets - multi-label is converted to mult-target (binary classification) task - i.e. for each sample we are predicting the probability that this label is assigned to that sample. In total there can be 40 000 labels - that is too much, so we choose only N the most frequent ones. \n    \n    After that - use any ML-model you like to make predictions. Start with Ridge as the the most simple and fast one. \n    ","metadata":{}},{"cell_type":"markdown","source":"Thanks to all  authors of the public notebooks and datasets which are quite helpful (please upvote them) and especially those ones:\n\nLEONID KULYK: https://www.kaggle.com/code/leonidkulyk/eda-cafa5-pfp-interactive-dags-plotly\n\nMARÍLIA PRATA: https://www.kaggle.com/code/mpwolke/cafa-5-protein-prediction\n\nDAREK KŁECZEK:  https://www.kaggle.com/code/thedrcat/cafa-eda\n\nD_KHATRI:  https://www.kaggle.com/code/dhruvkhatri/naive-submission-afa\n\n\nAnd special thanks again : to Grandmaster Sergei Fironov for sharing protein emebedding calculated by T5 protein language model from the Rost Lab:  https://www.kaggle.com/datasets/sergeifironov/t5embeds\n\n","metadata":{}},{"cell_type":"markdown","source":"# Key param(s)\n\n","metadata":{}},{"cell_type":"code","source":"n_labels_to_consider = 100 # We will choose only top frequent labels (in train) and predict only them. ","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:20:19.118255Z","iopub.execute_input":"2023-07-05T19:20:19.118833Z","iopub.status.idle":"2023-07-05T19:20:19.152232Z","shell.execute_reply.started":"2023-07-05T19:20:19.118789Z","shell.execute_reply":"2023-07-05T19:20:19.151225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport time\nt0start = time.time() \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-05T19:20:21.715729Z","iopub.execute_input":"2023-07-05T19:20:21.716397Z","iopub.status.idle":"2023-07-05T19:20:21.746198Z","shell.execute_reply.started":"2023-07-05T19:20:21.716362Z","shell.execute_reply":"2023-07-05T19:20:21.744744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare multi-target Y  ( transition from multi-label task to multi target task - binary classifiction ). ","metadata":{}},{"cell_type":"markdown","source":"## Load train labels and select the most frequent ones","metadata":{}},{"cell_type":"code","source":"%%time\ntrainTerms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\",sep=\"\\t\")\nprint(trainTerms.shape)\ndisplay(trainTerms.head(2))\nvec_freqCount = (trainTerms['term'].value_counts())\nprint(vec_freqCount )","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:20:28.162865Z","iopub.execute_input":"2023-07-05T19:20:28.163296Z","iopub.status.idle":"2023-07-05T19:20:32.515775Z","shell.execute_reply.started":"2023-07-05T19:20:28.163257Z","shell.execute_reply":"2023-07-05T19:20:32.514603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print()\nlabels_to_consider = list(vec_freqCount.index[:n_labels_to_consider] )\nprint('n_labels_to_consider:', len(labels_to_consider), 'First 10:', labels_to_consider[:10] ) ","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:20:39.278537Z","iopub.execute_input":"2023-07-05T19:20:39.278941Z","iopub.status.idle":"2023-07-05T19:20:39.285664Z","shell.execute_reply.started":"2023-07-05T19:20:39.278907Z","shell.execute_reply":"2023-07-05T19:20:39.284507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load protein Ids in train","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/train_ids.npy'\nvec_train_protein_ids = np.load(fn)\nprint(vec_train_protein_ids.shape)\nvec_train_protein_ids","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:20:46.842922Z","iopub.execute_input":"2023-07-05T19:20:46.843899Z","iopub.status.idle":"2023-07-05T19:20:46.89988Z","shell.execute_reply.started":"2023-07-05T19:20:46.843858Z","shell.execute_reply":"2023-07-05T19:20:46.898783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Y ","metadata":{}},{"cell_type":"code","source":"%%time \ntrain_size = 142246 # len(X)\nY = np.zeros( (train_size ,n_labels_to_consider) )\nprint(Y.shape)\n\nseries_train_protein_ids = pd.Series(vec_train_protein_ids ) # \n\ntrainTerms_smaller = trainTerms[ trainTerms['term'].isin( labels_to_consider ) ] # to speed-up the next step \nprint( trainTerms_smaller.shape)\n\nfor i in range(Y.shape[1]):\n    m = trainTerms_smaller['term'] ==  labels_to_consider[i]\n#     m.sum()\n    Y[:,i] =  series_train_protein_ids.isin(  set(trainTerms_smaller[m]['EntryID'] ) ).astype(float )\n    if (i % 10) == 0: \n        print(i, m.sum())\nY \n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:20:54.565715Z","iopub.execute_input":"2023-07-05T19:20:54.566334Z","iopub.status.idle":"2023-07-05T19:21:33.775225Z","shell.execute_reply.started":"2023-07-05T19:20:54.566293Z","shell.execute_reply":"2023-07-05T19:21:33.773934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n# save for possible future reuse \nfn4saveY = 'Y_'+str(Y.shape[1])\nprint(fn4saveY)\nnp.save( fn4saveY , Y) ","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:21:52.844575Z","iopub.execute_input":"2023-07-05T19:21:52.845214Z","iopub.status.idle":"2023-07-05T19:21:52.93264Z","shell.execute_reply.started":"2023-07-05T19:21:52.845173Z","shell.execute_reply":"2023-07-05T19:21:52.931379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfn4save_labels = 'Y_'+str(Y.shape[1]) + '_labels'\nnp.save(fn4save_labels, labels_to_consider )","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:21:56.251639Z","iopub.execute_input":"2023-07-05T19:21:56.252017Z","iopub.status.idle":"2023-07-05T19:21:56.259477Z","shell.execute_reply.started":"2023-07-05T19:21:56.251984Z","shell.execute_reply":"2023-07-05T19:21:56.258402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print( list(np.load(fn4save_labels +'.npy' ))[:10] )","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:22:00.07191Z","iopub.execute_input":"2023-07-05T19:22:00.072416Z","iopub.status.idle":"2023-07-05T19:22:00.080799Z","shell.execute_reply.started":"2023-07-05T19:22:00.072369Z","shell.execute_reply":"2023-07-05T19:22:00.07972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n# Someone may prefer  Y as dataframe \nif 1:\n    df_Y = pd.DataFrame(data = Y, columns = labels_to_consider)\n    display(df_Y.head(2))\n#     print( df.info().sum() )\n    print('memory_usage:', df_Y.memory_usage(index=True).sum() )\n    display(df_Y.describe() )    \n    fn4save =  'df_Y_'+str(Y.shape[1]) + '.csv'\n    df_Y.to_csv(fn4save)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:22:05.398551Z","iopub.execute_input":"2023-07-05T19:22:05.398924Z","iopub.status.idle":"2023-07-05T19:22:14.729782Z","shell.execute_reply.started":"2023-07-05T19:22:05.398892Z","shell.execute_reply":"2023-07-05T19:22:14.728494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load train features - precalculated embeddings for the proteins","metadata":{}},{"cell_type":"code","source":"%%time\n\n# fn = '/kaggle/input/protein-embeddings-1/reduced_embeddings_file.npy'\n# fn = '/kaggle/input/protein-embeddings-1/embed_protbert_train_clip_1200_first_70000_prot.csv'\nfn = '/kaggle/input/t5embeds/train_embeds.npy'\n# fn = '/kaggle/input/t5embeds/test_embeds.npy'\n\nprint(fn)\nif '.csv' in fn:\n    df = pd.read_csv(fn, index_col = 0)\n    X = df.values\nelif '.npy' in fn:\n    X = np.load(fn)\nprint(X.shape)\nX","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:22:14.732223Z","iopub.execute_input":"2023-07-05T19:22:14.732648Z","iopub.status.idle":"2023-07-05T19:22:25.248685Z","shell.execute_reply.started":"2023-07-05T19:22:14.732604Z","shell.execute_reply":"2023-07-05T19:22:25.247572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load protein Ids ","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/train_ids.npy'\nvec_train_protein_ids = np.load(fn)\nprint(vec_train_protein_ids.shape)\nvec_train_protein_ids","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:22:30.226381Z","iopub.execute_input":"2023-07-05T19:22:30.226976Z","iopub.status.idle":"2023-07-05T19:22:30.239657Z","shell.execute_reply.started":"2023-07-05T19:22:30.226928Z","shell.execute_reply":"2023-07-05T19:22:30.238429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sanity check \n\nIds from the train data are the same as from the train labels data","metadata":{}},{"cell_type":"code","source":"s = set(vec_train_protein_ids) &set (trainTerms['EntryID'] )\nprint( len(s), len( X ) )  # get same numbers ","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:22:34.774923Z","iopub.execute_input":"2023-07-05T19:22:34.775639Z","iopub.status.idle":"2023-07-05T19:22:35.570586Z","shell.execute_reply.started":"2023-07-05T19:22:34.775598Z","shell.execute_reply":"2023-07-05T19:22:35.569432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare Train-Test split ","metadata":{"execution":{"iopub.status.busy":"2023-04-24T10:45:13.246257Z","iopub.execute_input":"2023-04-24T10:45:13.247139Z","iopub.status.idle":"2023-04-24T10:45:23.612468Z","shell.execute_reply.started":"2023-04-24T10:45:13.247095Z","shell.execute_reply":"2023-04-24T10:45:23.611491Z"}}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nIX = np.arange(len(X))\nIX_train, IX_test, _,_ = train_test_split( IX, IX, train_size=0.1, random_state=42)\nprint(len(IX_train), len(IX_test),  IX_train[:10], IX_test[:10] )","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:22:38.930815Z","iopub.execute_input":"2023-07-05T19:22:38.931226Z","iopub.status.idle":"2023-07-05T19:22:39.44189Z","shell.execute_reply.started":"2023-07-05T19:22:38.931188Z","shell.execute_reply":"2023-07-05T19:22:39.44075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Ridge\n\nmodel = Ridge(alpha=1.0)\nstr_model_id = 'Ridge1'\n\ndf_models_stat = pd.DataFrame()\nmodel\n","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:22:45.428574Z","iopub.execute_input":"2023-07-05T19:22:45.429722Z","iopub.status.idle":"2023-07-05T19:22:45.528147Z","shell.execute_reply.started":"2023-07-05T19:22:45.429669Z","shell.execute_reply":"2023-07-05T19:22:45.527068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nimport time\nfrom sklearn.metrics import roc_auc_score\n\nt0 = time.time()\nmodel.fit(X[IX_train,:],Y[IX_train,:])\nY_pred_test = model.predict(X[IX_test,:])\ntt = time.time() - t0\nprint(str_model_id, tt)\nl = []\nfor i in range(Y.shape[1]):\n    if len(np.unique(Y[IX_test,i]) ) > 1:\n        s = roc_auc_score(Y[IX_test,i], Y_pred_test[:,i]);\n    else:\n        s = 0.5\n    l.append(s)        \n    if i %10 == 0:\n        print(i, s)\ndf_models_stat.loc[str_model_id,'RocAuc Mean Test'] = np.mean(l)\ndf_models_stat.loc[str_model_id,'Time'] = np.round(tt,1)\ndf_models_stat.loc[str_model_id,'Test Size'] = len(IX_test)\ndf_models_stat\n\n    ","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:22:55.656305Z","iopub.execute_input":"2023-07-05T19:22:55.657529Z","iopub.status.idle":"2023-07-05T19:23:02.406715Z","shell.execute_reply.started":"2023-07-05T19:22:55.657478Z","shell.execute_reply":"2023-07-05T19:23:02.405467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scores statistics over targets ","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.hist(l)\nplt.show()\npd.Series(l).describe()","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:23:51.931066Z","iopub.execute_input":"2023-07-05T19:23:51.931612Z","iopub.status.idle":"2023-07-05T19:23:52.162586Z","shell.execute_reply.started":"2023-07-05T19:23:51.931564Z","shell.execute_reply":"2023-07-05T19:23:52.161553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Retrain model on the the full sample ","metadata":{}},{"cell_type":"code","source":"%%time\nmodel.fit(X,Y)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:23:56.538578Z","iopub.execute_input":"2023-07-05T19:23:56.539313Z","iopub.status.idle":"2023-07-05T19:24:01.540894Z","shell.execute_reply.started":"2023-07-05T19:23:56.539272Z","shell.execute_reply":"2023-07-05T19:24:01.538302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission preparations Step 1 - load features and calculate predictions ","metadata":{}},{"cell_type":"markdown","source":"## Load features for submission","metadata":{}},{"cell_type":"code","source":"%%time\n# fn = '/kaggle/input/protein-embeddings-1/reduced_embeddings_file.npy'\n# fn = '/kaggle/input/protein-embeddings-1/embed_protbert_train_clip_1200_first_70000_prot.csv'\n# fn = '/kaggle/input/t5embeds/train_embeds.npy'\nfn = '/kaggle/input/t5embeds/test_embeds.npy'\nprint(fn)\nX_submit = np.load(fn)\nprint(X_submit.shape)\n# X_submit","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:24:08.982072Z","iopub.execute_input":"2023-07-05T19:24:08.982515Z","iopub.status.idle":"2023-07-05T19:24:19.459907Z","shell.execute_reply.started":"2023-07-05T19:24:08.982472Z","shell.execute_reply":"2023-07-05T19:24:19.458774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Calculate prediction for submission","metadata":{}},{"cell_type":"code","source":"%%time\nY_submit =  model.predict(X_submit)\nprint(Y_submit.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:24:23.14987Z","iopub.execute_input":"2023-07-05T19:24:23.150268Z","iopub.status.idle":"2023-07-05T19:24:24.090456Z","shell.execute_reply.started":"2023-07-05T19:24:23.150233Z","shell.execute_reply":"2023-07-05T19:24:24.089056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission preparations Step 2 - prepare submision in desired format  ","metadata":{}},{"cell_type":"code","source":"%%time \ndf_finalSubmission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:24:27.88341Z","iopub.execute_input":"2023-07-05T19:24:27.884049Z","iopub.status.idle":"2023-07-05T19:24:27.894912Z","shell.execute_reply.started":"2023-07-05T19:24:27.884008Z","shell.execute_reply":"2023-07-05T19:24:27.892764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load protein ids for the submission","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/test_ids.npy'\nvec_test_protein_ids = np.load(fn)\nprint(vec_test_protein_ids.shape)\nvec_test_protein_ids","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:24:31.544019Z","iopub.execute_input":"2023-07-05T19:24:31.544751Z","iopub.status.idle":"2023-07-05T19:24:31.605775Z","shell.execute_reply.started":"2023-07-05T19:24:31.544711Z","shell.execute_reply":"2023-07-05T19:24:31.604755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## \"Melt\" protein ids ","metadata":{}},{"cell_type":"code","source":"%%time \nl = []\nfor k in list(vec_test_protein_ids):\n    l += [ k] * Y_submit.shape[1]\nprint(len(l), l[:20])    \n\ndf_finalSubmission['Protein Id'] = l","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:24:36.252047Z","iopub.execute_input":"2023-07-05T19:24:36.253085Z","iopub.status.idle":"2023-07-05T19:24:41.32096Z","shell.execute_reply.started":"2023-07-05T19:24:36.253047Z","shell.execute_reply":"2023-07-05T19:24:41.319779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time \n# df_finalSubmission.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-04-24T15:33:14.673705Z","iopub.execute_input":"2023-04-24T15:33:14.674078Z","iopub.status.idle":"2023-04-24T15:33:14.679205Z","shell.execute_reply.started":"2023-04-24T15:33:14.674022Z","shell.execute_reply":"2023-04-24T15:33:14.677917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## \"Melt\" Labels (Gene ontology terms )","metadata":{"execution":{"iopub.status.busy":"2023-04-24T12:31:15.267721Z","iopub.execute_input":"2023-04-24T12:31:15.268181Z","iopub.status.idle":"2023-04-24T12:31:15.613742Z","shell.execute_reply.started":"2023-04-24T12:31:15.268143Z","shell.execute_reply":"2023-04-24T12:31:15.612514Z"}}},{"cell_type":"code","source":"df_finalSubmission['GO Term Id'] = labels_to_consider * Y_submit.shape[0]\n# df_finalSubmission.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:24:44.916962Z","iopub.execute_input":"2023-07-05T19:24:44.918011Z","iopub.status.idle":"2023-07-05T19:24:45.9416Z","shell.execute_reply.started":"2023-07-05T19:24:44.917971Z","shell.execute_reply":"2023-07-05T19:24:45.9403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Assign predictions ","metadata":{}},{"cell_type":"code","source":"df_finalSubmission['Prediction'] = Y_submit.ravel()","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:24:49.693018Z","iopub.execute_input":"2023-07-05T19:24:49.694155Z","iopub.status.idle":"2023-07-05T19:24:50.238493Z","shell.execute_reply.started":"2023-07-05T19:24:49.694087Z","shell.execute_reply":"2023-07-05T19:24:50.237231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save ","metadata":{}},{"cell_type":"code","source":"%%time \ndf_finalSubmission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:24:53.559299Z","iopub.execute_input":"2023-07-05T19:24:53.560394Z","iopub.status.idle":"2023-07-05T19:25:46.981798Z","shell.execute_reply.started":"2023-07-05T19:24:53.560344Z","shell.execute_reply":"2023-07-05T19:25:46.980564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show some info ","metadata":{}},{"cell_type":"code","source":"%%time \ndf_finalSubmission.info()","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:26:00.124497Z","iopub.execute_input":"2023-07-05T19:26:00.124955Z","iopub.status.idle":"2023-07-05T19:26:00.145576Z","shell.execute_reply.started":"2023-07-05T19:26:00.124913Z","shell.execute_reply":"2023-07-05T19:26:00.144555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \ndf_finalSubmission.describe()","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:26:02.646155Z","iopub.execute_input":"2023-07-05T19:26:02.646532Z","iopub.status.idle":"2023-07-05T19:26:03.305235Z","shell.execute_reply.started":"2023-07-05T19:26:02.646499Z","shell.execute_reply":"2023-07-05T19:26:03.304151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nplt.figure(figsize = (15,4))\nplt.hist(df_finalSubmission['Prediction'].values, bins = 1000 )\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:26:08.189038Z","iopub.execute_input":"2023-07-05T19:26:08.189768Z","iopub.status.idle":"2023-07-05T19:26:10.705526Z","shell.execute_reply.started":"2023-07-05T19:26:08.189727Z","shell.execute_reply":"2023-07-05T19:26:10.70438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:26:18.159676Z","iopub.execute_input":"2023-07-05T19:26:18.160065Z","iopub.status.idle":"2023-07-05T19:26:18.176239Z","shell.execute_reply.started":"2023-07-05T19:26:18.16003Z","shell.execute_reply":"2023-07-05T19:26:18.175024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )","metadata":{"execution":{"iopub.status.busy":"2023-07-05T19:26:21.565676Z","iopub.execute_input":"2023-07-05T19:26:21.566828Z","iopub.status.idle":"2023-07-05T19:26:21.574037Z","shell.execute_reply.started":"2023-07-05T19:26:21.566773Z","shell.execute_reply":"2023-07-05T19:26:21.572733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}