{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":41875,"databundleVersionId":5521661,"sourceType":"competition"},{"sourceId":5754528,"sourceType":"datasetVersion","datasetId":3198080},{"sourceId":5949133,"sourceType":"datasetVersion","datasetId":3317308},{"sourceId":6130265,"sourceType":"datasetVersion","datasetId":3413125},{"sourceId":6263177,"sourceType":"datasetVersion","datasetId":3520184},{"sourceId":6293005,"sourceType":"datasetVersion","datasetId":3401149},{"sourceId":6306350,"sourceType":"datasetVersion","datasetId":3455276},{"sourceId":13433186,"sourceType":"datasetVersion","datasetId":3628325},{"sourceId":127447831,"sourceType":"kernelVersion"},{"sourceId":127564481,"sourceType":"kernelVersion"},{"sourceId":127591342,"sourceType":"kernelVersion"},{"sourceId":129953701,"sourceType":"kernelVersion"},{"sourceId":130711425,"sourceType":"kernelVersion"},{"sourceId":131320671,"sourceType":"kernelVersion"},{"sourceId":131612344,"sourceType":"kernelVersion"},{"sourceId":131748626,"sourceType":"kernelVersion"},{"sourceId":132158771,"sourceType":"kernelVersion"},{"sourceId":133351227,"sourceType":"kernelVersion"},{"sourceId":133778155,"sourceType":"kernelVersion"},{"sourceId":134814805,"sourceType":"kernelVersion"},{"sourceId":136848294,"sourceType":"kernelVersion"},{"sourceId":139812222,"sourceType":"kernelVersion"},{"sourceId":139821981,"sourceType":"kernelVersion"},{"sourceId":157956682,"sourceType":"kernelVersion"},{"sourceId":179495611,"sourceType":"kernelVersion"}],"dockerImageVersionId":30458,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Notebooks referenced:**\n\ndf_minus_two is [Merge Datasets](https://www.kaggle.com/code/zmcxjt/merge-datasets) version 114 from the Synthetic Goose team in position 4 CAFA5 private score = 0.56245 LB, 0.56245 (code)\n\ndf_minus_one is [ESM1b](https://www.kaggle.com/code/wang0027/esm1b/output) version 111 from the GOCurator team in position 1, CAFA5 private score = 0.61623 LB, 0.61765 (code).\n\ndf_one is adapted by @mogalys900 from: [Fork - Baseline MultiLabel T5 embeds cafa5](https://www.kaggle.com/code/danofer/fork-baseline-multilabel-t5-embeds-cafa5) by @danofer, which is itself adapted from [Baseline MultiLabel to MultiTarget Binary](https://www.kaggle.com/code/alexandervc/baseline-multilabel-to-multitarget-binary) by @alexandervc.\n\ndf_two is from: [keops_knn](https://www.kaggle.com/code/geraseva/keops-knn) by @geraseva.\n\ndf_three is from: [ESM-2 Embeddings Starter](https://www.kaggle.com/code/viktorfairuschin/esm-2-embeddings-starter) by @viktorfairuschin.\n\ndf_four (= df_diamond), df_thirteen (= df_diamsens), df_diamfast, df_diammids, df_diamvery, and df_diamultra are from: [diamond](https://www.kaggle.com/code/geraseva/diamond) by @geraseva; df_diamcomb is adapted from these.\n\ndf_five is from: [Pytorch+Lightning Baseline EMS2 Embeddings+MLP](https://www.kaggle.com/code/henriupton/pytorch-lightning-baseline-ems2-embeddings-mlp) by @henriupton.\n\ndf_six is from: [Baseline 5 folds T5](https://www.kaggle.com/code/simonveitner/baseline-5-folds-t5) by @simonveitner.\n\ndf_seven is from: [Simple ANN approach](https://www.kaggle.com/code/simonveitner/simple-ann-approach) by @simonveitner.\n\ndf_eight is from: [Simple MLP](https://www.kaggle.com/code/simonveitner/simple-mlp) by @simonveitner.\n\ndf_nine is from: [Improve LB with PCA](https://www.kaggle.com/code/simonveitner/improve-lb-with-pca) by @simonveitner.\n\ndf_ten is from: [CAFA 5 protein function with TensorFlow](https://www.kaggle.com/code/gusthema/cafa-5-protein-function-with-tensorflow) by @gusthema.\n\ndf_eleven is from [Pytorch-CAFA 5 Prediction](https://www.kaggle.com/code/averma111/pytorch-cafa-5-prediction) by @averma111.\n\ndf_twelve is from [LinerNet for Protein PyTorch+ProtT5 Embeddings](https://www.kaggle.com/code/horikitasaku/linernet-for-protein-pytorch-prott5-embeddings) by @horikitasaku.\n\ndf_fourteen is from [Multi-Input Model with T5 Species](https://www.kaggle.com/code/joonyoungjang/multi-input-model-with-t5-species) by @joonyoungjang.\n\ndf_fifteen is from [CAFA-5.2](https://www.kaggle.com/code/dplg007/cafa-5-2) by @dplg007.\n\ndf_sixteen is from [CAFA-5.3 Dense Layer NN](https://www.kaggle.com/code/dplg007/cafa-5-3-dense-layer-nn) by @dplg007.\n\ndf_seventeen and df_eighteen are from [BLASTp + SPROF-GO](https://www.kaggle.com/code/samusram/blastp-sprof-go) by @samusram.\n\ndf_nineteen is from [merge_datasets](https://www.kaggle.com/code/mtinti/merge-datasets) by @mtinti.\n\ndf_twenty is from [ESM-2 3B embeddings with three pooling methods](https://www.kaggle.com/code/daehunbae/esm-2-3b-embeddings-with-three-pooling-methods) by @daehunbae.\n\ndf_twenty_one and df_twenty_five are from [CAFA 5 - T5 Embeds](https://www.kaggle.com/code/siddhvr/cafa-5-t5-embeds) by @siddhvr.\n\ndf_twenty_two is from [CombineEmbeddings](https://www.kaggle.com/code/szabo7zoltan/combineembeddings) by @szabo7zoltan.\n\ndf_twenty_three (= df_ngram) is from [Get Nearest Protein by Ngram statistics](https://www.kaggle.com/code/dmitryshibaev/get-nearest-protein-by-ngram-statistics) by @dmitryshibaev.\n\ndf_twenty_four is from [LB 0.53171 Tuning merge_datasets w. higher top](https://www.kaggle.com/code/bibanh/lb-0-53171-tuning-merge-datasets-w-higher-top) by @bibanh.\n\ndf_twenty_six is from [Leveraging Foldseek](https://www.kaggle.com/code/samusram/leveraging-foldseek) by @samusram.\n\ndf_twenty_seven is from [CAFA 5 - T5 Embeds + Ensemble+ zero_predic](https://www.kaggle.com/code/liudacheldieva/cafa-5-t5-embeds-ensemble-zero-predic) by @liudacheldieva.\n\ndf_twenty_eight is from [Foldseek/Blastp + Ensemble](https://www.kaggle.com/code/siddhvr/foldseek-blastp-ensemble) by @siddhvr.\n\ndf_twenty_nine is from [CAFA5- Using ProtBERT Embeds](https://www.kaggle.com/code/siddhvr/cafa5-using-protbert-embeds) by @siddhvr.\n\ndf_thirty is from [CAFA5- EMS2 embeds with Pytorch](https://www.kaggle.com/code/siddhvr/cafa5-ems2-embeds-with-pytorch) by @siddhvr.\n\ndf_thirty_one is from: [merge_datasets](https://www.kaggle.com/code/adaluodao/merge-datasets) by @adaluodao.\n\ndf_comput_annot is from: [test_notebook_quickgo](https://www.kaggle.com/code/mtinti/test-notebook-quickgo) by @mtinti.\n\ndf_deepGO is from: [deepGO](https://www.kaggle.com/code/geraseva/deepgo) by @geraseva.\n\ndf_deepGOplus is from: [deepGOPlus](https://www.kaggle.com/code/geraseva/deepgoplus) by @geraseva.\n\ndf_sprof is from: [sprof_predictions](https://www.kaggle.com/code/mtinti/sprof-predictions) by @mtinti.\n\n","metadata":{}},{"cell_type":"markdown","source":"df_one to df_twenty_eight, df_sprof, df_comput_annot, df_deepGO and df_deepGOplus are contained within df_zero.","metadata":{}},{"cell_type":"code","source":"df_minus_two = pd.read_csv('/kaggle/input/cafa5-fifth-set/zmcxjt_CAFA5_v114_submission.tsv', sep='\\t', header=None)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_minus_two.dropna(inplace= True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_minus_two.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_minus_one = pd.read_csv('/kaggle/input/cafa5-fifth-set/wang0027_CAFA5_v111_submission.tsv', sep='\\t', header=None)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_minus_one.dropna(inplace= True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_minus_one.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_zero = pd.read_csv('/kaggle/input/part-six-d/submission.tsv', sep='\\t', header=None)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_zero.shape","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_zero.dropna(inplace= True)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_zero.shape","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_zero = df_zero.sort_values(by=[0],ascending=True,axis=0)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_zero.head(30)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_zero.tail(30)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_zero[2].median()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_thirty_one = pd.read_csv('/kaggle/input/cafa-5-file-larder/57160.submission.tsv', sep='\\t', skiprows=0, header=None)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_thirty_one.shape","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_thirty_one = df_thirty_one[df_thirty_one[2]>0.1]","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_thirty_one.shape","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_thirty_one[2].median()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_thirty_one.head(30)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_presubmission = pd.concat([df_minus_two, df_minus_one, df_zero, df_thirty_one])","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_presubmission = df_presubmission.sort_values(by=[2],ascending=False,axis=0)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_presubmission = df_presubmission.drop_duplicates(subset=[0], keep='first')","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_presubmission = df_presubmission.sort_values(by=[0],ascending=True,axis=0)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_presubmission.loc[:,2] = np.clip(0.001, df_presubmission.loc[:,2], 1.0)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_presubmission[2] = df_presubmission[2].round(decimals=3)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_presubmission.shape","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_presubmission[2].median()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_presubmission.head(30)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_presubmission.tail(30)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_presubmission = df_presubmission.drop(df_presubmission.columns[[1,2]], axis=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_presubmission.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_presubmission.to_csv(\"codelist.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}