{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install protein-bert pyfastx","metadata":{"execution":{"iopub.status.busy":"2023-04-19T11:06:34.035471Z","iopub.execute_input":"2023-04-19T11:06:34.035906Z","iopub.status.idle":"2023-04-19T11:06:42.071767Z","shell.execute_reply.started":"2023-04-19T11:06:34.03586Z","shell.execute_reply":"2023-04-19T11:06:42.070653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pyfastx\nimport pandas as pd\n\nfrom tensorflow import keras\n\nfrom sklearn.model_selection import train_test_split\n\nfrom proteinbert import OutputType, OutputSpec, FinetuningModelGenerator, load_pretrained_model, finetune\nfrom proteinbert.conv_and_global_attention_model import get_model_with_hidden_layers_as_outputs","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare training data","metadata":{}},{"cell_type":"code","source":"train_set_file_path = '../input/cafa-5-protein-function-prediction/Train/train_terms.tsv'\ntrain_set = pd.read_csv(train_set_file_path, sep='\\t').dropna().drop_duplicates()\n\nUNIQUE_LABELS = train_set['term'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T11:06:42.073364Z","iopub.execute_input":"2023-04-19T11:06:42.074494Z","iopub.status.idle":"2023-04-19T11:06:48.935165Z","shell.execute_reply.started":"2023-04-19T11:06:42.074452Z","shell.execute_reply":"2023-04-19T11:06:48.934078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load FASTA sequences into the train set","metadata":{}},{"cell_type":"code","source":"# Copy train FASTA to writable directory to build index later\n![ ! -f train_sequences.fasta ] && cp ../input/cafa-5-protein-function-prediction/Train/train_sequences.fasta ./\n\n# This will build an index file at /kaggle/working/train_sequences.fasta.fxi\nfa = pyfastx.Fasta('train_sequences.fasta')\n\n# Each fa[x].seq still does a disk queries so we avoid repeating them\n# This will only make 140k queries instead of 5 millions (99.97% of them are duplicated)\nseqs = {x: fa[x].seq for x in train_set['EntryID'].unique()}\ntrain_set['seq'] = train_set['EntryID'].map(lambda x: seqs[x])","metadata":{"execution":{"iopub.status.busy":"2023-04-19T11:06:48.937437Z","iopub.execute_input":"2023-04-19T11:06:48.937734Z","iopub.status.idle":"2023-04-19T11:06:56.240388Z","shell.execute_reply.started":"2023-04-19T11:06:48.937707Z","shell.execute_reply":"2023-04-19T11:06:56.239089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train/validate split","metadata":{}},{"cell_type":"code","source":"train_set, valid_set = train_test_split(train_set, stratify = train_set['term'], test_size = 0.1, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T11:06:56.242537Z","iopub.execute_input":"2023-04-19T11:06:56.242979Z","iopub.status.idle":"2023-04-19T11:07:06.002573Z","shell.execute_reply.started":"2023-04-19T11:06:56.242929Z","shell.execute_reply":"2023-04-19T11:07:06.001124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Finetune ProteinBERT (adapted from [ProteinBERT demo notebook](https://github.com/nadavbra/protein_bert/blob/master/ProteinBERT%20demo.ipynb))","metadata":{}},{"cell_type":"code","source":"# Pretrained ProteinBERT weights\n![ ! -f ./epoch_92400_sample_23500000.pkl ] && wget ftp://ftp.cs.huji.ac.il/users/nadavb/protein_bert/epoch_92400_sample_23500000.pkl\n\n# In the ProteinBERT demo, output_type is `binary`, but our GO terms are categorical instead\nOUTPUT_TYPE = OutputType(is_seq = False, output_type = 'categorical')\nOUTPUT_SPEC = OutputSpec(OUTPUT_TYPE, UNIQUE_LABELS)\n\npretrained_model_generator, input_encoder = load_pretrained_model(\n    local_model_dump_dir = './',\n    local_model_dump_file_name = 'epoch_92400_sample_23500000.pkl'\n)\n\nmodel_generator = FinetuningModelGenerator(pretrained_model_generator, OUTPUT_SPEC, pretraining_model_manipulation_function = \\\n        get_model_with_hidden_layers_as_outputs, dropout_rate = 0.5)\n\ntraining_callbacks = [\n    keras.callbacks.ReduceLROnPlateau(patience = 1, factor = 0.25, min_lr = 1e-05, verbose = 1),\n    keras.callbacks.EarlyStopping(patience = 2, restore_best_weights = True),\n]\n\nfinetune(model_generator, input_encoder, OUTPUT_SPEC, train_set['seq'], train_set['term'], valid_set['seq'], valid_set['term'], \\\n        seq_len = 512, batch_size = 32, max_epochs_per_stage = 40, lr = 1e-04, begin_with_frozen_pretrained_layers = True, \\\n        lr_with_frozen_pretrained_layers = 1e-02, n_final_epochs = 1, final_seq_len = 1024, final_lr = 1e-05, callbacks = training_callbacks)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-19T11:08:05.52693Z","iopub.execute_input":"2023-04-19T11:08:05.527312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}