{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-30T15:37:24.020766Z","iopub.execute_input":"2023-05-30T15:37:24.021152Z","iopub.status.idle":"2023-05-30T15:37:24.053825Z","shell.execute_reply.started":"2023-05-30T15:37:24.021116Z","shell.execute_reply":"2023-05-30T15:37:24.052857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport progressbar","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:37:24.203141Z","iopub.execute_input":"2023-05-30T15:37:24.203518Z","iopub.status.idle":"2023-05-30T15:37:32.137586Z","shell.execute_reply.started":"2023-05-30T15:37:24.203489Z","shell.execute_reply":"2023-05-30T15:37:32.136616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Data","metadata":{}},{"cell_type":"code","source":"#Load Data \n#Terms for each protein fold\ntrain_terms = pd.read_csv('/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv',sep='\\t')\n#Embeddings for each aminoacid_sequence\ntrain_embeddings = np.load('/kaggle/input/t5embeds/train_embeds.npy')\n#Protein ID's for the embeddings\ntrain_id = np.load('/kaggle/input/t5embeds/train_ids.npy')","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:37:32.139496Z","iopub.execute_input":"2023-05-30T15:37:32.140278Z","iopub.status.idle":"2023-05-30T15:37:43.977891Z","shell.execute_reply.started":"2023-05-30T15:37:32.140243Z","shell.execute_reply":"2023-05-30T15:37:43.976792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_id.shape,train_embeddings.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:37:43.983303Z","iopub.execute_input":"2023-05-30T15:37:43.985709Z","iopub.status.idle":"2023-05-30T15:37:43.997366Z","shell.execute_reply.started":"2023-05-30T15:37:43.985673Z","shell.execute_reply":"2023-05-30T15:37:43.99653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert embeddings numpy array(train_embeddings) into pandas dataframe.\ncolumn_num = train_embeddings.shape[1]\ntrain_df = pd.DataFrame(train_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\ntrain_df['ID'] = train_id","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:37:44.003393Z","iopub.execute_input":"2023-05-30T15:37:44.005957Z","iopub.status.idle":"2023-05-30T15:37:44.046115Z","shell.execute_reply.started":"2023-05-30T15:37:44.005922Z","shell.execute_reply":"2023-05-30T15:37:44.045254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploratory Analysis","metadata":{}},{"cell_type":"code","source":"train_terms","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:37:44.050323Z","iopub.execute_input":"2023-05-30T15:37:44.052695Z","iopub.status.idle":"2023-05-30T15:37:44.072141Z","shell.execute_reply.started":"2023-05-30T15:37:44.052661Z","shell.execute_reply":"2023-05-30T15:37:44.07124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms['term'].unique().shape","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:37:44.076322Z","iopub.execute_input":"2023-05-30T15:37:44.078614Z","iopub.status.idle":"2023-05-30T15:37:44.734381Z","shell.execute_reply.started":"2023-05-30T15:37:44.078581Z","shell.execute_reply":"2023-05-30T15:37:44.733429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:37:44.738806Z","iopub.execute_input":"2023-05-30T15:37:44.741568Z","iopub.status.idle":"2023-05-30T15:37:44.902035Z","shell.execute_reply.started":"2023-05-30T15:37:44.741533Z","shell.execute_reply":"2023-05-30T15:37:44.901011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Number of different proteins with the same aminoacid sequence embedding \ndf_duplicated = train_df[train_df.loc[:, train_df.columns != 'ID'].duplicated(keep=False)]\n#Flag the pairs \ndf_duplicated['DuplicateGroup'] = df_duplicated.groupby([col for col in df_duplicated.columns]).ngroup()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:37:44.90696Z","iopub.execute_input":"2023-05-30T15:37:44.907879Z","iopub.status.idle":"2023-05-30T15:38:11.4562Z","shell.execute_reply.started":"2023-05-30T15:37:44.907842Z","shell.execute_reply":"2023-05-30T15:38:11.455066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_duplicated.shape[0])","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:11.457516Z","iopub.execute_input":"2023-05-30T15:38:11.459361Z","iopub.status.idle":"2023-05-30T15:38:11.464578Z","shell.execute_reply.started":"2023-05-30T15:38:11.459322Z","shell.execute_reply":"2023-05-30T15:38:11.463526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Collect all terms of a EntryID inside one row through a list\ndf_duplicated_terms = train_terms[train_terms['EntryID'].isin(df_duplicated['ID'])].groupby('EntryID')['term'].apply(list).reset_index(name='terms_collected')\ndf_duplicated_terms['duplicated_sequence_group'] = df_duplicated_terms.merge(df_duplicated, left_on='EntryID', right_on='ID')['DuplicateGroup']","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:11.469415Z","iopub.execute_input":"2023-05-30T15:38:11.470077Z","iopub.status.idle":"2023-05-30T15:38:12.507744Z","shell.execute_reply.started":"2023-05-30T15:38:11.470044Z","shell.execute_reply":"2023-05-30T15:38:12.506786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_duplicated_terms","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:12.509521Z","iopub.execute_input":"2023-05-30T15:38:12.509888Z","iopub.status.idle":"2023-05-30T15:38:12.527343Z","shell.execute_reply.started":"2023-05-30T15:38:12.509855Z","shell.execute_reply":"2023-05-30T15:38:12.526251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Number of the same amnoacid sequence and all the same go-terms\ndf_duplicated_terms['terms_collected'] = df_duplicated_terms['terms_collected'].apply(tuple)\ndf_dp_count = df_duplicated_terms[df_duplicated_terms[['terms_collected','duplicated_sequence_group']].duplicated(keep=False)] #Count the number of proteins with same sequence and same go-terms","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:12.528899Z","iopub.execute_input":"2023-05-30T15:38:12.5305Z","iopub.status.idle":"2023-05-30T15:38:12.561927Z","shell.execute_reply.started":"2023-05-30T15:38:12.530466Z","shell.execute_reply":"2023-05-30T15:38:12.561106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Different proteins that have different functions although having the same sequence\ndf_dp_count","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:12.563356Z","iopub.execute_input":"2023-05-30T15:38:12.563767Z","iopub.status.idle":"2023-05-30T15:38:12.572717Z","shell.execute_reply.started":"2023-05-30T15:38:12.563732Z","shell.execute_reply":"2023-05-30T15:38:12.571778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Distribution betwen aspects\npie_df = train_terms['aspect'].value_counts()\npalette_color = sns.color_palette('bright')\nplt.pie(pie_df.values, labels=np.array(pie_df.index), colors=palette_color, autopct='%.0f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:12.574352Z","iopub.execute_input":"2023-05-30T15:38:12.575242Z","iopub.status.idle":"2023-05-30T15:38:13.408802Z","shell.execute_reply.started":"2023-05-30T15:38:12.575209Z","shell.execute_reply":"2023-05-30T15:38:13.407764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from Bio import SeqIO\n\n# Specify the path to your FASTA file\nfasta_file = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta'\n\n# Read the FASTA file\nsequences = []\nfor record in SeqIO.parse(fasta_file, \"fasta\"):\n    # Access the sequence ID and sequence data\n    sequence_id = record.id\n    sequence_data = record.seq\n\n    # Add the sequence to the list\n    sequences.append((sequence_id, sequence_data))","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:13.410549Z","iopub.execute_input":"2023-05-30T15:38:13.411671Z","iopub.status.idle":"2023-05-30T15:38:16.234404Z","shell.execute_reply.started":"2023-05-30T15:38:13.411634Z","shell.execute_reply":"2023-05-30T15:38:16.233441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(sequences), sequences[0],sequences[0][1]","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:16.237951Z","iopub.execute_input":"2023-05-30T15:38:16.238254Z","iopub.status.idle":"2023-05-30T15:38:16.247346Z","shell.execute_reply.started":"2023-05-30T15:38:16.238228Z","shell.execute_reply":"2023-05-30T15:38:16.246398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preparing Data","metadata":{}},{"cell_type":"code","source":"# Set the limit for label\nnum_of_labels = 1500\ntrain_size = train_id.shape[0] # len(X)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:16.24881Z","iopub.execute_input":"2023-05-30T15:38:16.24917Z","iopub.status.idle":"2023-05-30T15:38:16.264349Z","shell.execute_reply.started":"2023-05-30T15:38:16.249138Z","shell.execute_reply":"2023-05-30T15:38:16.263235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import pad_sequences\n\n# Define the dictionary mapping for tokenization\namino_acid_dict = {\n    'A': 1, 'R': 2, 'N': 3, 'D': 4, 'C': 5, 'Q': 6, 'E': 7, 'G': 8,\n    'H': 9, 'I': 10, 'L': 11, 'K': 12, 'M': 13, 'F': 14, 'P': 15, 'S': 16,\n    'T': 17, 'W': 18, 'Y': 19, 'V': 20, 'B': 21, 'Z': 22, 'X': 23, 'U': 24,\n    'O': 25\n}\namnoacid_sequences = []\n\n# Define the maximum sequence length\nmax_sequence_length = 400\n\n# Loop through each label\nfor sequence in sequences:\n#     print(sequence[1])\n    # Convert sequences to integer tokens\n    tokenized_seq = [amino_acid_dict[aa] for aa in sequence[1]]\n    \n    amnoacid_sequences.append(tokenized_seq)\n    \n# Pad or truncate sequences to the desired length\namnoacid_sequences = pad_sequences(amnoacid_sequences, maxlen=max_sequence_length, padding='post', truncating='post', value=0)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:16.265905Z","iopub.execute_input":"2023-05-30T15:38:16.266324Z","iopub.status.idle":"2023-05-30T15:38:27.06552Z","shell.execute_reply.started":"2023-05-30T15:38:16.266293Z","shell.execute_reply":"2023-05-30T15:38:27.064494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amnoacid_sequences.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:27.066835Z","iopub.execute_input":"2023-05-30T15:38:27.067209Z","iopub.status.idle":"2023-05-30T15:38:27.07598Z","shell.execute_reply.started":"2023-05-30T15:38:27.067171Z","shell.execute_reply":"2023-05-30T15:38:27.075033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Take value counts in descending order and fetch first 1500 `GO term ID` as labels\nlabels = train_terms['term'].value_counts().index[:num_of_labels].tolist()\n\n# Fetch the train_terms data for the relevant labels only\ntrain_terms_updated = train_terms.loc[train_terms['term'].isin(labels)]\n\n# Setup progressbar settings.\nbar = progressbar.ProgressBar(maxval=num_of_labels, \\\n    widgets=[progressbar.Bar('=', '[', ']'), ' ', progressbar.Percentage()])\n\n# Create an empty dataframe of required size for storing the labels,\ntrain_labels = np.zeros((train_size ,num_of_labels))\nseries_train_protein_ids = pd.Series(train_id)\n\n# Loop through each label\nfor i in range(num_of_labels):\n    # For each label, fetch the corresponding train_terms data\n    n_train_terms = train_terms_updated[train_terms_updated['term'] ==  labels[i]]\n    \n    # Fetch all the unique EntryId aka proteins related to the current label(GO term ID)\n    label_related_proteins = n_train_terms['EntryID'].unique()\n    \n    # In the series_train_protein_ids pandas series, if a protein is related\n    # to the current label, then mark it as 1, else 0.\n    # Replace the ith column of train_Y with with that pandas series.\n    train_labels[:,i] =  series_train_protein_ids.isin(label_related_proteins).astype(float)\n    \n    # Progress bar percentage increase\n    bar.update(i+1)\n\n# Notify the end of progress bar \nbar.finish()\n\n# Convert train_Y numpy into pandas dataframe\nlabels_df = pd.DataFrame(data = train_labels, columns = labels)\nprint(labels_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:27.07765Z","iopub.execute_input":"2023-05-30T15:38:27.07849Z","iopub.status.idle":"2023-05-30T15:54:35.262865Z","shell.execute_reply.started":"2023-05-30T15:38:27.078457Z","shell.execute_reply":"2023-05-30T15:54:35.261935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract input features and labels from the DataFrame\nfeatures_input = train_df.loc[:, train_df.columns != 'ID'].values  # Extract the values from the DataFrame\nlabels_input = labels_df.values  # Extract the label column\n","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:54:35.264286Z","iopub.execute_input":"2023-05-30T15:54:35.264645Z","iopub.status.idle":"2023-05-30T15:54:35.645577Z","shell.execute_reply.started":"2023-05-30T15:54:35.264613Z","shell.execute_reply":"2023-05-30T15:54:35.644416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_input","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:54:35.64697Z","iopub.execute_input":"2023-05-30T15:54:35.648026Z","iopub.status.idle":"2023-05-30T15:54:35.659317Z","shell.execute_reply.started":"2023-05-30T15:54:35.647991Z","shell.execute_reply":"2023-05-30T15:54:35.658382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Eval in test data \ntest_embeddings = np.load('/kaggle/input/t5embeds/test_embeds.npy')\n\n# Convert test_embeddings to dataframe\ncolumn_num = test_embeddings.shape[1]\ntest_df = pd.DataFrame(test_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\nprint(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:54:35.660612Z","iopub.execute_input":"2023-05-30T15:54:35.661105Z","iopub.status.idle":"2023-05-30T15:54:45.181587Z","shell.execute_reply.started":"2023-05-30T15:54:35.66106Z","shell.execute_reply":"2023-05-30T15:54:45.180646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ResLSTSM model","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from keras.models import Model\nfrom keras.layers import LSTM, Conv1D, MaxPooling1D, Flatten, Dense, Bidirectional, Dropout, Add,Input, Embedding\nimport tensorflow as tf\n\n# Check if GPU devices are available\nphysical_devices = tf.config.list_physical_devices('GPU')\nnum_gpus = len(physical_devices)\nprint(\"Number of available GPUs:\", num_gpus)\n\n\n# Use MirroredStrategy for multi-GPU training\nstrategy = tf.distribute.MirroredStrategy()\nwith strategy.scope():\n    # Define the model\n    # Input layer\n    inputs = Input(shape=(1024, 1))\n\n    # Residual block\n    residual = inputs\n    \n    # LSTM layer\n    x = Bidirectional(LSTM(64, return_sequences=False))(inputs)\n    x = Add()([x, residual])\n\n    # Flatten layer\n    x = Flatten()(x)\n\n    # Dense layers\n    x = Dense(256, activation='relu')(x)\n    outputs = Dense(1500, activation='sigmoid')(x)\n\n    # Create the model\n    model = Model(inputs=inputs, outputs=outputs)\n\n    # Compile the model\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\n    # Print the model summary\n    model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T13:34:51.196794Z","iopub.execute_input":"2023-05-30T13:34:51.1974Z","iopub.status.idle":"2023-05-30T13:34:54.490361Z","shell.execute_reply.started":"2023-05-30T13:34:51.197367Z","shell.execute_reply":"2023-05-30T13:34:54.489604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"if num_gpus < 2:\n    print(\"Not enough GPUs available. Training on a single GPU.\")\n    history = model.fit(features_input, labels_input, epochs=10, batch_size=1024)# Train the model with GPU acceleration\nelse:\n    # Use MirroredStrategy for multi-GPU training\n    with strategy.scope():\n        history = model.fit(features_input, labels_input, epochs=10, batch_size=1024)# Train the model with GPU acceleration","metadata":{"execution":{"iopub.status.busy":"2023-05-30T13:34:54.491498Z","iopub.execute_input":"2023-05-30T13:34:54.491856Z","iopub.status.idle":"2023-05-30T13:41:25.965366Z","shell.execute_reply.started":"2023-05-30T13:34:54.491822Z","shell.execute_reply":"2023-05-30T13:41:25.9643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_input.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-30T13:41:25.96703Z","iopub.execute_input":"2023-05-30T13:41:25.967358Z","iopub.status.idle":"2023-05-30T13:41:25.979263Z","shell.execute_reply.started":"2023-05-30T13:41:25.96733Z","shell.execute_reply":"2023-05-30T13:41:25.978119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, ['accuracy']].plot(title=\"Accuracy\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T13:41:25.984987Z","iopub.execute_input":"2023-05-30T13:41:25.985279Z","iopub.status.idle":"2023-05-30T13:41:26.539185Z","shell.execute_reply.started":"2023-05-30T13:41:25.985256Z","shell.execute_reply":"2023-05-30T13:41:26.538183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tryng a CNN-LSTM","metadata":{}},{"cell_type":"code","source":"from keras.models import Sequential\n\n\n# Use MirroredStrategy for multi-GPU training\nstrategy = tf.distribute.MirroredStrategy()\nwith strategy.scope():\n    # Create a sequential model\n    model_CNN_LSTM = Sequential()\n    \n    # Add a Conv1D layer for spatial pattern detection\n    model_CNN_LSTM.add(Conv1D(32, kernel_size=3, input_shape = (1024,1), activation='relu'))\n    model_CNN_LSTM.add(MaxPooling1D(pool_size=2))\n    \n    # Add an LSTM layer for sequence modeling\n    model_CNN_LSTM.add(LSTM(units=64, dropout=0.2, recurrent_dropout=0.2))\n    \n    # Add another Dense layer for non-linear transformations\n    model_CNN_LSTM.add(Dense(128, activation='relu'))\n\n    # Add a fully connected layer for classification\n    model_CNN_LSTM.add(Dense(1500, activation='sigmoid'))\n    \n    # Compile the model\n    model_CNN_LSTM.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\n    # Print the model summary\n    model_CNN_LSTM.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T13:41:26.54077Z","iopub.execute_input":"2023-05-30T13:41:26.541133Z","iopub.status.idle":"2023-05-30T13:41:26.778205Z","shell.execute_reply.started":"2023-05-30T13:41:26.5411Z","shell.execute_reply":"2023-05-30T13:41:26.777431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if num_gpus < 2:\n    print(\"Not enough GPUs available. Training on a single GPU.\")\n    history_CNN = model_CNN_LSTM.fit(features_input, labels_input, epochs=10, batch_size=1024)# Train the model with GPU acceleration\nelse:\n    # Use MirroredStrategy for multi-GPU training\n    with strategy.scope():\n        history_CNN = model_CNN_LSTM.fit(features_input, labels_input, epochs=10, batch_size=1024)# Train the model with GPU acceleration","metadata":{"execution":{"iopub.status.busy":"2023-05-30T13:41:26.779473Z","iopub.execute_input":"2023-05-30T13:41:26.779815Z","iopub.status.idle":"2023-05-30T14:32:14.014075Z","shell.execute_reply.started":"2023-05-30T13:41:26.779782Z","shell.execute_reply":"2023-05-30T14:32:14.013133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df = pd.DataFrame(history_CNN.history)\nhistory_df.loc[:, ['loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, ['accuracy']].plot(title=\"Accuracy\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T14:32:14.015667Z","iopub.execute_input":"2023-05-30T14:32:14.01604Z","iopub.status.idle":"2023-05-30T14:32:14.555893Z","shell.execute_reply.started":"2023-05-30T14:32:14.016006Z","shell.execute_reply":"2023-05-30T14:32:14.554999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Trying a own embedding","metadata":{}},{"cell_type":"code","source":"num_aminoacids = 25\nsequence_length = amnoacid_sequences.shape[1]\n\n# Use MirroredStrategy for multi-GPU training\nstrategy = tf.distribute.MirroredStrategy()\nwith strategy.scope():\n    # Create a sequential model\n    model_emb_CNN_LSTM = Sequential()\n    \n    # Add an Embedding layer to transform the input sequence\n    model_emb_CNN_LSTM.add(Embedding(num_aminoacids, 1024, input_length=sequence_length))\n    \n    # Add a Conv1D layer for spatial pattern detection\n    model_emb_CNN_LSTM.add(Conv1D(32, kernel_size=3, input_shape = (1024,1), activation='relu'))\n    model_emb_CNN_LSTM.add(MaxPooling1D(pool_size=2))\n    \n#     # Flatten the output of the Conv1D layer\n#     model_emb_CNN_LSTM.add(Flatten())\n    \n    # Add an LSTM layer for sequence modeling\n    model_emb_CNN_LSTM.add(LSTM(units=64, dropout=0.2, recurrent_dropout=0.2))\n    \n    # Add another Dense layer for non-linear transformations\n    model_emb_CNN_LSTM.add(Dense(128, activation='relu'))\n\n    # Add a fmully connected layer for classification\n    model_emb_CNN_LSTM.add(Dense(1500, activation='sigmoid'))\n    \n    # Compile the model\n    model_emb_CNN_LSTM.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\n    # Print the model summary\n    model_emb_CNN_LSTM.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T14:32:14.557393Z","iopub.execute_input":"2023-05-30T14:32:14.557987Z","iopub.status.idle":"2023-05-30T14:32:14.793961Z","shell.execute_reply.started":"2023-05-30T14:32:14.557952Z","shell.execute_reply":"2023-05-30T14:32:14.793216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_emb_CNN_LSTM.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T14:32:14.794967Z","iopub.execute_input":"2023-05-30T14:32:14.795295Z","iopub.status.idle":"2023-05-30T14:32:14.816265Z","shell.execute_reply.started":"2023-05-30T14:32:14.795263Z","shell.execute_reply":"2023-05-30T14:32:14.815563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if num_gpus < 2:\n    print(\"Not enough GPUs available. Training on a single GPU.\")\n    history_emb_CNN = model_emb_CNN_LSTM.fit(amnoacid_sequences, labels_input, epochs=10, batch_size=1024)# Train the model with GPU acceleration\nelse:\n    # Use MirroredStrategy for multi-GPU training\n    with strategy.scope():\n        history_emb_CNN = model_emb_CNN_LSTM.fit(amnoacid_sequences, labels_input, epochs=10, batch_size=1024)# Train the model with GPU acceleration","metadata":{"execution":{"iopub.status.busy":"2023-05-30T14:35:34.679984Z","iopub.execute_input":"2023-05-30T14:35:34.680388Z","iopub.status.idle":"2023-05-30T14:58:03.057849Z","shell.execute_reply.started":"2023-05-30T14:35:34.680357Z","shell.execute_reply":"2023-05-30T14:58:03.056606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df = pd.DataFrame(history_emb_CNN.history)\nhistory_df.loc[:, ['loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, ['accuracy']].plot(title=\"Accuracy\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T14:58:03.05999Z","iopub.execute_input":"2023-05-30T14:58:03.060842Z","iopub.status.idle":"2023-05-30T14:58:03.630688Z","shell.execute_reply.started":"2023-05-30T14:58:03.060806Z","shell.execute_reply":"2023-05-30T14:58:03.629813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}