{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":41875,"databundleVersionId":5521661,"sourceType":"competition"},{"sourceId":5499219,"sourceType":"datasetVersion","datasetId":3167603},{"sourceId":8051704,"sourceType":"datasetVersion","datasetId":4748350},{"sourceId":10844181,"sourceType":"datasetVersion","datasetId":6734712}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pyvis\n!pip install obonet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:19:12.355017Z","iopub.execute_input":"2025-04-24T05:19:12.355414Z","iopub.status.idle":"2025-04-24T05:19:20.986556Z","shell.execute_reply.started":"2025-04-24T05:19:12.355347Z","shell.execute_reply":"2025-04-24T05:19:20.985525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install biopython\n!pip install progressbar","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:19:20.988321Z","iopub.execute_input":"2025-04-24T05:19:20.988625Z","iopub.status.idle":"2025-04-24T05:19:31.113704Z","shell.execute_reply.started":"2025-04-24T05:19:20.9886Z","shell.execute_reply":"2025-04-24T05:19:31.112848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install tensorflow","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:19:31.115011Z","iopub.execute_input":"2025-04-24T05:19:31.115319Z","iopub.status.idle":"2025-04-24T05:19:34.546407Z","shell.execute_reply.started":"2025-04-24T05:19:31.115294Z","shell.execute_reply":"2025-04-24T05:19:34.545601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport progressbar","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:19:34.547553Z","iopub.execute_input":"2025-04-24T05:19:34.547878Z","iopub.status.idle":"2025-04-24T05:19:48.520182Z","shell.execute_reply.started":"2025-04-24T05:19:34.547848Z","shell.execute_reply":"2025-04-24T05:19:48.519457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_terms=pd.read_csv('/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv', sep=\"\\t\")\ntrain_terms.head()\ntrain_terms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:19:48.521107Z","iopub.execute_input":"2025-04-24T05:19:48.521907Z","iopub.status.idle":"2025-04-24T05:19:51.63529Z","shell.execute_reply.started":"2025-04-24T05:19:48.52187Z","shell.execute_reply":"2025-04-24T05:19:51.634545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_terms.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:19:51.636108Z","iopub.execute_input":"2025-04-24T05:19:51.636341Z","iopub.status.idle":"2025-04-24T05:19:51.641791Z","shell.execute_reply.started":"2025-04-24T05:19:51.636321Z","shell.execute_reply":"2025-04-24T05:19:51.64074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nfrom typing import Dict\nfrom collections import Counter\n\nimport random\nimport obonet\nimport pandas as pd\nimport numpy as np\nfrom Bio import SeqIO\nimport re\nfrom transformers import T5Tokenizer, T5EncoderModel\nimport torch\ndevice = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\n\n# Load the tokenizer\ntokenizer = T5Tokenizer.from_pretrained('Rostlab/prot_t5_xl_half_uniref50-enc', do_lower_case=False) #.to(device)\n\n# Load the model\nmodel = T5EncoderModel.from_pretrained(\"Rostlab/prot_t5_xl_half_uniref50-enc\").to(device)\n\ndef get_embeddings(seq):\n    sequence_examples = [\" \".join(list(re.sub(r\"[UZOB]\", \"X\", seq)))]\n\n    ids = tokenizer.batch_encode_plus(sequence_examples, add_special_tokens=True, padding=\"longest\")\n\n    input_ids = torch.tensor(ids['input_ids']).to(device)\n    attention_mask = torch.tensor(ids['attention_mask']).to(device)\n\n    # generate embeddings\n    with torch.no_grad():\n        embedding_repr = model(input_ids=input_ids,\n                               attention_mask=attention_mask)\n\n    # extract residue embeddings for the first ([0,:]) sequence in the batch and remove padded & special tokens ([0,:7]) \n    emb_0 = embedding_repr.last_hidden_state[0]\n    emb_0_per_protein = emb_0.mean(dim=0)\n    \n    return emb_0_per_protein\n\nfile = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta'\nsequences = SeqIO.parse(file, \"fasta\")\nprint(sequences)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:19:51.645297Z","iopub.execute_input":"2025-04-24T05:19:51.645551Z","iopub.status.idle":"2025-04-24T05:20:16.843144Z","shell.execute_reply.started":"2025-04-24T05:19:51.645531Z","shell.execute_reply":"2025-04-24T05:20:16.84196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(next(iter(sequences)).seq)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:20:16.845697Z","iopub.execute_input":"2025-04-24T05:20:16.846861Z","iopub.status.idle":"2025-04-24T05:20:16.852885Z","shell.execute_reply.started":"2025-04-24T05:20:16.846818Z","shell.execute_reply":"2025-04-24T05:20:16.851916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sequences = SeqIO.parse(file, \"fasta\")\nsequences\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:20:16.854127Z","iopub.execute_input":"2025-04-24T05:20:16.854547Z","iopub.status.idle":"2025-04-24T05:20:17.786121Z","shell.execute_reply.started":"2025-04-24T05:20:16.854513Z","shell.execute_reply":"2025-04-24T05:20:17.785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"get_embeddings(str(next(iter(sequences)).seq))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:20:17.78709Z","iopub.execute_input":"2025-04-24T05:20:17.787434Z","iopub.status.idle":"2025-04-24T05:20:19.730297Z","shell.execute_reply.started":"2025-04-24T05:20:17.787402Z","shell.execute_reply":"2025-04-24T05:20:19.729233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_protein_ids = np.load('/kaggle/input/t5embeds/train_ids.npy')\ntrain_protein_ids_pd=pd.DataFrame(train_protein_ids)\ntrain_protein_ids_pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:20:19.731367Z","iopub.execute_input":"2025-04-24T05:20:19.731731Z","iopub.status.idle":"2025-04-24T05:20:19.831312Z","shell.execute_reply.started":"2025-04-24T05:20:19.731699Z","shell.execute_reply":"2025-04-24T05:20:19.830488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_protein_ids_pd.size","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:20:19.832309Z","iopub.execute_input":"2025-04-24T05:20:19.832594Z","iopub.status.idle":"2025-04-24T05:20:21.640741Z","shell.execute_reply.started":"2025-04-24T05:20:19.832571Z","shell.execute_reply":"2025-04-24T05:20:21.63968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_embeddings = np.load('/kaggle/input/t5embeds/train_embeds.npy')\ncolumn_num = train_embeddings.shape[1]\ntrain_df  = pd.DataFrame(train_embeddings, columns = [\"Colomn \" + str(i) for i in range(1, column_num+1)])\ntrain_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:20:21.641779Z","iopub.execute_input":"2025-04-24T05:20:21.642165Z","iopub.status.idle":"2025-04-24T05:20:29.385292Z","shell.execute_reply.started":"2025-04-24T05:20:21.642126Z","shell.execute_reply":"2025-04-24T05:20:29.384456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#There are more than 50,000 labels in the train_terms.tsv file, so we will choose the most frequent 1500 GO term IDs as labels of the dataset.\n#Extracting Go IDs\nnum_of_labels = 1500\nlabels = train_terms['term'].value_counts().index\nlabels=labels[:num_of_labels]\nlabels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:20:29.386177Z","iopub.execute_input":"2025-04-24T05:20:29.386492Z","iopub.status.idle":"2025-04-24T05:20:29.761643Z","shell.execute_reply.started":"2025-04-24T05:20:29.386455Z","shell.execute_reply":"2025-04-24T05:20:29.760722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extracting the dataset which contains the top 1500(labels) GO terms\ntrain_terms_updated = train_terms.loc[train_terms['term'].isin(labels)]\ntrain_terms_updated=train_terms_updated.reset_index(drop=True)\ntrain_terms_updated","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:20:29.762542Z","iopub.execute_input":"2025-04-24T05:20:29.762844Z","iopub.status.idle":"2025-04-24T05:20:30.331392Z","shell.execute_reply.started":"2025-04-24T05:20:29.762807Z","shell.execute_reply":"2025-04-24T05:20:30.330636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bar_df = train_terms_updated['aspect'].value_counts()\n\nplt.figure(figsize=(8, 5))\n\nsns.barplot(x=bar_df.index, y=bar_df.values, palette=\"bright\")\n\nplt.xlabel(\"Aspect\")\nplt.ylabel(\"Count\")\nplt.title(\"Distribution of Aspects\")\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:20:30.332048Z","iopub.execute_input":"2025-04-24T05:20:30.33225Z","iopub.status.idle":"2025-04-24T05:20:30.827745Z","shell.execute_reply.started":"2025-04-24T05:20:30.332233Z","shell.execute_reply":"2025-04-24T05:20:30.826887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_size = train_protein_ids.shape[0]\ntrain_labels = np.zeros((train_size ,num_of_labels))\nseries_train_protein_ids = pd.Series(train_protein_ids)\n\nbar = progressbar.ProgressBar(maxval=num_of_labels, widgets=[progressbar.Bar('=', '[', ']'), ' ', progressbar.Percentage()])\nbar.start()\nfor i in range(num_of_labels):\n    n_train_terms = train_terms_updated[train_terms_updated['term'] ==  labels[i]]\n    label_related_proteins = n_train_terms['EntryID'].unique()\n    train_labels[:,i] =  series_train_protein_ids.isin(label_related_proteins).astype(float)\n    bar.update(i+1)\nbar.finish()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:20:30.828746Z","iopub.execute_input":"2025-04-24T05:20:30.829063Z","iopub.status.idle":"2025-04-24T05:28:17.60685Z","shell.execute_reply.started":"2025-04-24T05:20:30.82903Z","shell.execute_reply":"2025-04-24T05:28:17.605908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir /kaggle/working/labels\nlabels_df = pd.DataFrame(data = train_labels, columns = labels)\nlabels_df.to_csv('/kaggle/working/labels/kaggledata.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:28:17.607815Z","iopub.execute_input":"2025-04-24T05:28:17.608097Z","iopub.status.idle":"2025-04-24T05:29:35.804592Z","shell.execute_reply.started":"2025-04-24T05:28:17.608066Z","shell.execute_reply":"2025-04-24T05:29:35.803542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:29:35.80558Z","iopub.execute_input":"2025-04-24T05:29:35.805813Z","iopub.status.idle":"2025-04-24T05:29:35.811561Z","shell.execute_reply.started":"2025-04-24T05:29:35.805791Z","shell.execute_reply":"2025-04-24T05:29:35.810646Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Training**","metadata":{}},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:29:35.812443Z","iopub.execute_input":"2025-04-24T05:29:35.812663Z","iopub.status.idle":"2025-04-24T05:29:35.826488Z","shell.execute_reply.started":"2025-04-24T05:29:35.812644Z","shell.execute_reply":"2025-04-24T05:29:35.825786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import train_test_split\n\n# Split the data into train and validation sets\ntrain_data, val_data, train_labels, val_labels = train_test_split(train_df, labels_df, test_size=0.2, random_state=42)\n\n# Define model architecture\nINPUT_SHAPE = [train_df.shape[1]]\nnum_of_labels = labels_df.shape[1]\n\nBATCH_SIZE = 512\n\nmodel = tf.keras.Sequential([\n    tf.keras.layers.BatchNormalization(input_shape=INPUT_SHAPE),\n    tf.keras.layers.Dense(units=256, activation='relu'),\n    tf.keras.layers.Dropout(0.3),\n    tf.keras.layers.Dense(units=256, activation='relu'),\n    tf.keras.layers.Dropout(0.3),\n    tf.keras.layers.Dense(units=256, activation='relu'),\n    tf.keras.layers.Dropout(0.3),\n    tf.keras.layers.Dense(units=256, activation='relu'),\n    tf.keras.layers.Dropout(0.3),\n    tf.keras.layers.Dense(units=256, activation='relu'),\n    tf.keras.layers.Dropout(0.3),\n    tf.keras.layers.Dense(units=num_of_labels, activation='sigmoid')\n])\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy', tf.keras.metrics.AUC()]\n)\n\n# Train the model with validation split\nhistory = model.fit(\n    train_data,\n    train_labels,\n    batch_size=BATCH_SIZE,\n    epochs=20,\n    validation_data=(val_data, val_labels)\n)\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:29:35.827512Z","iopub.execute_input":"2025-04-24T05:29:35.827789Z","iopub.status.idle":"2025-04-24T05:30:25.802709Z","shell.execute_reply.started":"2025-04-24T05:29:35.827758Z","shell.execute_reply":"2025-04-24T05:30:25.801931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the final training and validation metrics\nfinal_train_metrics = model.evaluate(train_data, train_labels, verbose=0)\nfinal_val_metrics = model.evaluate(val_data, val_labels, verbose=0)\n\n# Display in the desired format\nprint(\"\\nFinal Training Metrics:\")\nprint(\"Loss:\", final_train_metrics[0])\nprint(\"Binary Accuracy:\", final_train_metrics[2])\nprint(\"AUC:\", final_train_metrics[1])\n\nprint(\"\\nFinal Validation Metrics:\")\nprint(\"Loss:\", final_val_metrics[0])\nprint(\"Binary Accuracy:\", final_val_metrics[2])\nprint(\"AUC:\", final_val_metrics[1])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:30:25.809368Z","iopub.execute_input":"2025-04-24T05:30:25.809619Z","iopub.status.idle":"2025-04-24T05:30:39.412374Z","shell.execute_reply.started":"2025-04-24T05:30:25.809599Z","shell.execute_reply":"2025-04-24T05:30:39.411497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save(\"/kaggle/working/model1.h5\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:30:39.41369Z","iopub.execute_input":"2025-04-24T05:30:39.413925Z","iopub.status.idle":"2025-04-24T05:30:39.484661Z","shell.execute_reply.started":"2025-04-24T05:30:39.413905Z","shell.execute_reply":"2025-04-24T05:30:39.483792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save_weights(\"/kaggle/working/model.weights.h5\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:30:39.485723Z","iopub.execute_input":"2025-04-24T05:30:39.48604Z","iopub.status.idle":"2025-04-24T05:30:39.540244Z","shell.execute_reply.started":"2025-04-24T05:30:39.486008Z","shell.execute_reply":"2025-04-24T05:30:39.539432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.load_weights(\"/kaggle/working/model.weights.h5\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:30:39.541131Z","iopub.execute_input":"2025-04-24T05:30:39.541454Z","iopub.status.idle":"2025-04-24T05:30:39.598484Z","shell.execute_reply.started":"2025-04-24T05:30:39.541429Z","shell.execute_reply":"2025-04-24T05:30:39.597871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"custom_input_tensor = np.array([[1 for i in range(1024)]])\nprint(custom_input_tensor)\nprint(len(custom_input_tensor[0]))\n# Get predictions for custom input tensor\npredictions = model.predict(custom_input_tensor)\n\n# 'predictions' will contain the model's output for the custom input tensor\nprint(predictions)\nfor i in predictions[0]:\n    x=0 if i<0.5 else 1\n    #print(x)\n# print(len(predictions))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:30:39.59922Z","iopub.execute_input":"2025-04-24T05:30:39.59953Z","iopub.status.idle":"2025-04-24T05:30:39.9019Z","shell.execute_reply.started":"2025-04-24T05:30:39.599501Z","shell.execute_reply":"2025-04-24T05:30:39.90117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, ['binary_accuracy']].plot(title=\"Accuracy\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:30:39.902632Z","iopub.execute_input":"2025-04-24T05:30:39.902848Z","iopub.status.idle":"2025-04-24T05:30:40.378206Z","shell.execute_reply.started":"2025-04-24T05:30:39.902829Z","shell.execute_reply":"2025-04-24T05:30:40.377453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tqdm\nfrom Bio import SeqIO\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport os\nimport json\nfrom typing import Dict\nfrom collections import Counter\nimport random\nimport obonet\nfrom transformers import T5Tokenizer, T5EncoderModel\nimport torch\nimport re\n\ndevice = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\n\n# Load the tokenizer\ntokenizer = T5Tokenizer.from_pretrained('Rostlab/prot_t5_xl_half_uniref50-enc', do_lower_case=False) #.to(device)\n\n# Load the model\nmodel = T5EncoderModel.from_pretrained(\"Rostlab/prot_t5_xl_half_uniref50-enc\").to(device)\n\ndef get_embeddings(seq):\n    sequence_examples = [\" \".join(list(re.sub(r\"[UZOB]\", \"X\", seq)))]\n\n    ids = tokenizer.batch_encode_plus(sequence_examples, add_special_tokens=True, padding=\"longest\")\n\n    input_ids = torch.tensor(ids['input_ids']).to(device)\n    attention_mask = torch.tensor(ids['attention_mask']).to(device)\n\n    # generate embeddings\n    with torch.no_grad():\n        embedding_repr = model(input_ids=input_ids,\n                               attention_mask=attention_mask)\n\n    # extract residue embeddings for the first ([0,:]) sequence in the batch and remove padded & special tokens ([0,:7])\n    emb_0 = embedding_repr.last_hidden_state[0]\n    emb_0_per_protein = emb_0.mean(dim=0)\n\n    return emb_0_per_protein\n\ndef predict(fasta_file):\n    sequences = SeqIO.parse(fasta_file, \"fasta\")\n\n    ids = []\n    num_sequences=sum(1 for seq in sequences)\n    embeds = np.zeros((num_sequences, 1024))\n    i = 0\n    with open(fasta_file, \"r\") as fastafile:\n      # Iterate over each sequence in the file\n      for sequence in SeqIO.parse(fastafile, \"fasta\"):\n        seq_id = sequence.id\n        seq_data = str(sequence.seq)\n        embeds[i] = get_embeddings(seq_data).detach().cpu().numpy()\n        ids.append(seq_id)\n        i += 1\n        \n    INPUT_SHAPE=[1024]\n    num_of_labels=1500\n\n    model = tf.keras.Sequential([\n        tf.keras.layers.BatchNormalization(input_shape=INPUT_SHAPE),\n        tf.keras.layers.Dense(units=256, activation='relu'),\n        tf.keras.layers.Dropout(0.3),\n        tf.keras.layers.Dense(units=256, activation='relu'),\n        tf.keras.layers.Dropout(0.3),\n        tf.keras.layers.Dense(units=256, activation='relu'),\n        tf.keras.layers.Dropout(0.3),\n        tf.keras.layers.Dense(units=256, activation='relu'),\n        tf.keras.layers.Dropout(0.3),\n        tf.keras.layers.Dense(units=256, activation='relu'),\n        tf.keras.layers.Dropout(0.3),\n        tf.keras.layers.Dense(units=num_of_labels, activation='sigmoid')\n    ])\n\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n        loss='binary_crossentropy',\n        metrics=['binary_accuracy', tf.keras.metrics.AUC()]\n    )\n    \n    model.load_weights('/kaggle/working/model.weights.h5') #load model here\n    labels_df=pd.read_csv('/kaggle/input/labels/kaggledata.csv')\n    labels_df=labels_df.drop(columns='Unnamed: 0')\n\n    predictions = model.predict(embeds)\n    predictions_list1=[]\n    predictions_list2=[]\n\n    # 'predictions' will contain the model's output for the custom input tensor\n    for prediction in predictions:\n        tmp=[]\n        t2=[]\n        for i in prediction:\n            x=0 if i<0.4 else 1\n            tmp.append(x)\n            t2.append(i)\n        predictions_list1.append(tmp.copy())\n        predictions_list2.append(t2.copy())\n\n    label_columns = labels_df.columns\n\n    # Convert the predictions into a DataFrame\n    predictions_df = pd.DataFrame(predictions_list1, columns=label_columns)\n    p21=pd.DataFrame(predictions_list2, columns=label_columns)\n\n    # Save the DataFrame to a CSV file\n    predictions_df.to_csv(\"predictions.csv\", index=False)\n    p21.to_csv(\"decimal.csv\",index=False)\n    print(predictions_df.shape)\n    \n    \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:30:40.379071Z","iopub.execute_input":"2025-04-24T05:30:40.379352Z","iopub.status.idle":"2025-04-24T05:30:44.953657Z","shell.execute_reply.started":"2025-04-24T05:30:40.379323Z","shell.execute_reply":"2025-04-24T05:30:44.952486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predict('/kaggle/input/fastaexample/example.fasta') ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T05:30:44.954894Z","iopub.execute_input":"2025-04-24T05:30:44.95515Z","iopub.status.idle":"2025-04-24T05:31:12.279475Z","shell.execute_reply.started":"2025-04-24T05:30:44.95513Z","shell.execute_reply":"2025-04-24T05:31:12.278337Z"}},"outputs":[],"execution_count":null}]}