{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# used model from https://www.kaggle.com/code/gusthema/cafa-5-protein-function-with-tensorflow\n# and thanks to the video https://www.youtube.com/watch?v=-8XmD2zsFBI&t=365s\nimport numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import MultiLabelBinarizer\n\nsequence_train_path = \"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\"\n\n# Parsing fasta file\nid_sequence_dict = {}\nwith open(sequence_train_path) as f:\n    for line in f:\n        if line.startswith('>'):\n            protein_id = line.split(' ', 1)[0][1:].strip()\n            id_sequence_dict[protein_id] = ''\n        else:\n            id_sequence_dict[protein_id] += line.strip()\n\n# Reading labels\ntrain_path = \"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\"\ntrain_df_raw = pd.read_csv(train_path, sep='\\t')\n\n# these are simply take first 1500, which cannot represent the data well\n\n# # Selecting top num_labels most frequent labels\n# num_labels = 1500\n# labels_id = train_df_raw['term'].value_counts()[:num_labels].index.tolist()\n\n# # Grouping terms by EntryID\n# grouped = train_df_raw.groupby('EntryID')['term'].apply(list).reset_index()\n\n# labels_set = set(labels_id)\n\n# # filter it\n# grouped['term'] = grouped['term'].apply(lambda terms: [i for i in terms if i in labels_set])\n# # remove empty list\n# grouped = grouped[grouped['term'].apply(lambda x: len(x)>0)]","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:32:10.949762Z","iopub.execute_input":"2023-06-01T19:32:10.950251Z","iopub.status.idle":"2023-06-01T19:32:16.587532Z","shell.execute_reply.started":"2023-06-01T19:32:10.950217Z","shell.execute_reply":"2023-06-01T19:32:16.586228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Used code by https://www.kaggle.com/code/simonveitner/simple-ann-approach, the author took top 1500 label with ratio","metadata":{}},{"cell_type":"code","source":"terms = train_df_raw.groupby(['aspect', 'term'])['term'].count().reset_index(name='frequency')\nfractions = (terms.groupby('aspect')['term'].nunique() / terms['term'].nunique() * 1500).apply(round)\nprint(fractions)\n\nlabels_set = set()\nfor aspect, number in fractions.items():\n    selection = terms.loc[(terms.aspect == aspect)]\n    selection = selection.nlargest(number, columns='frequency', keep='first')\n    labels_set.update(selection.term.to_list())","metadata":{"execution":{"iopub.status.busy":"2023-06-01T18:20:20.217296Z","iopub.execute_input":"2023-06-01T18:20:20.217753Z","iopub.status.idle":"2023-06-01T18:20:22.32544Z","shell.execute_reply.started":"2023-06-01T18:20:20.217719Z","shell.execute_reply":"2023-06-01T18:20:22.324196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_id = list(labels_set)\n# Grouping terms by EntryID\ngrouped = train_df_raw.groupby('EntryID')['term'].apply(list).reset_index()\n# filter it\ngrouped['term'] = grouped['term'].apply(lambda terms: [i for i in terms if i in labels_set])\n# remove empty list\ngrouped = grouped[grouped['term'].apply(lambda x: len(x)>0)]","metadata":{"execution":{"iopub.status.busy":"2023-06-01T18:20:22.32681Z","iopub.execute_input":"2023-06-01T18:20:22.327385Z","iopub.status.idle":"2023-06-01T18:20:30.472887Z","shell.execute_reply.started":"2023-06-01T18:20:22.327353Z","shell.execute_reply":"2023-06-01T18:20:30.471674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One hot encoding of terms\nmlb = MultiLabelBinarizer(classes=labels_id)\ntrain_label_matrix = mlb.fit_transform(grouped['term'])","metadata":{"execution":{"iopub.status.busy":"2023-06-01T18:20:30.476522Z","iopub.execute_input":"2023-06-01T18:20:30.477006Z","iopub.status.idle":"2023-06-01T18:20:33.500285Z","shell.execute_reply.started":"2023-06-01T18:20:30.476964Z","shell.execute_reply":"2023-06-01T18:20:33.499145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences","metadata":{"execution":{"iopub.status.busy":"2023-06-01T18:20:33.501998Z","iopub.execute_input":"2023-06-01T18:20:33.50235Z","iopub.status.idle":"2023-06-01T18:20:42.309089Z","shell.execute_reply.started":"2023-06-01T18:20:33.50232Z","shell.execute_reply":"2023-06-01T18:20:42.307427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sequence_list = list(id_sequence_dict.values())\ntrain_sequence_with_spaces = []\nfor i in train_sequence_list:\n    cur = ''\n    for j in i:\n        cur += j+' '\n    train_sequence_with_spaces.append(cur)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T18:20:42.315328Z","iopub.execute_input":"2023-06-01T18:20:42.317685Z","iopub.status.idle":"2023-06-01T18:21:10.488002Z","shell.execute_reply.started":"2023-06-01T18:20:42.317615Z","shell.execute_reply":"2023-06-01T18:21:10.486388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sequence_list = train_sequence_with_spaces\ntokenizer = Tokenizer(num_words = 26)\ntokenizer.fit_on_texts(train_sequence_list)\nword_index = tokenizer.word_index","metadata":{"execution":{"iopub.status.busy":"2023-06-01T18:21:10.490096Z","iopub.execute_input":"2023-06-01T18:21:10.490514Z","iopub.status.idle":"2023-06-01T18:21:44.041855Z","shell.execute_reply.started":"2023-06-01T18:21:10.490478Z","shell.execute_reply":"2023-06-01T18:21:44.04066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequences = tokenizer.texts_to_sequences(train_sequence_list)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T18:21:44.043365Z","iopub.execute_input":"2023-06-01T18:21:44.043776Z","iopub.status.idle":"2023-06-01T18:22:24.99111Z","shell.execute_reply.started":"2023-06-01T18:21:44.043743Z","shell.execute_reply":"2023-06-01T18:22:24.989861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequences_lens = [len(i) for i in sequences if len(i)<2500]\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\nsns.kdeplot(sequences_lens, shade=True)\nplt.title('Distribution of Values')\nplt.xlabel('Values')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:37:55.936569Z","iopub.execute_input":"2023-06-01T19:37:55.937654Z","iopub.status.idle":"2023-06-01T19:37:57.099232Z","shell.execute_reply.started":"2023-06-01T19:37:55.937611Z","shell.execute_reply":"2023-06-01T19:37:57.09812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_length = 1000\ntrunc_type='post'\npadding_type='post'","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:39:03.828813Z","iopub.execute_input":"2023-06-01T19:39:03.829282Z","iopub.status.idle":"2023-06-01T19:39:03.835873Z","shell.execute_reply.started":"2023-06-01T19:39:03.829248Z","shell.execute_reply":"2023-06-01T19:39:03.834461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_padded = pad_sequences(sequences, maxlen=max_length, padding=padding_type, truncating=trunc_type)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:39:05.481869Z","iopub.execute_input":"2023-06-01T19:39:05.482303Z","iopub.status.idle":"2023-06-01T19:39:12.533105Z","shell.execute_reply.started":"2023-06-01T19:39:05.482268Z","shell.execute_reply":"2023-06-01T19:39:12.531916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del sequences","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_padded.shape, train_label_matrix.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:39:12.535228Z","iopub.execute_input":"2023-06-01T19:39:12.535678Z","iopub.status.idle":"2023-06-01T19:39:12.543638Z","shell.execute_reply.started":"2023-06-01T19:39:12.535642Z","shell.execute_reply":"2023-06-01T19:39:12.542305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_SHAPE = [training_padded.shape[1]]\nINPUT_SHAPE","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:39:12.545886Z","iopub.execute_input":"2023-06-01T19:39:12.546711Z","iopub.status.idle":"2023-06-01T19:39:12.55701Z","shell.execute_reply.started":"2023-06-01T19:39:12.546662Z","shell.execute_reply":"2023-06-01T19:39:12.555809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_labels = 1500\n# used model in https://www.kaggle.com/code/gusthema/cafa-5-protein-function-with-tensorflow\nmodel = tf.keras.Sequential([\n    tf.keras.layers.BatchNormalization(input_shape= [max_length]),   \n#     tf.keras.layers.GlobalAveragePooling1D(),\n    tf.keras.layers.Dense(units=512, activation='relu'),\n    tf.keras.layers.Dense(units=512, activation='relu'),\n    tf.keras.layers.Dense(units=512, activation='relu'),\n    tf.keras.layers.Dense(units=num_labels,activation='relu')\n])\n\n\n# Compile model\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy', tf.keras.metrics.AUC()],\n)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:39:12.560561Z","iopub.execute_input":"2023-06-01T19:39:12.561094Z","iopub.status.idle":"2023-06-01T19:39:13.112671Z","shell.execute_reply.started":"2023-06-01T19:39:12.561046Z","shell.execute_reply":"2023-06-01T19:39:13.111463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Assuming training_padded and train_label_matrix as your features and labels respectively\n\n# Split the data: 80% for training, 20% for validation\ntrain_padded, val_padded, train_labels, val_labels = train_test_split(training_padded, train_label_matrix, test_size=0.2)\n\n# Now, fit the model with the training data and validate using the validation data.\nhistory = model.fit(\n    train_padded, train_labels,\n    validation_data=(val_padded, val_labels),\n    batch_size=5120,\n    epochs=5\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:39:13.114061Z","iopub.execute_input":"2023-06-01T19:39:13.114445Z","iopub.status.idle":"2023-06-01T19:41:23.783485Z","shell.execute_reply.started":"2023-06-01T19:39:13.114395Z","shell.execute_reply":"2023-06-01T19:41:23.781734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get test data\ntest_dict = {}\nwith open('/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta') as f:\n    for line in f:\n        if line.startswith('>'):\n            protein_id = line.split(' ', 1)[0][1:].strip()\n            test_dict[protein_id] = ''\n        else:\n            test_dict[protein_id] += line.strip()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:41:23.785773Z","iopub.execute_input":"2023-06-01T19:41:23.786383Z","iopub.status.idle":"2023-06-01T19:41:25.392125Z","shell.execute_reply.started":"2023-06-01T19:41:23.786347Z","shell.execute_reply":"2023-06-01T19:41:25.390827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_list = list(test_dict.values())\ntest_with_spaces = []\nfor i in test_list:\n    cur = ''\n    for j in i:\n        cur += j+' '\n    test_with_spaces.append(cur)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:41:25.395655Z","iopub.execute_input":"2023-06-01T19:41:25.396172Z","iopub.status.idle":"2023-06-01T19:41:48.859116Z","shell.execute_reply.started":"2023-06-01T19:41:25.396128Z","shell.execute_reply":"2023-06-01T19:41:48.857938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_sequence_list = test_with_spaces\ntest_sequences = tokenizer.texts_to_sequences(test_sequence_list)\ntested = pad_sequences(test_sequences, maxlen=max_length, padding=padding_type, truncating=trunc_type)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:41:48.860682Z","iopub.execute_input":"2023-06-01T19:41:48.861049Z","iopub.status.idle":"2023-06-01T19:42:33.780585Z","shell.execute_reply.started":"2023-06-01T19:41:48.861019Z","shell.execute_reply":"2023-06-01T19:42:33.779585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(test_sequences[500])","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:42:33.782127Z","iopub.execute_input":"2023-06-01T19:42:33.782613Z","iopub.status.idle":"2023-06-01T19:42:33.789466Z","shell.execute_reply.started":"2023-06-01T19:42:33.782571Z","shell.execute_reply":"2023-06-01T19:42:33.788693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions =  model.predict(tested)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:42:33.793107Z","iopub.execute_input":"2023-06-01T19:42:33.793759Z","iopub.status.idle":"2023-06-01T19:43:00.178969Z","shell.execute_reply.started":"2023-06-01T19:42:33.793713Z","shell.execute_reply":"2023-06-01T19:43:00.177971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_path = \"/kaggle/working/submission.tsv\"","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:43:00.184063Z","iopub.execute_input":"2023-06-01T19:43:00.184497Z","iopub.status.idle":"2023-06-01T19:43:00.190374Z","shell.execute_reply.started":"2023-06-01T19:43:00.184463Z","shell.execute_reply":"2023-06-01T19:43:00.188914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Reference: https://www.kaggle.com/code/alexandervc/baseline-multilabel-to-multitarget-binary\n\nl = []\nfor i in test_dict.keys():\n    pro_id = i.split('\\t', 1)[0]\n    l += [pro_id]\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:43:00.192092Z","iopub.execute_input":"2023-06-01T19:43:00.192645Z","iopub.status.idle":"2023-06-01T19:43:00.299047Z","shell.execute_reply.started":"2023-06-01T19:43:00.192543Z","shell.execute_reply":"2023-06-01T19:43:00.297872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Open the file in write mode\nimport progressbar\nbar = progressbar.ProgressBar(max_value=len(l)).start()\n\ndict_items = list(test_dict.items())\nwith open(output_path, 'w') as file:\n    # Write the header row\n    for index in range(len(l)):\n        prot_id = l[index]\n        c = 0\n        for predic_val in predictions[index]:\n            if predic_val > 0.005:\n                func = labels_id[c]\n                predic_val = \"{:.4f}\".format(predic_val)\n                file.write(prot_id+'\\t'+func+'\\t'+str(predic_val)+'\\n')\n            c+=1\n        bar.update(index)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:59:16.323609Z","iopub.execute_input":"2023-06-01T19:59:16.324142Z","iopub.status.idle":"2023-06-01T20:03:33.786705Z","shell.execute_reply.started":"2023-06-01T19:59:16.324104Z","shell.execute_reply":"2023-06-01T20:03:33.784925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n# import progressbar\n\n# bar = progressbar.ProgressBar(max_value=len(l)).start()\n\n# with open(output_path, 'w') as file:\n#     for index, prot_id in enumerate(l):\n#         prediction_array = np.array(predictions[index])\n#         mask = prediction_array > 0.01\n#         selected_indices = np.nonzero(mask)[0]\n#         selected_predictions = prediction_array[mask]\n\n#         for c, predic_val in zip(selected_indices, selected_predictions):\n#             func = labels_id[c]\n#             predic_val = \"{:.4f}\".format(predic_val)\n#             file.write(f'{prot_id}\\t{func}\\t{predic_val}\\n')\n\n#         bar.update(index)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T20:04:45.515732Z","iopub.execute_input":"2023-06-01T20:04:45.516282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# import progressbar\n\n# # Assuming l, labels_id, and predictions are defined as per your requirements\n\n# # Convert the lists into Series objects\n# prot_ids = pd.Series(l, name='prot_id')\n# labels_ids = pd.Series(labels_id, name='func')\n\n# # Convert the predictions into a DataFrame\n# predictions_df = pd.DataFrame(predictions)\n\n# # Melt the predictions DataFrame to a long format and rename the columns\n# predictions_df = predictions_df.melt(ignore_index=False, var_name='func', value_name='predic_val')","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:43:00.307606Z","iopub.execute_input":"2023-06-01T19:43:00.308231Z","iopub.status.idle":"2023-06-01T19:43:16.119616Z","shell.execute_reply.started":"2023-06-01T19:43:00.308196Z","shell.execute_reply":"2023-06-01T19:43:16.11826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predictions_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:54:00.417271Z","iopub.execute_input":"2023-06-01T19:54:00.41803Z","iopub.status.idle":"2023-06-01T19:54:00.429396Z","shell.execute_reply.started":"2023-06-01T19:54:00.41798Z","shell.execute_reply":"2023-06-01T19:54:00.427638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predictions_df[predictions_df['predic_val']>0.01].shape","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:43:16.135016Z","iopub.execute_input":"2023-06-01T19:43:16.135948Z","iopub.status.idle":"2023-06-01T19:43:18.963606Z","shell.execute_reply.started":"2023-06-01T19:43:16.135895Z","shell.execute_reply":"2023-06-01T19:43:18.962603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Map func with labels_ids\n# predictions_df['func'] = predictions_df['func'].map(labels_ids)\n\n# # Map the index with prot_ids\n# predictions_df['prot_id'] = predictions_df.index.map(prot_ids)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:43:18.964879Z","iopub.execute_input":"2023-06-01T19:43:18.965189Z","iopub.status.idle":"2023-06-01T19:43:18.970369Z","shell.execute_reply.started":"2023-06-01T19:43:18.965163Z","shell.execute_reply":"2023-06-01T19:43:18.969278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Filter rows where predic_val is greater than 0.001\n# predictions_df = predictions_df[predictions_df['predic_val'] > 0.001]\n\n# # Round predic_val to 4 decimal places\n# predictions_df['predic_val'] = predictions_df['predic_val'].round(4)\n\n# # Write the DataFrame to a file\n# predictions_df.to_csv(output_path, sep='\\t', index=False, header=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:43:18.972121Z","iopub.execute_input":"2023-06-01T19:43:18.972497Z","iopub.status.idle":"2023-06-01T19:43:18.98367Z","shell.execute_reply.started":"2023-06-01T19:43:18.972464Z","shell.execute_reply":"2023-06-01T19:43:18.982494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# res_path = \"/kaggle/working/submission.tsv\"\n# res_pd = pd.read_csv(res_path, sep = '\\t')","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:43:18.98556Z","iopub.execute_input":"2023-06-01T19:43:18.985935Z","iopub.status.idle":"2023-06-01T19:43:18.99764Z","shell.execute_reply.started":"2023-06-01T19:43:18.985904Z","shell.execute_reply":"2023-06-01T19:43:18.996544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filtered_df = predictions_df[predictions_df['predic_val'] > 0.005].sample(frac = 0.005, random_state=1)\n# filtered_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:43:18.999156Z","iopub.execute_input":"2023-06-01T19:43:18.999793Z","iopub.status.idle":"2023-06-01T19:43:30.883937Z","shell.execute_reply.started":"2023-06-01T19:43:18.999754Z","shell.execute_reply":"2023-06-01T19:43:30.882705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n# import seaborn as sns\n\n\n# sns.kdeplot(filtered_df['predic_val'], shade=True)\n# plt.title('Distribution of Values')\n# plt.xlabel('Values')\n# plt.ylabel('Frequency')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:43:30.885485Z","iopub.execute_input":"2023-06-01T19:43:30.885928Z","iopub.status.idle":"2023-06-01T19:43:34.555604Z","shell.execute_reply.started":"2023-06-01T19:43:30.885888Z","shell.execute_reply":"2023-06-01T19:43:34.554746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-06-01T19:51:41.834394Z","iopub.execute_input":"2023-06-01T19:51:41.834864Z","iopub.status.idle":"2023-06-01T19:51:45.523196Z","shell.execute_reply.started":"2023-06-01T19:51:41.83483Z","shell.execute_reply":"2023-06-01T19:51:45.521314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}