{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# CAFA Keras Train: MLP GNN V2\n\n---\n### <a href='#hyperparameters'> ⚙️ Hyperparameters </a> | <a href='#data-processing'> 📦️ Data Processing </a> | <a href='#model'> 🧠️ Model</a>\n\n\n","metadata":{}},{"cell_type":"code","source":"\"\"\"\nTODO: \n\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:43:02.964801Z","iopub.execute_input":"2023-08-03T06:43:02.965326Z","iopub.status.idle":"2023-08-03T06:43:02.973361Z","shell.execute_reply.started":"2023-08-03T06:43:02.965279Z","shell.execute_reply":"2023-08-03T06:43:02.972123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Setup & Imports","metadata":{}},{"cell_type":"code","source":"# Installations, Setup and Imports\n!pip install -q omegaconf\n\n# Commonly Used Libraries\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nimport collections\nimport termcolor\nimport functools \nimport random\nimport pickle\nimport os\nimport gc\n\nfrom tqdm.auto import tqdm\ntqdm.pandas()\n\nimport omegaconf\nimport wandb\n!wandb login '3b335317f20548af7e3b941d09a6de9f1736bd8d'\n\nimport transformers\nimport datasets\nimport sklearn\nimport sklearn.metrics\n\n## PyTorch Setup\nimport torch\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\nfrom IPython.core.magic import register_line_cell_magic\n@register_line_cell_magic\ndef hyperparameters(hp_var_name, cell):\n    with open('experiment.yaml', 'w') as f:\n        f.write(cell)\n    HP = omegaconf.OmegaConf.load('experiment.yaml')\n    get_ipython().user_ns[hp_var_name] = HP","metadata":{"execution":{"iopub.status.busy":"2023-08-03T07:03:06.962793Z","iopub.execute_input":"2023-08-03T07:03:06.963278Z","iopub.status.idle":"2023-08-03T07:03:32.305507Z","shell.execute_reply.started":"2023-08-03T07:03:06.963246Z","shell.execute_reply":"2023-08-03T07:03:32.303695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_dir = Path('/kaggle/input/valid-proteins')\n\nterm_vectors = np.load(load_dir/'term_vectors.npy')\nprotein_embeds = np.load(load_dir/'valid_protein_embeds_dim256.npy')\nhf_dataset = datasets.Dataset.load_from_disk(load_dir/'valid_hf_dataset')\nterm_names = np.load(load_dir/'term_names.npy')","metadata":{"execution":{"iopub.status.busy":"2023-08-03T07:06:00.645549Z","iopub.execute_input":"2023-08-03T07:06:00.646005Z","iopub.status.idle":"2023-08-03T07:06:00.937217Z","shell.execute_reply.started":"2023-08-03T07:06:00.645972Z","shell.execute_reply":"2023-08-03T07:06:00.935841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('parent_adj_matrix.npy', parent_adj_matrix)\nnp.save('children_adj_matrix.npy', children_adj_matrix)","metadata":{"execution":{"iopub.status.busy":"2023-08-03T07:07:22.167952Z","iopub.execute_input":"2023-08-03T07:07:22.168402Z","iopub.status.idle":"2023-08-03T07:07:22.422261Z","shell.execute_reply.started":"2023-08-03T07:07:22.168369Z","shell.execute_reply":"2023-08-03T07:07:22.420937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CAFADataset(torch.utils.data.Dataset):\n    \n    def __init__(self, protein_ids, dataset_type):\n        self.protein_ids = protein_ids\n        self.dataset_type = dataset_type\n    \n    def __getitem__(self, idx):\n        protein_id = self.protein_ids[idx]\n        protein_embeds = protein_id_to_protein_embeddings[protein_id]\n        \n        if self.dataset_type == 'test':\n            return {\n                'protein_embeds': torch.tensor(protein_embeds, dtype=torch.float32), \n                'protein_id': protein_id,\n            }\n        \n        term_ids = protein_id_to_term_ids[protein_id]\n        \n        one_hot_label = np.zeros(NUM_TERMS) + args.label_smoothing\n        for term_id in term_ids:\n            one_hot_label[term_id] = 1 - args.label_smoothing\n        \n        return {\n            'protein_embeds': torch.tensor(protein_embeds, dtype=torch.float32),\n            'protein_id': protein_id,\n            'label': torch.tensor(one_hot_label, dtype=torch.float32),\n        }\n    \n    def __len__(self):\n        return len(self.protein_ids)\n\n\ntrain_dataset = CAFAMLPDataset(train_protein_ids, dataset_type='train')\nvalid_dataset = CAFAMLPDataset(valid_protein_ids, dataset_type='valid')\ntest_dataset = CAFAMLPDataset(test_protein_ids, dataset_type='test')\n\ntrain_dataloader = torch.utils.data.DataLoader(\n    dataset=train_dataset,\n    batch_size=args.train_batch_size,\n    shuffle=True,\n    num_workers=2,\n    pin_memory=True,\n    drop_last=True,\n)\nvalid_dataloader = torch.utils.data.DataLoader(\n    dataset=valid_dataset,\n    batch_size=args.test_batch_size,\n    shuffle=False,\n    num_workers=2,\n    pin_memory=True,\n    drop_last=False,\n)\ntest_dataloader = torch.utils.data.DataLoader(\n    dataset=test_dataset,\n    batch_size=args.test_batch_size,\n    shuffle=False,\n    num_workers=2,\n    pin_memory=True,\n    drop_last=False,\n)\n\nfor batch in tqdm(valid_dataloader):\n    pass","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ⚙️ Hyperparameters\n\n<a name='hyperparameters'/>","metadata":{}},{"cell_type":"code","source":"%%hyperparameters args\n\n## Model Training ##\ntrain_batch_size: 4096\ntest_batch_size: 4096\n\n## Misc ##\nvalidation_fold: 0\nmin_term_count: 100\ndebug: False","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:44:49.543809Z","iopub.execute_input":"2023-08-03T06:44:49.54527Z","iopub.status.idle":"2023-08-03T06:44:49.55707Z","shell.execute_reply.started":"2023-08-03T06:44:49.545228Z","shell.execute_reply":"2023-08-03T06:44:49.555871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EMBEDDING_NAMES = [\n    'esm15B_cls',\n    'esm15B_mean',\n    'esm3B_cls',\n    't5',\n    'protbert',\n]\n\nEMBEDDING_NAME_TO_DIM = {\n    'esm15B_cls': 5120, 'esm15B_mean': 5120, 'esm3B_cls': 1024, 't5': 1024, 'protbert': 1024\n}","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:44:49.558474Z","iopub.execute_input":"2023-08-03T06:44:49.558941Z","iopub.status.idle":"2023-08-03T06:44:49.568755Z","shell.execute_reply.started":"2023-08-03T06:44:49.558894Z","shell.execute_reply":"2023-08-03T06:44:49.567837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 📦️ Data Processing\n\n<a name='data-processing'>","metadata":{}},{"cell_type":"code","source":"# Read saved files from CAFA WandB\nwandb_save_dir = Path('/kaggle/input/cafa-wandb')\nprotein_df = pd.read_csv(wandb_save_dir/'protein_df.csv')\ninput_protein_ids = list(protein_df.EntryID.unique())\n# Build term name to term count map\nterm_name_to_term_count = protein_df.set_index('term')['term_count'].to_dict()\n\n# Build term name to IA weight map\nwith open('/kaggle/input/cafa-5-protein-function-prediction/IA.txt') as f:\n    lines = f.readlines()\nterm_name_to_ia_weight = {}\nfor line in lines:\n    term_name, ia_weight = line.replace('\\n', '').split('\\t')\n    term_name_to_ia_weight[term_name] = float(ia_weight)\n# Find dead terms\nROOT_TERMS = ['GO:0008150', 'GO:0005575', 'GO:0003674']\n\nfiltered_term_names = []\nfor term_name, term_count in term_name_to_term_count.items():\n    if term_count < args.min_term_count:\n        continue\n    if term_name_to_ia_weight[term_name] == 0 and term_name not in ROOT_TERMS:\n        continue\n    filtered_term_names.append(term_name)\n    \nprint('Alive terms:', len(filtered_term_names))\nfiltered_term_names.sort(key=lambda term_name: term_name_to_term_count[term_name], reverse=True)\n\nTERM_NAMES = filtered_term_names\nNUM_TERMS = len(TERM_NAMES)\n\nprint('Original Weight:', protein_df.term_weight.sum())\nprotein_df = protein_df[protein_df.term.isin(filtered_term_names)]\nprint('New weight:', protein_df.term_weight.sum())","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:44:49.570107Z","iopub.execute_input":"2023-08-03T06:44:49.571279Z","iopub.status.idle":"2023-08-03T06:45:07.817975Z","shell.execute_reply.started":"2023-08-03T06:44:49.571241Z","shell.execute_reply":"2023-08-03T06:45:07.816457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"term_name_to_term_id = {tname: tid for tid, tname in enumerate(filtered_term_names)}\nprotein_df['term_id'] = protein_df.term.map(term_name_to_term_id)\n\n# Split dataframe into training and test\ntrain_protein_df = protein_df[protein_df.fold!=args.validation_fold]\nvalid_protein_df = protein_df[protein_df.fold==args.validation_fold]\n\ntrain_protein_ids = list(train_protein_df.EntryID.unique())\nvalid_protein_ids = list(valid_protein_df.EntryID.unique())\ninput_protein_ids = list(protein_df.EntryID.unique())\ntrain_protein_df","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:45:07.819896Z","iopub.execute_input":"2023-08-03T06:45:07.820496Z","iopub.status.idle":"2023-08-03T06:45:09.887012Z","shell.execute_reply.started":"2023-08-03T06:45:07.820449Z","shell.execute_reply":"2023-08-03T06:45:09.885798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def whiten(X):\n    U, s, Vt = np.linalg.svd(X, full_matrices=False)\n    X_white = np.dot(U, Vt)\n    return X_white","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:45:09.888396Z","iopub.execute_input":"2023-08-03T06:45:09.888761Z","iopub.status.idle":"2023-08-03T06:45:09.89529Z","shell.execute_reply.started":"2023-08-03T06:45:09.888717Z","shell.execute_reply":"2023-08-03T06:45:09.893979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_train_protein_ids = np.load('/kaggle/input/cafa-protein-embeds-extract/train_protein_ids.npy')\n\nprotein_id_to_embeds = {name: {} for name in EMBEDDING_NAMES}\nfor embed_name in tqdm(EMBEDDING_NAMES):\n    embeds = np.load(f'/kaggle/input/cafa-protein-embeds-extract/train_{embed_name}_embeds.npy')\n    \n    # SKIPPED WHITNING and f32 conversion FOR GPU # \n    # embeds = np.array(embeds, dtype=np.float32)\n    # embeds = whiten(embeds)\n    \n    \n    np.save(f'train_{embed_name}_embeds_whitened.npy', embeds)\n    for idx, pid in enumerate(all_train_protein_ids):\n        protein_id_to_embeds[embed_name][pid] = embeds[idx]\n\ndel embeds\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:45:09.897146Z","iopub.execute_input":"2023-08-03T06:45:09.897867Z","iopub.status.idle":"2023-08-03T06:46:06.859556Z","shell.execute_reply.started":"2023-08-03T06:45:09.89783Z","shell.execute_reply":"2023-08-03T06:46:06.857835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q biopython\nfrom Bio import SeqIO\n\ncomp_dir = Path('/kaggle/input/cafa-5-protein-function-prediction')\ntest_sequences = SeqIO.parse(comp_dir/'Test (Targets)'/'testsuperset.fasta', 'fasta')\ntest_protein_ids = [test_seq.id for test_seq in test_sequences]\n\n# Build labels for training and validation\nprotein_id_to_term_ids = collections.defaultdict(list)\nfor protein_id, term_id in tqdm(zip(protein_df.EntryID.values, protein_df.term_id.values), total=len(protein_df)):\n    protein_id_to_term_ids[protein_id].append(int(term_id))","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:46:06.866488Z","iopub.execute_input":"2023-08-03T06:46:06.86691Z","iopub.status.idle":"2023-08-03T06:46:30.052129Z","shell.execute_reply.started":"2023-08-03T06:46:06.866875Z","shell.execute_reply":"2023-08-03T06:46:30.050553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_hf_dataset.save_to_disk('valid_hf_dataset')","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:59:34.950235Z","iopub.execute_input":"2023-08-03T06:59:34.950697Z","iopub.status.idle":"2023-08-03T06:59:34.977007Z","shell.execute_reply.started":"2023-08-03T06:59:34.950659Z","shell.execute_reply":"2023-08-03T06:59:34.975681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ndef build_one_hot_label_from_term_ids(term_ids, num_terms=NUM_TERMS):\n    one_hot_label = np.zeros(num_terms, dtype=np.bool)\n    for term_id in term_ids:\n        one_hot_label[term_id] = 1\n    return one_hot_label\n\ndef process_protein_sequence(protein_id):\n    output_dict = {'protein_id': protein_id}\n    \n    output_dict['label'] = -100 * np.ones(NUM_TERMS)\n    if protein_id in protein_id_to_term_ids:\n        term_ids = protein_id_to_term_ids[protein_id]\n        one_hot_label = build_one_hot_label_from_term_ids(term_ids)\n        output_dict['label'] = one_hot_label\n    return output_dict\n\ntrain_protein_id_dataset = datasets.Dataset.from_dict({'protein_id': train_protein_ids})\nvalid_protein_id_dataset = datasets.Dataset.from_dict({'protein_id': valid_protein_ids})\ntest_protein_id_dataset = datasets.Dataset.from_dict({'protein_id': test_protein_ids})\n\ntrain_hf_dataset = train_protein_id_dataset.map(\n    process_protein_sequence,\n    input_columns='protein_id',\n    desc='Processing Protein Sequences',\n    num_proc=2 if args.debug else 4,\n).with_format('np')\n\nvalid_hf_dataset = valid_protein_id_dataset.map(\n    process_protein_sequence,\n    input_columns='protein_id',\n    desc='Processing Protein Sequences',\n    num_proc=2 if args.debug else 4,\n).with_format('np')","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:46:30.053881Z","iopub.execute_input":"2023-08-03T06:46:30.054275Z","iopub.status.idle":"2023-08-03T06:46:48.049293Z","shell.execute_reply.started":"2023-08-03T06:46:30.05424Z","shell.execute_reply":"2023-08-03T06:46:48.047481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del protein_df, train_protein_df, valid_protein_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:46:48.051654Z","iopub.execute_input":"2023-08-03T06:46:48.052541Z","iopub.status.idle":"2023-08-03T06:46:48.567576Z","shell.execute_reply.started":"2023-08-03T06:46:48.052483Z","shell.execute_reply":"2023-08-03T06:46:48.566054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CPU times: user 16min 22s, sys: 43.1 s, total: 17min 5s\n# Wall time: 17min 53s","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:46:48.569541Z","iopub.execute_input":"2023-08-03T06:46:48.569947Z","iopub.status.idle":"2023-08-03T06:46:48.575554Z","shell.execute_reply.started":"2023-08-03T06:46:48.569915Z","shell.execute_reply":"2023-08-03T06:46:48.574141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_hf_dataset_to_train_ds(hf_dataset):\n    hf_dataset.set_format(type='numpy')\n    \n    model_inputs = {}\n    for embed_name in EMBEDDING_NAMES:\n        embed_name_embeds = protein_id_to_embeds[embed_name]\n        embeds = []\n        for protein_id in tqdm(hf_dataset['protein_id']):\n            embeds.append(embed_name_embeds[protein_id])\n        model_inputs[embed_name] = np.array(embeds).astype(np.float32)\n    input_ds = tf.data.Dataset.from_tensor_slices(model_inputs)\n    del model_inputs; gc.collect()\n    \n    model_outputs = {\n        'label': hf_dataset['label'].astype(np.float32),\n    }\n    output_ds = tf.data.Dataset.from_tensor_slices(model_outputs)\n    ds = tf.data.Dataset.zip((input_ds, output_ds))\n    return ds\n\n\ndef convert_hf_dataset_to_test_ds(hf_dataset):\n    model_inputs = {}\n    for embed_name in EMBEDDING_NAMES:\n        embed_name_embeds = protein_id_to_embeds[embed_name]\n        embeds = []\n        for protein_id in tqdm(hf_dataset['protein_id']):\n            embeds.append(embed_name_embeds[protein_id])\n        model_inputs[embed_name] = np.array(embeds).astype(np.float32)\n    input_ds = tf.data.Dataset.from_tensor_slices(model_inputs)\n    ds = tf.data.Dataset.zip((input_ds, input_ds))\n    return ds\n\n\ndef hf_dataset_to_tfds(hf_dataset, dataset_type, batch_size): \n    if dataset_type == 'train':\n        ds = convert_hf_dataset_to_train_ds(hf_dataset)\n        ds = ds.shuffle(len(hf_dataset), reshuffle_each_iteration=True).repeat()\n    elif dataset_type == 'valid': \n        ds = convert_hf_dataset_to_train_ds(hf_dataset)\n        ds = ds.cache()\n    elif dataset_type == 'test':\n        ds = convert_hf_dataset_to_test_ds(hf_dataset)\n    ds = ds.batch(batch_size)\n    steps = len(hf_dataset)//batch_size + 1\n    return ds.prefetch(tf.data.AUTOTUNE), steps\n\n\ntrain_ds, train_steps_per_epoch = hf_dataset_to_tfds(train_hf_dataset, 'train', args.train_batch_size)\nvalid_ds, valid_steps_per_epoch = hf_dataset_to_tfds(valid_hf_dataset, 'valid', args.test_batch_size)\n\nx, y = next(iter(train_ds))\nx","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:46:48.578119Z","iopub.execute_input":"2023-08-03T06:46:48.579097Z","iopub.status.idle":"2023-08-03T06:47:32.971001Z","shell.execute_reply.started":"2023-08-03T06:46:48.579048Z","shell.execute_reply":"2023-08-03T06:47:32.969648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del protein_id_to_embeds\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:32.973001Z","iopub.execute_input":"2023-08-03T06:47:32.973395Z","iopub.status.idle":"2023-08-03T06:47:33.53049Z","shell.execute_reply.started":"2023-08-03T06:47:32.973359Z","shell.execute_reply":"2023-08-03T06:47:33.529206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read saved files from CAFA WandB\nwandb_save_dir = Path('/kaggle/input/cafa-wandb')\nprotein_df = pd.read_csv(wandb_save_dir/'protein_df.csv')\ninput_protein_ids = list(protein_df.EntryID.unique())\n# Build term name to term count map\nterm_name_to_term_count = protein_df.set_index('term')['term_count'].to_dict()\n\n# Build term name to IA weight map\nwith open('/kaggle/input/cafa-5-protein-function-prediction/IA.txt') as f:\n    lines = f.readlines()\nterm_name_to_ia_weight = {}\nfor line in lines:\n    term_name, ia_weight = line.replace('\\n', '').split('\\t')\n    term_name_to_ia_weight[term_name] = float(ia_weight)\n# Find dead terms\nROOT_TERMS = ['GO:0008150', 'GO:0005575', 'GO:0003674']\n\nfiltered_term_names = []\nfor term_name, term_count in term_name_to_term_count.items():\n    if term_count < args.min_term_count:\n        continue\n    if term_name_to_ia_weight[term_name] == 0 and term_name not in ROOT_TERMS:\n        continue\n    filtered_term_names.append(term_name)\n    \nprint('Alive terms:', len(filtered_term_names))\nfiltered_term_names.sort(key=lambda term_name: term_name_to_term_count[term_name], reverse=True)\n\nTERM_NAMES = filtered_term_names\nNUM_TERMS = len(TERM_NAMES)\n\nprint('Original Weight:', protein_df.term_weight.sum())\nprotein_df = protein_df[protein_df.term.isin(filtered_term_names)]\nprint('New weight:', protein_df.term_weight.sum())","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:33.532154Z","iopub.execute_input":"2023-08-03T06:47:33.532539Z","iopub.status.idle":"2023-08-03T06:47:50.226446Z","shell.execute_reply.started":"2023-08-03T06:47:33.532505Z","shell.execute_reply":"2023-08-03T06:47:50.225111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model\n\n<a name='model'/>","metadata":{}},{"cell_type":"code","source":"args.num_train_epochs = 1000\nargs.peak_lr = 1e-3\nargs.weight_decay = 1e-5\nargs.beta_2 = 0.98\nargs.warmup_ratio = 0.10\nargs.dropout_ratio = 0.10\nargs.max_grad_norm = 1.00\nargs.bce_weight = 0.975\n\nargs.k_hop_neighbours = 3\nargs.parent_weight = 0.25\n\ntf.keras.backend.clear_session()","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:50.227822Z","iopub.execute_input":"2023-08-03T06:47:50.228223Z","iopub.status.idle":"2023-08-03T06:47:50.248559Z","shell.execute_reply.started":"2023-08-03T06:47:50.228189Z","shell.execute_reply":"2023-08-03T06:47:50.247111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"args.num_train_epochs = 100\nargs.peak_lr = 1e-3\nargs.weight_decay = 1e-5\nargs.beta_2 = 0.98\nargs.warmup_ratio = 0.10\nargs.dropout_ratio = 0.75\nargs.max_grad_norm = 1.00\nargs.bce_weight = 1.00\n\nargs.k_hop_neighbours = 3\nargs.parent_weight = 0.25\n\ntf.keras.backend.clear_session()","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:50.251613Z","iopub.execute_input":"2023-08-03T06:47:50.252146Z","iopub.status.idle":"2023-08-03T06:47:50.263799Z","shell.execute_reply.started":"2023-08-03T06:47:50.2521Z","shell.execute_reply":"2023-08-03T06:47:50.262438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_addons as tfa\n\ntrain_steps_per_epoch = len(train_hf_dataset) // args.train_batch_size\ntotal_train_steps = train_steps_per_epoch * args.num_train_epochs\nwarmup_steps = int(total_train_steps * args.warmup_ratio)\ndecay_steps = total_train_steps - warmup_steps\nprint('train steps per epoch:', train_steps_per_epoch)\nprint('warmup steps:', warmup_steps)\n\ndecay_schedule_fn = keras.optimizers.schedules.CosineDecay(\n    initial_learning_rate=args.peak_lr, \n    decay_steps=decay_steps, \n    alpha=0,\n)\nlr_scheduler = transformers.WarmUp(\n    initial_learning_rate=args.peak_lr,\n    decay_schedule_fn=decay_schedule_fn,\n    warmup_steps=warmup_steps,\n)\n\nwith STRATEGY.scope():\n    optimizer = tfa.optimizers.AdamW(\n        learning_rate=lr_scheduler,\n        weight_decay=args.weight_decay,\n        beta_1=0.9, beta_2=args.beta_2,\n        epsilon=1e-6,\n        global_clipnorm=args.max_grad_norm,\n    )","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:50.265929Z","iopub.execute_input":"2023-08-03T06:47:50.266432Z","iopub.status.idle":"2023-08-03T06:47:50.500074Z","shell.execute_reply.started":"2023-08-03T06:47:50.266388Z","shell.execute_reply":"2023-08-03T06:47:50.498625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"term_name_to_term_id = {term_name: idx for idx, term_name in enumerate(TERM_NAMES)}\n\nwith open('/kaggle/input/cafa-wandb/node_to_parents.pkl', 'rb') as f:\n    node_to_parents = pickle.load(f)\nparent_adj_matrix = np.zeros((NUM_TERMS, NUM_TERMS), dtype=np.float32)\nfor idx in tqdm(range(NUM_TERMS)):\n    term_name = TERM_NAMES[idx]\n    if term_name not in node_to_parents:\n        parent_adj_matrix[idx][idx] = 1.0\n        continue\n    parent_names = node_to_parents[term_name] \n    parent_names = [name for name in parent_names if name in term_name_to_term_id]\n    parent_names += [term_name]\n    parent_ids = [term_name_to_term_id[name] for name in parent_names]\n    for parent_id in parent_ids:\n        parent_adj_matrix[idx][parent_id] = 1/len(parent_ids)\n\nwith open('/kaggle/input/cafa-wandb/node_to_kids.pkl', 'rb') as f:\n    node_to_childrens = pickle.load(f)\nchildren_adj_matrix = np.zeros((NUM_TERMS, NUM_TERMS), dtype=np.float32)\nfor idx in tqdm(range(NUM_TERMS)):\n    term_name = TERM_NAMES[idx]\n    if term_name not in node_to_childrens:\n        children_adj_matrix[idx][idx] = 1.0\n        continue\n    children_names = node_to_childrens[term_name] \n    children_names = [name for name in children_names if name in term_name_to_term_id]\n    children_names += [term_name]\n    children_ids = [term_name_to_term_id[name] for name in children_names]\n    for children_id in children_ids:\n        children_adj_matrix[idx][children_id] = 1/len(children_ids)\n\nparent_adj_matrix\nchildren_adj_matrix","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:50.502266Z","iopub.execute_input":"2023-08-03T06:47:50.502641Z","iopub.status.idle":"2023-08-03T06:47:50.737287Z","shell.execute_reply.started":"2023-08-03T06:47:50.50261Z","shell.execute_reply":"2023-08-03T06:47:50.736133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.term_classifier","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:54:15.909928Z","iopub.execute_input":"2023-08-03T06:54:15.910379Z","iopub.status.idle":"2023-08-03T06:54:15.964355Z","shell.execute_reply.started":"2023-08-03T06:54:15.910347Z","shell.execute_reply":"2023-08-03T06:54:15.962641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow.keras.backend as K\n\nclass TermClassifier(tf.keras.layers.Layer):\n    \n    def __init__(self, hidden_size, name='term_classifier'):\n        super().__init__(name=name)\n        \n        # Calculate the fixed bias to be added to the logits\n        term_freq = protein_df.term_id.value_counts(normalize=True).values\n        fixed_bias = np.log(term_freq/(1-term_freq))\n        self.fixed_bias = tf.convert_to_tensor(fixed_bias, dtype=tf.float32)\n\n        # Precompute the final adj matrix\n        self.parent_adj_matrix = tf.convert_to_tensor(parent_adj_matrix, dtype=tf.float32)\n        self.children_adj_matrix = tf.convert_to_tensor(children_adj_matrix, dtype=tf.float32)\n        self.adj_matrix = args.parent_weight*self.parent_adj_matrix + (1-args.parent_weight)*self.children_adj_matrix\n        \n        final_adj_matrix = self.adj_matrix\n        for _ in range(args.k_hop_neighbours):\n            final_adj_matrix = tf.matmul(final_adj_matrix, self.adj_matrix)\n        self.final_adj_matrix = final_adj_matrix\n        \n        self.output_proj = tf.keras.Sequential([\n            tf.keras.layers.Dense(NUM_TERMS),\n            tfa.layers.GELU(),\n            tf.keras.layers.Dropout(args.dropout_ratio),\n            tf.keras.layers.Dense(NUM_TERMS),\n        ])\n        self.data_gate_proj = tf.keras.layers.Dense(NUM_TERMS)\n        \n        term_vectors_init = tf.keras.initializers.RandomNormal(stddev=1e-4)\n        self.term_vectors = tf.Variable(\n            initial_value=term_vectors_init(shape=(hidden_size, NUM_TERMS), dtype='float32'),\n            trainable=True,\n        )\n        \n    def call(self, protein_embeds):\n        aggregated_features = tf.matmul(self.term_vectors, self.final_adj_matrix)\n        # x = self.term_vectors + aggregated_features * self.data_gate_proj(self.term_vectors)\n        x = aggregated_features\n        weight = self.output_proj(x)\n        return tf.matmul(protein_embeds, weight) + self.fixed_bias\n    \n    \ndef build_model():\n    \n    # Model Inputs # \n    model_inputs = {\n        embed_name: tf.keras.Input(shape=(EMBEDDING_NAME_TO_DIM[embed_name]), dtype=tf.float32)\n        for embed_name in EMBEDDING_NAMES\n    }\n    \n    model_proj_arr = []\n    for model_input in model_inputs.values():\n        norm = tf.keras.layers.Normalization(axis=-1)\n        x = norm(model_input)\n        proj = tf.keras.Sequential([\n            tf.keras.layers.Dropout(args.dropout_ratio),\n            tf.keras.layers.Dense(4096),\n            tf.keras.layers.Dropout(args.dropout_ratio),\n            tfa.layers.GELU(),\n            tf.keras.layers.Dense(1024),\n            tfa.layers.GELU(),\n            tf.keras.layers.Dense(256),\n        ])\n        x_proj = proj(x)\n        model_proj_arr.append(x_proj)\n    \n    model_proj_sum = tf.math.reduce_sum(tf.stack(model_proj_arr, axis=-1), axis=-1)\n    x = model_proj_sum\n    \n    \n    # Term Classifier Head #\n    term_classifier = TermClassifier(256, name='term_classifier')\n    logits = term_classifier(x)\n    labels = tf.math.sigmoid(logits)\n    return tf.keras.Model(inputs=model_inputs, outputs={\n        'label':labels, 'model_proj_sum': model_proj_sum,\n    })\n\ndef bce_loss(y_true, y_pred):\n    return tf.keras.metrics.binary_crossentropy(y_true, y_pred)\n\ndef dice_coef(y_true, y_pred, smooth=1.0):\n    y_true_f = K.flatten(y_true)\n    y_pred_f = K.flatten(y_pred)\n    intersection = K.sum(y_true_f * y_pred_f)\n    dice = (2. * intersection + smooth) / (K.sum(y_true_f) + K.sum(y_pred_f) + smooth)\n    return dice\n\ndef dice_coef_loss(y_true, y_pred):\n    return 1 - dice_coef(y_true, y_pred)\n\ndef loss_fn(y_true, y_pred):\n    bce = bce_loss(y_true, y_pred)\n    dice = dice_coef_loss(y_true, y_pred)\n    return args.bce_weight * bce + (1-args.bce_weight)*dice\n\nwith STRATEGY.scope():\n    model = build_model()\n    model.compile(\n        optimizer=optimizer,\n        steps_per_execution=None if args.debug else 512,\n        run_eagerly=True if args.debug else False,\n        loss=loss_fn,\n        metrics=['accuracy', bce_loss, dice_coef],\n    )\n    model.load_weights('/kaggle/input/cafa-keras-mlp/model_ep600.h5')\n    # model.load_weights('model_v2')\n    # model.load_weights('/kaggle/working/model_val_loss0329.h5')\n    # model.load_weights('/kaggle/input/cafa-keras-mlp/model_valloss1014.h5')    ","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:55:13.331874Z","iopub.execute_input":"2023-08-03T06:55:13.332342Z","iopub.status.idle":"2023-08-03T06:55:22.059245Z","shell.execute_reply.started":"2023-08-03T06:55:13.332307Z","shell.execute_reply":"2023-08-03T06:55:22.058082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"term_vectors = model.get_layer('term_classifier').term_vectors\nterm_vectors = term_vectors.numpy()\nnp.save('term_vectors.npy', term_vectors)\nnp.save('term_names.npy', TERM_NAMES)","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:56:19.793785Z","iopub.execute_input":"2023-08-03T06:56:19.79442Z","iopub.status.idle":"2023-08-03T06:56:19.821038Z","shell.execute_reply.started":"2023-08-03T06:56:19.794369Z","shell.execute_reply":"2023-08-03T06:56:19.819713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_protein_embeds = model.predict(valid_ds, verbose=True)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:51:15.5904Z","iopub.execute_input":"2023-08-03T06:51:15.591103Z","iopub.status.idle":"2023-08-03T06:52:01.516269Z","shell.execute_reply.started":"2023-08-03T06:51:15.591053Z","shell.execute_reply":"2023-08-03T06:52:01.514791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('valid_protein_embeds_dim256.npy', valid_protein_embeds['model_proj_sum'])\nnp.save('valid_protein_ids.npy', valid_protein_ids)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-03T07:04:45.39808Z","iopub.execute_input":"2023-08-03T07:04:45.398519Z","iopub.status.idle":"2023-08-03T07:04:45.492449Z","shell.execute_reply.started":"2023-08-03T07:04:45.398488Z","shell.execute_reply":"2023-08-03T07:04:45.491094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.layers\n","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:53:33.081447Z","iopub.execute_input":"2023-08-03T06:53:33.081869Z","iopub.status.idle":"2023-08-03T06:53:33.091326Z","shell.execute_reply.started":"2023-08-03T06:53:33.081834Z","shell.execute_reply":"2023-08-03T06:53:33.09004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:52:43.950073Z","iopub.execute_input":"2023-08-03T06:52:43.950582Z","iopub.status.idle":"2023-08-03T06:52:45.932385Z","shell.execute_reply.started":"2023-08-03T06:52:43.950542Z","shell.execute_reply":"2023-08-03T06:52:45.931081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"step_multiplier = 10\nmodel.fit(\n    train_ds.repeat(), steps_per_epoch=train_steps_per_epoch*step_multiplier,\n    epochs=args.num_train_epochs//step_multiplier, verbose=True,\n    validation_data=valid_ds, validation_steps=valid_steps_per_epoch,\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.593176Z","iopub.status.idle":"2023-08-03T06:47:56.594148Z","shell.execute_reply.started":"2023-08-03T06:47:56.593919Z","shell.execute_reply":"2023-08-03T06:47:56.593943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights('model_ep600.h5')","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.595861Z","iopub.status.idle":"2023-08-03T06:47:56.596744Z","shell.execute_reply.started":"2023-08-03T06:47:56.596393Z","shell.execute_reply":"2023-08-03T06:47:56.596425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 0.356 was the best previous\n# TODO: Remove gating when adding parent + self\n\n# 0.0365 was the previous best","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.598318Z","iopub.status.idle":"2023-08-03T06:47:56.599163Z","shell.execute_reply.started":"2023-08-03T06:47:56.598932Z","shell.execute_reply":"2023-08-03T06:47:56.598955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# aEpoch 1/100\n# 7/7 [==============================] - 67s 10s/step - loss: 0.1204 - accuracy: 0.5173 - val_loss: 0.1215 - val_accuracy: 0.5125\n# Epoch 2/100\n# 7/7 [==============================] - 64s 9s/step - loss: 0.1198 - accuracy: 0.5074 - val_loss: 0.1210 - val_accuracy: 0.507","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.600467Z","iopub.status.idle":"2023-08-03T06:47:56.600871Z","shell.execute_reply.started":"2023-08-03T06:47:56.600665Z","shell.execute_reply":"2023-08-03T06:47:56.600683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# F1 score @ 0.1: 0.2817594124261782\n\n# oss: 0.0359 - accuracy: 0.5258 - val_loss: 0.0363 - val_accuracy: 0.5222","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.601967Z","iopub.status.idle":"2023-08-03T06:47:56.602359Z","shell.execute_reply.started":"2023-08-03T06:47:56.60217Z","shell.execute_reply":"2023-08-03T06:47:56.602189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import scipy.special\n\nterm_weights  = [term_name_to_ia_weight[term] for term in TERM_NAMES]\ndef validate_f1_score(preds):\n    for threshold in [0.01, 0.05, 0.10, 0.25, 0.50, 0.75, 0.90]:\n        bin_preds = np.where(preds>threshold, 1, 0).astype(np.bool_)\n        f1_scores = []\n        for idx, y_pred in tqdm(\n            zip(range(len(valid_hf_dataset)), bin_preds), \n            total=len(valid_hf_dataset)\n        ):\n            y_true = valid_hf_dataset[idx]['label']\n            y_true = np.where(y_true>0.50, 1.00, 0.00)\n            f1 = sklearn.metrics.f1_score(\n                y_true, y_pred, sample_weight=term_weights, zero_division=0.0\n            )\n            f1_scores.append(f1)\n        f1_score = sum(f1_scores) / len(f1_scores)\n        print(f\"F1 score @ {termcolor.colored(threshold, 'blue')}: {termcolor.colored(f1_score, 'red')}\")\n        \n\nlogits = model.predict(valid_ds, verbose=True)\nvalid_preds = logits['label']\n# valid_preds = scipy.special.expit(logits['label'])\nvalidate_f1_score(valid_preds)","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.603446Z","iopub.status.idle":"2023-08-03T06:47:56.603847Z","shell.execute_reply.started":"2023-08-03T06:47:56.603636Z","shell.execute_reply":"2023-08-03T06:47:56.603654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_hf_dataset, valid_hf_dataset, train_ds, valid_ds\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.605253Z","iopub.status.idle":"2023-08-03T06:47:56.605638Z","shell.execute_reply.started":"2023-08-03T06:47:56.605445Z","shell.execute_reply":"2023-08-03T06:47:56.605464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_test_protein_ids = np.load('/kaggle/input/cafa-protein-embeds-extract/test_protein_ids.npy')\n\nprotein_id_to_embeds = {name: {} for name in EMBEDDING_NAMES}\nfor embed_name in tqdm(EMBEDDING_NAMES):\n    embeds = np.load(f'/kaggle/input/cafa-protein-embeds-extract/test_{embed_name}_embeds.npy')\n    embeds = np.array(embeds, dtype=np.float32)\n    \n    \n    \n    # embeds = whiten(embeds)\n    # np.save(f'test_{embed_name}_embeds_whitened.npy', embeds)\n    \n    \n    \n    for idx, pid in enumerate(all_test_protein_ids):\n        protein_id_to_embeds[embed_name][pid] = embeds[idx]\n\ndel embeds\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.606765Z","iopub.status.idle":"2023-08-03T06:47:56.60716Z","shell.execute_reply.started":"2023-08-03T06:47:56.606959Z","shell.execute_reply":"2023-08-03T06:47:56.606976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\ndef process_test_protein_sequence(protein_id):\n    output_dict = {'protein_id': protein_id}\n    return output_dict\n\ntest_hf_dataset = test_protein_id_dataset.map(\n    process_test_protein_sequence,\n    input_columns='protein_id',\n    desc='Processing Test Protein Sequences',\n    #num_proc=2 if args.debug else 16,\n).with_format('np')\n\ntest_ds, test_steps_per_epoch = hf_dataset_to_tfds(test_hf_dataset, 'test', args.test_batch_size)","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.608307Z","iopub.status.idle":"2023-08-03T06:47:56.608749Z","shell.execute_reply.started":"2023-08-03T06:47:56.608539Z","shell.execute_reply":"2023-08-03T06:47:56.608558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 3:08 to ","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.610007Z","iopub.status.idle":"2023-08-03T06:47:56.610403Z","shell.execute_reply.started":"2023-08-03T06:47:56.610209Z","shell.execute_reply":"2023-08-03T06:47:56.610228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_hf_dataset\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.611601Z","iopub.status.idle":"2023-08-03T06:47:56.612022Z","shell.execute_reply.started":"2023-08-03T06:47:56.611822Z","shell.execute_reply":"2023-08-03T06:47:56.611841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test Prediction\nlogits = model.predict(test_ds, verbose=True)\npreds = logits['label']","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.613752Z","iopub.status.idle":"2023-08-03T06:47:56.617396Z","shell.execute_reply.started":"2023-08-03T06:47:56.617042Z","shell.execute_reply":"2023-08-03T06:47:56.617081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TERMS = np.array(TERM_NAMES)\nTHRESHOLD = 0.05\n\nprotein_pred_df = {'protein_id': [], 'term': [], 'pred': []}\nfor protein_id, protein_preds in tqdm(zip(test_protein_ids, preds), total=len(preds)):\n    \n    mask = protein_preds>THRESHOLD\n    pred_terms, protein_preds = TERMS[mask], protein_preds[mask]\n    protein_pred_df['protein_id']+= [protein_id]*len(list(pred_terms))\n    protein_pred_df['term'] += list(pred_terms)\n    protein_pred_df['pred'] += list(protein_preds)\n\nsub_df = pd.DataFrame.from_dict(protein_pred_df)\nsub_df.to_csv('submission.tsv', sep='\\t', index=False, header=False)\ndf = pd.read_csv('submission.tsv', sep='\\t')\ndf","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.619223Z","iopub.status.idle":"2023-08-03T06:47:56.621235Z","shell.execute_reply.started":"2023-08-03T06:47:56.620885Z","shell.execute_reply":"2023-08-03T06:47:56.62092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !cp '/kaggle/input/cafa-keras-mlp/model_valloss1014.h5' 'model_valloss1014.h5'","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.623928Z","iopub.status.idle":"2023-08-03T06:47:56.625242Z","shell.execute_reply.started":"2023-08-03T06:47:56.624907Z","shell.execute_reply":"2023-08-03T06:47:56.624941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !ls","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.626948Z","iopub.status.idle":"2023-08-03T06:47:56.627801Z","shell.execute_reply.started":"2023-08-03T06:47:56.627455Z","shell.execute_reply":"2023-08-03T06:47:56.627495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"step_multiplier = 10\nmodel.fit(\n    train_ds.repeat(), steps_per_epoch=train_steps_per_epoch*step_multiplier,\n    epochs=args.num_train_epochs//step_multiplier, verbose=True,\n    validation_data=valid_ds, validation_steps=valid_steps_per_epoch,\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-03T06:47:56.62994Z","iopub.status.idle":"2023-08-03T06:47:56.630542Z","shell.execute_reply.started":"2023-08-03T06:47:56.630237Z","shell.execute_reply":"2023-08-03T06:47:56.630265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}