{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport torch\nimport matplotlib.pyplot as plt\nimport random\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-08T19:03:05.453369Z","iopub.execute_input":"2023-08-08T19:03:05.453748Z","iopub.status.idle":"2023-08-08T19:03:05.459255Z","shell.execute_reply.started":"2023-08-08T19:03:05.453713Z","shell.execute_reply":"2023-08-08T19:03:05.457925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed = 42\n\n#torch.use_deterministic_algorithms(True)\ntorch.manual_seed(seed)\nrandom.seed(seed)\nnp.random.seed(seed)","metadata":{"execution":{"iopub.status.busy":"2023-08-08T19:03:06.846451Z","iopub.execute_input":"2023-08-08T19:03:06.846866Z","iopub.status.idle":"2023-08-08T19:03:06.860817Z","shell.execute_reply.started":"2023-08-08T19:03:06.846836Z","shell.execute_reply":"2023-08-08T19:03:06.85973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CafaDataset(torch.utils.data.Dataset):\n    def __init__(\n        self, \n        indexes: list,\n        is_train: bool = True\n    ):\n        self.indexes = indexes\n        self.is_train = is_train\n        \n    def __getitem__(self, index): \n        real_index = self.indexes[index]\n        \n        if self.is_train:\n            path = f\"/kaggle/input/ds-train-test-cafa-v0/train_test_cafa/train_batch_{real_index}.pt\"\n        else:\n            path = f\"/kaggle/input/ds-train-test-cafa-v0/train_test_cafa/test_batch_{real_index}.pt\"            \n            \n        batch = torch.load(path)\n        \n        if self.is_train:\n            x, y = batch            \n            x = torch.from_numpy(x)\n            y = torch.from_numpy(y)\n        else:\n            y = None\n            x = torch.from_numpy(batch)\n        \n        if x is not None:\n            x[:, 2566] /= 15.0\n        return x, y\n        \n    def __len__(self):\n        return len(self.indexes)","metadata":{"execution":{"iopub.status.busy":"2023-08-08T19:03:07.574102Z","iopub.execute_input":"2023-08-08T19:03:07.574479Z","iopub.status.idle":"2023-08-08T19:03:07.583662Z","shell.execute_reply.started":"2023-08-08T19:03:07.574448Z","shell.execute_reply":"2023-08-08T19:03:07.582475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Model5(torch.nn.Module):\n    def __init__(self):\n        super().__init__()\n        \n        input_features = 3584\n        output_features = 2000\n        self.activation = torch.nn.PReLU()\n        \n        self.bn1 = torch.nn.BatchNorm1d(input_features)\n        self.fc1 = torch.nn.Linear(input_features, 800)\n        self.ln1 = torch.nn.LayerNorm(800, elementwise_affine=True)\n        \n        self.bn2 = torch.nn.BatchNorm1d(800)\n        self.fc2 = torch.nn.Linear(800, 600)\n        self.ln2 = torch.nn.LayerNorm(600, elementwise_affine=True)\n        \n        self.bn3 = torch.nn.BatchNorm1d(600)\n        self.fc3 = torch.nn.Linear(600, 400)\n        self.ln3 = torch.nn.LayerNorm(400, elementwise_affine=True)\n        \n        self.bn4 = torch.nn.BatchNorm1d(1200)\n        self.fc4 = torch.nn.Linear(1200, output_features)\n        self.ln4 = torch.nn.LayerNorm(output_features, elementwise_affine=True)        \n        \n    def forward(self,inputs):\n#         print(inputs.shape)\n\n        fc1_out = self.bn1(inputs)\n        fc1_out = self.ln1(self.fc1(inputs))\n        fc1_out = self.activation(fc1_out)\n        \n        x = self.bn2(fc1_out)\n        \n        x = self.ln2(self.fc2(x))\n        x = self.activation(x)\n        \n        x = self.bn3(x)\n        \n        x = self.ln3(self.fc3(x))\n        x = self.activation(x)\n        \n        x = torch.cat([x, fc1_out], axis = -1)\n        \n        x = self.bn4(x)\n        \n        x = self.ln4(self.fc4(x))\n        out = x\n        return out","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_indexes, val_indexes = train_test_split(list(range(0, 4448)), test_size=100, random_state=seed)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = torch.nn.DataParallel(Model5()).cuda()\ntrain_dataset = CafaDataset(train_indexes)\nval_dataset = CafaDataset(val_indexes)\n\ncriterion = torch.nn.BCEWithLogitsLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-3)\ngrad_clip = 1.0\nscaler = torch.cuda.amp.GradScaler()\n\ni = 0\ntrain_loss_mean = 0.0\nlog_step = 100\nval_step = 300\nstop_step = 18000\n\nwhile True:\n    for x_batch, y_batch in train_dataset:\n        x_batch = x_batch.cuda()\n        y_batch = y_batch.cuda()\n\n        model.train(True)\n        model.zero_grad()\n\n        with torch.autocast(device_type='cuda', dtype=torch.float16):\n            out = model(x_batch)        \n            loss = criterion(out, y_batch)\n\n        scaler.scale(loss).backward()      \n        scaler.unscale_(optimizer)\n        torch.nn.utils.clip_grad_norm_(model.parameters(), grad_clip)\n        scaler.step(optimizer)\n        scaler.update()\n\n        train_loss_mean += loss.item()\n        if i != 0 and i % log_step == 0:\n            train_loss_mean = np.round(train_loss_mean/log_step, 5)\n            print(f\"{i}) train_loss_mean: {train_loss_mean}\")\n            train_loss_mean = 0.0\n\n\n        if i != 0 and i % val_step == 0:        \n            val_loss_mean = 0.0 \n            model.train(False)\n            with torch.no_grad():\n                for val_x_batch, val_y_batch in val_dataset:\n                    val_x_batch = val_x_batch.cuda()\n                    val_y_batch = val_y_batch.cuda()\n                    val_out = model(val_x_batch)        \n                    val_loss = criterion(val_out, val_y_batch)\n                    val_loss_mean += val_loss.item()\n\n            val_loss_mean = np.round(val_loss_mean/val_step, 5)\n            print(f\"{i}) Val loss: {val_loss_mean}\") \n\n        if i >= stop_step:\n            break\n\n        i += 1\n    if i >= stop_step:\n        break\n    \n    \n        \n    #break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = CafaDataset(list(range(0, 1109)), is_train=False)\nmodel.train(False)\n\nY_submit = []\nfor x_batch, _ in test_dataset:\n    x_batch = x_batch.cuda()\n    with torch.no_grad():    \n        Y_predict = torch.sigmoid(model(x_batch))\n        Y_submit.append(Y_predict.detach().cpu().numpy())\n        \nY_submit = np.concatenate(Y_submit)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit_protein_ids = np.load(\"/kaggle/input/4637427/test_ids_esm2_t36_3B_UR50D.npy\")\nY_labels = np.load(\"/kaggle/input/submit-meta-cafa/submit_meta_cafa/Y_labels.npy\", allow_pickle=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = \"submission.tsv\"\ncutoff_threshold_low = 0.1\n\nglobal_i = 0\nsize = Y_submit.shape[0] * Y_submit.shape[1]\nstep_log = int(size / 100.0)\nwith open(file_path, 'w') as file:\n    for i in range(Y_submit.shape[0]):\n        for j in range(Y_submit.shape[1]):\n            val = Y_submit[i,j]            \n            if val >= cutoff_threshold_low:\n                str_go_term = Y_labels[j]            \n                str_protein_id = submit_protein_ids[i]\n\n                str_save = str_protein_id+'\\t'+str_go_term + '\\t' + '%.3f'%val + '\\n'            \n                file.write(str_save)  \n                if global_i != 0 and global_i % step_log == 0:\n                    percent = np.round(100 * global_i / size, 3)\n                    print(f\"Step writing: {percent} %\")\n            global_i += 1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}