{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210}],"dockerImageVersionId":31328,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-22T16:39:01.030288Z","iopub.execute_input":"2026-03-22T16:39:01.030647Z","iopub.status.idle":"2026-03-22T16:39:42.561539Z","shell.execute_reply.started":"2026-03-22T16:39:01.030606Z","shell.execute_reply":"2026-03-22T16:39:42.56058Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\ntrain_sequences = pd.read_csv(\"/kaggle/input/competitions/stanford-rna-3d-folding-2/train_sequences.csv\")\ntrain_sequences\n\ntrain_labels = pd.read_csv(\"/kaggle/input/competitions/stanford-rna-3d-folding-2/train_labels.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T22:32:34.564722Z","iopub.execute_input":"2026-03-23T22:32:34.565711Z","iopub.status.idle":"2026-03-23T22:32:46.949196Z","shell.execute_reply.started":"2026-03-23T22:32:34.565668Z","shell.execute_reply":"2026-03-23T22:32:46.948037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/competitions/stanford-rna-3d-folding-2/sample_submission.csv\")\nprint(sample_sub.head())\nprint(f\"Columns: {sample_sub.columns.tolist()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T00:05:35.71113Z","iopub.execute_input":"2026-03-24T00:05:35.712005Z","iopub.status.idle":"2026-03-24T00:05:35.749904Z","shell.execute_reply.started":"2026-03-24T00:05:35.711893Z","shell.execute_reply":"2026-03-24T00:05:35.748898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T20:44:22.822766Z","iopub.execute_input":"2026-03-22T20:44:22.823758Z","iopub.status.idle":"2026-03-22T20:44:22.836381Z","shell.execute_reply.started":"2026-03-22T20:44:22.823717Z","shell.execute_reply":"2026-03-22T20:44:22.83523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences.stoichiometry.unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T21:00:23.419831Z","iopub.execute_input":"2026-03-22T21:00:23.42101Z","iopub.status.idle":"2026-03-22T21:00:23.431455Z","shell.execute_reply.started":"2026-03-22T21:00:23.42096Z","shell.execute_reply":"2026-03-22T21:00:23.430279Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences[train_sequences['stoichiometry']=='B:1;A:1']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T21:01:03.245561Z","iopub.execute_input":"2026-03-22T21:01:03.24634Z","iopub.status.idle":"2026-03-22T21:01:03.264849Z","shell.execute_reply.started":"2026-03-22T21:01:03.246302Z","shell.execute_reply":"2026-03-22T21:01:03.26378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences[train_sequences['target_id']=='1ELH']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T21:01:53.750967Z","iopub.execute_input":"2026-03-22T21:01:53.751378Z","iopub.status.idle":"2026-03-22T21:01:53.765865Z","shell.execute_reply.started":"2026-03-22T21:01:53.751344Z","shell.execute_reply":"2026-03-22T21:01:53.764328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\n\n# 1. LOAD DATA\nDATA_DIR = \"/kaggle/input/competitions/stanford-rna-3d-folding-2/\"\ntrain_seq_df = pd.read_csv(f\"{DATA_DIR}train_sequences.csv\")\ntest_seq_df = pd.read_csv(f\"{DATA_DIR}test_sequences.csv\")\n\n# 2. TOKENIZER SETUP\ntokenizer = {\"[PAD]\": 0, \"G\": 1, \"C\": 2, \"A\": 3, \"U\": 4, \"[UNK]\": 5}\nL_MAX = max(train_seq_df['sequence'].str.len().max(), \n            test_seq_df['sequence'].str.len().max())\n\ndef tokenize_sequence(seq, max_len):\n    tokens = [tokenizer.get(ch, 5) for ch in seq]\n    n_pad = max_len - len(tokens)\n    return tokens + [0] * n_pad\n\n# 3. PYTORCH DATASET\nclass RNADataset(Dataset):\n    def __init__(self, df, max_len):\n        self.ids = df['target_id'].values \n        self.sequences = df['sequence'].values\n        self.lengths = df['sequence'].str.len().values\n        self.max_len = max_len\n\n    def __len__(self):\n        return len(self.ids)\n\n    def __getitem__(self, idx):\n        tokens = tokenize_sequence(self.sequences[idx], self.max_len)\n        return {\n            \"tokens\": torch.tensor(tokens, dtype=torch.long),\n            \"target_id\": str(self.ids[idx]), \n            \"original_len\": self.lengths[idx],\n            \"full_sequence\": self.sequences[idx]\n        }\n\n# 4. SIMPLE 1-LAYER NN MODEL\nclass SimpleRNAFoldingModel(nn.Module):\n    def __init__(self, vocab_size=6, embed_dim=64):\n        super().__init__()\n        self.embedding = nn.Embedding(vocab_size, embed_dim, padding_idx=0)\n        self.head = nn.Linear(embed_dim, 15) # 3 coords * 5 shots = 15\n\n    def forward(self, x):\n        x = self.embedding(x)\n        return self.head(x)\n\n# 5. INFERENCE & CLIPPING\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = SimpleRNAFoldingModel().to(device)\nmodel.eval()\n\ntest_ds = RNADataset(test_seq_df, L_MAX)\ntest_loader = DataLoader(test_ds, batch_size=64, shuffle=False)\n\nsubmission_data = []\n\nprint(\"Generating predictions...\")\nwith torch.no_grad():\n    for batch in test_loader:\n        tokens = batch[\"tokens\"].to(device)\n        preds = model(tokens).cpu().numpy()\n        \n        # Required clipping per competition rules\n        preds = np.clip(preds, -999.999, 9999.999)\n        \n        for b_idx in range(len(batch[\"target_id\"])):\n            t_id = batch[\"target_id\"][b_idx]\n            o_len = batch[\"original_len\"][b_idx].item()\n            seq_str = batch[\"full_sequence\"][b_idx]\n            \n            for res_idx in range(o_len):\n                res_coords = preds[b_idx, res_idx, :]\n                \n                # Aligning with sample_submission.csv format\n                row = {\n                    'ID': f\"{t_id}_{res_idx + 1}\",\n                    'resname': seq_str[res_idx],\n                    'resid': res_idx + 1,\n                    'x_1': res_coords[0],  'y_1': res_coords[1],  'z_1': res_coords[2],\n                    'x_2': res_coords[3],  'y_2': res_coords[4],  'z_2': res_coords[5],\n                    'x_3': res_coords[6],  'y_3': res_coords[7],  'z_3': res_coords[8],\n                    'x_4': res_coords[9],  'y_4': res_coords[10], 'z_4': res_coords[11],\n                    'x_5': res_coords[12], 'y_5': res_coords[13], 'z_5': res_coords[14]\n                }\n                submission_data.append(row)\n\n# 6. SAVE WITH EXACT COLUMN ORDER\nsubmission_df = pd.DataFrame(submission_data)\n\n# Re-ordering to match the image exactly\ncols = ['ID', 'resname', 'resid', \n        'x_1', 'y_1', 'z_1', 'x_2', 'y_2', 'z_2', \n        'x_3', 'y_3', 'z_3', 'x_4', 'y_4', 'z_4', \n        'x_5', 'y_5', 'z_5']\n\nsubmission_df = submission_df[cols]\nsubmission_df.to_csv(\"submission.csv\", index=False)\n\nprint(f\"Success! Submission file saved with {len(submission_df)} rows.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T23:49:42.975993Z","iopub.execute_input":"2026-03-23T23:49:42.976446Z","iopub.status.idle":"2026-03-23T23:49:44.788202Z","shell.execute_reply.started":"2026-03-23T23:49:42.976407Z","shell.execute_reply":"2026-03-23T23:49:44.7869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Train columns:\", train_seq_df.columns)\nprint(\"Test columns:\", test_seq_df.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T23:49:10.415781Z","iopub.execute_input":"2026-03-23T23:49:10.416782Z","iopub.status.idle":"2026-03-23T23:49:10.422522Z","shell.execute_reply.started":"2026-03-23T23:49:10.416737Z","shell.execute_reply":"2026-03-23T23:49:10.421546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels[train_labels['ID'].str.contains('1ELH')]\n\n\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T21:03:36.548409Z","iopub.execute_input":"2026-03-22T21:03:36.54939Z","iopub.status.idle":"2026-03-22T21:03:39.116101Z","shell.execute_reply.started":"2026-03-22T21:03:36.549352Z","shell.execute_reply":"2026-03-22T21:03:39.114948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/competitions/stanford-rna-3d-folding-2/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T21:14:20.207938Z","iopub.execute_input":"2026-03-22T21:14:20.208791Z","iopub.status.idle":"2026-03-22T21:14:20.229228Z","shell.execute_reply.started":"2026-03-22T21:14:20.208749Z","shell.execute_reply":"2026-03-22T21:14:20.227776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub[sample_sub[\"ID\"].str.contains(\"8ZNQ\")]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T21:15:05.150979Z","iopub.execute_input":"2026-03-22T21:15:05.151491Z","iopub.status.idle":"2026-03-22T21:15:05.177532Z","shell.execute_reply.started":"2026-03-22T21:15:05.151456Z","shell.execute_reply":"2026-03-22T21:15:05.176033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}