{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210},{"sourceType":"datasetVersion","sourceId":15264082,"datasetId":9764674,"databundleVersionId":16163297},{"sourceType":"modelInstanceVersion","sourceId":803216,"databundleVersionId":16279282,"modelInstanceId":611351,"modelId":623173}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import shutil\nimport os\n\n# This copies your uploaded file to the 'working' directory\n# Replace 'my-entry-file' with whatever name you gave your upload\ninput_path = '/kaggle/input/datasets/ashishjd/my-entry-file/submission.csv'\noutput_path = '/kaggle/working/submission.csv'\n\nif os.path.exists(input_path):\n    shutil.copyfile(input_path, output_path)\n    print(\"Submission file successfully placed!\")\nelse:\n    print(\"Error: Check your dataset name/path in the sidebar.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-25T16:40:19.868871Z","iopub.execute_input":"2026-03-25T16:40:19.869256Z","iopub.status.idle":"2026-03-25T16:40:19.887306Z","shell.execute_reply.started":"2026-03-25T16:40:19.869218Z","shell.execute_reply":"2026-03-25T16:40:19.886259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input/'):\n    for filename in filenames:\n        if filename.endswith('.pt'):\n            print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-25T17:00:14.437244Z","iopub.execute_input":"2026-03-25T17:00:14.437616Z","iopub.status.idle":"2026-03-25T17:00:34.804519Z","shell.execute_reply.started":"2026-03-25T17:00:14.437582Z","shell.execute_reply":"2026-03-25T17:00:34.80349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\n\n# 1. Architecture (V8)\nclass RNAArchitectModel_v8(nn.Module):\n    def __init__(self, vocab_size=4, emb_dim=64, n_heads=4, n_layers=2, dim_feedforward=128):\n        super().__init__()\n        self.emb = nn.Embedding(vocab_size, emb_dim)\n        self.pos_emb = nn.Parameter(torch.zeros(1, 256, emb_dim))\n        layer = nn.TransformerEncoderLayer(d_model=emb_dim, nhead=n_heads, dim_feedforward=dim_feedforward, batch_first=True)\n        self.transformer = nn.TransformerEncoder(layer, num_layers=n_layers)\n        self.coord_head = nn.Linear(emb_dim, 3)\n\n    def forward(self, x):\n        L = x.size(1)\n        if L <= 256: p_emb = self.pos_emb[:, :L, :]\n        else: p_emb = torch.cat([self.pos_emb, self.pos_emb[:, -1:, :].repeat(1, L-256, 1)], dim=1)\n        return self.coord_head(self.transformer(self.emb(x) + p_emb))\n\n# 2. Load Weights\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nmodel = RNAArchitectModel_v8().to(device)\n# Use the path we verified\nmodel_path = '/kaggle/input/models/ashishjd/rna-models-v6-v8/other/v6_distogram_gem_best/1/train_v8_distogram_gem_best.pt'\nmodel.load_state_dict(torch.load(model_path, map_location=device), strict=False)\nmodel.eval()\n\n# 3. Inference & Wide Format\ntest_df = pd.read_csv('/kaggle/input/competitions/stanford-rna-3d-folding-2/test_sequences.csv')\nvocab = {\"A\": 0, \"C\": 1, \"G\": 2, \"U\": 3}\nfinal_rows = []\n\nfor _, row in tqdm(test_df.iterrows(), total=len(test_df)):\n    seq = str(row['sequence'])\n    ids = torch.tensor([vocab.get(c, 0) for c in seq]).unsqueeze(0).to(device)\n    with torch.no_grad():\n        coords = model(ids).squeeze(0).cpu().numpy()\n    \n    for i, res_char in enumerate(seq):\n        x, y, z = coords[i]\n        final_rows.append({\n            'ID': f\"{row['target_id']}_{i+1}\", 'resname': res_char, 'resid': i+1,\n            'x_1': x, 'y_1': y, 'z_1': z, 'x_2': x, 'y_2': y, 'z_2': z,\n            'x_3': x, 'y_3': y, 'z_3': z, 'x_4': x, 'y_4': y, 'z_4': z,\n            'x_5': x, 'y_5': y, 'z_5': z\n        })\n\n# 4. Save\npd.DataFrame(final_rows).to_csv('submission.csv', index=False)\nprint(\"✅ Submission CSV (Wide Format) Generated!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-25T18:36:55.857963Z","iopub.execute_input":"2026-03-25T18:36:55.858478Z","iopub.status.idle":"2026-03-25T18:36:57.023608Z","shell.execute_reply.started":"2026-03-25T18:36:55.858431Z","shell.execute_reply":"2026-03-25T18:36:57.022194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\n\n# 1. Architecture (V8) - Keeping your specific structure\nclass RNAArchitectModel_v8(nn.Module):\n    def __init__(self, vocab_size=4, emb_dim=64, n_heads=4, n_layers=2, dim_feedforward=128):\n        super().__init__()\n        self.emb = nn.Embedding(vocab_size, emb_dim)\n        self.pos_emb = nn.Parameter(torch.zeros(1, 256, emb_dim))\n        layer = nn.TransformerEncoderLayer(d_model=emb_dim, nhead=n_heads, \n                                           dim_feedforward=dim_feedforward, batch_first=True)\n        self.transformer = nn.TransformerEncoder(layer, num_layers=n_layers)\n        # Predicting 3 coords (x,y,z). We will map these to the 5 atoms required.\n        self.coord_head = nn.Linear(emb_dim, 3) \n\n    def forward(self, x):\n        L = x.size(1)\n        if L <= 256:\n            p_emb = self.pos_emb[:, :L, :]\n        else:\n            p_emb = torch.cat([self.pos_emb, self.pos_emb[:, -1:, :].repeat(1, L-256, 1)], dim=1)\n        return self.coord_head(self.transformer(self.emb(x) + p_emb))\n\n# 2. Setup & Load Weights\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nmodel = RNAArchitectModel_v8().to(device)\nmodel_path = '/kaggle/input/models/ashishjd/rna-models-v6-v8/other/v6_distogram_gem_best/1/train_v8_distogram_gem_best.pt'\nmodel.load_state_dict(torch.load(model_path, map_location=device), strict=False)\nmodel.eval()\n\n# 3. Robust Inference Loop\ntest_df = pd.read_csv('/kaggle/input/competitions/stanford-rna-3d-folding-2/test_sequences.csv')\nvocab = {\"A\": 0, \"C\": 1, \"G\": 2, \"U\": 3}\nfinal_rows = []\n\nfor _, row in tqdm(test_df.iterrows(), total=len(test_df)):\n    seq = str(row['sequence'])\n    ids = torch.tensor([vocab.get(c, 0) for c in seq]).unsqueeze(0).to(device)\n    \n    with torch.no_grad():\n        coords = model(ids).squeeze(0).cpu().numpy() # Shape: (seq_len, 3)\n    \n    for i, res_char in enumerate(seq):\n        x, y, z = coords[i]\n        # Mapping the predicted (x,y,z) to all 5 required atom positions\n        final_rows.append({\n            'ID': f\"{row['target_id']}_{i+1}\",\n            'resname': res_char,\n            'resid': i + 1,\n            'x_1': x, 'y_1': y, 'z_1': z,\n            'x_2': x, 'y_2': y, 'z_2': z,\n            'x_3': x, 'y_3': y, 'z_3': z,\n            'x_4': x, 'y_4': y, 'z_4': z,\n            'x_5': x, 'y_5': y, 'z_5': z\n        })\n\n# 4. Final Formatting & Verification\nsubmission = pd.DataFrame(final_rows)\n\n# Fill any NaNs and ensure coordinates aren't impossibly large/small\ncoord_cols = [c for c in submission.columns if any(p in c for p in ['x_', 'y_', 'z_'])]\nsubmission[coord_cols] = submission[coord_cols].fillna(0).clip(-1000, 1000)\n\nsubmission.to_csv('submission.csv', index=False)\nprint(f\"✅ Submission Generated with {len(submission)} rows!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T00:40:42.2056Z","iopub.execute_input":"2026-03-26T00:40:42.20638Z","iopub.status.idle":"2026-03-26T00:40:43.558529Z","shell.execute_reply.started":"2026-03-26T00:40:42.206339Z","shell.execute_reply":"2026-03-26T00:40:43.55757Z"}},"outputs":[],"execution_count":null}]}