{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":118765,"databundleVersionId":15231210,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n#import os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-11T11:39:39.564569Z","iopub.execute_input":"2026-01-11T11:39:39.564766Z","iopub.status.idle":"2026-01-11T11:39:39.850305Z","shell.execute_reply.started":"2026-01-11T11:39:39.564748Z","shell.execute_reply":"2026-01-11T11:39:39.849505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install knot_pull\n!pip install -U plotly","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T11:39:39.851355Z","iopub.execute_input":"2026-01-11T11:39:39.851804Z","iopub.status.idle":"2026-01-11T11:39:45.847448Z","shell.execute_reply.started":"2026-01-11T11:39:39.851771Z","shell.execute_reply":"2026-01-11T11:39:45.846368Z"},"_kg_hide-input":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"📌 Step 0 — Imports & paths","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nDATA_DIR = \"/kaggle/input/stanford-rna-3d-folding-2\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T11:39:45.849131Z","iopub.execute_input":"2026-01-11T11:39:45.849483Z","iopub.status.idle":"2026-01-11T11:39:45.854302Z","shell.execute_reply.started":"2026-01-11T11:39:45.849418Z","shell.execute_reply":"2026-01-11T11:39:45.853537Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  Step 1 — Load test sequences","metadata":{}},{"cell_type":"code","source":"train_seq = pd.read_csv(f\"{DATA_DIR}/train_sequences.csv\")\ntrain_lbl = pd.read_csv(f\"{DATA_DIR}/train_labels.csv\")\n\nval_seq = pd.read_csv(f\"{DATA_DIR}/validation_sequences.csv\")\nval_lbl = pd.read_csv(f\"{DATA_DIR}/validation_labels.csv\")\n\ntest_seq = pd.read_csv(f\"{DATA_DIR}/test_sequences.csv\")\nsample_sub = pd.read_csv(f\"{DATA_DIR}/sample_submission.csv\")\nprint(test_seq.head())\nprint(\"Number of targets:\", len(test_seq))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T11:39:45.855361Z","iopub.execute_input":"2026-01-11T11:39:45.855662Z","iopub.status.idle":"2026-01-11T11:39:51.525751Z","shell.execute_reply.started":"2026-01-11T11:39:45.855642Z","shell.execute_reply":"2026-01-11T11:39:51.524647Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 2 — Core idea of the baseline (important)","metadata":{}},{"cell_type":"markdown","source":"Geometry choice (safe & common)\n\t•\tFixed bond length ≈ 6.0 Å (C1′–C1′)\n\t•\tRandom walk in 3D\n\t•\tSlight noise for diversity\n\t•\tSame residue indexing\n\nThis already scores > 0 TM-score.\n","metadata":{}},{"cell_type":"markdown","source":"# Step 3 — 3D Visualization","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nfrom collections import Counter\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nimport plotly.io as pio\npio.renderers.default = 'iframe'\n\nSAMPLE_SIZE = 6000\nvalid_coords = train_lbl[['x_1', 'y_1', 'z_1', 'resid', 'resname', 'chain']].dropna()\nsample_coords = valid_coords.sample(n=min(SAMPLE_SIZE, len(valid_coords)), random_state=42)\n\n# ✅ Use add_scatter3d for speed\nfig = go.Figure()\n\nfig.add_trace(go.Scatter3d(\n    x=sample_coords['x_1'],\n    y=sample_coords['y_1'], \n    z=sample_coords['z_1'],\n    mode='markers',\n    marker=dict(\n        size=4,\n        color=sample_coords['resid'],\n        colorscale='Viridis',\n        opacity=0.6,\n        showscale=True,  # ← This makes the colorbar appear\n        colorbar=dict(title=\"Residue #\")\n    ),\n    hovertemplate='<b>Res %{customdata[0]}</b><br>Chain: %{customdata[1]}<br>X: %{x:.1f}Å<br>Y: %{y:.1f}Å<br>Z: %{z:.1f}Å<extra></extra>',\n    customdata=sample_coords[['resid', 'chain']].values\n))\n\nfig.add_scatter3d(\n    x=sample_coords['x_1'],\n    y=sample_coords['y_1'],\n    z=sample_coords['z_1'],\n    mode='markers',\n    marker=dict(\n        size=3,\n        color=sample_coords['resid'],\n        colorscale='Viridis',\n        opacity=0.6\n    ),\n    hovertemplate='<b>Res %{customdata[0]}</b><br>Chain: %{customdata[1]}<extra></extra>',\n    customdata=sample_coords[['resid', 'chain']].values\n)\n\nfig.update_layout(\n    title=f\"🧬 RNA C1' Atoms (n={len(sample_coords):,})\",\n    scene=dict(xaxis_title='X (Å)', yaxis_title='Y (Å)', zaxis_title='Z (Å)'),\n    height=600,\n    margin=dict(l=0, r=0, b=0, t=40)\n)\nfig.show()   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T11:39:51.527739Z","iopub.execute_input":"2026-01-11T11:39:51.527967Z","iopub.status.idle":"2026-01-11T11:39:55.244927Z","shell.execute_reply.started":"2026-01-11T11:39:51.527947Z","shell.execute_reply":"2026-01-11T11:39:55.244054Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 4— Function to generate one structure\n","metadata":{"_kg_hide-input":true}},{"cell_type":"code","source":"def generate_random_chain(length, bond_length=6.0, noise=0.5, seed=None):\n    if seed is not None:\n        np.random.seed(seed)\n\n    coords = np.zeros((length, 3), dtype=np.float32)\n\n    direction = np.random.randn(3)\n    direction /= np.linalg.norm(direction)\n\n    for i in range(1, length):\n        # small random rotation\n        direction += noise * np.random.randn(3)\n        direction /= np.linalg.norm(direction)\n\n        coords[i] = coords[i-1] + bond_length * direction\n\n    return coords","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T11:39:55.245966Z","iopub.execute_input":"2026-01-11T11:39:55.246167Z","iopub.status.idle":"2026-01-11T11:39:55.251726Z","shell.execute_reply.started":"2026-01-11T11:39:55.246149Z","shell.execute_reply":"2026-01-11T11:39:55.250765Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 5 — Generate 5 predictions per target","metadata":{}},{"cell_type":"code","source":"NUM_PRED = 5\nrows = []\n\nfor _, row in test_seq.iterrows():\n    target_id = row[\"target_id\"]\n    sequence = row[\"sequence\"]\n    L = len(sequence)\n\n    structures = [\n        generate_random_chain(L, seed=100 + i)\n        for i in range(NUM_PRED)\n    ]\n\n    for i in range(L):\n        out = {\n            \"ID\": f\"{target_id}_{i+1}\",\n            \"resname\": sequence[i],\n            \"resid\": i+1,\n        }\n\n        for k in range(NUM_PRED):\n            out[f\"x_{k+1}\"] = structures[k][i, 0]\n            out[f\"y_{k+1}\"] = structures[k][i, 1]\n            out[f\"z_{k+1}\"] = structures[k][i, 2]\n\n        rows.append(out)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T11:39:55.252401Z","iopub.execute_input":"2026-01-11T11:39:55.252622Z","iopub.status.idle":"2026-01-11T11:39:55.734916Z","shell.execute_reply.started":"2026-01-11T11:39:55.252598Z","shell.execute_reply":"2026-01-11T11:39:55.734069Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 6— Create submission.csv","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame(rows)\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T11:39:55.73594Z","iopub.execute_input":"2026-01-11T11:39:55.736136Z","iopub.status.idle":"2026-01-11T11:39:55.790387Z","shell.execute_reply.started":"2026-01-11T11:39:55.736118Z","shell.execute_reply":"2026-01-11T11:39:55.789498Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 7 — Save file","metadata":{}},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)\nprint(\"submission.csv saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-11T11:39:55.79187Z","iopub.execute_input":"2026-01-11T11:39:55.792173Z","iopub.status.idle":"2026-01-11T11:39:55.908209Z","shell.execute_reply.started":"2026-01-11T11:39:55.792145Z","shell.execute_reply":"2026-01-11T11:39:55.907022Z"}},"outputs":[],"execution_count":null}]}