{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":118765,"databundleVersionId":15231210,"sourceType":"competition"},{"sourceId":10855324,"sourceType":"datasetVersion","datasetId":6742586},{"sourceId":11118830,"sourceType":"datasetVersion","datasetId":6933267},{"sourceId":14519720,"sourceType":"datasetVersion","datasetId":9271415},{"sourceId":311741,"sourceType":"modelInstanceVersion","modelInstanceId":264400,"modelId":285488},{"sourceId":731240,"sourceType":"modelInstanceVersion","modelInstanceId":557082,"modelId":569638}],"dockerImageVersionId":31236,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":2354.045049,"end_time":"2026-01-13T06:54:44.524542","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-01-13T06:15:30.479493","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport random\nimport time\nimport warnings\n\nfrom Bio.Seq import Seq\nfrom Bio.Align import PairwiseAligner\n\nwarnings.filterwarnings('ignore')\n\n# ==============================\n# NUCLEOTIDE NORMALIZATION\n# ==============================\nNUCLEOTIDE_MAPPING = {\n    'A': 'A', 'U': 'U', 'G': 'G', 'C': 'C',\n    'I': 'A', '1MA': 'A', 'PSU': 'U', 'M2G': 'G', '5MC': 'C', 'T': 'U',\n}\n\ndef clean_sequence(seq: str) -> str:\n    \"\"\"Map modified nucleotides to canonical A/U/G/C.\"\"\"\n    return \"\".join(NUCLEOTIDE_MAPPING.get(b, 'A') for b in seq)\n\n\n# ==============================\n# DATA LOADING\n# ==============================\nDATA_PATH = '/kaggle/input/stanford-rna-3d-folding-2/'\n\ntrain_seqs = pd.read_csv(DATA_PATH + 'train_sequences.csv')\ntest_seqs = pd.read_csv(DATA_PATH + 'test_sequences.csv')\ntrain_labels = pd.read_csv(DATA_PATH + 'train_labels.csv')\n\ndef process_labels(labels_df: pd.DataFrame) -> dict:\n    \"\"\"\n    Build a mapping: target_id prefix -> (N, 3) float32 coordinates sorted by resid.\n    \"\"\"\n    id_prefixes = labels_df['ID'].str.rsplit('_', 1).str[0]\n    coords_dict = {}\n    for prefix, group in labels_df.groupby(id_prefixes):\n        group = group.sort_values('resid')\n        coords = group[['x_1', 'y_1', 'z_1']].to_numpy(dtype=np.float32)\n        coords_dict[prefix] = coords\n    return coords_dict\n\ntrain_coords_dict = process_labels(train_labels)\n\n# Pre-materialize train sequences for fast iteration (avoid iterrows in hot loop)\ntrain_records = list(\n    zip(train_seqs['target_id'].tolist(), train_seqs['sequence'].tolist())\n)\n\n\n# ==============================\n# GLOBAL ALIGNER CONFIG\n# ==============================\nALIGNER = PairwiseAligner()\nALIGNER.mode = \"global\"\nALIGNER.match_score = 2.0\nALIGNER.mismatch_score = -1.0\nALIGNER.open_gap_score = -7.0\nALIGNER.extend_gap_score = -0.25  # keep quirky tuned penalties\n\n\n# ==============================\n# PHASE 2: LOCAL GEOMETRY REFINEMENT\n# ==============================\ndef adaptive_rna_constraints(coordinates: np.ndarray,\n                             sequence: str,\n                             confidence: float = 1.0) -> np.ndarray:\n    \"\"\"\n    Apply simple distance-based constraints along the backbone.\n    Strength decays with confidence (higher confidence -> weaker adjustment).\n    \"\"\"\n    coords = np.asarray(coordinates, dtype=np.float32)\n    n_coords = coords.shape[0]\n    n_seq = len(sequence)\n    n = min(n_coords, n_seq)\n\n    refined = coords.copy()\n    strength = 0.69 * (1.0 - min(confidence, 0.96))  # keep original magic number\n\n    if n <= 1 or strength == 0.0:\n        return refined\n\n    for _ in range(2):  # two relaxation sweeps\n        for i in range(n - 1):\n            p1 = refined[i]\n            p2 = refined[i + 1]\n            dist = np.linalg.norm(p2 - p1)\n            if dist > 0.0:\n                adj = (5.95 - dist) * strength * 0.45\n                refined[i + 1] = p2 + (p2 - p1) / dist * adj\n\n            if i < n - 2:\n                p3 = refined[i + 2]\n                dist2 = np.linalg.norm(p3 - p1)\n                if dist2 > 0.0:\n                    adj2 = (10.2 - dist2) * strength * 0.25\n                    refined[i + 2] = p3 + (p3 - p1) / dist2 * adj2\n\n    return refined\n\n\n# ==============================\n# PHASE 3: TEMPLATE WARPING\n# ==============================\ndef adapt_template_to_query(query_seq: str,\n                            template_seq: str,\n                            template_coords: np.ndarray) -> np.ndarray:\n    \"\"\"\n    Align template sequence to query and map template coordinates onto query,\n    then inpaint gaps linearly along the chain.\n    \"\"\"\n    q_c = clean_sequence(query_seq)\n    t_c = clean_sequence(template_seq)\n\n    aln = ALIGNER.align(q_c, t_c)\n    L = len(query_seq)\n\n    if len(aln) == 0:\n        # No alignment: straight-line fallback\n        x = np.arange(L, dtype=np.float32) * 3.5\n        return np.stack([x, np.zeros(L, np.float32), np.zeros(L, np.float32)], axis=1)\n\n    # Initialize output with NaNs\n    new_coords = np.full((L, 3), np.nan, dtype=np.float32)\n    t_len = len(template_coords)\n\n    # Use aligned blocks to avoid constructing gapped strings\n    blocks_q, blocks_t = aln[0].aligned\n    for (q_start, q_end), (t_start, t_end) in zip(blocks_q, blocks_t):\n        for qi, ti in zip(range(q_start, q_end), range(t_start, t_end)):\n            if qi >= L or ti >= t_len:\n                break\n            new_coords[qi] = template_coords[ti]\n\n    # Fast O(L) inpainting of missing coordinates\n    mask_valid = ~np.isnan(new_coords[:, 0])\n    if not mask_valid.any():\n        x = np.arange(L, dtype=np.float32) * 3.5\n        return np.stack([x, np.zeros(L, np.float32), np.zeros(L, np.float32)], axis=1)\n\n    idx = np.arange(L, dtype=np.int32)\n\n    # forward fill indices\n    prev_idx = np.where(mask_valid, idx, -1)\n    prev_idx = np.maximum.accumulate(prev_idx)\n\n    # backward fill indices\n    next_idx = np.where(mask_valid, idx, L)\n    next_idx = np.minimum.accumulate(next_idx[::-1])[::-1]\n\n    for i in range(L):\n        if mask_valid[i]:\n            continue\n        pv = prev_idx[i]\n        nv = next_idx[i]\n        if pv >= 0 and nv < L:\n            # Linear interpolation between nearest known coords\n            w = (i - pv) / float(nv - pv)\n            new_coords[i] = (1.0 - w) * new_coords[pv] + w * new_coords[nv]\n        elif pv >= 0:\n            new_coords[i] = new_coords[pv] + np.array([3.5, 0.0, 0.0], dtype=np.float32)\n        elif nv < L:\n            new_coords[i] = new_coords[nv] + np.array([3.5, 0.0, 0.0], dtype=np.float32)\n        else:\n            new_coords[i] = np.array([i * 3.5, 0.0, 0.0], dtype=np.float32)\n\n    return new_coords\n\n\n# ==============================\n# PHASE 4: TEMPLATE SEARCH\n# ==============================\ndef find_similar_sequences(query_seq: str,\n                           train_records,\n                           train_coords_dict: dict,\n                           top_n: int = 5):\n    \"\"\"\n    Scan all training targets with fast global alignment scores (score-only),\n    then return top_n with normalized similarity and coordinates.\n    \"\"\"\n    q_c = clean_sequence(query_seq)\n    q_len = len(query_seq)\n    if q_len == 0:\n        return []\n\n    candidates = []\n    for t_id, t_seq in train_records:\n        if t_id not in train_coords_dict:\n            continue\n\n        t_len = len(t_seq)\n        max_len = max(t_len, q_len)\n        if max_len == 0:\n            continue\n\n        # Length filter: skip very different lengths\n        if abs(t_len - q_len) / max_len > 0.4:\n            continue\n\n        t_c = clean_sequence(t_seq)\n        score = ALIGNER.score(q_c, t_c)  # score-only, no traceback\n        denom = 2 * min(q_len, t_len)\n        if denom <= 0:\n            continue\n        norm_score = score / denom\n        candidates.append((t_id, t_seq, norm_score))\n\n    if not candidates:\n        return []\n\n    candidates.sort(key=lambda x: x[2], reverse=True)\n    top = candidates[:top_n]\n\n    # Attach coordinates for the reduced set\n    result = []\n    for t_id, t_seq, s in top:\n        result.append((t_id, t_seq, s, train_coords_dict[t_id]))\n    return result\n\n\ndef predict_rna_structures(sequence: str,\n                           target_id: str,\n                           train_records,\n                           train_coords_dict: dict,\n                           n_predictions: int = 5):\n    \"\"\"\n    For a query sequence:\n      1. Find similar training templates.\n      2. Warp their coordinates to the query.\n      3. Apply local constraints and confidence-dependent noise.\n      4. If not enough templates, fill with straight-line backbone.\n    \"\"\"\n    predictions = []\n    similar_seqs = find_similar_sequences(\n        sequence, train_records, train_coords_dict, top_n=n_predictions\n    )\n    n = len(sequence)\n\n    for i in range(n_predictions):\n        if i < len(similar_seqs):\n            t_id, t_seq, sim, t_coords = similar_seqs[i]\n            adapted = adapt_template_to_query(sequence, t_seq, t_coords)\n            refined = adaptive_rna_constraints(adapted, sequence, confidence=sim)\n\n            # Keep original noise heuristic\n            noise = 0.0 if i == 0 else max(0.006, (0.38 - sim) * 0.07)\n            if noise > 0.0:\n                refined = refined + np.random.normal(0.0, noise, refined.shape).astype(np.float32)\n            predictions.append(refined.astype(np.float32))\n        else:\n            # Fallback: straight-line coordinates\n            coords = np.zeros((n, 3), dtype=np.float32)\n            if n > 0:\n                step = np.array([4.0, 0.0, 0.0], dtype=np.float32)\n                for j in range(1, n):\n                    coords[j] = coords[j - 1] + step\n            predictions.append(coords)\n\n    return predictions\n\n\n# ==============================\n# PHASE 5: INFERENCE LOOP\n# ==============================\n# Make predictions reproducible\nnp.random.seed(42)\nrandom.seed(42)\n\nall_predictions = []\nstart_time = time.time()\n\nfor idx, row in test_seqs.iterrows():\n    if idx % 10 == 0:\n        print(f\"Processing {idx} | {time.time() - start_time:.1f}s\")\n\n    tid = row['target_id']\n    seq = row['sequence']\n\n    preds = predict_rna_structures(seq, tid, train_records, train_coords_dict)\n    L = len(seq)\n\n    for j in range(L):\n        res = {\n            'ID': f\"{tid}_{j + 1}\",\n            'resname': seq[j],\n            'resid': j + 1,\n        }\n        for i in range(5):\n            x, y, z = preds[i][j]\n            res[f'x_{i + 1}'] = x\n            res[f'y_{i + 1}'] = y\n            res[f'z_{i + 1}'] = z\n        all_predictions.append(res)\n\nsub = pd.DataFrame(all_predictions)\ncols = ['ID', 'resname', 'resid'] + [f'{c}_{i}' for i in range(1, 6) for c in ['x', 'y', 'z']]\nsub[cols].to_csv('submission.csv', index=False)\nprint(\"ALL COMPLETED. SUBMIT!!\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}