{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210,"isSourceIdPinned":false}],"dockerImageVersionId":31328,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom collections import Counter\n\n# =========================================\n# V7: CPU-friendly Retrieval Baseline\n# - k-mer similarity (3-mer + 4-mer)\n# - position-aware segment similarity\n# - stricter length filtering\n# - score-aware blending\n# - true top-5 hypotheses\n# - NaN-safe throughout\n# =========================================\n\n# =========================================\n# 0. Find data directory\n# =========================================\nCANDIDATE_DIRS = [\n    \"/kaggle/input/competitions/stanford-rna-3d-folding-2\",\n    \"/kaggle/input/stanford-rna-3d-folding-2\",\n]\n\nDATA_DIR = None\nfor d in CANDIDATE_DIRS:\n    if os.path.exists(d):\n        DATA_DIR = d\n        break\n\nif DATA_DIR is None:\n    print(\"Available under /kaggle/input:\")\n    for dirname, _, filenames in os.walk(\"/kaggle/input\"):\n        print(dirname)\n        for f in filenames[:10]:\n            print(\"   \", f)\n    raise FileNotFoundError(\"Competition data directory not found.\")\n\nprint(\"Using DATA_DIR =\", DATA_DIR)\n\n# =========================================\n# 1. Read data\n# =========================================\ntrain_seq = pd.read_csv(\n    f\"{DATA_DIR}/train_sequences.csv\",\n    usecols=[\"target_id\", \"sequence\"]\n)\n\ntrain_labels = pd.read_csv(\n    f\"{DATA_DIR}/train_labels.csv\",\n    usecols=[\"ID\", \"resname\", \"resid\", \"x_1\", \"y_1\", \"z_1\"],\n    low_memory=False\n)\n\ntest_seq = pd.read_csv(\n    f\"{DATA_DIR}/test_sequences.csv\",\n    usecols=[\"target_id\", \"sequence\"]\n)\n\nprint(\"train_seq shape:\", train_seq.shape)\nprint(\"train_labels shape:\", train_labels.shape)\nprint(\"test_seq shape:\", test_seq.shape)\n\n# =========================================\n# 2. Helper functions\n# =========================================\ndef clip_coords(arr):\n    return np.clip(arr, -999.999, 9999.999)\n\ndef safe_center(coords):\n    center = np.nanmean(coords, axis=0, keepdims=True)\n    center = np.where(np.isnan(center), 0.0, center)\n    return coords - center\n\ndef fill_nan_coords(coords):\n    \"\"\"\n    coords: [L, 3]\n    Fill NaNs column-wise with linear interpolation.\n    If an entire column is NaN -> fill 0.\n    If only one valid point -> fill constant.\n    \"\"\"\n    coords = coords.copy().astype(np.float32)\n    L = len(coords)\n    idx = np.arange(L)\n\n    for d in range(3):\n        col = coords[:, d]\n        valid = ~np.isnan(col)\n\n        if valid.sum() == 0:\n            coords[:, d] = 0.0\n        elif valid.sum() == 1:\n            coords[:, d] = col[valid][0]\n        elif valid.sum() < L:\n            coords[:, d] = np.interp(idx, idx[valid], col[valid])\n\n    return coords\n\ndef resample_coords(coords, new_len):\n    \"\"\"\n    Linear interpolation from [L,3] -> [new_len,3]\n    \"\"\"\n    coords = np.asarray(coords, dtype=np.float32)\n\n    if np.isnan(coords).any():\n        coords = fill_nan_coords(coords)\n\n    old_len = len(coords)\n    if old_len == new_len:\n        return coords.copy()\n\n    if old_len == 1:\n        return np.repeat(coords, new_len, axis=0)\n\n    old_x = np.linspace(0.0, 1.0, old_len)\n    new_x = np.linspace(0.0, 1.0, new_len)\n\n    new_coords = np.zeros((new_len, 3), dtype=np.float32)\n    for d in range(3):\n        col = coords[:, d]\n        valid = ~np.isnan(col)\n\n        if valid.sum() == 0:\n            new_coords[:, d] = 0.0\n        elif valid.sum() == 1:\n            new_coords[:, d] = col[valid][0]\n        else:\n            new_coords[:, d] = np.interp(new_x, old_x[valid], col[valid])\n\n    return new_coords\n\ndef seq_to_kmer_counter(seq, k):\n    if len(seq) < k:\n        return Counter()\n    return Counter(seq[i:i+k] for i in range(len(seq) - k + 1))\n\ndef weighted_jaccard(counter_a, counter_b):\n    if len(counter_a) == 0 and len(counter_b) == 0:\n        return 0.0\n    keys = set(counter_a) | set(counter_b)\n    inter = 0\n    union = 0\n    for key in keys:\n        a = counter_a.get(key, 0)\n        b = counter_b.get(key, 0)\n        inter += min(a, b)\n        union += max(a, b)\n    if union == 0:\n        return 0.0\n    return inter / union\n\ndef split_three_segments(seq):\n    L = len(seq)\n    a = int(round(L * 0.25))\n    b = int(round(L * 0.75))\n    return [seq[:a], seq[a:b], seq[b:]]\n\ndef get_segment_counters(seq, k):\n    return [seq_to_kmer_counter(seg, k) for seg in split_three_segments(seq)]\n\n# =========================================\n# 3. Build train maps\n# =========================================\ntrain_labels[\"target_id\"] = train_labels[\"ID\"].str.rsplit(\"_\", n=1).str[0]\ntrain_labels = train_labels[train_labels[\"target_id\"].isin(train_seq[\"target_id\"])].copy()\ntrain_labels = train_labels.sort_values([\"target_id\", \"resid\"]).reset_index(drop=True)\n\ntrain_seq_map = dict(zip(train_seq[\"target_id\"], train_seq[\"sequence\"]))\n\ntrain_coord_map = {}\ntrain_base_map = {}\n\nfor tid, g in train_labels.groupby(\"target_id\", sort=False):\n    coords = g[[\"x_1\", \"y_1\", \"z_1\"]].to_numpy(dtype=np.float32)\n    bases = g[\"resname\"].tolist()\n\n    coords = fill_nan_coords(coords)\n    coords = safe_center(coords)\n\n    if np.isnan(coords).any():\n        continue\n\n    train_coord_map[tid] = coords\n    train_base_map[tid] = bases\n\nprint(\"num train targets with usable coords:\", len(train_coord_map))\n\n# Keep only train targets with both sequence and coords\ntrain_items = []\nfor tid, seq in train_seq_map.items():\n    if tid in train_coord_map:\n        train_items.append((tid, seq))\n\nprint(\"usable train items:\", len(train_items))\n\n# =========================================\n# 4. Precompute train k-mer features\n# =========================================\nprint(\"Precomputing train k-mer features...\")\n\ntrain_features = []\nfor tid, seq in train_items:\n    feat = {\n        \"target_id\": tid,\n        \"sequence\": seq,\n        \"length\": len(seq),\n        \"k3\": seq_to_kmer_counter(seq, 3),\n        \"k4\": seq_to_kmer_counter(seq, 4),\n        \"seg3\": get_segment_counters(seq, 3),\n        \"seg4\": get_segment_counters(seq, 4),\n    }\n    train_features.append(feat)\n\nprint(\"train feature items:\", len(train_features))\n\n# =========================================\n# 5. Build fallback baseline tables\n# =========================================\ntrain_seq[\"seq_len\"] = train_seq[\"sequence\"].str.len()\n\ntrain_labels = train_labels.merge(\n    train_seq[[\"target_id\", \"seq_len\"]],\n    on=\"target_id\",\n    how=\"left\"\n)\n\ntrain_labels = train_labels.dropna(subset=[\"seq_len\"]).copy()\ntrain_labels[\"seq_len\"] = train_labels[\"seq_len\"].astype(int)\n\nfor col in [\"x_1\", \"y_1\", \"z_1\"]:\n    train_labels[col] = pd.to_numeric(train_labels[col], errors=\"coerce\")\n\nxyz_global_mean = train_labels[[\"x_1\", \"y_1\", \"z_1\"]].mean()\ntrain_labels[\"x_1\"] = train_labels[\"x_1\"].fillna(xyz_global_mean[\"x_1\"])\ntrain_labels[\"y_1\"] = train_labels[\"y_1\"].fillna(xyz_global_mean[\"y_1\"])\ntrain_labels[\"z_1\"] = train_labels[\"z_1\"].fillna(xyz_global_mean[\"z_1\"])\n\nN_BINS = 20\ntrain_labels[\"rel_pos\"] = (train_labels[\"resid\"] - 1) / np.maximum(train_labels[\"seq_len\"] - 1, 1)\ntrain_labels[\"rel_bin\"] = np.floor(train_labels[\"rel_pos\"] * (N_BINS - 1e-8)).astype(int)\ntrain_labels[\"rel_bin\"] = train_labels[\"rel_bin\"].clip(0, N_BINS - 1)\n\nmean_by_resid_base = (\n    train_labels\n    .groupby([\"resid\", \"resname\"])[[\"x_1\", \"y_1\", \"z_1\"]]\n    .mean()\n    .reset_index()\n)\nresid_base_dict = {\n    (int(r[\"resid\"]), r[\"resname\"]): (r[\"x_1\"], r[\"y_1\"], r[\"z_1\"])\n    for _, r in mean_by_resid_base.iterrows()\n}\n\nmean_by_resid = (\n    train_labels\n    .groupby(\"resid\")[[\"x_1\", \"y_1\", \"z_1\"]]\n    .mean()\n    .reset_index()\n)\nresid_dict = {\n    int(r[\"resid\"]): (r[\"x_1\"], r[\"y_1\"], r[\"z_1\"])\n    for _, r in mean_by_resid.iterrows()\n}\n\nmean_by_bin_base = (\n    train_labels\n    .groupby([\"rel_bin\", \"resname\"])[[\"x_1\", \"y_1\", \"z_1\"]]\n    .mean()\n    .reset_index()\n)\nbin_base_dict = {\n    (int(r[\"rel_bin\"]), r[\"resname\"]): (r[\"x_1\"], r[\"y_1\"], r[\"z_1\"])\n    for _, r in mean_by_bin_base.iterrows()\n}\n\nmean_by_bin = (\n    train_labels\n    .groupby(\"rel_bin\")[[\"x_1\", \"y_1\", \"z_1\"]]\n    .mean()\n    .reset_index()\n)\nbin_dict = {\n    int(r[\"rel_bin\"]): (r[\"x_1\"], r[\"y_1\"], r[\"z_1\"])\n    for _, r in mean_by_bin.iterrows()\n}\n\nglobal_mean = train_labels[[\"x_1\", \"y_1\", \"z_1\"]].mean()\nglobal_xyz = (global_mean[\"x_1\"], global_mean[\"y_1\"], global_mean[\"z_1\"])\n\ndef get_rel_bin(resid, seq_len, n_bins=20):\n    rel_pos = (resid - 1) / max(seq_len - 1, 1)\n    b = int(np.floor(rel_pos * (n_bins - 1e-8)))\n    return max(0, min(n_bins - 1, b))\n\ndef fallback_pred_xyz(resid, base, seq_len):\n    rel_bin = get_rel_bin(resid, seq_len, N_BINS)\n\n    xyz_rb = resid_base_dict.get((resid, base), None)\n    xyz_r = resid_dict.get(resid, None)\n    xyz_bb = bin_base_dict.get((rel_bin, base), None)\n    xyz_b = bin_dict.get(rel_bin, None)\n\n    if xyz_rb is not None and xyz_bb is not None:\n        return (\n            0.7 * xyz_rb[0] + 0.3 * xyz_bb[0],\n            0.7 * xyz_rb[1] + 0.3 * xyz_bb[1],\n            0.7 * xyz_rb[2] + 0.3 * xyz_bb[2],\n        )\n    if xyz_rb is not None:\n        return xyz_rb\n    if xyz_r is not None and xyz_bb is not None:\n        return (\n            0.6 * xyz_r[0] + 0.4 * xyz_bb[0],\n            0.6 * xyz_r[1] + 0.4 * xyz_bb[1],\n            0.6 * xyz_r[2] + 0.4 * xyz_bb[2],\n        )\n    if xyz_r is not None:\n        return xyz_r\n    if xyz_bb is not None:\n        return xyz_bb\n    if xyz_b is not None:\n        return xyz_b\n    return global_xyz\n\ndef fallback_coords_for_sequence(seq):\n    L = len(seq)\n    out = np.zeros((L, 3), dtype=np.float32)\n    for i, base in enumerate(seq, start=1):\n        out[i - 1] = np.array(fallback_pred_xyz(i, base, L), dtype=np.float32)\n    out = out - out.mean(axis=0, keepdims=True)\n    return out\n\n# =========================================\n# 6. Similarity scoring\n# =========================================\ndef compute_similarity_features(query_seq):\n    return {\n        \"length\": len(query_seq),\n        \"k3\": seq_to_kmer_counter(query_seq, 3),\n        \"k4\": seq_to_kmer_counter(query_seq, 4),\n        \"seg3\": get_segment_counters(query_seq, 3),\n        \"seg4\": get_segment_counters(query_seq, 4),\n    }\n\ndef score_candidate(query_feat, cand_feat):\n    qlen = query_feat[\"length\"]\n    clen = cand_feat[\"length\"]\n\n    len_ratio = min(qlen, clen) / max(qlen, clen)\n\n    # strict-ish length filter\n    if len_ratio < 0.45:\n        return None\n\n    j3 = weighted_jaccard(query_feat[\"k3\"], cand_feat[\"k3\"])\n    j4 = weighted_jaccard(query_feat[\"k4\"], cand_feat[\"k4\"])\n\n    seg3_scores = [\n        weighted_jaccard(query_feat[\"seg3\"][i], cand_feat[\"seg3\"][i])\n        for i in range(3)\n    ]\n    seg4_scores = [\n        weighted_jaccard(query_feat[\"seg4\"][i], cand_feat[\"seg4\"][i])\n        for i in range(3)\n    ]\n\n    seg3 = float(np.mean(seg3_scores))\n    seg4 = float(np.mean(seg4_scores))\n\n    # final score\n    score = (\n        0.35 * j3 +\n        0.25 * j4 +\n        0.20 * seg3 +\n        0.10 * seg4 +\n        0.10 * len_ratio\n    )\n\n    return {\n        \"score\": score,\n        \"len_ratio\": len_ratio,\n        \"j3\": j3,\n        \"j4\": j4,\n        \"seg3\": seg3,\n        \"seg4\": seg4,\n    }\n\ndef top_k_similar_sequences_v7(query_seq, train_features, k=5):\n    query_feat = compute_similarity_features(query_seq)\n    scored = []\n\n    for cand in train_features:\n        info = score_candidate(query_feat, cand)\n        if info is None:\n            continue\n\n        scored.append({\n            \"target_id\": cand[\"target_id\"],\n            \"sequence\": cand[\"sequence\"],\n            **info\n        })\n\n    scored = sorted(scored, key=lambda x: x[\"score\"], reverse=True)\n    return scored[:k]\n\n# =========================================\n# 7. Score-aware blending\n# =========================================\ndef blend_with_fallback_score_aware(pred_coords, seq, score):\n    \"\"\"\n    Stronger template if score is high, stronger fallback if score is weak\n    \"\"\"\n    fb = fallback_coords_for_sequence(seq)\n\n    # clamp score to [0,1]\n    s = max(0.0, min(1.0, float(score)))\n\n    # template weight from 0.55 to 0.90\n    w_template = 0.55 + 0.35 * s\n    w_fallback = 1.0 - w_template\n\n    out = w_template * pred_coords + w_fallback * fb\n    out = out - out.mean(axis=0, keepdims=True)\n    return out\n\n# =========================================\n# 8. Build 5 predictions\n# =========================================\ndef make_prediction_from_template(test_sequence, neighbor_info):\n    L = len(test_sequence)\n    tid = neighbor_info[\"target_id\"]\n    score = neighbor_info[\"score\"]\n\n    coords = train_coord_map.get(tid, None)\n    if coords is None or np.isnan(coords).any():\n        return fallback_coords_for_sequence(test_sequence)\n\n    pred = resample_coords(coords, L)\n    pred = blend_with_fallback_score_aware(pred, test_sequence, score)\n\n    if np.isnan(pred).any():\n        pred = fallback_coords_for_sequence(test_sequence)\n\n    return clip_coords(pred)\n\ndef make_five_predictions_for_sequence_v7(test_sequence, k=5):\n    L = len(test_sequence)\n    neighbors = top_k_similar_sequences_v7(test_sequence, train_features, k=max(5, k))\n\n    preds = []\n\n    # pred_1: top-1 template\n    if len(neighbors) >= 1:\n        preds.append(make_prediction_from_template(test_sequence, neighbors[0]))\n    else:\n        preds.append(fallback_coords_for_sequence(test_sequence))\n\n    # pred_2: top-2 template\n    if len(neighbors) >= 2:\n        preds.append(make_prediction_from_template(test_sequence, neighbors[1]))\n    else:\n        preds.append(fallback_coords_for_sequence(test_sequence))\n\n    # pred_3: top-3 template\n    if len(neighbors) >= 3:\n        preds.append(make_prediction_from_template(test_sequence, neighbors[2]))\n    else:\n        preds.append(fallback_coords_for_sequence(test_sequence))\n\n    # pred_4: weighted ensemble of top 2\n    if len(neighbors) >= 2:\n        p1 = make_prediction_from_template(test_sequence, neighbors[0])\n        p2 = make_prediction_from_template(test_sequence, neighbors[1])\n        s1 = max(neighbors[0][\"score\"], 1e-6)\n        s2 = max(neighbors[1][\"score\"], 1e-6)\n        w1 = s1 / (s1 + s2)\n        w2 = s2 / (s1 + s2)\n        p4 = w1 * p1 + w2 * p2\n        p4 = p4 - p4.mean(axis=0, keepdims=True)\n        preds.append(clip_coords(p4))\n    else:\n        preds.append(fallback_coords_for_sequence(test_sequence))\n\n    # pred_5: weighted ensemble of top 3 if available, else fallback blend\n    if len(neighbors) >= 3:\n        tmp_preds = []\n        tmp_scores = []\n        for nb in neighbors[:3]:\n            p = make_prediction_from_template(test_sequence, nb)\n            tmp_preds.append(p)\n            tmp_scores.append(max(nb[\"score\"], 1e-6))\n\n        w = np.array(tmp_scores, dtype=np.float32)\n        w = w / w.sum()\n        p5 = sum(wi * pi for wi, pi in zip(w, tmp_preds))\n        p5 = p5 - p5.mean(axis=0, keepdims=True)\n        preds.append(clip_coords(p5))\n    else:\n        fb = fallback_coords_for_sequence(test_sequence)\n        p1 = preds[0]\n        p5 = 0.5 * p1 + 0.5 * fb\n        p5 = p5 - p5.mean(axis=0, keepdims=True)\n        preds.append(clip_coords(p5))\n\n    # final safety\n    for i in range(5):\n        if np.isnan(preds[i]).any():\n            preds[i] = clip_coords(fallback_coords_for_sequence(test_sequence))\n\n    return preds, neighbors[:5]\n\n# =========================================\n# 9. Build submission\n# =========================================\nrows = []\ndebug_info = []\n\nfor _, row in test_seq.iterrows():\n    target_id = row[\"target_id\"]\n    seq = row[\"sequence\"]\n\n    five_preds, neighbors = make_five_predictions_for_sequence_v7(seq, k=5)\n\n    debug_info.append({\n        \"target_id\": target_id,\n        \"seq_len\": len(seq),\n        \"top_neighbor_ids\": [x[\"target_id\"] for x in neighbors],\n        \"top_neighbor_scores\": [round(float(x[\"score\"]), 4) for x in neighbors],\n        \"top_neighbor_j3\": [round(float(x[\"j3\"]), 4) for x in neighbors],\n        \"top_neighbor_j4\": [round(float(x[\"j4\"]), 4) for x in neighbors],\n        \"top_neighbor_len_ratio\": [round(float(x[\"len_ratio\"]), 4) for x in neighbors],\n    })\n\n    for i, base in enumerate(seq, start=1):\n        out = {\n            \"ID\": f\"{target_id}_{i}\",\n            \"resname\": base,\n            \"resid\": i,\n        }\n\n        for k in range(5):\n            x, y, z = five_preds[k][i - 1]\n            out[f\"x_{k+1}\"] = float(x)\n            out[f\"y_{k+1}\"] = float(y)\n            out[f\"z_{k+1}\"] = float(z)\n\n        rows.append(out)\n\nsubmission = pd.DataFrame(rows)\n\nsubmission = submission[\n    [\"ID\", \"resname\", \"resid\"] +\n    [f\"{axis}_{k}\" for k in range(1, 6) for axis in [\"x\", \"y\", \"z\"]]\n]\n\nprint(submission.head())\nprint(\"submission shape:\", submission.shape)\nprint(\"NA count before save:\", submission.isna().sum().sum())\n\nassert submission.isna().sum().sum() == 0, \"Submission still contains NaN!\"\n\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"✅ submission.csv saved\")\n\ndebug_df = pd.DataFrame(debug_info)\nprint(debug_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T16:48:05.842699Z","iopub.execute_input":"2026-03-21T16:48:05.843431Z","iopub.status.idle":"2026-03-21T16:49:20.709852Z","shell.execute_reply.started":"2026-03-21T16:48:05.84339Z","shell.execute_reply":"2026-03-21T16:49:20.708807Z"}},"outputs":[],"execution_count":null}]}