{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":11788894,"datasetId":7402173,"databundleVersionId":12276346},{"sourceType":"datasetVersion","sourceId":14604295,"datasetId":9328538,"databundleVersionId":15440074},{"sourceType":"datasetVersion","sourceId":11695366,"datasetId":6791615,"databundleVersionId":12170579},{"sourceType":"datasetVersion","sourceId":15338221,"datasetId":9810805,"databundleVersionId":16246649},{"sourceType":"datasetVersion","sourceId":5123458,"datasetId":2975803,"databundleVersionId":5194879},{"sourceType":"datasetVersion","sourceId":10880374,"datasetId":6760482,"databundleVersionId":11247092},{"sourceType":"datasetVersion","sourceId":14941601,"datasetId":9562090,"databundleVersionId":15810757},{"sourceType":"datasetVersion","sourceId":15273191,"datasetId":9770020,"databundleVersionId":16173360},{"sourceType":"datasetVersion","sourceId":11118830,"datasetId":6933267,"databundleVersionId":11511771},{"sourceType":"datasetVersion","sourceId":11899194,"datasetId":7479946,"databundleVersionId":12404228},{"sourceType":"datasetVersion","sourceId":11451236,"datasetId":7174725,"databundleVersionId":11892125},{"sourceType":"datasetVersion","sourceId":11230242,"datasetId":7014687,"databundleVersionId":11640305},{"sourceType":"datasetVersion","sourceId":13282339,"datasetId":7162026,"databundleVersionId":13982667},{"sourceType":"datasetVersion","sourceId":14874339,"datasetId":9502242,"databundleVersionId":15736806},{"sourceType":"datasetVersion","sourceId":11469248,"datasetId":7187409,"databundleVersionId":11912757},{"sourceType":"datasetVersion","sourceId":11969392,"datasetId":7526656,"databundleVersionId":12482691},{"sourceType":"datasetVersion","sourceId":10923077,"datasetId":6785143,"databundleVersionId":11294302},{"sourceType":"datasetVersion","sourceId":10855324,"datasetId":6742586,"databundleVersionId":11219268},{"sourceType":"modelInstanceVersion","sourceId":311741,"databundleVersionId":11641144,"modelInstanceId":264400,"modelId":285488,"isSourceIdPinned":false}],"dockerImageVersionId":31260,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Reference\n\nhttps://www.kaggle.com/code/qiweiyin/protenix-v1-inference-2026\n\nhttps://www.kaggle.com/code/nihilisticneuralnet/0-409-stanford-rna-folding-2-protenix-template\n\nhttps://www.kaggle.com/code/alexxanderlarko/protenix-v1","metadata":{"execution":{"iopub.execute_input":"2026-02-16T19:38:36.011993Z","iopub.status.busy":"2026-02-16T19:38:36.011656Z","iopub.status.idle":"2026-02-16T19:38:36.016065Z","shell.execute_reply":"2026-02-16T19:38:36.015433Z","shell.execute_reply.started":"2026-02-16T19:38:36.011965Z"}}},{"cell_type":"markdown","source":"| version | description             | LB    \n|---------|-------------------------|-------\n| 3       | protenix+TBM            | 0.408 \n| 4       | pure protenix           | 0.249 \n| 5       | USE_MSA and USE_RNA_MSA | TBC   \n| 6       | USE_MSA                 | TBC\n| 7       | USE_RNA_MSA             | TBC","metadata":{}},{"cell_type":"code","source":"#!pip install /kaggle/input/datasets/ogurtsov/biopython/biopython-1.85-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n\n# # ── Setup finetuned Protenix (rmsa repo + AIDO.RNA) ──\n# import os\n\n# STANDARD_PTX = (\n#     \"/kaggle/input/datasets/qiweiyin/protenix-v1-adjusted\"\n#     \"/Protenix-v1-adjust-v2/Protenix-v1-adjust-v2/Protenix-v1\"\n# )\n\n# # Install extra packages needed by rmsa repo\n# !cp -r /kaggle/input/datasets/zoushuxian/protenix-packages/packages /kaggle/working\n# %cd /kaggle/working/packages\n# !pip install --no-deps --exists-action=i *.whl -q\n# %cd /kaggle/working\n\n# !mv /kaggle/working/packages/ihm-2.3/ihm-2.3 /kaggle/working\n# !mv /kaggle/working/packages/modelcif-0.7/modelcif-0.7 /kaggle/working\n# !pip install /kaggle/working/ihm-2.3 -q\n# !pip install /kaggle/working/modelcif-0.7 -q\n# !rm -rf /kaggle/working/ihm-2.3 /kaggle/working/modelcif-0.7 /kaggle/working/packages\n\n# !pip install /kaggle/input/datasets/ogurtsov/ml-collections/ml_collections-1.0.0-py3-none-any.whl -q\n\n# !cp -r /kaggle/input/datasets/zoushuxian/protenix-mg-packages/protenix_mg_packages /kaggle/working\n# %cd /kaggle/working/protenix_mg_packages\n# !pip install --no-deps --exists-action=i *.whl -q\n# %cd /kaggle/working\n# !rm -rf /kaggle/working/protenix_mg_packages\n\n# # Copy rmsa Protenix repo\n# !cp -R /kaggle/input/datasets/zoushuxian/protenix-rmsa-repo/protenix_kaggle /kaggle/working/\n# !mv /kaggle/working/protenix_kaggle /kaggle/working/Protenix_FT\n\n# # Symlink CCD files and checkpoint from standard Protenix\n# import shutil\n# from pathlib import Path\n\n# ft_dir = Path(\"/kaggle/working/Protenix_FT\")\n\n# # CCD files\n# (ft_dir / \"common\").mkdir(exist_ok=True)\n# for f in [\"components.cif\", \"components.cif.rdkit_mol.pkl\"]:\n#     src = Path(STANDARD_PTX) / \"common\" / f\n#     dst = ft_dir / \"common\" / f\n#     if src.exists() and not dst.exists():\n#         os.symlink(str(src), str(dst))\n#         print(f\"✓ Linked {f}\")\n\n# # Finetuned checkpoint\n# (ft_dir / \"checkpoint\").mkdir(exist_ok=True)\n# ft_ckpt_src = \"/kaggle/input/datasets/zoushuxian/protenix-finetuned-rna3db-all-1599/1599_ema_0.999.pt\"\n# ft_ckpt_dst = ft_dir / \"checkpoint\" / \"finetuned_rna3db.pt\"\n# if not ft_ckpt_dst.exists():\n#     os.symlink(ft_ckpt_src, str(ft_ckpt_dst))\n#     print(f\"✓ Linked finetuned checkpoint\")\n\n# print(\"✓ Finetuned Protenix setup complete\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T17:14:41.948854Z","iopub.execute_input":"2026-03-19T17:14:41.949612Z","iopub.status.idle":"2026-03-19T17:15:14.938401Z","shell.execute_reply.started":"2026-03-19T17:14:41.949583Z","shell.execute_reply":"2026-03-19T17:15:14.937703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport sys\nimport pandas as pd\n\n# ── Local vs Kaggle mode ─────────────────────────────────────────────────────\n# On Kaggle competition rerun, KAGGLE_IS_COMPETITION_RERUN is set to a truthy value.\n# When running locally we do NOT exit — instead we cap the test set to a small\n# number of samples so the notebook finishes quickly.\n\nIS_KAGGLE = bool(os.environ.get(\"KAGGLE_IS_COMPETITION_RERUN\", \"\"))\n\n# How many test samples to use when running locally\nLOCAL_N_SAMPLES = 2\n\nif IS_KAGGLE:\n    print(\"Running in KAGGLE COMPETITION mode — all test targets will be processed.\")\nelse:\n    print(f\"Running in LOCAL mode — only the first {LOCAL_N_SAMPLES} test targets \"\n          f\"will be processed to save time.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T18:02:25.77215Z","iopub.execute_input":"2026-03-21T18:02:25.77261Z","iopub.status.idle":"2026-03-21T18:02:26.965384Z","shell.execute_reply.started":"2026-03-21T18:02:25.772578Z","shell.execute_reply":"2026-03-21T18:02:26.964663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\nimport json\nimport os\nimport time\n\nos.environ[\"LAYERNORM_TYPE\"] = \"torch\"\nos.environ.setdefault(\"RNA_MSA_DEPTH_LIMIT\", \"512\")\n\nimport sys\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport torch\nfrom Bio.Align import PairwiseAligner\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T18:02:30.369167Z","iopub.execute_input":"2026-03-21T18:02:30.370037Z","iopub.status.idle":"2026-03-21T18:02:34.479705Z","shell.execute_reply.started":"2026-03-21T18:02:30.370003Z","shell.execute_reply":"2026-03-21T18:02:34.479079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_c1_mask(data: dict, atom_array) -> torch.Tensor:\n    # 1. Try atom_array attributes first\n    if atom_array is not None:\n        try:\n            if hasattr(atom_array, \"centre_atom_mask\"):\n                m = atom_array.centre_atom_mask == 1\n                if hasattr(atom_array, \"is_rna\"):\n                    m = m & atom_array.is_rna\n                return torch.from_numpy(m).bool()\n            \n            if hasattr(atom_array, \"atom_name\"):\n                base = atom_array.atom_name == \"C1'\"\n                if hasattr(atom_array, \"is_rna\"):\n                    base = base & atom_array.is_rna\n                return torch.from_numpy(base).bool()\n        except Exception:\n            pass\n\n    # 2. Fallback to feature dict\n    f = data[\"input_feature_dict\"]\n    \n    if \"centre_atom_mask\" in f:\n        return (f[\"centre_atom_mask\"] == 1).bool()\n    if \"center_atom_mask\" in f:\n        return (f[\"center_atom_mask\"] == 1).bool()\n        \n    # Heuristic fallback: check which index gives us roughly N_token atoms\n    n_tokens = data.get(\"N_token\", torch.tensor(0)).item()\n    mask11 = (f[\"atom_to_tokatom_idx\"] == 11).bool()\n    mask12 = (f[\"atom_to_tokatom_idx\"] == 12).bool()\n    \n    c11 = mask11.sum().item()\n    c12 = mask12.sum().item()\n    \n    # Return the one closer to N_tokens (likely one per residue)\n    if abs(c11 - n_tokens) < abs(c12 - n_tokens):\n        return mask11\n    else:\n        return mask12\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T18:02:37.180006Z","iopub.execute_input":"2026-03-21T18:02:37.180919Z","iopub.status.idle":"2026-03-21T18:02:37.187783Z","shell.execute_reply.started":"2026-03-21T18:02:37.18087Z","shell.execute_reply":"2026-03-21T18:02:37.186952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────── BOLTZ SETUP AND PREDICTION ──────────────────────────────────\n\nBOLTZ_AVAILABLE = False\nBOLTZ_MAX_LEN = 400  # Boltz sequence length limit\n\ndef setup_boltz():\n    \"\"\"Setup Boltz-1 model for predictions.\"\"\"\n    global BOLTZ_AVAILABLE\n    \n    try:\n        import sys\n        import types\n        \n        # Install fairscale\n        os.system(\"pip install /kaggle/input/datasets/igorkrashenyi/fairscale-0413/fairscale-0.4.13-py3-none-any.whl -q\")\n        \n        # Mock modules for offline use\n        sys.modules['ihm'] = types.ModuleType('ihm')\n        \n        modelcif_mock = types.ModuleType('modelcif')\n        modelcif_model = types.ModuleType('modelcif.model')\n        class MockClass: pass\n        modelcif_model.AbInitioModel = MockClass\n        modelcif_model.Atom = MockClass\n        modelcif_model.ModelGroup = MockClass\n        sys.modules['modelcif'] = modelcif_mock\n        sys.modules['modelcif.model'] = modelcif_model\n        \n        # Import Boltz\n        from boltz.main import Boltz1\n        \n        # Load model\n        device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n        \n        checkpoint_path = \"/kaggle/input/datasets/youhanlee/rna-prediction-boltz/boltz1_conf.ckpt\"\n        ccd_path = \"/kaggle/input/datasets/youhanlee/rna-prediction-boltz/ccd.pkl\"\n        \n        if not os.path.exists(checkpoint_path):\n            print(\"  Boltz checkpoint not found, skipping Boltz\")\n            return None, None, None\n        \n        model = Boltz1.from_pretrained(checkpoint_path, map_location=device)\n        model.eval()\n        \n        # Load CCD\n        import pickle\n        with open(ccd_path, \"rb\") as f:\n            ccd = pickle.load(f)\n        \n        BOLTZ_AVAILABLE = True\n        print(f\"  ✓ Boltz-1 loaded on {device}\")\n        return model, ccd, device\n        \n    except Exception as e:\n        print(f\"  ✗ Boltz setup failed: {e}\")\n        return None, None, None\n\n\ndef predict_boltz_single(sequence, target_id, model, ccd, device, n_samples=5):\n    \"\"\"Generate Boltz predictions for a single sequence.\"\"\"\n    \n    if len(sequence) > BOLTZ_MAX_LEN:\n        print(f\"  {target_id}: sequence too long for Boltz ({len(sequence)} > {BOLTZ_MAX_LEN})\")\n        return None\n    \n    try:\n        from boltz.main import BoltzDiffusionParams, BoltzProcessedInput\n        \n        # Create temp directory\n        temp_dir = f\"/kaggle/working/boltz_temp_{target_id}\"\n        os.makedirs(temp_dir, exist_ok=True)\n        \n        # Write FASTA\n        fasta_path = os.path.join(temp_dir, \"input.fasta\")\n        with open(fasta_path, 'w') as f:\n            f.write(f\">{target_id}\\n{sequence}\\n\")\n        \n        # Process input\n        processed = BoltzProcessedInput.from_fasta(\n            fasta_path,\n            ccd=ccd,\n            max_msa_seqs=1,\n        )\n        \n        # Predict\n        params = BoltzDiffusionParams(\n            recycling_steps=3,\n            sampling_steps=200,\n            diffusion_samples=n_samples,\n        )\n        \n        with torch.no_grad():\n            output = model.predict(\n                processed,\n                params,\n                output_dir=temp_dir,\n                data_dir=temp_dir,\n            )\n        \n        # Parse output - find PDB/CIF files\n        coords_list = []\n        \n        # Check for output files\n        for fname in os.listdir(temp_dir):\n            if fname.endswith('.pdb') or fname.endswith('.cif'):\n                fpath = os.path.join(temp_dir, fname)\n                coords = parse_boltz_output(fpath, len(sequence))\n                if coords is not None:\n                    coords_list.append(coords)\n        \n        # Cleanup\n        import shutil\n        shutil.rmtree(temp_dir, ignore_errors=True)\n        \n        if coords_list:\n            result = np.stack(coords_list[:n_samples], axis=0)\n            return result\n        \n        return None\n        \n    except Exception as e:\n        print(f\"  {target_id}: Boltz prediction failed - {e}\")\n        # Cleanup on error\n        import shutil\n        shutil.rmtree(f\"/kaggle/working/boltz_temp_{target_id}\", ignore_errors=True)\n        return None\n\n\ndef parse_boltz_output(file_path, expected_len):\n    \"\"\"Extract C1' coordinates from Boltz output file.\"\"\"\n    coords = []\n    \n    try:\n        with open(file_path, 'r') as f:\n            for line in f:\n                # PDB format\n                if line.startswith('ATOM') and \"C1'\" in line:\n                    try:\n                        x = float(line[30:38])\n                        y = float(line[38:46])\n                        z = float(line[46:54])\n                        coords.append([x, y, z])\n                    except:\n                        pass\n        \n        if len(coords) == 0:\n            return None\n        \n        coords = np.array(coords)\n        \n        # Adjust length if needed\n        if len(coords) > expected_len:\n            coords = coords[:expected_len]\n        elif len(coords) < expected_len:\n            # Pad with extrapolation\n            n_pad = expected_len - len(coords)\n            if len(coords) > 1:\n                direction = coords[-1] - coords[-2]\n                padding = np.array([coords[-1] + direction * (i + 1) for i in range(n_pad)])\n            else:\n                padding = np.zeros((n_pad, 3))\n            coords = np.vstack([coords, padding])\n        \n        return coords\n        \n    except Exception as e:\n        return None\n\n\n# ─────────────── SMART ENSEMBLE SELECTION ────────────────────────────────────\n\ndef kabsch_rmsd(coords1, coords2):\n    \"\"\"\n    RMSD after optimal Kabsch superposition.\n    Removes translation + rotation so we measure TRUE structural difference.\n    \"\"\"\n    if coords1.shape != coords2.shape:\n        min_len = min(len(coords1), len(coords2))\n        coords1 = coords1[:min_len]\n        coords2 = coords2[:min_len]\n    \n    if len(coords1) < 3:\n        diff = coords1 - coords2\n        return np.sqrt(np.mean(np.sum(diff ** 2, axis=1)))\n    \n    c1 = coords1 - coords1.mean(axis=0)\n    c2 = coords2 - coords2.mean(axis=0)\n    \n    H = c1.T @ c2\n    U, S, Vt = np.linalg.svd(H)\n    d = np.linalg.det(Vt.T @ U.T)\n    R = Vt.T @ np.diag([1, 1, np.sign(d)]) @ U.T\n    \n    c1_rotated = c1 @ R.T\n    diff = c1_rotated - c2\n    return np.sqrt(np.mean(np.sum(diff ** 2, axis=1)))\n\n\ndef smart_ensemble_selection(candidates, n_select=5):\n    \"\"\"\n    Consistency-focused selection (inspired by paper's agentic tree search):\n    1. Anchor with best Protenix prediction\n    2. Include best TBM prediction\n    3. Fill remaining with predictions CLOSEST to anchor (consistency > diversity)\n    \"\"\"\n    if len(candidates) == 0:\n        return []\n    if len(candidates) <= n_select:\n        return [c[1] for c in candidates]\n    \n    selected = []\n    selected_indices = set()\n    anchor = None\n    \n    # Step 1: Anchor with best Protenix (highest weight model)\n    protenix_cands = [(i, c) for i, c in enumerate(candidates) if c[0] == 'Protenix']\n    if protenix_cands:\n        if len(protenix_cands) >= 3:\n            best_idx = protenix_cands[0][0]\n            best_avg = float('inf')\n            for i, (pi, pc) in enumerate(protenix_cands):\n                avg = np.mean([\n                    kabsch_rmsd(pc[1], protenix_cands[j][1][1])\n                    for j, (pj, pjc) in enumerate(protenix_cands) if j != i\n                ])\n                if avg < best_avg:\n                    best_avg = avg\n                    best_idx = pi\n            selected.append(candidates[best_idx][1])\n            selected_indices.add(best_idx)\n            anchor = candidates[best_idx][1]\n        else:\n            idx = protenix_cands[0][0]\n            selected.append(candidates[idx][1])\n            selected_indices.add(idx)\n            anchor = candidates[idx][1]\n    \n    # Step 2: Add best TBM (structural prior from template)\n    tbm_cands = [(i, c) for i, c in enumerate(candidates) if c[0] == 'TBM']\n    if tbm_cands and len(selected) < n_select:\n        idx = tbm_cands[0][0]\n        if idx not in selected_indices:\n            selected.append(candidates[idx][1])\n            selected_indices.add(idx)\n    \n    # If no anchor yet, use first available\n    if anchor is None and selected:\n        anchor = selected[0]\n    elif anchor is None and candidates:\n        anchor = candidates[0][1]\n        selected.append(candidates[0][1])\n        selected_indices.add(0)\n    \n    # Step 3: Fill with predictions most CONSISTENT with anchor\n    # Sort remaining by: closeness to anchor (low RMSD) weighted by model reliability\n    remaining = []\n    for i, (model_name, coords, weight) in enumerate(candidates):\n        if i in selected_indices:\n            continue\n        rmsd_to_anchor = kabsch_rmsd(coords, anchor)\n        # Score: prefer low RMSD (consistent) and high weight (reliable)\n        # Lower score = better candidate\n        consistency_score = rmsd_to_anchor / max(weight, 0.01)\n        remaining.append((i, consistency_score))\n    \n    remaining.sort(key=lambda x: x[1])\n    \n    for idx, score in remaining:\n        if len(selected) >= n_select:\n            break\n        selected.append(candidates[idx][1])\n        selected_indices.add(idx)\n    \n    return selected","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T18:02:40.293184Z","iopub.execute_input":"2026-03-21T18:02:40.293752Z","iopub.status.idle":"2026-03-21T18:02:40.317197Z","shell.execute_reply.started":"2026-03-21T18:02:40.29371Z","shell.execute_reply":"2026-03-21T18:02:40.316318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────── DRfold2 SETUP ───────────────────────────────────────────────\nimport shutil\n\nDRFOLD2_DIR = \"/kaggle/input/datasets/error1249x/drfold2-weights/DRfold2\"  # adjust to your dataset name\nDRFOLD2_WORK = \"/kaggle/working/DRfold2\"\nDRFOLD2_MAX_LEN = 500  # skip very long sequences\n\ndef setup_drfold2():\n    \"\"\"Copy DRfold2 to working dir (needs write access).\"\"\"\n    try:\n        if os.path.exists(DRFOLD2_WORK):\n            shutil.rmtree(DRFOLD2_WORK)\n        shutil.copytree(DRFOLD2_DIR, DRFOLD2_WORK)\n        \n        # Make Arena executable\n        arena = os.path.join(DRFOLD2_WORK, \"PotentialFold\", \"bin\", \"Arena\")\n        if os.path.exists(arena):\n            os.chmod(arena, 0o755)\n        \n        # Also check Arena in Arena/ dir\n        arena2 = os.path.join(DRFOLD2_WORK, \"Arena\", \"Arena\")\n        if os.path.exists(arena2):\n            os.chmod(arena2, 0o755)\n        \n        print(f\"✓ DRfold2 ready at {DRFOLD2_WORK}\")\n        return True\n    except Exception as e:\n        print(f\"✗ DRfold2 setup failed: {e}\")\n        return False\n\n\ndef run_drfold2_single(target_id, sequence):\n    \"\"\"Run DRfold2 on a single sequence, return list of (seq_len, 3) arrays.\"\"\"\n    if len(sequence) > DRFOLD2_MAX_LEN:\n        return []\n    \n    import subprocess\n    \n    work = os.path.join(DRFOLD2_WORK, f\"run_{target_id}\")\n    os.makedirs(work, exist_ok=True)\n    \n    # Write FASTA\n    fasta_path = os.path.join(work, \"input.fasta\")\n    with open(fasta_path, 'w') as f:\n        f.write(f\">{target_id}\\n{sequence}\\n\")\n    \n    out_dir = os.path.join(work, \"output\")\n    \n    try:\n        # Run DRfold2 (single model mode — faster)\n        result = subprocess.run(\n            ['python', os.path.join(DRFOLD2_WORK, 'DRfold_infer.py'),\n             fasta_path, out_dir],\n            capture_output=True, text=True, timeout=300,  # 5 min max per target\n            cwd=DRFOLD2_WORK\n        )\n        \n        # Parse all PDB files from rets_dir (raw predictions, no refinement needed)\n        import glob\n        pdb_files = glob.glob(os.path.join(out_dir, 'rets_dir', '*.pdb'))\n        \n        # Also check relax/ and folds/ for refined outputs\n        pdb_files += glob.glob(os.path.join(out_dir, 'relax', '*.pdb'))\n        pdb_files += glob.glob(os.path.join(out_dir, 'folds', '*.pdb'))\n        \n        coords_list = []\n        for pdb_path in pdb_files:\n            coords = parse_drfold2_pdb(pdb_path, len(sequence))\n            if coords is not None:\n                coords_list.append(coords)\n        \n        return coords_list\n        \n    except subprocess.TimeoutExpired:\n        print(f\"    {target_id}: DRfold2 timeout\")\n        return []\n    except Exception as e:\n        print(f\"    {target_id}: DRfold2 error — {e}\")\n        return []\n    finally:\n        # Cleanup to save disk space\n        shutil.rmtree(work, ignore_errors=True)\n\n\ndef parse_drfold2_pdb(pdb_path, expected_len):\n    \"\"\"Extract C1' coordinates from DRfold2 PDB output.\"\"\"\n    coords = []\n    try:\n        with open(pdb_path, 'r') as f:\n            for line in f:\n                if line.startswith('ATOM') and \" C1'\" in line:\n                    try:\n                        x = float(line[30:38])\n                        y = float(line[38:46])\n                        z = float(line[46:54])\n                        coords.append([x, y, z])\n                    except:\n                        pass\n        \n        if len(coords) == 0:\n            return None\n        \n        coords = np.array(coords, dtype=np.float64)\n        \n        # Adjust to expected length\n        if len(coords) > expected_len:\n            coords = coords[:expected_len]\n        elif len(coords) < expected_len:\n            # Pad with last valid coord\n            last = coords[-1].copy()\n            padding = np.tile(last, (expected_len - len(coords), 1))\n            coords = np.vstack([coords, padding])\n        \n        return coords\n    except:\n        return None\n\n\ndef drfold2_phase(test_df):\n    \"\"\"Phase 1.5: Run DRfold2 on all targets.\"\"\"\n    print(f\"\\n{'='*60}\")\n    print(f\"PHASE 1.5: DRfold2 Ab Initio Predictions\")\n    print(f\"{'='*60}\")\n    \n    drfold2_preds = {}\n    t0 = time.time()\n    \n    for idx, row in test_df.iterrows():\n        tid = row['target_id']\n        seq = row['sequence']\n        \n        if len(seq) > DRFOLD2_MAX_LEN:\n            print(f\"  {tid} ({len(seq)} nt): skipped (>{DRFOLD2_MAX_LEN})\")\n            continue\n        \n        print(f\"  {tid} ({len(seq)} nt): running...\", end=\"\", flush=True)\n        preds = run_drfold2_single(tid, seq)\n        \n        if preds:\n            # Take up to 5 diverse predictions\n            drfold2_preds[tid] = preds[:5]\n            print(f\" ✓ {len(preds)} predictions\")\n        else:\n            print(f\" ✗ no output\")\n    \n    elapsed = time.time() - t0\n    print(f\"\\n✓ DRfold2 done in {elapsed/60:.1f} min\")\n    print(f\"  Predictions for: {len(drfold2_preds)}/{len(test_df)} targets\")\n    \n    # Free GPU memory\n    gc.collect()\n    torch.cuda.empty_cache()\n    \n    return drfold2_preds\n\nDRFOLD2_AVAILABLE = setup_drfold2()\nprint(f\"DRfold2 available: {DRFOLD2_AVAILABLE}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────── Paths & Constants ───────────────────────────────────────────\nDATA_BASE              = \"/kaggle/input/stanford-rna-3d-folding-2\"\nDEFAULT_TEST_CSV       = f\"{DATA_BASE}/test_sequences.csv\"\nDEFAULT_TRAIN_CSV      = f\"{DATA_BASE}/train_sequences.csv\"\nDEFAULT_TRAIN_LBLS     = f\"{DATA_BASE}/train_labels.csv\"\nDEFAULT_VAL_CSV        = f\"{DATA_BASE}/validation_sequences.csv\"\nDEFAULT_VAL_LBLS       = f\"{DATA_BASE}/validation_labels.csv\"\nDEFAULT_OUTPUT         = \"/kaggle/working/submission.csv\"\n\nDEFAULT_CODE_DIR = (\n    \"/kaggle/input/datasets/qiweiyin/protenix-v1-adjusted\"\n    \"/Protenix-v1-adjust-v2/Protenix-v1-adjust-v2/Protenix-v1\"\n)\nDEFAULT_ROOT_DIR = DEFAULT_CODE_DIR\n\nMODEL_NAME    = \"protenix_base_20250630_v1.0.0\"\nN_SAMPLE      = 5\nSEED          = 42\nMAX_SEQ_LEN   = int(os.environ.get(\"MAX_SEQ_LEN\",   \"512\"))\nCHUNK_OVERLAP = int(os.environ.get(\"CHUNK_OVERLAP\",  \"128\"))\n\n# TBM quality thresholds — sequences below these get routed to Protenix\nMIN_SIMILARITY       = float(os.environ.get(\"MIN_SIMILARITY\",       \"0.0\"))\nMIN_PERCENT_IDENTITY = float(os.environ.get(\"MIN_PERCENT_IDENTITY\", \"50.0\"))\n\n# Set False to skip Protenix and use de-novo fallback instead\nUSE_PROTENIX = True\n\n\ndef parse_bool(value: str, default: bool = False) -> str:\n    v = str(value).strip().lower()\n    if v in {\"1\", \"true\", \"t\", \"yes\", \"y\", \"on\"}:\n        return \"true\"\n    if v in {\"0\", \"false\", \"f\", \"no\", \"n\", \"off\"}:\n        return \"false\"\n    return \"true\" if default else \"false\"\n\n\nUSE_MSA      = parse_bool(os.environ.get(\"USE_MSA\",      \"false\"))\nUSE_TEMPLATE = parse_bool(os.environ.get(\"USE_TEMPLATE\", \"false\"))\nUSE_RNA_MSA  = parse_bool(os.environ.get(\"USE_RNA_MSA\",  \"true\"))\n\nMODEL_N_SAMPLE = int(os.environ.get(\"MODEL_N_SAMPLE\", str(N_SAMPLE)))\n\n\n# ─────────────── General Utilities ───────────────────────────────────────────\ndef seed_everything(seed: int) -> None:\n    os.environ[\"CUBLAS_WORKSPACE_CONFIG\"] = \":4096:8\"\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    np.random.seed(seed)\n    torch.backends.cudnn.benchmark = False\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.enabled = True\n    torch.use_deterministic_algorithms(True)\n\n\ndef resolve_paths():\n    test_csv   = os.environ.get(\"TEST_CSV\",           DEFAULT_TEST_CSV)\n    output_csv = os.environ.get(\"SUBMISSION_CSV\",     DEFAULT_OUTPUT)\n    code_dir   = os.environ.get(\"PROTENIX_CODE_DIR\",  DEFAULT_CODE_DIR)\n    root_dir   = os.environ.get(\"PROTENIX_ROOT_DIR\",  DEFAULT_ROOT_DIR)\n    return test_csv, output_csv, code_dir, root_dir\n\n\ndef ensure_required_files(root_dir: str) -> None:\n    for p, name in [\n        (Path(root_dir) / \"checkpoint\" / f\"{MODEL_NAME}.pt\",          \"checkpoint\"),\n        (Path(root_dir) / \"common\" / \"components.cif\",                \"CCD file\"),\n        (Path(root_dir) / \"common\" / \"components.cif.rdkit_mol.pkl\",  \"CCD cache\"),\n    ]:\n        if not p.exists():\n            raise FileNotFoundError(f\"Missing {name}: {p}\")\n\n\n# ─────────────── Protenix Input / Config Helpers ─────────────────────────────\ndef build_input_json(df: pd.DataFrame, json_path: str) -> None:\n    data = [\n        {\n            \"name\": row[\"target_id\"],\n            \"covalent_bonds\": [],\n            \"sequences\": [{\"rnaSequence\": {\"sequence\": row[\"sequence\"], \"count\": 1}}],\n        }\n        for _, row in df.iterrows()\n    ]\n    with open(json_path, \"w\", encoding=\"utf-8\") as f:\n        json.dump(data, f)\n\n\ndef build_configs(input_json_path: str, dump_dir: str, model_name: str):\n    from configs.configs_base import configs as configs_base\n    from configs.configs_data import data_configs\n    from configs.configs_inference import inference_configs\n    from configs.configs_model_type import model_configs\n    from protenix.config.config import parse_configs\n\n    base = {**configs_base, **{\"data\": data_configs}, **inference_configs}\n\n    def deep_update(t, p):\n        for k, v in p.items():\n            if isinstance(v, dict) and k in t and isinstance(t[k], dict):\n                deep_update(t[k], v)\n            else:\n                t[k] = v\n\n    deep_update(base, model_configs[model_name])\n    arg_str = \" \".join([\n        f\"--model_name {model_name}\",\n        f\"--input_json_path {input_json_path}\",\n        f\"--dump_dir {dump_dir}\",\n        f\"--use_msa {USE_MSA}\",\n        f\"--use_template {USE_TEMPLATE}\",\n        f\"--use_rna_msa {USE_RNA_MSA}\",\n        f\"--sample_diffusion.N_sample {MODEL_N_SAMPLE}\",\n        f\"--seeds {SEED}\",\n    ])\n    return parse_configs(configs=base, arg_str=arg_str, fill_required_with_null=True)\n\n\ndef get_c1_mask(data: dict, atom_array) -> torch.Tensor:\n    # 1. Try atom_array attributes first\n    if atom_array is not None:\n        try:\n            if hasattr(atom_array, \"centre_atom_mask\"):\n                m = atom_array.centre_atom_mask == 1\n                if hasattr(atom_array, \"is_rna\"):\n                    m = m & atom_array.is_rna\n                return torch.from_numpy(m).bool()\n            \n            if hasattr(atom_array, \"atom_name\"):\n                base = atom_array.atom_name == \"C1'\"\n                if hasattr(atom_array, \"is_rna\"):\n                    base = base & atom_array.is_rna\n                return torch.from_numpy(base).bool()\n        except Exception:\n            pass\n\n    # 2. Fallback to feature dict\n    f = data[\"input_feature_dict\"]\n    \n    # CASE A: center_atom_mask exists\n    if \"center_atom_mask\" in f:\n        return (f[\"center_atom_mask\"] == 1).bool()\n    if \"centre_atom_mask\" in f:\n        return (f[\"centre_atom_mask\"] == 1).bool()\n        \n    # CASE B: Use atom_name\n    if \"atom_name\" in f:\n        # Check against \"C1'\" (byte encoded or string?)\n        # For now assume typical behavior is center_atom_mask is present.\n        pass\n\n    # CASE C: atom_to_tokatom_idx fallback\n    # The index for C1' is typically 11 or 12 depending on featurizer.\n    # Let's try to match exactly C1' if possible.\n    # But usually 'centre_atom_mask' should be there.\n    \n    # If we fall through, assume standard mask\n    return (f[\"atom_to_tokatom_idx\"] == 11).bool()\n\n\ndef get_feature_c1_mask(data: dict) -> torch.Tensor:\n    f = data[\"input_feature_dict\"]\n    if \"centre_atom_mask\" in f:\n        return f[\"centre_atom_mask\"].long() == 1\n    return f[\"atom_to_tokatom_idx\"].long() == 12\n\n\ndef coords_to_rows(target_id: str, seq: str, coords: np.ndarray) -> list:\n    \"\"\"coords shape: (N_SAMPLE, seq_len, 3)\"\"\"\n    rows = []\n    for i in range(len(seq)):\n        row = {\"ID\": f\"{target_id}_{i + 1}\", \"resname\": seq[i], \"resid\": i + 1}\n        for s in range(N_SAMPLE):\n            if s < coords.shape[0] and i < coords.shape[1]:\n                x, y, z = coords[s, i]\n            else:\n                x, y, z = 0.0, 0.0, 0.0\n            row[f\"x_{s + 1}\"] = float(x)\n            row[f\"y_{s + 1}\"] = float(y)\n            row[f\"z_{s + 1}\"] = float(z)\n        rows.append(row)\n    return rows\n\n\ndef pad_samples(coords: np.ndarray, n: int) -> np.ndarray:\n    if coords.shape[0] >= n:\n        return coords[:n]\n    if coords.shape[0] == 0:\n        return np.zeros((n, coords.shape[1], 3), dtype=coords.dtype)\n    extra = np.repeat(coords[:1], n - coords.shape[0], axis=0)\n    return np.concatenate([coords, extra], axis=0)\n\n\n# ─────────────── TBM Core Functions ──────────────────────────────────────────\ndef _make_aligner() -> PairwiseAligner:\n    al = PairwiseAligner()\n    al.mode                           = \"global\"\n    al.match_score                    = 2\n    al.mismatch_score                 = -1.5\n    al.open_gap_score                 = -8\n    al.extend_gap_score               = -0.4\n    al.query_left_open_gap_score      = -8\n    al.query_left_extend_gap_score    = -0.4\n    al.query_right_open_gap_score     = -8\n    al.query_right_extend_gap_score   = -0.4\n    al.target_left_open_gap_score     = -8\n    al.target_left_extend_gap_score   = -0.4\n    al.target_right_open_gap_score    = -8\n    al.target_right_extend_gap_score  = -0.4\n    return al\n\n\n_aligner = _make_aligner()\n\n\ndef parse_stoichiometry(stoich: str) -> list:\n    if pd.isna(stoich) or str(stoich).strip() == \"\":\n        return []\n    return [(ch.strip(), int(cnt)) for part in str(stoich).split(\";\")\n            for ch, cnt in [part.split(\":\")]]\n\n\ndef parse_fasta(fasta_content: str) -> dict:\n    out, cur, parts = {}, None, []\n    for line in str(fasta_content).splitlines():\n        line = line.strip()\n        if not line:\n            continue\n        if line.startswith(\">\"):\n            if cur is not None:\n                out[cur] = \"\".join(parts)\n            cur = line[1:].split()[0]\n            parts = []\n        else:\n            parts.append(line.replace(\" \", \"\"))\n    if cur is not None:\n        out[cur] = \"\".join(parts)\n    return out\n\n\ndef get_chain_segments(row) -> list:\n    seq    = row[\"sequence\"]\n    stoich = row.get(\"stoichiometry\", \"\")\n    all_sq = row.get(\"all_sequences\", \"\")\n    if (pd.isna(stoich) or pd.isna(all_sq)\n            or str(stoich).strip() == \"\" or str(all_sq).strip() == \"\"):\n        return [(0, len(seq))]\n    try:\n        chain_dict = parse_fasta(all_sq)\n        order = parse_stoichiometry(stoich)\n        segs, pos = [], 0\n        for ch, cnt in order:\n            base = chain_dict.get(ch)\n            if base is None:\n                return [(0, len(seq))]\n            for _ in range(cnt):\n                segs.append((pos, pos + len(base)))\n                pos += len(base)\n        return segs if pos == len(seq) else [(0, len(seq))]\n    except Exception:\n        return [(0, len(seq))]\n\n\ndef build_segments_map(df: pd.DataFrame) -> tuple:\n    seg_map, stoich_map = {}, {}\n    for _, r in df.iterrows():\n        tid               = r[\"target_id\"]\n        seg_map[tid]      = get_chain_segments(r)\n        raw_s             = r.get(\"stoichiometry\", \"\")\n        stoich_map[tid]   = \"\" if pd.isna(raw_s) else str(raw_s)\n    return seg_map, stoich_map\n\n\ndef process_labels(labels_df: pd.DataFrame) -> dict:\n    coords = {}\n    id_arr = labels_df[\"ID\"].values.astype(str)\n    x_arr  = labels_df[\"x_1\"].values\n    y_arr  = labels_df[\"y_1\"].values\n    z_arr  = labels_df[\"z_1\"].values\n    \n    from collections import defaultdict\n    buckets = defaultdict(list)\n    \n    for i in range(len(id_arr)):\n        parts = id_arr[i].rsplit(\"_\", 1)\n        prefix = parts[0]\n        try:\n            pos = int(parts[1])\n        except:\n            pos = 0\n        buckets[prefix].append((pos, x_arr[i], y_arr[i], z_arr[i]))\n    \n    for prefix, items in buckets.items():\n        items.sort(key=lambda x: x[0])\n        arr = np.array([[x, y, z] for _, x, y, z in items], dtype=np.float64)\n        arr[np.abs(arr) > 1e10] = np.nan\n        coords[prefix] = np.nan_to_num(arr, nan=0.0)\n    \n    return coords\n\n\ndef _build_aligned_strings(query_seq, template_seq, alignment):\n    q_segs, t_segs = alignment.aligned\n    aq, at, qi, ti = [], [], 0, 0\n    for (qs, qe), (ts, te) in zip(q_segs, t_segs):\n        while qi < qs: aq.append(query_seq[qi]);    at.append(\"-\");              qi += 1\n        while ti < ts: aq.append(\"-\");              at.append(template_seq[ti]); ti += 1\n        for qp, tp in zip(range(qs, qe), range(ts, te)):\n            aq.append(query_seq[qp]); at.append(template_seq[tp])\n        qi, ti = qe, te\n    while qi < len(query_seq):    aq.append(query_seq[qi]);    at.append(\"-\");              qi += 1\n    while ti < len(template_seq): aq.append(\"-\");              at.append(template_seq[ti]); ti += 1\n    return \"\".join(aq), \"\".join(at)\n\n\ndef find_similar_sequences_detailed(query_seq, train_seqs_df, train_coords_dict, top_n=30):\n    results = []\n    for _, row in train_seqs_df.iterrows():\n        tid, tseq = row[\"target_id\"], row[\"sequence\"]\n        if tid not in train_coords_dict:\n            continue\n        if abs(len(tseq) - len(query_seq)) / max(len(tseq), len(query_seq)) > 0.3:\n            continue\n        aln       = next(iter(_aligner.align(query_seq, tseq)))\n        norm_s    = aln.score / (2 * min(len(query_seq), len(tseq)))\n        identical = sum(\n            1 for (qs, qe), (ts, te) in zip(*aln.aligned)\n            for qp, tp in zip(range(qs, qe), range(ts, te))\n            if query_seq[qp] == tseq[tp]\n        )\n        pct_id = 100 * identical / len(query_seq)\n        aq, at = _build_aligned_strings(query_seq, tseq, aln)\n        results.append((tid, tseq, norm_s, train_coords_dict[tid], pct_id, aq, at))\n    results.sort(key=lambda x: x[2], reverse=True)\n    return results[:top_n]\n\n\ndef adapt_template_to_query(query_seq, template_seq, template_coords) -> np.ndarray:\n    aln        = next(iter(_aligner.align(query_seq, template_seq)))\n    new_coords = np.full((len(query_seq), 3), np.nan)\n    for (qs, qe), (ts, te) in zip(*aln.aligned):\n        chunk = template_coords[ts:te]\n        if len(chunk) == (qe - qs):\n            new_coords[qs:qe] = chunk\n    for i in range(len(new_coords)):\n        if np.isnan(new_coords[i, 0]):\n            pv = next((j for j in range(i - 1, -1, -1) if not np.isnan(new_coords[j, 0])), -1)\n            nv = next((j for j in range(i + 1, len(new_coords)) if not np.isnan(new_coords[j, 0])), -1)\n            if pv >= 0 and nv >= 0:\n                w = (i - pv) / (nv - pv)\n                new_coords[i] = (1 - w) * new_coords[pv] + w * new_coords[nv]\n            elif pv >= 0:\n                new_coords[i] = new_coords[pv] + [3, 0, 0]\n            elif nv >= 0:\n                new_coords[i] = new_coords[nv] + [3, 0, 0]\n            else:\n                new_coords[i] = [i * 3, 0, 0]\n    return np.nan_to_num(new_coords)\n\n\ndef adaptive_rna_constraints(coords, target_id, segments_map, confidence=1.0, passes=2) -> np.ndarray:\n    X        = coords.copy()\n    segments = segments_map.get(target_id, [(0, len(X))])\n    strength = max(0.75 * (1.0 - min(confidence, 0.97)), 0.02)\n    for _ in range(passes):\n        for s, e in segments:\n            C = X[s:e]; L = e - s\n            if L < 3:\n                continue\n            # bond i–i+1  ~5.95 Å\n            d    = C[1:] - C[:-1]; dist = np.linalg.norm(d, axis=1) + 1e-6\n            adj  = d * ((5.95 - dist) / dist)[:, None] * (0.22 * strength)\n            C[:-1] -= adj; C[1:] += adj\n            # soft i–i+2  ~10.2 Å\n            d2   = C[2:] - C[:-2]; d2n = np.linalg.norm(d2, axis=1) + 1e-6\n            adj2 = d2 * ((10.2 - d2n) / d2n)[:, None] * (0.10 * strength)\n            C[:-2] -= adj2; C[2:] += adj2\n            # Laplacian smoothing\n            C[1:-1] += (0.06 * strength) * (0.5 * (C[:-2] + C[2:]) - C[1:-1])\n            # self-avoidance\n            if L >= 25:\n                idx  = np.linspace(0, L - 1, min(L, 160)).astype(int) if L > 220 else np.arange(L)\n                P    = C[idx]; diff = P[:, None, :] - P[None, :, :]\n                dm   = np.linalg.norm(diff, axis=2) + 1e-6\n                sep  = np.abs(idx[:, None] - idx[None, :])\n                mask = (sep > 2) & (dm < 3.2)\n                if np.any(mask):\n                    vec = (diff * ((3.2 - dm) / dm)[:, :, None] * mask[:, :, None]).sum(axis=1)\n                    C[idx] += (0.015 * strength) * vec\n            X[s:e] = C\n    return X\n\n\ndef _rotmat(axis, ang):\n    a = np.asarray(axis, float); a /= np.linalg.norm(a) + 1e-12\n    x, y, z = a; c, s = np.cos(ang), np.sin(ang); CC = 1 - c\n    return np.array([[c+x*x*CC, x*y*CC-z*s, x*z*CC+y*s],\n                     [y*x*CC+z*s, c+y*y*CC, y*z*CC-x*s],\n                     [z*x*CC-y*s, z*y*CC+x*s, c+z*z*CC]])\n\n\ndef apply_hinge(coords, seg, rng, deg=22):\n    s, e = seg; L = e - s\n    if L < 30: return coords\n    pivot = s + int(rng.integers(10, L - 10))\n    R = _rotmat(rng.normal(size=3), np.deg2rad(float(rng.uniform(-deg, deg))))\n    X = coords.copy(); p0 = X[pivot].copy()\n    X[pivot+1:e] = (X[pivot+1:e] - p0) @ R.T + p0\n    return X\n\n\ndef jitter_chains(coords, segs, rng, deg=12, trans=1.5):\n    X = coords.copy(); gc_ = X.mean(0, keepdims=True)\n    for s, e in segs:\n        R     = _rotmat(rng.normal(size=3), np.deg2rad(float(rng.uniform(-deg, deg))))\n        shift = rng.normal(size=3); shift = shift / (np.linalg.norm(shift) + 1e-12) * float(rng.uniform(0, trans))\n        c     = X[s:e].mean(0, keepdims=True)\n        X[s:e] = (X[s:e] - c) @ R.T + c + shift\n    X -= X.mean(0, keepdims=True) - gc_\n    return X\n\n\ndef smooth_wiggle(coords, segs, rng, amp=0.8):\n    X = coords.copy()\n    for s, e in segs:\n        L = e - s\n        if L < 20: continue\n        ctrl = np.linspace(0, L - 1, 6); disp = rng.normal(0, amp, (6, 3)); t = np.arange(L)\n        X[s:e] += np.vstack([np.interp(t, ctrl, disp[:, k]) for k in range(3)]).T\n    return X\n\n\ndef generate_rna_structure(sequence: str, seed=None) -> np.ndarray:\n    \"\"\"Idealized A-form RNA helix — last-resort de-novo fallback.\"\"\"\n    if seed is not None:\n        np.random.seed(seed)\n    n = len(sequence); coords = np.zeros((n, 3))\n    for i in range(n):\n        ang = i * 0.6\n        coords[i] = [10.0 * np.cos(ang), 10.0 * np.sin(ang), i * 2.5]\n    return coords\n\n\n# ─────────────── TBM Phase ───────────────────────────────────────────────────\ndef tbm_phase(test_df, train_seqs_df, train_coords_dict, segments_map):\n    \"\"\"\n    Phase 1 — Template-Based Modeling.\n\n    Returns\n    -------\n    template_predictions : {target_id: [np.ndarray(seq_len, 3), ...]}\n        0 to N_SAMPLE predictions per target, from real templates.\n    protenix_queue : {target_id: (n_needed, full_sequence)}\n        Targets that still need more predictions.\n    \"\"\"\n    print(f\"\\n{'='*60}\")\n    print(f\"PHASE 1: Template-Based Modeling\")\n    print(f\"  MIN_SIMILARITY = {MIN_SIMILARITY}  |  MIN_PCT_IDENTITY = {MIN_PERCENT_IDENTITY}\")\n    print(f\"{'='*60}\")\n    t0 = time.time()\n\n    template_predictions: dict = {}\n    protenix_queue:       dict = {}\n\n    for _, row in test_df.iterrows():\n        tid = row[\"target_id\"]\n        seq = row[\"sequence\"]\n        segs = segments_map.get(tid, [(0, len(seq))])\n\n        similar = find_similar_sequences_detailed(seq, train_seqs_df, train_coords_dict, top_n=30)\n        preds   = []\n        used    = set()\n\n        for i, (tmpl_id, tmpl_seq, sim, tmpl_coords, pct_id, _, _) in enumerate(similar):\n            if len(preds) >= N_SAMPLE:\n                break\n            if sim < MIN_SIMILARITY or pct_id < MIN_PERCENT_IDENTITY:\n                break           # list is sorted by sim, so no point continuing\n            if tmpl_id in used:\n                continue\n\n            rng     = np.random.default_rng((row.name * 10000000000 + i * 10007) % (2**32))\n            adapted = adapt_template_to_query(seq, tmpl_seq, tmpl_coords)\n\n            # Diversity transforms (same strategy as the 0-409 TBM notebook)\n            slot = len(preds)\n            if slot == 0:\n                X = adapted\n            elif slot == 1:\n                X = adapted + rng.normal(0, max(0.01, (0.40 - sim) * 0.06), adapted.shape)\n            elif slot == 2:\n                longest = max(segs, key=lambda se: se[1] - se[0])\n                X = apply_hinge(adapted, longest, rng)\n            elif slot == 3:\n                X = jitter_chains(adapted, segs, rng)\n            else:\n                X = smooth_wiggle(adapted, segs, rng)\n\n            refined = adaptive_rna_constraints(X, tid, segments_map, confidence=sim)\n            preds.append(refined)\n            used.add(tmpl_id)\n\n        template_predictions[tid] = preds\n        n_needed = N_SAMPLE - len(preds)\n        if n_needed > 0:\n            protenix_queue[tid] = (n_needed, seq)\n            print(f\"  {tid} ({len(seq)} nt): {len(preds)} TBM → need {n_needed} from Protenix\")\n        else:\n            print(f\"  {tid} ({len(seq)} nt): all {N_SAMPLE} from TBM ✓\")\n\n    elapsed = time.time() - t0\n    n_full  = len(test_df) - len(protenix_queue)\n    print(f\"\\nPhase 1 done in {elapsed:.1f}s\")\n    print(f\"  Fully covered by TBM : {n_full}\")\n    print(f\"  Need Protenix        : {len(protenix_queue)}\")\n    return template_predictions, protenix_queue\n\n# ─────────────── Chunking for Long Sequences ────────────────────────────────\n\ndef plan_chunks(seq_len, max_len=MAX_SEQ_LEN, overlap=CHUNK_OVERLAP):\n    \"\"\"Plan overlapping chunks for sequences longer than max_len.\"\"\"\n    if seq_len <= max_len:\n        return [(0, seq_len)]\n    chunks = []\n    stride = max_len - overlap\n    start = 0\n    while start < seq_len:\n        end = min(start + max_len, seq_len)\n        chunks.append((start, end))\n        if end == seq_len:\n            break\n        start += stride\n    if len(chunks) > 1 and chunks[-1][1] - chunks[-1][0] < overlap:\n        chunks.pop()\n    if len(chunks) > 1:\n        last_s, last_e = chunks[-1]\n        if last_e - last_s > max_len:\n            chunks[-1] = (last_e - max_len, last_e)\n    return chunks\n\n\ndef stitch_chunks(chunk_coords_list, chunk_ranges, full_len):\n    \"\"\"Kabsch-align overlapping chunk predictions and blend with cosine smoothing.\"\"\"\n    full_coords = np.full((full_len, 3), np.nan)\n    \n    if len(chunk_coords_list) == 1:\n        s, e = chunk_ranges[0]\n        clen = min(len(chunk_coords_list[0]), e - s)\n        full_coords[s:s+clen] = chunk_coords_list[0][:clen]\n        return np.nan_to_num(full_coords)\n    \n    s0, e0 = chunk_ranges[0]\n    c0 = chunk_coords_list[0]\n    clen0 = min(len(c0), e0 - s0)\n    full_coords[s0:s0+clen0] = c0[:clen0]\n    placed_end = s0 + clen0\n    \n    for k in range(1, len(chunk_coords_list)):\n        sk, ek = chunk_ranges[k]\n        ck = chunk_coords_list[k]\n        clenk = min(len(ck), ek - sk)\n        \n        overlap_start = sk\n        overlap_end = min(placed_end, sk + clenk)\n        overlap_len = overlap_end - overlap_start\n        \n        ck_aligned = ck[:clenk].copy()\n        \n        if overlap_len >= 3:\n            ref_overlap = full_coords[overlap_start:overlap_end]\n            local_overlap = ck[:overlap_len]\n            valid = ~np.any(np.isnan(ref_overlap), axis=1)\n            if valid.sum() >= 3:\n                ref_pts = ref_overlap[valid]\n                chk_pts = local_overlap[valid]\n                ref_c = ref_pts.mean(axis=0)\n                chk_c = chk_pts.mean(axis=0)\n                H = (chk_pts - chk_c).T @ (ref_pts - ref_c)\n                U, S, Vt = np.linalg.svd(H)\n                d = np.linalg.det(Vt.T @ U.T)\n                R = Vt.T @ np.diag([1, 1, np.sign(d)]) @ U.T\n                ck_aligned = (ck[:clenk] - chk_c) @ R.T + ref_c\n        \n        for pos in range(clenk):\n            gpos = sk + pos\n            if gpos >= full_len:\n                break\n            if np.isnan(full_coords[gpos, 0]):\n                full_coords[gpos] = ck_aligned[pos]\n            else:\n                # Cosine blending: smoother S-curve transition in overlap\n                t_linear = pos / max(overlap_len - 1, 1) if overlap_len > 1 else 0.5\n                t = 0.5 * (1 - np.cos(np.pi * t_linear))  # S-curve: 0→0, 0.5→0.5, 1→1\n                full_coords[gpos] = (1 - t) * full_coords[gpos] + t * ck_aligned[pos]\n        \n        placed_end = max(placed_end, sk + clenk)\n    \n    # Fill remaining NaN with last valid coord (not zeros)\n    last_valid = np.array([0.0, 0.0, 0.0])\n    for i in range(len(full_coords)):\n        if not np.isnan(full_coords[i, 0]):\n            last_valid = full_coords[i].copy()\n        else:\n            full_coords[i] = last_valid\n    return full_coords\n\n\n\n# ─────────────── Main ────────────────────────────────────────────────────────\ndef main() -> None:\n    test_csv, output_csv, code_dir, root_dir = resolve_paths()\n\n    if not os.path.isdir(code_dir):\n        raise FileNotFoundError(\n            f\"Missing PROTENIX_CODE_DIR: {code_dir}. \"\n            \"Set PROTENIX_CODE_DIR to the repo path.\"\n        )\n\n    os.environ[\"PROTENIX_ROOT_DIR\"] = root_dir\n    sys.path.append(code_dir)\n    ensure_required_files(root_dir)\n    seed_everything(SEED)\n\n    # ── Load test data ──────────────────────────────────────────────────────\n    test_df_full = pd.read_csv(test_csv)\n    test_df      = (test_df_full.head(LOCAL_N_SAMPLES) if not IS_KAGGLE\n                    else test_df_full).reset_index(drop=True)\n    print(f\"Test targets : {len(test_df)}\"\n          + (\" (LOCAL MODE)\" if not IS_KAGGLE else \"\"))\n\n    seq_by_id = dict(zip(test_df[\"target_id\"], test_df[\"sequence\"]))\n\n    # Truncated copy for Protenix (Protenix has token limits)\n    test_df_trunc = test_df.copy()\n    test_df_trunc[\"sequence\"] = test_df_trunc[\"sequence\"].str[:MAX_SEQ_LEN]\n\n    # ── Load training data for TBM ──────────────────────────────────────────\n    print(\"\\nLoading training data for TBM …\")\n    train_seqs   = pd.read_csv(DEFAULT_TRAIN_CSV)\n    val_seqs     = pd.read_csv(DEFAULT_VAL_CSV)\n    train_labels = pd.read_csv(DEFAULT_TRAIN_LBLS,low_memory=False)\n    val_labels   = pd.read_csv(DEFAULT_VAL_LBLS)\n\n    combined_seqs   = pd.concat([train_seqs,   val_seqs],    ignore_index=True)\n    combined_labels = pd.concat([train_labels, val_labels],  ignore_index=True)\n    train_coords    = process_labels(combined_labels)\n    del combined_labels; gc.collect()\n    segments_map, _ = build_segments_map(test_df)\n\n    print(f\"Template pool: {len(combined_seqs)} sequences, {len(train_coords)} structures\")\n   \n\n    # ─── PHASE 1: TBM ──────────────────────────────────────────────────────\n    template_preds, protenix_queue = tbm_phase(\n        test_df, combined_seqs, train_coords, segments_map\n    )\n\n    # ─── PHASE 1.5: DRfold2 ────────────────────────────────────────────────\n    drfold2_preds = {}\n    if DRFOLD2_AVAILABLE:\n        try:\n            drfold2_preds = drfold2_phase(test_df)\n        except Exception as e:\n            print(f\"⚠ DRfold2 phase failed: {e}\")\n            drfold2_preds = {}\n    # ─── PHASE 2: Protenix (only for targets that need extra predictions) ──\n    # ─── PHASE 2: Protenix with chunking for long sequences ────────────────\n    protenix_preds: dict = {}\n\n    if protenix_queue and USE_PROTENIX:\n        print(f\"\\n{'='*60}\")\n        print(f\"PHASE 2: Protenix for {len(protenix_queue)} targets (with chunking)\")\n        print(f\"{'='*60}\")\n\n        work_dir = Path(\"/kaggle/working\")\n        work_dir.mkdir(parents=True, exist_ok=True)\n\n        from protenix.data.inference.infer_dataloader import InferenceDataset\n        from runner.inference import (InferenceRunner,\n                                      update_gpu_compatible_configs,\n                                      update_inference_configs)\n\n        # Build chunk map: for each target, plan chunks\n        chunk_map = {}  # target_id -> list of (start, end)\n        chunk_entries = []  # list of (chunk_id, target_id, chunk_seq, chunk_range)\n        \n        for tid, (n_needed, full_seq) in protenix_queue.items():\n            chunks = plan_chunks(len(full_seq))\n            chunk_map[tid] = chunks\n            if len(chunks) == 1:\n                # Short sequence: just truncate as before\n                chunk_entries.append((f\"{tid}\", tid, full_seq[:MAX_SEQ_LEN], chunks[0]))\n                print(f\"  {tid} ({len(full_seq)} nt): 1 chunk (no splitting needed)\")\n            else:\n                for ci, (cs, ce) in enumerate(chunks):\n                    chunk_seq = full_seq[cs:ce]\n                    chunk_id = f\"{tid}__chunk{ci}\"\n                    chunk_entries.append((chunk_id, tid, chunk_seq, (cs, ce)))\n                print(f\"  {tid} ({len(full_seq)} nt): {len(chunks)} chunks {chunks}\")\n\n        # Build input JSON with all chunks as separate entries\n        chunk_data = []\n        for chunk_id, tid, chunk_seq, _ in chunk_entries:\n            chunk_data.append({\n                \"name\": chunk_id,\n                \"covalent_bonds\": [],\n                \"sequences\": [{\"rnaSequence\": {\"sequence\": chunk_seq, \"count\": 1}}],\n            })\n        \n        input_json_path = str(work_dir / \"protenix_chunk_input.json\")\n        with open(input_json_path, \"w\", encoding=\"utf-8\") as f:\n            json.dump(chunk_data, f)\n\n        configs = build_configs(input_json_path, str(work_dir / \"outputs\"), MODEL_NAME)\n        configs = update_gpu_compatible_configs(configs)\n        runner  = InferenceRunner(configs)\n        dataset = InferenceDataset(configs)\n\n        # Run inference with multiple seeds for diversity\n        seeds_to_run = [42, 137]  # Two seeds for diversity\n        chunk_results = {}  # chunk_id -> np.array (n_samples, chunk_len, 3)\n\n        for seed_idx, run_seed in enumerate(seeds_to_run):\n            seed_everything(run_seed)\n            print(f\"\\n  --- Protenix pass {seed_idx+1}/{len(seeds_to_run)} (seed={run_seed}) ---\")\n            \n            for i in tqdm(range(len(dataset)), desc=f\"Protenix seed={run_seed}\"):\n                data, atom_array, error_message = dataset[i]\n                sample_name = data.get(\"sample_name\", f\"sample_{i}\")\n\n                if error_message:\n                    print(f\"  {sample_name}: data error — {error_message}\")\n                    del data, atom_array, error_message\n                    gc.collect(); torch.cuda.empty_cache(); gc.collect()\n                    continue\n\n                matching = [e for e in chunk_entries if e[0] == sample_name]\n                if not matching:\n                    print(f\"  {sample_name}: not in chunk_entries, skipping\")\n                    del data, atom_array, error_message\n                    gc.collect(); torch.cuda.empty_cache(); gc.collect()\n                    continue\n                \n                chunk_id, parent_tid, chunk_seq, chunk_range = matching[0]\n                n_needed = protenix_queue[parent_tid][0]\n\n                try:\n                    new_cfg = update_inference_configs(configs, data[\"N_token\"].item())\n                    # For multi-seed, generate fewer samples per seed\n                    samples_this_pass = max(2, (n_needed + len(seeds_to_run) - 1) // len(seeds_to_run))\n                    new_cfg.sample_diffusion.N_sample = samples_this_pass\n                    new_cfg.seeds = [run_seed]\n                    runner.update_model_configs(new_cfg)\n\n                    prediction = runner.predict(data)\n                    raw_coords = prediction[\"coordinate\"]\n                    feat = data[\"input_feature_dict\"]\n\n                    if \"centre_atom_mask\" in feat:\n                        mask = (feat[\"centre_atom_mask\"] == 1).to(raw_coords.device)\n                    elif \"atom_to_tokatom_idx\" in feat:\n                        m11 = (feat[\"atom_to_tokatom_idx\"] == 11).to(raw_coords.device)\n                        m12 = (feat[\"atom_to_tokatom_idx\"] == 12).to(raw_coords.device)\n                        c11, c12 = m11.sum(), m12.sum()\n                        target_len = len(chunk_seq)\n                        mask = m11 if abs(c11 - target_len) < abs(c12 - target_len) else m12\n                    else:\n                        mask = torch.zeros(raw_coords.shape[1], dtype=torch.bool, device=raw_coords.device)\n\n                    coords = raw_coords[:, mask, :].detach().cpu().numpy()\n                    \n                    # Accumulate results across seeds\n                    if chunk_id in chunk_results and chunk_results[chunk_id] is not None:\n                        chunk_results[chunk_id] = np.concatenate(\n                            [chunk_results[chunk_id], coords], axis=0\n                        )\n                    else:\n                        chunk_results[chunk_id] = coords\n                    \n                    print(f\"  {chunk_id} seed={run_seed}: +{coords.shape[0]} samples (total: {chunk_results[chunk_id].shape[0]})\")\n\n                except Exception as exc:\n                    print(f\"  {chunk_id}: Protenix FAILED — {exc}\")\n                    import traceback\n                    traceback.print_exc()\n                    if chunk_id not in chunk_results:\n                        chunk_results[chunk_id] = None\n\n                finally:\n                    del prediction, raw_coords, mask, data, atom_array\n                    gc.collect(); torch.cuda.empty_cache(); gc.collect()\n\n        # Stitch chunks back together per target\n        for tid, (n_needed, full_seq) in protenix_queue.items():\n            chunks = chunk_map[tid]\n            full_len = len(full_seq)\n            \n            if len(chunks) == 1:\n                # Single chunk (short sequence) — same as before\n                result = chunk_results.get(tid)\n                if result is None:\n                    protenix_preds[tid] = None\n                    continue\n                coords = result\n                if coords.shape[1] != full_len:\n                    padded = np.zeros((coords.shape[0], full_len, 3), dtype=np.float32)\n                    ml = min(coords.shape[1], full_len)\n                    padded[:, :ml, :] = coords[:, :ml, :]\n                    # Pad remaining with last valid coord\n                    for si in range(padded.shape[0]):\n                        if ml > 0 and ml < full_len:\n                            padded[si, ml:] = padded[si, ml-1]\n                    coords = padded\n                protenix_preds[tid] = coords\n            else:\n                # Multiple chunks — stitch per sample\n                # Gather chunk results\n                chunk_coords = []\n                all_ok = True\n                for ci in range(len(chunks)):\n                    cid = f\"{tid}__chunk{ci}\"\n                    cr = chunk_results.get(cid)\n                    if cr is None:\n                        all_ok = False\n                        break\n                    chunk_coords.append(cr)\n                \n                if not all_ok or not chunk_coords:\n                    print(f\"  {tid}: some chunks failed, falling back to first chunk only\")\n                    first_cr = chunk_results.get(f\"{tid}__chunk0\")\n                    if first_cr is not None:\n                        padded = np.zeros((first_cr.shape[0], full_len, 3), dtype=np.float32)\n                        ml = min(first_cr.shape[1], full_len)\n                        padded[:, :ml, :] = first_cr[:, :ml, :]\n                        protenix_preds[tid] = padded\n                    else:\n                        protenix_preds[tid] = None\n                    continue\n                \n                # Stitch each sample independently\n                n_samples = chunk_coords[0].shape[0]\n                stitched_all = np.zeros((n_samples, full_len, 3), dtype=np.float32)\n                \n                for si in range(n_samples):\n                    sample_chunks = [cc[min(si, cc.shape[0]-1)] for cc in chunk_coords]\n                    stitched_all[si] = stitch_chunks(sample_chunks, chunks, full_len)\n                \n                protenix_preds[tid] = stitched_all\n                print(f\"  {tid}: stitched {len(chunks)} chunks → ({n_samples}, {full_len}, 3) ✓\")\n\n    elif protenix_queue and not USE_PROTENIX:\n        print(f\"\\nPHASE 2 skipped (USE_PROTENIX=False). \"\n              f\"De-novo fallback will cover {len(protenix_queue)} targets.\")\n\n    # ─── PHASE 2B: BOLTZ ────────────────────────────────────────────────────\n    boltz_preds: dict = {}\n    \n    # print(f\"\\n{'='*60}\")\n    # print(\"PHASE 2B: Boltz Predictions are diabled for this run\")\n    # print(f\"{'='*60}\")\n    \n    boltz_model, ccd, boltz_device = setup_boltz()\n    \n    if boltz_model is not None:\n        for _, row in tqdm(test_df.iterrows(), desc=\"Boltz\", total=len(test_df)):\n            tid = row[\"target_id\"]\n            seq = row[\"sequence\"]\n            \n            # Skip if sequence too long\n            if len(seq) > BOLTZ_MAX_LEN:\n                print(f\"  {tid}: skipped (len={len(seq)} > {BOLTZ_MAX_LEN})\")\n                continue\n            \n            preds = predict_boltz_single(seq, tid, boltz_model, ccd, boltz_device, n_samples=5)\n            \n            if preds is not None:\n                boltz_preds[tid] = preds\n                print(f\"  {tid}: {preds.shape[0]} Boltz predictions ✓\")\n            else:\n                print(f\"  {tid}: Boltz failed\")\n        \n        # Cleanup Boltz model to free GPU memory\n        del boltz_model, ccd\n        gc.collect()\n        torch.cuda.empty_cache()\n        \n        print(f\"\\nBoltz completed: {len(boltz_preds)}/{len(test_df)} targets\")\n    else:\n        print(\"Boltz not available, skipping\")\n\n\n    # ─── PHASE 3: Combine everything ───────────────────────────────────────\n    # ─── PHASE 3: Smart Ensemble Selection ─────────────────────────────────\n    print(f\"\\n{'='*60}\")\n    print(\"PHASE 3: Smart Ensemble Selection (TBM + Protenix + Boltz)\")\n    print(f\"{'='*60}\")\n\n    all_rows = []\n    \n    stats = {'tbm_only': 0, 'protenix_used': 0, 'drfold2_used': 0, 'boltz_used': 0, 'denovo': 0}\n\n    for _, row in test_df.iterrows():\n        tid = row[\"target_id\"]\n        seq = row[\"sequence\"]\n\n        # Gather all candidates: (model_name, coords, weight)\n        candidates = []\n        \n        # Add TBM predictions (weight: 0.15)\n        tbm_list = template_preds.get(tid, [])\n        for i, pred in enumerate(tbm_list):\n            candidates.append(('TBM', pred, 0.05))\n        \n        # Add Protenix predictions (weight: 0.75)\n        ptx = protenix_preds.get(tid)\n        if ptx is not None and hasattr(ptx, 'ndim') and ptx.ndim == 3:\n            for j in range(ptx.shape[0]):\n                candidates.append(('Protenix', ptx[j], 0.85))\n\n        # Add DRfold2 predictions (weight: 0.20)\n        dr2 = drfold2_preds.get(tid, [])\n        for pred in dr2:\n            candidates.append(('DRfold2', pred, 0.30))\n        \n        # Add Boltz predictions (weight: 0.10)\n        bltz = boltz_preds.get(tid)\n        if bltz is not None and hasattr(bltz, 'ndim') and bltz.ndim == 3:\n            for j in range(bltz.shape[0]):\n                candidates.append(('Boltz', bltz[j], 0.20))\n        \n        # Smart selection\n        if candidates:\n            selected = smart_ensemble_selection(candidates, n_select=N_SAMPLE)\n            \n            # Track stats\n            if len(tbm_list) > 0:\n                stats['tbm_only'] += 1\n            if ptx is not None:\n                stats['protenix_used'] += 1\n            if bltz is not None:\n                stats['boltz_used'] += 1\n            if dr2:\n                stats['drfold2_used'] += 1\n        else:\n            selected = []\n\n        # De-novo fallback for any still-empty slots\n        n_denovo = 0\n        while len(selected) < N_SAMPLE:\n            seed_val = row.name * 1000000 + len(selected) * 1000\n            dn = generate_rna_structure(seq, seed=seed_val)\n            selected.append(adaptive_rna_constraints(dn, tid, segments_map, confidence=0.2))\n            n_denovo += 1\n        \n        if n_denovo > 0:\n            stats['denovo'] += 1\n\n        # Stack to (N_SAMPLE, seq_len, 3) and write rows\n        stacked = np.stack(selected[:N_SAMPLE], axis=0)\n        all_rows.extend(coords_to_rows(tid, seq, stacked))\n\n    print(f\"\\nEnsemble stats:\")\n    print(f\"  Targets with TBM: {stats['tbm_only']}\")\n    print(f\"  Targets with Protenix: {stats['protenix_used']}\")\n    print(f\"  Targets with Boltz: {stats['boltz_used']}\")\n    print(f\"  Targets with de-novo fallback: {stats['denovo']}\")\n    # ── Save ───────────────────────────────────────────────────────────────\n    sub = pd.DataFrame(all_rows)\n    cols = [\"ID\", \"resname\", \"resid\"] + [\n        f\"{c}_{i}\" for i in range(1, N_SAMPLE + 1) for c in [\"x\", \"y\", \"z\"]\n    ]\n    coord_cols = [c for c in cols if c.startswith((\"x_\", \"y_\", \"z_\"))]\n    sub[coord_cols] = sub[coord_cols].fillna(0.0)\n    sub[coord_cols] = sub[coord_cols].replace([np.inf, -np.inf], 0.0)\n    sub[coord_cols] = sub[coord_cols].clip(-999.999, 9999.999)\n    sub[cols].to_csv(output_csv, index=False)\n    print(f\"\\n✓ Saved submission to {output_csv}  ({len(sub):,} rows)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T18:02:46.100848Z","iopub.execute_input":"2026-03-21T18:02:46.101599Z","iopub.status.idle":"2026-03-21T18:02:46.205875Z","shell.execute_reply.started":"2026-03-21T18:02:46.101565Z","shell.execute_reply":"2026-03-21T18:02:46.20514Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Main","metadata":{}},{"cell_type":"code","source":"\nif __name__ == \"__main__\":\n    main()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T18:02:55.711025Z","iopub.execute_input":"2026-03-21T18:02:55.711458Z","iopub.status.idle":"2026-03-21T18:06:01.726877Z","shell.execute_reply.started":"2026-03-21T18:02:55.711427Z","shell.execute_reply":"2026-03-21T18:06:01.726204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#read submission.csv\nsubmission_path = \"/kaggle/working/submission.csv\"\nsubmission_df = pd.read_csv(submission_path)\nprint(submission_df.head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T16:39:36.027403Z","iopub.execute_input":"2026-03-19T16:39:36.028031Z","iopub.status.idle":"2026-03-19T16:39:36.04567Z","shell.execute_reply.started":"2026-03-19T16:39:36.028001Z","shell.execute_reply":"2026-03-19T16:39:36.045117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}