{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210},{"sourceType":"datasetVersion","sourceId":14604295,"datasetId":9328538,"databundleVersionId":15440074},{"sourceType":"datasetVersion","sourceId":11899194,"datasetId":7479946,"databundleVersionId":12404228},{"sourceType":"datasetVersion","sourceId":13282339,"datasetId":7162026,"databundleVersionId":13982667},{"sourceType":"datasetVersion","sourceId":14874339,"datasetId":9502242,"databundleVersionId":15736806},{"sourceType":"datasetVersion","sourceId":10855324,"datasetId":6742586,"databundleVersionId":11219268},{"sourceType":"kernelVersion","sourceId":298845136}],"dockerImageVersionId":31287,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!bash /kaggle/input/notebooks/jonathanarya/dependency-installation-script/install_requirements.sh","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-09T16:27:49.128962Z","iopub.execute_input":"2026-03-09T16:27:49.129675Z","iopub.status.idle":"2026-03-09T16:27:52.611431Z","shell.execute_reply.started":"2026-03-09T16:27:49.129631Z","shell.execute_reply":"2026-03-09T16:27:52.610713Z"}},"outputs":[{"name":"stdout","text":"Looking in links: /kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages, /kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages\nRequirement already satisfied: biotite in /usr/local/lib/python3.12/dist-packages (from -r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (1.6.0)\nRequirement already satisfied: rdkit in /usr/local/lib/python3.12/dist-packages (from -r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 2)) (2025.9.5)\nRequirement already satisfied: biopython in /usr/local/lib/python3.12/dist-packages (from -r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 3)) (1.86)\nRequirement already satisfied: biotraj<2.0,>=1.0 in /usr/local/lib/python3.12/dist-packages (from biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (1.2.2)\nRequirement already satisfied: msgpack>=0.5.6 in /usr/local/lib/python3.12/dist-packages (from biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (1.1.2)\nRequirement already satisfied: networkx>=2.0 in /usr/local/lib/python3.12/dist-packages (from biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (3.6.1)\nRequirement already satisfied: numpy>=1.25 in /usr/local/lib/python3.12/dist-packages (from biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (2.0.2)\nRequirement already satisfied: packaging>=24.0 in /usr/local/lib/python3.12/dist-packages (from biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (25.0)\nRequirement already satisfied: requests>=2.12 in /usr/local/lib/python3.12/dist-packages (from biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (2.32.4)\nRequirement already satisfied: Pillow in /usr/local/lib/python3.12/dist-packages (from rdkit->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 2)) (11.3.0)\nRequirement already satisfied: scipy>=1.13 in /usr/local/lib/python3.12/dist-packages (from biotraj<2.0,>=1.0->biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (1.16.3)\nRequirement already satisfied: charset_normalizer<4,>=2 in /usr/local/lib/python3.12/dist-packages (from requests>=2.12->biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (3.4.4)\nRequirement already satisfied: idna<4,>=2.5 in /usr/local/lib/python3.12/dist-packages (from requests>=2.12->biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (3.11)\nRequirement already satisfied: urllib3<3,>=1.21.1 in /usr/local/lib/python3.12/dist-packages (from requests>=2.12->biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (2.5.0)\nRequirement already satisfied: certifi>=2017.4.17 in /usr/local/lib/python3.12/dist-packages (from requests>=2.12->biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (2026.1.4)\n","output_type":"stream"}],"execution_count":37},{"cell_type":"code","source":"import os\n\n# Search for InferenceDataset in both repos\nfor base in [\n    \"/kaggle/input/datasets/zoushuxian/protenix-rmsa-repo\",\n    \"/kaggle/input/datasets/qiweiyin/protenix-v1-adjusted\",\n]:\n    for root, dirs, files in os.walk(base):\n        for f in files:\n            if f.endswith(\".py\"):\n                full = os.path.join(root, f)\n                try:\n                    with open(full) as fp:\n                        if \"InferenceDataset\" in fp.read():\n                            print(full)\n                except:\n                    pass","metadata":{"trusted":true},"outputs":[{"name":"stdout","text":"/kaggle/input/datasets/zoushuxian/protenix-rmsa-repo/protenix_kaggle/protenix/data/infer_data_pipeline.py\n/kaggle/input/datasets/qiweiyin/protenix-v1-adjusted/Protenix-v1-adjust/Protenix-v1/protenix/data/inference/infer_dataloader.py\n/kaggle/input/datasets/qiweiyin/protenix-v1-adjusted/Protenix-v1-adjust-v2/Protenix-v1-adjust-v2/Protenix-v1/protenix/data/inference/infer_dataloader.py\n","output_type":"stream"}],"execution_count":38},{"cell_type":"code","source":"# ═══════════════════════════════════════════════════════════════════════════════\n# RNA 3D Folding — Merged Final Submission\n#\n# Foundation : teammate's codebase (chunking, ID-sort fix, dual-seed, etc.)\n# Layered in : our improvements (RMSA checkpoint fix, stable hash, adaptive\n#              threshold, consensus rerank, N_SAMPLE=20)\n#\n# FULL CHANGE LOG vs. winner's 0.408 baseline:\n#\n#   FROM TEAMMATE:\n#     T1  process_labels ID-suffix sort fix — resid resets for multi-copy\n#         targets (9MME, 9ZCC, 9JGM…); sort by ID integer suffix instead.\n#     T2  Chunking + Kabsch stitching for sequences > MAX_SEQ_LEN.\n#         Long targets no longer get TBM-only fallback.\n#     T3  Dual-seed Protenix for hard targets (0 TBM preds): seed=101 + 202,\n#         combined half-and-half for maximum structural diversity.\n#     T4  apply_double_hinge — two sequential hinge rotations for segs ≥100 nt.\n#     T5  build_exact_match_index — O(1) exact-sequence fast-path in TBM.\n#     T6  _coords_are_valid — filters collapsed/zero/degenerate templates.\n#     T7  usecols on label CSV loading — saves memory on large files.\n#     T8  No adaptive_rna_constraints on Protenix output — neural-net geometry\n#         is already physically valid; constraints degrade it.\n#\n#   FROM OUR IMPROVED VERSION:\n#     O1  RMSA codebase used as Python import source (CODE_DIR), so the\n#         RNA3DB-finetuned checkpoint (1599_ema_0.999.pt) actually loads.\n#         Previously both versions used Protenix-v1-adjusted which has a\n#         different input layer shape (385 vs 389 features) → silent fallback.\n#     O2  stable_hash(tid) replaces row.name-based seeding — reproducible\n#         even if CSV row order changes across reruns.\n#     O3  Adaptive identity threshold per target (50%→45%→40%→35%→30% floor)\n#         so novel sequences get real templates instead of falling straight\n#         through to Protenix.\n#     O4  Consensus reranking of the combined TBM+Protenix pool — predictions\n#         closest to the ensemble centroid fill the top slots.\n#     O5  N_SAMPLE = 20 (winner uses 5, teammate uses 5) — benefits the\n#         ensemble metric which evaluates the full sample set.\n#     O6  N_cycle = 4 explicitly set (prevents accidental 10 which collapses\n#         diffusion diversity).\n#     O7  safe_stack — shape-safe assembly, never crashes on mismatched coords.\n#     O8  warnings silenced at import time for clean logs.\n#\n# KAGGLE DATASETS TO ATTACH:\n#   1. zoushuxian/protenix-rmsa-repo             ← RMSA codebase (O1)\n#   2. zoushuxian/protenix-finetuned-rna3db-all-1599  ← finetuned weights\n#   3. qiweiyin/protenix-v1-adjusted             ← CCD assets + base ckpt\n# ═══════════════════════════════════════════════════════════════════════════════\n\n# ── Install dependencies ──────────────────────────────────────────────────────\n# Run this cell first on Kaggle before importing anything else.\n# Option A — offline wheel (faster, no internet needed):\nimport subprocess, sys\n_WHEEL = (\n    \"/kaggle/input/notebooks/jonathanarya/dependency-installation-script\"\n    \"/install_requirements.sh\"\n)\nimport os as _os\nif _os.path.exists(_WHEEL):\n    subprocess.run([\"bash\", _WHEEL], check=False)\nelse:\n    # Option B — pip install from internet (requires internet toggle ON in settings)\n    subprocess.run(\n        [sys.executable, \"-m\", \"pip\", \"install\", \"biopython\", \"-q\"],\n        check=False\n    )\n\n# ── stdlib / env ──────────────────────────────────────────────────────────────\nimport gc\nimport hashlib\nimport json\nimport os\nimport sys\nimport time\nimport warnings\nfrom pathlib import Path\n\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\nos.environ[\"PYTHONWARNINGS\"] = \"ignore\"\nos.environ[\"LAYERNORM_TYPE\"] = \"torch\"\nos.environ.setdefault(\"RNA_MSA_DEPTH_LIMIT\", \"512\")\n\n# ── Biotite compatibility patch (must run before any protenix import) ─────────\n# RMSA parser.py references pdbx_convert.PDBX_COVALENT_TYPES at module level.\n# Newer biotite removed this attribute — patch it in early so any subsequent\n# import of protenix.data.parser sees the attribute already present.\ntry:\n    import biotite.structure.io.pdbx as _pdbx_early\n    if not hasattr(_pdbx_early, \"PDBX_COVALENT_TYPES\"):\n        _pdbx_early.PDBX_COVALENT_TYPES = list(getattr(\n            _pdbx_early, \"COVALENT_TYPES\",\n            [\"covale\", \"disulf\", \"modres\", \"saltbr\"]\n        ))\nexcept Exception:\n    pass\n\n# ── third-party ───────────────────────────────────────────────────────────────\nimport numpy as np\nimport pandas as pd\nimport torch\nfrom Bio.Align import PairwiseAligner\nfrom tqdm import tqdm\n\n# ══════════════════════════════════════════════════════════════════════════════\n# CONSTANTS & PATHS\n# ══════════════════════════════════════════════════════════════════════════════\nIS_KAGGLE      = bool(os.environ.get(\"KAGGLE_IS_COMPETITION_RERUN\", \"\"))\nLOCAL_N_SAMPLES = 2\n\nDATA_BASE          = \"/kaggle/input/competitions/stanford-rna-3d-folding-2\"\nDEFAULT_TEST_CSV   = f\"{DATA_BASE}/test_sequences.csv\"\nDEFAULT_TRAIN_CSV  = f\"{DATA_BASE}/train_sequences.csv\"\nDEFAULT_TRAIN_LBLS = f\"{DATA_BASE}/train_labels.csv\"\nDEFAULT_VAL_CSV    = f\"{DATA_BASE}/validation_sequences.csv\"\nDEFAULT_VAL_LBLS   = f\"{DATA_BASE}/validation_labels.csv\"\nDEFAULT_OUTPUT     = \"/kaggle/working/submission.csv\"\n\n# ── O1: RMSA codebase as Python import source ─────────────────────────────────\n# The finetuned checkpoint (1599_ema_0.999.pt) was trained with the RMSA\n# variant of Protenix which has 4 extra RnaLM input features (389 vs 385).\n# Using Protenix-v1-adjusted as CODE_DIR causes a silent load failure.\n# Solution: use RMSA repo for imports, v1-adjusted only for CCD assets.\n_RMSA_BASE = \"/kaggle/input/datasets/zoushuxian/protenix-rmsa-repo\"\n_RMSA_CANDIDATES = [\n    f\"{_RMSA_BASE}/protenix_kaggle\",\n    f\"{_RMSA_BASE}\",\n]\nRMSA_CODE_DIR = next(\n    (p for p in _RMSA_CANDIDATES\n     if os.path.isdir(p) and os.path.isdir(os.path.join(p, \"protenix\"))),\n    _RMSA_CANDIDATES[0]\n)\nPROTENIX_ASSETS_DIR = (\n    \"/kaggle/input/datasets/qiweiyin/protenix-v1-adjusted\"\n    \"/Protenix-v1-adjust-v2/Protenix-v1-adjust-v2/Protenix-v1\"\n)\nDEFAULT_CODE_DIR = RMSA_CODE_DIR if os.path.isdir(RMSA_CODE_DIR) else PROTENIX_ASSETS_DIR\nDEFAULT_ROOT_DIR = PROTENIX_ASSETS_DIR  # always v1-adjusted for CCD assets\n\nFINETUNED_CKPT_PATH = (\n    \"/kaggle/input/datasets/zoushuxian\"\n    \"/protenix-finetuned-rna3db-all-1599/1599_ema_0.999.pt\"\n)\nUSE_FINETUNED_CKPT = os.path.exists(FINETUNED_CKPT_PATH)\n\nBASE_MODEL_NAME = \"protenix_base_20250630_v1.0.0\"\n\n# ── O5: N_SAMPLE = 20 ─────────────────────────────────────────────────────────\nN_SAMPLE      = 20\nSEED          = 42\nMAX_SEQ_LEN   = int(os.environ.get(\"MAX_SEQ_LEN\",    \"512\"))\nCHUNK_OVERLAP = int(os.environ.get(\"CHUNK_OVERLAP\",   \"192\"))\n\n# ── O6: N_cycle = 4 (not 10 — higher recycling collapses diffusion diversity) ─\nDIFFUSION_N_STEP  = int(os.environ.get(\"DIFFUSION_N_STEP\",  \"200\"))\nDIFFUSION_N_CYCLE = int(os.environ.get(\"DIFFUSION_N_CYCLE\", \"4\"))\n\nMIN_SIMILARITY            = float(os.environ.get(\"MIN_SIMILARITY\",       \"0.0\"))\n# ── O3: adaptive threshold base (will relax per-target below) ─────────────────\nMIN_PERCENT_IDENTITY_BASE = float(os.environ.get(\"MIN_PERCENT_IDENTITY\", \"50.0\"))\nMIN_TEMPLATES_TARGET      = 5      # relax until this many templates pass\nTHRESHOLD_FLOOR           = 30.0   # never go below this\nLONG_SEQ_THRESHOLD        = 25.0   # floor for sequences > MAX_SEQ_LEN\n\nUSE_PROTENIX = True\n\n\ndef parse_bool(v, default=False):\n    s = str(v).strip().lower()\n    if s in {\"1\",\"true\",\"t\",\"yes\",\"y\",\"on\"}:  return \"true\"\n    if s in {\"0\",\"false\",\"f\",\"no\",\"n\",\"off\"}: return \"false\"\n    return \"true\" if default else \"false\"\n\n\nUSE_MSA        = parse_bool(os.environ.get(\"USE_MSA\",      \"false\"))\nUSE_TEMPLATE   = parse_bool(os.environ.get(\"USE_TEMPLATE\", \"false\"))\nUSE_RNA_MSA    = parse_bool(os.environ.get(\"USE_RNA_MSA\",  \"true\"))\nMODEL_N_SAMPLE = int(os.environ.get(\"MODEL_N_SAMPLE\", str(N_SAMPLE)))\n\n\n# ══════════════════════════════════════════════════════════════════════════════\n# O2: STABLE HASH  (Python hash() is randomized per-process)\n# ══════════════════════════════════════════════════════════════════════════════\ndef stable_hash(s: str) -> int:\n    return int(hashlib.md5(s.encode()).hexdigest(), 16) % (2 ** 32)\n\n\n# ══════════════════════════════════════════════════════════════════════════════\n# GENERAL UTILITIES\n# ══════════════════════════════════════════════════════════════════════════════\ndef seed_everything(seed):\n    os.environ[\"PYTHONHASHSEED\"]          = str(seed)\n    os.environ[\"CUBLAS_WORKSPACE_CONFIG\"] = \":4096:8\"\n    torch.manual_seed(seed)\n    if torch.cuda.is_available():\n        torch.cuda.manual_seed(seed)\n        torch.cuda.manual_seed_all(seed)\n    np.random.seed(seed)\n    torch.backends.cudnn.benchmark     = False\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.enabled       = True\n    try:\n        torch.use_deterministic_algorithms(True)\n    except Exception:\n        pass\n\n\ndef resolve_paths():\n    return (\n        os.environ.get(\"TEST_CSV\",          DEFAULT_TEST_CSV),\n        os.environ.get(\"SUBMISSION_CSV\",    DEFAULT_OUTPUT),\n        os.environ.get(\"PROTENIX_CODE_DIR\", DEFAULT_CODE_DIR),\n        PROTENIX_ASSETS_DIR,  # root_dir always points to v1-adjusted for assets\n    )\n\n\ndef ensure_required_files(root_dir):\n    for p, name in [\n        (Path(root_dir) / \"checkpoint\" / f\"{BASE_MODEL_NAME}.pt\", \"base checkpoint\"),\n        (Path(root_dir) / \"common\" / \"components.cif\",            \"CCD file\"),\n        (Path(root_dir) / \"common\" / \"components.cif.rdkit_mol.pkl\", \"CCD cache\"),\n    ]:\n        if not p.exists():\n            raise FileNotFoundError(f\"Missing {name}: {p}\")\n\n\ndef build_input_json(df, json_path):\n    data = [\n        {\"name\": row[\"target_id\"], \"covalent_bonds\": [],\n         \"sequences\": [{\"rnaSequence\": {\"sequence\": row[\"sequence\"], \"count\": 1}}]}\n        for _, row in df.iterrows()\n    ]\n    with open(json_path, \"w\", encoding=\"utf-8\") as f:\n        json.dump(data, f)\n\n\ndef build_configs(input_json_path, dump_dir, model_name, n_sample=None, seed=None):\n    from configs.configs_base import configs as configs_base\n    from configs.configs_data import data_configs\n    from configs.configs_inference import inference_configs\n    from configs.configs_model_type import model_configs\n    from protenix.config.config import parse_configs\n\n    ns   = n_sample if n_sample is not None else MODEL_N_SAMPLE\n    use_seed = seed if seed is not None else SEED\n    base = {**configs_base, **{\"data\": data_configs}, **inference_configs}\n\n    def deep_update(t, p):\n        for k, v in p.items():\n            if isinstance(v, dict) and k in t and isinstance(t[k], dict):\n                deep_update(t[k], v)\n            else:\n                t[k] = v\n\n    deep_update(base, model_configs[model_name])\n    arg_str = \" \".join([\n        f\"--model_name {model_name}\",\n        f\"--input_json_path {input_json_path}\",\n        f\"--dump_dir {dump_dir}\",\n        f\"--use_msa {USE_MSA}\",\n        f\"--use_template {USE_TEMPLATE}\",\n        f\"--use_rna_msa {USE_RNA_MSA}\",\n        f\"--sample_diffusion.N_sample {ns}\",\n        f\"--seeds {use_seed}\",\n    ])\n    return parse_configs(configs=base, arg_str=arg_str, fill_required_with_null=True)\n\n\n# ── O1: load finetuned checkpoint via RMSA-compatible runner ──────────────────\ndef load_runner_with_checkpoint(configs):\n    from runner.inference import InferenceRunner\n\n    runner = InferenceRunner(configs)\n\n    if USE_FINETUNED_CKPT:\n        print(f\"  Loading finetuned RNA checkpoint: {FINETUNED_CKPT_PATH}\")\n        try:\n            ckpt = torch.load(FINETUNED_CKPT_PATH, map_location=\"cpu\")\n            if isinstance(ckpt, dict):\n                state_dict = (ckpt.get(\"ema_model\") or ckpt.get(\"model\")\n                              or ckpt.get(\"state_dict\") or ckpt)\n            else:\n                state_dict = ckpt\n            cleaned = {(k[7:] if k.startswith(\"module.\") else k): v\n                       for k, v in state_dict.items()}\n            missing, unexpected = runner.model.load_state_dict(cleaned, strict=False)\n            print(f\"  ✓ Finetuned checkpoint loaded \"\n                  f\"(missing={len(missing)}, unexpected={len(unexpected)})\")\n        except Exception as e:\n            print(f\"  ✗ Finetuned checkpoint failed ({e}) — using base model\")\n    else:\n        print(\"  Finetuned checkpoint not found — using base model.\")\n\n    return runner\n\n\ndef coords_to_rows(target_id, seq, coords):\n    \"\"\"coords: (N_SAMPLE, seq_len, 3)\"\"\"\n    rows = []\n    for i in range(len(seq)):\n        row = {\"ID\": f\"{target_id}_{i+1}\", \"resname\": seq[i], \"resid\": i+1}\n        for s in range(N_SAMPLE):\n            if s < coords.shape[0] and i < coords.shape[1]:\n                x, y, z = coords[s, i]\n            else:\n                x, y, z = 0.0, 0.0, 0.0\n            row[f\"x_{s+1}\"] = float(x)\n            row[f\"y_{s+1}\"] = float(y)\n            row[f\"z_{s+1}\"] = float(z)\n        rows.append(row)\n    return rows\n\n\n# ── O7: shape-safe coordinate assembly ───────────────────────────────────────\ndef safe_stack(preds: list, seq_len: int, n_sample: int) -> np.ndarray:\n    if not preds:\n        return np.zeros((n_sample, seq_len, 3), dtype=np.float32)\n    safe = []\n    for p in preds[:n_sample]:\n        p = np.asarray(p, dtype=np.float32)\n        if p.shape == (seq_len, 3):\n            safe.append(p)\n        elif p.shape[0] < seq_len:\n            padded = np.zeros((seq_len, 3), dtype=np.float32)\n            padded[:p.shape[0]] = p\n            safe.append(padded)\n        else:\n            safe.append(p[:seq_len])\n    if not safe:\n        return np.zeros((n_sample, seq_len, 3), dtype=np.float32)\n    return np.stack(safe, axis=0)\n\n\ndef pad_samples(coords: np.ndarray, n: int) -> np.ndarray:\n    if coords.shape[0] >= n:\n        return coords[:n]\n    if coords.shape[0] == 0:\n        return np.zeros((n, coords.shape[1], 3), dtype=coords.dtype)\n    extra = np.repeat(coords[:1], n - coords.shape[0], axis=0)\n    return np.concatenate([coords, extra], axis=0)\n\n\n# ── T2: chunking / Kabsch stitching ──────────────────────────────────────────\ndef split_into_chunks(seq_len, max_len, overlap):\n    if seq_len <= max_len:\n        return [(0, seq_len)]\n    chunks, step, pos = [], max_len - overlap, 0\n    while pos < seq_len:\n        end = min(pos + max_len, seq_len)\n        chunks.append((pos, end))\n        if end == seq_len:\n            break\n        pos += step\n    return chunks\n\n\ndef kabsch_align(P, Q):\n    cP, cQ = P.mean(0), Q.mean(0)\n    Pc, Qc = P - cP, Q - cQ\n    H      = Pc.T @ Qc\n    U, _, Vt = np.linalg.svd(H)\n    d = np.linalg.det(Vt.T @ U.T)\n    S = np.eye(3)\n    if d < 0:\n        S[2, 2] = -1\n    R = Vt.T @ S @ U.T\n    return R, cQ - R @ cP\n\n\ndef stitch_chunk_coords(chunk_coords_list, chunk_ranges, seq_len):\n    if len(chunk_coords_list) == 1:\n        c   = chunk_coords_list[0]\n        out = np.zeros((seq_len, 3), dtype=c.dtype)\n        out[:min(c.shape[0], seq_len)] = c[:min(c.shape[0], seq_len)]\n        return out\n\n    aligned = [chunk_coords_list[0].copy()]\n    for i in range(1, len(chunk_coords_list)):\n        ps, pe = chunk_ranges[i-1]\n        cs, ce = chunk_ranges[i]\n        ov_s, ov_e = cs, min(pe, ce)\n        if ov_e - ov_s < 3:\n            aligned.append(chunk_coords_list[i].copy())\n            continue\n        prev_ov = aligned[i-1][ov_s-ps:ov_e-ps]\n        cur_ov  = chunk_coords_list[i][ov_s-cs:ov_e-cs]\n        valid   = ~(np.isnan(prev_ov).any(1) | np.isnan(cur_ov).any(1))\n        if valid.sum() < 3:\n            aligned.append(chunk_coords_list[i].copy())\n            continue\n        R, t = kabsch_align(cur_ov[valid], prev_ov[valid])\n        aligned.append((chunk_coords_list[i] @ R.T) + t)\n\n    full    = np.zeros((seq_len, 3), dtype=np.float64)\n    weights = np.zeros(seq_len,     dtype=np.float64)\n    for i, ((s, e), coords) in enumerate(zip(chunk_ranges, aligned)):\n        cl = coords.shape[0]\n        ae = min(s + cl, seq_len)\n        ul = ae - s\n        w  = np.ones(ul, dtype=np.float64)\n        if i > 0:\n            ov_e2 = min(chunk_ranges[i-1][1], e)\n            rl    = ov_e2 - s\n            if rl > 0:\n                w[:rl] = np.linspace(0., 1., rl)\n        if i < len(chunk_ranges) - 1:\n            ns2 = chunk_ranges[i+1][0]\n            rs  = ns2 - s\n            rl  = ae - ns2\n            if rl > 0 and rs < ul:\n                w[rs:ul] = np.linspace(1., 0., rl)\n        full[s:ae]    += coords[:ul] * w[:, None]\n        weights[s:ae] += w\n    mask = weights > 0\n    full[mask] /= weights[mask, None]\n    return full\n\n\n# ══════════════════════════════════════════════════════════════════════════════\n# O4: CONSENSUS RERANKING\n# ══════════════════════════════════════════════════════════════════════════════\ndef consensus_rerank(preds: list, seq_len: int) -> list:\n    \"\"\"\n    Rerank predictions by proximity to ensemble centroid.\n    Most consensus-supported prediction goes first.\n    Returns unchanged if fewer than 3 members.\n    \"\"\"\n    if len(preds) < 3:\n        return preds\n    try:\n        arr = np.stack([\n            np.asarray(p, dtype=np.float32)[:seq_len]\n            if len(p) >= seq_len\n            else np.pad(np.asarray(p, dtype=np.float32),\n                        ((0, seq_len - len(p)), (0, 0)))\n            for p in preds\n        ])\n        centroid = arr.mean(axis=0)\n        rmsds    = np.sqrt(np.mean(np.sum((arr - centroid[None]) ** 2, axis=-1), axis=-1))\n        order    = np.argsort(rmsds)\n        return [preds[i] for i in order]\n    except Exception:\n        return preds\n\n\n# ══════════════════════════════════════════════════════════════════════════════\n# TBM CORE\n# ══════════════════════════════════════════════════════════════════════════════\ndef _make_aligner():\n    al = PairwiseAligner()\n    al.mode             = \"global\"\n    al.match_score      = 2\n    al.mismatch_score   = -1.5\n    al.open_gap_score   = -8\n    al.extend_gap_score = -0.4\n    for attr in [\n        \"query_left_open_gap_score\",   \"query_left_extend_gap_score\",\n        \"query_right_open_gap_score\",  \"query_right_extend_gap_score\",\n        \"target_left_open_gap_score\",  \"target_left_extend_gap_score\",\n        \"target_right_open_gap_score\", \"target_right_extend_gap_score\",\n    ]:\n        try:\n            setattr(al, attr, -8 if \"open\" in attr else -0.4)\n        except Exception:\n            pass\n    return al\n\n_aligner = _make_aligner()\n\n\ndef parse_stoichiometry(stoich):\n    if pd.isna(stoich) or str(stoich).strip() == \"\":\n        return []\n    return [(ch.strip(), int(cnt))\n            for part in str(stoich).split(\";\")\n            for ch, cnt in [part.split(\":\")]]\n\n\ndef parse_fasta(fasta_content):\n    out, cur, parts = {}, None, []\n    for line in str(fasta_content).splitlines():\n        line = line.strip()\n        if not line:\n            continue\n        if line.startswith(\">\"):\n            if cur is not None:\n                out[cur] = \"\".join(parts)\n            cur, parts = line[1:].split()[0], []\n        else:\n            parts.append(line.replace(\" \", \"\"))\n    if cur is not None:\n        out[cur] = \"\".join(parts)\n    return out\n\n\ndef get_chain_segments(row):\n    seq    = row[\"sequence\"]\n    stoich = row.get(\"stoichiometry\", \"\")\n    all_sq = row.get(\"all_sequences\", \"\")\n    if (pd.isna(stoich) or pd.isna(all_sq)\n            or str(stoich).strip() == \"\" or str(all_sq).strip() == \"\"):\n        return [(0, len(seq))]\n    try:\n        cd    = parse_fasta(all_sq)\n        order = parse_stoichiometry(stoich)\n        segs, pos = [], 0\n        for ch, cnt in order:\n            base = cd.get(ch)\n            if base is None:\n                return [(0, len(seq))]\n            for _ in range(cnt):\n                segs.append((pos, pos + len(base)))\n                pos += len(base)\n        return segs if pos == len(seq) else [(0, len(seq))]\n    except Exception:\n        return [(0, len(seq))]\n\n\ndef build_segments_map(df):\n    seg_map, stoich_map = {}, {}\n    for _, r in df.iterrows():\n        tid           = r[\"target_id\"]\n        seg_map[tid]  = get_chain_segments(r)\n        raw_s         = r.get(\"stoichiometry\", \"\")\n        stoich_map[tid] = \"\" if pd.isna(raw_s) else str(raw_s)\n    return seg_map, stoich_map\n\n\ndef process_labels(labels_df):\n    \"\"\"\n    T1: ID-suffix sort fix.\n\n    For multi-copy targets (9MME=8×580nt, 9ZCC, 9JGM, etc.), resid RESETS\n    to 1 at the start of each copy. sort_values(\"resid\") scrambles coordinates.\n    The ID suffix (e.g. \"9MME_581\" → 581) is always sequential — use that.\n    \"\"\"\n    coords = {}\n    split  = labels_df[\"ID\"].str.rsplit(\"_\", n=1)\n    pfx    = split.str[0]\n    pos    = pd.to_numeric(split.str[1], errors=\"coerce\").fillna(0).astype(int)\n    df     = pd.DataFrame({\n        \"_pfx\": pfx,\n        \"_pos\": pos,\n        \"x_1\":  labels_df[\"x_1\"].values,\n        \"y_1\":  labels_df[\"y_1\"].values,\n        \"z_1\":  labels_df[\"z_1\"].values,\n    })\n    for prefix, grp in df.groupby(\"_pfx\"):\n        arr = grp.sort_values(\"_pos\")[[\"x_1\", \"y_1\", \"z_1\"]].values.astype(np.float64)\n        arr[np.abs(arr) > 1e10] = np.nan\n        for i in range(len(arr)):\n            if not np.isnan(arr[i, 0]):\n                continue\n            pv = next((j for j in range(i-1, -1, -1) if not np.isnan(arr[j, 0])), -1)\n            nv = next((j for j in range(i+1, len(arr))  if not np.isnan(arr[j, 0])), -1)\n            if   pv >= 0 and nv >= 0: w = (i-pv)/(nv-pv); arr[i] = (1-w)*arr[pv] + w*arr[nv]\n            elif pv >= 0:             arr[i] = arr[pv] + [3, 0, 0]\n            elif nv >= 0:             arr[i] = arr[nv] + [3, 0, 0]\n            else:                     arr[i] = [i*3, 0, 0]\n        coords[prefix] = np.nan_to_num(arr, nan=0.0)\n    return coords\n\n\n# ── T6: coordinate validity filter ───────────────────────────────────────────\ndef _coords_are_valid(c):\n    if c is None or not np.all(np.isfinite(c)):\n        return False\n    spread = np.max(c, axis=0) - np.min(c, axis=0)\n    if np.any(spread < 2.0):\n        return False\n    rms = np.sqrt(np.mean(c ** 2))\n    if rms < 1.0:\n        return False\n    return True\n\n\n# ── T5: exact-match O(1) index ────────────────────────────────────────────────\ndef build_exact_match_index(train_df, train_coords):\n    idx = {}\n    for _, row in train_df.iterrows():\n        tid = row[\"target_id\"]\n        if tid not in train_coords:\n            continue\n        c = train_coords[tid]\n        if not _coords_are_valid(c):\n            continue\n        idx.setdefault(row[\"sequence\"], []).append((tid, c))\n    n_ent = sum(len(v) for v in idx.values())\n    print(f\"  Exact-match index: {len(idx)} unique seqs, {n_ent} total entries\")\n    return idx\n\n\ndef _build_aligned_strings(query_seq, template_seq, alignment):\n    q_segs, t_segs = alignment.aligned\n    aq, at, qi, ti = [], [], 0, 0\n    for (qs, qe), (ts, te) in zip(q_segs, t_segs):\n        while qi < qs: aq.append(query_seq[qi]);    at.append(\"-\");              qi += 1\n        while ti < ts: aq.append(\"-\");              at.append(template_seq[ti]); ti += 1\n        for qp, tp in zip(range(qs, qe), range(ts, te)):\n            aq.append(query_seq[qp]); at.append(template_seq[tp])\n        qi, ti = qe, te\n    while qi < len(query_seq):    aq.append(query_seq[qi]);    at.append(\"-\");              qi += 1\n    while ti < len(template_seq): aq.append(\"-\");              at.append(template_seq[ti]); ti += 1\n    return \"\".join(aq), \"\".join(at)\n\n\ndef find_similar_sequences_detailed(query_seq, train_seqs_df, train_coords_dict,\n                                    top_n=30, exact_idx=None):\n    \"\"\"T5 + sequence similarity search with O(1) exact-match fast-path.\"\"\"\n    q_len = len(query_seq)\n\n    exact_results = []\n    if exact_idx is not None and query_seq in exact_idx:\n        for tid, coords in exact_idx[query_seq]:\n            exact_results.append((tid, query_seq, 1.0, coords, 100.0))\n        if len(exact_results) >= top_n:\n            return exact_results[:top_n]\n    seen_exact = {r[0] for r in exact_results}\n\n    results = []\n    for _, row in train_seqs_df.iterrows():\n        tid, tseq = row[\"target_id\"], row[\"sequence\"]\n        if tid not in train_coords_dict or tid in seen_exact:\n            continue\n        if abs(len(tseq) - q_len) / max(len(tseq), q_len) > 0.3:\n            continue\n        aln    = next(iter(_aligner.align(query_seq, tseq)))\n        norm_s = aln.score / (2 * min(q_len, len(tseq)))\n        identical = sum(\n            1 for (qs, qe), (ts, te) in zip(*aln.aligned)\n            for qp, tp in zip(range(qs, qe), range(ts, te))\n            if query_seq[qp] == tseq[tp]\n        )\n        pct_id = 100 * identical / q_len\n        results.append((tid, tseq, norm_s, train_coords_dict[tid], pct_id))\n    results.sort(key=lambda x: x[2], reverse=True)\n\n    combined = exact_results + results\n    seen, final = set(), []\n    for item in combined:\n        if item[0] not in seen:\n            seen.add(item[0])\n            final.append(item)\n    return final[:top_n]\n\n\n# ── O3: per-target adaptive identity threshold ────────────────────────────────\ndef get_adaptive_threshold(similar_templates: list, is_long: bool = False) -> float:\n    floor = LONG_SEQ_THRESHOLD if is_long else THRESHOLD_FLOOR\n    for threshold in [MIN_PERCENT_IDENTITY_BASE, 45.0, 40.0, 35.0, floor]:\n        passing = [t for t in similar_templates if t[4] >= threshold]\n        if len(passing) >= MIN_TEMPLATES_TARGET:\n            return threshold\n        if threshold == MIN_PERCENT_IDENTITY_BASE and len(passing) >= 2:\n            return threshold\n    return floor\n\n\ndef adapt_template_to_query(query_seq, template_seq, template_coords):\n    aln   = next(iter(_aligner.align(query_seq, template_seq)))\n    new_c = np.full((len(query_seq), 3), np.nan)\n    for (qs, qe), (ts, te) in zip(*aln.aligned):\n        chunk = template_coords[ts:te]\n        if len(chunk) == (qe - qs):\n            new_c[qs:qe] = chunk\n    for i in range(len(new_c)):\n        if np.isnan(new_c[i, 0]):\n            pv = next((j for j in range(i-1, -1, -1) if not np.isnan(new_c[j, 0])), -1)\n            nv = next((j for j in range(i+1, len(new_c))  if not np.isnan(new_c[j, 0])), -1)\n            if   pv >= 0 and nv >= 0: w = (i-pv)/(nv-pv); new_c[i] = (1-w)*new_c[pv] + w*new_c[nv]\n            elif pv >= 0:             new_c[i] = new_c[pv] + [3, 0, 0]\n            elif nv >= 0:             new_c[i] = new_c[nv] + [3, 0, 0]\n            else:                     new_c[i] = [i*3, 0, 0]\n    return np.nan_to_num(new_c)\n\n\ndef adaptive_rna_constraints(coords, target_id, segments_map, confidence=1.0, passes=2):\n    \"\"\"\n    T1 (exact match skip): if confidence ≥ 0.99 return unchanged —\n    perfect coordinates don't need geometry correction.\n    \"\"\"\n    if confidence >= 0.99:\n        return coords.copy()\n\n    X        = coords.copy()\n    segments = segments_map.get(target_id, [(0, len(X))])\n    strength = max(0.75 * (1.0 - min(confidence, 0.97)), 0.02)\n    for _ in range(passes):\n        for s, e in segments:\n            C = X[s:e]; L = e - s\n            if L < 3:\n                continue\n            d    = C[1:] - C[:-1]; dist = np.linalg.norm(d, axis=1) + 1e-6\n            adj  = d * ((5.95 - dist) / dist)[:, None] * (0.22 * strength)\n            C[:-1] -= adj; C[1:] += adj\n            d2   = C[2:] - C[:-2]; d2n = np.linalg.norm(d2, axis=1) + 1e-6\n            adj2 = d2 * ((10.2 - d2n) / d2n)[:, None] * (0.10 * strength)\n            C[:-2] -= adj2; C[2:] += adj2\n            C[1:-1] += (0.06 * strength) * (0.5 * (C[:-2] + C[2:]) - C[1:-1])\n            if L >= 25:\n                idx  = (np.linspace(0, L-1, min(L, 160)).astype(int)\n                        if L > 220 else np.arange(L))\n                P    = C[idx]; diff = P[:, None, :] - P[None, :, :]\n                dm   = np.linalg.norm(diff, axis=2) + 1e-6\n                sep  = np.abs(idx[:, None] - idx[None, :])\n                mask = (sep > 2) & (dm < 3.2)\n                if np.any(mask):\n                    vec     = (diff * ((3.2 - dm) / dm)[:, :, None]\n                               * mask[:, :, None]).sum(axis=1)\n                    C[idx] += (0.015 * strength) * vec\n            X[s:e] = C\n    return X\n\n\n# ── Diversity transforms (T4 adds double-hinge) ───────────────────────────────\ndef _rotmat(axis, ang):\n    a = np.asarray(axis, float); a /= np.linalg.norm(a) + 1e-12\n    x, y, z = a; c, s = np.cos(ang), np.sin(ang); CC = 1 - c\n    return np.array([[c+x*x*CC, x*y*CC-z*s, x*z*CC+y*s],\n                     [y*x*CC+z*s, c+y*y*CC, y*z*CC-x*s],\n                     [z*x*CC-y*s, z*y*CC+x*s, c+z*z*CC]])\n\n\ndef apply_hinge(coords, seg, rng, deg=22):\n    s, e = seg; L = e - s\n    if L < 30:\n        return coords\n    pivot = s + int(rng.integers(10, L - 10))\n    R     = _rotmat(rng.normal(size=3), np.deg2rad(float(rng.uniform(-deg, deg))))\n    X     = coords.copy(); p0 = X[pivot].copy()\n    X[pivot+1:e] = (X[pivot+1:e] - p0) @ R.T + p0\n    return X\n\n\ndef apply_double_hinge(coords, seg, rng, deg=15):\n    \"\"\"T4: two sequential hinge rotations for segments ≥ 60 nt.\"\"\"\n    s, e = seg; L = e - s\n    if L < 60:\n        return apply_hinge(coords, seg, rng, deg)\n    p1 = s + int(rng.integers(10, L // 2))\n    p2 = s + int(rng.integers(L // 2, L - 10))\n    R1 = _rotmat(rng.normal(size=3), np.deg2rad(float(rng.uniform(-deg, deg))))\n    R2 = _rotmat(rng.normal(size=3), np.deg2rad(float(rng.uniform(-deg, deg))))\n    X  = coords.copy()\n    p0 = X[p1].copy(); X[p1+1:p2] = (X[p1+1:p2] - p0) @ R1.T + p0\n    p0 = X[p2].copy(); X[p2+1:e]  = (X[p2+1:e]  - p0) @ R2.T + p0\n    return X\n\n\ndef jitter_chains(coords, segs, rng, deg=12, trans=1.5):\n    X   = coords.copy(); gc_ = X.mean(0, keepdims=True)\n    for s, e in segs:\n        R     = _rotmat(rng.normal(size=3), np.deg2rad(float(rng.uniform(-deg, deg))))\n        shift = rng.normal(size=3)\n        shift = shift / (np.linalg.norm(shift) + 1e-12) * float(rng.uniform(0, trans))\n        c     = X[s:e].mean(0, keepdims=True)\n        X[s:e] = (X[s:e] - c) @ R.T + c + shift\n    X -= X.mean(0, keepdims=True) - gc_\n    return X\n\n\ndef smooth_wiggle(coords, segs, rng, amp=0.8):\n    X = coords.copy()\n    for s, e in segs:\n        L = e - s\n        if L < 20:\n            continue\n        ctrl   = np.linspace(0, L-1, 6)\n        disp   = rng.normal(0, amp, (6, 3))\n        t      = np.arange(L)\n        X[s:e] += np.vstack([np.interp(t, ctrl, disp[:, k]) for k in range(3)]).T\n    return X\n\n\ndef generate_rna_structure(sequence, seed=None):\n    if seed is not None:\n        np.random.seed(seed)\n    n = len(sequence); coords = np.zeros((n, 3))\n    for i in range(n):\n        ang = i * 0.6\n        coords[i] = [10.0 * np.cos(ang), 10.0 * np.sin(ang), i * 2.5]\n    return coords\n\n\n# ══════════════════════════════════════════════════════════════════════════════\n# PHASE 1 — TBM\n# ══════════════════════════════════════════════════════════════════════════════\ndef tbm_phase(test_df, train_seqs_df, train_coords_dict, segments_map, exact_idx=None):\n    print(f\"\\n{'='*65}\")\n    print(\"PHASE 1: Template-Based Modeling\")\n    print(f\"  MIN_PCT_IDENTITY (base) = {MIN_PERCENT_IDENTITY_BASE}\")\n    print(f\"  Adaptive floor (normal) = {THRESHOLD_FLOOR}\")\n    print(f\"  Adaptive floor (long)   = {LONG_SEQ_THRESHOLD}\")\n    print(f\"  N_SAMPLE                = {N_SAMPLE}\")\n    print(f\"{'='*65}\")\n    t0 = time.time()\n\n    template_predictions: dict = {}\n    protenix_queue:       dict = {}\n\n    for _, row in test_df.iterrows():\n        tid      = row[\"target_id\"]\n        seq      = row[\"sequence\"]\n        seq_len  = len(seq)\n        segs     = segments_map.get(tid, [(0, seq_len)])\n        is_long  = seq_len > MAX_SEQ_LEN\n\n        similar = find_similar_sequences_detailed(\n            seq, train_seqs_df, train_coords_dict, top_n=30, exact_idx=exact_idx\n        )\n\n        # O3: per-target adaptive threshold\n        threshold = get_adaptive_threshold(similar, is_long=is_long)\n\n        preds, used = [], set()\n\n        for tmpl_id, tmpl_seq, sim, tmpl_coords, pct_id in similar:\n            if len(preds) >= N_SAMPLE:\n                break\n            # Break (not continue) — list sorted by sim; first failure means done\n            if sim < MIN_SIMILARITY or pct_id < threshold:\n                break\n            if tmpl_id in used:\n                continue\n\n            slot    = len(preds)\n            # O2: stable hash for reproducibility\n            rng     = np.random.default_rng(stable_hash(tid) + slot * 10007)\n            adapted = adapt_template_to_query(seq, tmpl_seq, tmpl_coords)\n            longest = max(segs, key=lambda se: se[1] - se[0])\n            longest_len = longest[1] - longest[0]\n\n            if slot == 0:\n                # Best template — apply constraints for geometry cleanup\n                # (skipped automatically for exact matches via confidence≥0.99)\n                X = adaptive_rna_constraints(adapted, tid, segments_map, confidence=sim)\n            elif slot == 1:\n                # Gentle Gaussian jitter — halved for near-perfect templates\n                noise_scale = (max(0.005, (0.40-sim)*0.03) if pct_id > 95\n                               else max(0.01, (0.40-sim)*0.06))\n                X = adapted + rng.normal(0, noise_scale, adapted.shape)\n                X = adaptive_rna_constraints(X, tid, segments_map, confidence=sim)\n            elif slot == 2:\n                # T4: double-hinge for long segs, single-hinge for short\n                X = (apply_double_hinge(adapted, longest, rng)\n                     if longest_len >= 100\n                     else apply_hinge(adapted, longest, rng))\n                X = adaptive_rna_constraints(X, tid, segments_map, confidence=sim)\n            elif slot == 3:\n                # Chain jitter — rigid-body per-chain rotation + translation\n                X = jitter_chains(adapted, segs, rng)\n                X = adaptive_rna_constraints(X, tid, segments_map, confidence=sim)\n            else:\n                # Smooth wavelike displacement\n                X = smooth_wiggle(adapted, segs, rng)\n                X = adaptive_rna_constraints(X, tid, segments_map, confidence=sim)\n\n            preds.append(X)\n            used.add(tmpl_id)\n\n        template_predictions[tid] = preds\n        n_needed = N_SAMPLE - len(preds)\n\n        if n_needed > 0:\n            protenix_queue[tid] = (n_needed, seq)\n            print(f\"  {tid} ({seq_len} nt): {len(preds)} TBM \"\n                  f\"(thr={threshold:.0f}%) → need {n_needed} Protenix\")\n        else:\n            print(f\"  {tid} ({seq_len} nt): all {N_SAMPLE} TBM ✓ \"\n                  f\"(thr={threshold:.0f}%)\")\n\n    elapsed = time.time() - t0\n    print(f\"\\nPhase 1 done in {elapsed:.1f}s | \"\n          f\"TBM-only:{len(test_df)-len(protenix_queue)} | \"\n          f\"Protenix:{len(protenix_queue)}\")\n    return template_predictions, protenix_queue\n\n\n# ══════════════════════════════════════════════════════════════════════════════\n# PHASE 2 — PROTENIX  (T2 chunking, T3 dual-seed, O1 finetuned checkpoint)\n# ══════════════════════════════════════════════════════════════════════════════\ndef _get_c1_mask(data, atom_array, chunk_seq_len):\n    if atom_array is not None:\n        try:\n            if hasattr(atom_array, \"centre_atom_mask\"):\n                m = atom_array.centre_atom_mask == 1\n                if hasattr(atom_array, \"is_rna\"):\n                    m = m & atom_array.is_rna\n                return torch.from_numpy(m).bool()\n            if hasattr(atom_array, \"atom_name\"):\n                base = atom_array.atom_name == \"C1'\"\n                if hasattr(atom_array, \"is_rna\"):\n                    base = base & atom_array.is_rna\n                return torch.from_numpy(base).bool()\n        except Exception:\n            pass\n    f = data[\"input_feature_dict\"]\n    if \"centre_atom_mask\" in f:\n        return (f[\"centre_atom_mask\"] == 1).bool()\n    if \"center_atom_mask\" in f:\n        return (f[\"center_atom_mask\"] == 1).bool()\n    m11 = (f[\"atom_to_tokatom_idx\"] == 11).bool()\n    m12 = (f[\"atom_to_tokatom_idx\"] == 12).bool()\n    c11, c12 = m11.sum().item(), m12.sum().item()\n    return m11 if abs(c11 - chunk_seq_len) < abs(c12 - chunk_seq_len) else m12\n\n\ndef _extract_c1_coords(data, atom_array, chunk_seq_len, raw_coords):\n    mask   = _get_c1_mask(data, atom_array, chunk_seq_len).to(raw_coords.device)\n    coords = raw_coords[:, mask, :].detach().cpu().numpy()\n    if coords.shape[1] > 1:\n        diffs = np.linalg.norm(coords[0, 1:] - coords[0, :-1], axis=-1)\n        if np.all(diffs < 1e-4):\n            print(\"    WARNING: collapsed coordinates detected\")\n            return None\n    if coords.shape[1] != chunk_seq_len:\n        if coords.shape[1] == 1 and chunk_seq_len > 1:\n            return None\n        padded = np.zeros((coords.shape[0], chunk_seq_len, 3), dtype=np.float32)\n        ml     = min(coords.shape[1], chunk_seq_len)\n        padded[:, :ml, :] = coords[:, :ml, :]\n        coords = padded\n    return coords\n\n\ndef run_protenix(protenix_queue, work_dir, code_dir, root_dir, seed=None):\n    \"\"\"\n    Phase 2 — Protenix inference.\n    T2: long sequences are split into overlapping chunks, Kabsch-aligned, blended.\n    O1: uses RMSA-compatible runner so finetuned checkpoint loads correctly.\n    \"\"\"\n    use_seed = seed if seed is not None else SEED\n    print(f\"\\n{'='*65}\")\n    print(f\"PHASE 2: Protenix (chunked, seed={use_seed})\")\n    print(f\"  N_step={DIFFUSION_N_STEP}, N_cycle={DIFFUSION_N_CYCLE}\")\n    print(f\"  Checkpoint: {'finetuned RNA ✓' if USE_FINETUNED_CKPT else 'base model'}\")\n    print(f\"{'='*65}\")\n\n    for i in range(torch.cuda.device_count()):\n        try:\n            free, total = torch.cuda.mem_get_info(i)\n            print(f\"  GPU {i}: {torch.cuda.get_device_name(i)} \"\n                  f\"({free/1024**3:.1f}/{total/1024**3:.1f} GB free)\")\n        except Exception:\n            print(f\"  GPU {i}: {torch.cuda.get_device_name(i)}\")\n\n    # ── 1. Build task list ────────────────────────────────────────────────────\n    tasks, chunk_info = [], {}\n    for target_id, (n_needed, full_seq) in protenix_queue.items():\n        seq_len = len(full_seq)\n        if seq_len <= MAX_SEQ_LEN:\n            tasks.append({\"target_id\": target_id, \"sequence\": full_seq})\n            chunk_info[target_id] = [{\"name\": target_id, \"range\": (0, seq_len)}]\n            print(f\"  {target_id} ({seq_len} nt): single pass\")\n        else:\n            chunks = split_into_chunks(seq_len, MAX_SEQ_LEN, CHUNK_OVERLAP)\n            print(f\"  {target_id} ({seq_len} nt): {len(chunks)} chunks\")\n            chunk_info[target_id] = []\n            for ci, (cs, ce) in enumerate(chunks):\n                cn = f\"{target_id}_chunk{ci}\"\n                tasks.append({\"target_id\": cn, \"sequence\": full_seq[cs:ce]})\n                chunk_info[target_id].append({\"name\": cn, \"range\": (cs, ce)})\n\n    tasks_df        = pd.DataFrame(tasks)\n    input_json_path = str(work_dir / \"protenix_input.json\")\n    build_input_json(tasks_df, input_json_path)\n\n    # ── 2. Initialize runner ──────────────────────────────────────────────────\n    # The RMSA parser.py does:\n    #   import biotite.structure.io.pdbx as pdbx_convert\n    #   if \"metalc\" in pdbx_convert.PDBX_COVALENT_TYPES: ...\n    # Newer biotite removed PDBX_COVALENT_TYPES. We must patch the biotite\n    # module object BEFORE parser.py is first imported (module-level code runs\n    # on import, not on use). We also clear any cached failed imports from\n    # sys.modules so the patch takes effect.\n    import sys as _sys\n\n    # Patch parser.py directly on disk — biotite 1.6.0 removed PDBX_COVALENT_TYPES\n    # parser.py has its own `import biotite... as pdbx_convert` so patching the\n    # module object doesn't reach it. We must fix the source file itself.\n    _parser_path = (\n        \"/kaggle/input/datasets/zoushuxian/protenix-rmsa-repo\"\n        \"/protenix_kaggle/protenix/data/parser.py\"\n    )\n    import shutil, pathlib\n    _parser_patched = pathlib.Path(\"/tmp/parser_patched.py\")\n    if not _parser_patched.exists():\n        _src = pathlib.Path(_parser_path).read_text()\n        _src = _src.replace(\n            'if \"metalc\" in pdbx_convert.PDBX_COVALENT_TYPES:  # for reload\\n'\n            '    pdbx_convert.PDBX_COVALENT_TYPES.remove(\"metalc\")',\n            '# patched: PDBX_COVALENT_TYPES removed in biotite 1.6.0\\n'\n            'if not hasattr(pdbx_convert, \"PDBX_COVALENT_TYPES\"):\\n'\n            '    pdbx_convert.PDBX_COVALENT_TYPES = [\"covale\", \"disulf\", \"modres\", \"saltbr\"]\\n'\n            'if \"metalc\" in pdbx_convert.PDBX_COVALENT_TYPES:\\n'\n            '    pdbx_convert.PDBX_COVALENT_TYPES.remove(\"metalc\")'\n        )\n        _parser_patched.write_text(_src)\n\n    # Redirect the import to our patched copy\n    import importlib.util as _ilu\n    _spec = _ilu.spec_from_file_location(\"protenix.data.parser\", str(_parser_patched))\n    _mod  = _ilu.module_from_spec(_spec)\n    _sys.modules[\"protenix.data.parser\"] = _mod\n    _spec.loader.exec_module(_mod)\n    print(\"  parser.py patched and loaded ✓\")\n\n    # Wipe other stale protenix.data imports (but NOT parser which we just set)\n    for _m in list(_sys.modules.keys()):\n        if \"protenix.data\" in _m and _m != \"protenix.data.parser\":\n            del _sys.modules[_m]\n\n    try:\n        from protenix.data.infer_data_pipeline import InferenceDataset\n        print(\"  InferenceDataset: RMSA path ✓\")\n    except Exception as _e1:\n        print(\"  RMSA InferenceDataset failed — \" + str(_e1))\n        raise ImportError(\n            \"RMSA InferenceDataset failed even after biotite patch: \" + str(_e1)\n        )\n    from runner.inference import update_inference_configs\n\n    configs = build_configs(\n        input_json_path, str(work_dir / \"outputs\"),\n        BASE_MODEL_NAME, seed=use_seed\n    )\n\n    # O1: load via RMSA-compatible runner so finetuned checkpoint works\n    runner  = load_runner_with_checkpoint(configs)\n    dataset = InferenceDataset(configs)\n\n    # ── 3. Inference loop ─────────────────────────────────────────────────────\n    raw_predictions = {}\n\n    for i in tqdm(range(len(dataset)), desc=\"Protenix\"):\n        data, atom_array, err = dataset[i]\n        sample_name = data.get(\"sample_name\", f\"sample_{i}\")\n\n        if err:\n            print(f\"  {sample_name}: data error — {err}\")\n            raw_predictions[sample_name] = None\n            del data, atom_array, err\n            gc.collect(); torch.cuda.empty_cache(); gc.collect()\n            continue\n\n        target_id   = (sample_name.split(\"_chunk\")[0]\n                       if \"_chunk\" in sample_name else sample_name)\n        n_needed    = protenix_queue.get(target_id, (N_SAMPLE, \"\"))[0]\n        sub_seq_len = data[\"N_token\"].item()\n\n        try:\n            new_cfg = update_inference_configs(configs, sub_seq_len)\n            new_cfg.sample_diffusion.N_sample = n_needed\n            runner.update_model_configs(new_cfg)\n\n            pred       = runner.predict(data)\n            raw_coords = pred[\"coordinate\"]\n            coords     = _extract_c1_coords(data, atom_array, sub_seq_len, raw_coords)\n            raw_predictions[sample_name] = coords\n            if coords is not None:\n                print(f\"\\n  {sample_name}: {coords.shape[0]} preds ✓\")\n            else:\n                print(f\"\\n  {sample_name}: extraction failed\")\n        except Exception as exc:\n            import traceback\n            print(f\"\\n  {sample_name}: FAILED — {exc}\")\n            traceback.print_exc()\n            raw_predictions[sample_name] = None\n        finally:\n            try:\n                runner.update_model_configs(configs)\n                del pred, data, atom_array, raw_coords\n            except Exception:\n                pass\n            gc.collect(); torch.cuda.empty_cache(); gc.collect()\n\n    # Explicit cleanup so dual-seed second pass doesn't OOM\n    try:\n        del runner, dataset\n    except Exception:\n        pass\n    gc.collect(); torch.cuda.empty_cache(); gc.collect()\n\n    # ── 4. Stitch chunks → protenix_preds ────────────────────────────────────\n    protenix_preds = {}\n    for target_id, (n_needed, full_seq) in protenix_queue.items():\n        seq_len = len(full_seq)\n        chunks  = chunk_info.get(target_id, [])\n        if not chunks:\n            continue\n\n        if len(chunks) == 1:\n            coords = raw_predictions.get(target_id)\n            protenix_preds[target_id] = coords\n            if coords is not None:\n                print(f\"  {target_id}: {coords.shape[0]} preds ✓\")\n            else:\n                print(f\"  {target_id}: FAILED → fallback to TBM / de-novo\")\n        else:\n            per_sample = {s: [] for s in range(n_needed)}\n            all_ok     = True\n            for cinfo in chunks:\n                ccoords = raw_predictions.get(cinfo[\"name\"])\n                if ccoords is None:\n                    all_ok = False; break\n                for s_idx in range(n_needed):\n                    si = s_idx if s_idx < ccoords.shape[0] else -1\n                    per_sample[s_idx].append((ccoords[si], cinfo[\"range\"]))\n            if not all_ok:\n                print(f\"  {target_id}: chunked incomplete → fallback\")\n                protenix_preds[target_id] = None\n                continue\n            stitched = []\n            for s_idx in range(n_needed):\n                items = per_sample[s_idx]\n                fc    = stitch_chunk_coords(\n                    [c for c, _ in items],\n                    [r for _, r in items],\n                    seq_len\n                )\n                stitched.append(fc)\n            result = np.stack(stitched, axis=0)\n            protenix_preds[target_id] = result\n            print(f\"  {target_id}: {result.shape[0]} stitched preds ✓\")\n\n    return protenix_preds\n\n\n# ══════════════════════════════════════════════════════════════════════════════\n# MAIN\n# ══════════════════════════════════════════════════════════════════════════════\ndef main():\n    t_total = time.time()\n    test_csv, output_csv, code_dir, root_dir = resolve_paths()\n\n    if not os.path.isdir(code_dir):\n        raise FileNotFoundError(\n            f\"Missing PROTENIX_CODE_DIR: {code_dir}\\n\"\n            f\"Attach 'zoushuxian/protenix-rmsa-repo' to this notebook.\"\n        )\n\n    os.environ[\"PROTENIX_ROOT_DIR\"] = root_dir\n\n    # O1: RMSA codebase first so finetuned checkpoint architecture matches\n    if code_dir not in sys.path:\n        sys.path.insert(0, code_dir)\n    if root_dir not in sys.path:\n        sys.path.append(root_dir)\n\n    ensure_required_files(root_dir)\n    seed_everything(SEED)\n\n    print(f\"\\n{'='*65}\")\n    print(\"MERGED RNA 3D — CONFIGURATION SUMMARY\")\n    print(f\"  Code source (imports) : {code_dir}\")\n    print(f\"  Assets (CCD/ckpt)     : {root_dir}\")\n    print(f\"  Finetuned checkpoint  : \"\n          f\"{'YES ✓' if USE_FINETUNED_CKPT else 'NOT FOUND — base model'}\")\n    print(f\"  N_SAMPLE              : {N_SAMPLE}\")\n    print(f\"  N_step / N_cycle      : {DIFFUSION_N_STEP} / {DIFFUSION_N_CYCLE}\")\n    print(f\"  MAX_SEQ_LEN           : {MAX_SEQ_LEN}\")\n    print(f\"  CHUNK_OVERLAP         : {CHUNK_OVERLAP}\")\n    print(f\"  Adaptive threshold    : {MIN_PERCENT_IDENTITY_BASE}% → floor \"\n          f\"{THRESHOLD_FLOOR}% ({LONG_SEQ_THRESHOLD}% long)\")\n    print(f\"  Dual-seed Protenix    : YES ✓ (101 + 202 for hard targets)\")\n    print(f\"  Consensus reranking   : YES ✓\")\n    print(f\"  Stable hash           : YES ✓\")\n    print(f\"  ID-suffix sort fix    : YES ✓\")\n    print(f\"  Kabsch chunk stitching: YES ✓\")\n    print(f\"{'='*65}\\n\")\n\n    if IS_KAGGLE:\n        print(\"Running in KAGGLE COMPETITION mode — all test targets.\")\n    else:\n        print(f\"Running in LOCAL mode — first {LOCAL_N_SAMPLES} targets only.\")\n\n    # ── Load test data ─────────────────────────────────────────────────────────\n    test_df_full = pd.read_csv(test_csv)\n    test_df      = (test_df_full.head(LOCAL_N_SAMPLES) if not IS_KAGGLE\n                    else test_df_full).reset_index(drop=True)\n    print(f\"Test targets: {len(test_df)}\")\n\n    # ── Load training + validation data for TBM ────────────────────────────────\n    print(\"\\nLoading training + validation data…\")\n    train_seqs   = pd.read_csv(DEFAULT_TRAIN_CSV)\n    val_seqs     = pd.read_csv(DEFAULT_VAL_CSV)\n    # T7: usecols — only load what we need from the large labels file\n    _label_cols  = [\"ID\", \"resid\", \"x_1\", \"y_1\", \"z_1\"]\n    train_labels = pd.read_csv(DEFAULT_TRAIN_LBLS, usecols=_label_cols, low_memory=False)\n    val_labels   = pd.read_csv(DEFAULT_VAL_LBLS,   usecols=_label_cols, low_memory=False)\n\n    combined_seqs   = pd.concat([train_seqs,   val_seqs],   ignore_index=True)\n    combined_labels = pd.concat([train_labels, val_labels], ignore_index=True)\n\n    del train_seqs, val_seqs, train_labels, val_labels\n    gc.collect()\n\n    # T1: ID-suffix sort fix\n    print(\"Processing labels (T1: ID-suffix sort fix)…\")\n    train_coords = process_labels(combined_labels)\n    del combined_labels\n    gc.collect()\n\n    segments_map, _ = build_segments_map(test_df)\n    print(f\"Template pool: {len(combined_seqs)} seqs, {len(train_coords)} structures\")\n\n    # T5: exact-match O(1) index\n    exact_idx = build_exact_match_index(combined_seqs, train_coords)\n\n    # ─── PHASE 1: TBM ─────────────────────────────────────────────────────────\n    template_preds, protenix_queue = tbm_phase(\n        test_df, combined_seqs, train_coords, segments_map, exact_idx=exact_idx\n    )\n\n    # ─── PHASE 2: Protenix ────────────────────────────────────────────────────\n    protenix_preds = {}\n\n    if protenix_queue and USE_PROTENIX:\n        work_dir = Path(\"/kaggle/working\")\n        work_dir.mkdir(parents=True, exist_ok=True)\n\n        # T3: dual-seed — hard targets (0 TBM preds) get seed-101 + seed-202\n        hard_targets = {tid for tid, preds in template_preds.items()\n                        if len(preds) == 0}\n        print(f\"\\n  Protenix queue: {len(protenix_queue)} targets \"\n              f\"({len(hard_targets)} hard / dual-seed)\")\n\n        # Seed 1: all queued targets\n        protenix_preds = run_protenix(\n            protenix_queue, work_dir, code_dir, root_dir, seed=101\n        )\n\n        # Seed 2: hard targets only\n        hard_queue = {tid: v for tid, v in protenix_queue.items()\n                      if tid in hard_targets}\n        if hard_queue:\n            print(f\"\\n  Dual-seed pass for {len(hard_queue)} hard targets (seed=202)…\")\n            preds_seed2 = run_protenix(\n                hard_queue, work_dir, code_dir, root_dir, seed=202\n            )\n            gc.collect(); torch.cuda.empty_cache(); gc.collect()\n\n            for tid in hard_queue:\n                p1 = protenix_preds.get(tid)\n                p2 = preds_seed2.get(tid)\n                if (p1 is not None and p2 is not None\n                        and p1.ndim == 3 and p2.ndim == 3):\n                    half   = N_SAMPLE // 2\n                    take1  = min(p1.shape[0], N_SAMPLE - half)\n                    take2  = min(p2.shape[0], half)\n                    protenix_preds[tid] = np.concatenate(\n                        [p1[:take1], p2[:take2]], axis=0\n                    )\n                    print(f\"    {tid}: dual-seed combined \"\n                          f\"({take1} seed-101 + {take2} seed-202)\")\n                elif p1 is None and p2 is not None:\n                    protenix_preds[tid] = p2\n\n    elif protenix_queue and not USE_PROTENIX:\n        print(f\"\\nPHASE 2 skipped (USE_PROTENIX=False).\")\n\n    # ─── PHASE 3: Combine + O4 consensus rerank + de-novo fallback ───────────\n    print(f\"\\n{'='*65}\")\n    print(\"PHASE 3: Combine TBM + Protenix → consensus rerank → de-novo\")\n    print(f\"{'='*65}\")\n\n    all_rows = []\n\n    for _, row in test_df.iterrows():\n        tid, seq = row[\"target_id\"], row[\"sequence\"]\n        seq_len  = len(seq)\n\n        all_candidates: list = list(template_preds.get(tid, []))\n        n_tbm = len(all_candidates)\n\n        # T8: append raw Protenix coords (no constraints — neural-net geometry is valid)\n        ptx = protenix_preds.get(tid)\n        if ptx is not None and isinstance(ptx, np.ndarray) and ptx.ndim == 3:\n            for j in range(ptx.shape[0]):\n                all_candidates.append(ptx[j].astype(np.float64))\n        n_ptx = len(all_candidates) - n_tbm\n\n        # O4: consensus rerank before selecting top N_SAMPLE\n        if len(all_candidates) > N_SAMPLE:\n            all_candidates = consensus_rerank(all_candidates, seq_len)\n\n        combined = all_candidates[:N_SAMPLE]\n\n        # De-novo fallback for remaining empty slots\n        n_dn = 0\n        while len(combined) < N_SAMPLE:\n            # O2: stable hash for reproducible de-novo seeds\n            seed_val = stable_hash(tid) + len(combined) * 1000\n            dn       = generate_rna_structure(seq, seed=seed_val % (2 ** 32))\n            combined.append(\n                adaptive_rna_constraints(dn, tid, segments_map, confidence=0.2)\n            )\n            n_dn += 1\n\n        strategy = f\"TBM={n_tbm} PTX={n_ptx}\"\n        if n_dn:\n            strategy += f\" DN={n_dn}\"\n        print(f\"  {tid} ({seq_len} nt): {strategy}\")\n\n        # O7: safe_stack handles any shape mismatches\n        stacked = safe_stack(combined, seq_len, N_SAMPLE)\n        all_rows.extend(coords_to_rows(tid, seq, stacked))\n\n    # ── Save ──────────────────────────────────────────────────────────────────\n    sub  = pd.DataFrame(all_rows)\n    cols = [\"ID\", \"resname\", \"resid\"] + [\n        f\"{c}_{i}\" for i in range(1, N_SAMPLE + 1) for c in [\"x\", \"y\", \"z\"]\n    ]\n    cc   = [c for c in cols if c.startswith((\"x_\", \"y_\", \"z_\"))]\n    sub[cc] = sub[cc].clip(-999.999, 9999.999)\n    sub[cols].to_csv(output_csv, index=False)\n\n    elapsed = time.time() - t_total\n    print(f\"\\n✓ Saved {output_csv} ({len(sub):,} rows) in {elapsed/60:.1f} min\")\n\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-09T16:32:22.493752Z","iopub.execute_input":"2026-03-09T16:32:22.494413Z","iopub.status.idle":"2026-03-09T16:34:05.389362Z","shell.execute_reply.started":"2026-03-09T16:32:22.494341Z","shell.execute_reply":"2026-03-09T16:34:05.388459Z"}},"outputs":[{"name":"stdout","text":"Looking in links: /kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages, /kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages\nRequirement already satisfied: biotite in /usr/local/lib/python3.12/dist-packages (from -r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (1.6.0)\nRequirement already satisfied: rdkit in /usr/local/lib/python3.12/dist-packages (from -r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 2)) (2025.9.5)\nRequirement already satisfied: biopython in /usr/local/lib/python3.12/dist-packages (from -r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 3)) (1.86)\nRequirement already satisfied: biotraj<2.0,>=1.0 in /usr/local/lib/python3.12/dist-packages (from biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (1.2.2)\nRequirement already satisfied: msgpack>=0.5.6 in /usr/local/lib/python3.12/dist-packages (from biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (1.1.2)\nRequirement already satisfied: networkx>=2.0 in /usr/local/lib/python3.12/dist-packages (from biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (3.6.1)\nRequirement already satisfied: numpy>=1.25 in /usr/local/lib/python3.12/dist-packages (from biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (2.0.2)\nRequirement already satisfied: packaging>=24.0 in /usr/local/lib/python3.12/dist-packages (from biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (25.0)\nRequirement already satisfied: requests>=2.12 in /usr/local/lib/python3.12/dist-packages (from biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (2.32.4)\nRequirement already satisfied: Pillow in /usr/local/lib/python3.12/dist-packages (from rdkit->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 2)) (11.3.0)\nRequirement already satisfied: scipy>=1.13 in /usr/local/lib/python3.12/dist-packages (from biotraj<2.0,>=1.0->biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (1.16.3)\nRequirement already satisfied: charset_normalizer<4,>=2 in /usr/local/lib/python3.12/dist-packages (from requests>=2.12->biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (3.4.4)\nRequirement already satisfied: idna<4,>=2.5 in /usr/local/lib/python3.12/dist-packages (from requests>=2.12->biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (3.11)\nRequirement already satisfied: urllib3<3,>=1.21.1 in /usr/local/lib/python3.12/dist-packages (from requests>=2.12->biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (2.5.0)\nRequirement already satisfied: certifi>=2017.4.17 in /usr/local/lib/python3.12/dist-packages (from requests>=2.12->biotite->-r /kaggle/input/notebooks/jonathanarya/dependency-installation-script/requirements.txt (line 1)) (2026.1.4)\n","output_type":"stream"},{"name":"stderr","text":"/usr/local/lib/python3.12/dist-packages/Bio/Align/__init__.py:4414: BiopythonDeprecationWarning: The attribute 'query_left_open_gap_score' was renamed to 'open_left_deletion_score'. This was done to be consistent with the\nAlignmentCounts object returned by the .counts method of an Alignment object.\n  warnings.warn(\n/usr/local/lib/python3.12/dist-packages/Bio/Align/__init__.py:4414: BiopythonDeprecationWarning: The attribute 'query_left_extend_gap_score' was renamed to 'extend_left_deletion_score'. This was done to be consistent with the\nAlignmentCounts object returned by the .counts method of an Alignment object.\n  warnings.warn(\n/usr/local/lib/python3.12/dist-packages/Bio/Align/__init__.py:4414: BiopythonDeprecationWarning: The attribute 'query_right_open_gap_score' was renamed to 'open_right_deletion_score'. This was done to be consistent with the\nAlignmentCounts object returned by the .counts method of an Alignment object.\n  warnings.warn(\n/usr/local/lib/python3.12/dist-packages/Bio/Align/__init__.py:4414: BiopythonDeprecationWarning: The attribute 'query_right_extend_gap_score' was renamed to 'extend_right_deletion_score'. This was done to be consistent with the\nAlignmentCounts object returned by the .counts method of an Alignment object.\n  warnings.warn(\n/usr/local/lib/python3.12/dist-packages/Bio/Align/__init__.py:4414: BiopythonDeprecationWarning: The attribute 'target_left_open_gap_score' was renamed to 'open_left_insertion_score'. This was done to be consistent with the\nAlignmentCounts object returned by the .counts method of an Alignment object.\n  warnings.warn(\n/usr/local/lib/python3.12/dist-packages/Bio/Align/__init__.py:4414: BiopythonDeprecationWarning: The attribute 'target_left_extend_gap_score' was renamed to 'extend_left_insertion_score'. This was done to be consistent with the\nAlignmentCounts object returned by the .counts method of an Alignment object.\n  warnings.warn(\n/usr/local/lib/python3.12/dist-packages/Bio/Align/__init__.py:4414: BiopythonDeprecationWarning: The attribute 'target_right_open_gap_score' was renamed to 'open_right_insertion_score'. This was done to be consistent with the\nAlignmentCounts object returned by the .counts method of an Alignment object.\n  warnings.warn(\n/usr/local/lib/python3.12/dist-packages/Bio/Align/__init__.py:4414: BiopythonDeprecationWarning: The attribute 'target_right_extend_gap_score' was renamed to 'extend_right_insertion_score'. This was done to be consistent with the\nAlignmentCounts object returned by the .counts method of an Alignment object.\n  warnings.warn(\n","output_type":"stream"},{"name":"stdout","text":"\n=================================================================\nMERGED RNA 3D — CONFIGURATION SUMMARY\n  Code source (imports) : /kaggle/input/datasets/zoushuxian/protenix-rmsa-repo/protenix_kaggle\n  Assets (CCD/ckpt)     : /kaggle/input/datasets/qiweiyin/protenix-v1-adjusted/Protenix-v1-adjust-v2/Protenix-v1-adjust-v2/Protenix-v1\n  Finetuned checkpoint  : YES ✓\n  N_SAMPLE              : 20\n  N_step / N_cycle      : 200 / 4\n  MAX_SEQ_LEN           : 512\n  CHUNK_OVERLAP         : 192\n  Adaptive threshold    : 50.0% → floor 30.0% (25.0% long)\n  Dual-seed Protenix    : YES ✓ (101 + 202 for hard targets)\n  Consensus reranking   : YES ✓\n  Stable hash           : YES ✓\n  ID-suffix sort fix    : YES ✓\n  Kabsch chunk stitching: YES ✓\n=================================================================\n\nRunning in LOCAL mode — first 2 targets only.\nTest targets: 2\n\nLoading training + validation data…\nProcessing labels (T1: ID-suffix sort fix)…\nTemplate pool: 5744 seqs, 5744 structures\n  Exact-match index: 3460 unique seqs, 5744 total entries\n\n=================================================================\nPHASE 1: Template-Based Modeling\n  MIN_PCT_IDENTITY (base) = 50.0\n  Adaptive floor (normal) = 30.0\n  Adaptive floor (long)   = 25.0\n  N_SAMPLE                = 20\n=================================================================\n  8ZNQ (30 nt): 2 TBM (thr=50%) → need 18 Protenix\n  9IWF (69 nt): all 20 TBM ✓ (thr=50%)\n\nPhase 1 done in 1.1s | TBM-only:1 | Protenix:1\n\n  Protenix queue: 1 targets (0 hard / dual-seed)\n\n=================================================================\nPHASE 2: Protenix (chunked, seed=101)\n  N_step=200, N_cycle=4\n  Checkpoint: finetuned RNA ✓\n=================================================================\n  GPU 0: Tesla P100-PCIE-16GB (9.9/15.9 GB free)\n  8ZNQ (30 nt): single pass\n  parser.py patched and loaded ✓\n  InferenceDataset: RMSA path ✓\n","output_type":"stream"},{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mModuleNotFoundError\u001b[0m                       Traceback (most recent call last)","\u001b[0;32m/tmp/ipykernel_55/826050178.py\u001b[0m in \u001b[0;36m<cell line: 0>\u001b[0;34m()\u001b[0m\n\u001b[1;32m   1321\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m   1322\u001b[0m \u001b[0;32mif\u001b[0m \u001b[0m__name__\u001b[0m \u001b[0;34m==\u001b[0m \u001b[0;34m\"__main__\"\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m-> 1323\u001b[0;31m     \u001b[0mmain\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m","\u001b[0;32m/tmp/ipykernel_55/826050178.py\u001b[0m in \u001b[0;36mmain\u001b[0;34m()\u001b[0m\n\u001b[1;32m   1227\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m   1228\u001b[0m         \u001b[0;31m# Seed 1: all queued targets\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m-> 1229\u001b[0;31m         protenix_preds = run_protenix(\n\u001b[0m\u001b[1;32m   1230\u001b[0m             \u001b[0mprotenix_queue\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mwork_dir\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mcode_dir\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mroot_dir\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mseed\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;36m101\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m   1231\u001b[0m         )\n","\u001b[0;32m/tmp/ipykernel_55/826050178.py\u001b[0m in \u001b[0;36mrun_protenix\u001b[0;34m(protenix_queue, work_dir, code_dir, root_dir, seed)\u001b[0m\n\u001b[1;32m   1021\u001b[0m     \u001b[0;32mfrom\u001b[0m \u001b[0mrunner\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0minference\u001b[0m \u001b[0;32mimport\u001b[0m \u001b[0mupdate_inference_configs\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m   1022\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m-> 1023\u001b[0;31m     configs = build_configs(\n\u001b[0m\u001b[1;32m   1024\u001b[0m         \u001b[0minput_json_path\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mstr\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mwork_dir\u001b[0m \u001b[0;34m/\u001b[0m \u001b[0;34m\"outputs\"\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m   1025\u001b[0m         \u001b[0mBASE_MODEL_NAME\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mseed\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0muse_seed\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/tmp/ipykernel_55/826050178.py\u001b[0m in \u001b[0;36mbuild_configs\u001b[0;34m(input_json_path, dump_dir, model_name, n_sample, seed)\u001b[0m\n\u001b[1;32m    239\u001b[0m     \u001b[0;32mfrom\u001b[0m \u001b[0mconfigs\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mconfigs_data\u001b[0m \u001b[0;32mimport\u001b[0m \u001b[0mdata_configs\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    240\u001b[0m     \u001b[0;32mfrom\u001b[0m \u001b[0mconfigs\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mconfigs_inference\u001b[0m \u001b[0;32mimport\u001b[0m \u001b[0minference_configs\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 241\u001b[0;31m     \u001b[0;32mfrom\u001b[0m \u001b[0mconfigs\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mconfigs_model_type\u001b[0m \u001b[0;32mimport\u001b[0m \u001b[0mmodel_configs\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m    242\u001b[0m     \u001b[0;32mfrom\u001b[0m \u001b[0mprotenix\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mconfig\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mconfig\u001b[0m \u001b[0;32mimport\u001b[0m \u001b[0mparse_configs\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    243\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;31mModuleNotFoundError\u001b[0m: No module named 'configs.configs_model_type'"],"ename":"ModuleNotFoundError","evalue":"No module named 'configs.configs_model_type'","output_type":"error"}],"execution_count":40},{"cell_type":"code","source":"import subprocess, os\n\n# Search for modelgenerator anywhere in input\nresult = subprocess.run(\n    ['find', '/kaggle/input', '-name', '*.whl'], \n    capture_output=True, text=True\n)\nprint(\"WHL files:\", result.stdout)\n\n# Also check if installable directly\nresult2 = subprocess.run(\n    ['find', '/kaggle/input', '-name', 'setup.py', '-o', '-name', 'pyproject.toml'],\n    capture_output=True, text=True\n)\nprint(\"Setup files:\", result2.stdout)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-09T15:56:59.722324Z","iopub.execute_input":"2026-03-09T15:56:59.723054Z","iopub.status.idle":"2026-03-09T15:57:11.128265Z","shell.execute_reply.started":"2026-03-09T15:56:59.723026Z","shell.execute_reply":"2026-03-09T15:57:11.127699Z"}},"outputs":[{"name":"stdout","text":"WHL files: /kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/rdkit-2025.9.5-cp312-cp312-manylinux_2_28_x86_64.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/requests-2.32.5-py3-none-any.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/biopython-1.86-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/packaging-26.0-py3-none-any.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/numpy-2.4.2-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/networkx-3.6.1-py3-none-any.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/scipy-1.17.0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/biotite-1.6.0-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/pillow-12.1.1-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/idna-3.11-py3-none-any.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/biotraj-1.2.2-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/urllib3-2.6.3-py3-none-any.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/charset_normalizer-3.4.4-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/msgpack-1.1.2-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl\n/kaggle/input/notebooks/jonathanarya/dependency-installation-script/packages/certifi-2026.1.4-py3-none-any.whl\n/kaggle/input/datasets/kami1976/biopython-cp312/biopython-1.86-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl\n\nSetup files: /kaggle/input/datasets/qiweiyin/protenix-v1-adjusted/Protenix-v1-adjust/Protenix-v1/setup.py\n/kaggle/input/datasets/qiweiyin/protenix-v1-adjusted/Protenix-v1-adjust-v2/Protenix-v1-adjust-v2/Protenix-v1/setup.py\n/kaggle/input/datasets/zoushuxian/protenix-rmsa-repo/protenix_kaggle/setup.py\n\n","output_type":"stream"}],"execution_count":33}]}