{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210},{"sourceType":"datasetVersion","sourceId":15280851,"datasetId":9774740,"databundleVersionId":16182340}],"dockerImageVersionId":31287,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ── DIAGNOSTIC: Run this first to see all input paths ────────────────────────\nimport os\nprint(\"=== All files in /kaggle/input ===\")\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    level = root.count(os.sep)\n    if level > 5:\n        continue\n    indent = \"  \" * level\n    print(f\"{indent}{os.path.basename(root)}/\")\n    for f in files[:3]:\n        print(f\"{indent}  {f}\")\n\n# ══════════════════════════════════════════════════════════════════════════════\n# STANFORD RNA 3D FOLDING — FINAL OFFLINE SUBMISSION NOTEBOOK\n# ══════════════════════════════════════════════════════════════════════════════\n\nimport sys, subprocess\nfrom pathlib import Path\n\nKAGGLE_INPUT = Path(\"/kaggle/input\")\nWORK_DIR     = Path(\"/kaggle/working\")\n\ndef run_cmd(cmd):\n    r = subprocess.run(cmd, shell=True, capture_output=True, text=True)\n    if r.stdout.strip(): print(r.stdout[-600:])\n    if r.returncode != 0: print(\"⚠️\", r.stderr[-300:])\n    return r\n\n# ── CELL 1: Find competition data ─────────────────────────────────────────────\nDATA_DIR = None\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    if \"test_sequences.csv\" in files:\n        DATA_DIR = Path(root)\n        break\nif DATA_DIR is None:\n    raise FileNotFoundError(\"Add the competition dataset via the Input panel.\")\nprint(f\"\\n✅ Competition data : {DATA_DIR}\")\n\n# ── CELL 2: Find offline RhoFold package ──────────────────────────────────────\n# Search EVERYWHERE under /kaggle/input for RhoFold folder\nRHOFOLD_DIR = None\nWHEELS_DIR  = None\n\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    if \"inference.py\" in files:\n        RHOFOLD_DIR = Path(root)\n        WHEELS_DIR  = RHOFOLD_DIR.parent / \"wheels\"\n        print(f\"✅ RhoFold found    : {RHOFOLD_DIR}\")\n        print(f\"✅ Wheels found     : {WHEELS_DIR} (exists={WHEELS_DIR.exists()})\")\n        break\n\nif RHOFOLD_DIR is None:\n    raise FileNotFoundError(\n        \"❌ RhoFold not found in /kaggle/input!\\n\"\n        \"Make sure rhofold-offline-package is added via the Input panel.\"\n    )\n\nCKPT_PATH = RHOFOLD_DIR / \"pretrained\" / \"RhoFold_pretrained.pt\"\nprint(f\"✅ CKPT exists      : {CKPT_PATH.exists()}\")\n\n# ── CELL 3: Copy RhoFold to writable location ─────────────────────────────────\n# /kaggle/input is read-only so we copy RhoFold to /kaggle/working first\nimport shutil\nRHOFOLD_WORK = WORK_DIR / \"RhoFold\"\nif not RHOFOLD_WORK.exists():\n    print(\"Copying RhoFold to working directory...\")\n    shutil.copytree(str(RHOFOLD_DIR), str(RHOFOLD_WORK))\n    print(\"✅ Copied to /kaggle/working/RhoFold\")\nelse:\n    print(\"✅ RhoFold already in working directory\")\n\n# Update paths to writable copy\nRHOFOLD_DIR = RHOFOLD_WORK\nCKPT_PATH   = RHOFOLD_DIR / \"pretrained\" / \"RhoFold_pretrained.pt\"\nprint(f\"✅ CKPT exists      : {CKPT_PATH.exists()}\")\n\n# ── CELL 4: Install packages ──────────────────────────────────────────────────\nprint(\"\\nInstalling packages...\")\nif WHEELS_DIR.exists():\n    run_cmd(f\"pip install -q --no-index --find-links={WHEELS_DIR} ml-collections einops openmm dm-tree biopython\")\n    print(\"✅ Installed from offline wheels\")\nelse:\n    print(\"⚠️  No wheels folder — trying online install\")\n    run_cmd(\"pip install -q ml-collections einops openmm\")\n\nrun_cmd(f\"pip install -q -e {RHOFOLD_DIR}\")\nprint(\"✅ RhoFold installed\")\n\n# ── CELL 5: Patch relax.py (now writable!) ────────────────────────────────────\nRELAX_FILE = RHOFOLD_DIR / \"rhofold\" / \"relax\" / \"relax.py\"\nif RELAX_FILE.exists():\n    content = RELAX_FILE.read_text()\n    if \"from simtk.openmm.app import *\" in content and \"try:\" not in content[:300]:\n        RELAX_FILE.write_text(content.replace(\n            \"from simtk.openmm.app import *\",\n            \"try:\\n    from simtk.openmm.app import *\\nexcept ImportError:\\n    try:\\n        from openmm.app import *\\n    except ImportError:\\n        pass\"\n        ))\n        print(\"✅ relax.py patched\")\n    else:\n        print(\"✅ relax.py already patched\")\n\n# ── CELL 5: Imports ───────────────────────────────────────────────────────────\nimport pandas as pd\nimport numpy as np\nimport torch\n\nsys.path.insert(0, str(RHOFOLD_DIR))\n\nTEST_SEQ_CSV = DATA_DIR / \"test_sequences.csv\"\nSAMPLE_SUB   = DATA_DIR / \"sample_submission.csv\"\nMSA_DIR      = DATA_DIR / \"MSA\"\nPRED_DIR     = WORK_DIR / \"predictions\"\nPRED_DIR.mkdir(exist_ok=True)\n\ntest_df    = pd.read_csv(TEST_SEQ_CSV)\nsample_sub = pd.read_csv(SAMPLE_SUB)\n\nprint(f\"\\nTest sequences : {len(test_df)}\")\nprint(f\"GPU available  : {torch.cuda.is_available()}\")\n\n# ── CELL 6: Helper functions ──────────────────────────────────────────────────\nMAX_GPU_LEN = 1000\n\ndef write_fasta(path, seq_id, sequence):\n    with open(path, \"w\") as f:\n        f.write(f\">{seq_id}\\n{sequence}\\n\")\n\ndef parse_c1prime(pdb_path):\n    coords = []\n    with open(pdb_path) as f:\n        for line in f:\n            if line.startswith(\"ATOM\") and line[12:16].strip() == \"C1'\":\n                try:\n                    coords.append([float(line[30:38]),\n                                   float(line[38:46]),\n                                   float(line[46:54])])\n                except ValueError:\n                    pass\n    return np.array(coords) if coords else np.zeros((0, 3))\n\ndef find_pdb(folder):\n    for p in Path(folder).rglob(\"*.pdb\"):\n        return p\n    return None\n\ndef run_rhofold(seq_id, sequence, msa_path=None):\n    out_dir = PRED_DIR / seq_id\n    out_dir.mkdir(exist_ok=True)\n    write_fasta(out_dir / \"input.fasta\", seq_id, sequence)\n\n    if len(sequence) > MAX_GPU_LEN:\n        device = \"cpu\"\n        print(f\"  → len={len(sequence)}, using CPU to avoid OOM\")\n    else:\n        device = \"cuda:0\" if torch.cuda.is_available() else \"cpu\"\n\n    extra = (f\"--input_a3m {msa_path}\"\n             if (msa_path and msa_path.exists())\n             else \"--single_seq_pred True\")\n\n    cmd = (\n        f\"PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True \"\n        f\"python {RHOFOLD_DIR}/inference.py \"\n        f\"--input_fas {out_dir}/input.fasta \"\n        f\"--output_dir {out_dir} \"\n        f\"--ckpt {CKPT_PATH} \"\n        f\"--device {device} \"\n        f\"--relax_steps 0 \"\n        f\"{extra}\"\n    )\n    run_cmd(cmd)\n    return out_dir\n\n# ── CELL 7: Run all 28 predictions ────────────────────────────────────────────\nNOISE_SCALE = 0.5\nall_rows    = []\n\nfor idx, row in test_df.iterrows():\n    seq_id   = row[\"target_id\"]\n    sequence = row[\"sequence\"]\n    n_res    = len(sequence)\n    print(f\"\\n[{idx+1}/{len(test_df)}] {seq_id}  (len={n_res})\")\n\n    out_dir  = run_rhofold(seq_id, sequence, MSA_DIR / f\"{seq_id}.MSA.fasta\")\n    pdb_path = find_pdb(out_dir)\n\n    if pdb_path:\n        base = parse_c1prime(pdb_path)\n        print(f\"  ✅ {len(base)} real C1' coords\")\n        if len(base) < n_res:\n            base = np.vstack([base, np.zeros((n_res - len(base), 3))])\n        elif len(base) > n_res:\n            base = base[:n_res]\n    else:\n        print(f\"  ⚠️  No PDB — using zeros\")\n        base = np.zeros((n_res, 3))\n\n    structs = [base] + [\n        base + np.random.normal(0, NOISE_SCALE, base.shape)\n        for _ in range(4)\n    ]\n\n    for res0, nuc in enumerate(sequence):\n        resid = res0 + 1\n        entry = {\"ID\": f\"{seq_id}_{resid}\", \"resname\": nuc, \"resid\": resid}\n        for i, s in enumerate(structs, 1):\n            entry[f\"x_{i}\"] = float(np.clip(s[res0, 0], -999.999, 9999.999))\n            entry[f\"y_{i}\"] = float(np.clip(s[res0, 1], -999.999, 9999.999))\n            entry[f\"z_{i}\"] = float(np.clip(s[res0, 2], -999.999, 9999.999))\n        all_rows.append(entry)\n\nprint(f\"\\n✅ Done! Total rows: {len(all_rows)}\")\nreal = sum(1 for r in all_rows if r[\"x_1\"] != 0.0)\nprint(f\"Real predictions : {real}/{len(all_rows)}\")\nprint(f\"Zero fallback    : {len(all_rows)-real}/{len(all_rows)}\")\n\n# ── CELL 8: Save submission.csv ───────────────────────────────────────────────\ncoord_cols = []\nfor i in range(1, 6):\n    coord_cols += [f\"x_{i}\", f\"y_{i}\", f\"z_{i}\"]\n\nsubmission = pd.DataFrame(all_rows, columns=[\"ID\", \"resname\", \"resid\"] + coord_cols)\nprint(\"\\nShape      :\", submission.shape)\nprint(\"NaNs       :\", submission.isnull().sum().sum())\nprint(\"Cols match :\", list(submission.columns) == list(sample_sub.columns))\n\nOUT = WORK_DIR / \"submission.csv\"\nsubmission.to_csv(OUT, index=False)\nprint(f\"\\n✅ Saved → {OUT}  ({OUT.stat().st_size/1024:.1f} KB)\")\nprint(\"→ Click 'Submit to Competition' in the right panel!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T22:13:50.590955Z","iopub.execute_input":"2026-03-19T22:13:50.591427Z","iopub.status.idle":"2026-03-19T22:43:51.429845Z","shell.execute_reply.started":"2026-03-19T22:13:50.591398Z","shell.execute_reply":"2026-03-19T22:43:51.428914Z"}},"outputs":[],"execution_count":null}]}