{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210},{"sourceType":"datasetVersion","sourceId":14750391,"datasetId":9427257,"databundleVersionId":15600400}],"dockerImageVersionId":31260,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":149.496912,"end_time":"2026-02-08T20:20:13.150906","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-02-08T20:17:43.653994","version":"2.6.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"0cb1011dcc2c4e868b9a657f661f35ab":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"18a99f2ab6f4401d9614bb92f60b141c":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"1ef79c53e47e4a168a091c8d8136ad1f":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"26f4d0faf657400f8927c810e64d09aa":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"304831e8c0ab4813bcd69406e30c0f29":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_f89049ee982b45cfaf9f376fcc5fa814","placeholder":"​","style":"IPY_MODEL_80422042b3134cefb450800bd391c39d","tabbable":null,"tooltip":null,"value":"Caching Templates: 100%"}},"33b71cfbb6f14d3c9c817f9e7a726f71":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_c2b2c59ef10441949f08a7bf2756f8ab","IPY_MODEL_ffd46c5fad434410907e9140e2773091","IPY_MODEL_9e7efc40821e4487b728922b44f0646c"],"layout":"IPY_MODEL_18a99f2ab6f4401d9614bb92f60b141c","tabbable":null,"tooltip":null}},"414eb6e387154d65bbec3f9bfd836c6b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"5efe1eea096641aa83dbc01ede03af9a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_985f860d9e564d759351abc66dd4ab47","placeholder":"​","style":"IPY_MODEL_92ea3d90293a462a9f8a74c97594e115","tabbable":null,"tooltip":null,"value":" 5716/5716 [00:04&lt;00:00, 1342.55it/s]"}},"6826a59c004a4c6ebd7961eb14b0b3a1":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"7131685f582c47a59d41ac6bacafc3e7":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_26f4d0faf657400f8927c810e64d09aa","max":5716,"min":0,"orientation":"horizontal","style":"IPY_MODEL_ea55b525d1b14f75a0638c08534f4c72","tabbable":null,"tooltip":null,"value":5716}},"739a2bd7e8d842a6a95f9997a3015a71":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"80422042b3134cefb450800bd391c39d":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"92ea3d90293a462a9f8a74c97594e115":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"985f860d9e564d759351abc66dd4ab47":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"994175e86ae94e9ca65f4379b4ece069":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9e7efc40821e4487b728922b44f0646c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_994175e86ae94e9ca65f4379b4ece069","placeholder":"​","style":"IPY_MODEL_1ef79c53e47e4a168a091c8d8136ad1f","tabbable":null,"tooltip":null,"value":" 28/28 [01:54&lt;00:00,  2.52it/s]"}},"c2b2c59ef10441949f08a7bf2756f8ab":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_739a2bd7e8d842a6a95f9997a3015a71","placeholder":"​","style":"IPY_MODEL_cde99ba85fc04a368225fbc64b5e850b","tabbable":null,"tooltip":null,"value":"100%"}},"cde99ba85fc04a368225fbc64b5e850b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"ea55b525d1b14f75a0638c08534f4c72":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"ecf51759162d4b159aa819a658f2a10a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_304831e8c0ab4813bcd69406e30c0f29","IPY_MODEL_7131685f582c47a59d41ac6bacafc3e7","IPY_MODEL_5efe1eea096641aa83dbc01ede03af9a"],"layout":"IPY_MODEL_0cb1011dcc2c4e868b9a657f661f35ab","tabbable":null,"tooltip":null}},"f89049ee982b45cfaf9f376fcc5fa814":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"ffd46c5fad434410907e9140e2773091":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_6826a59c004a4c6ebd7961eb14b0b3a1","max":28,"min":0,"orientation":"horizontal","style":"IPY_MODEL_414eb6e387154d65bbec3f9bfd836c6b","tabbable":null,"tooltip":null,"value":28}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## TBM BASELINE\n\nMain TBM logic, later used with Protenix and other modifications in full submission pipeline.","metadata":{}},{"cell_type":"code","source":"!pip install /kaggle/input/datasets/yuriygreben/biopython-cp312/biopython-1.86-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl","metadata":{"execution":{"iopub.execute_input":"2026-02-08T20:17:47.138567Z","iopub.status.busy":"2026-02-08T20:17:47.138317Z","iopub.status.idle":"2026-02-08T20:17:55.38751Z","shell.execute_reply":"2026-02-08T20:17:55.386732Z"},"papermill":{"duration":8.25361,"end_time":"2026-02-08T20:17:55.388894","exception":false,"start_time":"2026-02-08T20:17:47.135284","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport sys\nimport gc\nimport time\nimport warnings\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nfrom Bio.Align import PairwiseAligner\n\n# Suppress Biopython warnings for cleaner output\nwarnings.filterwarnings('ignore')\n\nIS_KAGGLE = os.path.exists('/kaggle/input')\nINPUT_DIR = Path('/kaggle/input/stanford-rna-3d-folding-2') if IS_KAGGLE else Path('./input')\nOUTPUT_DIR = Path('/kaggle/working')\n\npd.set_option('display.max_columns', None)\n\n# --- Segmentation Utils\n# These functions parse the 'stoichiometry' (e.g., \"A:1;B:1\") to identify \n# where one chain ends and the next begins. This prevents \"ghost bonds\".\n\ndef parse_fasta(fasta_content):\n    \"\"\"Parses raw FASTA string into {chain_id: sequence} dict.\"\"\"\n    out = {}\n    cur = None\n    seq_parts = []\n    if pd.isna(fasta_content): return {}\n    for line in str(fasta_content).splitlines():\n        line = line.strip()\n        if not line: continue\n        if line.startswith(\">\"):\n            if cur is not None: out[cur] = \"\".join(seq_parts)\n            cur = line[1:].split()[0]\n            seq_parts = []\n        else:\n            seq_parts.append(line.replace(\" \", \"\"))\n    if cur is not None: out[cur] = \"\".join(seq_parts)\n    return out\n\ndef parse_stoichiometry(stoich):\n    \"\"\"Parses 'A:1;B:2' into list of (Chain, Count) tuples.\"\"\"\n    if pd.isna(stoich) or str(stoich).strip() == \"\": return []\n    out = []\n    try:\n        for part in str(stoich).split(';'):\n            if ':' in part:\n                ch, cnt = part.split(':')\n                out.append((ch.strip(), int(cnt)))\n    except: pass\n    return out\n\ndef get_chain_segments(row):\n    \"\"\"Returns (start, end) indices for each chain in the sequence.\"\"\"\n    seq = row['sequence']\n    stoich = row.get('stoichiometry', '')\n    all_seq = row.get('all_sequences', '')\n    if pd.isna(stoich) or pd.isna(all_seq): return [(0, len(seq))]\n    try:\n        chain_dict = parse_fasta(all_seq)\n        order = parse_stoichiometry(stoich)\n        segs = []\n        pos = 0\n        for ch, cnt in order:\n            base = chain_dict.get(ch)\n            if base is None: return [(0, len(seq))]\n            for _ in range(cnt):\n                L = len(base)\n                if pos + L <= len(seq):\n                    segs.append((pos, pos + L))\n                    pos += L\n        return segs if segs else [(0, len(seq))]\n    except:\n        return [(0, len(seq))]\n\n# Pre-calculate segments for all test sequences\ntest_seqs = pd.read_csv(INPUT_DIR / 'test_sequences.csv')\nTEST_SEGMENTS = {}\nfor idx, row in test_seqs.iterrows():\n    TEST_SEGMENTS[row['target_id']] = get_chain_segments(row)\nprint(\"Segmentation map built.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_and_cache_labels(labels_path):\n    \"\"\"Loads 3D coordinates into RAM. Uses float32 to save memory.\"\"\"\n    if not labels_path.exists(): return {}\n    df = pd.read_csv(labels_path, usecols=['ID', 'resid', 'x_1', 'y_1', 'z_1'], \n                     dtype={'x_1': 'float32', 'y_1': 'float32', 'z_1': 'float32', 'resid': 'int32'})\n    df['target_id'] = df['ID'].str.rsplit('_', n=1).str[0]\n    coords_dict = {}\n    grouped = df.groupby('target_id')\n    for name, group in tqdm(grouped, desc=\"Caching Templates\"):\n        coords = group.sort_values('resid')[['x_1', 'y_1', 'z_1']].values\n        coords_dict[name] = coords\n    return coords_dict\n\ntrain_seqs = pd.read_csv(INPUT_DIR / 'train_sequences.csv')\ntrain_coords = load_and_cache_labels(INPUT_DIR / 'train_labels.csv')\n\n# --- Aligner Configuration ---\n# Strict penalties prevent the aligner from matching disparate fragments.\naligner = PairwiseAligner()\naligner.mode = 'global'\naligner.match_score = 2\naligner.mismatch_score = -1.5\naligner.open_gap_score = -8      # High cost to open a gap\naligner.extend_gap_score = -0.4  # Low cost to extend it\n\n# Handle Biopython version differences (1.86 vs older)\ntry:\n    aligner.open_left_deletion_score = -8\n    aligner.open_left_insertion_score = -8\nexcept:\n    aligner.query_left_open_gap_score = -8\n    aligner.target_left_open_gap_score = -8","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def _rotmat(axis, ang):\n    \"\"\"Math helper: Creates a 3D rotation matrix.\"\"\"\n    axis = np.asarray(axis, float)\n    axis = axis / (np.linalg.norm(axis) + 1e-12)\n    x, y, z = axis\n    c, s = np.cos(ang), np.sin(ang)\n    C = 1.0 - c\n    return np.array([\n        [c + x*x*C,     x*y*C - z*s, x*z*C + y*s],\n        [y*x*C + z*s, c + y*y*C,     y*z*C - x*s],\n        [z*x*C - y*s, z*y*C + x*s, c + z*z*C],\n    ])\n\ndef apply_hinge(coords, seg, rng, max_angle_deg=25): \n    \"\"\"Bends the backbone at a random pivot. Relaxed angle (25) for diversity.\"\"\"\n    s, e = seg\n    L = e - s\n    if L < 30: return coords\n    pivot = s + int(rng.integers(10, L - 10))\n    axis = rng.normal(size=3)\n    ang = np.deg2rad(float(rng.uniform(-max_angle_deg, max_angle_deg)))\n    R = _rotmat(axis, ang)\n    X = coords.copy()\n    p0 = X[pivot].copy()\n    X[pivot+1:e] = (X[pivot+1:e] - p0) @ R.T + p0\n    return X\n\ndef jitter_chains(coords, segments, rng, max_angle_deg=12, max_trans=1.5): \n    \"\"\"Rotates/Shifts chains relative to each other. Relaxed trans (1.5) for diversity.\"\"\"\n    X = coords.copy()\n    global_center = X.mean(axis=0, keepdims=True)\n    for (s, e) in segments:\n        if (e - s) < 5: continue\n        axis = rng.normal(size=3)\n        ang = np.deg2rad(float(rng.uniform(-max_angle_deg, max_angle_deg)))\n        R = _rotmat(axis, ang)\n        shift = rng.normal(size=3)\n        shift = shift / (np.linalg.norm(shift) + 1e-12) * float(rng.uniform(0, max_trans))\n        c = X[s:e].mean(axis=0, keepdims=True)\n        X[s:e] = (X[s:e] - c) @ R.T + c + shift\n    X -= X.mean(axis=0, keepdims=True) - global_center\n    return X\n\ndef smooth_wiggle(coords, segments, rng, amp=0.8):\n    \"\"\"Adds gentle sine-wave distortion.\"\"\"\n    X = coords.copy()\n    for (s, e) in segments:\n        L = e - s\n        if L < 20: continue\n        n_ctrl = 6\n        ctrl_x = np.linspace(0, L-1, n_ctrl)\n        ctrl_disp = rng.normal(0, amp, size=(n_ctrl, 3))\n        t = np.arange(L)\n        disp = np.vstack([np.interp(t, ctrl_x, ctrl_disp[:, k]) for k in range(3)]).T\n        X[s:e] += disp\n    return X","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def find_similar_sequences(query_seq, train_seqs_df, top_n=30):\n    \"\"\"Finds top N training sequences matching the query.\"\"\"\n    similar_seqs = []\n    q_len = len(query_seq)\n    \n    # Fast filter by length\n    candidates = train_seqs_df[\n        (train_seqs_df['sequence'].str.len() >= q_len * 0.7) & \n        (train_seqs_df['sequence'].str.len() <= q_len * 1.3)\n    ]\n    \n    for _, row in candidates.iterrows():\n        tid = row['target_id']\n        if tid not in train_coords: continue\n        t_seq = row['sequence']\n        raw_score = aligner.score(query_seq, t_seq)\n        norm_score = raw_score / (2 * min(q_len, len(t_seq)))\n        similar_seqs.append((tid, t_seq, norm_score, train_coords[tid]))\n    \n    similar_seqs.sort(key=lambda x: x[2], reverse=True)\n    return similar_seqs[:top_n]\n\ndef adapt_template_to_query(query_seq, template_seq, template_coords):\n    \"\"\"Maps template coordinates to query, handling insertions/deletions.\"\"\"\n    alignment = next(iter(aligner.align(query_seq, template_seq)))\n    new_coords = np.full((len(query_seq), 3), np.nan, dtype=np.float32)\n    \n    for (q_start, q_end), (t_start, t_end) in zip(*alignment.aligned):\n        t_chunk = template_coords[t_start:t_end]\n        if len(t_chunk) == (q_end - q_start):\n            new_coords[q_start:q_end] = t_chunk\n            \n    # Interpolate missing atoms (Linear)\n    for i in range(len(new_coords)):\n        if np.isnan(new_coords[i, 0]):\n            prev_v = next((j for j in range(i-1, -1, -1) if not np.isnan(new_coords[j, 0])), -1)\n            next_v = next((j for j in range(i+1, len(new_coords)) if not np.isnan(new_coords[j, 0])), -1)\n            if prev_v >= 0 and next_v >= 0:\n                w = (i - prev_v) / (next_v - prev_v)\n                new_coords[i] = (1-w)*new_coords[prev_v] + w*new_coords[next_v]\n            elif prev_v >= 0: new_coords[i] = new_coords[prev_v] + [3, 0, 0]\n            elif next_v >= 0: new_coords[i] = new_coords[next_v] - [3, 0, 0]\n            else: new_coords[i] = [i*3, 0, 0]\n            \n    return np.nan_to_num(new_coords)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def adaptive_rna_constraints(coordinates, segments, confidence=1.0, passes=2):\n    \"\"\"\n    Refines geometry. \n    Passes=2 is key: fixes broken bonds but PRESERVES the template's natural curve.\n    Higher passes (10+) tends to over-smooth and lower the score.\n    \"\"\"\n    coords = coordinates.copy()\n    strength = 0.8 * (1.0 - min(confidence, 0.95))\n    strength = max(strength, 0.05)\n\n    for _ in range(passes):\n        for (s, e) in segments:\n            X = coords[s:e]\n            L = e - s\n            if L < 3: \n                coords[s:e] = X\n                continue\n\n            # 1. Neighbor Constraint (Target 5.95A)\n            d = X[1:] - X[:-1]\n            dist = np.linalg.norm(d, axis=1) + 1e-6\n            scale = (5.95 - dist) / dist\n            adj = (d * scale[:, None]) * (0.25 * strength)\n            X[:-1] -= adj\n            X[1:]  += adj\n\n            # 2. Next-Neighbor Constraint (Target 10.2A)\n            if L > 2:\n                d2 = X[2:] - X[:-2]\n                dist2 = np.linalg.norm(d2, axis=1) + 1e-6\n                scale2 = (10.2 - dist2) / dist2\n                adj2 = (d2 * scale2[:, None]) * (0.12 * strength)\n                X[:-2] -= adj2\n                X[2:]  += adj2\n            \n            coords[s:e] = X\n\n        # 3. Global Repulsion: Pushes overlapping chains apart\n        if len(coords) > 10:\n            indices = np.linspace(0, len(coords)-1, min(len(coords), 300)).astype(int)\n            sub_X = coords[indices]\n            diff = sub_X[:, None, :] - sub_X[None, :, :]\n            dist_sq = np.sum(diff**2, axis=2) + 1e-6\n            mask = (dist_sq < 12.25) & (dist_sq > 0.1)\n            \n            if np.any(mask):\n                dist = np.sqrt(dist_sq)\n                force = (3.5 - dist) / dist\n                delta = diff * force[:, :, None] * mask[:, :, None]\n                move = np.sum(delta, axis=1) * (0.05 * strength)\n                coords[indices] += move\n\n    return coords","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def main():\n    print(f\"Predicting for {len(test_seqs)} sequences (Greedy Temp 0.03 + 2 Passes)...\")\n    results = []\n    \n    # Fallback helix generator (Segment Aware)\n    def get_segmented_helix(seq_len, segments):\n        coords = np.zeros((seq_len, 3), dtype=np.float32)\n        for (s, e) in segments:\n            length = e - s\n            if length > 0:\n                t = np.arange(length)\n                local_coords = np.column_stack([10*np.cos(0.5*t), 10*np.sin(0.5*t), 3*t]).astype(np.float32)\n                z_offset = 0 if s == 0 else coords[s-1, 2] + 10.0\n                local_coords[:, 2] += z_offset\n                coords[s:e] = local_coords\n        return coords\n\n    for idx, row in tqdm(test_seqs.iterrows(), total=len(test_seqs)):\n        tid = row['target_id']\n        seq = row['sequence']\n        segments = TEST_SEGMENTS.get(tid, [(0, len(seq))])\n        \n        templates = find_similar_sequences(seq, train_seqs, top_n=30)\n        \n        preds = []\n        used_templates = set()\n        \n        for i in range(5):\n            seed = (abs(hash(tid)) + i * 10007) % (2**32)\n            rng = np.random.default_rng(seed)\n            \n            try:\n                if not templates:\n                    base_coords = get_segmented_helix(len(seq), segments)\n                    sim = 0.0\n                else:\n                    # Model 0: Best template\n                    if i == 0:\n                        t_id, t_seq, sim, t_coords = templates[0]\n                    # Models 1-4: Weighted random choice\n                    else:\n                        K = min(12, len(templates))\n                        scores = np.array([t[2] for t in templates[:K]])\n                        # GREEDY: Temp 0.03 strongly prefers top scores\n                        weights = np.exp((scores - scores.max()) / 0.03) \n                        for k in range(K):\n                            if templates[k][0] in used_templates: weights[k] *= 0.1\n                        weights /= weights.sum()\n                        k_idx = rng.choice(np.arange(K), p=weights)\n                        t_id, t_seq, sim, t_coords = templates[k_idx]\n\n                    used_templates.add(t_id)\n                    base_coords = adapt_template_to_query(seq, t_seq, t_coords)\n\n                # --- Relaxed Augmentations ---\n                if i == 0:\n                    X = base_coords\n                elif i == 1:\n                    noise = max(0.01, (0.40 - sim) * 0.06)\n                    X = base_coords + rng.normal(0, noise, base_coords.shape)\n                elif i == 2:\n                    longest_seg = max(segments, key=lambda s: s[1]-s[0])\n                    X = apply_hinge(base_coords, longest_seg, rng, max_angle_deg=25)\n                elif i == 3:\n                    X = jitter_chains(base_coords, segments, rng, max_angle_deg=12, max_trans=1.5)\n                else:\n                    X = smooth_wiggle(base_coords, segments, rng, amp=0.8)\n\n                # Physics: 2 passes only\n                refined = adaptive_rna_constraints(X, segments, confidence=sim, passes=2)\n                \n                # Sanitizer\n                if np.isnan(refined).any() or np.abs(refined).max() > 10000:\n                    refined = get_segmented_helix(len(seq), segments)\n                \n                preds.append(refined)\n                \n            except Exception:\n                preds.append(get_segmented_helix(len(seq), segments))\n\n        # Format output\n        for j, char in enumerate(seq):\n            res_data = {'ID': f\"{tid}_{j+1}\", 'resname': char, 'resid': j+1}\n            for m in range(5):\n                x, y, z = preds[m][j]\n                res_data[f'x_{m+1}'] = x\n                res_data[f'y_{m+1}'] = y\n                res_data[f'z_{m+1}'] = z\n            results.append(res_data)\n        \n        # Periodic Garbage Collection\n        if idx % 50 == 0: gc.collect()\n            \n    submission = pd.DataFrame(results)\n    cols = ['ID', 'resname', 'resid'] + [f'{c}_{i}' for i in range(1,6) for c in ['x','y','z']]\n    coord_cols = [c for c in cols if c.startswith(('x_','y_','z_'))]\n    submission[coord_cols] = submission[coord_cols].clip(-999.999, 9999.999)\n    \n    submission[cols].to_csv('submission.csv', index=False)\n    print(\"✅ Submission saved: submission.csv\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}