{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210}],"dockerImageVersionId":31328,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-23T21:40:03.383892Z","iopub.execute_input":"2026-03-23T21:40:03.384348Z","iopub.status.idle":"2026-03-23T21:40:14.462513Z","shell.execute_reply.started":"2026-03-23T21:40:03.384296Z","shell.execute_reply":"2026-03-23T21:40:14.461386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===============================\n# Stanford RNA 3D Folding - SIMPLE BASELINE (SUBMITTABLE)\n# ===============================\n\nimport numpy as np\nimport pandas as pd\nfrom difflib import SequenceMatcher\n\nBASE_PATH = \"/kaggle/input/competitions/stanford-rna-3d-folding-2\"\n\n# ===============================\n# 1. Load data\n# ===============================\ntrain_seq = pd.read_csv(f\"{BASE_PATH}/train_sequences.csv\")\ntrain_lab = pd.read_csv(f\"{BASE_PATH}/train_labels.csv\")\ntest_seq  = pd.read_csv(f\"{BASE_PATH}/test_sequences.csv\")\n\nprint(\"Loaded data:\")\nprint(train_seq.shape, train_lab.shape, test_seq.shape)\n\n# ===============================\n# 2. Parse labels\n# ===============================\ndef extract_target_id(id_value):\n    return str(id_value).rsplit(\"_\", 1)[0]\n\ntrain_lab[\"target_id\"] = train_lab[\"ID\"].apply(extract_target_id)\n\n# use ONLY first structure (x_1,y_1,z_1)\ntemplate_cols = [\"x_1\", \"y_1\", \"z_1\"]\n\nlabel_dict = {}\n\nfor tid, grp in train_lab.groupby(\"target_id\"):\n    grp = grp.sort_values(\"resid\")\n    coords = grp[template_cols].values\n    label_dict[tid] = coords\n\n# keep only usable sequences\ntrain_seq = train_seq[train_seq[\"target_id\"].isin(label_dict.keys())].copy()\ntrain_seq[\"seq_len\"] = train_seq[\"sequence\"].str.len()\n\nprint(\"Train usable:\", train_seq.shape)\n\n# ===============================\n# 3. Similarity function\n# ===============================\ndef similarity(a, b):\n    return SequenceMatcher(None, a, b).ratio()\n\ndef find_best_template(seq):\n    candidates = train_seq.copy()\n    candidates[\"len_diff\"] = (candidates[\"seq_len\"] - len(seq)).abs()\n    candidates = candidates.sort_values(\"len_diff\").head(200)\n\n    candidates[\"sim\"] = candidates[\"sequence\"].apply(lambda x: similarity(seq, x))\n    best = candidates.sort_values([\"sim\",\"len_diff\"], ascending=[False, True]).iloc[0]\n\n    return best[\"target_id\"]\n\n# ===============================\n# 4. Resize coordinates\n# ===============================\ndef resize(coords, new_len):\n    old_len = coords.shape[0]\n    if old_len == new_len:\n        return coords.copy()\n\n    old_x = np.linspace(0, 1, old_len)\n    new_x = np.linspace(0, 1, new_len)\n\n    out = np.zeros((new_len, 3))\n    for d in range(3):\n        out[:, d] = np.interp(new_x, old_x, coords[:, d])\n\n    return out\n\n# ===============================\n# 5. Create 5 predictions\n# ===============================\ndef make_5(coords):\n    preds = [coords.copy()]\n    rng = np.random.default_rng(42)\n\n    for i in range(4):\n        noise = rng.normal(0, 0.2 + 0.05*i, coords.shape)\n        preds.append(coords + noise)\n\n    return preds\n\n# ===============================\n# 6. Build submission\n# ===============================\nrows = []\n\nfor _, row in test_seq.iterrows():\n\n    tid = row[\"target_id\"]\n    seq = row[\"sequence\"]\n    L = len(seq)\n\n    # find template\n    best_tid = find_best_template(seq)\n\n    template = label_dict[best_tid]\n    coords = resize(template, L)\n\n    preds = make_5(coords)\n\n    for i, base in enumerate(seq, start=1):\n\n        record = {\n            \"ID\": f\"{tid}_{i}\",\n            \"resname\": base,\n            \"resid\": i,\n        }\n\n        for k in range(5):\n            record[f\"x_{k+1}\"] = float(preds[k][i-1,0])\n            record[f\"y_{k+1}\"] = float(preds[k][i-1,1])\n            record[f\"z_{k+1}\"] = float(preds[k][i-1,2])\n\n        rows.append(record)\n\nsubmission = pd.DataFrame(rows)\n\n# ===============================\n# 7. Final format check\n# ===============================\ncols = [\"ID\",\"resname\",\"resid\"]\nfor i in range(1,6):\n    cols += [f\"x_{i}\", f\"y_{i}\", f\"z_{i}\"]\n\nsubmission = submission[cols]\n\n# check\nprint(\"Submission shape:\", submission.shape)\nprint(submission.head())\n\n# save\nsubmission.to_csv(\"submission.csv\", index=False)\n\nprint(\"✅ submission.csv ready!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T21:42:56.95551Z","iopub.execute_input":"2026-03-23T21:42:56.955904Z","iopub.status.idle":"2026-03-23T21:43:24.384436Z","shell.execute_reply.started":"2026-03-23T21:42:56.955871Z","shell.execute_reply":"2026-03-23T21:43:24.383574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===============================\n# FIX ALL NaN / INVALID VALUES\n# ===============================\n\nimport numpy as np\n\n\ncoord_cols = [c for c in submission.columns if c.startswith((\"x_\", \"y_\", \"z_\"))]\n\n\nsubmission[coord_cols] = submission[coord_cols].replace([np.inf, -np.inf], np.nan)\n\n\nsubmission[coord_cols] = submission[coord_cols].fillna(0)\n\n\nsubmission[coord_cols] = submission[coord_cols].astype(float)\n\n\nsubmission[coord_cols] = submission[coord_cols].clip(-100, 100)\n\n\nprint(\"NaN count:\", submission[coord_cols].isna().sum().sum())\nprint(\"Inf count:\", np.isinf(submission[coord_cols]).sum().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T22:02:54.024805Z","iopub.execute_input":"2026-03-23T22:02:54.025166Z","iopub.status.idle":"2026-03-23T22:02:54.068001Z","shell.execute_reply.started":"2026-03-23T22:02:54.025134Z","shell.execute_reply":"2026-03-23T22:02:54.067042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-23T22:03:16.514351Z","iopub.execute_input":"2026-03-23T22:03:16.515249Z","iopub.status.idle":"2026-03-23T22:03:16.675369Z","shell.execute_reply.started":"2026-03-23T22:03:16.515205Z","shell.execute_reply":"2026-03-23T22:03:16.674389Z"}},"outputs":[],"execution_count":null}]}