{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"1867f1d7","cell_type":"markdown","source":"# Merge CAFA5 & CAFA6 Train Datasets","metadata":{}},{"id":"79ec14e6","cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom pathlib import Path","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T11:12:09.069437Z","iopub.execute_input":"2026-07-26T11:12:09.069789Z","iopub.status.idle":"2026-07-26T11:12:09.405417Z","shell.execute_reply.started":"2026-07-26T11:12:09.069756Z","shell.execute_reply":"2026-07-26T11:12:09.404455Z"}},"outputs":[],"execution_count":null},{"id":"5ffd2045","cell_type":"code","source":"cafa5_train_terms = pd.read_csv('/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_terms.tsv', sep='\\t')\ncafa6_train_terms = pd.read_csv('/kaggle/input/competitions/cafa-6-protein-function-prediction/Train/train_terms.tsv', sep='\\t')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T11:12:10.264507Z","iopub.execute_input":"2026-07-26T11:12:10.264994Z","iopub.status.idle":"2026-07-26T11:12:12.692447Z","shell.execute_reply.started":"2026-07-26T11:12:10.264963Z","shell.execute_reply":"2026-07-26T11:12:12.69122Z"}},"outputs":[],"execution_count":null},{"id":"10957791","cell_type":"code","source":"print(len(cafa5_train_terms))\nprint(len(cafa6_train_terms))\n\nprint(cafa5_train_terms.aspect.unique())\nprint(cafa6_train_terms.aspect.unique())\n\nprint(len(cafa5_train_terms.EntryID.unique()))\nprint(len(cafa6_train_terms.EntryID.unique()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T11:12:13.86488Z","iopub.execute_input":"2026-07-26T11:12:13.865253Z","iopub.status.idle":"2026-07-26T11:12:14.510228Z","shell.execute_reply.started":"2026-07-26T11:12:13.865222Z","shell.execute_reply":"2026-07-26T11:12:14.509329Z"}},"outputs":[],"execution_count":null},{"id":"bd8f8732","cell_type":"code","source":"# Standardize the aspect column\n\nASPECT_MAP = {\"P\": \"BPO\", \"C\": \"CCO\", \"F\": \"MFO\"}\n\ncafa6_train_terms[\"aspect\"] = cafa6_train_terms[\"aspect\"].map(ASPECT_MAP)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T11:12:16.21552Z","iopub.execute_input":"2026-07-26T11:12:16.215888Z","iopub.status.idle":"2026-07-26T11:12:16.257244Z","shell.execute_reply.started":"2026-07-26T11:12:16.215859Z","shell.execute_reply":"2026-07-26T11:12:16.256322Z"}},"outputs":[],"execution_count":null},{"id":"081693ca","cell_type":"code","source":"# Merge annotations: stack rows, then dedupe (EntryID, term).\n\ncafa5_cafa6_train_terms = (\n    pd.concat([cafa5_train_terms, cafa6_train_terms], ignore_index=True)\n    .drop_duplicates(subset=[\"EntryID\", \"term\"])\n)\n\nprint(f\"Merged rows: {len(cafa5_cafa6_train_terms):,}\")\nprint(f\"Unique proteins: {cafa5_cafa6_train_terms['EntryID'].nunique():,}\")\nprint(f\"RAM ~{cafa5_cafa6_train_terms.memory_usage(deep=True).sum() / 1e6:.0f} MB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T11:12:18.115417Z","iopub.execute_input":"2026-07-26T11:12:18.11575Z","iopub.status.idle":"2026-07-26T11:12:24.262869Z","shell.execute_reply.started":"2026-07-26T11:12:18.115721Z","shell.execute_reply":"2026-07-26T11:12:24.26189Z"}},"outputs":[],"execution_count":null},{"id":"ae9dbe75","cell_type":"code","source":"print(len(cafa5_cafa6_train_terms))\n\nprint(cafa5_cafa6_train_terms.aspect.unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T11:12:26.690811Z","iopub.execute_input":"2026-07-26T11:12:26.691138Z","iopub.status.idle":"2026-07-26T11:12:26.952401Z","shell.execute_reply.started":"2026-07-26T11:12:26.69111Z","shell.execute_reply":"2026-07-26T11:12:26.951276Z"}},"outputs":[],"execution_count":null},{"id":"107273cf","cell_type":"code","source":"# Check missing values and duplicates\nprint(cafa5_cafa6_train_terms.isnull().sum())\nprint(cafa5_cafa6_train_terms.duplicated().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T11:12:28.57493Z","iopub.execute_input":"2026-07-26T11:12:28.575246Z","iopub.status.idle":"2026-07-26T11:12:31.344802Z","shell.execute_reply.started":"2026-07-26T11:12:28.575218Z","shell.execute_reply":"2026-07-26T11:12:31.343908Z"}},"outputs":[],"execution_count":null},{"id":"23b951b3-5477-478e-9852-c925ee3972dd","cell_type":"code","source":"MERGED_DIR = Path(\"/kaggle/working/cafa-5-cafa-6-train-datasets/\")\nMERGED_DIR.mkdir(parents=True, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T11:12:33.344149Z","iopub.execute_input":"2026-07-26T11:12:33.344473Z","iopub.status.idle":"2026-07-26T11:12:33.349494Z","shell.execute_reply.started":"2026-07-26T11:12:33.344443Z","shell.execute_reply":"2026-07-26T11:12:33.348553Z"}},"outputs":[],"execution_count":null},{"id":"f1f2747f","cell_type":"code","source":"# Save merged terms \nOUT_TERMS = MERGED_DIR / \"train_terms.tsv\"\ncafa5_cafa6_train_terms.to_csv(OUT_TERMS, sep=\"\\t\", index=False)\nprint(f\"Saved → {OUT_TERMS}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T11:12:35.43267Z","iopub.execute_input":"2026-07-26T11:12:35.432999Z","iopub.status.idle":"2026-07-26T11:12:42.399028Z","shell.execute_reply.started":"2026-07-26T11:12:35.43297Z","shell.execute_reply":"2026-07-26T11:12:42.397962Z"}},"outputs":[],"execution_count":null},{"id":"831ac4a0","cell_type":"markdown","source":"## Merge FASTA \n\nUnion ~145K proteins without loading both full FASTA files into a dict. Writes `data/cafa-5-cafa-6-protein-function-prediction/Train/train_sequences.fasta`","metadata":{}},{"id":"9614c437","cell_type":"code","source":"\n\nCAFA5_FASTA = Path(\"/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_sequences.fasta\")\nCAFA6_FASTA = Path(\"/kaggle/input/competitions/cafa-6-protein-function-prediction/Train/train_sequences.fasta\")\nOUT_FASTA = MERGED_DIR / \"train_sequences.fasta\"\n\n\ndef extract_entry_id(header: str) -> str:\n    h = header.strip().lstrip(\">\")\n    if \"|\" in h:\n        parts = h.split(\"|\")\n        if len(parts) >= 2 and parts[1]:\n            return parts[1]\n    return h.split()[0]\n\n\ndef stream_fasta(path: Path):\n    header, chunks = None, []\n    with path.open() as f:\n        for line in f:\n            line = line.rstrip(\"\\n\")\n            if line.startswith(\">\"):\n                if header is not None:\n                    yield extract_entry_id(header), header, \"\".join(chunks)\n                header, chunks = line, []\n            elif line:\n                chunks.append(line)\n        if header is not None:\n            yield extract_entry_id(header), header, \"\".join(chunks)\n\n\nseen_ids: set[str] = set()\nwritten = 0\n\nwith OUT_FASTA.open(\"w\") as out:\n    for fasta_path in (CAFA5_FASTA, CAFA6_FASTA):\n        for entry_id, header, seq in stream_fasta(fasta_path):\n            if entry_id in seen_ids:\n                continue\n            seen_ids.add(entry_id)\n            out.write(f\"{header}\\n\")\n            for i in range(0, len(seq), 80):\n                out.write(seq[i : i + 80] + \"\\n\")\n            written += 1\n\nprint(f\"Wrote {written:,} sequences → {OUT_FASTA}\")\nprint(f\"IDs tracked in memory: {len(seen_ids):,} (~{len(seen_ids) * 72 / 1e6:.0f} MB for ID set)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-26T11:13:05.880949Z","iopub.execute_input":"2026-07-26T11:13:05.881968Z","iopub.status.idle":"2026-07-26T11:13:07.5508Z","shell.execute_reply.started":"2026-07-26T11:13:05.881929Z","shell.execute_reply":"2026-07-26T11:13:07.549676Z"}},"outputs":[],"execution_count":null},{"id":"7430485e","cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}