{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":41875,"databundleVersionId":5521661},{"sourceType":"datasetVersion","sourceId":16081083,"datasetId":10311822,"databundleVersionId":17050540},{"sourceType":"kernelVersion","sourceId":134063325},{"sourceType":"kernelVersion","sourceId":281424068},{"sourceType":"kernelVersion","sourceId":286397083},{"sourceType":"kernelVersion","sourceId":294261447},{"sourceType":"kernelVersion","sourceId":312548759},{"sourceType":"kernelVersion","sourceId":312596935},{"sourceType":"kernelVersion","sourceId":312860814},{"sourceType":"kernelVersion","sourceId":314249659},{"sourceType":"kernelVersion","sourceId":314988882},{"sourceType":"kernelVersion","sourceId":315396838}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-02T03:37:02.122082Z","iopub.execute_input":"2026-05-02T03:37:02.122355Z","iopub.status.idle":"2026-05-02T03:37:03.162573Z","shell.execute_reply.started":"2026-05-02T03:37:02.122325Z","shell.execute_reply":"2026-05-02T03:37:03.161656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install biopython\n!pip install obonet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-02T03:37:03.164276Z","iopub.execute_input":"2026-05-02T03:37:03.165007Z","iopub.status.idle":"2026-05-02T03:37:12.124302Z","shell.execute_reply.started":"2026-05-02T03:37:03.164982Z","shell.execute_reply":"2026-05-02T03:37:12.123308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport numpy as np\nimport pandas as pd\nimport json\nimport scipy.sparse as sp\nimport collections\nfrom torch.cuda.amp import autocast, GradScaler\nfrom torch.optim.lr_scheduler import OneCycleLR\nfrom torch.utils.data import Dataset, DataLoader, Subset\nfrom sklearn.model_selection import StratifiedKFold\nimport gc\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T06:21:13.646101Z","iopub.execute_input":"2026-04-25T06:21:13.646477Z","iopub.status.idle":"2026-04-25T06:21:17.791303Z","shell.execute_reply.started":"2026-04-25T06:21:13.646442Z","shell.execute_reply":"2026-04-25T06:21:17.790314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nimport requests\nimport pandas as pd\nimport io\nimport time\nfrom tqdm import tqdm\n\ndef fetch_uniprot_dates_robust(entry_ids):\n    \"\"\"使用 TSV 格式向 UniProt API 請求，並修正官方欄位名稱\"\"\"\n    url = \"https://rest.uniprot.org/uniprotkb/search\"\n    all_dfs = []\n    \n    chunk_size = 100 \n    for i in tqdm(range(0, len(entry_ids), chunk_size), desc=\"Fetching Dates (TSV Mode)\"):\n        chunk = entry_ids[i:i + chunk_size]\n        query = f\"accession:({' OR '.join(chunk)})\"\n        \n        params = {\n            \"query\": query,\n            \"format\": \"tsv\", \n            \"fields\": \"accession,date_created,date_modified\", \n            \"size\": chunk_size\n        }\n        \n        try:\n            response = requests.get(url, params=params, timeout=30)\n            if response.status_code == 200:\n                # 讀取 TSV，UniProt 會回傳 3 個欄位\n                chunk_df = pd.read_csv(io.StringIO(response.text), sep='\\t')\n                all_dfs.append(chunk_df)\n            else:\n                print(f\"\\n⚠️ Chunk {i} 請求失敗: 狀態碼 {response.status_code}\")\n                print(f\"錯誤訊息: {response.text[:200]}\")\n                \n        except Exception as e:\n            print(f\"\\n⚠️ Chunk {i} 發生連線錯誤: {e}\")\n\n        time.sleep(0.2)\n            \n    # 合併所有 DataFrame\n    if all_dfs:\n        final_df = pd.concat(all_dfs, ignore_index=True)\n        # 強制重新命名欄位，方便我們後續的 DataFrame 操作\n        final_df.columns = [\"EntryID\", \"Date_Created\", \"Date_Modified\"]\n        return final_df\n    else:\n        print(\"❌ 抓取失敗，回傳空資料表。\")\n        return pd.DataFrame(columns=[\"EntryID\", \"Date_Created\", \"Date_Modified\"])\n\n# ==========================================\n# 執行區塊\n# ==========================================\ntrain_ids_df = pd.read_csv(\"/kaggle/input/notebooks/t8101349/cafa-5-protein-function-prediction-embeddings/train_ids.csv\")\n\ndate_df = fetch_uniprot_dates_robust(train_ids_df['EntryID'].tolist())\n\n# 轉換時間並存檔\nif not date_df.empty:\n    date_df['Date_Created'] = pd.to_datetime(date_df['Date_Created'])\n    date_df['Date_Modified'] = pd.to_datetime(date_df['Date_Modified'])\n    \n    print(\"\\n✅ 成功抓取！前五筆資料如下：\")\n    print(date_df.head())\n    \n    date_df.to_csv(\"uniprot_dates.csv\", index=False)\n    print(\"\\n💾 已安全儲存為 uniprot_dates.csv！\")\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-04T04:59:07.112855Z","iopub.execute_input":"2026-05-04T04:59:07.113168Z","iopub.status.idle":"2026-05-04T05:21:43.152922Z","shell.execute_reply.started":"2026-05-04T04:59:07.113141Z","shell.execute_reply":"2026-05-04T05:21:43.15183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport scipy.sparse as sp\nimport json\nfrom tqdm import tqdm\nimport os\n\ndef align_all_features_pure_mode(base_path): \n    print(\" 啟動對齊工程 ...\")\n    \n    # ==========================================\n    # 1. 確立絕對座標系 (The Anchor)\n    # ==========================================\n    train_ids_df = pd.read_csv(f\"{base_path}/train_ids.csv\")\n    anchor_ids = train_ids_df['EntryID'].tolist()\n    N_train = len(anchor_ids)\n\n    print(\" 掃描 Train FASTA，驗證特徵排序絕對一致性...\")\n    fasta_path = \"/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_sequences.fasta\"\n    \n    with open(fasta_path, \"r\") as f:\n        idx = 0\n        for line in f:\n            if line.startswith(\">\"):\n                pid = line.strip()[1:].split()[0] \n                if pid != anchor_ids[idx]:\n                    raise ValueError(f\"❌ 致命錯位！FASTA: {pid} vs CSV: {anchor_ids[idx]}\")\n                idx += 1\n                \n    print(f\"   ✅ 完美過關！FASTA 與 CSV 的 {idx} 筆資料 1 對 1 絕對吻合。\")\n    anchor_dict = {pid: i for i, pid in enumerate(anchor_ids)}\n\n    # ==========================================\n    # 2. 浴火重生：自己建立 Y_true 與 專屬標籤清單\n    # ==========================================\n    print(\"\\n 建立 Y_true...\")\n    train_terms = pd.read_csv(\"/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_terms.tsv\", sep=\"\\t\")\n\n    train_terms = train_terms[train_terms['EntryID'].isin(anchor_dict)]\n\n    term_counts = train_terms['term'].value_counts()\n    my_term_list = term_counts.index.tolist()\n    total_labels = len(my_term_list)\n    \n    # 建立 {GO_Term: 欄位索引} 的查表\n    term_to_col = {term: i for i, term in enumerate(my_term_list)}\n    \n    with open(\"my_term_list.json\", \"w\") as f:\n        json.dump(my_term_list, f)\n        \n    print(f\"   ✅ 專屬標籤清單建立完成！共 {total_labels} 個有效 GO Terms。已儲存為 my_term_list.json\")\n    \n    # 填寫稀疏矩陣\n    y_true_aligned = sp.lil_matrix((N_train, total_labels), dtype=np.float32)\n    for _, row in tqdm(train_terms.iterrows(), total=len(train_terms), desc=\"填寫 Y_true\"):\n        pid, term = row['EntryID'], row['term']\n        y_true_aligned[anchor_dict[pid], term_to_col[term]] = 1.0\n            \n    y_true_aligned = y_true_aligned.tocsr()\n    print(f\"   ✅ Y_true 填入完成！形狀: {y_true_aligned.shape}\")\n\n    # ==========================================\n    # 3. 重新建立對齊的 IA 權重 (IA Weights)\n    # ==========================================\n    print(\"\\n 正在對齊 IA 權重...\")\n    ia_df = pd.read_csv(\"/kaggle/input/competitions/cafa-5-protein-function-prediction/IA.txt\", sep=\"\\t\", header=None, names=[\"term\", \"ia\"])\n    ia_dict = dict(zip(ia_df['term'], ia_df['ia']))\n\n    aligned_ia_weights = [ia_dict.get(term, 0.0) for term in my_term_list]\n    ia_tensor = torch.tensor(aligned_ia_weights, dtype=torch.float32)\n    \n    torch.save(ia_tensor, \"my_ia_weights.pt\")\n    print(f\"   ✅ IA Tensor 儲存完成！形狀: {ia_tensor.shape}\")\n\n    # ==========================================\n    # 4. 對齊 Taxonomy 特徵\n    # ==========================================\n    print(\"\\n 正在對齊 Taxonomy 特徵...\")\n    tax_df = pd.read_csv(\"/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv\", sep=\"\\t\")\n    test_tax_df = pd.read_csv(\"/kaggle/input/competitions/cafa-5-protein-function-prediction/Test (Targets)/testsuperset-taxon-list.tsv\", sep=\"\\t\", encoding=\"ISO-8859-1\")\n    \n    taxon_to_idx = {tax_id: i + 1 for i, tax_id in enumerate(test_tax_df['ID'])}\n    taxon_aligned = np.zeros(N_train, dtype=np.int64)\n    tax_lookup = dict(zip(tax_df['EntryID'], tax_df['taxonomyID']))\n    \n    for pid, row_idx in anchor_dict.items():\n        tax_id = tax_lookup.get(pid, -1)\n        taxon_aligned[row_idx] = taxon_to_idx.get(tax_id, 0)\n        \n    print(f\"   ✅ Taxon Array 對齊完成！形狀: {taxon_aligned.shape}\")\n\n    # ==========================================\n    # 5. 儲存大一統檔案\n    # ==========================================\n    print(\"\\n 正在儲存對齊後資料...\")\n    sp.save_npz(\"aligned_Y_true.npz\", y_true_aligned)\n    np.save(\"aligned_taxon.npy\", taxon_aligned)\n    \n    print(\"\\n對齊全數完成！所有檔案皆已完美對齊！\")\n\n\nalign_all_features_pure_mode(\n    base_path=\"/kaggle/input/notebooks/t8101349/cafa-5-protein-function-prediction-embeddings\"\n)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 資源與權重準備 (Data & Weights Preparation)\n# ==========================================\n\n# 模型是 BP, CC, MF 全維度輸出，IA 權重也必須是「三合一」且加入 Offset 對齊。\ndef get_combined_ia_weights_tensor(ia_df, go_maps, ontologies=['MF', 'BP', 'CC']):\n    \"\"\"\n    讀取 IA.txt 與 maps.json，建構與模型輸出維度完美對齊的 IA 權重 Tensor。\n    \n    Args:\n        ia_file_path (str): 官方 IA.txt 的路徑\n        maps_file_path (str): 神級對照表 maps.json 的路徑\n        ontology (str): 要轉換的本體論分類，可選 \"MF\", \"BP\", \"CC\"\n        \n    Returns:\n        torch.Tensor: 形狀為 (num_labels,) 的 1D 權重 Tensor\n    \"\"\"\n    \"\"\"建構嚴格對齊全維度 (MF+BP+CC) 的 IA 權重 Tensor\"\"\"\n    ia_dict = dict(zip(ia_df['term'], ia_df['ia']))\n    total_labels = sum(len(go_maps[ont]) for ont in ontologies)\n    ia_tensor_list = [0.0] * total_labels\n    \n    current_offset = 0\n    missing_count = 0\n    zero_ia_count = 0\n    for ont in ontologies:\n        target_map = go_maps[ont]\n        term_to_col = {v: int(k) for k, v in target_map.items()}\n        \n        for term, col_idx in term_to_col.items():\n            absolute_col_idx = current_offset + col_idx\n            \n            if term in ia_dict:\n                \n                ia_tensor_list[absolute_col_idx] = ia_dict[term]\n                if ia_dict[term] == 0.0:\n                    zero_ia_count += 1\n                    \n            else:\n                missing_count += 1\n        current_offset += len(target_map)\n        \n    ia_weights_tensor = torch.tensor(ia_tensor_list, dtype=torch.float32)\n    \n    print(f\"✅ 全維度 IA 權重對齊完成！Shape: {ia_weights_tensor.shape}\")\n    print(f\" 統計情報：\")\n    print(f\"   ├─ 官方指定 IA 為 0 的節點數: {zero_ia_count}\")\n    print(f\"   └─ 官方檔案缺失的節點數: {missing_count} (已一併棄用)\")\n\n    return ia_weights_tensor\n\n\nIA_TXT_PATH = \"/kaggle/input/competitions/cafa-5-protein-function-prediction/IA.txt\"\nia_df = pd.read_csv(IA_TXT_PATH, sep='\\t', header=None, names=['term', 'ia'])\nprint(f\"成功載入 IA 權重表，共 {len(ia_df)} 筆資料\")\n\n# 建立 {GO_Term: IA_Weight} 的查表字典\nia_dict = dict(zip(ia_df['term'], ia_df['ia']))\n\nwith open(\"my_term_list.json\", 'r') as f:\n    my_term_list = json.load(f) \nprint(f\"成功載入自定義標籤清單，共 {len(my_term_list)} 筆標籤\")\n\naligned_ia_weights = [ia_dict.get(term, 0.0) for term in my_term_list]\n\nia_tensor = torch.tensor(aligned_ia_weights, dtype=torch.float32)\n\nmissing_or_zero_count = aligned_ia_weights.count(0.0)\nprint(f\" 全維度 IA 權重對齊完成！Shape: {ia_tensor.shape}\")\nprint(f\" 統計情報：官方缺失或 IA 本身為 0.0 的節點數: {missing_or_zero_count}\")\n\ntorch.save(ia_tensor, \"ia_weights_combined.pt\")\nprint(\" IA 權重已儲存為 ia_weights_combined.pt\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 資源與權重準備 (Data Preparation)\n# ==========================================\n\ndef load_aligned_cafa_resources():\n    print(\"啟動純淨版 (Pure Sequence) 資源載入...\")\n    ALIGNED_DIR = \"./\" \n    \n    # ---------------------------------------------------------\n    # 1. 載入 序列特徵 (The Anchor 絕對座標基準)\n    # ---------------------------------------------------------\n    print(\"載入序列特徵 (ESM & ProtBERT)...\")\n    esm_emb = np.load(\"/kaggle/input/notebooks/t8101349/cafa-5-protein-function-prediction-embeddings/esm_embeddings_full.npy\", mmap_mode='r')\n    pb_emb = np.load(\"/kaggle/input/notebooks/t8101349/cafa-5-protein-function-prediction-embeddings1-2/protbert_embeddings_full.npy\", mmap_mode='r')\n    \n    # ---------------------------------------------------------\n    # 2. 載入 對齊後的 Ground Truth 與 Taxonomy\n    # ---------------------------------------------------------\n    print(\"載入對齊後的 Y_true 與 Taxonomy...\")\n    y_true_aligned = sp.load_npz(f\"{ALIGNED_DIR}aligned_Y_true.npz\")\n    taxon_aligned = np.load(f\"{ALIGNED_DIR}aligned_taxon.npy\", mmap_mode='r')\n    \n    # ---------------------------------------------------------\n    # 3. 載入 專屬的標籤清單、IA 權重 與 Ancestors\n    # ---------------------------------------------------------\n    print(\"載入專屬標籤清單、IA 權重與圖論 Ancestors...\")\n    with open(f\"{ALIGNED_DIR}my_term_list.json\", \"r\") as f:\n        term_list = json.load(f)\n        \n    ia_weights_tensor = torch.load(f\"{ALIGNED_DIR}my_ia_weights.pt\")\n    \n    with open(\"/kaggle/input/notebooks/t8101349/cafa-5-protein-function-prediction-embeddings2/ancestors.json\", \"r\") as f:\n        ancestors = json.load(f)\n    \n    # ---------------------------------------------------------\n    # 4. 終極防呆斷言\n    # ---------------------------------------------------------\n    N_proteins = esm_emb.shape[0]\n    total_labels = ia_weights_tensor.shape[0] \n    \n    assert pb_emb.shape[0] == N_proteins, \"❌ ESM 與 PB 數量不一致！\"\n    assert y_true_aligned.shape[0] == N_proteins, \"❌ Y_true 數量不一致！\"\n    assert taxon_aligned.shape[0] == N_proteins, \"❌ Taxon 數量不一致！\"\n    assert y_true_aligned.shape[1] == total_labels, \"❌ 標籤矩陣與 IA 維度不一致！\"\n    assert len(term_list) == total_labels, \"❌ 標籤清單的長度與 IA 維度不一致！\"\n\n    print(f\"✅ 載入完美通過！共有 {N_proteins} 筆蛋白質，預測維度: {total_labels}\")\n    return term_list, ancestors, esm_emb, pb_emb, y_true_aligned, taxon_aligned, ia_weights_tensor","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 資料前處理模組 (Data Preprocessing)\n# ==========================================\ndef prepare_taxon_array_top90(base_path, prot_mapping):\n    \"\"\"嚴格對齊測試集的 90 個 Taxonomy ID\"\"\"\n    print(\"執行 Taxonomy 截斷 (Top 90)...\")\n    tax_df = pd.read_csv(f\"/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv\", sep=\"\\t\")\n    test_taxon_df = pd.read_csv(f\"/kaggle/input/competitions/cafa-5-protein-function-prediction/Test (Targets)/testsuperset-taxon-list.tsv\", sep=\"\\t\", encoding=\"ISO-8859-1\")\n        \n    valid_taxons = test_taxon_df['ID'].tolist()\n    taxon_to_idx = {tax_id: i + 1 for i, tax_id in enumerate(valid_taxons)}\n    num_taxons = len(valid_taxons) + 1  # 總共 91 維\n    \n    taxon_arr = np.zeros(len(prot_mapping), dtype=np.int64)\n    tax_dict = dict(zip(tax_df['EntryID'], tax_df['taxonomyID']))\n    \n    for pid, row_idx in prot_mapping.items():\n        tax_id = tax_dict.get(pid, -1)\n        taxon_arr[row_idx] = taxon_to_idx.get(tax_id, 0)\n\n    print(f\"✅ Taxonomy Array 建立完成！)\")\n    return taxon_arr, num_taxons\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_dual_mask_optimized(y_true_aligned_csr, ancestors_dict, term_list, save_path=\"aligned_final_mask.npy\"):\n    \"\"\"\n    純淨版圖論遮罩 (Pure Graph Mask)\n    不再依賴 go_maps，完全根據 ancestors_dict 與單一 term_list 進行父輩未知傳播剔除。\n    \"\"\"\n    print(\"\\n[開始] 建構純圖論遮罩 (Pure Graph Mask) ...\")\n    \n    num_proteins, total_labels = y_true_aligned_csr.shape\n    \n    if os.path.exists(save_path):\n        os.remove(save_path)\n    \n    # ==========================================\n    # 1. 直接建立硬碟映射 (完全不佔 RAM)\n    # ==========================================\n    print(f\"   ├─ 建立硬碟映射檔案 (Memmap): {save_path}\")\n    final_mask = np.memmap(save_path, dtype='float16', mode='w+', shape=(num_proteins, total_labels))\n    \n\n    final_mask[:] = 1.0 \n    final_mask.flush()\n    print(\"   ✅ 初始全域 Mask (全 1.0) 已寫入硬碟。\")\n\n    # ==========================================\n    # 2. 條件機率 Mask (使用 CSC 稀疏查詢)\n    # ==========================================\n    print(\"   ├─ 準備稀疏矩陣 (CSC 格式加速查詢)...\")\n    y_csc = y_true_aligned_csr.tocsc() \n    \n    # 建立快速查找表 (從一維的 term_list 直接建立)\n    term_to_col = {term: i for i, term in enumerate(term_list)}\n        \n    for child_term, col_idx in tqdm(term_to_col.items(), desc=\"   ├─ Build Cond Mask\"):\n        if child_term in ancestors_dict:\n            # 找出所有父節點的絕對欄位索引\n            parent_cols = [term_to_col[p] for p in ancestors_dict[child_term] if p in term_to_col and p != child_term]\n            \n            if parent_cols:\n                # 直接對稀疏矩陣加總\n                parents_sum = np.array(y_csc[:, parent_cols].sum(axis=1)).flatten()\n                \n                # 找出「所有父輩都是 0」的孤立節點行號\n                isolated_idx = np.where(parents_sum == 0)[0]\n                if len(isolated_idx) > 0:\n                    # 直接修改硬碟裡的 memmap，將這些沒有根據的子節點設為 0.0\n                    final_mask[isolated_idx, col_idx] = 0.0\n                    \n    final_mask.flush() \n    \n    print(f\"\\n   預期維度: ({num_proteins}, {total_labels})\")\n    print(f\"   實際維度: {final_mask.shape}\")\n    assert final_mask.shape == (num_proteins, total_labels), \"❌ 致命錯誤：Final Mask 維度不正確！\"\n    \n    # ==========================================\n    # 3. 安全轉移 (釋放寫入權，改為唯讀)\n    # ==========================================\n    del final_mask\n    del y_csc\n    gc.collect()\n\n    readonly_mask = np.memmap(save_path, dtype='float16', mode='r', shape=(num_proteins, total_labels))\n    \n    print(f\"  圖論遮罩建構完畢！\")\n    \n    return readonly_mask\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_valid_training_indices(prot_mapping_dict, seq_ids_csv_path, train_terms_tsv_path):\n    print(\"\\n[資料清洗] 正在過濾無效的蛋白質 (取特徵與標籤的交集)...\")\n    \n    seq_df = pd.read_csv(seq_ids_csv_path)\n    seq_ids = set(seq_df['EntryID'])\n    \n    labels_df = pd.read_csv(train_terms_tsv_path, sep=\"\\t\")\n    label_ids = set(labels_df['EntryID'])\n    \n    mapping_ids = set(prot_mapping_dict.keys())\n    \n    # 核心：取三個集合的交集！\n    valid_ids = seq_ids.intersection(label_ids).intersection(mapping_ids)\n    \n    # 將這些 ID 轉換為神經網路對應的絕對 row index\n    valid_indices = [prot_mapping_dict[pid] for pid in valid_ids]\n    \n    print(f\"   - 總 Mapping 數量: {len(mapping_ids)}\")\n    print(f\"   - 具備序列的數量: {len(seq_ids)}\")\n    print(f\"   - 具備標籤的數量: {len(label_ids)}\")\n    print(f\"   ✅ 最終獲准進入訓練的有效蛋白質數量: {len(valid_indices)}\")\n    \n    return np.array(valid_indices), list(valid_ids)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 資料分割模組 \n# ==========================================\n\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\n\ndef prepare_ultimate_validation_splits(train_ids_list, dates_csv_path, taxon_arr, n_splits=5, vault_cutoff_date='2022-01-01'):\n    print(\"\\n 啟動切分 (Temporal + Aligned Taxon Split)...\")\n    \n    # ==========================================\n    # 1. 讀取基準 IDs 與 時間表\n    # ==========================================\n    train_df = train_ids_list\n    dates_df = pd.read_csv(dates_csv_path)\n    \n    train_df['original_idx'] = np.arange(len(train_df))\n    \n    train_df['taxon_index'] = taxon_arr\n    \n    # ==========================================\n    # 2. 合併時間並處理缺失值\n    # ==========================================\n    merged_df = train_df.merge(dates_df, on='EntryID', how='left')\n    merged_df['Date_Created'] = pd.to_datetime(merged_df['Date_Created']).fillna(pd.to_datetime('1990-01-01'))\n\n    # ==========================================\n    # 3. 定義終極驗證集 (The Vault)\n    # ==========================================\n    vault_mask = (merged_df['Date_Created'] >= vault_cutoff_date) & (merged_df['taxon_index'] > 0)\n    \n    vault_df = merged_df[vault_mask].copy()\n    forge_pool_df = merged_df[~vault_mask].copy()\n    \n    print(f\"   ├─ Vault 規模: {len(vault_df)} 筆 (最新且符合目標物種)\")\n    print(f\"   ├─ Forge Pool 規模: {len(forge_pool_df)} 筆\")\n\n    # ==========================================\n    # 4. 在 Forge Pool 內部進行物種分層 K-Fold\n    # ==========================================\n    taxon_counts = forge_pool_df['taxon_index'].value_counts()\n    valid_taxons = taxon_counts[taxon_counts >= n_splits].index\n\n    def get_stratify_label(row):\n        idx = row['taxon_index']\n        if idx > 0 and idx in valid_taxons:\n            return idx\n        return 0 \n        \n    forge_pool_df['stratify_col'] = forge_pool_df.apply(get_stratify_label, axis=1)\n\n    skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n    folds = []\n    \n    for f_train_idx, f_val_idx in skf.split(forge_pool_df, forge_pool_df['stratify_col']):\n        train_indices = forge_pool_df.iloc[f_train_idx]['original_idx'].values\n        val_indices = forge_pool_df.iloc[f_val_idx]['original_idx'].values\n        folds.append((train_indices, val_indices))\n        \n    print(\"  切分完成！Vault 與 K-Fold 索引皆已完美對齊原始特徵矩陣。\")\n    return vault_df['original_idx'].values, folds","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# PyTorch Dataset\n# ==========================================\n\"\"\"\n參數說明：\nesm_matrix, bert_matrix: numpy.memmap (mmap_mode='r')\ntaxon_arr: numpy 1D array\nnon_exp_pos, non_exp_neg: scipy.sparse.csr_matrix\nlabels_matrix: scipy.sparse.csr_matrix\nmask_matrix: numpy 2D array \nsubset_indices: 該 fold 被選中的 valid_indices (list or 1D array)\n\"\"\"\nclass CAFA_Ultimate_Dataset(Dataset):\n    def __init__(self, esm_matrix, bert_matrix, taxon_arr, labels_matrix, mask_matrix, subset_indices):\n        self.esm = esm_matrix\n        self.bert = bert_matrix\n        self.taxon_arr = taxon_arr\n        self.indices = subset_indices\n        \n        print(\"[Dataset] 正在將稀疏矩陣預取 (Pre-fetch) 為連續記憶體區塊...\")\n        \n        self.y_true_tensor = torch.from_numpy(labels_matrix[self.indices].toarray()).float()\n        \n        if isinstance(mask_matrix, np.ndarray):\n            self.mask_tensor = torch.from_numpy(mask_matrix[self.indices]).float()\n        else:\n            self.mask_tensor = torch.from_numpy(mask_matrix[self.indices].toarray()).float()\n\n    def __len__(self):\n        return len(self.indices)\n\n    def __getitem__(self, i):\n        real_idx = self.indices[i]\n        \n        x_esm = torch.from_numpy(self.esm[real_idx].copy()).float()\n        x_bert = torch.from_numpy(self.bert[real_idx].copy()).float()\n        x_tax = torch.tensor(self.taxon_arr[real_idx], dtype=torch.long)\n\n        y_true = self.y_true_tensor[i]\n        y_mask = self.mask_tensor[i]\n\n        return x_esm, x_bert, x_tax, y_true, y_mask","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CAFA_Ultimate_Dataset_LowRAM(Dataset):\n    def __init__(self, esm_matrix, bert_matrix, taxon_arr, non_exp_pos, non_exp_neg, labels_matrix, mask_matrix, subset_indices):\n        self.esm = esm_matrix\n        self.bert = bert_matrix\n        self.taxon_arr = taxon_arr\n        \n        # 保持為 CSR 稀疏矩陣\n        self.non_exp_pos = non_exp_pos\n        self.non_exp_neg = non_exp_neg\n        self.labels_mat = labels_matrix\n        self.mask_mat = mask_matrix\n        \n        self.indices = subset_indices\n        self.is_mask_sparse = not isinstance(mask_matrix, np.ndarray)\n\n    def __len__(self):\n        return len(self.indices)\n\n    def __getitem__(self, i):\n        real_idx = self.indices[i]\n        \n        x_esm = torch.from_numpy(self.esm[real_idx].copy()).float()\n        x_bert = torch.from_numpy(self.bert[real_idx].copy()).float()\n        x_tax = torch.tensor(self.taxon_arr[real_idx], dtype=torch.long)\n        \n        # 使用 CSR 矩陣的特定方法 getrow() 來提取，會比直接索引稍微快一點\n        # 然後再 toarray()\n        pos_row = self.non_exp_pos.getrow(real_idx).toarray()[0]\n        neg_row = self.non_exp_neg.getrow(real_idx).toarray()[0]\n        x_non_exp = torch.from_numpy(np.stack([pos_row, neg_row])).float()\n        \n        y_true_row = self.labels_mat.getrow(real_idx).toarray()[0]\n        y_true = torch.from_numpy(y_true_row).float()\n        \n        if self.is_mask_sparse:\n            y_mask = torch.from_numpy(self.mask_mat.getrow(real_idx).toarray()[0]).float()\n        else:\n            y_mask = torch.from_numpy(self.mask_mat[real_idx].copy()).float()\n            \n        return x_esm, x_bert, x_tax, x_non_exp, y_true, y_mask","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# [模型] 融合 ESM + T5 + Taxon Emb + Non-Exp CNN\n# ==========================================\nclass CAFA_Ultimate_Model_MultiHead(nn.Module):\n    def __init__(self, esm_dim=1280, bert_dim=1024, num_taxons=91, taxon_emb_dim=64, total_labels=31466): \n        super().__init__()\n        self.taxon_embedding = nn.Embedding(num_embeddings=num_taxons, embedding_dim=taxon_emb_dim)\n\n        fusion_dim = esm_dim + bert_dim + taxon_emb_dim \n        \n        self.shared_mlp = nn.Sequential(\n            nn.Linear(fusion_dim, 2048), \n            nn.LayerNorm(2048),  \n            nn.GELU(), \n            nn.Dropout(0.3),\n            nn.Linear(2048, 1024), \n            nn.LayerNorm(1024),   \n            nn.GELU(), \n            nn.Dropout(0.2)\n        )\n\n        self.unified_head = nn.Sequential(\n            nn.Linear(1024, 1024), \n            nn.GELU(), \n            nn.Dropout(0.1),\n            nn.Linear(1024, total_labels)\n        )\n\n    def forward(self, x_esm, x_bert, x_tax):\n        tax_emb = self.taxon_embedding(x_tax)\n\n        merged = torch.cat([x_esm, x_bert, tax_emb], dim=1)\n\n        shared_features = self.shared_mlp(merged)\n\n        logits = self.unified_head(shared_features)\n        \n        return logits","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# [Loss] 支援雙重 Mask 的 IA Asymmetric Loss\n# ==========================================\n\nclass IA_AsymmetricLossMasked(nn.Module):\n    def __init__(self, ia_weights_tensor, gamma_neg=4, gamma_pos=1, clip=0.05):\n        \"\"\"\n        ia_weights_tensor: shape 為 (num_labels,) 的 1D Tensor，必須與模型輸出維度完全對齊\n        \"\"\"\n        super().__init__()\n        self.gamma_neg = gamma_neg\n        self.gamma_pos = gamma_pos\n        self.clip = clip\n\n        self.register_buffer('ia_weights', ia_weights_tensor)\n\n    def forward(self, inputs, targets, valid_mask):\n        xs_pos = torch.sigmoid(inputs)\n        xs_neg = 1 - xs_pos\n\n        if self.clip is not None and self.clip > 0:\n            xs_neg = (xs_neg + self.clip).clamp(max=1)\n\n        loss_pos = targets * torch.log(xs_pos.clamp(min=1e-8))\n        loss_neg = (1 - targets) * torch.log(xs_neg.clamp(min=1e-8))\n        \n        # 1.  ASL 計算\n        loss = -(loss_pos * (1 - xs_pos)**self.gamma_pos + loss_neg * (xs_neg)**self.gamma_neg)\n        \n        # 2. 乘上 Mask (濾除無實驗證據與條件機率 NaN)\n        masked_loss = loss * valid_mask\n        \n        # 3. 乘上 IA 權重\n        # masked_loss shape: [Batch_size, num_labels]\n        # ia_weights shape:  [num_labels]\n        ia_masked_loss = masked_loss * self.ia_weights\n        \n        # 4. 最終平均 (依照有效 Mask 的數量平均，保持梯度穩定)\n        return ia_masked_loss.sum() / inputs.size(0)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 4. 訓練引擎 (Training and validation Loop)\n# ==========================================\nfrom torch.cuda.amp import autocast, GradScaler\nfrom tqdm import tqdm\n\ndef train_ultimate_stage(model, train_loader, val_loader, criterion, epochs=15, lr=1e-4, stage=1, save_path=\"best_model.pth\"):\n    print(f\" 啟動模型訓練管線 (Stage {stage})... | Epochs: {epochs}, LR: {lr}\")\n    \n    optimizer = torch.optim.AdamW(model.parameters(), lr=lr, weight_decay=1e-4)\n    \n    if stage == 1:\n        # Stage 1: OneCycleLR warm up\n        scheduler = torch.optim.lr_scheduler.OneCycleLR(\n            optimizer, max_lr=lr, steps_per_epoch=len(train_loader), epochs=epochs\n        )\n    else:\n        # Stage 2: CosineAnnealingLR \n        # 讓學習率在整個 stage 期間緩慢降到接近 0 (1e-6)\n        total_steps = epochs * len(train_loader)\n        scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(\n            optimizer, T_max=total_steps, eta_min=1e-6\n        )\n        \n    scaler = torch.amp.GradScaler('cuda')\n    best_val_loss = float('inf') \n    \n    for epoch in range(epochs):\n        # ==========================================\n        #  Training Phase\n        # ==========================================\n        model.train()\n        total_train_loss = 0\n        train_loop = tqdm(train_loader, leave=False, desc=f\"Stage {stage} | Epoch [{epoch+1}/{epochs}] Train\")\n        \n        for i, batch in enumerate(train_loop):\n            x_esm, x_pb, x_tax, y_true, y_mask = [b.cuda() for b in batch]\n\n            if epoch == 0 and i == 0:  \n                print(f\"\\n🔍 [標籤檢查] 一個 Batch 總共有 {y_true.numel()} 個格子\")\n                print(f\"🔍 [標籤檢查] 裡面 1 的數量只有: {y_true.sum().item()}\")\n                        \n            # Label Smoothing\n            y_smoothed = torch.where(y_true == 1.0, 0.95, 0.0)\n            \n            optimizer.zero_grad()\n            with torch.amp.autocast('cuda'):\n                logits = model(x_esm, x_pb, x_tax)\n                loss = criterion(logits, y_smoothed, y_mask)\n\n            if epoch == 0 and i == 0:\n                print(\"\\n\" + \"=\"*40)\n                print(f\"📊 [標籤診斷] Batch 裡面 1 的數量: {y_true.sum().item()} / 總共 {y_true.numel()} 個格子\")\n                print(f\"📈 [Loss 診斷] 初始 Loss 數值: {loss.item():.4f}\")\n                print(f\"🌡️ [Train Logits 診斷] Min={logits.min().item():.2f}, Max={logits.max().item():.2f}\")\n                print(\"=\"*40 + \"\\n\")\n            \n            scaler.scale(loss).backward()\n            scaler.unscale_(optimizer) \n            torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n            \n            scaler.step(optimizer)\n            scaler.update()\n            \n            scheduler.step() \n            \n            total_train_loss += loss.item()\n            train_loop.set_postfix(loss=f\"{loss.item():.4e}\", lr=f\"{scheduler.get_last_lr()[0]:.6f}\")\n            \n        avg_train_loss = total_train_loss / len(train_loader)\n\n        # ==========================================\n        #  Validation Phase\n        # ==========================================\n        model.eval() \n        total_val_loss = 0\n        val_loop = tqdm(val_loader, leave=False, desc=f\"Stage {stage} | Epoch [{epoch+1}/{epochs}] Valid\")\n        \n        with torch.no_grad():\n            for batch in val_loop:\n                x_esm, x_pb, x_tax, y_true, y_mask = [b.cuda() for b in batch]\n                \n                with torch.amp.autocast('cuda'):\n                    logits = model(x_esm, x_pb, x_tax)\n                    loss = criterion(logits, y_true, y_mask)\n                \n                total_val_loss += loss.item()\n                val_loop.set_postfix(val_loss=f\"{loss.item():.4e}\")\n                \n        avg_val_loss = total_val_loss / len(val_loader)\n        \n        # ==========================================\n        # Model Checkpointing\n        # ==========================================\n        print(f\"Stage {stage} | Epoch [{epoch+1}/{epochs}] 結算 | Train Loss: {avg_train_loss:.4e} | Val Loss: {avg_val_loss:.4e}\")\n        \n        if avg_val_loss < best_val_loss:\n            best_val_loss = avg_val_loss\n            if isinstance(model, torch.nn.DataParallel):\n                torch.save(model.module.state_dict(), save_path)\n            else:\n                torch.save(model.state_dict(), save_path)\n            print(f\"   🏆 驗證集 Loss 破紀錄！已儲存模型至 {save_path}\")\n\n    print(f\" Stage {stage} 訓練全數完成！\")\n    \n    # 自動載回最佳權重交接給下一個階段或推論\n    if isinstance(model, torch.nn.DataParallel):\n        model.module.load_state_dict(torch.load(save_path))\n    else:\n        model.load_state_dict(torch.load(save_path))\n        \n    print(f\"🔄 已重新載入最佳權重 (Val Loss: {best_val_loss:.4e})\")\n    \n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T10:46:43.54224Z","iopub.execute_input":"2026-05-01T10:46:43.542709Z","iopub.status.idle":"2026-05-01T10:46:46.760279Z","shell.execute_reply.started":"2026-05-01T10:46:43.542675Z","shell.execute_reply":"2026-05-01T10:46:46.759519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# [後處理] 公式：[Raw + Max(Children) + Min(Parents)] / 3\n# ==========================================\n\nimport collections\nimport numpy as np\nfrom tqdm import tqdm\n\ndef apply_3way_average_formula_fixed(preds_matrix, ancestors_dict, term_list):\n    \"\"\"\n    根據 ancestors.json 執行第二名三向平均後處理 (已修正 Self-Loop)。\n    \n    參數:\n        preds_matrix (np.ndarray): 模型輸出的預測機率矩陣，形狀為 (N_samples, 1500)\n        ancestors_dict (dict): 讀取自 ancestors.json，格式如 {\"GO:子\": [\"GO:父1\", \"GO:父2\", \"GO:子\"]}\n        term_list (list): 1500 個標籤的順序列表，必須與 preds_matrix 的欄位對齊\n        \n    回傳:\n        final_preds (np.ndarray): 修正後的機率矩陣\n    \"\"\"\n    print(\"執行三向平均後處理 (修正 Self-Loop 迴避)...\")\n    num_samples, num_terms = preds_matrix.shape\n    final_preds = np.copy(preds_matrix)\n    term_to_idx = {term: i for i, term in enumerate(term_list)}\n    \n    # ---------------------------------------------------------\n    # 步驟 1: 從 Ancestors 反推建立 Children Dictionary\n    # ---------------------------------------------------------\n    children_dict = collections.defaultdict(list)\n    for child, parents in ancestors_dict.items():\n        for p in parents:\n            if p != child: \n                children_dict[p].append(child)\n            \n    # ---------------------------------------------------------\n    # 步驟 2: 對每個 GO Term 獨立進行推播計算\n    # ---------------------------------------------------------\n    for term, idx in tqdm(term_to_idx.items(), desc=\"Graph Formula Post-Processing\"):\n        \n        # [A] 獲取該 Term 的原始預測機率 (Shape: N_samples,)\n        p_raw = preds_matrix[:, idx] \n        \n        # [B] 計算 Max(Children)\n        p_child_max = np.zeros(num_samples) # 如果沒子節點，預設給 0\n        if term in children_dict:\n            valid_children = [term_to_idx[c] for c in children_dict[term] if c in term_to_idx and c != term]\n            if valid_children:\n                p_child_max = np.max(preds_matrix[:, valid_children], axis=1)\n                \n        # [C] 計算 Min(Parents)\n        p_parent_min = np.ones(num_samples) # 如果沒父節點，預設給 1 (避免拉低平均)\n        if term in ancestors_dict:\n            valid_parents = [term_to_idx[p] for p in ancestors_dict[term] if p in term_to_idx and p != term]\n            if valid_parents:\n                p_parent_min = np.min(preds_matrix[:, valid_parents], axis=1)\n        \n        # ---------------------------------------------------------\n        # 步驟 3: 套用最終公式 \n        # [ 原始預測 + Max(所有子節點預測) + Min(所有父節點預測) ] / 3\n        # ---------------------------------------------------------\n        final_preds[:, idx] = (p_raw + p_child_max + p_parent_min) / 3.0\n        \n    return final_preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T10:46:40.332467Z","iopub.execute_input":"2026-05-01T10:46:40.332834Z","iopub.status.idle":"2026-05-01T10:46:40.351728Z","shell.execute_reply.started":"2026-05-01T10:46:40.332803Z","shell.execute_reply":"2026-05-01T10:46:40.350544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def run_kfold_pipeline(\n    model_class, \n    model_params, \n    esm_matrix,       \n    bert_matrix, \n    taxon_arr, \n    labels_matrix, \n    mask_matrix, \n    kfolds, \n    ultimate_val_idx, \n    criterion_stage1,  \n    criterion_stage2,  \n    term_list,         \n    ancestors          \n):\n\n    fold_val_scores = []\n    \n    # ==========================================\n    # 建立驗證 DataLoader (The Vault)\n    # ==========================================\n    print(\"\\n準備 Vault Dataset...\")\n    ultimate_ds = CAFA_Ultimate_Dataset(\n        esm_matrix, bert_matrix, taxon_arr, \n        labels_matrix, mask_matrix, subset_indices=ultimate_val_idx\n    )\n    ultimate_loader = DataLoader(ultimate_ds, batch_size=64, shuffle=False, num_workers=2, pin_memory=True) \n\n    # ==========================================\n    # 啟動 K-Fold 引擎\n    # ==========================================\n    for fold, (t_idx, v_idx) in enumerate(kfolds):\n        print(f\"\\n\" + \"=\"*40 + f\"\\n        🚩 啟動第 {fold+1} 摺疊訓練\\n\" + \"=\"*40)\n        \n        model = model_class(**model_params)\n        model = nn.DataParallel(model).cuda()\n        \n        train_ds = CAFA_Ultimate_Dataset(\n            esm_matrix, bert_matrix, taxon_arr, \n            labels_matrix, mask_matrix, subset_indices=t_idx\n        )\n        val_ds = CAFA_Ultimate_Dataset(\n            esm_matrix, bert_matrix, taxon_arr, \n            labels_matrix, mask_matrix, subset_indices=v_idx\n        )\n        \n        train_loader = DataLoader(train_ds, batch_size=64, shuffle=True, drop_last=True, num_workers=2, pin_memory=True)\n        val_loader = DataLoader(val_ds, batch_size=64, shuffle=False, num_workers=2, pin_memory=True)\n        \n        # 第一階段 :使用平權 Loss\n        model = train_ultimate_stage(\n            model=model,\n            train_loader=train_loader,\n            val_loader=val_loader,\n            criterion=criterion_stage1, \n            epochs=10, \n            lr=1e-3, \n            stage=1,\n            save_path=f\"model_fold_{fold+1}_stage1.pth\"\n        )\n    \n        # 第二階段：IA 權重\n        best_fold_model = train_ultimate_stage(\n            model=model, \n            train_loader=train_loader,\n            val_loader=val_loader,\n            criterion=criterion_stage2, \n            epochs=5, \n            lr=5e-5, \n            stage=2,\n            save_path=f\"model_fold_{fold+1}_final.pth\"\n        )\n        \n        # ==========================================\n        # 4. 終極保險箱 (Vault) 真實壓力測試\n        # ==========================================\n        print(f\"✅ 第 {fold+1} 摺疊對『Vault』進行真實壓力測試...\")\n        \n        best_fold_model.eval() \n        vault_total_loss = 0.0\n        \n        all_vault_preds = []\n        all_vault_labels = []\n        \n        with torch.no_grad():\n            for batch in ultimate_loader:\n                x_esm, x_bert, x_tax, y_true_batch, y_mask = [b.cuda() for b in batch]\n                \n                with torch.amp.autocast('cuda'):\n                    logits = best_fold_model(x_esm, x_bert, x_tax)\n                    \n                    loss = criterion_stage2(logits, y_true_batch, y_mask)\n                    vault_total_loss += loss.item()\n\n                    all_vault_preds.append(torch.sigmoid(logits).cpu().numpy())\n                    all_vault_labels.append(y_true_batch.cpu().numpy())\n                    \n        avg_vault_loss = vault_total_loss / len(ultimate_loader)\n        \n        # --- 啟動模擬 Submission ---\n        print(\"   正在套用圖論後處理與計算 F_max...\")\n        full_preds = np.vstack(all_vault_preds)\n        full_labels = np.vstack(all_vault_labels)\n        \n        # 3. 套用三向圖論平均\n        full_preds_post = apply_3way_average_formula_fixed(full_preds, ancestors, term_list)\n        \n        # 4. 算出真實的 F_max 分數\n        vault_fmax = calculate_protein_centric_fmax(full_labels, full_preds_post)\n        \n        print(f\"   神經網路 Loss: {avg_vault_loss:.5f}\")\n        print(f\"   Vault F_max: {vault_fmax:.4f}\")\n        \n        # 把 f_max 存起來，這比 val_loss 更有參考價值\n        fold_val_scores.append(vault_fmax)\n        \n        del model, best_fold_model, train_ds, val_ds, train_loader, val_loader\n        gc.collect()\n        torch.cuda.empty_cache()\n\n    print(f\"\\n 所有摺疊訓練完成，5 個最佳模型已就位。\")\n    print(f\"各摺疊測試分數 (Vault Scores):\")\n    for i, score in enumerate(fold_val_scores):\n        print(f\"   🚩 Fold {i+1}: {score:.5f}\")\n        \n    print(f\"\\n 平均分數 (Mean Score): {np.mean(fold_val_scores):.5f}\")\n    return fold_val_scores","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\ndef calculate_protein_centric_fmax(y_true, y_pred_prob, num_thresholds=100):\n    \"\"\"\n    計算以蛋白質為中心的 F_max 分數 (CAFA 5 官方評分標準)\n    \n    參數:\n        y_true (np.ndarray): 真實標籤矩陣，Shape = (num_proteins, num_labels)\n        y_pred_prob (np.ndarray): 模型輸出的機率矩陣，Shape = (num_proteins, num_labels)\n        num_thresholds (int): 掃描的閾值數量 (預設 100 階層)\n        \n    回傳:\n        float: F_max 分數\n    \"\"\"\n    print(f\"  開始計算 Protein-centric F_max (掃描 {num_thresholds} 個閾值)...\")\n    \n\n    assert y_true.shape == y_pred_prob.shape, \"標籤與預測機率的維度不符！\"\n    \n\n    true_counts = y_true.sum(axis=1)\n    true_counts_safe = np.where(true_counts == 0, 1.0, true_counts)\n    \n    # 建立一系列閾值\n    thresholds = np.linspace(0.01, 0.99, num_thresholds)\n    fmax = 0.0\n    \n    # 對每個閾值進行向量化計算\n    for t in thresholds:\n        # 將機率轉為 0 或 1 (布林矩陣)\n        preds_binary = (y_pred_prob >= t)\n        \n        pred_counts = preds_binary.sum(axis=1)\n        pred_counts_safe = np.where(pred_counts == 0, 1.0, pred_counts)\n\n        correct_counts = (preds_binary & (y_true == 1)).sum(axis=1)\n\n        precision_per_protein = correct_counts / pred_counts_safe\n        recall_per_protein = correct_counts / true_counts_safe\n        \n        valid_mask = (true_counts > 0) | (pred_counts > 0)\n        \n        if valid_mask.sum() == 0:\n            continue\n\n        avg_precision = precision_per_protein[valid_mask].mean()\n        avg_recall = recall_per_protein[valid_mask].mean()\n\n        if avg_precision + avg_recall > 0:\n            f1 = 2 * (avg_precision * avg_recall) / (avg_precision + avg_recall)\n        else:\n            f1 = 0.0\n\n        if f1 > fmax:\n            fmax = f1\n            \n    return fmax","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Trainning","metadata":{}},{"cell_type":"code","source":"# ==========================================\n# 6. 主程式執行區塊 (Main Execution)\n# ==========================================\nif __name__ == \"__main__\":\n    print(\"=== 啟動 CAFA 5 訓練 ===\")\n\n    # ---------------------------------------------------------\n    # [步驟 A] 全域設定與測試目標載入\n    # ---------------------------------------------------------\n    BASE_PATH = \"/kaggle/input/competitions/cafa-5-protein-function-prediction\"\n    TRAIN_IDS_CSV_PATH = \"/kaggle/input/notebooks/t8101349/cafa-5-protein-function-prediction-embeddings/train_ids.csv\"\n    IA_WEIGHTS_PATH = \"ia_weights_combined.pt\" \n    \n    print(\"[階段 1] 讀取測試集 Taxonomy 名單...\")\n    test_taxon_path = f\"{BASE_PATH}/Test (Targets)/testsuperset-taxon-list.tsv\"\n    test_taxon_df = pd.read_csv(test_taxon_path, sep=\"\\t\", encoding=\"ISO-8859-1\")\n    TEST_TAXONS = test_taxon_df['ID'].astype(int).tolist()\n    print(f\"✅ 成功讀取，共 {len(TEST_TAXONS)} 個目標物種。\")\n\n    # ---------------------------------------------------------\n    # [步驟 B] 載入已對齊的「大一統資源」\n    # ---------------------------------------------------------\n    print(\"\\n[階段 2] 載入全域統一資源...\")\n    term_list, ancestors, esm_emb, pb_emb, y_true, taxon_arr, _ = load_aligned_cafa_resources()\n    \n    num_taxons = int(taxon_arr.max()) + 1 \n    total_labels = len(term_list)         \n\n    print(f\"✅ term_list 與神經網路輸出頭對齊，總維度: {total_labels}\")\n\n    # ---------------------------------------------------------\n    # [步驟 C] 秒殺級：獲取有效訓練索引 (Valid Indices)\n    # ---------------------------------------------------------\n    print(\"\\n[階段 3] 篩選有效訓練樣本...\")\n    row_sums = np.array(y_true.sum(axis=1)).flatten() \n    \n    valid_indices = np.where(row_sums > 0)[0] \n    \n    train_ids_df = pd.read_csv(TRAIN_IDS_CSV_PATH)\n    print(f\"   - 總庫存: {len(train_ids_df)} 筆\")\n    print(f\"   ✅ 獲准進入訓練的有效蛋白質: {len(valid_indices)} 筆\")\n\n    # ---------------------------------------------------------\n    # [步驟 D] 建立雙重遮罩 (Dual Mask)\n    # ---------------------------------------------------------\n    print(\"\\n[階段 4] 建立雙重遮罩 (Dual Mask)...\")\n    final_mask = build_dual_mask_optimized(y_true, ancestors, term_list)\n\n    # ---------------------------------------------------------\n    # [步驟 E] 戰術切分 (K-Fold & Vault)\n    # ---------------------------------------------------------\n    print(\"\\n[階段 5] 準備 Vault 與 K-Fold 嚴格切分...\")\n\n    vault_idx_raw, kfolds_raw = prepare_ultimate_validation_splits(\n        train_ids_list=train_ids_df,\n        dates_csv_path=\"/kaggle/input/datasets/t8101349/uniprot-dates/uniprot_dates.csv\",\n        taxon_arr=taxon_arr,\n        n_splits=5,\n        vault_cutoff_date='2022-03-01' \n    )\n    \n\n    vault_idx = np.intersect1d(vault_idx_raw, valid_indices)\n    \n    kfolds = []\n    for t_idx_raw, v_idx_raw in kfolds_raw:\n        t_idx_clean = np.intersect1d(t_idx_raw, valid_indices)\n        v_idx_clean = np.intersect1d(v_idx_raw, valid_indices)\n        kfolds.append((t_idx_clean, v_idx_clean))\n        \n    print(f\"   ✅ 切分洗牌完成！乾淨的 Vault 數量: {len(vault_idx)}\")\n\n    # ---------------------------------------------------------\n    # [步驟 F] 實例化終極 Loss 函數\n    # ---------------------------------------------------------\n    print(\"\\n[階段 6] 載入 IA 權重與 Loss Function...\")\n    ia_tensor = torch.load(IA_WEIGHTS_PATH)\n    ia_tensor = ia_tensor / (ia_tensor.mean() + 1e-8) \n    \n    criterion_stage1 = IA_AsymmetricLossMasked(\n        ia_weights_tensor=torch.ones(total_labels, dtype=torch.float32), \n        gamma_neg=4, gamma_pos=1, clip=0.05\n    ).cuda()\n    \n    criterion_stage2 = IA_AsymmetricLossMasked(\n        ia_weights_tensor=ia_tensor, \n        gamma_neg=4, gamma_pos=1, clip=0.05\n    ).cuda()\n\n    # ---------------------------------------------------------\n    # [步驟 G] 啟動 5-Fold 引擎\n    # ---------------------------------------------------------\n    print(\"\\n[階段 7]  啟動 5-Fold 交叉驗證...\")\n\n    # 打包神經網路架構參數\n    model_params = {\n        'num_taxons': num_taxons,\n        'total_labels': total_labels, \n        'esm_dim': 1280,\n        'bert_dim': 1024\n    }\n    \n\n    run_kfold_pipeline(\n        model_class=CAFA_Ultimate_Model_MultiHead,\n        model_params=model_params,\n        esm_matrix=esm_emb,         # 傳入 memmap\n        bert_matrix=pb_emb,         # 傳入 memmap\n        taxon_arr=taxon_arr,\n        labels_matrix=y_true,      \n        mask_matrix=final_mask,       # 傳入 float16 矩陣\n        kfolds=kfolds,\n        ultimate_val_idx=vault_idx,\n        criterion_stage1=criterion_stage1, \n        criterion_stage2=criterion_stage2, \n        term_list=term_list,       \n        ancestors=ancestors         \n    )\n\n\n    print(\"\\n 所有摺疊訓練完成，5 個最佳模型已完成！\")\n    print(\"下個階段：請載入這 5 個 .pth 權重，對 Vault 與 Test Set 進行預測與融合。\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Test prediction","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport scipy.sparse as sp\nimport json\nfrom tqdm import tqdm\nimport collections\n\n# ==========================================\n# 1. 測試集專用 Dataset\n# ==========================================\nclass CAFA_Test_Dataset(Dataset):\n    def __init__(self, esm_matrix, bert_matrix, taxon_arr):\n        self.esm = esm_matrix\n        self.bert = bert_matrix\n        self.taxon_arr = taxon_arr\n        self.length = len(self.taxon_arr)\n\n    def __len__(self):\n        return self.length\n\n    def __getitem__(self, i):\n        x_esm = torch.from_numpy(self.esm[i].copy()).float()\n        x_bert = torch.from_numpy(self.bert[i].copy()).float()\n        x_tax = torch.tensor(self.taxon_arr[i], dtype=torch.long)\n        \n        return x_esm, x_bert, x_tax\n\n# ==========================================\n# 2. Test 資料對齊器 (解析 FASTA)\n# ==========================================\ndef prepare_test_data(base_path, test_taxon_list, test_ids_csv_path, test_esm_path, test_pb_path):\n    print(\" Test 對齊工程...\")\n    \n    test_ids_df = pd.read_csv(test_ids_csv_path)\n    test_ids = test_ids_df['EntryID'].tolist()\n    N_test = len(test_ids)\n    \n    print(\"解析 FASTA 獲取 Taxon，並驗證排序一致性...\")\n    taxon_to_idx = {tax_id: i + 1 for i, tax_id in enumerate(test_taxon_list)}\n    test_taxons = np.zeros(N_test, dtype=np.int64)\n    \n    with open(f\"{base_path}/Test (Targets)/testsuperset.fasta\", \"r\") as f:\n        idx = 0\n        for line in f:\n            if line.startswith(\">\"):\n                parts = line.strip()[1:].split() \n                pid = parts[0]\n                \n                if pid != test_ids[idx]:\n                    raise ValueError(f\"❌ 致命錯位！FASTA 讀到 {pid}，但 CSV 第 {idx} 筆是 {test_ids[idx]}\")\n                \n                tax_id = int(parts[1]) if len(parts) > 1 else 0\n                test_taxons[idx] = taxon_to_idx.get(tax_id, 0)\n                idx += 1\n                \n    print(f\"   ✅ 順序確認無誤！FASTA 與 CSV 的 {idx} 筆資料完美 1 對 1 吻合。\")\n\n\n    print(\"載入 Embeddings...\")\n    test_esm_emb = np.load(test_esm_path, mmap_mode='r')\n    test_pb_emb = np.load(test_pb_path, mmap_mode='r')\n\n\n    return test_ids, test_taxons, test_esm_emb, test_pb_emb","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def batch_3way_average_fixed(batch_preds, ancestors_dict, term_list):\n    \"\"\"\n    對一個 Batch 的預測結果 (batch_size, 26125) 進行三向圖論修正。\n    \"\"\"\n    term_to_idx = {term: i for i, term in enumerate(term_list)}\n    final_preds = np.copy(batch_preds)\n    \n    children_dict = collections.defaultdict(list)\n    for child, parents in ancestors_dict.items():\n        for p in parents:\n            if p != child: \n                children_dict[p].append(child)\n                \n    for term, idx in term_to_idx.items():\n        p_raw = batch_preds[:, idx]\n        \n        p_child_max = np.zeros(batch_preds.shape[0])\n        if term in children_dict:\n            valid_children = [term_to_idx[c] for c in children_dict[term] if c in term_to_idx and c != term]\n            if valid_children:\n                p_child_max = np.max(batch_preds[:, valid_children], axis=1)\n                \n        p_parent_min = np.ones(batch_preds.shape[0])\n        if term in ancestors_dict:\n            valid_parents = [term_to_idx[p] for p in ancestors_dict[term] if p in term_to_idx and p != term]\n            if valid_parents:\n                p_parent_min = np.min(batch_preds[:, valid_parents], axis=1)\n                \n        final_preds[:, idx] = (p_raw + p_child_max + p_parent_min) / 3.0\n        \n    return final_preds","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def generate_cafa5_submission(\n    models, test_loader, test_ids, term_list, ancestors_dict, \n    output_path=\"submission.tsv\", threshold=0.01, top_k=500\n):\n    print(f\"\\n 5-Fold 融合推論與提交生成 (閾值: {threshold}, 最高保留: {top_k})...\")\n    \n    for model in models:\n        model.eval()\n\n    with open(output_path, \"w\") as f:\n        \n        test_loop = tqdm(test_loader, desc=\"Inference & Post-processing\")\n        current_idx = 0\n        \n        with torch.no_grad():\n            for batch in test_loop:\n                x_esm, x_pb, x_tax = [b.cuda() for b in batch]\n\n                if isinstance(models[0], torch.nn.DataParallel):\n                    max_idx = models[0].module.taxon_embedding.num_embeddings - 1\n                else:\n                    max_idx = models[0].taxon_embedding.num_embeddings - 1\n\n                x_tax = x_tax.long() # Embeddings require LongTensors\n                # Force everything to be between 0 and max_idx\n                x_tax = torch.clamp(x_tax, min=0, max=max_idx)\n                \n                batch_size = x_esm.size(0)\n                \n                ensemble_preds = torch.zeros((batch_size, len(term_list)), device='cuda')\n                \n                with torch.amp.autocast('cuda'):\n                    for model in models:\n                        logits = model(x_esm, x_pb, x_tax)\n                        ensemble_preds += torch.sigmoid(logits)\n                        \n                ensemble_preds = ensemble_preds / len(models)\n                \n                ensemble_preds_np = ensemble_preds.cpu().numpy()\n\n                # Batch-Level 圖論後處理 \n                final_preds_np = batch_3way_average_fixed(ensemble_preds_np, ancestors_dict, term_list)\n                \n                for b_idx in range(batch_size):\n                    pid = test_ids[current_idx]\n                    p_scores = final_preds_np[b_idx]\n                    \n                    valid_indices = np.where(p_scores > threshold)[0]\n                    valid_scores = p_scores[valid_indices]\n                    \n                    # 如果超過 Top K，則進行排序並截斷 (保留最高分的 K 個)\n                    if len(valid_scores) > top_k:\n                        sort_idx = np.argsort(valid_scores)[::-1][:top_k]\n                        valid_indices = valid_indices[sort_idx]\n                        valid_scores = valid_scores[sort_idx]\n                        \n                    for v_idx, score in zip(valid_indices, valid_scores):\n                        go_term = term_list[v_idx]\n                        f.write(f\"{pid}\\t{go_term}\\t{score:.3f}\\n\")\n                        \n                    current_idx += 1\n                    \n    print(f\"\\n Submission 已成功生成至：{output_path}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 終極預測與提交階段 (The Final Submission)\n# ==========================================\nprint(\"\\n\" + \"=\"*50)\nprint(\" Submission 生成程序\")\nprint(\"=\"*50)\n\n\nTEST_IDS_PATH = \"/kaggle/input/notebooks/t8101349/cafa-5-protein-function-prediction-test-embed/test_ids.csv\"\nTEST_ESM_PATH = \"/kaggle/input/notebooks/t8101349/cafa-5-protein-function-prediction-test-embed/test_esm_emb_aligned.npy\"\nTEST_PB_PATH = \"/kaggle/input/notebooks/t8101349/cafa-5-protein-function-prediction-test-embed/test_pb_emb_aligned.npy\"\nBASE_PATH = \"/kaggle/input/competitions/cafa-5-protein-function-prediction\"\n\n\ntest_ids, test_taxons, test_esm_emb, test_pb_emb = prepare_test_data(\n    base_path=BASE_PATH,\n    test_taxon_list=TEST_TAXONS, # 請確保你前面有定義這個 list\n    test_ids_csv_path=TEST_IDS_PATH,\n    test_esm_path=TEST_ESM_PATH,\n    test_pb_path=TEST_PB_PATH\n)\n\ntest_ds = CAFA_Test_Dataset(test_esm_emb, test_pb_emb, test_taxons)\ntest_loader = DataLoader(test_ds, batch_size=128, shuffle=False, num_workers=2, pin_memory=True) # 拔除 non-exp 後，Batch size 可以開大一點\n\n\nprint(f\"✅ term_list 建立完成，總共包含 {len(term_list)} 個 GO Terms。\")\n\nmodels = []\nfor i in range(1, 6):\n    model = CAFA_Ultimate_Model_MultiHead(**model_params)\n    model.load_state_dict(torch.load(f\"model_fold_{i}_final.pth\"))\n    model = nn.DataParallel(model).cuda()\n    models.append(model)\n\n# 5. 推論\ngenerate_cafa5_submission(\n    models=models,             # 裝有 5 個最佳 fold 權重的 model list\n    test_loader=test_loader,\n    test_ids=test_ids,         # CSV 的絕對 ID 順序\n    term_list=term_list,\n    ancestors_dict=ancestors,\n    output_path=\"submission.tsv\",\n    threshold=0.01,            # 濾除 1% 以下的微弱訊號，保護檔案大小\n    top_k=500                  # 避免超出官方上傳限制 每個蛋白質最多上傳 500 個標籤\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}