{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":116062,"databundleVersionId":14875579,"sourceType":"competition"},{"sourceId":41875,"databundleVersionId":5521661,"sourceType":"competition"},{"sourceId":13665986,"sourceType":"datasetVersion","datasetId":8688768},{"sourceId":13680623,"sourceType":"datasetVersion","datasetId":8699749},{"sourceId":268508255,"sourceType":"kernelVersion"},{"sourceId":284979869,"sourceType":"kernelVersion"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q tensorflow","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-13T04:32:50.482964Z","iopub.execute_input":"2025-12-13T04:32:50.483456Z","iopub.status.idle":"2025-12-13T04:32:54.902777Z","shell.execute_reply.started":"2025-12-13T04:32:50.483396Z","shell.execute_reply":"2025-12-13T04:32:54.901954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport os\nimport gc\nfrom sklearn.model_selection import train_test_split\n\n# ==========================================\n# 1. CẤU HÌNH (CONFIG)\n# ==========================================\nclass CFG:\n    RAW_DIR = \"/kaggle/input/cafa-6-protein-function-prediction\"\n    TRAIN_TERMS = os.path.join(RAW_DIR, \"Train/train_terms.tsv\")\n    \n    EMBED_DIR = \"/kaggle/input/cafa-6-t5-embeddings\" \n    TRAIN_EMBEDS = os.path.join(EMBED_DIR, \"train_embeds.npy\")\n    TRAIN_IDS = os.path.join(EMBED_DIR, \"train_ids.npy\")\n    TEST_EMBEDS = os.path.join(EMBED_DIR, \"test_embeds.npy\")\n    TEST_IDS = os.path.join(EMBED_DIR, \"test_ids.npy\")\n\n    IA_PATH = \"/kaggle/input/cafa-6-protein-function-prediction/IA.tsv\" \n    \n    INPUT_DIM = 1024\n    BATCH_SIZE = 256\n    EPOCHS = 30\n    LEARNING_RATE = 1e-4\n    \n    # Giữ nguyên key là BPO/CCO/MFO để dễ quản lý\n    NUM_LABELS = {'BPO': 1500, 'CCO': 800, 'MFO': 800}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T04:32:54.904157Z","iopub.execute_input":"2025-12-13T04:32:54.904464Z","iopub.status.idle":"2025-12-13T04:33:05.18Z","shell.execute_reply.started":"2025-12-13T04:32:54.904413Z","shell.execute_reply":"2025-12-13T04:33:05.179365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 2. HÀM LOAD & PREPARE DỮ LIỆU\n# ==========================================\ndef load_dataset_fixed():\n    print(\"--- 1. LOADING & CLEANING IDs ---\")\n    \n    # A. Load Embeddings\n    X = np.load(CFG.TRAIN_EMBEDS)\n    train_ids_raw = np.load(CFG.TRAIN_IDS)\n    \n    # B. Clean Embeddings IDs (Cắt bỏ 'sp|...|')\n    train_ids_clean = []\n    for uid in train_ids_raw:\n        uid_str = str(uid).strip()\n        if '|' in uid_str:\n            parts = uid_str.split('|')\n            if len(parts) >= 2:\n                train_ids_clean.append(parts[1])\n            else:\n                train_ids_clean.append(uid_str)\n        else:\n            train_ids_clean.append(uid_str)\n            \n    # Ép kiểu string\n    train_ids_clean = [str(x) for x in train_ids_clean]\n    \n    # Tạo Dictionary Map\n    id_to_idx = {uid: i for i, uid in enumerate(train_ids_clean)}\n    \n    # C. Load Labels\n    terms_df = pd.read_csv(CFG.TRAIN_TERMS, sep='\\t')\n    terms_df['EntryID'] = terms_df['EntryID'].astype(str).str.strip()\n    \n    # Check khớp lệnh\n    common = set(train_ids_clean).intersection(set(terms_df['EntryID']))\n    print(f\"   > IDs khớp nhau: {len(common)}\")\n    \n    return X, id_to_idx, terms_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T04:33:05.180686Z","iopub.execute_input":"2025-12-13T04:33:05.18108Z","iopub.status.idle":"2025-12-13T04:33:05.187317Z","shell.execute_reply.started":"2025-12-13T04:33:05.181062Z","shell.execute_reply":"2025-12-13T04:33:05.186605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_data_for_aspect(aspect_name, X, id_to_idx, terms_df):\n    print(f\"\\n--- PREPARING {aspect_name} ---\")\n    \n    # [SỬA LỖI QUAN TRỌNG] Mapping từ tên Aspect sang ký hiệu trong file (P, C, F)\n    # BPO -> P, CCO -> C, MFO -> F\n    aspect_map = {\n        'BPO': 'P',\n        'CCO': 'C',\n        'MFO': 'F'\n    }\n    target_code = aspect_map[aspect_name]\n    print(f\"   > Mapping {aspect_name} -> '{target_code}' in dataset\")\n    \n    # 1. Filter aspect\n    df_aspect = terms_df[terms_df['aspect'] == target_code].copy()\n    \n    if len(df_aspect) == 0:\n        raise ValueError(f\"Không tìm thấy dữ liệu cho code '{target_code}'. Kiểm tra lại file terms!\")\n\n    # 2. Top-K Terms\n    top_k = CFG.NUM_LABELS[aspect_name]\n    top_terms = df_aspect['term'].value_counts().head(top_k).index.tolist()\n    df_aspect = df_aspect[df_aspect['term'].isin(top_terms)]\n    \n    # 3. Pivot Table (Tạo One-Hot)\n    print(\"   > Creating pivot table (this may take a moment)...\")\n    df_aspect['val'] = 1\n    label_matrix = df_aspect.pivot_table(index='EntryID', columns='term', values='val', fill_value=0)\n    \n    # Ép kiểu index\n    label_ids = [str(x) for x in label_matrix.index]\n    \n    # 4. Tìm giao thoa\n    valid_ids = list(set(label_ids).intersection(set(id_to_idx.keys())))\n    \n    print(f\"   > Label Matrix Rows: {len(label_matrix)}\")\n    print(f\"   > Valid IDs (Intersection): {len(valid_ids)}\")\n    \n    if len(valid_ids) == 0:\n        raise ValueError(f\"Không tìm thấy protein nào cho aspect {aspect_name}!\")\n\n    # 5. Tạo dữ liệu train\n    # Lấy vector embedding tương ứng\n    indices = [id_to_idx[uid] for uid in valid_ids]\n    X_sub = X[indices]\n    \n    # Lấy nhãn tương ứng (cần reindex label_matrix theo valid_ids để đảm bảo thứ tự)\n    # Lưu ý: label_matrix.loc[valid_ids] sẽ tự sắp xếp theo thứ tự valid_ids\n    Y_sub = label_matrix.loc[valid_ids].values\n    \n    print(f\"   > Final Train Data: X={X_sub.shape}, Y={Y_sub.shape}\")\n    return X_sub, Y_sub, label_matrix.columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T04:33:05.189242Z","iopub.execute_input":"2025-12-13T04:33:05.189663Z","iopub.status.idle":"2025-12-13T04:33:05.207976Z","shell.execute_reply.started":"2025-12-13T04:33:05.189633Z","shell.execute_reply":"2025-12-13T04:33:05.207333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# 2.1 HELPER FUNCTIONS CHO IA LOSS\n# ==========================================\ndef load_ia_weights(ia_path):\n    \"\"\"Đọc file IA.tsv và trả về dictionary {GO_ID: Score}\"\"\"\n    print(f\"Loading IA weights from: {ia_path}\")\n    \n    # Đọc file TSV (không có header hoặc header tùy file, thường là cot 1: Term, cot 2: IA)\n    # Giả định file có 2 cột: TermID và IA_Score\n    try:\n        df_ia = pd.read_csv(ia_path, sep='\\t', header=None, names=['term', 'ia'])\n        # Chuyển về dict để tra cứu cho nhanh\n        return dict(zip(df_ia['term'], df_ia['ia']))\n    except Exception as e:\n        print(f\"⚠️ Error reading IA file: {e}\")\n        return {}\n\ndef get_weighted_loss(class_weights):\n    \"\"\"\n    Tạo hàm loss tùy chỉnh: Weighted Binary Crossentropy.\n    class_weights: Numpy array chứa trọng số IA tương ứng với từng label.\n    \"\"\"\n    # Chuyển weights thành Tensor hằng số (shape: 1, num_classes)\n    weights_tensor = tf.constant(class_weights[None, :], dtype=tf.float32)\n    \n    def weighted_loss(y_true, y_pred):\n        # Tránh lỗi log(0)\n        epsilon = tf.keras.backend.epsilon()\n        y_pred = tf.clip_by_value(y_pred, epsilon, 1. - epsilon)\n        \n        # Công thức Binary Crossentropy chuẩn\n        bce = -(y_true * tf.math.log(y_pred) + (1 - y_true) * tf.math.log(1 - y_pred))\n        \n        # Nhân với trọng số IA\n        weighted_bce = bce * weights_tensor\n        \n        # Trả về trung bình loss\n        return tf.reduce_mean(weighted_bce)\n    \n    return weighted_loss","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T04:33:05.208618Z","iopub.execute_input":"2025-12-13T04:33:05.208786Z","iopub.status.idle":"2025-12-13T04:33:05.219456Z","shell.execute_reply.started":"2025-12-13T04:33:05.208772Z","shell.execute_reply":"2025-12-13T04:33:05.218687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_model(num_classes, weights_array=None):\n    model = tf.keras.Sequential([\n        tf.keras.layers.Input(shape=(CFG.INPUT_DIM,)),\n        tf.keras.layers.Dense(512, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.2),\n        tf.keras.layers.Dense(512, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.2),\n        tf.keras.layers.Dense(num_classes, activation='sigmoid')\n    ])\n    \n    # --- PHẦN THAY ĐỔI ---\n    if weights_array is not None:\n        # Nếu có weights, dùng Custom Loss\n        loss_fn = get_weighted_loss(weights_array)\n        #print(\"   > Using Weighted Binary Crossentropy (IA based)\")\n    else:\n        # Nếu không (hoặc lỗi), dùng mặc định\n        loss_fn = 'binary_crossentropy'\n        #print(\"   > Using Standard Binary Crossentropy\")\n        \n    model.compile(optimizer='adam', loss=loss_fn, metrics=['binary_accuracy'])\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T04:33:05.220195Z","iopub.execute_input":"2025-12-13T04:33:05.220508Z","iopub.status.idle":"2025-12-13T04:33:05.234612Z","shell.execute_reply.started":"2025-12-13T04:33:05.220481Z","shell.execute_reply":"2025-12-13T04:33:05.233802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# A. Load Global Data\nX_global, id_to_idx, terms_df = load_dataset_fixed()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T04:33:05.235247Z","iopub.execute_input":"2025-12-13T04:33:05.235488Z","iopub.status.idle":"2025-12-13T04:33:10.832028Z","shell.execute_reply.started":"2025-12-13T04:33:05.235468Z","shell.execute_reply":"2025-12-13T04:33:10.831204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# B. Load Test Data\nprint(\"\\nLoading Test Data...\")\nif os.path.exists(CFG.TEST_EMBEDS):\n    X_test = np.load(CFG.TEST_EMBEDS)\n    test_ids = np.load(CFG.TEST_IDS)\n    print(f\"Test loaded: {X_test.shape}\")\nelse:\n    print(\"WARNING: Test files not found. Creating dummy test data just to run code.\")\n    # Dummy để code không crash nếu bạn chưa có file test\n    X_test = np.zeros((10, CFG.INPUT_DIM)) \n    test_ids = np.array([f\"TEST_{i}\" for i in range(10)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T04:33:10.832911Z","iopub.execute_input":"2025-12-13T04:33:10.833213Z","iopub.status.idle":"2025-12-13T04:33:21.843586Z","shell.execute_reply.started":"2025-12-13T04:33:10.833187Z","shell.execute_reply":"2025-12-13T04:33:21.842909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submissions = []\n\nglobal_ia_weights = load_ia_weights(CFG.IA_PATH)\n\n# C. Loop Training\nfor aspect in ['BPO', 'CCO', 'MFO']:\n    try:\n        # Get Data\n        X_train, Y_train, target_terms = get_data_for_aspect(aspect, X_global, id_to_idx, terms_df)\n        \n        # Split\n        x_tr, x_val, y_tr, y_val = train_test_split(X_train, Y_train, test_size=0.1, random_state=42)\n\n        print(f\"Preparing weights for {len(target_terms)} terms...\")\n        weights_list = []\n        for term in target_terms:\n            # Nếu term có trong IA file thì lấy, không thì mặc định là 1.0\n            # Mẹo: Có thể lấy mặc định là trung bình IA hoặc 1.0\n            w = global_ia_weights.get(term, 1.0) \n            weights_list.append(w)\n        \n        weights_array = np.array(weights_list, dtype=np.float32)\n        \n        # Train\n        print(f\"Training {aspect} model...\")\n        model = create_model(len(target_terms), weights_array=weights_array) \n        \n        early_stop = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=3, restore_best_weights=True)\n        \n        model.fit(x_tr, y_tr, validation_data=(x_val, y_val), \n                  epochs=CFG.EPOCHS, batch_size=CFG.BATCH_SIZE, \n                  callbacks=[early_stop], verbose=1)\n        \n        # Predict\n        print(f\"Predicting {aspect}...\")\n        preds = model.predict(X_test, batch_size=CFG.BATCH_SIZE, verbose=1)\n        \n        # Format Result\n        df_pred = pd.DataFrame(preds, columns=target_terms)\n        df_pred['EntryID'] = test_ids\n        melted = df_pred.melt(id_vars='EntryID', var_name='term', value_name='score')\n        melted = melted[melted['score'] > 0.006] # Chỉ lấy điểm > 0.005\n        submissions.append(melted)\n        \n        # Cleanup\n        del model, x_tr, y_tr, df_pred, melted, X_train, Y_train\n        tf.keras.backend.clear_session()\n        gc.collect()\n        \n    except Exception as e:\n        print(f\"Error processing {aspect}: {e}\")\n        import traceback\n        traceback.print_exc()\n        continue","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T04:33:21.844319Z","iopub.execute_input":"2025-12-13T04:33:21.844523Z","iopub.status.idle":"2025-12-13T04:36:07.723885Z","shell.execute_reply.started":"2025-12-13T04:33:21.844508Z","shell.execute_reply":"2025-12-13T04:36:07.723243Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # ==========================================\n# # D. SAVE SUBMISSION (TSV FORMAT)\n# # ==========================================\n# print(\"\\nSaving final submission...\")\n\n# if len(submissions) > 0:\n#     # 1. Gộp tất cả các aspect lại\n#     final_df = pd.concat(submissions, axis=0, ignore_index=True)\n    \n#     # 2. Sắp xếp lại cho đẹp (Protein ID tăng dần, Score giảm dần) - Optional\n#     # Giúp file dễ nhìn hơn nếu bạn mở ra check\n#     print(\"Sorting data...\")\n#     final_df.sort_values(by=['EntryID', 'score'], ascending=[True, False], inplace=True)\n    \n#     # 3. Lưu file .tsv\n#     output_filename = \"submission.tsv\"\n    \n#     print(f\"Writing to {output_filename}...\")\n#     # sep='\\t': Dùng tab làm dấu phân cách\n#     # index=False: Không lưu cột số thứ tự dòng (0,1,2...)\n#     # float_format='%.3f': Làm tròn 3 số thập phân để giảm dung lượng file (Optional)\n#     final_df.to_csv(output_filename, sep='\\t', index=False, float_format='%.3f', header=False) \n    \n#     print(f\"Done! Saved {len(final_df)} rows to {output_filename}\")\n#     print(final_df.head())\n\n# else:\n#     print(\"No predictions made.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T04:36:07.728393Z","iopub.execute_input":"2025-12-13T04:36:07.728649Z","iopub.status.idle":"2025-12-13T04:36:07.732907Z","shell.execute_reply.started":"2025-12-13T04:36:07.72863Z","shell.execute_reply":"2025-12-13T04:36:07.732083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# E. HELPER FUNCTIONS CHO POST-PROCESSING\n# ==========================================\nfrom collections import defaultdict\n\ndef parse_obo(obo_file):\n    \"\"\"Đọc file .obo để hiểu quan hệ cha-con giữa các GO terms\"\"\"\n    print(f\"Loading OBO file: {obo_file}\")\n    parents = defaultdict(list)\n    children = defaultdict(list)\n    term_id = None\n    \n    with open(obo_file, 'r') as f:\n        for line in f:\n            line = line.strip()\n            if line.startswith('id: '):\n                term_id = line.split('id: ')[1]\n            elif line.startswith('is_a: ') and term_id:\n                parent_id = line.split('is_a: ')[1].split(' ! ')[0]\n                parents[term_id].append(parent_id)\n                children[parent_id].append(term_id)\n    return parents, children\n\ndef get_descendants(term, children_map, cache=None):\n    \"\"\"Tìm tất cả các con cháu của một GO term (để lan truyền tính chất NOT)\"\"\"\n    if cache is None: cache = {}\n    if term in cache: return cache[term]\n    \n    descendants = set()\n    stack = [term]\n    while stack:\n        current = stack.pop()\n        if current in children_map:\n            for child in children_map[current]:\n                if child not in descendants:\n                    descendants.add(child)\n                    stack.append(child)\n    \n    cache[term] = descendants\n    return descendants","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T04:36:07.733756Z","iopub.execute_input":"2025-12-13T04:36:07.733962Z","iopub.status.idle":"2025-12-13T04:36:07.749705Z","shell.execute_reply.started":"2025-12-13T04:36:07.733947Z","shell.execute_reply":"2025-12-13T04:36:07.748681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, gc\nimport pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\n\n# ==========================================\n# F. POST-PROCESSING (DIRECT MEMORY PROCESSING)\n# ==========================================\nprint(\"\\n[START] Post-processing pipeline (Direct from RAM)...\")\n\nOBO_PATH = \"/kaggle/input/cafa-6-protein-function-prediction/Train/go-basic.obo\"\nGOA_PATH = \"/kaggle/input/protein-go-annotations/goa_uniprot_all.csv\"\n\n# Kiểm tra xem list submissions có dữ liệu không\nif 'submissions' in globals() and len(submissions) > 0 and os.path.exists(OBO_PATH) and os.path.exists(GOA_PATH):\n    \n    # 1. GỘP BIẾN SUBMISSIONS TỪ RAM\n    print(\"[1/6] Consolidating predictions from memory...\")\n    # Gộp tất cả các mảnh (aspects) lại thành 1 DataFrame\n    sub = pd.concat(submissions, axis=0, ignore_index=True)\n    \n    # [QUAN TRỌNG] Đổi tên cột khớp với logic xử lý bên dưới\n    # Code train sinh ra: ['EntryID', 'term', 'score']\n    # Code xử lý cần: ['protein_id', 'go_term', 'score']\n    sub.rename(columns={'EntryID': 'protein_id', 'term': 'go_term'}, inplace=True)\n    \n    # Xóa biến submissions gốc để giải phóng RAM ngay lập tức\n    del submissions\n    gc.collect()\n    \n    # Lấy danh sách ID mục tiêu\n    target_ids = set(sub['protein_id'])\n    print(f\"   > Focusing on {len(target_ids)} target proteins.\")\n\n    # 2. PARSE OBO\n    print(\"[2/6] Parsing Ontology Tree...\")\n    parents_map, children_map = parse_obo(OBO_PATH)\n\n    # 3. ĐỌC GOA FILE THEO CHUNK\n    print(\"[3/6] Scanning GOA file in chunks...\")\n    \n    ground_truth_pairs = set() \n    negative_dict = {} \n\n    chunk_size = 1_000_000 \n    reader = pd.read_csv(GOA_PATH, chunksize=chunk_size, usecols=['protein_id', 'go_term', 'qualifier'])\n\n    for chunk in tqdm(reader, desc=\"Processing GOA Chunks\"):\n        # Chỉ giữ lại protein nằm trong target_ids\n        chunk = chunk[chunk['protein_id'].isin(target_ids)]\n        if chunk.empty: continue\n        \n        is_neg = chunk['qualifier'].str.contains('NOT', na=False)\n        \n        # Gom Negative\n        neg_chunk = chunk[is_neg]\n        for pid, term in zip(neg_chunk['protein_id'], neg_chunk['go_term']):\n            if pid not in negative_dict: negative_dict[pid] = set()\n            negative_dict[pid].add(term)\n            \n        # Gom Positive (Ground Truth)\n        pos_chunk = chunk[~is_neg]\n        ground_truth_pairs.update(zip(pos_chunk['protein_id'], pos_chunk['go_term']))\n    \n    print(f\"   > Found {len(ground_truth_pairs)} Ground Truth pairs.\")\n    print(f\"   > Found negative evidence for {len(negative_dict)} proteins.\")\n    \n    del reader, chunk\n    gc.collect()\n\n    # 4. NEGATIVE PROPAGATION\n    print(\"[4/6] Propagating Negative Terms...\")\n    blacklist_pairs = set()\n    cache_descendants = {}\n    \n    for pid, terms in tqdm(negative_dict.items(), desc=\"Propagating\"):\n        all_neg_terms = set(terms)\n        for t in list(all_neg_terms):\n            descendants = get_descendants(t, children_map, cache_descendants)\n            all_neg_terms.update(descendants)\n            \n        for t in all_neg_terms:\n            blacklist_pairs.add((pid, t))\n            \n    print(f\"   > Final Blacklist size: {len(blacklist_pairs)} pairs.\")\n    del negative_dict, cache_descendants, parents_map, children_map\n    gc.collect()\n\n    # 5. LỌC SUBMISSION (TRỰC TIẾP TRÊN BIẾN SUB)\n    print(\"[5/6] Refining Submission...\")\n    print(f\"   > Original rows: {len(sub)}\")\n    \n    # Tạo mask lọc (nhanh hơn drop)\n    # Logic: Giữ lại nếu (KHÔNG nằm trong blacklist) VÀ (KHÔNG nằm trong Ground Truth)\n    # Ground truth sẽ được gộp vào sau với điểm 1.0\n    valid_mask = []\n    \n    # Chuyển columns sang list/numpy để zip nhanh hơn truy cập dataframe\n    pids = sub['protein_id'].values\n    terms = sub['go_term'].values\n    \n    for pid, term in zip(pids, terms):\n        pair = (pid, term)\n        if (pair in blacklist_pairs) or (pair in ground_truth_pairs):\n            valid_mask.append(False)\n        else:\n            valid_mask.append(True)\n            \n    sub = sub[valid_mask]\n    print(f\"   > Rows after filtering: {len(sub)}\")\n    \n    del blacklist_pairs, valid_mask, pids, terms\n    gc.collect()\n\n    # 6. MERGE VÀ LƯU FILE\n    print(\"[6/6] Merging Ground Truth and Saving...\")\n    \n    if len(ground_truth_pairs) > 0:\n        gt_df = pd.DataFrame(list(ground_truth_pairs), columns=['protein_id', 'go_term'])\n        gt_df['score'] = 1.0\n        final_submission = pd.concat([gt_df, sub], axis=0, ignore_index=True)\n    else:\n        final_submission = sub\n\n    # Lưu file kết quả cuối cùng\n    output_filename = \"submission.tsv\"\n    # Sắp xếp nhẹ để dễ nhìn (có thể bỏ qua nếu RAM quá căng)\n    # final_submission.sort_values(by=['protein_id', 'score'], ascending=[True, False], inplace=True)\n    \n    final_submission.to_csv(output_filename, sep='\\t', index=False, header=False, float_format='%.3f')\n    \n    print(f\"[SUCCESS] Saved {len(final_submission)} rows to {output_filename}\")\n    \n    # Dọn dẹp sạch sẽ\n    del sub, ground_truth_pairs, final_submission\n    gc.collect()\n\nelse:\n    print(\"[ERROR] 'submissions' list is empty/missing OR dataset files not found.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-13T04:36:07.750351Z","iopub.execute_input":"2025-12-13T04:36:07.750547Z","iopub.status.idle":"2025-12-13T04:41:26.224274Z","shell.execute_reply.started":"2025-12-13T04:36:07.750531Z","shell.execute_reply":"2025-12-13T04:41:26.223549Z"}},"outputs":[],"execution_count":null}]}