{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-07-13T06:17:06.727024Z","iopub.execute_input":"2026-07-13T06:17:06.727296Z","iopub.status.idle":"2026-07-13T06:17:09.143549Z","shell.execute_reply.started":"2026-07-13T06:17:06.72727Z","shell.execute_reply":"2026-07-13T06:17:09.142571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install biopython -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-13T06:55:48.264599Z","iopub.execute_input":"2026-07-13T06:55:48.264968Z","iopub.status.idle":"2026-07-13T06:55:55.845317Z","shell.execute_reply.started":"2026-07-13T06:55:48.264934Z","shell.execute_reply":"2026-07-13T06:55:55.844365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom itertools import product\nfrom Bio import SeqIO\nfrom scipy.sparse import csr_matrix\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.multioutput import MultiOutputClassifier\nimport joblib\n\n# =====================================================================\n# 1. SETUP THE 3-MER FEATURIZER\n# =====================================================================\n# The 20 standard amino acids used in GOLabeler\nAMINO_ACIDS = \"ACDEFGHIKLMNPQRSTVWY\"\nTHREE_MERS = [\"\".join(p) for p in product(AMINO_ACIDS, repeat=3)]\nTHREE_MER_TO_IDX = {mer: i for i, mer in enumerate(THREE_MERS)}\n\ndef extract_3mer_features(fasta_path):\n    \"\"\"\n    Scans sequences to construct an N x 8000 normalized frequency matrix.\n    Uses sparse representations to drastically reduce RAM usage.\n    \"\"\"\n    print(f\"Parsing 3-mer frequencies from {fasta_path}...\")\n    indptr = [0]\n    indices = []\n    data = []\n    protein_ids = []\n    \n    for record in SeqIO.parse(fasta_path, \"fasta\"):\n        seq = str(record.seq).upper()\n        total_3mers = len(seq) - 2\n        protein_ids.append(record.id)\n        \n        if total_3mers <= 0:\n            indptr.append(len(indices))\n            continue\n            \n        # Count occurrences of valid 3-mers locally\n        local_counts = {}\n        for i in range(total_3mers):\n            mer = seq[i:i+3]\n            if mer in THREE_MER_TO_IDX:\n                mer_idx = THREE_MER_TO_IDX[mer]\n                local_counts[mer_idx] = local_counts.get(mer_idx, 0) + 1\n                \n        # Append rows dynamically for CSR matrix\n        for mer_idx, count in local_counts.items():\n            indices.append(mer_idx)\n            data.append(count / total_3mers) # Normalized frequency\n        indptr.append(len(indices))\n        \n    X_sparse = csr_matrix((data, indices, indptr), shape=(len(protein_ids), 8000), dtype=np.float64)\n    return X_sparse, protein_ids\n\n# =====================================================================\n# 2. TARGET PREPARATION (CAFA 5 MULTI-LABEL)\n# =====================================================================\ndef load_labels(tsv_path, protein_ids, top_n_terms=4000):\n    \"\"\"\n    Builds the binary target matrix Y for the top N most frequent GO terms.\n    \"\"\"\n    print(\"Loading CAFA 5 training annotations...\")\n    train_terms = pd.read_csv(tsv_path, sep=\"\\t\")\n    \n    # Restrict to top N terms due to the massive scale of CAFA 5\n    frequent_terms = train_terms['term'].value_counts().index[:top_n_terms].tolist()\n    term_to_idx = {term: i for i, term in enumerate(frequent_terms)}\n    \n    # Multi-label binary array setup\n    Y = np.zeros((len(protein_ids), top_n_terms), dtype=np.float32)\n    protein_to_idx = {pid: i for i, pid in enumerate(protein_ids)}\n    \n    for row in train_terms.itertuples():\n        if row.EntryID in protein_to_idx and row.term in term_to_idx:\n            p_idx = protein_to_idx[row.EntryID]\n            t_idx = term_to_idx[row.term]\n            Y[p_idx, t_idx] = 1.0\n            \n    return Y, frequent_terms\n\n# =====================================================================\n# 3. MAIN TRAINING PIPELINE\n# =====================================================================\nif __name__ == \"__main__\":\n    # Change these paths to point to your local CAFA 5 dataset files\n    TRAIN_FASTA = \"/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_sequences.fasta\"\n    TRAIN_TERMS = \"/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_terms.tsv\"\n    TEST_FASTA  = \"/kaggle/input/competitions/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta\"\n    \n    # Step 1: Feature Extraction\n    X_train, train_ids = extract_3mer_features(TRAIN_FASTA)\n    \n    # Step 2: Target Generation (Using Top 1500 terms as an example)\n    Y_train, selected_go_terms = load_labels(TRAIN_TERMS, train_ids, top_n_terms=4000)\n    \n    # Step 3: Model Setup\n    # GOLabeler uses classic Logistic Regression.\n    # 'saga' is selected here because it scales efficiently with large, sparse samples.\n    base_lr = LogisticRegression(max_iter=500, solver='liblinear', penalty='l2', C=1.0, tol=1e-3)\n    \n    # Wrap it to predict all binary labels independently across multiple CPU threads\n    clf = MultiOutputClassifier(base_lr, n_jobs=4)\n    \n    # Step 4: Training\n    print(\"Training GOLabeler-style LR-3mer... (This can take a while depending on RAM/CPU)\")\n    clf.fit(X_train, Y_train)\n    \n    # Step 5: Inference on Test Data\n    print(\"Predicting on test data...\")\n    X_test, test_ids = extract_3mer_features(TEST_FASTA)\n    \n    # predict_proba returns a list of arrays (one list entry per GO term)\n    prob_predictions = clf.predict_proba(X_test)\n    \n    # Reassemble probabilities into a single comprehensive matrix (N_samples x N_GO_terms)\n    Y_pred = np.column_stack([pred[:, 1] for pred in prob_predictions])\n    \n    # =====================================================================\n    # 6. EXPORTING TO CAFA 5 SUBMISSION FORMAT\n    # =====================================================================\n    print(\"Formatting submission file...\")\n    submission_records = []\n    \n    for i, protein_id in enumerate(test_ids):\n        for j, go_term in enumerate(selected_go_terms):\n            score = Y_pred[i, j]\n            if score > 0.01: # Avoid saving near-zero predictions to save file space\n                submission_records.append([protein_id, go_term, round(score, 3)])\n                \n    submission_df = pd.DataFrame(submission_records, columns=[\"Protein_ID\", \"GO_Term\", \"Score\"])\n    submission_df.to_csv(\"lr3mer_submission.tsv\", sep=\"\\t\", index=False, header=False)\n    print(\"Saved submission to lr3mer_submission.tsv successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-13T06:57:12.38476Z","iopub.execute_input":"2026-07-13T06:57:12.385565Z","iopub.status.idle":"2026-07-13T06:57:46.093163Z","shell.execute_reply.started":"2026-07-13T06:57:12.385531Z","shell.execute_reply":"2026-07-13T06:57:46.091957Z"}},"outputs":[],"execution_count":null}]}