{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-07-20T04:09:05.54113Z","iopub.execute_input":"2026-07-20T04:09:05.541449Z","iopub.status.idle":"2026-07-20T04:09:08.186053Z","shell.execute_reply.started":"2026-07-20T04:09:05.541421Z","shell.execute_reply":"2026-07-20T04:09:08.184927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install biopython -q\nimport pandas as pd\nimport numpy as np\nfrom Bio import SeqIO","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T04:09:08.188285Z","iopub.execute_input":"2026-07-20T04:09:08.188961Z","iopub.status.idle":"2026-07-20T04:09:16.033073Z","shell.execute_reply.started":"2026-07-20T04:09:08.188925Z","shell.execute_reply":"2026-07-20T04:09:16.031967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install the correct NCBI BLAST+ suite\n!apt-get update && apt-get install -y ncbi-blast+","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T04:09:16.034542Z","iopub.execute_input":"2026-07-20T04:09:16.03502Z","iopub.status.idle":"2026-07-20T04:09:34.717805Z","shell.execute_reply.started":"2026-07-20T04:09:16.03497Z","shell.execute_reply":"2026-07-20T04:09:34.716588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load your custom parquet file\nparquet_path = '/kaggle/input/datasets/pradippokhrel77/test-interpro/test-00000-of-00001.parquet'\ndf_all = pd.read_parquet(parquet_path)\n\n# Drop any rows that don't have sequences\ndf_all = df_all.dropna(subset=['sequence'])\n\n# Split into reference (has known GO IDs) and queries (missing GO IDs or targets)\n# Note: Adjust the condition if you want to split it differently (e.g., random 80/20 split)\ndf_ref = df_all[df_all['go_ids'].notna()]\ndf_query = df_all[df_all['go_ids'].isna()]\n\n# If no rows are explicitly missing GO IDs, let's just do a quick 80/20 train/test split for your experiment:\nif len(df_query) == 0:\n    df_ref = df_all.sample(frac=0.8, random_state=42)\n    df_query = df_all.drop(df_ref.index)\n\n# 1. Write the Custom Reference Database FASTA\nwith open('custom_ref_database.fasta', 'w') as f:\n    for _, row in df_ref.iterrows():\n        f.write(f\">{row['protein_id']}\\n{row['sequence']}\\n\")\n\n# 2. Write the Custom Query FASTA\nwith open('custom_query_sequences.fasta', 'w') as f:\n    for _, row in df_query.iterrows():\n        f.write(f\">{row['protein_id']}\\n{row['sequence']}\\n\")\n\nprint(f\"Custom Database: {len(df_ref)} sequences.\")\nprint(f\"Custom Queries: {len(df_query)} sequences.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T04:09:34.719382Z","iopub.execute_input":"2026-07-20T04:09:34.719929Z","iopub.status.idle":"2026-07-20T04:09:35.984294Z","shell.execute_reply.started":"2026-07-20T04:09:34.719763Z","shell.execute_reply":"2026-07-20T04:09:35.98304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Build the BLAST database using ONLY your custom reference sequences\n!makeblastdb -in custom_ref_database.fasta -dbtype prot -out custom_blast_db","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T04:09:35.985573Z","iopub.execute_input":"2026-07-20T04:09:35.986111Z","iopub.status.idle":"2026-07-20T04:09:36.422313Z","shell.execute_reply.started":"2026-07-20T04:09:35.986079Z","shell.execute_reply":"2026-07-20T04:09:36.42106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import subprocess\nfrom tqdm.notebook import tqdm\nfrom Bio import SeqIO\n\n# 1. Read your query sequences from the split file\nquery_records = list(SeqIO.parse(\"custom_query_sequences.fasta\", \"fasta\"))\ntotal_seqs = len(query_records)\n\n# 2. Clear or initialize the output file\noutput_file = \"blast_results.txt\"\nwith open(output_file, \"w\") as f:\n    pass\n\nprint(f\"Aligning {total_seqs} sequences against custom_blast_db...\")\n\n# 3. Run BLAST sequence-by-sequence with a progress bar\nfor record in tqdm(query_records, desc=\"BLAST Processing\"):\n    # Create a temporary fasta string for the single sequence\n    query_str = f\">{record.id}\\n{record.seq}\\n\"\n    \n    # Run the blastp command for this single protein\n    process = subprocess.Popen(\n        [\n            \"blastp\", \n            \"-db\", \"custom_blast_db\", \n            \"-outfmt\", \"6 qseqid sseqid pident length evalue bitscore\", \n            \"-max_target_seqs\", \"10\", \n            \"-num_threads\", \"4\"\n        ],\n        stdin=subprocess.PIPE, \n        stdout=subprocess.PIPE, \n        stderr=subprocess.PIPE,\n        text=True\n    )\n    \n    # Send the sequence to stdin and capture output\n    stdout, stderr = process.communicate(input=query_str)\n    \n    # Append the alignment result line straight to your master text file\n    if stdout:\n        with open(output_file, \"a\") as f:\n            f.write(stdout)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T04:09:36.424558Z","iopub.execute_input":"2026-07-20T04:09:36.424949Z","iopub.status.idle":"2026-07-20T04:24:45.172278Z","shell.execute_reply.started":"2026-07-20T04:09:36.424911Z","shell.execute_reply":"2026-07-20T04:24:45.171195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a lookup dictionary from your custom reference DataFrame\n# Assuming go_ids is stored as a list or string representation of a list\nimport ast\n\nprotein_to_go = {}\nfor _, row in df_ref.iterrows():\n    go_data = row['go_ids']\n    if isinstance(go_data, str):\n        # Convert string representation of list \"['GO:0003674', ...]\" to an actual list\n        try:\n            go_list = ast.literal_eval(go_data)\n        except:\n            go_list = [go_data]\n    else:\n        go_list = list(go_data)\n    protein_to_go[row['protein_id']] = set(go_list)\n\n# Simplified k-NN prediction function using our local python variables\ndef predict_custom_knn(blast_results_path, protein_to_go, k=5):\n    columns = ['qseqid', 'sseqid', 'pident', 'length', 'evalue', 'bitscore']\n    blast_df = pd.read_csv(blast_results_path, sep='\\t', names=columns)\n    \n    predictions = {}\n    for qseqid, group in blast_df.groupby('qseqid'):\n        top_k = group.nsmallest(k, 'evalue')\n        go_term_scores = {}\n        total_weight = 0\n        \n        for _, row in top_k.iterrows():\n            neighbor = row['sseqid']\n            weight = row['pident'] / 100.0  \n            \n            if neighbor in protein_to_go:\n                total_weight += weight\n                for term in protein_to_go[neighbor]:\n                    go_term_scores[term] = go_term_scores.get(term, 0) + weight\n                    \n        if total_weight > 0:\n            predictions[qseqid] = {term: score / total_weight for term, score in go_term_scores.items()}\n            \n    return predictions\n\n# Run predictions\npreds = predict_custom_knn('blast_results.txt', protein_to_go, k=5)\n\n# Format to submission format\nsubmission_data = []\nfor prot_id, terms in preds.items():\n    for term, score in terms.items():\n        submission_data.append([prot_id, term, round(score, 3)])\n\nsub_df = pd.DataFrame(submission_data, columns=['Protein Id', 'GO Term ID', 'Prediction Score'])\nsub_df.to_csv('submission.tsv', sep='\\t', index=False, header=False)\nprint(f\"Generated clean custom submission with {len(sub_df)} records!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T04:24:45.17498Z","iopub.execute_input":"2026-07-20T04:24:45.175398Z","iopub.status.idle":"2026-07-20T04:24:49.344434Z","shell.execute_reply.started":"2026-07-20T04:24:45.175369Z","shell.execute_reply":"2026-07-20T04:24:49.343286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_after_blast_knn=pd.read_csv(\"submission.tsv\",sep=\"\\t\")\ndf_after_blast_knn.iloc[:30]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T04:25:28.122532Z","iopub.execute_input":"2026-07-20T04:25:28.12303Z","iopub.status.idle":"2026-07-20T04:25:28.201084Z","shell.execute_reply.started":"2026-07-20T04:25:28.122999Z","shell.execute_reply":"2026-07-20T04:25:28.200174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install cafaeval -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T04:25:55.321662Z","iopub.execute_input":"2026-07-20T04:25:55.322126Z","iopub.status.idle":"2026-07-20T04:25:59.869205Z","shell.execute_reply.started":"2026-07-20T04:25:55.322098Z","shell.execute_reply":"2026-07-20T04:25:59.868181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport ast\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nfrom cafaeval.evaluation import cafa_eval\n\n# 1. File Paths\nSUBMISSION_FILE = \"/kaggle/working/submission.tsv\"\nGROUND_TRUTH_PARQUET = \"/kaggle/input/datasets/pradippokhrel77/test-interpro/test-00000-of-00001.parquet\"\nOBO_FILE = \"/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/go-basic.obo\"\nIA_FILE = \"/kaggle/input/competitions/cafa-5-protein-function-prediction/IA.txt\"\n\nPRED_DIR = \"/kaggle/working/predictions\"\nGT_FILE = \"/kaggle/working/query_ground_truth.tsv\"\n\nos.makedirs(PRED_DIR, exist_ok=True)\n\n# 2. Save predictions inside dedicated folder with progress bar\npred_df = pd.read_csv(SUBMISSION_FILE, sep=\"\\t\", names=[\"EntryID\", \"term\", \"score\"])\nfor _ in tqdm(range(1), desc=\"Saving Predictions\"):\n    pred_df.to_csv(os.path.join(PRED_DIR, \"submission.tsv\"), sep=\"\\t\", index=False, header=False)\nprint(f\"Total predictions saved: {len(pred_df):,}\")\n\n# 3. Format Ground Truth into long format with tqdm progress bar\ngt_df = pd.read_parquet(GROUND_TRUTH_PARQUET).dropna(subset=['go_ids'])\n\ngt_records = []\nfor _, row in tqdm(gt_df.iterrows(), total=len(gt_df), desc=\"Processing Ground Truth\"):\n    prot_id = row['protein_id']\n    raw_go = row['go_ids']\n    \n    if isinstance(raw_go, str):\n        try:\n            go_list = ast.literal_eval(raw_go)\n        except Exception:\n            go_list = [raw_go]\n    else:\n        go_list = list(raw_go)\n        \n    for term in go_list:\n        gt_records.append([prot_id, term])\n\ngt_formatted = pd.DataFrame(gt_records, columns=['protein_id', 'go_id'])\ngt_formatted.to_csv(GT_FILE, sep=\"\\t\", index=False, header=False)\nprint(f\"Formatted ground truth annotations: {len(gt_formatted):,}\")\n\n# 4. Run CAFA Evaluation (n_cpu=4 for multi-core speedup)\nprint(\"\\nRunning CAFA Evaluation across thresholds...\")\nfor _ in tqdm(range(1), desc=\"Calculating CAFA Metrics\"):\n    results = cafa_eval(\n        obo_file=OBO_FILE,\n        pred_dir=PRED_DIR,\n        gt_file=GT_FILE,\n        ia=IA_FILE,\n        no_orphans=False,\n        norm=\"cafa\",\n        prop=\"max\",\n        th_step=0.01,\n        n_cpu=4  # Utilizing 4 CPU cores\n    )\n\nprint(\"\\n==============================\")\nprint(\"CAFA Evaluation Results\")\nprint(\"==============================\")\nprint(results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T04:54:08.823746Z","iopub.execute_input":"2026-07-20T04:54:08.824791Z","iopub.status.idle":"2026-07-20T04:58:27.368254Z","shell.execute_reply.started":"2026-07-20T04:54:08.824751Z","shell.execute_reply":"2026-07-20T04:58:27.367051Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print formatted peak summary from the results tuple\nprint(\"=== CAFA 5 PEAK PERFORMANCE SUMMARY ===\")\n\n# 1. Move 'ns' and 'tau' from Index to regular DataFrame columns\npeak_fmax = results[1]['f'].reset_index()\n\n# 2. Select and rename the summary metrics\nsummary_df = peak_fmax[['ns', 'tau', 'f', 'pr', 'rc', 'cov']].rename(\n    columns={\n        'ns': 'Ontology Aspect',\n        'tau': 'Optimal Threshold',\n        'f': 'F_max',\n        'pr': 'Precision',\n        'rc': 'Recall',\n        'cov': 'Coverage'\n    }\n)\n\nsummary_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T05:02:37.385834Z","iopub.execute_input":"2026-07-20T05:02:37.386336Z","iopub.status.idle":"2026-07-20T05:02:37.42831Z","shell.execute_reply.started":"2026-07-20T05:02:37.386306Z","shell.execute_reply":"2026-07-20T05:02:37.427306Z"}},"outputs":[],"execution_count":null}]}