{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Introduction\n* just for saving df_term_full","metadata":{}},{"cell_type":"markdown","source":"## Setup","metadata":{}},{"cell_type":"code","source":"GLOBAL_SEED = 42\n\nimport os\nos.environ['PYTHONHASHSEED'] = str(GLOBAL_SEED)\n\nimport numpy as np # linear algebra\nfrom numpy import random as np_rnd\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport pickle\nimport gc\n\nfrom sklearn.model_selection import StratifiedKFold, StratifiedGroupKFold\n\nfrom Bio import SeqIO\nfrom Bio.SeqUtils.ProtParam import ProteinAnalysis\nfrom collections import Counter\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-17T02:07:20.029288Z","iopub.execute_input":"2023-07-17T02:07:20.029802Z","iopub.status.idle":"2023-07-17T02:07:21.208318Z","shell.execute_reply.started":"2023-07-17T02:07:20.029765Z","shell.execute_reply":"2023-07-17T02:07:21.207129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pickleIO(obj, src, op=\"r\"):\n    if op==\"w\":\n        with open(src, op + \"b\") as f:\n            pickle.dump(obj, f)\n    elif op==\"r\":\n        with open(src, op + \"b\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    else:\n        print(\"unknown operation\")\n        return obj\n    \ndef seed_everything(seed=42):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    # python random\n    rnd.seed(seed)\n    # numpy random\n    np_rnd.seed(seed)\n    # RAPIDS random\n    try:\n        cupy.random.seed(seed)\n    except:\n        pass\n    # tf random\n    try:\n        tf_rnd.set_seed(seed)\n    except:\n        pass\n    # pytorch random\n    try:\n        torch.manual_seed(seed)\n        torch.cuda.manual_seed(seed)\n        torch.backends.cudnn.deterministic = True\n    except:\n        pass","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:21.210895Z","iopub.execute_input":"2023-07-17T02:07:21.211349Z","iopub.status.idle":"2023-07-17T02:07:21.220774Z","shell.execute_reply.started":"2023-07-17T02:07:21.21131Z","shell.execute_reply":"2023-07-17T02:07:21.219995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set amino acid mapper\naa_map = {'VAL': 'V', 'PRO': 'P', 'ASN': 'N', 'GLU': 'E', 'ASP': 'D', 'ALA': 'A', 'THR': 'T', 'SER': 'S',\n          'LEU': 'L', 'LYS': 'K', 'GLY': 'G', 'GLN': 'Q', 'ILE': 'I', 'PHE': 'F', 'CYS': 'C', 'TRP': 'W',\n          'ARG': 'R', 'TYR': 'Y', 'HIS': 'H', 'MET': 'M'}\naa_map[\"X\"] = \"X\"\naa_map_encoder = {x:y for x,y in zip(list(aa_map.values()),np.arange(21))}\n\n# set amino acid group mapper\naa_groups = {\n    # Electrically Charged Side Chains - positive\n    \"AAG0\": [\"R\", \"H\", \"K\"],\n    # Electrically Charged Side Chains - negative\n    \"AAG1\": [\"D\", \"E\"],\n    # Polar Uncharged Side Chains\n    \"AAG2\": [\"S\", \"T\", \"N\", \"Q\"],\n    # Hydrophobic Side Chains\n    \"AAG3\": [\"A\", \"V\", \"I\", \"L\", \"M\", \"F\", \"Y\", \"W\"],\n    # Not including any group\n    \"AAG4\": [\"P\", \"G\", \"C\", \"X\"],\n}\naa_groups_encoder = {}\nvalue = 0\nfor i in aa_groups.values():\n    for j in i: aa_groups_encoder[j] = value\n    value += 1\ndef get_amino_acids_group_percent(seq):\n    counter = Counter([aa_groups_encoder[i] for i in seq])\n    norm = sum(counter.values())\n    return {f\"AAG{k}\": v / norm for k, v in counter.items()}\n\n# get amino acid properties\n# https://www.kaggle.com/datasets/alejopaullier/aminoacids-physical-and-chemical-properties\naa_props = pd.read_csv(\"/kaggle/input/aminoacids-physical-and-chemical-properties/aminoacids.csv\").set_index('Letter')\n# set property variable for analysis\nPROPS = ['Molecular Weight', 'Residue Weight', 'pKa1', 'pKb2', 'pKx3', 'pl4', \n         'H', 'VSC', 'P1', 'P2', 'SASA', 'NCISC', 'carbon', 'hydrogen', 'oxygen']\n# remove pKx3 which includes na values\nPROPS.remove(\"pKx3\")\n# remove special case amino acid\naa_props = aa_props.drop([\"O\", \"U\"])\n# impute X amino acid property values with mean of other amino acids\nvalue = aa_props.mean()\nfor i in [\"Name\", \"Abbr\", \"Molecular Formula\", \"Residue Formula\"]:\n    value[i] = \"X\"\naa_props.loc[\"X\"] = value\n# shape check\nprint('Amino Acid properties dataframe. Shape:', aa_props.shape )\n# validation check\nfor i in aa_props.index:\n    if i not in aa_map.values():\n        print(i)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:21.222228Z","iopub.execute_input":"2023-07-17T02:07:21.222845Z","iopub.status.idle":"2023-07-17T02:07:21.283006Z","shell.execute_reply.started":"2023-07-17T02:07:21.222815Z","shell.execute_reply":"2023-07-17T02:07:21.281827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aa_props","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:21.284261Z","iopub.execute_input":"2023-07-17T02:07:21.284595Z","iopub.status.idle":"2023-07-17T02:07:21.358517Z","shell.execute_reply.started":"2023-07-17T02:07:21.284566Z","shell.execute_reply":"2023-07-17T02:07:21.357389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(aa_map_encoder)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:21.362353Z","iopub.execute_input":"2023-07-17T02:07:21.362993Z","iopub.status.idle":"2023-07-17T02:07:21.369278Z","shell.execute_reply.started":"2023-07-17T02:07:21.36295Z","shell.execute_reply":"2023-07-17T02:07:21.368164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(aa_groups_encoder)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:21.371314Z","iopub.execute_input":"2023-07-17T02:07:21.372075Z","iopub.status.idle":"2023-07-17T02:07:21.383952Z","shell.execute_reply.started":"2023-07-17T02:07:21.372034Z","shell.execute_reply":"2023-07-17T02:07:21.382683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading Data","metadata":{}},{"cell_type":"markdown","source":"### sequence data","metadata":{}},{"cell_type":"code","source":"fasta_table = {\n    \"uniprot_id\": [],\n    \"seq\": [],\n}\nfor i in SeqIO.parse(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\", \"fasta\"):\n    fasta_table[\"uniprot_id\"].append(i.id)\n    fasta_table[\"seq\"].append(str(i.seq))\nfasta_table = pd.DataFrame(fasta_table)\nfasta_table = fasta_table.drop_duplicates(subset=\"uniprot_id\").reset_index(drop=True)\nfasta_table[\"len\"] = fasta_table[\"seq\"].apply(len)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:21.38554Z","iopub.execute_input":"2023-07-17T02:07:21.38633Z","iopub.status.idle":"2023-07-17T02:07:24.386938Z","shell.execute_reply.started":"2023-07-17T02:07:21.386287Z","shell.execute_reply":"2023-07-17T02:07:24.385559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fasta_table.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:24.388585Z","iopub.execute_input":"2023-07-17T02:07:24.389072Z","iopub.status.idle":"2023-07-17T02:07:24.402799Z","shell.execute_reply.started":"2023-07-17T02:07:24.389027Z","shell.execute_reply":"2023-07-17T02:07:24.401124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### weight data","metadata":{}},{"cell_type":"code","source":"# Label weight data\nwith open(\"/kaggle/input/cafa-5-protein-function-prediction/IA.txt\", \"r\") as f:\n    tmp = f.readlines()\n    tmp = [i.rstrip(\"\\n\").split(\"\\t\") for i in tmp]\n    res1, res2 = map(list, zip(*tmp))\n    term_weight = pd.Series({k: v for k, v in zip(res1, res2)}).astype(\"float32\")","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:24.404315Z","iopub.execute_input":"2023-07-17T02:07:24.404644Z","iopub.status.idle":"2023-07-17T02:07:24.665989Z","shell.execute_reply.started":"2023-07-17T02:07:24.404617Z","shell.execute_reply":"2023-07-17T02:07:24.664812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"term_weight = term_weight.to_frame(\"weight\")\nterm_weight.index.name = \"term\"\nterm_weight[\"weight_level\"] = pd.cut(term_weight[\"weight\"], 5, labels=list(range(5))).astype(\"int8\")\nterm_weight.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:24.667299Z","iopub.execute_input":"2023-07-17T02:07:24.667646Z","iopub.status.idle":"2023-07-17T02:07:24.689529Z","shell.execute_reply.started":"2023-07-17T02:07:24.667617Z","shell.execute_reply":"2023-07-17T02:07:24.688439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pickleIO(term_weight, \"./term_weight.pkl\", \"w\")","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:24.69093Z","iopub.execute_input":"2023-07-17T02:07:24.691292Z","iopub.status.idle":"2023-07-17T02:07:24.706785Z","shell.execute_reply.started":"2023-07-17T02:07:24.691261Z","shell.execute_reply":"2023-07-17T02:07:24.70584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### train data","metadata":{}},{"cell_type":"code","source":"# Label data\ndf_term_full = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\", sep=\"\\t\")\ndf_term_full = df_term_full.rename(columns={\"EntryID\": \"uniprot_id\"})\ndf_term_full.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:24.70805Z","iopub.execute_input":"2023-07-17T02:07:24.708549Z","iopub.status.idle":"2023-07-17T02:07:29.041274Z","shell.execute_reply.started":"2023-07-17T02:07:24.708521Z","shell.execute_reply":"2023-07-17T02:07:29.040044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merge the weight\ndf_term_full[\"weight\"] = term_weight.loc[df_term_full[\"term\"].values, \"weight\"].values\n# transform term weight to difficulty\ndf_term_full[\"weight_level\"] = term_weight.loc[df_term_full[\"term\"].values, \"weight_level\"].values ","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:29.042898Z","iopub.execute_input":"2023-07-17T02:07:29.043232Z","iopub.status.idle":"2023-07-17T02:07:32.779637Z","shell.execute_reply.started":"2023-07-17T02:07:29.0432Z","shell.execute_reply":"2023-07-17T02:07:32.778645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aspect_mapper = df_term_full[[\"term\", \"aspect\"]].drop_duplicates().set_index(\"term\")[\"aspect\"]\npickleIO(aspect_mapper, \"./aspect_mapper.pkl\", \"w\")","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:32.788108Z","iopub.execute_input":"2023-07-17T02:07:32.788477Z","iopub.status.idle":"2023-07-17T02:07:34.299978Z","shell.execute_reply.started":"2023-07-17T02:07:32.78845Z","shell.execute_reply":"2023-07-17T02:07:34.298805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_term_full[\"weight\"].describe().round(3)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:34.301722Z","iopub.execute_input":"2023-07-17T02:07:34.302703Z","iopub.status.idle":"2023-07-17T02:07:34.535131Z","shell.execute_reply.started":"2023-07-17T02:07:34.302653Z","shell.execute_reply":"2023-07-17T02:07:34.534139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_term_full[\"weight_level\"].value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:34.536443Z","iopub.execute_input":"2023-07-17T02:07:34.536785Z","iopub.status.idle":"2023-07-17T02:07:34.591996Z","shell.execute_reply.started":"2023-07-17T02:07:34.536757Z","shell.execute_reply":"2023-07-17T02:07:34.590589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_term_full.info()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:34.593841Z","iopub.execute_input":"2023-07-17T02:07:34.594268Z","iopub.status.idle":"2023-07-17T02:07:34.610945Z","shell.execute_reply.started":"2023-07-17T02:07:34.59423Z","shell.execute_reply":"2023-07-17T02:07:34.609757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_term_full.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:34.612621Z","iopub.execute_input":"2023-07-17T02:07:34.613359Z","iopub.status.idle":"2023-07-17T02:07:34.625651Z","shell.execute_reply.started":"2023-07-17T02:07:34.613327Z","shell.execute_reply":"2023-07-17T02:07:34.6247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Negative Sampling","metadata":{}},{"cell_type":"markdown","source":"### Class sampling by frequency","metadata":{}},{"cell_type":"code","source":"threshold = 0.85\n\nterm_frequency_vc = {i: df_term_full.loc[(df_term_full[\"aspect\"] == i).values, \"term\"].value_counts(normalize=True) for i in df_term_full[\"aspect\"].unique()}\nmajority_terms = []\nfor k in term_frequency_vc.keys():\n    majority_terms.extend(term_frequency_vc[k].index[term_frequency_vc[k].cumsum() <= threshold].to_list())\nmajority_terms = pd.Series(majority_terms)\n\nprint(\"=== number of classes ===\")\ndisplay(df_term_full.loc[df_term_full[\"term\"].isin(majority_terms), [\"term\", \"aspect\"]].groupby(\"aspect\")[\"term\"].nunique())\nprint(\"total ->\", df_term_full.loc[df_term_full[\"term\"].isin(majority_terms), [\"term\", \"aspect\"]].groupby(\"aspect\")[\"term\"].nunique().sum())","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:34.627123Z","iopub.execute_input":"2023-07-17T02:07:34.627783Z","iopub.status.idle":"2023-07-17T02:07:44.291872Z","shell.execute_reply.started":"2023-07-17T02:07:34.627752Z","shell.execute_reply":"2023-07-17T02:07:44.290646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"term_level_group = df_term_full.loc[df_term_full[\"term\"].isin(majority_terms), [\"term\", \"aspect\", \"weight_level\"]].groupby([\"aspect\", \"weight_level\"])[\"term\"].apply(lambda x: np.array(list(set(x))))","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:44.293294Z","iopub.execute_input":"2023-07-17T02:07:44.293762Z","iopub.status.idle":"2023-07-17T02:07:46.956316Z","shell.execute_reply.started":"2023-07-17T02:07:44.293721Z","shell.execute_reply":"2023-07-17T02:07:46.955113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"term_level_group","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:46.957744Z","iopub.execute_input":"2023-07-17T02:07:46.958102Z","iopub.status.idle":"2023-07-17T02:07:46.971322Z","shell.execute_reply.started":"2023-07-17T02:07:46.958072Z","shell.execute_reply":"2023-07-17T02:07:46.970212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pickleIO(majority_terms, \"./majority_terms.pkl\", \"w\")","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:46.972769Z","iopub.execute_input":"2023-07-17T02:07:46.97311Z","iopub.status.idle":"2023-07-17T02:07:46.98589Z","shell.execute_reply.started":"2023-07-17T02:07:46.97308Z","shell.execute_reply":"2023-07-17T02:07:46.984722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Sample Pruning","metadata":{}},{"cell_type":"code","source":"print(\"before pruning shape ->\", df_term_full.shape)\n# === pruning train sample (drop samples) ===\n# get only sequence length 50 >= x & 1500 <= x\ndf_term_full = df_term_full[df_term_full[\"uniprot_id\"].isin(fasta_table.loc[(fasta_table[\"len\"] >= 50) & (fasta_table[\"len\"] <= 1500), \"uniprot_id\"])].reset_index(drop=True)\n# get term frequency is greater than equal threshold\ndf_term_full = df_term_full[df_term_full['term'].isin(majority_terms)].reset_index(drop=True)\nprint(\"after pruning ->\", df_term_full.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:46.987802Z","iopub.execute_input":"2023-07-17T02:07:46.988236Z","iopub.status.idle":"2023-07-17T02:07:49.24146Z","shell.execute_reply.started":"2023-07-17T02:07:46.988197Z","shell.execute_reply":"2023-07-17T02:07:49.240234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_term_full[\"weight\"].describe().round(3)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:49.242789Z","iopub.execute_input":"2023-07-17T02:07:49.243119Z","iopub.status.idle":"2023-07-17T02:07:49.424946Z","shell.execute_reply.started":"2023-07-17T02:07:49.243092Z","shell.execute_reply":"2023-07-17T02:07:49.423722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_term_full[\"weight_level\"].value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:49.426533Z","iopub.execute_input":"2023-07-17T02:07:49.426977Z","iopub.status.idle":"2023-07-17T02:07:49.469771Z","shell.execute_reply.started":"2023-07-17T02:07:49.426939Z","shell.execute_reply":"2023-07-17T02:07:49.468863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_term_full","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:49.471107Z","iopub.execute_input":"2023-07-17T02:07:49.471781Z","iopub.status.idle":"2023-07-17T02:07:49.489892Z","shell.execute_reply.started":"2023-07-17T02:07:49.471748Z","shell.execute_reply":"2023-07-17T02:07:49.488582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Merging with query sequence ","metadata":{}},{"cell_type":"code","source":"lbe_dic = {\"term\": {term: n for n, term in zip(range(len(majority_terms)), majority_terms)}}\npickleIO(majority_terms, \"./majority_terms.pkl\", \"w\")","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:49.491696Z","iopub.execute_input":"2023-07-17T02:07:49.492418Z","iopub.status.idle":"2023-07-17T02:07:49.499914Z","shell.execute_reply.started":"2023-07-17T02:07:49.492386Z","shell.execute_reply":"2023-07-17T02:07:49.498757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df_term_full.copy()\n\nterm_mapped_list = df[\"term\"].map(lbe_dic[\"term\"])\nassert term_mapped_list.isna().sum() == 0\n\ndf[\"term\"] = term_mapped_list.values\ndf = df.groupby(\"uniprot_id\", sort=False, as_index=False)[[\"term\", \"weight\"]].agg(list)\n\n# create label vector & weight vector\npos_label_vector = np.zeros((len(df), len(majority_terms)), dtype=\"float32\")\nfor idx, (_, row) in enumerate(df.iterrows()):\n    pos_idx = np.array(row[\"term\"])\n    pos_label_vector[idx, pos_idx] = 1.0","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:07:49.501406Z","iopub.execute_input":"2023-07-17T02:07:49.501774Z","iopub.status.idle":"2023-07-17T02:08:12.02847Z","shell.execute_reply.started":"2023-07-17T02:07:49.501744Z","shell.execute_reply":"2023-07-17T02:08:12.027218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:09:39.666834Z","iopub.execute_input":"2023-07-17T02:09:39.667289Z","iopub.status.idle":"2023-07-17T02:09:39.68985Z","shell.execute_reply.started":"2023-07-17T02:09:39.667253Z","shell.execute_reply":"2023-07-17T02:09:39.688611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pos_label_vector[:5]","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:09:34.062347Z","iopub.execute_input":"2023-07-17T02:09:34.062879Z","iopub.status.idle":"2023-07-17T02:09:34.073166Z","shell.execute_reply.started":"2023-07-17T02:09:34.062835Z","shell.execute_reply":"2023-07-17T02:09:34.072069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_container = {}\nfor i in [\"t5\", \"esm2\", \"protbert\"]:\n    feature_container[i] = pickleIO(None, f\"/kaggle/input/cafa-create-sequence-embedding-v2/seq_embed_{i}_train.pkl\", \"r\").loc[df[\"uniprot_id\"]]\n\nnp.savez(\"./df_full.npz\", label=pos_label_vector, **feature_container)\npickleIO(df, \"df_full_meta.pkl\", \"w\")","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:08:12.030242Z","iopub.execute_input":"2023-07-17T02:08:12.030998Z","iopub.status.idle":"2023-07-17T02:08:33.749423Z","shell.execute_reply.started":"2023-07-17T02:08:12.030963Z","shell.execute_reply":"2023-07-17T02:08:33.748043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test data","metadata":{}},{"cell_type":"code","source":"seq_embed = pickleIO(None, \"/kaggle/input/cafa-create-sequence-embedding-v2/seq_embed_t5_test.pkl\", \"r\")","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:08:33.752739Z","iopub.execute_input":"2023-07-17T02:08:33.753231Z","iopub.status.idle":"2023-07-17T02:08:38.254007Z","shell.execute_reply.started":"2023-07-17T02:08:33.753189Z","shell.execute_reply":"2023-07-17T02:08:38.252921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading fasta rawdata & transforming to dataframe\ndf_fasta = {\n    \"uniprot_id\": [],\n    \"seq\": [],\n}\nfor i in tqdm(list(SeqIO.parse(\"/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta\", \"fasta\"))):\n    df_fasta[\"uniprot_id\"].append(i.id)\n    df_fasta[\"seq\"].append(str(i.seq))\ndf_fasta = pd.DataFrame(df_fasta)\ndf_fasta = df_fasta.drop_duplicates(subset=\"uniprot_id\")\ndf_fasta.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:08:38.255362Z","iopub.execute_input":"2023-07-17T02:08:38.255714Z","iopub.status.idle":"2023-07-17T02:08:43.006313Z","shell.execute_reply.started":"2023-07-17T02:08:38.255643Z","shell.execute_reply":"2023-07-17T02:08:43.005088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_container = {}\nfor i in [\"t5\", \"esm2\", \"protbert\"]:\n    feature_container[i] = pickleIO(None, f\"/kaggle/input/cafa-create-sequence-embedding-v2/seq_embed_{i}_test.pkl\", \"r\").loc[df_fasta[\"uniprot_id\"]]\n\nnp.savez(\"./df_test.npz\", **feature_container)\npickleIO(df_fasta, \"df_test_meta.pkl\", \"w\")","metadata":{"execution":{"iopub.status.busy":"2023-07-17T02:08:43.008175Z","iopub.execute_input":"2023-07-17T02:08:43.008649Z","iopub.status.idle":"2023-07-17T02:09:09.442366Z","shell.execute_reply.started":"2023-07-17T02:08:43.008605Z","shell.execute_reply":"2023-07-17T02:09:09.440353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}