{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport gc\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.model_selection import train_test_split\nfrom lightgbm import LGBMClassifier\nfrom lightgbm import early_stopping\nfrom lightgbm import log_evaluation\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-08T20:00:32.415041Z","iopub.execute_input":"2023-06-08T20:00:32.415409Z","iopub.status.idle":"2023-06-08T20:00:34.555808Z","shell.execute_reply.started":"2023-06-08T20:00:32.415378Z","shell.execute_reply":"2023-06-08T20:00:34.554952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analisando o banco de dados","metadata":{}},{"cell_type":"code","source":"def read_fasta_file(file_path):\n    entries = []\n    with open(file_path, \"r\") as fasta_file:\n        lines = fasta_file.readlines()\n        i = 0\n        while i < len(lines):\n            if lines[i].startswith(\">\"):\n                entry = {}\n                entry['EntryID'] = lines[i][1:].split()[0]\n                \n                #entry[\"OX\"] = lines[i].split(\"OX=\")[1].split()[0]\n                i += 1\n                sequence_lines = []\n                while i < len(lines) and not lines[i].startswith(\">\"):\n                    sequence_lines.append(lines[i].strip())\n                    i += 1\n                entry['seq'] = \"\".join(sequence_lines)\n                entries.append(entry)\n            else:\n                i += 1\n    df = pd.DataFrame(entries)\n    df.set_index('EntryID', inplace=True)\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-11T02:58:12.208879Z","iopub.execute_input":"2023-06-11T02:58:12.209543Z","iopub.status.idle":"2023-06-11T02:58:12.218592Z","shell.execute_reply.started":"2023-06-11T02:58:12.209511Z","shell.execute_reply":"2023-06-11T02:58:12.217487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Entrada\ntrainFasta = read_fasta_file('/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta')\n\n# Saída\ntrainTermsID = pd.read_csv('/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv',\n                           sep='\\t')\n# Teste\ntestFasta = read_fasta_file('/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta')\n\nprint(trainTermsID['term'].value_counts(normalize=True)*100)\nprint(trainTermsID['term'].nunique())","metadata":{"execution":{"iopub.status.busy":"2023-06-08T20:00:34.566216Z","iopub.execute_input":"2023-06-08T20:00:34.566792Z","iopub.status.idle":"2023-06-08T20:00:43.513798Z","shell.execute_reply.started":"2023-06-08T20:00:34.56675Z","shell.execute_reply":"2023-06-08T20:00:43.512757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainTermsID['aspect'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T20:00:43.516717Z","iopub.execute_input":"2023-06-08T20:00:43.517203Z","iopub.status.idle":"2023-06-08T20:00:43.873419Z","shell.execute_reply.started":"2023-06-08T20:00:43.517164Z","shell.execute_reply":"2023-06-08T20:00:43.872417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainFasta","metadata":{"execution":{"iopub.status.busy":"2023-06-08T20:00:43.874693Z","iopub.execute_input":"2023-06-08T20:00:43.875512Z","iopub.status.idle":"2023-06-08T20:00:43.899268Z","shell.execute_reply.started":"2023-06-08T20:00:43.875459Z","shell.execute_reply":"2023-06-08T20:00:43.89845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testFasta","metadata":{"execution":{"iopub.status.busy":"2023-06-08T20:00:43.900406Z","iopub.execute_input":"2023-06-08T20:00:43.901406Z","iopub.status.idle":"2023-06-08T20:00:43.911121Z","shell.execute_reply.started":"2023-06-08T20:00:43.901375Z","shell.execute_reply":"2023-06-08T20:00:43.910214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainTermsID","metadata":{"execution":{"iopub.status.busy":"2023-06-08T20:00:43.912572Z","iopub.execute_input":"2023-06-08T20:00:43.912971Z","iopub.status.idle":"2023-06-08T20:00:43.927984Z","shell.execute_reply.started":"2023-06-08T20:00:43.912921Z","shell.execute_reply":"2023-06-08T20:00:43.926981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainTermsID['EntryID'].value_counts().mean()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T20:00:43.929351Z","iopub.execute_input":"2023-06-08T20:00:43.929895Z","iopub.status.idle":"2023-06-08T20:00:44.326003Z","shell.execute_reply.started":"2023-06-08T20:00:43.92986Z","shell.execute_reply":"2023-06-08T20:00:44.324997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainFull = trainTermsID.merge(trainFasta, on='EntryID', how='left')\n#trainFull.drop('aspect', axis=1, inplace=True)\ntrainFull","metadata":{"execution":{"iopub.status.busy":"2023-06-08T20:00:44.327753Z","iopub.execute_input":"2023-06-08T20:00:44.328145Z","iopub.status.idle":"2023-06-08T20:00:46.093435Z","shell.execute_reply.started":"2023-06-08T20:00:44.328108Z","shell.execute_reply":"2023-06-08T20:00:46.092534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def keep_most_frequent_outputs(df, percentage):\n    # Calculate output frequency\n    output_counts = df['term'].value_counts()\n\n    # Calculate the number of classes to keep based on the percentage\n    n = int(len(output_counts) * percentage)\n\n    # Get the \"n\" most frequent outputs\n    most_frequent_outputs = output_counts.head(n).index\n\n    # Filter the DataFrame to keep rows with the most frequent outputs\n    filtered_df = df[df['term'].isin(most_frequent_outputs)].copy()\n\n    # Calculate the percentage of the database removed\n    removed_percentage = (1 - len(filtered_df) / len(df)) * 100\n    print(\"Percentage of database removed: {:.2f}%\".format(removed_percentage))\n    \n    return filtered_df\n\ntrainFullD = keep_most_frequent_outputs(trainFull, 0.1)\ntrainFullD.drop('EntryID', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T20:00:46.096253Z","iopub.execute_input":"2023-06-08T20:00:46.09656Z","iopub.status.idle":"2023-06-08T20:00:48.764849Z","shell.execute_reply.started":"2023-06-08T20:00:46.096535Z","shell.execute_reply":"2023-06-08T20:00:48.763879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainFullD","metadata":{"execution":{"iopub.status.busy":"2023-06-08T20:00:48.766203Z","iopub.execute_input":"2023-06-08T20:00:48.766526Z","iopub.status.idle":"2023-06-08T20:00:48.778214Z","shell.execute_reply.started":"2023-06-08T20:00:48.766476Z","shell.execute_reply":"2023-06-08T20:00:48.777101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Separando em 3 grupos","metadata":{}},{"cell_type":"code","source":"trainBPO = trainFullD[trainFullD['aspect'] == 'BPO'].copy()\ntrainBPO.drop('aspect', axis=1, inplace=True)\ntrainCCO = trainFullD[trainFullD['aspect'] == 'CCO'].copy()\ntrainCCO.drop('aspect', axis=1, inplace=True)\ntrainMFO = trainFullD[trainFullD['aspect'] == 'MFO'].copy()\ntrainMFO.drop('aspect', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T20:00:48.779898Z","iopub.execute_input":"2023-06-08T20:00:48.780975Z","iopub.status.idle":"2023-06-08T20:00:50.531102Z","shell.execute_reply.started":"2023-06-08T20:00:48.780936Z","shell.execute_reply":"2023-06-08T20:00:50.530035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(trainMFO['term'].value_counts())\nprint(trainMFO['seq'].nunique())","metadata":{"execution":{"iopub.status.busy":"2023-06-08T20:00:50.532453Z","iopub.execute_input":"2023-06-08T20:00:50.532792Z","iopub.status.idle":"2023-06-08T20:00:51.129429Z","shell.execute_reply.started":"2023-06-08T20:00:50.532765Z","shell.execute_reply":"2023-06-08T20:00:51.128205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_most_common(df):\n    # Get value counts of the term column\n    term_counts = df['term'].value_counts()\n    \n    # Add a column to df with the count of each term\n    df['term_count'] = df['term'].map(term_counts)\n    \n    # Create an output dataframe with unique sequences\n    unique_seqs = df['seq'].unique()\n    output_df = pd.DataFrame({'seq': unique_seqs, 'term': ''})\n    \n    # Create a dictionary to store the most common term for each sequence\n    most_common_terms = {}\n    \n    # Iterate over each unique sequence with tqdm progress bar\n    for sequence in tqdm(unique_seqs, desc='Processing sequences', leave=True):\n        # Filter the df to find all rows with the sequence\n        seq_rows = df.query(\"seq == @sequence\")\n        \n        # Get the term with the highest count\n        most_common_term = seq_rows.nlargest(3, 'term_count')['term'].iloc[-1]\n        \n        # Store the most common term in the dictionary\n        most_common_terms[sequence] = most_common_term\n    \n    # Update the 'term' column in the output dataframe using the stored most common terms\n    output_df['term'] = output_df['seq'].map(most_common_terms)\n    \n    return output_df\n\ntrainBPOs = get_most_common(trainBPO)\nprint('BPO pronto')\ntrainCCOs = get_most_common(trainCCO)\nprint('CCO pronto')\ntrainMFOs = get_most_common(trainMFO)\nprint('MFO pronto')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T20:00:51.131003Z","iopub.execute_input":"2023-06-08T20:00:51.131666Z","iopub.status.idle":"2023-06-08T23:44:04.580007Z","shell.execute_reply.started":"2023-06-08T20:00:51.131628Z","shell.execute_reply":"2023-06-08T23:44:04.578887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modelo","metadata":{}},{"cell_type":"code","source":"# Copiando para processar\ntrainBPO = trainBPOs.copy()\ntrainCCO = trainCCOs.copy()\ntrainMFO = trainMFOs.copy()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T23:44:04.581082Z","iopub.execute_input":"2023-06-08T23:44:04.581507Z","iopub.status.idle":"2023-06-08T23:44:04.677049Z","shell.execute_reply.started":"2023-06-08T23:44:04.581451Z","shell.execute_reply":"2023-06-08T23:44:04.675933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainBPO","metadata":{"execution":{"iopub.status.busy":"2023-06-08T23:44:04.678391Z","iopub.execute_input":"2023-06-08T23:44:04.678726Z","iopub.status.idle":"2023-06-08T23:44:04.699464Z","shell.execute_reply.started":"2023-06-08T23:44:04.678693Z","shell.execute_reply":"2023-06-08T23:44:04.69833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(df):\n    # Calculate class counts\n    class_counts = df['term'].value_counts()\n    \n    # Find the class with only one sample\n    class_to_drop = class_counts[class_counts == 1].index\n    \n    if len(class_to_drop) > 0:\n        # Drop the class with only one sample from df\n        df = df[~df['term'].isin(class_to_drop)]\n    \n    gc.collect()\n    # Split the data into training and testing sets with stratification\n    Xtrain, Xval, ytrain, yval = train_test_split(df['seq'], df['term'], test_size=0.2,\n                                                  random_state=42, stratify=df['term'])\n    \n    # Preprocess the training and validation data\n    todasLetras = 'abcdefghijklmnopqrstuvwxyz'\n    vectorizer = CountVectorizer(analyzer='char', vocabulary=todasLetras)\n    Xtrain = vectorizer.fit_transform(Xtrain)\n    Xval = vectorizer.transform(Xval)\n    \n    # Convert to float64\n    Xtrain = Xtrain.astype('float64')\n    Xval = Xval.astype('float64')\n    gc.collect()\n    \n    # Create the LGBMClassifier model\n    model =  LGBMClassifier(random_state=42)\n    gc.collect()\n    # Train the model with early stopping\n    model.fit(Xtrain, ytrain,\n              callbacks=[early_stopping(100), log_evaluation(100)],\n              eval_metric='logloss',\n              eval_set=[(Xval, yval)])\n    gc.collect()\n    \n    print(model.score(Xval,yval))\n    return vectorizer, model\n\nvectBPO, modelBPO = create_model(trainBPO)\nvectCCO, modelCCO = create_model(trainCCO)\nvectMFO, modelMFO = create_model(trainMFO)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T23:44:04.70098Z","iopub.execute_input":"2023-06-08T23:44:04.702111Z","iopub.status.idle":"2023-06-08T23:47:07.05242Z","shell.execute_reply.started":"2023-06-08T23:44:04.702066Z","shell.execute_reply":"2023-06-08T23:47:07.051215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T23:47:07.053918Z","iopub.execute_input":"2023-06-08T23:47:07.054258Z","iopub.status.idle":"2023-06-08T23:47:07.219175Z","shell.execute_reply.started":"2023-06-08T23:47:07.054228Z","shell.execute_reply":"2023-06-08T23:47:07.217887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Geração da submissão","metadata":{}},{"cell_type":"code","source":"XBPO = vectBPO.transform(testFasta['seq']).astype('float64')\nXCCO = vectCCO.transform(testFasta['seq']).astype('float64')\nXMFO = vectMFO.transform(testFasta['seq']).astype('float64')\nprint('Transformação feita')\n\npredBPO = modelBPO.predict(XBPO)\npredCCO = modelCCO.predict(XCCO)\npredMFO = modelMFO.predict(XMFO)\nprint('Predição feita')\n\nprobBPO = modelBPO.predict_proba(XBPO).max(axis=1)\nprobCCO = modelCCO.predict_proba(XCCO).max(axis=1)\nprobMFO = modelMFO.predict_proba(XMFO).max(axis=1)\nprint('Probabilidade feita')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T23:47:07.220755Z","iopub.execute_input":"2023-06-08T23:47:07.221123Z","iopub.status.idle":"2023-06-08T23:47:55.750631Z","shell.execute_reply.started":"2023-06-08T23:47:07.221095Z","shell.execute_reply":"2023-06-08T23:47:55.749887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combinando predições\n\n# Create separate DataFrames for each class\ndf_bpo = pd.DataFrame({'EntryID': testFasta.index.values,\n                       'Prediction': predBPO,\n                       'Probability': probBPO})\n\ndf_cco = pd.DataFrame({'EntryID': testFasta.index.values,\n                       'Prediction': predCCO,\n                       'Probability': probCCO})\n\ndf_mfo = pd.DataFrame({'EntryID': testFasta.index.values,\n                       'Prediction': predMFO,\n                       'Probability': probMFO})\n\n# Concatenate the DataFrames vertically\nsubmission_df = pd.concat([df_bpo, df_cco, df_mfo], ignore_index=True)\n\n# Save the DataFrame as submission.tsv without headers\nsubmission_df.to_csv('submission.tsv', sep='\\t', header=False, index=False)\nprint(submission_df['Prediction'].value_counts())\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2023-06-08T23:47:55.75175Z","iopub.execute_input":"2023-06-08T23:47:55.752211Z","iopub.status.idle":"2023-06-08T23:47:57.080723Z","shell.execute_reply.started":"2023-06-08T23:47:55.752183Z","shell.execute_reply":"2023-06-08T23:47:57.079637Z"},"trusted":true},"execution_count":null,"outputs":[]}]}