{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **FOREWORD**","metadata":{}},{"cell_type":"markdown","source":"This is a general kernel script for my tabular baseline model training purposes. <br>\n\n### **KEY CHANGES**\n1. Common invocation of metric across utility, hill climb and Optuna blends\n2. Support for TabNet models\n3. Improved feature importance for boosted trees, TabNet and linear models using model.coef_\n4. Support for Hill Climber\n5. Separate requirements and imports for AutoGluon models","metadata":{}},{"cell_type":"markdown","source":"# **PACKAGE INSTALLATIONS**","metadata":{}},{"cell_type":"code","source":"%%writefile -a req_kaggle.txt\n\nscikit-learn==1.5.2\nlightgbm==4.5.0\nxgboost==2.1.2\nnumpy==1.26.4\nscipy==1.14.1\npolars==1.15.0\npytorch_tabnet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:46:32.671436Z","iopub.execute_input":"2024-11-27T07:46:32.671896Z","iopub.status.idle":"2024-11-27T07:46:32.677972Z","shell.execute_reply.started":"2024-11-27T07:46:32.671857Z","shell.execute_reply":"2024-11-27T07:46:32.676871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a req_colab.txt\n\nxgboost==2.1.2\ncatboost==1.2.7\nnumpy==1.26.4\nscipy==1.14.1\npolars==1.15.0\ncolorama\ncloudpickle\noptuna\npytorch_tabnet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:46:34.782906Z","iopub.execute_input":"2024-11-27T07:46:34.783275Z","iopub.status.idle":"2024-11-27T07:46:34.789455Z","shell.execute_reply.started":"2024-11-27T07:46:34.783241Z","shell.execute_reply":"2024-11-27T07:46:34.788451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile req_ag.txt\n\npolars==1.15.0\ncolorama\noptuna\ncatboost==1.2.7\nautogluon.tabular\nray==2.10.0\ndask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:46:37.112405Z","iopub.execute_input":"2024-11-27T07:46:37.113342Z","iopub.status.idle":"2024-11-27T07:46:37.118649Z","shell.execute_reply.started":"2024-11-27T07:46:37.113302Z","shell.execute_reply":"2024-11-27T07:46:37.11755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a requirements_lama.txt\n\nlightautoml\ncolorama\npolars==1.15.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:48:30.672915Z","iopub.execute_input":"2024-11-27T07:48:30.673282Z","iopub.status.idle":"2024-11-27T07:48:30.67891Z","shell.execute_reply.started":"2024-11-27T07:48:30.673248Z","shell.execute_reply":"2024-11-27T07:48:30.678009Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **IMPORTS**","metadata":{}},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nprint(f\"\\n---> Commencing imports-part1\")\n\nfrom gc import collect\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\nfrom IPython.display import display_html, clear_output\nclear_output()\nimport os, sys, logging, re, joblib, ctypes, shutil, random, torch\nfrom copy import deepcopy\n\nimport xgboost as xgb, lightgbm as lgb, catboost as cb, sklearn as sk, pandas as pd\nprint(f\"---> XGBoost = {xgb.__version__} | LightGBM = {lgb.__version__} | Catboost = {cb.__version__}\")\nprint(f\"---> Sklearn = {sk.__version__}| Pandas = {pd.__version__}\")\ncollect()\n\n# General library imports:-\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\nfrom gc import collect\n\nfrom os import path, walk, getpid\nfrom psutil import Process\nimport re\nfrom collections import Counter\nfrom itertools import product\n\nimport ctypes\nlibc = ctypes.CDLL(\"libc.so.6\")\n\nfrom IPython.display import display_html, clear_output\nfrom pprint import pprint\nfrom functools import partial\nfrom copy import deepcopy\nimport pandas as pd, numpy as np, os, joblib\nimport polars as pl\nimport polars.selectors as cs\nimport re\n\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom colorama import Fore, Style, init\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\nfrom tqdm.notebook import tqdm\n\nprint(f\"---> Imports- part 1 done\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:46:39.447496Z","iopub.execute_input":"2024-11-27T07:46:39.447913Z","iopub.status.idle":"2024-11-27T07:46:39.454728Z","shell.execute_reply.started":"2024-11-27T07:46:39.447877Z","shell.execute_reply":"2024-11-27T07:46:39.453367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myimports.py\n\n# Pipeline specifics:-\nfrom sklearn.preprocessing import (RobustScaler,\n                                   MinMaxScaler,\n                                   StandardScaler,\n                                   FunctionTransformer as FT,\n                                   PowerTransformer,\n                                  )\nfrom sklearn.impute import SimpleImputer as SI\nfrom sklearn.model_selection import (RepeatedStratifiedKFold as RSKF,\n                                     StratifiedKFold as SKF,\n                                     StratifiedGroupKFold as SGKF,\n                                     KFold,\n                                     GroupKFold as GKF,\n                                     RepeatedKFold as RKF,\n                                     PredefinedSplit as PDS,\n                                     cross_val_score,\n                                     cross_val_predict,\n                                    )\nfrom sklearn.inspection import permutation_importance\nfrom sklearn.feature_selection import VarianceThreshold as VT\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.base import BaseEstimator, TransformerMixin, clone\nfrom sklearn.compose import ColumnTransformer, make_column_selector\n\n# ML Model training:-\nfrom sklearn.metrics import (\nroc_auc_score, brier_score_loss, accuracy_score, cohen_kappa_score, f1_score , \nr2_score, root_mean_squared_error as rmse, mean_squared_error as mse,\nroot_mean_squared_log_error as rmsle,\nmake_scorer,      \n)\n\nfrom xgboost import QuantileDMatrix, XGBClassifier as XGBC, XGBRegressor as XGBR\nfrom lightgbm import log_evaluation, early_stopping, LGBMClassifier as LGBMC, LGBMRegressor as LGBMR\nfrom catboost import CatBoostClassifier as CBC, Pool, CatBoostRegressor as CBR\nfrom sklearn.ensemble import HistGradientBoostingClassifier as HGBC, RandomForestClassifier as RFC\nfrom sklearn.ensemble import HistGradientBoostingRegressor as HGBR, RandomForestRegressor as RFR\nfrom sklearn.linear_model import LogisticRegression as LRC, Ridge, Lasso\n\n# TabNet models\nfrom pytorch_tabnet.tab_model import (TabNetRegressor as TNR, TabNetClassifier as TNC)\n\n# Ensemble and tuning:-\nimport optuna\nfrom optuna import Trial, trial, create_study\nfrom optuna.pruners import HyperbandPruner\nfrom optuna.samplers import TPESampler, CmaEsSampler\n\n# Setting rc parameters in seaborn for plots and graphs-\nsns.set({\"axes.facecolor\"       : \"white\",\n         \"figure.facecolor\"     : \"#ffffff\",\n         \"axes.edgecolor\"       : \"black\",\n         \"grid.color\"           : '#b0b0b0',\n         \"font.family\"          : ['Cambria'],\n         \"axes.labelcolor\"      : \"#000000\",\n         \"xtick.color\"          : \"#000000\",\n         \"ytick.color\"          : \"#000000\",\n         \"grid.linewidth\"       : 0.50,\n         \"grid.linestyle\"       : \"--\",\n         \"axes.titlecolor\"      : 'maroon',\n         'axes.titlesize'       : 9,\n         'axes.labelweight'     : \"bold\",\n         'legend.fontsize'      : 7.0,\n         'legend.title_fontsize': 7.0,\n         'font.size'            : 7.5,\n         'xtick.labelsize'      : 12.5,\n         'ytick.labelsize'      : 9.0,\n        }\n       )\n\n# Color printing\ndef PrintColor(text: str, color = Fore.BLUE, style = Style.BRIGHT):\n    \"Prints color outputs using colorama using a text F-string\"\n    print(style + color + text + Style.RESET_ALL)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:46:42.231435Z","iopub.execute_input":"2024-11-27T07:46:42.231854Z","iopub.status.idle":"2024-11-27T07:46:42.239683Z","shell.execute_reply.started":"2024-11-27T07:46:42.231817Z","shell.execute_reply":"2024-11-27T07:46:42.238631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nprint(f\"---> Commencing imports-part2\")\noptuna.logging.set_verbosity = optuna.logging.ERROR\noptuna.logging.disable_default_handler()\nprint(f\"---> XGBoost = {xgb.__version__} | LightGBM = {lgb.__version__}\")\n\n##################################################################\n# Customizing logging for LGBM\nclass MyLogger:\n    \"\"\"\n    This class helps to suppress logs in lightgbm and Optuna\n    Source - https://github.com/microsoft/LightGBM/issues/6014\n    \"\"\"\n\n    def init(self, logging_lbl: str):\n        self.logger = logging.getLogger(logging_lbl)\n        self.logger.setLevel(logging.ERROR)\n\n    def info(self, message):\n        pass\n\n    def warning(self, message):\n        pass\n\n    def error(self, message):\n        self.logger.error(message)\n\nl = MyLogger()\nl.init(logging_lbl = \"lightgbm_custom\")\nlgb.register_logger(l)\n\n##################################################################\n# Customizing logging for XGBoost\nfor handler in logging.root.handlers[:]:\n    logging.root.removeHandler(handler)\n\nlogger = logging.getLogger(__name__)\nlogger.setLevel(logging.ERROR)\nformatter = logging.Formatter('%(asctime)s | %(levelname)s | %(message)s')\n\nstdout_handler = logging.StreamHandler(sys.stdout)\nstdout_handler.setLevel(logging.INFO)\nstdout_handler.setFormatter(formatter)\n\nfile_handler = logging.FileHandler(f'xgb_optimize.log')\nfile_handler.setLevel(logging.ERROR)\nfile_handler.setFormatter(formatter)\n\nlogger.addHandler(file_handler)\nlogger.addHandler(stdout_handler)\n\nclass XGBLogging(xgb.callback.TrainingCallback):\n    \"\"\"log train logs to file\"\"\"\n\n    def __init__(self, epoch_log_interval=100):\n        self.epoch_log_interval = epoch_log_interval\n\n    def after_iteration(self, model, epoch:int,\n                        evals_log:xgb.callback.TrainingCallback.EvalsLog\n                        ):\n\n        if self.epoch_log_interval <= 0:\n            pass\n\n        elif (epoch %  self.epoch_log_interval == 0):\n            for data, metric in evals_log.items():\n                for metric_name, log in metric.items():\n                    score = log[-1][0] if isinstance(log[-1], tuple) else log[-1]\n                    logger.info(f\"XGBLogging epoch {epoch} dataset {data} {metric_name} {score}\")\n\n        return False\n\n# Making sklearn pipeline outputs as dataframe:-\nfrom sklearn import set_config\npd.set_option('display.max_columns', 1000)\npd.set_option('display.max_rows', 200)\nprint(f\"---> Imports- part 2 done\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:46:45.396082Z","iopub.execute_input":"2024-11-27T07:46:45.396445Z","iopub.status.idle":"2024-11-27T07:46:45.404042Z","shell.execute_reply.started":"2024-11-27T07:46:45.396411Z","shell.execute_reply":"2024-11-27T07:46:45.403028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myimports.py\n\nprint(f\"---> Seeding everything\")\n\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\n\nseed_everything(2024)\nprint(f\"\\n---> Imports done\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:46:49.175843Z","iopub.execute_input":"2024-11-27T07:46:49.176181Z","iopub.status.idle":"2024-11-27T07:46:49.182954Z","shell.execute_reply.started":"2024-11-27T07:46:49.176151Z","shell.execute_reply":"2024-11-27T07:46:49.181998Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **TRAINING ELEMENTS**","metadata":{}},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass Utils:\n    \"\"\"\n    This class creates and uses several utility methods to be used across the code\n    \"\"\";\n\n    def __init__(self):\n        pass\n\n    def ScoreMetric(self, ytrue, ypred)-> float:\n        \"\"\"\n        This method calculates the metric for the competition\n        Inputs- ytrue, ypred:- input truth and predictions\n        Output- float:- competition metric\n        \"\"\";\n\n        score = rmsle(ytrue,ypred)\n        return score\n\n    def CleanMemory(self):\n        \"This method cleans the memory off unused objects and displays the cleaned state RAM usage\"\n\n        collect();\n        libc.malloc_trim(0)\n        pid        = getpid()\n        py         = Process(pid)\n        memory_use = py.memory_info()[0] / 2. ** 30\n        return f\"\\nRAM usage = {memory_use :.4} GB\"\n\n    def DisplayAdjTbl(self, *args):\n        \"\"\"\n        This function displays pandas tables in an adjacent manner, sourced from the below link-\n        https://stackoverflow.com/questions/38783027/jupyter-notebook-display-two-pandas-tables-side-by-side\n        \"\"\"\n\n        html_str = ''\n        for df in args:\n            html_str += df.to_html()\n        display_html(html_str.replace('table','table style=\"display:inline\"'),raw=True)\n        collect()\n\n    def DisplayScores(\n        self, Scores: pd.DataFrame, TrainScores: pd.DataFrame, methods: list\n    ):\n        \"This method displays the scores and their means\"\n\n        args = \\\n        [Scores.style.format(precision = 5).\\\n         background_gradient(cmap = \"Blues\", subset = methods + [\"Ensemble\"]).\\\n         set_caption(f\"\\nOOF scores across methods and folds\\n\"),\n\n         TrainScores.style.format(precision = 5).\\\n         background_gradient(cmap = \"Pastel2\", subset = methods).\\\n         set_caption(f\"\\nTrain scores across methods and folds\\n\")\n        ];\n\n        PrintColor(f\"\\n\\n\\n---> OOF score across all methods and folds\\n\",\n                   color = Fore.LIGHTMAGENTA_EX\n                   )\n        self.DisplayAdjTbl(*args)\n\n        print('\\n')\n        display(Scores.mean().to_frame().\\\n                transpose().\\\n                style.format(precision = 5).\\\n                background_gradient(cmap = \"mako\", axis=1,\n                                    subset = Scores.columns\n                                   ).\\\n                set_caption(f\"\\nOOF mean scores across methods and folds\\n\")\n               )\n\n\nutils = Utils()\ncollect()\nprint()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:46:51.055636Z","iopub.execute_input":"2024-11-27T07:46:51.056021Z","iopub.status.idle":"2024-11-27T07:46:51.06337Z","shell.execute_reply.started":"2024-11-27T07:46:51.055986Z","shell.execute_reply":"2024-11-27T07:46:51.062368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a training.py\n\ndef MakePermImp(\n        method, mdl, X, y, ygrp,\n        myscorer, \n        n_repeats = 2,\n        state = 42,\n        ntop: int = 15,\n        **params,\n):\n    \"\"\"\n    This function makes the permutation importance for the provided model and returns the importance scores for all features\n    \n    Note-\n    myscorer - scikit-learn -> metrics -> make_scorer object with the corresponding eval metric and relevant details\n    \"\"\"\n\n    cv        = PDS(ygrp)\n    n_splits  = ygrp.nunique()\n    drop_cols = [\"Source\", \"id\", \"Id\", \"Label\", \"fold_nb\"]\n\n    for fold_nb, (train_idx, dev_idx) in tqdm(enumerate(cv.split(X, y))):\n        Xtr  = X.iloc[train_idx].drop(drop_cols, axis=1, errors = \"ignore\")\n        Xdev = X.iloc[dev_idx].drop(drop_cols, axis=1, errors = \"ignore\")\n        ytr  = y.loc[Xtr.index]\n        ydev = y.loc[Xdev.index]\n\n        model = clone(mdl)\n        sel_cols = list(Xdev.columns)\n        model.fit(Xtr, ytr)\n\n        imp_ = permutation_importance(model,\n                                      Xdev, ydev,\n                                      scoring = myscorer,\n                                      n_repeats = n_repeats,\n                                      random_state = state,\n                                      )[\"importances_mean\"]\n        imp_ = pd.Series(index = sel_cols, data = imp_)\n\n        display(\n            imp_.\\\n            sort_values(ascending = False).\\\n            head(ntop).\\\n            to_frame().\\\n            transpose().\\\n            style.\\\n            format(formatter = '{:,.3f}').\\\n            background_gradient(\"icefire\", axis=1).\\\n            set_caption(f\"Top {ntop} features\")\n            )\n\n        return imp_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:46:55.680937Z","iopub.execute_input":"2024-11-27T07:46:55.681796Z","iopub.status.idle":"2024-11-27T07:46:55.688085Z","shell.execute_reply.started":"2024-11-27T07:46:55.681755Z","shell.execute_reply":"2024-11-27T07:46:55.68716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass ModelTrainer:\n    \"This class trains the provided model on the train-test data and returns the predictions and fitted models\"\n\n    def __init__(\n        self,\n        problem_type   : str   = \"regression\", \n        es             : int   = 100,\n        target         : str   = \"\",\n        metric_lbl     : str   = \"rmse\",\n        orig_req       : bool  = False,\n        orig_all_folds : bool  = False,\n        drop_cols      : list  = [\"Source\", \"id\", \"Id\", \"Label\", \"fold_nb\"],\n        pp_preds       : bool  = False,\n    ):\n        \"\"\"\n        Key parameters-\n        es_iter  - early stopping rounds for boosted trees\n        pp_preds - do you want to post-process predictions (true/ false boolean)\n        \"\"\"\n\n        self.problem_type   = problem_type\n        self.es_iter        = es\n        self.target         = target\n        self.drop_cols      = drop_cols + [self.target]\n        self.metric_lbl     = metric_lbl\n        self.orig_req       = orig_req\n        self.orig_all_folds = orig_all_folds\n        self.pp_preds       = pp_preds\n\n    def ScoreMetric(self, ytrue, ypred):\n        \"\"\"\n        This is the metric function for the competition scoring\n        \"\"\"\n\n        if self.pp_preds :\n            y_true = self.PostProcessPreds(ytrue)\n        else:\n            y_true = ytrue\n            \n        if self.metric_lbl == \"rmse\":\n            return rmse(y_true, ypred)\n        elif self.metric_lbl == \"rmsle\":\n            return rmsle(y_true, ypred)\n\n    def PlotFtreImp(\n        self, \n        ftreimp: pd.Series, \n        method: str,\n        ntop: int = 50,\n        title_specs: dict = {'fontsize': 12,'fontweight' : 'bold','color': '#992600'},\n        **params,\n    ):\n        \"This function plots the feature importances for the model provided\"\n\n        print()\n        \n        with sns.axes_style(\"white\"):\n            fig, ax = plt.subplots(1, 1, figsize = (25, 7.5))\n    \n            ftreimp.sort_values(ascending = False).\\\n            head(ntop).\\\n            plot.bar(ax = ax, color = \"#1285c7\")\n            ax.set_title(\n                f\"Feature Importances - {method}\", \n                **title_specs\n            )\n    \n            plt.tight_layout()\n            plt.show()\n        print()\n\n    def PostProcessPreds(self, ypred):\n        \"This method post-processes predictions optionally\"\n        if self.pp_preds :\n            return np.clip(np.expm1(ypred), a_min = 20.0, a_max = 4999.00)\n        else:\n            return ypred\n\n    def LoadData(\n            self, X, y, Xtest,\n            train_idx : list = [],\n            dev_idx   : list = [],\n            ):\n        \"This method loads the train and test data for the model fold using/ not using the original data\"\n\n        if self.orig_req == False:\n            Xtr  = X.iloc[train_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ytr  = y.iloc[Xtr.index]\n            Xdev = X.iloc[dev_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ydev = y.iloc[Xdev.index]\n\n        elif self.orig_req == True and self.orig_all_folds == True:\n            Xtr  = X.iloc[train_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ytr  = y.iloc[Xtr.index]\n            Xdev = X.iloc[dev_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ydev = y.iloc[Xdev.index]\n\n            orig_x = X.query(\"Source == 'Original'\")[Xtr.columns]\n            orig_y = y.iloc[orig_x.index]\n\n            Xtr = pd.concat([Xtr, orig_x], axis = 0, ignore_index = True)\n            ytr = pd.concat([ytr, orig_y], axis = 0, ignore_index = True)\n\n        elif self.orig_req == True and self.orig_all_folds == False:\n            Xtr  = X.iloc[train_idx].drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ytr  = y.iloc[Xtr.index]\n            Xdev = X.iloc[dev_idx].query(\"Source == 'Competition'\").drop(self.drop_cols, axis=1, errors = \"ignore\")\n            ydev = y.iloc[Xdev.index]\n\n        Xt = Xtest[Xdev.columns]\n\n        print(f\"\\n---> Shapes = {Xtr.shape} {ytr.shape} -- {Xdev.shape} {ydev.shape} -- {Xt.shape}\")\n        return (Xtr, ytr, Xdev, ydev, Xt)\n    \n    def MakePreds(self, X, fitted_model):\n        \"This method creates the model predictions based on the model provided, with optional post-processing\"\n\n        if self.problem_type == \"regression\":\n            if isinstance(fitted_model, (TNC, TNR)) == True:\n                return self.PostProcessPreds(fitted_model.predict(X.to_numpy()).flatten())\n            else:\n                return self.PostProcessPreds(fitted_model.predict(X))\n        elif self.problem_type == \"binary\":\n            if isinstance(fitted_model, (TNC, TNR)) == True:\n                return self.PostProcessPreds(fitted_model.predict_proba(X.to_numpy()[:,1]).flatten())\n            else:\n                return self.PostProcessPreds(fitted_model.predict_proba(X)[:, 1])\n        elif self.problem_type == \"multiclass\":\n            if isinstance(fitted_model, (TNC, TNR)) == True:\n                return self.PostProcessPreds(fitted_model.predict_proba(X.to_numpy()))\n            else:\n                return self.PostProcessPreds(fitted_model.predict_proba(X))\n\n    def MakeOrigPreds(\n            self, orig: pd.DataFrame, fitted_models: list, n_splits : int, ygrp: pd.Series,\n            ):\n        \"This method creates the original data predictions separately only if required\"\n\n        if self.orig_req == False:\n            orig_preds = 0\n\n        elif self.orig_req == True and self.orig_all_folds == True:\n            orig_preds = 0\n            df = orig.drop(self.drop_cols, axis = 1, errors = \"ignore\")\n\n            for fitted_model in fitted_models:\n                orig_preds = orig_preds + (self.MakePreds(df, fitted_model) / n_splits)\n\n        elif self.orig_req == True and self.orig_all_folds == False:\n            len_orig   = orig.shape[0]\n            orig.index = range(len_orig)\n            orig_ygrp  = ygrp[-1 * len_orig:]\n            orig_ygrp.index = range(len_orig)\n            \n            orig_preds = np.zeros(len_orig)\n            for fold_nb, fitted_model in enumerate(fitted_models):\n                df = \\\n                orig.iloc[orig_ygrp.loc[orig_ygrp == fold_nb].index].\\\n                drop(self.drop_cols, axis=1, errors = \"ignore\")\n                \n                orig_preds[df.index] = self.MakePreds(df, fitted_model)\n                del df\n        return orig_preds\n\n    def MakeOfflineModel(\n        self, X, y, ygrp, Xtest, mdl, method,\n        test_preds_req   : bool = True,\n        ftreimp_plot_req : bool = True,\n        ntop             : int  = 50,\n        **params,\n    ):\n        \"\"\"\n        This function trains the provided model on the dataset and cross-validates appropriately\n\n        Inputs-\n        X, y, ygrp       - training data components (Xtrain, ytrain, fold_nb)\n        Xtest            - test data (optional)\n        model            - model object for training\n        method           - model method label\n        test_preds_req   - boolean flag to extract test set predictions\n        ftreimp_plot_req - boolean flag to plot tree feature importances\n        ntop             - top n features for feature importances plot\n\n        Returns-\n        oof_preds, test_preds - prediction arrays\n        fitted_models         - fitted model list for test set\n        ftreimp               - feature importances across selected features\n        mdl_best_iter         - model average best iteration across folds\n        \"\"\"\n\n        oof_preds     = np.zeros(len(X.loc[X.Source == \"Competition\"]))\n        orig_preds    = np.zeros(len(X.loc[X.Source == \"Original\"]))\n        test_preds    = []\n        mdl_best_iter = []\n        ftreimp       = 0\n\n        scores, tr_scores, fitted_models = [], [], []\n\n        if self.orig_req == True:\n            cv = PDS(ygrp)\n        elif self.orig_req == False:\n            X  = X.loc[X.Source == \"Competition\"]\n            y  = y.iloc[X.index]\n            cv = PDS(ygrp.iloc[0 : len(X)])\n\n        n_splits = ygrp.nunique()\n\n        for fold_nb, (train_idx, dev_idx) in tqdm(enumerate(cv.split(X, y))):\n            Xtr, ytr, Xdev, ydev, Xt = \\\n            self.LoadData(X, y, Xtest, train_idx, dev_idx)\n\n            model = clone(mdl)\n\n            if \"CB\" in method and self.es_iter > 0:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          verbose = 0,\n                          early_stopping_rounds = self.es_iter,\n                          )\n                best_iter = model.get_best_iteration()\n\n            elif \"LGB\" in method and self.es_iter > 0:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          callbacks = [log_evaluation(0),\n                                       early_stopping(stopping_rounds = self.es_iter, verbose = False,),\n                                       ],\n                          eval_metric = mymetric,\n                          )\n                best_iter = model.best_iteration_\n\n            elif \"XGB\" in method and self.es_iter > 0:\n                model.fit(Xtr, ytr,\n                          eval_set = [(Xdev, ydev)],\n                          verbose  = 0,\n                          )\n                best_iter = model.best_iteration\n\n            elif \"TN\" in method :\n                model.fit(\n                    Xtr.to_numpy(), ytr.to_numpy().reshape(-1,1),\n                    eval_set    = [(Xdev.to_numpy(), ydev.to_numpy().reshape(-1,1))],\n                    eval_name   = [\"dev\"],\n                    eval_metric = ['rmse'],\n                    max_epochs  = 100,\n                    patience    = 6,\n                    batch_size  = 128,\n                    virtual_batch_size = 64,\n                )\n\n            else:\n                model.fit(Xtr, ytr)\n                best_iter = -1\n\n            fitted_models.append(model)\n\n            try:\n                ftreimp += model.feature_importances_\n            except:\n                try:\n                    ftreimp += model.coef_.flatten()\n                except:\n                    pass\n            \n            dev_preds = self.MakePreds(Xdev, model)\n            oof_preds[Xdev.index] = dev_preds\n\n            train_preds  = self.MakePreds(Xtr, model)\n            tr_score     = self.ScoreMetric(ytr.values.flatten(), train_preds)\n            score        = self.ScoreMetric(ydev.values.flatten(), dev_preds)\n\n            scores.append(score)\n            tr_scores.append(tr_score)\n\n            nspace = 15 - len(method) - 2 if fold_nb <= 9 else 15 - len(method) - 1\n\n            if best_iter > 0 :\n                PrintColor(f\"{method} Fold{fold_nb} {' ' * nspace} OOF = {score:.6f} | Train = {tr_score:.6f} | Iter = {best_iter:,.0f} \")\n            else:\n                PrintColor(f\"{method} Fold{fold_nb} {' ' * nspace} OOF = {score:.6f} | Train = {tr_score:.6f} \")\n                \n            mdl_best_iter.append(best_iter)\n\n            if test_preds_req:\n                test_preds.append(self.MakePreds(Xt, model))\n            else:\n                pass\n\n        test_preds    = np.mean(np.stack(test_preds, axis = 1), axis=1)\n        ftreimp       = pd.Series(ftreimp, index = Xdev.columns)\n        mdl_best_iter = np.uint16(np.amax(mdl_best_iter))\n\n        if ftreimp_plot_req :\n            print()\n            self.PlotFtreImp(ftreimp, method = method, ntop = ntop,)\n        else:\n            pass\n\n        PrintColor(f\"\\n---> {np.mean(scores):.6f} +- {np.std(scores):.6f} | OOF\", color = Fore.RED)\n        PrintColor(f\"---> {np.mean(tr_scores):.6f} +- {np.std(tr_scores):.6f} | Train\", color = Fore.RED)\n\n        if mdl_best_iter < 0 or mdl_best_iter > 50_000:\n            pass\n        else:\n            PrintColor(\n                f\"---> Max best iteration = {mdl_best_iter :,.0f}\",\n                color = Fore.RED\n            )\n\n        if self.orig_req:\n            print(f\"---> Collecting original predictions\")\n            orig_preds = self.MakeOrigPreds(X.loc[X.Source == \"Original\"],\n                                            fitted_models,\n                                            n_splits,\n                                            ygrp,\n                                            )\n            oof_preds = np.concatenate([oof_preds, orig_preds], axis= 0)\n        else:\n            pass\n        return (fitted_models, oof_preds, test_preds, ftreimp, mdl_best_iter)\n\n    def MakeOnlineModel(\n        self, X, y, Xtest, model, method,\n        test_preds_req : bool = False,\n    ):\n        \"This method refits the model on the complete train data and returns the model fitted object and predictions\"\n\n        try:\n            model.early_stopping_rounds = None\n        except:\n            pass\n\n        if \"TN\" in method:\n            model.fit(\n                X.to_numpy(), y.to_numpy().reshape(-1,1),\n                max_epochs  = 100,\n                batch_size  = 128,\n                virtual_batch_size = 64,\n                )\n        else:\n            try:\n                model.fit(X, y, verbose = 0)\n            except:\n                model.fit(X, y,)\n\n        oof_preds  = self.MakePreds(X, model)\n        if test_preds_req:\n            test_preds = self.MakePreds(Xtest[X.columns], model)\n        else:\n            test_preds = 0\n            \n        return (model, oof_preds, test_preds)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass OptunaEnsembler:\n    \"\"\"\n    This is the Optuna ensemble class-\n    Source- https://www.kaggle.com/code/arunklenin/ps3e26-cirrhosis-survial-prediction-multiclass\n    \"\"\";\n\n    def __init__(\n        self, \n        state: int = 42, \n        ntrials: int = 300, \n        metric_obj: str = \"minimize\", \n        metric_lbl: str = \"rmse\",\n        **params\n    ):\n        self.study        = None\n        self.weights      = None\n        self.random_state = state\n        self.n_trials     = ntrials\n        self.direction    = metric_obj\n        self.metric_lbl   = metric_lbl\n        self.ScoreMetric  = utils.ScoreMetric\n\n    def _objective(\n        self, trial, y_true, y_preds\n    ):\n        \"\"\"\n        This method defines the objective function for the ensemble\n        \"\"\";\n\n        if isinstance(y_preds, pd.DataFrame) or isinstance(y_preds, np.ndarray):\n            weights = [trial.suggest_float(f\"weight{n}\", 0.001, 0.999)\n                       for n in range(y_preds.shape[-1])\n                      ]\n            axis = 1\n\n        elif isinstance(y_preds, list):\n            weights = [trial.suggest_float(f\"weight{n}\", 0.001, 0.999)\n                       for n in range(len(y_preds))\n                      ]\n            axis = 0\n\n        # Calculating the weighted prediction:-\n        weighted_pred  = np.average(np.array(y_preds), axis = axis, weights = weights)\n        score          = self.ScoreMetric(y_true, weighted_pred)\n        return score\n\n    def fit(self, y_true, y_preds):\n        \"This method fits the Optuna objective on the fold level data\";\n\n        optuna.logging.set_verbosity = optuna.logging.ERROR\n\n        self.study = \\\n        optuna.create_study(sampler    = TPESampler(seed = self.random_state),\n                            pruner     = HyperbandPruner(),\n                            study_name = \"Ensemble\",\n                            direction  = self.direction,\n                           )\n\n        obj = partial(self._objective, y_true = y_true, y_preds = y_preds)\n        self.study.optimize(obj, n_trials = self.n_trials)\n\n        if isinstance(y_preds, list):\n            self.weights = [self.study.best_params[f\"weight{n}\"] for n in range(len(y_preds))]\n\n        else:\n            self.weights = [self.study.best_params[f\"weight{n}\"] for n in range(y_preds.shape[-1])]\n\n    def predict(self, y_preds):\n        \"This method predicts using the fitted Optuna objective\";\n\n        assert self.weights is not None, 'OptunaWeights error, must be fitted before predict';\n\n        if isinstance(y_preds, list):\n            weighted_pred = np.average(np.array(y_preds), axis=0, weights = self.weights)\n\n        else:\n            weighted_pred = np.average(np.array(y_preds), axis=1, weights = self.weights)\n\n        return weighted_pred\n\n    def fit_predict(self, y_true, y_preds):\n        \"\"\"\n        This method fits the Optuna objective on the fold data, then predicts the test set\n        \"\"\";\n        self.fit(y_true, y_preds)\n        return self.predict(y_preds)\n\n    def weights(self):\n        \"This method returns the non-normalized weights for all models in a fold\"\n        return self.weights\n\nprint()\ncollect();","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:47:14.586785Z","iopub.execute_input":"2024-11-27T07:47:14.587157Z","iopub.status.idle":"2024-11-27T07:47:14.595356Z","shell.execute_reply.started":"2024-11-27T07:47:14.587122Z","shell.execute_reply":"2024-11-27T07:47:14.594231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a training.py\n\ndef NormWeights(weights: dict, methods: list):\n    \"This function normalizes the weights and returns a dataframe of normalized weights across folds and models\"\n\n    weights = pd.DataFrame.from_dict(weights).T\n    weights[\"row_sum\"] = weights.sum(axis=1)\n\n    for col in weights.columns:\n        weights[col] = weights[col] / weights[\"row_sum\"]\n\n    weights.drop(\"row_sum\", axis = 1, inplace = True, errors = \"ignore\")\n    weights.columns    = methods\n    weights.index.name = \"Fold_Nb\"\n    return weights\n\ndef MakeEnsemble(target: str, ntrials: int = 300):\n    \"This function implements the Optuna ensemble on the OOF and test prediction datasets\"\n\n    global OOF_Preds, Mdl_Preds\n\n    PrintColor(f\"\\n{'=' * 20} ENSEMBLE {'=' * 20}\\n\")\n\n    ygrp       = OOF_Preds[\"fold_nb\"]\n    cv         = PDS(ygrp)\n    oof_preds  = np.zeros(len(OOF_Preds))\n    test_preds = []\n    scores     = []\n    weights    = {}\n    drop_cols  = [\"fold_nb\", target, \"Ensemble\"]\n    n_splits   = ygrp.nunique()\n\n    for fold_nb, (_, dev_idx) in tqdm(enumerate(cv.split(OOF_Preds, OOF_Preds[target]))):\n        Xdev = OOF_Preds.iloc[dev_idx].drop(drop_cols, axis=1, errors = \"ignore\")\n        ydev = OOF_Preds.loc[dev_idx, target]\n\n        ens = OptunaEnsembler(ntrials = ntrials)\n        ens.fit(ydev, Xdev,)\n\n        dev_preds = ens.predict(Xdev)\n        score     = ens.ScoreMetric(ydev.values, dev_preds)\n        oof_preds[dev_idx] = dev_preds\n        test_preds.append(\n            ens.predict(Mdl_Preds.drop(drop_cols, axis=1, errors = \"ignore\"))\n        )\n\n        PrintColor(f\"---> {score: .6f} | Fold {fold_nb}\", color = Fore.CYAN)\n        scores.append(score)\n\n        weights[f\"Fold{fold_nb}\"] = ens.weights\n\n    PrintColor(f\"\\n---> OOF = {np.mean(scores): .6f} +- {np.std(scores): .6f} | Ensemble\",\n               color = Fore.RED\n              )\n\n    test_preds = np.mean(np.stack(test_preds, axis=1), axis=1,)\n\n    OOF_Preds[\"Ensemble\"] = oof_preds\n    Mdl_Preds[\"Ensemble\"] = test_preds\n\n    weights = \\\n    NormWeights(\n        weights,\n        methods = Mdl_Preds.drop(drop_cols, axis=1, errors = \"ignore\").columns\n    )\n\n    print(\"\\n\\n\\n\")\n    display(\n        weights.\\\n        style.\\\n        set_caption(\"Normalized weights\").\\\n        format(precision = 6).\\\n        set_properties(\n            props = \"color:red; background-color:white; font-weight: bold; border: maroon dashed 1.6px\"\n        )\n    )\n\n    return weights\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:47:19.049361Z","iopub.execute_input":"2024-11-27T07:47:19.050509Z","iopub.status.idle":"2024-11-27T07:47:19.057735Z","shell.execute_reply.started":"2024-11-27T07:47:19.050438Z","shell.execute_reply":"2024-11-27T07:47:19.056688Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **HILL CLIMBER**","metadata":{}},{"cell_type":"code","source":"%%writefile -a training.py\n\nclass HillClimber:\n    \"This class develops the Hill Climber algorithm for the provided datasets\"\n\n    def __init__(self):\n        self.ScoreMetric = utils.ScoreMetric\n\n    def DoHillClimb(\n        target:str,\n        direction:str,\n        cutoff:float,\n        neg_wgt:str,\n        OOF_Preds: pd.DataFrame,\n        Mdl_Preds: pd.DataFrame,\n        y: pd.Series,\n        **kwargs\n    ):\n        \"\"\"\n        This method performs hill-climbing on the OOF and Test predictions dataset and returns the below-\n        1. OOF ensemble predictions\n        2. Test set predictions\n        3. Score dataframe (with scores in sort-order)\n        \"\"\"\n\n        oof_df     = OOF_Preds\n        test_preds = Mdl_Preds\n    \n        # Scoring the individual models:-\n        Scores = pd.DataFrame(index = oof_df.columns, columns = ['Score'])\n    \n        for col in oof_df.columns:\n            Scores.at[col, 'Score'] = self.ScoreMetric(y, oof_df[col].values.flatten())\n    \n        # Sorting scores\n        Scores.sort_values(\n            by= 'Score',\n            ascending = [True if direction == 'minimize' else False],\n            inplace = True,\n        )\n    \n        PrintColor(f\"\\n----- Data preparation: ------ \\n\");\n        display(\n            Scores.\n            transpose().\n            style.\n            format(precision = 5)\n            )\n    \n        PrintColor(f\"\\n ----- Initiating hill-climb ----- \\n\");\n        STOP = False\n        current_best_ensemble   = oof_df.iloc[:,0]\n        current_best_test_preds = test_preds.iloc[:,0]\n        MODELS                  = oof_df.iloc[:,1:]\n    \n        if neg_wgt == \"Y\":\n            weight_range = np.arange(-0.5,0.51,0.01);\n        else:\n            weight_range = np.arange(0.01,0.51,0.01);\n    \n        history = [self.ScoreMetric(y, current_best_ensemble)]\n    \n        i=0\n    \n        # Hill climbing algorithm:-\n        while not STOP:\n            i+=1\n    \n            potential_new_best_cv_score = self.ScoreMetric(y, current_best_ensemble)\n            k_best, wgt_best = None, None\n    \n            for k in MODELS:\n                for wgt in weight_range:\n                    potential_ensemble = (1- wgt) * current_best_ensemble + wgt * MODELS[k]\n                    cv_score = self.ScoreMetric(y, potential_ensemble)\n    \n                    if direction == 'minimize':\n                        if cv_score < potential_new_best_cv_score:\n                            potential_new_best_cv_score, k_best, wgt_best = cv_score, k, wgt\n    \n                    if direction == 'maximize':\n                        if cv_score > potential_new_best_cv_score:\n                            potential_new_best_cv_score, k_best, wgt_best = cv_score, k, wgt\n    \n            if k_best is not None:\n                current_best_ensemble   = (1- wgt_best) * current_best_ensemble + wgt_best * MODELS[k_best]\n                current_best_test_preds = (1- wgt_best) * current_best_test_preds + wgt_best * test_preds[k_best]\n                MODELS.drop(k_best, axis=1, inplace=True)\n    \n                if MODELS.shape[1]==0:  STOP = True\n    \n                num_space = 50 - len(k_best) if i <= 9 else 49 - len(k_best)\n                PrintColor(f\" {i}.{k_best} {' ' * num_space} Weight = {wgt_best: .4f} {' ' * 5} Score = {potential_new_best_cv_score:.6f}\",\n                           color = Fore.CYAN\n                          )\n                del num_space\n    \n                history.append(potential_new_best_cv_score)\n    \n            else:\n                STOP = True\n    \n        return (current_best_ensemble, current_best_test_preds, Scores)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:48:10.586452Z","iopub.execute_input":"2024-11-27T07:48:10.586842Z","iopub.status.idle":"2024-11-27T07:48:10.594896Z","shell.execute_reply.started":"2024-11-27T07:48:10.58681Z","shell.execute_reply":"2024-11-27T07:48:10.593835Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **LAMA IMPORTS**","metadata":{}},{"cell_type":"code","source":"%%writefile -a myimports_lama.py\n\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\n\nimport os\nfrom os import path, walk, getpid\nfrom psutil import Process\nimport re\nfrom collections import Counter\nfrom itertools import product\nfrom gc import collect\n\nimport ctypes\nlibc = ctypes.CDLL(\"libc.so.6\")\n\nfrom IPython.display import display_html, clear_output\nfrom pprint import pprint\nfrom functools import partial\nfrom copy import deepcopy\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom colorama import Fore, Style, init\nfrom tqdm.notebook import tqdm\nimport tempfile\n\n# Essential DS libraries\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport polars.selectors as cs\nfrom sklearn.metrics import (log_loss, roc_auc_score, accuracy_score, f1_score, \nroot_mean_squared_error as rmse, root_mean_squared_log_error as rmsle)\n\nfrom sklearn.model_selection import PredefinedSplit as PDS\nfrom sklearn.preprocessing import RobustScaler\nimport torch\nimport torch.nn as nn\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\n\n# LightAutoML presets, task and report generation\nfrom lightautoml.automl.presets.tabular_presets import TabularAutoML\nfrom lightautoml.tasks import Task\n\n# Color printing\ndef PrintColor(text: str, color = Fore.BLUE, style = Style.BRIGHT):\n    \"Prints color outputs using colorama using a text F-string\"\n    print(style + color + text + Style.RESET_ALL)\n\nprint(f\"---> CUDA available = {torch.cuda.is_available()}\\n\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:48:49.684685Z","iopub.execute_input":"2024-11-27T07:48:49.685043Z","iopub.status.idle":"2024-11-27T07:48:49.691698Z","shell.execute_reply.started":"2024-11-27T07:48:49.685012Z","shell.execute_reply":"2024-11-27T07:48:49.690639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myimports_lama.py\n\nclass Utils:\n    \"\"\"\n    This class creates and uses several utility methods to be used across the code\n    \"\"\";\n\n    def __init__(self):\n        pass\n\n    def ScoreMetric(self, ytrue, ypred)-> float:\n        \"\"\"\n        This method calculates the metric for the competition\n        Inputs- ytrue, ypred:- input truth and predictions\n        Output- float:- competition metric\n        \"\"\";\n        return rmse(ytrue, ypred)\n\n    def CleanMemory(self):\n        \"This method cleans the memory off unused objects and displays the cleaned state RAM usage\"\n\n        collect();\n        libc.malloc_trim(0)\n        pid        = getpid()\n        py         = Process(pid)\n        memory_use = py.memory_info()[0] / 2. ** 30\n        return f\"\\nRAM usage = {memory_use :.4} GB\"\n\n    def DisplayAdjTbl(self, *args):\n        \"\"\"\n        This function displays pandas tables in an adjacent manner, sourced from the below link-\n        https://stackoverflow.com/questions/38783027/jupyter-notebook-display-two-pandas-tables-side-by-side\n        \"\"\"\n\n        html_str = ''\n        for df in args:\n            html_str += df.to_html()\n        display_html(html_str.replace('table','table style=\"display:inline\"'),raw=True)\n        collect()\n\n    def DisplayScores(\n        self, Scores: pd.DataFrame, TrainScores: pd.DataFrame, methods: list\n    ):\n        \"This method displays the scores and their means\"\n\n        args = \\\n        [Scores.style.format(precision = 5).\\\n         background_gradient(cmap = \"Blues\", subset = methods + [\"Ensemble\"]).\\\n         set_caption(f\"\\nOOF scores across methods and folds\\n\"),\n\n         TrainScores.style.format(precision = 5).\\\n         background_gradient(cmap = \"Pastel2\", subset = methods).\\\n         set_caption(f\"\\nTrain scores across methods and folds\\n\")\n        ];\n\n        PrintColor(f\"\\n\\n\\n---> OOF score across all methods and folds\\n\",\n                   color = Fore.LIGHTMAGENTA_EX\n                   )\n        self.DisplayAdjTbl(*args)\n\n        print('\\n')\n        display(Scores.mean().to_frame().\\\n                transpose().\\\n                style.format(precision = 5).\\\n                background_gradient(cmap = \"mako\", axis=1,\n                                    subset = Scores.columns\n                                   ).\\\n                set_caption(f\"\\nOOF mean scores across methods and folds\\n\")\n               )\n\n\nutils = Utils()\ncollect()\nprint()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:48:52.671708Z","iopub.execute_input":"2024-11-27T07:48:52.672197Z","iopub.status.idle":"2024-11-27T07:48:52.680204Z","shell.execute_reply.started":"2024-11-27T07:48:52.672148Z","shell.execute_reply":"2024-11-27T07:48:52.678707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myimports_ag.py\n\nimport numpy as np, pandas as pd\nimport polars as pl\nimport polars.selectors as cs\nimport re, os, joblib, logging\nfrom gc import collect\n\nfrom IPython.display import display_html, clear_output\nfrom pprint import pprint\nfrom tqdm.notebook import tqdm\nfrom colorama import Fore, Back, Style\nfrom os import path, walk, getpid\nfrom psutil import Process\nimport ctypes\nlibc = ctypes.CDLL(\"libc.so.6\")\n\nfrom warnings import filterwarnings\nfilterwarnings(\"ignore\")\n\nfrom sklearn.model_selection import StratifiedKFold as SKF, GroupKFold as GKF\nfrom sklearn.metrics import (roc_auc_score, mean_squared_error as mse, \nroot_mean_squared_error as rmse, root_mean_squared_log_error as rmsle)\nfrom autogluon.tabular import TabularPredictor, TabularDataset\n\n# Color printing\ndef PrintColor(text: str, color = Fore.BLUE, style = Style.BRIGHT):\n    \"Prints color outputs using colorama using a text F-string\"\n    print(style + color + text + Style.RESET_ALL)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:48:56.111288Z","iopub.execute_input":"2024-11-27T07:48:56.111685Z","iopub.status.idle":"2024-11-27T07:48:56.118476Z","shell.execute_reply.started":"2024-11-27T07:48:56.111649Z","shell.execute_reply":"2024-11-27T07:48:56.117334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile -a myimports_ag.py\n\nclass Utils:\n    \"\"\"\n    This class creates and uses several utility methods to be used across the code\n    \"\"\"\n\n    def __init__(self):\n        pass\n\n    def ScoreMetric(self, ytrue, ypred)-> float:\n        \"\"\"\n        This method calculates the metric for the competition\n        Inputs- ytrue, ypred:- input truth and predictions\n        Output- float:- competition metric\n        \"\"\";\n        return rmse(ytrue, ypred)\n\n    def CleanMemory(self):\n        \"This method cleans the memory off unused objects and displays the cleaned state RAM usage\"\n\n        collect();\n        libc.malloc_trim(0)\n        pid        = getpid()\n        py         = Process(pid)\n        memory_use = py.memory_info()[0] / 2. ** 30\n        return f\"\\nRAM usage = {memory_use :.4} GB\"\n\n    def DisplayAdjTbl(self, *args):\n        \"\"\"\n        This function displays pandas tables in an adjacent manner, sourced from the below link-\n        https://stackoverflow.com/questions/38783027/jupyter-notebook-display-two-pandas-tables-side-by-side\n        \"\"\"\n\n        html_str = ''\n        for df in args:\n            html_str += df.to_html()\n        display_html(html_str.replace('table','table style=\"display:inline\"'),raw=True)\n        collect()\n\n    def DisplayScores(\n        self, Scores: pd.DataFrame, TrainScores: pd.DataFrame, methods: list\n    ):\n        \"This method displays the scores and their means\"\n\n        args = \\\n        [Scores.style.format(precision = 5).\\\n         background_gradient(cmap = \"Blues\", subset = methods + [\"Ensemble\"]).\\\n         set_caption(f\"\\nOOF scores across methods and folds\\n\"),\n\n         TrainScores.style.format(precision = 5).\\\n         background_gradient(cmap = \"Pastel2\", subset = methods).\\\n         set_caption(f\"\\nTrain scores across methods and folds\\n\")\n        ];\n\n        PrintColor(f\"\\n\\n\\n---> OOF score across all methods and folds\\n\",\n                   color = Fore.LIGHTMAGENTA_EX\n                   )\n        self.DisplayAdjTbl(*args)\n\n        print('\\n')\n        display(Scores.mean().to_frame().\\\n                transpose().\\\n                style.format(precision = 5).\\\n                background_gradient(cmap = \"mako\", axis=1,\n                                    subset = Scores.columns\n                                   ).\\\n                set_caption(f\"\\nOOF mean scores across methods and folds\\n\")\n               )\n\n\nutils = Utils()\ncollect()\nprint()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T07:48:59.125269Z","iopub.execute_input":"2024-11-27T07:48:59.125678Z","iopub.status.idle":"2024-11-27T07:48:59.132975Z","shell.execute_reply.started":"2024-11-27T07:48:59.125643Z","shell.execute_reply":"2024-11-27T07:48:59.131972Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **STANDARD PREPROCESSOR**","metadata":{}},{"cell_type":"code","source":"%%writefile pp.py\n\nclass Preprocessor():\n    \"\"\"\n    This class aims to do the below-\n    1. Read the datasets\n    2. In this case, we need to process the original data target column to be compatible with the competition dataset\n    3. Check information and description\n    4. Check unique values and nulls\n    5. Collate starting features \n    \"\"\";\n    \n    def __init__(self):\n        self.train             = pd.read_csv(os.path.join(CFG.ip_path,\"train.csv\"), index_col = 'id')\n        self.test              = pd.read_csv(os.path.join(CFG.ip_path ,\"test.csv\"), index_col = 'id')\n        self.target            = CFG.target \n        self.conjoin_orig_data = True if CFG.nb_orig > 0 else False\n        self.dtl_preproc_req   = CFG.dtl_preproc_req\n        self.test_req          = CFG.test_req\n        self.cv                = cv_selector[CFG.mdlcv_mthd]\n         \n        self.original = pd.read_csv(CFG.orig_path)\n        self.original.index = range(len(self.original))\n        self.original.index.name = \"id\"    \n        self.original = self.original[self.train.columns]\n\n        self.sub_fl = pd.read_csv(os.path.join(CFG.ip_path, \"sample_submission.csv\"))\n        PrintColor(f\"Data shapes - train-test-original | {self.train.shape} {self.test.shape} {self.original.shape}\")\n        \n        for tbl in [self.train, self.original, self.test]:\n            obj_cols      = tbl.select_dtypes(include = [\"object\", \"category\"]).columns\n            tbl.columns   = tbl.columns.str.replace(r\"\\(|\\)|\\.|\\?|/|\\s+\",\"\", regex = True)\n            \n    def _VisualizeDF(self):\n        \"This method visualizes the heads for the train, test and original data\"\n        \n        PrintColor(f\"\\nTrain set head\", color = Fore.CYAN)\n        display(self.train.head(5).style.format(precision = 3))\n        \n        PrintColor(f\"\\nTest set head\", color = Fore.CYAN)\n        display(self.test.head(5).style.format(precision = 3))\n        \n        PrintColor(f\"\\nOriginal set head\", color = Fore.CYAN)\n        display(self.original.head(5).style.format(precision = 3))\n              \n    def _AddSourceCol(self):\n        self.train['Source']    = \"Competition\";\n        self.test['Source']     = \"Competition\";\n        self.original['Source'] = 'Original';\n        \n        self.strt_ftre = self.test.columns;\n        return self;\n          \n    def _CollateInfoDesc(self):\n        if self.dtl_preproc_req == \"Y\":\n            PrintColor(f\"\\n{'-' * 20} Information and description {'-' * 20}\\n\", color = Fore.MAGENTA);\n\n            # Creating dataset information and description:\n            for lbl, df in {'Train': self.train, 'Test': self.test, 'Original': self.original}.items():\n                PrintColor(f\"\\n{lbl} description\\n\");\n                display(df.describe(percentiles= [0.05, 0.25, 0.50, 0.75, 0.9, 0.95, 0.99]).\\\n                        transpose().\\\n                        drop(columns = ['count'], errors = 'ignore').\\\n                        drop([self.target], axis=0, errors = 'ignore').\\\n                        style.format(formatter = '{:,.2f}').\\\n                        background_gradient(cmap = 'Blues')\n                       );\n\n                PrintColor(f\"\\n{lbl} information\\n\");\n                display(df.info());\n                collect();\n        return self;\n    \n    def _CollateUnqNull(self):\n        \n        if self.dtl_preproc_req == \"Y\":\n            # Dislaying the unique values across train-test-original:-\n            PrintColor(f\"\\nUnique and null values\\n\")\n            _ = pd.concat([self.train[self.strt_ftre].nunique(), \n                           self.test[self.strt_ftre].nunique(), \n                           self.original[self.strt_ftre].nunique(),\n                           self.train[self.strt_ftre].isna().sum(axis=0),\n                           self.test[self.strt_ftre].isna().sum(axis=0),\n                           self.original[self.strt_ftre].isna().sum(axis=0)\n                          ], \n                          axis=1)\n            _.columns = ['Train_Nunq', 'Test_Nunq', 'Original_Nunq', \n                         'Train_Nulls', 'Test_Nulls', 'Original_Nulls'\n                        ]\n            display(_.T.style.background_gradient(cmap = 'Blues', axis=1).\\\n                    format(formatter = '{:,.0f}')\n                   )\n            \n        return self;\n       \n    def _ConjoinTrainOrig(self):\n        if self.conjoin_orig_data :\n            PrintColor(f\"\\n\\nTrain shape before conjoining with original = {self.train.shape}\")\n            train = pd.concat([self.train] + [self.original] * CFG.nb_orig, \n                              axis=0, \n                              ignore_index = True\n                             )\n            PrintColor(f\"Train shape after conjoining with original= {train.shape}\")\n\n            train.index = range(len(train))\n            train.index.name = 'id'\n\n        else:\n            PrintColor(f\"\\nWe are using the competition training data only\")\n            train = self.train\n        return train\n       \n    def DoPreprocessing(self):\n        self._VisualizeDF()\n        self._AddSourceCol()\n        self._CollateInfoDesc()\n        self._CollateUnqNull()\n        self.train = self._ConjoinTrainOrig()\n\n        self.train = self.train.dropna(subset = [self.target])\n        self.train.index = range(len(self.train))\n        \n        self.cat_cols  = list(self.test.drop(\"Source\", axis=1).select_dtypes(\"object\").columns)\n        self.cont_cols = [c for c in self.strt_ftre if c not in self.cat_cols + ['Source']]\n        return self \n            \ncollect();\nprint();","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}