{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":213026665,"sourceType":"kernelVersion"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":7114.455447,"end_time":"2024-12-02T13:14:26.97532","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-12-02T11:15:52.519873","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports and configs","metadata":{"papermill":{"duration":0.009186,"end_time":"2024-12-02T11:15:55.401053","exception":false,"start_time":"2024-12-02T11:15:55.391867","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!pip install -q scikit-learn==1.5.2","metadata":{"trusted":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightgbm import LGBMRegressor, early_stopping, log_evaluation\nfrom ydf import GradientBoostedTreesLearner\nfrom catboost import CatBoostRegressor, Pool\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.metrics import root_mean_squared_error\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.linear_model import Ridge, Lasso\nfrom sklearn.model_selection import KFold\nfrom sklearn.base import clone\nimport matplotlib.pyplot as plt\nimport contextlib, io\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\nimport warnings\nimport pickle\nimport shutil\nimport optuna\nimport json\nimport glob\nimport ydf\nimport os\nimport gc\n\nydf.verbose(2)\nwarnings.filterwarnings(\"ignore\")","metadata":{"papermill":{"duration":5.125566,"end_time":"2024-12-02T11:16:00.535519","exception":false,"start_time":"2024-12-02T11:15:55.409953","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    train_path = \"/kaggle/input/playground-series-s4e12/train.csv\"\n    test_path = \"/kaggle/input/playground-series-s4e12/test.csv\"\n    sample_sub_path = \"/kaggle/input/playground-series-s4e12/sample_submission.csv\"\n    \n    target = \"Premium Amount\"\n    metric = \"RMSLE\"\n    n_folds = 10\n    seed = 42\n\n    n_optuna_trials = 100","metadata":{"papermill":{"duration":0.018849,"end_time":"2024-12-02T11:16:00.565557","exception":false,"start_time":"2024-12-02T11:16:00.546708","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data loading and preprocessing","metadata":{"papermill":{"duration":0.007855,"end_time":"2024-12-02T11:16:00.581781","exception":false,"start_time":"2024-12-02T11:16:00.573926","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def get_data(model):\n    train = pd.read_csv(CFG.train_path, index_col=\"id\")\n    test = pd.read_csv(CFG.test_path, index_col=\"id\")\n    \n    train[\"Policy Start Date\"] = pd.to_datetime(train[\"Policy Start Date\"])\n    test[\"Policy Start Date\"] = pd.to_datetime(test[\"Policy Start Date\"])\n    train[\"Year\"] = train[\"Policy Start Date\"].dt.year\n    test[\"Year\"] = test[\"Policy Start Date\"].dt.year\n    train.drop(\"Policy Start Date\", axis=1, inplace=True)\n    test.drop(\"Policy Start Date\", axis=1, inplace=True)\n\n    cat_cols = test.select_dtypes(include=\"object\").columns.tolist()\n    train[cat_cols] = train[cat_cols].astype(str).astype(\"category\")\n    test[cat_cols] = test[cat_cols].astype(str).astype(\"category\")\n\n    if model in [\"histgb\"]:\n        encoder = OrdinalEncoder(handle_unknown=\"use_encoded_value\", unknown_value=-1)\n        train[cat_cols] = encoder.fit_transform(train[cat_cols])\n        test[cat_cols] = encoder.transform(test[cat_cols])\n\n    X = train.drop(CFG.target, axis=1)\n    y = np.log1p(train[CFG.target])\n    X_test = test\n\n    return X, y, X_test","metadata":{"papermill":{"duration":0.029528,"end_time":"2024-12-02T11:16:28.139548","exception":false,"start_time":"2024-12-02T11:16:28.11002","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training base models","metadata":{"papermill":{"duration":0.012551,"end_time":"2024-12-02T11:16:28.1651","exception":false,"start_time":"2024-12-02T11:16:28.152549","status":"completed"},"tags":[]}},{"cell_type":"code","source":"class Trainer:\n    def __init__(self, model, config=CFG):\n        self.model = model\n        self.config = config\n        \n    def train(self, X, y, X_test, model_name):       \n        print(f\"Training {self.model.__class__.__name__}\\n\")\n                \n        scores = []\n        oof_preds = np.zeros(X.shape[0])\n        test_preds = np.zeros(X_test.shape[0])\n        coeffs = np.zeros((1, X.shape[1]))\n        \n        split = KFold(n_splits=self.config.n_folds, random_state=self.config.seed, shuffle=True).split(X, y)\n        for fold_idx, (train_idx, val_idx) in enumerate(split):\n            X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n            y_train, y_val = y[train_idx], y[val_idx]\n            \n            y_preds, temp_test_preds, score, _coeff = self._fit_predict(fold_idx, X_train, y_train, X_val, y_val, X_test)\n            oof_preds[val_idx] = y_preds\n            test_preds += temp_test_preds / self.config.n_folds\n            scores.append(score)\n            if _coeff is not None:\n                coeffs += _coeff / self.config.n_folds\n            \n            del X_train, y_train, X_val, y_val, y_preds, temp_test_preds\n            gc.collect() \n            \n        overall_score = root_mean_squared_error(y, np.maximum(oof_preds, 0))\n        average_score = np.mean(scores)\n        \n        self._save_results(oof_preds, test_preds, overall_score, model_name)\n        print(f\"\\n------ Overall {CFG.metric}: {overall_score:.6f} - Average {CFG.metric}: {average_score:.6f}\")\n        \n        return oof_preds, test_preds, scores, coeffs\n    \n    def tune(self, X, y):       \n        scores = []\n        split = KFold(n_splits=self.config.n_folds, random_state=self.config.seed, shuffle=True).split(X, y)\n        for fold_idx, (train_idx, val_idx) in enumerate(split):\n            X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n            y_train, y_val = y[train_idx], y[val_idx]\n            \n            model = clone(self.model)\n            model.fit(X_train, y_train)\n            \n            y_preds = model.predict(X_val)\n            score = root_mean_squared_error(y_val, np.maximum(y_preds, 0))\n            scores.append(score)\n            \n            del X_train, y_train, X_val, y_val, y_preds\n            gc.collect()\n            \n        return np.mean(scores)\n        \n    def _save_results(self, oof_preds, test_preds, cv_score, model_name):            \n        if isinstance(self.model, (Ridge, Lasso)):\n            pass\n        else:\n            prefix = model_name.replace(\"-\", \"_\")\n            os.makedirs(f\"oof/{model_name}\", exist_ok=True)\n            \n            with open(f\"oof/{model_name}/{prefix}_oof_preds_{cv_score:.6f}.pkl\", \"wb\") as f:\n                pickle.dump(np.expm1(oof_preds), f)\n                \n            with open(f\"oof/{model_name}/{prefix}_test_preds_{cv_score:.6f}.pkl\", \"wb\") as f:\n                pickle.dump(np.expm1(test_preds), f)\n            \n    def _fit_predict(self, fold_idx, X_train, y_train, X_val, y_val, X_test):\n        if isinstance(self.model, LGBMRegressor):\n            y_preds, test_preds = self._fit_lightgbm(X_train, y_train, X_val, y_val, X_test)\n        elif isinstance(self.model, XGBRegressor):\n            y_preds, test_preds = self._fit_xgboost(X_train, y_train, X_val, y_val, X_test)\n        elif isinstance(self.model, CatBoostRegressor):\n            y_preds, test_preds = self._fit_catboost(X_train, y_train, X_val, y_val, X_test)\n        else:\n            model = clone(self.model)\n            model.fit(X_train, y_train)\n            y_preds = model.predict(X_val)\n            test_preds = model.predict(X_test)\n            \n        score = root_mean_squared_error(y_val, np.maximum(y_preds, 0))\n        \n        if isinstance(self.model, (CatBoostRegressor, LGBMRegressor, XGBRegressor)):\n            print(f\"\\n--- Fold {fold_idx + 1} - {CFG.metric}: {score:.6f}\\n\\n\")\n        else:\n            print(f\"--- Fold {fold_idx + 1} - {CFG.metric}: {score:.6f}\")\n            \n        coeff = model.coef_ if isinstance(self.model, (Ridge, Lasso)) else None\n        return y_preds, test_preds, score, coeff\n        \n    def _fit_lightgbm(self, X_train, y_train, X_val, y_val, X_test):\n        model = clone(self.model)\n        model.fit(\n            X_train, \n            y_train, \n            eval_metric=\"rmse\",\n            eval_set=[(X_val, y_val)], \n            callbacks=[\n                log_evaluation(period=200), \n                early_stopping(stopping_rounds=100)\n            ]\n        )\n        \n        y_preds = model.predict(X_val)\n        test_preds = model.predict(X_test)\n        del model\n        \n        return y_preds, test_preds\n    \n    def _fit_xgboost(self, X_train, y_train, X_val, y_val, X_test):\n        model = clone(self.model)\n        model.fit(\n                X_train, \n                y_train, \n                eval_set=[(X_val, y_val)], \n                verbose=200\n            )\n        \n        y_preds = model.predict(X_val)\n        test_preds = model.predict(X_test)\n        del model\n        \n        return y_preds, test_preds\n    \n    def _fit_catboost(self, X_train, y_train, X_val, y_val, X_test):\n        model = clone(self.model)\n\n        cat_cols = X_train.select_dtypes(include=\"category\").columns.tolist()\n        if len(cat_cols) > 0:\n            train_pool = Pool(X_train, y_train, cat_features=cat_cols)\n            val_pool = Pool(X_val, y_val, cat_features=cat_cols)\n            test_pool = Pool(X_test, cat_features=cat_cols)\n        else:\n            train_pool = Pool(X_train, y_train)\n            val_pool = Pool(X_val, y_val)\n            test_pool = Pool(X_test)\n\n        model.fit(\n            X=train_pool, \n            eval_set=val_pool, \n            verbose=200, \n            early_stopping_rounds=100,\n            use_best_model=True\n        )\n        \n        y_preds = model.predict(val_pool)\n        test_preds = model.predict(test_pool)\n        del model\n        \n        return y_preds, test_preds","metadata":{"papermill":{"duration":0.036819,"end_time":"2024-12-02T11:16:28.214856","exception":false,"start_time":"2024-12-02T11:16:28.178037","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_results(y_preds, y, model_name):\n    y = np.expm1(y)\n    y_preds = np.expm1(y_preds)\n    \n    sns.set_style(\"whitegrid\")\n    fig, axes = plt.subplots(1, 2, figsize=(15, 5))\n\n    sns.ecdfplot(y_preds, label=\"Predictions\", ax=axes[0])\n    sns.ecdfplot(y, label=\"Target\", ax=axes[0])\n    axes[0].set_title(f\"{model_name} CDF\")\n    axes[0].legend(loc=\"best\")\n    axes[0].set_ylim(0, 1.1)\n\n    sns.histplot(y_preds, kde=True, ax=axes[1], label='Predictions')\n    sns.histplot(y, kde=True, ax=axes[1], label='Target')\n    axes[1].set_title(f\"{model_name} prediction and target distributions\")\n    axes[1].legend(loc=\"best\")\n\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def save_sub(name, test_preds, score):\n    sub = pd.read_csv(CFG.sample_sub_path)\n    sub[CFG.target] = np.expm1(test_preds)\n    sub.to_csv(f\"sub_{name}_{score:.6f}.csv\", index=False)\n    return sub","metadata":{"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"histgb_params = {\n    \"l2_regularization\": 97.57932943804968,\n    \"learning_rate\": 0.04240839552862035,\n    \"max_depth\": 354,\n    \"max_features\": 0.8247981165620517,\n    \"max_iter\": 2310,\n    \"max_leaf_nodes\": 97,\n    \"min_samples_leaf\": 36,\n    \"random_state\": 42\n}\n\nlgbm_params = {\n    \"boosting_type\": \"gbdt\",\n    \"colsample_bytree\": 0.9872314257127408,\n    \"learning_rate\": 0.05343422655561087,\n    \"min_child_samples\": 63,\n    \"min_child_weight\": 0.6687637013029941,\n    \"n_estimators\": 5000,\n    \"n_jobs\": -1,\n    \"num_leaves\": 79,\n    \"random_state\": 42,\n    \"reg_alpha\": 13.597853416168482,\n    \"reg_lambda\": 30.849379423298384,\n    \"subsample\": 0.158082661630902,\n    \"verbose\": -1\n}\n\nlgbm_goss_params = {\n    \"boosting_type\": \"goss\",\n    \"colsample_bytree\": 0.99540568767702,\n    \"learning_rate\": 0.009551897950587227,\n    \"min_child_samples\": 19,\n    \"min_child_weight\": 0.011150633860352643,\n    \"n_estimators\": 5000,\n    \"n_jobs\": -1,\n    \"num_leaves\": 82,\n    \"random_state\": 42,\n    \"reg_alpha\": 10.213340637119938,\n    \"reg_lambda\": 60.86384994349396,\n    \"subsample\": 0.9974198155466328,\n    \"verbose\": -1\n}\n\nxgb_params = {\n    \"colsample_bylevel\": 0.992856916206705,\n    \"colsample_bynode\": 0.9544347759352364,\n    \"colsample_bytree\": 0.9961529490270978,\n    \"early_stopping_rounds\": 100,\n    \"enable_categorical\": True,\n    \"eval_metric\": \"rmse\",\n    \"gamma\": 4.054739566104445,\n    \"learning_rate\": 0.03889684314255277,\n    \"max_depth\": 19,\n    \"max_leaves\": 83,\n    \"min_child_weight\": 23,\n    \"n_estimators\": 5000,\n    \"n_jobs\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 83.07529157853307,\n    \"reg_lambda\": 2.67831942399408,\n    \"subsample\": 0.9438059983649152,\n    \"verbosity\": 0\n}\n\ncb_params = {\n    \"border_count\": 221,\n    \"colsample_bylevel\": 0.7769024750363183,\n    \"depth\": 8,\n    \"eval_metric\": \"RMSE\",\n    \"iterations\": 5000,\n    \"l2_leaf_reg\": 13.06280331208708,\n    \"learning_rate\": 0.08084212863274641,\n    \"min_child_samples\": 50,\n    \"random_state\": 42,\n    \"random_strength\": 0.8965166707564866,\n    \"subsample\": 0.8932974737961973,\n    \"verbose\": False\n}\n\nydf_params = {\n    \"num_trees\": 1000,\n    \"max_depth\": 8\n}","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"papermill":{"duration":0.029097,"end_time":"2024-12-02T11:16:28.293517","exception":false,"start_time":"2024-12-02T11:16:28.26442","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores = {}\noof_preds = {}\ntest_preds = {}","metadata":{"papermill":{"duration":0.021554,"end_time":"2024-12-02T11:16:28.327978","exception":false,"start_time":"2024-12-02T11:16:28.306424","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## HistGradientBoosting","metadata":{}},{"cell_type":"code","source":"X, y, X_test = get_data(\"histgb\")\nhistgb_model = HistGradientBoostingRegressor(**histgb_params)\nhistgb_trainer = Trainer(histgb_model)\noof_preds[\"HistGB\"], test_preds[\"HistGB\"], scores[\"HistGB\"], _ = histgb_trainer.train(X, y, X_test, \"histgb\")","metadata":{"papermill":{"duration":218.820817,"end_time":"2024-12-02T12:36:34.682349","exception":false,"start_time":"2024-12-02T12:32:55.861532","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_results(oof_preds[\"HistGB\"], y, \"HistGB\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## LightGBM (GBDT)","metadata":{}},{"cell_type":"code","source":"X, y, X_test = get_data(\"lgbm-gbdt\")\nlgbm_model = LGBMRegressor(**lgbm_params)\nlgbm_trainer = Trainer(lgbm_model)\noof_preds[\"LightGBM\"], test_preds[\"LightGBM\"], scores[\"LightGBM\"], _ = lgbm_trainer.train(X, y, X_test, \"lgbm-gbdt\")","metadata":{"papermill":{"duration":143.399369,"end_time":"2024-12-02T11:39:56.731875","exception":false,"start_time":"2024-12-02T11:37:33.332506","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_results(oof_preds[\"LightGBM\"], y, \"LightGBM\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## LightGBM (GOSS)","metadata":{}},{"cell_type":"code","source":"X, y, X_test = get_data(\"lgbm-goss\")\nlgbm_goss_model = LGBMRegressor(**lgbm_goss_params)\nlgbm_goss_trainer = Trainer(lgbm_goss_model)\noof_preds[\"LightGBM (goss)\"], test_preds[\"LightGBM (goss)\"], scores[\"LightGBM (goss)\"], _ = lgbm_goss_trainer.train(X, y, X_test, \"lgbm-goss\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_results(oof_preds[\"LightGBM (goss)\"], y, \"LightGBM (goss)\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## XGBoost","metadata":{}},{"cell_type":"code","source":"X, y, X_test = get_data(\"xgb\")\nxgb_model = XGBRegressor(**xgb_params)\nxgb_trainer = Trainer(xgb_model)\noof_preds[\"XGBoost\"], test_preds[\"XGBoost\"], scores[\"XGBoost\"], _ = xgb_trainer.train(X, y, X_test, \"xgb\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_results(oof_preds[\"XGBoost\"], y, \"XGBoost\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CatBoost","metadata":{}},{"cell_type":"code","source":"X, y, X_test = get_data(\"cb\")\ncb_model = CatBoostRegressor(**cb_params)\ncb_trainer = Trainer(cb_model)\noof_preds[\"CatBoost\"], test_preds[\"CatBoost\"], scores[\"CatBoost\"], _ = cb_trainer.train(X, y, X_test, \"cb\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_results(oof_preds[\"CatBoost\"], y, \"CatBoost\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Yggdrasil Decision Forests","metadata":{}},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/competitions/playground-series-s4e12/discussion/549959#3064621\n\ndef YDFRegressor(learner_class):\n\n    class YDFXRegressor(BaseEstimator, RegressorMixin):\n\n        def __init__(self, params={}):\n            self.params = params\n\n        def fit(self, X, y):\n            assert isinstance(X, pd.DataFrame)\n            assert isinstance(y, pd.Series)\n            target = y.name\n            params = self.params.copy()\n            params['label'] = target\n            params['task'] = ydf.Task.REGRESSION\n            X = pd.concat([X, y], axis=1)\n            with contextlib.redirect_stderr(io.StringIO()), contextlib.redirect_stdout(io.StringIO()):\n                self.model = learner_class(**params).train(X)\n            return self\n\n        def predict(self, X):\n            assert isinstance(X, pd.DataFrame)\n            with contextlib.redirect_stderr(io.StringIO()), contextlib.redirect_stdout(io.StringIO()):\n                return self.model.predict(X)\n\n    return YDFXRegressor","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/competitions/playground-series-s4e12/discussion/549959#3064621\n\nX, y, X_test = get_data(\"ydf\")\nydf_model = YDFRegressor(GradientBoostedTreesLearner)(ydf_params)\nydf_trainer = Trainer(ydf_model)\noof_preds[\"Yggdrasil DF\"], test_preds[\"Yggdrasil DF\"], scores[\"Yggdrasil DF\"], _ = ydf_trainer.train(X, y, X_test, \"ydf\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_results(oof_preds[\"Yggdrasil DF\"], y, \"Yggdrasil DF\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## AutoGluon","metadata":{}},{"cell_type":"code","source":"oof_preds_files = glob.glob(f'/kaggle/input/s04e12-insurance-premium-prediction-autogluon/*_oof_preds_*.pkl')\ntest_preds_files = glob.glob(f'/kaggle/input/s04e12-insurance-premium-prediction-autogluon/*_test_preds_*.pkl')\n\nag_oof_preds = np.log1p(pickle.load(open(oof_preds_files[0], 'rb')))\nag_test_preds = np.log1p(pickle.load(open(test_preds_files[0], 'rb')))\n\nag_scores = []\nsplit = KFold(n_splits=CFG.n_folds, random_state=CFG.seed, shuffle=True).split(X, y)\nfor _, val_idx in split:\n    y_val = y[val_idx]\n    y_preds = ag_oof_preds[val_idx]   \n    score = root_mean_squared_error(y_preds, y_val)\n    ag_scores.append(score)\n    \noof_preds[\"AutoGluon\"], test_preds[\"AutoGluon\"], scores[\"AutoGluon\"] = ag_oof_preds, ag_test_preds, ag_scores","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_results(oof_preds[\"AutoGluon\"], y, \"AutoGluon\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# L2 Ridge","metadata":{"papermill":{"duration":0.019963,"end_time":"2024-12-02T13:07:16.365655","exception":false,"start_time":"2024-12-02T13:07:16.345692","status":"completed"},"tags":[]}},{"cell_type":"code","source":"l2_oof_preds = {}\nl2_test_preds = {}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_weights(weights, title):\n    sorted_indices = np.argsort(weights[0])[::-1]\n    sorted_coeffs = np.array(weights[0])[sorted_indices]\n    sorted_model_names = np.array(list(oof_preds.keys()))[sorted_indices]\n\n    plt.figure(figsize=(10, weights.shape[1] * 0.5))\n    ax = sns.barplot(x=sorted_coeffs, y=sorted_model_names, palette=\"RdYlGn_r\")\n\n    for i, (value, name) in enumerate(zip(sorted_coeffs, sorted_model_names)):\n        if value >= 0:\n            ax.text(value, i, f\"{value:.3f}\", va=\"center\", ha=\"left\", color=\"black\")\n        else:\n            ax.text(value, i, f\"{value:.3f}\", va=\"center\", ha=\"right\", color=\"black\")\n\n    xlim = ax.get_xlim()\n    ax.set_xlim(xlim[0] - 0.1 * abs(xlim[0]), xlim[1] + 0.1 * abs(xlim[1]))\n\n    plt.title(title)\n    plt.xlabel(\"\")\n    plt.ylabel(\"\")\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = pd.DataFrame(oof_preds)\nX_test = pd.DataFrame(test_preds)","metadata":{"papermill":{"duration":0.041269,"end_time":"2024-12-02T13:07:16.426997","exception":false,"start_time":"2024-12-02T13:07:16.385728","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial):    \n    params = {\n        'random_state': CFG.seed,\n        'alpha': trial.suggest_float('alpha', 0, 50),\n        'tol': trial.suggest_float('tol', 1e-6, 1e-3)\n    }\n    \n    model = Ridge(**params)\n    trainer = Trainer(model)\n    return trainer.tune(X, y)\n\nsampler = optuna.samplers.TPESampler(seed=CFG.seed, multivariate=True)\nstudy = optuna.create_study(direction='minimize', sampler=sampler)\nstudy.optimize(objective, n_trials=CFG.n_optuna_trials, n_jobs=-1)\nbest_params = study.best_params","metadata":{"trusted":true,"_kg_hide-output":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ridge_params = {\n    'random_state': CFG.seed,\n    'alpha': best_params['alpha'],\n    'tol': best_params['tol']\n}\nprint(json.dumps(ridge_params, indent=2))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ridge_model = Ridge(**ridge_params)\nridge_trainer = Trainer(ridge_model)\nl2_oof_preds[\"L2 Ensemble Ridge\"], l2_test_preds[\"L2 Ensemble Ridge\"], scores[\"L2 Ensemble Ridge\"], ridge_coeffs = ridge_trainer.train(X, y, X_test, \"ensemble-ridge\")","metadata":{"papermill":{"duration":1.367105,"end_time":"2024-12-02T13:14:18.387825","exception":false,"start_time":"2024-12-02T13:14:17.02072","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_weights(ridge_coeffs, \"Ridge Coefficients\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_results(l2_oof_preds[\"L2 Ensemble Ridge\"], y, \"Ridge\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"save_sub(\"l2-ensemble-ridge\", l2_test_preds[\"L2 Ensemble Ridge\"], np.mean(scores[\"L2 Ensemble Ridge\"]))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# L2 Lasso","metadata":{}},{"cell_type":"code","source":"X = pd.DataFrame(oof_preds)\nX_test = pd.DataFrame(test_preds)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial):    \n    params = {\n        'random_state': CFG.seed,\n        'alpha': trial.suggest_float('alpha', 1e-7, 1e-3),\n        'tol': trial.suggest_float('tol', 1e-6, 1e-3)\n    }\n    \n    model = Lasso(**params)\n    trainer = Trainer(model)\n    return trainer.tune(X, y)\n\nsampler = optuna.samplers.TPESampler(seed=CFG.seed, multivariate=True)\nstudy = optuna.create_study(direction='minimize', sampler=sampler)\nstudy.optimize(objective, n_trials=CFG.n_optuna_trials, n_jobs=-1)\nbest_params = study.best_params","metadata":{"trusted":true,"_kg_hide-output":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lasso_params = {\n    'random_state': CFG.seed,\n    'alpha': best_params['alpha'],\n    'tol': best_params['tol']\n}\nprint(json.dumps(lasso_params, indent=2))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lasso_model = Lasso(**lasso_params)\nlasso_trainer = Trainer(lasso_model)\nl2_oof_preds[\"L2 Ensemble Lasso\"], l2_test_preds[\"L2 Ensemble Lasso\"], scores[\"L2 Ensemble Lasso\"], lasso_coeffs = lasso_trainer.train(X, y, X_test, \"ensemble-ridge\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_weights(lasso_coeffs, \"Lasso Coefficients\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_results(l2_oof_preds[\"L2 Ensemble Lasso\"], y, \"Lasso\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"save_sub(\"l2-ensemble-lasso\", l2_test_preds[\"L2 Ensemble Lasso\"], np.mean(scores[\"L2 Ensemble Lasso\"]))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# L3 weighted ensemble","metadata":{}},{"cell_type":"code","source":"def objective(trial):\n    weights = np.array([trial.suggest_float(l2_model_name, 0, 1) for l2_model_name in l2_oof_preds.keys()])\n    weights /= np.sum(weights)\n    \n    preds = np.zeros(len(y))\n    for model, weight in zip(l2_oof_preds.keys(), weights):\n        preds += l2_oof_preds[model] * weight\n            \n    return root_mean_squared_error(y, preds)\n\nsampler = optuna.samplers.TPESampler(seed=CFG.seed, multivariate=True)\nstudy = optuna.create_study(direction='minimize', sampler=sampler)\nstudy.optimize(objective, n_trials=CFG.n_optuna_trials * 2, n_jobs=-1)","metadata":{"trusted":true,"_kg_hide-output":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores['L3 Weighted Ensemble'] = [study.best_value] * CFG.n_folds","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_weights = np.array([study.best_params[l2_model] for l2_model in l2_oof_preds.keys()])\nbest_weights /= np.sum(best_weights)\nprint(json.dumps({model: weight for model, weight in zip(l2_oof_preds.keys(), best_weights)}, indent=2))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 5))\nplt.pie(\n    study.best_params.values(), \n    labels=study.best_params.keys(), \n    autopct='%1.1f%%',\n    colors=sns.color_palette('Set2', 2)\n)\nplt.title(\"L3 Ensemble Weights\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"weighted_test_preds = np.zeros((X_test.shape[0]))\nfor model, weight in zip(l2_test_preds.keys(), best_weights):\n    weighted_test_preds += l2_test_preds[model] * weight\n    \nsave_sub('l3-weighted-ensemble', weighted_test_preds, np.mean(scores['L3 Weighted Ensemble']))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Results","metadata":{"papermill":{"duration":0.031862,"end_time":"2024-12-02T13:14:23.322797","exception":false,"start_time":"2024-12-02T13:14:23.290935","status":"completed"},"tags":[]}},{"cell_type":"code","source":"scores = pd.DataFrame(scores)\nmean_scores = scores.mean().sort_values(ascending=True)\norder = scores.mean().sort_values(ascending=True).index.tolist()\n\nmin_score = mean_scores.min()\nmax_score = mean_scores.max()\npadding = (max_score - min_score) * 0.5\nlower_limit = min_score - padding\nupper_limit = max_score + padding\n\nfig, axs = plt.subplots(1, 2, figsize=(15, scores.shape[1] * 0.3))\n\nboxplot = sns.boxplot(data=scores, order=order, ax=axs[0], orient=\"h\", color=\"grey\")\naxs[0].set_title(f\"Fold {CFG.metric}\")\naxs[0].set_xlabel(\"\")\naxs[0].set_ylabel(\"\")\n\nbarplot = sns.barplot(x=mean_scores.values, y=mean_scores.index, ax=axs[1], color=\"grey\")\naxs[1].set_title(f\"Average {CFG.metric}\")\naxs[1].set_xlabel(\"\")\naxs[1].set_xlim(left=lower_limit, right=upper_limit)\naxs[1].set_ylabel(\"\")\n\nfor i, (score, model) in enumerate(zip(mean_scores.values, mean_scores.index)):\n    color = \"skyblue\" if \"ensemble\" in model.lower() else \"grey\"\n    barplot.patches[i].set_facecolor(color)\n    boxplot.patches[i].set_facecolor(color)\n    barplot.text(score, i, round(score, 6), va=\"center\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.set_style(\"whitegrid\")\nfig, axes = plt.subplots(1, 2, figsize=(15, 5))\n\nsns.ecdfplot(np.expm1(y), label=\"Target\", ax=axes[0])\nsns.ecdfplot(np.expm1(l2_oof_preds[\"L2 Ensemble Lasso\"]), label=\"L2 Ensemble Lasso\", ax=axes[0])\nsns.ecdfplot(np.expm1(l2_oof_preds[\"L2 Ensemble Ridge\"]), label=\"L2 Ensemble Ridge\", ax=axes[0])\nfor model in oof_preds:\n    sns.ecdfplot(np.expm1(oof_preds[model]), label=model, ax=axes[0])\naxes[0].set_title(\"CDF\")\naxes[0].legend(loc=\"best\")\naxes[0].set_ylim(0, 1.1)\n\nsns.kdeplot(np.expm1(y), ax=axes[1], label='Target')\nsns.kdeplot(np.expm1(l2_oof_preds[\"L2 Ensemble Lasso\"]), ax=axes[1], label='L2 Ensemble Lasso')\nsns.kdeplot(np.expm1(l2_oof_preds[\"L2 Ensemble Ridge\"]), ax=axes[1], label='L2 Ensemble Ridge')\nfor model in oof_preds:\n    sns.kdeplot(np.expm1(oof_preds[model]), ax=axes[1], label=model)\naxes[1].set_title(\"Prediction and target distributions\")\naxes[1].legend(loc=\"best\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"shutil.rmtree(\"catboost_info\", ignore_errors=True)","metadata":{"papermill":{"duration":0.043514,"end_time":"2024-12-02T13:14:24.22101","exception":false,"start_time":"2024-12-02T13:14:24.177496","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null}]}