{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nfrom scipy.optimize import minimize\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport re\nimport tensorflow as tf\nimport warnings\nimport os\nimport tqdm\nfrom sklearn.pipeline import make_pipeline, Pipeline\nfrom sklearn.preprocessing import FunctionTransformer, StandardScaler\nfrom sklearn.neural_network import MLPRegressor\nfrom category_encoders import OrdinalEncoder, OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer, TransformedTargetRegressor\nfrom sklearn.base import clone\nfrom sklearn.inspection import permutation_importance\nfrom sklearn.linear_model import Ridge, ElasticNet\nfrom sklearn.metrics import mean_squared_log_error, make_scorer\nfrom sklearn.model_selection import KFold\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\nimport time\nfrom sklearn import set_config\nfrom colorama import Fore, Style\n\nset_config(transform_output='pandas')\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \nplt.style.use('ggplot')\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-03T19:17:00.681082Z","iopub.execute_input":"2024-12-03T19:17:00.681506Z","iopub.status.idle":"2024-12-03T19:17:16.347627Z","shell.execute_reply.started":"2024-12-03T19:17:00.681452Z","shell.execute_reply":"2024-12-03T19:17:16.346705Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reading Data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv', parse_dates=['Policy Start Date'], index_col='id')\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv', parse_dates=['Policy Start Date'], index_col='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:17:16.349115Z","iopub.execute_input":"2024-12-03T19:17:16.349716Z","iopub.status.idle":"2024-12-03T19:17:25.859474Z","shell.execute_reply.started":"2024-12-03T19:17:16.349685Z","shell.execute_reply":"2024-12-03T19:17:25.858766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:17:25.86037Z","iopub.execute_input":"2024-12-03T19:17:25.860634Z","iopub.status.idle":"2024-12-03T19:17:25.867015Z","shell.execute_reply.started":"2024-12-03T19:17:25.86061Z","shell.execute_reply":"2024-12-03T19:17:25.866129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nnum_cols = test.select_dtypes('number').columns.tolist()\ncat_cols = [f for f in test.columns if f not in num_cols]\ntarget = 'Premium Amount'\nfor c in cat_cols:    \n    if not re.search(r'Date',c):\n        print(f'column: {c}')\n        A = train[c].fillna('None').astype(str).unique()\n        B = test[c].fillna('None').astype(str).unique()\n        C = np.setdiff1d(B,A)\n        if C.size>0:\n            print(C)\n            train.iloc[~ [c].isin(C), c ] = 'None'\n        train[c] = train[c].astype('category')\n        test[c] = test[c].astype('category')    \n\ninitial_features = test.columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:17:25.869346Z","iopub.execute_input":"2024-12-03T19:17:25.869771Z","iopub.status.idle":"2024-12-03T19:17:29.02631Z","shell.execute_reply.started":"2024-12-03T19:17:25.869734Z","shell.execute_reply":"2024-12-03T19:17:29.025458Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Distribution of Training and Test feature","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(5,4, figsize=(15,12), constrained_layout=True)\nfor col,ax in zip(initial_features, axs.ravel()):    \n    if train[col].dtype == 'float':\n        ax.hist(train[col], color='lightgreen',alpha=0.5,label='train')\n        ax.hist(test[col], color='red',alpha=0.5,label='test')        \n    elif train[col].dtype == 'category':\n        vc = train[col].value_counts() / len(train)\n        vc2 = test[col].value_counts() / len(test)\n        ax.bar(vc.index, vc,color='lightgreen',alpha=0.5,label='train')        \n        ax.yaxis.set_major_formatter('{x:.0%}')\n        ax.bar(vc2.index, vc2,color='red',alpha=0.5,label='test')   \n        \n        if len(vc)<=15:\n            ax.set_xticks(np.arange(len(train[col].dtype.categories)), train[col].dtype.categories,rotation=90)\n        else:\n            ax.set_xticks([])\n    elif np.issubdtype(train[col].dtype, np.datetime64): \n        trainc = train.copy()\n        trainc['year'] = trainc[col].dt.year\n        trainc['month'] = trainc[col].dt.month\n        monthly_counts = trainc.groupby(['year', 'month']).size().reset_index(name='count')    \n        ax.bar(monthly_counts['year'].astype(str) + '-' + monthly_counts['month'].astype(str),\n               monthly_counts['count'], color='lightblue',label='train')\n        ax.set_xticks([])\n        \n    ax.set_title(f'{col}',fontweight='bold')    \nfor ax in axs.ravel():\n    if not ax.has_data():\n        ax.axis('off') \nfig.legend(bbox_to_anchor=[1,1.02],labels=['Train','Test'])\nplt.suptitle('Distribution of Training and Test feature',fontweight='bold',size=20);\ndel trainc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:17:29.027282Z","iopub.execute_input":"2024-12-03T19:17:29.027571Z","iopub.status.idle":"2024-12-03T19:17:33.363987Z","shell.execute_reply.started":"2024-12-03T19:17:29.027544Z","shell.execute_reply":"2024-12-03T19:17:33.363171Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* The distribution of features between the training and testing sets appears to be quite similar for most variables. This suggests that the testing set is representative of the training set, which is a good sign for model generalization.\n* **Annual Income:** Has a tail on the right indicating the presence of some individuals with very high incomes.\n* **Credit Score:** The distribution of credit scores is similar, with a concentration in higher credit ranges.\n* **Insurance Duration:** The distribution of insurance duration is similar, with a concentration in shorter periods.","metadata":{}},{"cell_type":"code","source":"cat_cols.remove('Policy Start Date')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:17:33.36523Z","iopub.execute_input":"2024-12-03T19:17:33.365631Z","iopub.status.idle":"2024-12-03T19:17:33.369869Z","shell.execute_reply.started":"2024-12-03T19:17:33.36559Z","shell.execute_reply":"2024-12-03T19:17:33.368987Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Target Distribuition","metadata":{}},{"cell_type":"code","source":"_, ax = plt.subplots(1,2, figsize=(8,3), constrained_layout=True)\ntrain[target].hist(ax=ax[0],bins=100)\nskew_value = train[target].skew()\nax[0].text(0.95, 0.9, f'Skew: {skew_value:.2f}', \n           transform=ax[0].transAxes, ha='right', va='top')\nsns.ecdfplot(train[target],ax=ax[1])\nplt.grid(True)\nplt.suptitle('Target Distribuition');","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:17:33.370813Z","iopub.execute_input":"2024-12-03T19:17:33.371092Z","iopub.status.idle":"2024-12-03T19:17:35.030978Z","shell.execute_reply.started":"2024-12-03T19:17:33.371065Z","shell.execute_reply":"2024-12-03T19:17:35.030038Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Positive Skewness: The distribution of the variable is highly skewed to the right. This means that most of the values ​​are concentrated in a smaller range, with a few very high values ​​pulling the mean to the right. The skew value, 1.24, confirms the strong positive skewness of the distribution. The higher the skew value, the greater the skewness, usually a logarithmic transformation can be applied.","metadata":{}},{"cell_type":"markdown","source":"# Feature Engineering\n","metadata":{}},{"cell_type":"code","source":"for df in [train, test]:\n    df['year'] = df['Policy Start Date'].dt.year\n    df['month'] = df['Policy Start Date'].dt.month\n    df['day'] = df['Policy Start Date'].dt.day    \n    df['dayofweek'] = df['Policy Start Date'].dt.dayofweek    \n    df['month_sin'] = np.sin(2 * np.pi * df['month'] / 12) \n    df['month_cos'] = np.cos(2 * np.pi * df['month'] / 12)\n    df['day_sin'] = np.sin(2 * np.pi * df['day'] / 31)  \n    df['day_cos'] = np.cos(2 * np.pi * df['day'] / 31)\n    \ntime_features = ['year','month','day','dayofweek']\ntime_features_ridge = ['month_sin','month_cos','day_sin','day_cos']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:18:38.944705Z","iopub.execute_input":"2024-12-03T19:18:38.945558Z","iopub.status.idle":"2024-12-03T19:18:39.488128Z","shell.execute_reply.started":"2024-12-03T19:18:38.945524Z","shell.execute_reply":"2024-12-03T19:18:39.487438Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Metric","metadata":{}},{"cell_type":"markdown","source":"* RMSLE penalizes underestimation errors more.","metadata":{}},{"cell_type":"code","source":"def root_mean_squared_log_error(y_true, y_pred):\n    y_pred = np.maximum(0, y_pred)\n    return np.sqrt(np.mean((np.log1p(y_true) - np.log1p(y_pred)) ** 2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:18:41.318971Z","iopub.execute_input":"2024-12-03T19:18:41.319864Z","iopub.status.idle":"2024-12-03T19:18:41.32419Z","shell.execute_reply.started":"2024-12-03T19:18:41.319827Z","shell.execute_reply":"2024-12-03T19:18:41.323219Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* permutation importance (to be used later) does not have a scoring for rmsle, so we must make one using make_scorer.","metadata":{}},{"cell_type":"code","source":"rmsle_scorer = make_scorer(root_mean_squared_log_error, greater_is_better=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:18:42.577914Z","iopub.execute_input":"2024-12-03T19:18:42.578718Z","iopub.status.idle":"2024-12-03T19:18:42.582635Z","shell.execute_reply.started":"2024-12-03T19:18:42.578681Z","shell.execute_reply":"2024-12-03T19:18:42.58169Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"def convert_to_category(df):\n    for col in df.select_dtypes(include='object').columns:\n        df[col] = df[col].astype('category')\n    return df\n    \npreprocessing = make_pipeline(ColumnTransformer([('drop','drop','Policy Start Date')],\n                                             remainder='passthrough', \n                                             verbose_feature_names_out=False),\n                            ColumnTransformer([('num',SimpleImputer(),num_cols)],\n                                             remainder='passthrough', \n                                             verbose_feature_names_out=False),\n                            ColumnTransformer([('cat',SimpleImputer(strategy='most_frequent'),cat_cols)],\n                                             remainder='passthrough', \n                                             verbose_feature_names_out=False), \n                            FunctionTransformer(convert_to_category, validate=False))\n\npreprocessing_ridge = make_pipeline(preprocessing,\n                                 OneHotEncoder(cols=cat_cols),\n                                StandardScaler())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:18:43.740984Z","iopub.execute_input":"2024-12-03T19:18:43.741328Z","iopub.status.idle":"2024-12-03T19:18:43.747768Z","shell.execute_reply.started":"2024-12-03T19:18:43.741297Z","shell.execute_reply":"2024-12-03T19:18:43.746787Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores, oofs, test_preds = pd.DataFrame(), pd.DataFrame(), pd.DataFrame()\nkf = KFold(5,random_state=42,shuffle=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:18:45.030934Z","iopub.execute_input":"2024-12-03T19:18:45.031705Z","iopub.status.idle":"2024-12-03T19:18:45.036649Z","shell.execute_reply.started":"2024-12-03T19:18:45.031671Z","shell.execute_reply":"2024-12-03T19:18:45.035729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def score_model(estimator, label = '', features=[], show_importance = True):\n    X = train.copy()\n    y = X.pop(target)\n    \n    val_predictions = np.zeros((len(X)))\n    test_predictions = np.zeros((len(test)))\n    val_scores= []\n                \n    start_time = time.time()\n    for fold, (train_idx, val_idx) in enumerate(kf.split(X, y)):\n        \n        model = clone(estimator)\n        \n        X_train = X.iloc[train_idx][features]\n        X_val   = X.iloc[val_idx][features]\n        y_train = y.iloc[train_idx]\n        y_val   = y.iloc[val_idx]\n\n        model.fit(X_train, y_train)    \n        y_pred = model.predict(X_val).clip(0,None)\n        val_predictions[val_idx] += y_pred                \n        val_score = root_mean_squared_log_error(y_val, y_pred)        \n        val_scores.append(val_score)\n        \n        test_predictions += model.predict(test[features]).clip(0,None) / kf.get_n_splits()            \n        \n        print(f\"# Fold {fold}: rmsle={val_score:.5f}\")\n\n        if (show_importance) and (fold == 0): \n            rmsle = val_score\n            result = permutation_importance(model, X_val, y_val,\n                                            scoring=rmsle_scorer, \n                                            n_repeats=5, random_state=42)\n            \n            print(f\"{Fore.BLUE}{Style.BRIGHT}Important features: {(result['importances_mean'] > 0).mean():.0%}   ({rmsle=:.3f}){Style.RESET_ALL}\")\n            importance_df = pd.DataFrame({'importance': result['importances_mean'],\n                                           'std': result['importances_std']}, \n                                         index=X_val.columns).sort_values('importance', ascending=False)\n            display(importance_df.head(30))\n                \n    print(f\" Average rmsle={Fore.BLUE}{Style.BRIGHT}{np.mean(val_scores):.5f}{Style.RESET_ALL}\")\n\n    \n    return val_scores, val_predictions, test_predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:18:46.470708Z","iopub.execute_input":"2024-12-03T19:18:46.471519Z","iopub.status.idle":"2024-12-03T19:18:46.479967Z","shell.execute_reply.started":"2024-12-03T19:18:46.471484Z","shell.execute_reply":"2024-12-03T19:18:46.47912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores['Ridge'], oofs['Ridge'], test_preds['Ridge'] = score_model(make_pipeline(preprocessing_ridge,\n                                                            TransformedTargetRegressor(\n                                                                Ridge(),\n                                                            func=np.log1p,\n                                                            inverse_func=np.expm1\n                                                        )), \n                                              'Ridge', \n                                              initial_features+time_features_ridge,False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:18:48.190195Z","iopub.execute_input":"2024-12-03T19:18:48.191159Z","iopub.status.idle":"2024-12-03T19:20:15.823727Z","shell.execute_reply.started":"2024-12-03T19:18:48.191117Z","shell.execute_reply":"2024-12-03T19:20:15.822635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {'objective': 'regression', \n          'colsample_by_tree':0.78,\n          'learning_rate':0.12,\n          'n_estimators':150,\n          'subsample':0.80,\n          'verbose': -1, \n          'n_jobs': -1}\n\nscores['lgbm'], oofs['lgbm'], test_preds['lgbm'] = score_model(make_pipeline(preprocessing,\n                                                            TransformedTargetRegressor(\n                                                                LGBMRegressor(**params),\n                                                            func=np.log1p,\n                                                            inverse_func=np.expm1\n                                                        )), \n                                              'lgbm', \n                                              initial_features+time_features,True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:20:15.825776Z","iopub.execute_input":"2024-12-03T19:20:15.826192Z","iopub.status.idle":"2024-12-03T19:21:14.364516Z","shell.execute_reply.started":"2024-12-03T19:20:15.82615Z","shell.execute_reply":"2024-12-03T19:21:14.363675Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* The three most important features were Annual Income, Credit Score and Previous Claims.\n\n**Warning:** Permutation importance can be a bit misleading when dealing with multicollinearity","metadata":{}},{"cell_type":"code","source":"params = {'iterations':1000,\n          'cat_features':cat_cols,\n          'task_type': 'GPU',\n          'verbose': 0}\n\nscores['catboost'], oofs['catboost'], test_preds['catboost'] = score_model(\n                                            make_pipeline(ColumnTransformer([('drop','drop','Policy Start Date')],\n                                                             remainder='passthrough', \n                                                             verbose_feature_names_out=False),\n                                                          ColumnTransformer([('num',SimpleImputer(),num_cols)],\n                                                             remainder='passthrough', \n                                                             verbose_feature_names_out=False),\n                                                          ColumnTransformer([('cat',SimpleImputer(strategy='most_frequent'),cat_cols)],\n                                                             remainder='passthrough', \n                                                             verbose_feature_names_out=False),                                                             \n                                                            TransformedTargetRegressor(\n                                                                CatBoostRegressor(**params),\n                                                            func=np.log1p,\n                                                            inverse_func=np.expm1\n                                                        )), \n                                              'catboost', \n                                              initial_features+time_features,False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:21:14.365522Z","iopub.execute_input":"2024-12-03T19:21:14.365825Z","iopub.status.idle":"2024-12-03T19:27:04.896566Z","shell.execute_reply.started":"2024-12-03T19:21:14.365794Z","shell.execute_reply":"2024-12-03T19:27:04.895633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {'objective': 'regression',\n          'boosting':'dart',\n          'n_estimators':500,          \n          'verbose': -1, \n          'n_jobs': -1}\n\nscores['dart'], oofs['dart'], test_preds['dart'] = score_model(make_pipeline(preprocessing,\n                                                            TransformedTargetRegressor(\n                                                                LGBMRegressor(**params),\n                                                            func=np.log1p,\n                                                            inverse_func=np.expm1\n                                                        )), \n                                              'dart', \n                                              initial_features+time_features,False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:38:22.121263Z","iopub.execute_input":"2024-12-03T19:38:22.122021Z","iopub.status.idle":"2024-12-03T19:53:36.051177Z","shell.execute_reply.started":"2024-12-03T19:38:22.121986Z","shell.execute_reply":"2024-12-03T19:53:36.050298Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ensemble","metadata":{}},{"cell_type":"code","source":"def obj_fun(alpha, y_true, oofs):\n    if isinstance(oofs,dict):\n        oofs_df = pd.DataFrame(oofs)\n    else:\n        oofs_df = oofs.copy()\n    weighted_preds = oofs_df @ alpha         \n    rmsle = root_mean_squared_log_error(y_true, weighted_preds) \n    \n    return -rmsle\n\t\n\t\nresult = minimize(obj_fun, x0=[0.5]*oofs.shape[1],                  \n          args=(train[target],oofs),method='Nelder-Mead')\n\nw_nelder = result.x / np.sum(result.x)\nscores['nelder-mead'] = root_mean_squared_log_error(train[target],oofs@w_nelder)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:53:36.054541Z","iopub.execute_input":"2024-12-03T19:53:36.055139Z","iopub.status.idle":"2024-12-03T19:54:22.531077Z","shell.execute_reply.started":"2024-12-03T19:53:36.0551Z","shell.execute_reply":"2024-12-03T19:54:22.528308Z"},"_kg_hide-input":false},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Scores","metadata":{}},{"cell_type":"code","source":"ax = scores.mean().sort_values(ascending=False).plot(kind='barh')\nax.bar_label(ax.containers[0],label_type='center',color='white',fontweight='bold')\nax.patches[-1].set_facecolor('green');","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:54:22.532388Z","iopub.execute_input":"2024-12-03T19:54:22.532803Z","iopub.status.idle":"2024-12-03T19:54:22.800533Z","shell.execute_reply.started":"2024-12-03T19:54:22.532747Z","shell.execute_reply":"2024-12-03T19:54:22.799674Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* The set with Nelder Mead did not perform better than the LGBM, this may be due to the low diversity of models. For best results using ensembles, consider using neural networks (the amount of data is good enough for one network)","metadata":{}},{"cell_type":"code","source":"sns.ecdfplot(y=train[target],label='y_true')\nsns.ecdfplot(y=oofs['lgbm'],label='lgbm')\nsns.ecdfplot(y=oofs['catboost'],label='catboost')\nsns.ecdfplot(y=oofs['Ridge'],label='ridge')\nsns.ecdfplot(y=oofs['dart'],label='dart')\nplt.legend(bbox_to_anchor=(1.05,1));","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:58:59.401152Z","iopub.execute_input":"2024-12-03T19:58:59.402102Z","iopub.status.idle":"2024-12-03T19:59:13.742319Z","shell.execute_reply.started":"2024-12-03T19:58:59.402054Z","shell.execute_reply":"2024-12-03T19:59:13.741453Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The actual value (red line) shows an exponential behavior, especially after the ratio 0.6, with a very sharp increase close to 1.0, reaching almost 5000\nThe forecast models (other lines) have very similar behavior to each other, remaining relatively stable around 1000\nThere is a significant underestimation by the models in the upper part of the distribution, especially for the higher values","metadata":{}},{"cell_type":"markdown","source":"# Submssion","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\nsub[target] = test_preds['lgbm']\nsub.to_csv('submission.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T19:17:36.045737Z","iopub.status.idle":"2024-12-03T19:17:36.046156Z","shell.execute_reply.started":"2024-12-03T19:17:36.045939Z","shell.execute_reply":"2024-12-03T19:17:36.04596Z"}},"outputs":[],"execution_count":null}]}