{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%pip install --upgrade scikit-learn lightgbm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:03:56.405918Z","iopub.execute_input":"2024-12-17T15:03:56.406362Z","iopub.status.idle":"2024-12-17T15:04:19.23325Z","shell.execute_reply.started":"2024-12-17T15:03:56.406323Z","shell.execute_reply":"2024-12-17T15:04:19.230295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pylab as plt\nimport seaborn as sns\nimport warnings\nimport os\n\nwarnings.filterwarnings('ignore', category=FutureWarning)\nsns.set_context('talk')\n\n\ninput_files = {}\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        file_path = os.path.join(dirname, filename)\n        if file_path.endswith('.csv'):\n            print( f'Importing {file_path}')\n            input_files[filename] = pd.read_csv( file_path)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:19.237343Z","iopub.execute_input":"2024-12-17T15:04:19.238294Z","iopub.status.idle":"2024-12-17T15:04:33.586305Z","shell.execute_reply.started":"2024-12-17T15:04:19.238222Z","shell.execute_reply":"2024-12-17T15:04:33.585044Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"df_train = input_files['train.csv'].drop(columns='id')\ndf_test = input_files['test.csv']\ndf_submission = input_files['sample_submission.csv']\ndf_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:33.587984Z","iopub.execute_input":"2024-12-17T15:04:33.588387Z","iopub.status.idle":"2024-12-17T15:04:34.542581Z","shell.execute_reply.started":"2024-12-17T15:04:33.588351Z","shell.execute_reply":"2024-12-17T15:04:34.541203Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Datatypes","metadata":{}},{"cell_type":"code","source":"df_train.dtypes.value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:34.545748Z","iopub.execute_input":"2024-12-17T15:04:34.546607Z","iopub.status.idle":"2024-12-17T15:04:34.55789Z","shell.execute_reply.started":"2024-12-17T15:04:34.546544Z","shell.execute_reply":"2024-12-17T15:04:34.556595Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"11 categorical input features and 4 numerical ones. Let's check the cardinality of the former.","metadata":{}},{"cell_type":"code","source":"df_train.select_dtypes(include=object).nunique().sort_values(ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:34.559492Z","iopub.execute_input":"2024-12-17T15:04:34.560006Z","iopub.status.idle":"2024-12-17T15:04:35.854835Z","shell.execute_reply.started":"2024-12-17T15:04:34.559929Z","shell.execute_reply":"2024-12-17T15:04:35.852562Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Cardinality of these features is very low, the one with the higher entropy being a date, so ordinal. At first sight it seems easy to preprocess them (the two binary ones are trivial for instance).","metadata":{}},{"cell_type":"markdown","source":"### Missing inputs","metadata":{}},{"cell_type":"markdown","source":"Counting the missing values in absolute and fractional units.","metadata":{}},{"cell_type":"code","source":"pd.concat( (df_train.isna().sum().sort_values(ascending=False).rename('#'), df_train.isna().mean().mul(100).round(2).rename('%')), axis=1) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:35.857993Z","iopub.execute_input":"2024-12-17T15:04:35.858833Z","iopub.status.idle":"2024-12-17T15:04:37.207472Z","shell.execute_reply.started":"2024-12-17T15:04:35.858682Z","shell.execute_reply":"2024-12-17T15:04:37.205806Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"A number of features have many missing inputs. Missing `Previous Claims` no necessarily means that there is no history, same for `Customer Feedback`.","metadata":{}},{"cell_type":"code","source":"df_train['Previous Claims'].value_counts(dropna=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:37.208764Z","iopub.execute_input":"2024-12-17T15:04:37.209177Z","iopub.status.idle":"2024-12-17T15:04:37.242372Z","shell.execute_reply.started":"2024-12-17T15:04:37.209137Z","shell.execute_reply":"2024-12-17T15:04:37.241058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['Policy Start Date'].min(), df_train['Policy Start Date'].max()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:37.243933Z","iopub.execute_input":"2024-12-17T15:04:37.244424Z","iopub.status.idle":"2024-12-17T15:04:37.462842Z","shell.execute_reply.started":"2024-12-17T15:04:37.244374Z","shell.execute_reply":"2024-12-17T15:04:37.461524Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Target distribution","metadata":{}},{"cell_type":"code","source":"sns.histplot(data=df_train, x='Premium Amount');","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:37.464341Z","iopub.execute_input":"2024-12-17T15:04:37.464697Z","iopub.status.idle":"2024-12-17T15:04:38.645935Z","shell.execute_reply.started":"2024-12-17T15:04:37.464663Z","shell.execute_reply":"2024-12-17T15:04:38.643575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_info = df_train['Premium Amount'].describe()\ntarget_info['skewness'] = df_train['Premium Amount'].skew()\ntarget_info","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:38.650076Z","iopub.execute_input":"2024-12-17T15:04:38.650511Z","iopub.status.idle":"2024-12-17T15:04:38.750118Z","shell.execute_reply.started":"2024-12-17T15:04:38.650472Z","shell.execute_reply":"2024-12-17T15:04:38.748298Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Worthy notes on the distribution:\n- it has a long tail, and fairly large skewness.\n- models assuming it's normaly distributed will be negatively impacted.\n- it does look like a [log-normal distribution](https://en.wikipedia.org/wiki/Log-normal_distribution) but it has a mode near `x=10` and a void near `2000`.","metadata":{"execution":{"iopub.status.busy":"2024-12-02T09:19:49.426163Z","iopub.execute_input":"2024-12-02T09:19:49.426642Z","iopub.status.idle":"2024-12-02T09:19:49.435332Z","shell.execute_reply.started":"2024-12-02T09:19:49.426604Z","shell.execute_reply":"2024-12-02T09:19:49.433486Z"}}},{"cell_type":"code","source":"df_train['Premium Amount'].value_counts().head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:38.752426Z","iopub.execute_input":"2024-12-17T15:04:38.752992Z","iopub.status.idle":"2024-12-17T15:04:38.788333Z","shell.execute_reply.started":"2024-12-17T15:04:38.752917Z","shell.execute_reply":"2024-12-17T15:04:38.786692Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Note that the left most mode account for many values and missclassifiying it can account for a large contributions to the RMLE. It seems worth the investment to predict it as good as possible.","metadata":{}},{"cell_type":"code","source":"sns.histplot(data=np.log10(df_train['Premium Amount']));","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:38.790405Z","iopub.execute_input":"2024-12-17T15:04:38.79077Z","iopub.status.idle":"2024-12-17T15:04:40.342388Z","shell.execute_reply.started":"2024-12-17T15:04:38.790738Z","shell.execute_reply":"2024-12-17T15:04:40.341025Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"A naive log transform is not enough to give it a normal distribution shape. Now the skewness is negative and it has a mode at the end of a tail which is very undesirable.","metadata":{}},{"cell_type":"markdown","source":"## Data Preprocessing","metadata":{"execution":{"iopub.status.busy":"2024-12-02T09:22:16.288479Z","iopub.execute_input":"2024-12-02T09:22:16.289665Z","iopub.status.idle":"2024-12-02T09:22:16.298236Z","shell.execute_reply.started":"2024-12-02T09:22:16.289601Z","shell.execute_reply":"2024-12-02T09:22:16.296271Z"}}},{"cell_type":"markdown","source":"#### Let's prepare this data for model ingestion","metadata":{}},{"cell_type":"markdown","source":"Manual encoding of ordinal variables","metadata":{}},{"cell_type":"code","source":"ordinal_variables = {\n    'Education Level': ['High School', \"Bachelor's\", \"Master's\", 'PhD'],\n    'Location': ['Urban', 'Suburban', 'Rural'],\n    'Property Type' : ['House',  'Condo', 'Apartment'],\n    'Policy Type' : ['Premium', 'Comprehensive', 'Basic'],\n    'Customer Feedback' : ['Poor', 'Average', 'Good'],\n    'Exercise Frequency' : ['Daily', 'Weekly', 'Monthly', 'Rarely'],\n    'Occupation': ['Unemployed','Self-Employed','Employed'],\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:40.343801Z","iopub.execute_input":"2024-12-17T15:04:40.344308Z","iopub.status.idle":"2024-12-17T15:04:40.35161Z","shell.execute_reply.started":"2024-12-17T15:04:40.344257Z","shell.execute_reply":"2024-12-17T15:04:40.350229Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Sorry for the self-employed people but I am sure insurance companies decided to see you in a lesser regard than fully employed clients.","metadata":{}},{"cell_type":"markdown","source":"#### Checking price inflation over timeArithmeticError","metadata":{}},{"cell_type":"code","source":"df_train['Health Score'] = df_train['Health Score'].fillna(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:40.353322Z","iopub.execute_input":"2024-12-17T15:04:40.353738Z","iopub.status.idle":"2024-12-17T15:04:40.381965Z","shell.execute_reply.started":"2024-12-17T15:04:40.353653Z","shell.execute_reply":"2024-12-17T15:04:40.380351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hue = df_train['Premium Amount'].gt(50)\n\nsns.histplot(data= df_train, x='Health Score' , hue=hue, common_norm=False, stat='percent', bins=25);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:40.383657Z","iopub.execute_input":"2024-12-17T15:04:40.384143Z","iopub.status.idle":"2024-12-17T15:04:41.323153Z","shell.execute_reply.started":"2024-12-17T15:04:40.384095Z","shell.execute_reply":"2024-12-17T15:04:41.321888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def transform_policy_date(df):\n    dt = pd.to_datetime( df['Policy Start Date'])\n    df['start_date_month'] = dt.dt.month\n    df['start_date_dow'] = dt.dt.dayofweek\n    df['start_date_year'] = dt.dt.year\n    return df#.drop(columns=['Policy Start Date'])\n\n\ndf_train = transform_policy_date(df_train)\ndf_test = transform_policy_date(df_test)\n\ndef transform_claims(df):\n    df['Claims vs Duration'] = df['Previous Claims'] / df['Insurance Duration']\n    df['Health vs Claims'] = np.clip( df['Health Score'] / df['Previous Claims'], 0, 1_000)\n    df['Health Score'] = df['Health Score'].round().fillna(0).astype('category')\n    return df\n\ndf_train = transform_claims(df_train)\ndf_test = transform_claims(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:41.324704Z","iopub.execute_input":"2024-12-17T15:04:41.325163Z","iopub.status.idle":"2024-12-17T15:04:42.475621Z","shell.execute_reply.started":"2024-12-17T15:04:41.325117Z","shell.execute_reply":"2024-12-17T15:04:42.474268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# most_frequent_values = df_train['Credit Score Cat'].value_counts(dropna=False).head(25).index\n\n# def enrich_credit_score_cat(df):\nmost_frequent_values = df_train['Credit Score'].value_counts(dropna=False).head(25).index\n\ndf_train['Credit Score Cat'] = df_train['Credit Score'].where( df_train['Credit Score'].isin(most_frequent_values), 'RATE').map(str)\ndf_test['Credit Score Cat'] = df_test['Credit Score'].where( df_test['Credit Score'].isin(most_frequent_values), 'RATE').map(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:42.476704Z","iopub.execute_input":"2024-12-17T15:04:42.477066Z","iopub.status.idle":"2024-12-17T15:04:43.120474Z","shell.execute_reply.started":"2024-12-17T15:04:42.477031Z","shell.execute_reply":"2024-12-17T15:04:43.118914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Interestingly, even if inflation has been one of the buzzwords of the last few years it does not seem to have driven the prices up over the last 5 years. For simplicity let's drop the purchase time from the model input variables for now.","metadata":{}},{"cell_type":"code","source":"categorical_variables = list(set(df_train.select_dtypes(include=('category', object))) - set(ordinal_variables) - {'Policy Start Date'})\nnumerical_variables = df_train.drop(columns='Premium Amount').select_dtypes(exclude=('category', object)).columns.tolist()\nprint('Numerical variables:', numerical_variables)\nprint('Categorical variables:', categorical_variables)\nprint('Ordinal variables:', list(ordinal_variables.keys()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:43.122361Z","iopub.execute_input":"2024-12-17T15:04:43.122845Z","iopub.status.idle":"2024-12-17T15:04:44.316729Z","shell.execute_reply.started":"2024-12-17T15:04:43.122804Z","shell.execute_reply":"2024-12-17T15:04:44.315213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder, LabelEncoder, OrdinalEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.utils import check_array\nfrom sklearn.compose import ColumnTransformer\n\nclass OE(OrdinalEncoder):\n    def transform(self, X):\n        if not hasattr(self, 'infrequent_categories_'):\n            return super().transform(X)\n        self.unknown_value = -1 if self.encoded_missing_value != -1 else -2\n        X = check_array(super().transform(X), force_all_finite=False)\n        for i in range(X.shape[1]):\n            infreq_cat = [] if self.infrequent_categories_[i] is None else self.infrequent_categories_[i]\n            infreq_cat_count = len([c for c in infreq_cat if not (isinstance(c, float) and np.isnan(c))])\n            cat_count = len([c for c in self.categories_[i] if not (isinstance(c, float) and np.isnan(c))])\n            X[:,i] = np.where(\n                X[:,i]==self.unknown_value, \n                cat_count - infreq_cat_count, \n                X[:,i]\n            )\n        if hasattr(self, '_output_transform'): \n            return self._output_transform(X) \n        else: \n            return X\n\n# Combine the numerical and categorical pipelines\n\ncombined_categoricals = list(ordinal_variables.keys())+categorical_variables\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', SimpleImputer(strategy='median'), numerical_variables),\n        ('ordin', OE(dtype=np.int32, handle_unknown='use_encoded_value', unknown_value=-1, encoded_missing_value=-1), combined_categoricals),\n        # ('ohe', OneHotEncoder(drop='first'), categorical_variables),\n    ], remainder='drop'\n)\n\nfeature_names = numerical_variables+combined_categoricals\n\n# Apply the transformations to the training and test sets\nX_train_preprocessed = preprocessor.fit_transform(df_train)\ny_train = df_train['Premium Amount'].values\nX_test_preprocessed = preprocessor.transform(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:44.319258Z","iopub.execute_input":"2024-12-17T15:04:44.319909Z","iopub.status.idle":"2024-12-17T15:04:54.246872Z","shell.execute_reply.started":"2024-12-17T15:04:44.31986Z","shell.execute_reply":"2024-12-17T15:04:54.245613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_preprocessed.shape, X_train_preprocessed.dtype","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:04:54.248382Z","iopub.execute_input":"2024-12-17T15:04:54.249147Z","iopub.status.idle":"2024-12-17T15:04:54.257289Z","shell.execute_reply.started":"2024-12-17T15:04:54.249095Z","shell.execute_reply":"2024-12-17T15:04:54.256013Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Optimizing LGB regressor model","metadata":{}},{"cell_type":"markdown","source":"Note that I am usually more of a fan of LightGBM as it's fast - at least in my experience - so let's go for it. It does not have natively training against log error - unlike XGB, so let's log1p the target and invert instead.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_log_error\nimport lightgbm as lgb\nimport optuna\n\nLGBM_DEFAULT_PARAMS = {\n    \"verbosity\": 0,\n    \"objective\": \"regression\",\n    'seed': 666_666,\n    'boosting':'gbdt',\n    'metric': 'rmse',\n    'force_row_wise': True\n}\n\n# def unweighted_rmse(preds, train_data):\n#     y_true = train_data.get_label()\n#     metric = np.sqrt( (y_true-preds)**2)\n#     higher_better = False\n#     return (\"unweighted_rmse\", metric, higher_better)\n\ncatetorical_idxs = list(range(X_train_preprocessed.shape[0]-len(combined_categoricals), X_train_preprocessed.shape[0]))\n\ndef lgbm_error(trial):\n\n    lgb_params = LGBM_DEFAULT_PARAMS | {\n        # \"booster\": trial.suggest_categorical(\"booster\", [\"gbtree\", \"gblinear\", \"dart\"]),\n        \"num_boost_round\": trial.suggest_int(\"num_boost_round\", 500, 700),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 8, 128),\n        # \"max_bin\": trial.suggest_int(\"max_bin\", 50, 500),\n        # \"lambda_l2\": trial.suggest_float(\"lambda_l2\", 1e-5, 1e0, log=True),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 1e-3, 0.5, log=True),\n        \"feature_fraction\": trial.suggest_float(\"feature_fraction\", 0.9, 1.0),\n        # 'max_bin': trial.suggest_int(\"max_bin\", 100, 500, log=True)\n    }\n    max_bin = trial.suggest_int(\"max_bin\", 10, 900)\n\n    train_data = lgb.Dataset(X_train_preprocessed, label=np.log1p(y_train), categorical_feature=catetorical_idxs, params=dict(max_bin=max_bin))\n\n    res = lgb.cv(\n        lgb_params,\n        train_data,\n        nfold=5,\n        metrics={'rmse'},\n        seed=666,\n        stratified=False,\n        # maximize=False,\n        # feval=unweighted_rmse\n        )\n    return np.mean(res['valid rmse-mean']) - 0.01*np.std(res['valid rmse-mean'])\n\nstudy = optuna.create_study(sampler=optuna.samplers.TPESampler(seed=666, n_startup_trials=33), direction=\"minimize\")\nstudy.optimize(lgbm_error, n_trials=100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:09:11.948993Z","iopub.execute_input":"2024-12-17T15:09:11.94942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# optuna.visualization.matplotlib.plot_param_importances(study);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.445825Z","iopub.status.idle":"2024-12-17T15:05:59.446245Z","shell.execute_reply.started":"2024-12-17T15:05:59.446056Z","shell.execute_reply":"2024-12-17T15:05:59.446076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# study.best_params, study.best_value","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.448365Z","iopub.status.idle":"2024-12-17T15:05:59.448983Z","shell.execute_reply.started":"2024-12-17T15:05:59.4487Z","shell.execute_reply":"2024-12-17T15:05:59.44873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# best_params = LGBM_DEFAULT_PARAMS | {'max_depth': 19, 'learning_rate': 0.31035270088788264}\n# s_params = {'max_depth': 31, 'lambda_l2': 0.2661611633251218, 'learning_rate': 0.34224581142203603, 'feature_fraction': 0.9649420579568114, 'weight_decay': 0.5}\ns_params = dict(study.best_params)\nmax_bin = s_params.pop('max_bin')\n# weight_decay = s_params.pop('weight_decay')\n# train_data.set_weight( train_data.get_label()**(-weight_decay))\nbest_params = LGBM_DEFAULT_PARAMS | s_params\ntrain_data = lgb.Dataset(X_train_preprocessed, label=np.log1p(y_train), categorical_feature=catetorical_idxs, params=dict(max_bin=max_bin))\n\nmodel = lgb.cv(params=best_params, train_set=train_data, return_cvbooster=True, seed=666, stratified=False)\ncv_preds = model['cvbooster'].predict(X_test_preprocessed)\npreds = np.expm1( np.mean(cv_preds, axis=0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.450642Z","iopub.status.idle":"2024-12-17T15:05:59.451117Z","shell.execute_reply.started":"2024-12-17T15:05:59.450878Z","shell.execute_reply":"2024-12-17T15:05:59.450897Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Let's look at the contributions to the loss function","metadata":{}},{"cell_type":"code","source":"cv_preds = model['cvbooster'].predict( X_train_preprocessed)\ntraining_preds = np.expm1( np.array(cv_preds).mean(axis=0))\n\n\nscaled_target = np.log1p(y_train)\nscaled_preds = np.log1p(training_preds)\n\ns = np.argsort((scaled_target-scaled_preds)**2)[::-1][:25]\n\nprint( np.c_[y_train, training_preds][s, :])\n\nplt.hist(y_train, density=True, bins=100, alpha=0.5, color='green');\nplt.hist(preds, density=True, bins=100, alpha=0.5, color='red');","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.453308Z","iopub.status.idle":"2024-12-17T15:05:59.453827Z","shell.execute_reply.started":"2024-12-17T15:05:59.453623Z","shell.execute_reply":"2024-12-17T15:05:59.453649Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Plotting training and evaluation distributions","metadata":{}},{"cell_type":"code","source":"plt.hist(y_train, density=True, bins=100, alpha=0.5, color='green');\nplt.hist(preds, density=True, bins=100, alpha=0.5, color='red');\n# plt.gca().set(xlim=(-10, 500));","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.456311Z","iopub.status.idle":"2024-12-17T15:05:59.45682Z","shell.execute_reply.started":"2024-12-17T15:05:59.45662Z","shell.execute_reply":"2024-12-17T15:05:59.456642Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Score distributions in test is very different from training, still this is what you get when minimizing for RMLE.","metadata":{}},{"cell_type":"code","source":"df_submission['Premium Amount'] = np.round(preds)\ndf_submission.head()\ndf_submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.458594Z","iopub.status.idle":"2024-12-17T15:05:59.459019Z","shell.execute_reply.started":"2024-12-17T15:05:59.458808Z","shell.execute_reply":"2024-12-17T15:05:59.458826Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Check which features can classify [0-50] range from [1000+] one","metadata":{}},{"cell_type":"code","source":"neg = y_train < 50\npos = y_train >= 50\n\nX_bin = np.vstack((X_train_preprocessed[neg], X_train_preprocessed[pos]))\ny_bin = np.r_[ np.ones(len(neg[neg])), np.zeros( len(pos[pos]))]\nX_bin = pd.DataFrame( X_bin, columns = feature_names)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.46034Z","iopub.status.idle":"2024-12-17T15:05:59.460716Z","shell.execute_reply.started":"2024-12-17T15:05:59.46054Z","shell.execute_reply":"2024-12-17T15:05:59.460557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# len(X_bin)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.462211Z","iopub.status.idle":"2024-12-17T15:05:59.462634Z","shell.execute_reply.started":"2024-12-17T15:05:59.462436Z","shell.execute_reply":"2024-12-17T15:05:59.462465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from collections import Counter\n# Counter( y_bin)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.464322Z","iopub.status.idle":"2024-12-17T15:05:59.464821Z","shell.execute_reply.started":"2024-12-17T15:05:59.464631Z","shell.execute_reply":"2024-12-17T15:05:59.464651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nLGBM_DEFAULT_PARAMS = {\n    \"verbosity\": -1,\n    \"objective\": \"classification\",\n    'seed': 666_666,\n    'boosting':'gbdt',\n    'metric': 'logloss',\n    'force_row_wise': True\n}\n\nimport lightgbm as lgb\ncatetorical_idxs = list(range(X_train_preprocessed.shape[0]-len(combined_categoricals), X_train_preprocessed.shape[0]))\nX_bin_tr, X_bin_te, y_bin_tr, y_bin_te = train_test_split( X_bin, y_bin, test_size=0.5)\n\nmodel = lgb.LGBMClassifier(max_bins=500, num_boost_rounds=50).fit( X_bin_tr, y_bin_tr, categorical_feature=catetorical_idxs, eval_set=(X_bin_te, y_bin_te))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.466577Z","iopub.status.idle":"2024-12-17T15:05:59.467023Z","shell.execute_reply.started":"2024-12-17T15:05:59.4668Z","shell.execute_reply":"2024-12-17T15:05:59.466819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb.plot_importance(model);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.469365Z","iopub.status.idle":"2024-12-17T15:05:59.469987Z","shell.execute_reply.started":"2024-12-17T15:05:59.469749Z","shell.execute_reply":"2024-12-17T15:05:59.469772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.inspection import permutation_importance\n\n# X_bin_tr, X_bin_te, y_bin_tr, y_bin_te = train_test_split( X_bin, y_bin, test_size=0.5)\n# model = lgb.LGBMClassifier(max_bins=500, num_boost_rounds=50).fit( X_bin_tr, y_bin_tr, categorical_feature=catetorical_idxs)\n\n# r = permutation_importance(model, X_bin_te, y_bin_te,\n#                            n_repeats=5,\n#                            scoring='roc_auc',\n#                            random_state=0)\n\n# for i in r.importances_mean.argsort()[::-1]:\n#     if r.importances_mean[i] - r.importances_std[i] > 0:\n#         print(f\"{feature_names[i]:<25}\"\n#               f\"{r.importances_mean[i]:.3f}\"\n#               f\" +/- {r.importances_std[i]:.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.471625Z","iopub.status.idle":"2024-12-17T15:05:59.472044Z","shell.execute_reply.started":"2024-12-17T15:05:59.471832Z","shell.execute_reply":"2024-12-17T15:05:59.47185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# r = permutation_importance(model, X_bin_te, y_bin_te,\n#                            n_repeats=5,\n#                            scoring='f1',\n#                            random_state=0)\n\n# for i in r.importances_mean.argsort()[::-1]:\n#     if r.importances_mean[i] - r.importances_std[i] > 0:\n#         print(f\"{feature_names[i]:<25}\"\n#               f\"{r.importances_mean[i]:.3f}\"\n#               f\" +/- {r.importances_std[i]:.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.473548Z","iopub.status.idle":"2024-12-17T15:05:59.47397Z","shell.execute_reply.started":"2024-12-17T15:05:59.473764Z","shell.execute_reply":"2024-12-17T15:05:59.473784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# hue = df_train['Premium Amount'].gt(50)\n# sns.histplot(data= df_train, x='Annual Income' , hue=hue, common_norm=False, stat='percent', bins=25);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.477395Z","iopub.status.idle":"2024-12-17T15:05:59.477858Z","shell.execute_reply.started":"2024-12-17T15:05:59.477654Z","shell.execute_reply":"2024-12-17T15:05:59.477676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# hue = df_train['Premium Amount'].gt(50)\n# sns.histplot(data= df_train, x='Credit Score' , hue=hue, common_norm=False, stat='percent', bins=25);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.479809Z","iopub.status.idle":"2024-12-17T15:05:59.480258Z","shell.execute_reply.started":"2024-12-17T15:05:59.480061Z","shell.execute_reply":"2024-12-17T15:05:59.480082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# hue = df_train['Premium Amount'].gt(50)\n# sns.histplot(data= df_train, x='Previous Claims' , hue=hue, common_norm=False, stat='percent', bins=25);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.482239Z","iopub.status.idle":"2024-12-17T15:05:59.48266Z","shell.execute_reply.started":"2024-12-17T15:05:59.482471Z","shell.execute_reply":"2024-12-17T15:05:59.482492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import pandas as pd\n# import numpy as np\n# from sklearn.compose import make_column_transformer\n# from sklearn.pipeline import make_pipeline\n# from sklearn.preprocessing import TargetEncoder, FunctionTransformer\n# from sklearn.tree import DecisionTreeRegressor\n\n# X = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv', index_col='id')\n# y = X.pop('Premium Amount')\n\n# model = make_pipeline(\n#     make_column_transformer(\n#         ('passthrough', list(X.select_dtypes('number').columns)),\n#         (FunctionTransformer(\n#             lambda X: pd.to_datetime(X.squeeze()).astype(int).to_frame()\n#         ), ['Policy Start Date']),\n#         remainder=TargetEncoder(target_type='continuous', random_state=0),\n#         force_int_remainder_cols=False\n#     ),\n#     DecisionTreeRegressor(max_depth=18, random_state=0)\n# )\n\n# model.fit(X, np.log1p(y))\n# ct = model.steps[0][1]\n# dtr = model.steps[1][1]\n# print(F'Number of cells: {dtr.get_n_leaves()}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.484722Z","iopub.status.idle":"2024-12-17T15:05:59.485152Z","shell.execute_reply.started":"2024-12-17T15:05:59.484926Z","shell.execute_reply":"2024-12-17T15:05:59.484973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df = pd.DataFrame({'y_true': y, 'leaf': dtr.apply(ct.transform(X))})\n# # get leaf indices for low-premium samples\n# leaves = df[(df['y_true']>=20)&(df['y_true']<=40)]['leaf'].values\n\n# def ExpMeanLog(s): return np.expm1(np.mean(np.log1p(s)))\n# def MSLE(s): return np.var(np.log1p(s))\n\n# # all samples with these leaf indices\n# df2 = df[df['leaf'].isin(leaves)].groupby('leaf').agg(['min','max', 'count', ExpMeanLog, MSLE])\n# df2.columns = df2.columns.droplevel(0)\n# display(df2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.487287Z","iopub.status.idle":"2024-12-17T15:05:59.487735Z","shell.execute_reply.started":"2024-12-17T15:05:59.487535Z","shell.execute_reply":"2024-12-17T15:05:59.48756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df2.MSLE.plot(kind='hist', bins=50);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.489293Z","iopub.status.idle":"2024-12-17T15:05:59.489816Z","shell.execute_reply.started":"2024-12-17T15:05:59.489603Z","shell.execute_reply":"2024-12-17T15:05:59.489625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# mask = df2['min'].lt(40) & df2['max'].gt(1000)\n# mask = df2.MSLE.gt(3)\n# mixed_leaves = df2[mask].index.tolist()\n# df['is_in_mixed_leaf'] = df['leaf'].isin( set(mixed_leaves))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.491374Z","iopub.status.idle":"2024-12-17T15:05:59.491769Z","shell.execute_reply.started":"2024-12-17T15:05:59.491575Z","shell.execute_reply":"2024-12-17T15:05:59.491594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df[df['is_in_mixed_leaf']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.493129Z","iopub.status.idle":"2024-12-17T15:05:59.493516Z","shell.execute_reply.started":"2024-12-17T15:05:59.493329Z","shell.execute_reply":"2024-12-17T15:05:59.493348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:05:59.495622Z","iopub.status.idle":"2024-12-17T15:05:59.496017Z","shell.execute_reply.started":"2024-12-17T15:05:59.495809Z","shell.execute_reply":"2024-12-17T15:05:59.495825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}