{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Regression with an Insurance Dataset","metadata":{}},{"cell_type":"markdown","source":"## Import Libraries","metadata":{}},{"cell_type":"code","source":"pip install --upgrade scikit-learn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:17:44.445407Z","iopub.execute_input":"2024-12-14T14:17:44.445721Z","iopub.status.idle":"2024-12-14T14:17:56.360161Z","shell.execute_reply.started":"2024-12-14T14:17:44.445689Z","shell.execute_reply":"2024-12-14T14:17:56.359115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install optuna-integration[sklearn]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:17:58.783805Z","iopub.execute_input":"2024-12-14T14:17:58.784086Z","iopub.status.idle":"2024-12-14T14:18:05.936799Z","shell.execute_reply.started":"2024-12-14T14:17:58.784062Z","shell.execute_reply":"2024-12-14T14:18:05.935483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sklearn\nsklearn.__version__","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:18:05.938631Z","iopub.execute_input":"2024-12-14T14:18:05.938934Z","iopub.status.idle":"2024-12-14T14:18:06.509659Z","shell.execute_reply.started":"2024-12-14T14:18:05.938901Z","shell.execute_reply":"2024-12-14T14:18:06.508668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Importing Libraries\nimport numpy as np \nimport pandas as pd\n# Data Visualization\nimport missingno as msno\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n# Ignore warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\", \"use_inf_as_na\")\n# Correlation Matrix\nimport phik\nfrom phik.report import plot_correlation_matrix\n\nimport lightgbm as lgb\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler, MinMaxScaler, RobustScaler\nfrom sklearn.impute import SimpleImputer \nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import root_mean_squared_error\nimport optuna","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:18:10.16303Z","iopub.execute_input":"2024-12-14T14:18:10.163984Z","iopub.status.idle":"2024-12-14T14:18:11.79158Z","shell.execute_reply.started":"2024-12-14T14:18:10.163958Z","shell.execute_reply":"2024-12-14T14:18:11.790804Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load Datasets","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest_data = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:18:33.409119Z","iopub.execute_input":"2024-12-14T14:18:33.409621Z","iopub.status.idle":"2024-12-14T14:18:40.138852Z","shell.execute_reply.started":"2024-12-14T14:18:33.409597Z","shell.execute_reply":"2024-12-14T14:18:40.138114Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train Data Overview","metadata":{}},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:18:50.206686Z","iopub.execute_input":"2024-12-14T14:18:50.206953Z","iopub.status.idle":"2024-12-14T14:18:50.237493Z","shell.execute_reply.started":"2024-12-14T14:18:50.206932Z","shell.execute_reply":"2024-12-14T14:18:50.236839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:18:53.080825Z","iopub.execute_input":"2024-12-14T14:18:53.081114Z","iopub.status.idle":"2024-12-14T14:18:53.531798Z","shell.execute_reply.started":"2024-12-14T14:18:53.081091Z","shell.execute_reply":"2024-12-14T14:18:53.530905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert objects to datetime\ntrain_data['Policy Start Date'] = pd.to_datetime(train_data['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:18:56.8472Z","iopub.execute_input":"2024-12-14T14:18:56.847508Z","iopub.status.idle":"2024-12-14T14:18:57.086721Z","shell.execute_reply.started":"2024-12-14T14:18:56.847485Z","shell.execute_reply":"2024-12-14T14:18:57.085889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:20:03.023299Z","iopub.execute_input":"2024-12-14T14:20:03.023662Z","iopub.status.idle":"2024-12-14T14:20:03.420432Z","shell.execute_reply.started":"2024-12-14T14:20:03.023639Z","shell.execute_reply":"2024-12-14T14:20:03.419587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"msno.bar(train_data)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:20:08.580788Z","iopub.execute_input":"2024-12-14T14:20:08.581074Z","iopub.status.idle":"2024-12-14T14:20:10.049918Z","shell.execute_reply.started":"2024-12-14T14:20:08.581051Z","shell.execute_reply":"2024-12-14T14:20:10.048951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:20:11.665034Z","iopub.execute_input":"2024-12-14T14:20:11.665335Z","iopub.status.idle":"2024-12-14T14:20:12.594117Z","shell.execute_reply.started":"2024-12-14T14:20:11.665311Z","shell.execute_reply":"2024-12-14T14:20:12.593187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in train_data.select_dtypes(include=['object']).columns:\n    print(train_data[col].unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:22:42.540346Z","iopub.execute_input":"2024-12-14T14:22:42.540671Z","iopub.status.idle":"2024-12-14T14:22:43.265589Z","shell.execute_reply.started":"2024-12-14T14:22:42.54065Z","shell.execute_reply":"2024-12-14T14:22:43.264721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print top 10 unique value counts for each categorical column\nfor column in train_data.select_dtypes(include=['object']).columns:\n    print(f\"\\nTop value counts in '{column}':\\n{train_data[column].value_counts()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:24:25.231298Z","iopub.execute_input":"2024-12-14T14:24:25.231604Z","iopub.status.idle":"2024-12-14T14:24:26.160633Z","shell.execute_reply.started":"2024-12-14T14:24:25.231583Z","shell.execute_reply":"2024-12-14T14:24:26.159717Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualization of Numerical Features","metadata":{}},{"cell_type":"code","source":"def plot_numerical_features(df):\n    numerical_cols = df.select_dtypes(include=['number']).columns\n\n    num_cols = len(numerical_cols)\n    fig, axes = plt.subplots(nrows=num_cols, ncols=2, figsize=(15, 5*num_cols)) \n\n    for i, col in enumerate(numerical_cols):\n        # Histogram\n        sns.histplot(data=df, x=col, kde=True, ax=axes[i, 0]) \n        axes[i, 0].set_title(f'Histogram {col}')\n        axes[i, 0].set_xlabel(col)\n        axes[i, 0].set_ylabel('Frequency')\n\n        # Boxplot\n        sns.boxplot(y=df[col], ax=axes[i, 1])\n        axes[i, 1].set_title(f'Boxplot {col}')\n        axes[i, 1].set_ylabel(col)\n\n    plt.tight_layout() \n    plt.show()\n\n\nplot_numerical_features(train_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:26:51.5221Z","iopub.execute_input":"2024-12-14T14:26:51.522465Z","iopub.status.idle":"2024-12-14T14:27:28.786002Z","shell.execute_reply.started":"2024-12-14T14:26:51.522436Z","shell.execute_reply":"2024-12-14T14:27:28.785187Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualization of Categorical Features","metadata":{}},{"cell_type":"code","source":"def plot_categorical_features(df, target_column):\n    categorical_cols = df.select_dtypes(include=['object']).columns\n\n    cat_cols = len(categorical_cols)\n    fig, axes = plt.subplots(nrows=cat_cols, ncols=2, figsize=(15, 5*cat_cols)) \n\n    for i, col in enumerate(categorical_cols):\n        # Histogram\n        sns.countplot(data=df, x=col, ax=axes[i, 0]) \n        axes[i, 0].set_title(f'Histogram {col}')\n        axes[i, 0].set_xlabel(col)\n        axes[i, 0].set_ylabel('Frequency')\n\n        # Boxplot\n        sns.boxplot(x=df[col], y=df[target_column], ax=axes[i, 1])\n        axes[i, 1].set_title(f'Boxplot {col}')\n        axes[i, 1].set_ylabel(col)\n\n    plt.tight_layout() \n    plt.show()\n\n\nplot_categorical_features(train_data, target_column='Premium Amount')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:28:08.296182Z","iopub.execute_input":"2024-12-14T14:28:08.296935Z","iopub.status.idle":"2024-12-14T14:28:20.059805Z","shell.execute_reply.started":"2024-12-14T14:28:08.296908Z","shell.execute_reply":"2024-12-14T14:28:20.058929Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Correlation Heatmap of Numerical Features","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (9, 6))\nsns.heatmap(train_data[train_data.select_dtypes(include=['number']).columns].corr(), annot=True, fmt='.2f', cmap='crest')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:29:38.285054Z","iopub.execute_input":"2024-12-14T14:29:38.285352Z","iopub.status.idle":"2024-12-14T14:29:38.919335Z","shell.execute_reply.started":"2024-12-14T14:29:38.28533Z","shell.execute_reply":"2024-12-14T14:29:38.91836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_col_names = ['Age',\n 'Annual Income',\n 'Number of Dependents',\n 'Health Score',\n 'Previous Claims',\n 'Vehicle Age',\n 'Credit Score',\n 'Insurance Duration']\n\ncat_col_names = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location',\n       'Policy Type', 'Customer Feedback', 'Smoking Status',\n       'Exercise Frequency', 'Property Type']\n\ntarget_column = 'Premium Amount'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:29:50.63917Z","iopub.execute_input":"2024-12-14T14:29:50.639508Z","iopub.status.idle":"2024-12-14T14:29:50.644113Z","shell.execute_reply.started":"2024-12-14T14:29:50.639485Z","shell.execute_reply":"2024-12-14T14:29:50.643274Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Split Data: Train and Validation Datasets","metadata":{}},{"cell_type":"code","source":"RANDOM_STATE = 42\nTEST_SIZE = 0.25\n\n\nX = train_data[num_col_names + cat_col_names]\ny = train_data[target_column]\n\nX_train, X_val, y_train, y_val = train_test_split(\n    X,\n    y,\n    test_size = TEST_SIZE, \n    random_state = RANDOM_STATE)\n\nprint(X_train.shape, X_val.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:30:02.350004Z","iopub.execute_input":"2024-12-14T14:30:02.350278Z","iopub.status.idle":"2024-12-14T14:30:02.903302Z","shell.execute_reply.started":"2024-12-14T14:30:02.350257Z","shell.execute_reply":"2024-12-14T14:30:02.902357Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Build Pipeline","metadata":{}},{"cell_type":"code","source":"# SimpleImputer + OHE\nohe_pipe = Pipeline(\n    [\n        (\n            'simpleImputer_ohe', \n            SimpleImputer(missing_values=np.nan, strategy='most_frequent')\n        ),\n        (\n            'ohe', \n            OneHotEncoder(drop='first', handle_unknown='ignore', sparse_output=False)\n        )\n    ]\n) \n\nnum_pipeline = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median'))])\n\ndata_preprocessor = ColumnTransformer([\n    ('ohe', ohe_pipe, cat_col_names),\n    ('num', num_pipeline, num_col_names)\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:30:25.204872Z","iopub.execute_input":"2024-12-14T14:30:25.205146Z","iopub.status.idle":"2024-12-14T14:30:25.210156Z","shell.execute_reply.started":"2024-12-14T14:30:25.205125Z","shell.execute_reply":"2024-12-14T14:30:25.20927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_preprocessed = data_preprocessor.fit_transform(X_train)\nX_val_preprocessed = data_preprocessor.transform(X_val)\nX_test = test_data[num_col_names + cat_col_names]\nX_test_preprocessed = data_preprocessor.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:30:35.452151Z","iopub.execute_input":"2024-12-14T14:30:35.452483Z","iopub.status.idle":"2024-12-14T14:30:42.602133Z","shell.execute_reply.started":"2024-12-14T14:30:35.452458Z","shell.execute_reply":"2024-12-14T14:30:42.601421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the train and validation data into the LightGBM dataset object\nlgb_train = lgb.Dataset(X_train_preprocessed, y_train)\nlgb_eval = lgb.Dataset(X_val_preprocessed, y_val, reference=lgb_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:31:26.741293Z","iopub.execute_input":"2024-12-14T14:31:26.742221Z","iopub.status.idle":"2024-12-14T14:31:26.74588Z","shell.execute_reply.started":"2024-12-14T14:31:26.74218Z","shell.execute_reply":"2024-12-14T14:31:26.744909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial):\n    params = {\n        'objective': 'regression',\n        'metric': 'rmse',\n        'boosting_type': trial.suggest_categorical(\"boosting_type\", [\"gbdt\", \"dart\"]),\n        'learning_rate': trial.suggest_float('learning_rate', 0.001, 0.1, log=True),\n        'num_leaves': trial.suggest_int('num_leaves', 20, 300),\n        'max_depth': trial.suggest_int('max_depth', -1, 20),  # -1 означает отсутствие ограничения\n        'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 10, 100),\n        'min_sum_hessian_in_leaf': trial.suggest_loguniform('min_sum_hessian_in_leaf', 1e-3, 10.0),\n        'feature_fraction': trial.suggest_uniform('feature_fraction', 0.5, 1.0),\n        'bagging_fraction': trial.suggest_uniform('bagging_fraction', 0.5, 1.0),\n        'bagging_freq': trial.suggest_int('bagging_freq', 1, 10),\n        'lambda_l1': trial.suggest_loguniform('lambda_l1', 1e-8, 10.0),\n        'lambda_l2': trial.suggest_loguniform('lambda_l2', 1e-8, 10.0),\n        'max_bin': trial.suggest_int('max_bin', 100, 500)\n    }\n\n    model = lgb.LGBMRegressor(**params)\n    model.fit(X_train_preprocessed, y_train)\n    predictions = model.predict(X_val_preprocessed)\n    rmse = root_mean_squared_error(y_val, predictions)\n    return rmse","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:38:42.349084Z","iopub.execute_input":"2024-12-14T14:38:42.349354Z","iopub.status.idle":"2024-12-14T14:38:42.355924Z","shell.execute_reply.started":"2024-12-14T14:38:42.349333Z","shell.execute_reply":"2024-12-14T14:38:42.354807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study = optuna.create_study(direction='minimize')\nstudy.optimize(objective, n_trials=20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:38:46.909932Z","iopub.execute_input":"2024-12-14T14:38:46.91026Z","iopub.status.idle":"2024-12-14T14:45:41.610824Z","shell.execute_reply.started":"2024-12-14T14:38:46.910231Z","shell.execute_reply":"2024-12-14T14:45:41.609884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_params = study.best_params\nfinal_model = lgb.LGBMRegressor(**final_params)\nfinal_model.fit(X_train_preprocessed, y_train, eval_set=[(X_val_preprocessed, y_val)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:50:04.296233Z","iopub.execute_input":"2024-12-14T14:50:04.296574Z","iopub.status.idle":"2024-12-14T14:50:19.19929Z","shell.execute_reply.started":"2024-12-14T14:50:04.296549Z","shell.execute_reply":"2024-12-14T14:50:19.197829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test_preprocessed = data_preprocessor.transform(X_test)\nfinal_preds = final_model.predict(X_test_preprocessed)\npd.DataFrame({'id': test_data['id'], 'Premium Amount': final_preds}).to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T14:50:46.41051Z","iopub.execute_input":"2024-12-14T14:50:46.410838Z","iopub.status.idle":"2024-12-14T14:50:54.034137Z","shell.execute_reply.started":"2024-12-14T14:50:46.410815Z","shell.execute_reply":"2024-12-14T14:50:54.033009Z"}},"outputs":[],"execution_count":null}]}