{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Insurance Premiums Prediction Using LightGBM & Optuna\n\nTuning hyperparameters is always a great hassle in training machine learning models.\n\nHere we will combine LigbtGBM with Optuna, which is a powerful and flexible tool for hyperparameter optimization.    \n\nIts ease of use, coupled with advanced features like pruning and trial visualization, makes it an excellent choice for parameter tuning. \n\nRegression with an Insurance Dataset (train & test csvs)   \nhttps://www.kaggle.com/competitions/playground-series-s4e12/overview","metadata":{}},{"cell_type":"markdown","source":"## 1. Importe Libraries & Data","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\n\nimport optuna\nimport lightgbm as lgb\nfrom optuna import Trial\nfrom plotly.io import renderers\n\n# Set renderer to inline for Kaggle compatibility\nrenderers.default = \"notebook\"\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nsns.set_style('whitegrid')\nplt.rcParams['figure.figsize'] = 10, 6\n\ncolors = sns.color_palette('tab10')\n\npd.set_option('display.float_format', '{:.2f}'.format)\npd.options.display.float_format = '{:,.2f}'.format\nnp.set_printoptions(formatter={'float': '{: 0.2f}'.format})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:20.484588Z","iopub.execute_input":"2024-12-22T22:57:20.485004Z","iopub.status.idle":"2024-12-22T22:57:20.492958Z","shell.execute_reply.started":"2024-12-22T22:57:20.48497Z","shell.execute_reply":"2024-12-22T22:57:20.491833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv', )\n\ntrain.drop(columns=['id', 'Policy Start Date'], axis=1, inplace=True)\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:20.494516Z","iopub.execute_input":"2024-12-22T22:57:20.494913Z","iopub.status.idle":"2024-12-22T22:57:25.52487Z","shell.execute_reply.started":"2024-12-22T22:57:20.494875Z","shell.execute_reply":"2024-12-22T22:57:25.523823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:25.526741Z","iopub.execute_input":"2024-12-22T22:57:25.527097Z","iopub.status.idle":"2024-12-22T22:57:26.090016Z","shell.execute_reply.started":"2024-12-22T22:57:25.527069Z","shell.execute_reply":"2024-12-22T22:57:26.088925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:26.091767Z","iopub.execute_input":"2024-12-22T22:57:26.092175Z","iopub.status.idle":"2024-12-22T22:57:26.754788Z","shell.execute_reply.started":"2024-12-22T22:57:26.092133Z","shell.execute_reply":"2024-12-22T22:57:26.753765Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:26.755699Z","iopub.execute_input":"2024-12-22T22:57:26.755978Z","iopub.status.idle":"2024-12-22T22:57:27.30965Z","shell.execute_reply.started":"2024-12-22T22:57:26.755957Z","shell.execute_reply":"2024-12-22T22:57:27.308654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:27.310642Z","iopub.execute_input":"2024-12-22T22:57:27.310919Z","iopub.status.idle":"2024-12-22T22:57:27.988096Z","shell.execute_reply.started":"2024-12-22T22:57:27.310895Z","shell.execute_reply":"2024-12-22T22:57:27.983597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv', )\n\ntest.drop(columns=['id', 'Policy Start Date'], axis=1, inplace=True)\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:27.989364Z","iopub.execute_input":"2024-12-22T22:57:27.989785Z","iopub.status.idle":"2024-12-22T22:57:31.021283Z","shell.execute_reply.started":"2024-12-22T22:57:27.989746Z","shell.execute_reply":"2024-12-22T22:57:31.020254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:31.024109Z","iopub.execute_input":"2024-12-22T22:57:31.024407Z","iopub.status.idle":"2024-12-22T22:57:31.401246Z","shell.execute_reply.started":"2024-12-22T22:57:31.024383Z","shell.execute_reply":"2024-12-22T22:57:31.400248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:31.402957Z","iopub.execute_input":"2024-12-22T22:57:31.403234Z","iopub.status.idle":"2024-12-22T22:57:31.806499Z","shell.execute_reply.started":"2024-12-22T22:57:31.403212Z","shell.execute_reply":"2024-12-22T22:57:31.805372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:31.807567Z","iopub.execute_input":"2024-12-22T22:57:31.807852Z","iopub.status.idle":"2024-12-22T22:57:32.183166Z","shell.execute_reply.started":"2024-12-22T22:57:31.807829Z","shell.execute_reply":"2024-12-22T22:57:32.182262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:32.183913Z","iopub.execute_input":"2024-12-22T22:57:32.184147Z","iopub.status.idle":"2024-12-22T22:57:32.579554Z","shell.execute_reply.started":"2024-12-22T22:57:32.184127Z","shell.execute_reply":"2024-12-22T22:57:32.578494Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"### 2.1 Handle the Missing Values","metadata":{}},{"cell_type":"code","source":"# Create two subset for future use\ncategorical_cols = train.select_dtypes(include='object').columns\nprint(categorical_cols)\n\nnumerical_cols = train.select_dtypes(include='number').columns\nprint(numerical_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:32.580479Z","iopub.execute_input":"2024-12-22T22:57:32.580737Z","iopub.status.idle":"2024-12-22T22:57:32.742039Z","shell.execute_reply.started":"2024-12-22T22:57:32.580715Z","shell.execute_reply":"2024-12-22T22:57:32.741112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Exclude the last element in numerical_cols\nnumerical_cols_excluding_last = numerical_cols[:-1]\n\n# Filling missing values in train data\nfor col in train[numerical_cols_excluding_last]:\n    train[col].fillna(train[col].median(), inplace=True)\n\n# Filling missing values in test data\nfor col in test[numerical_cols_excluding_last]:\n    test[col].fillna(test[col].median(), inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:32.742884Z","iopub.execute_input":"2024-12-22T22:57:32.743132Z","iopub.status.idle":"2024-12-22T22:57:33.223717Z","shell.execute_reply.started":"2024-12-22T22:57:32.743111Z","shell.execute_reply":"2024-12-22T22:57:33.222462Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filling Missing Categorical Values\nfor col in train[categorical_cols]:\n    train[col].fillna('Unknown', inplace=True)\n\nfor col in test[categorical_cols]:\n    test[col].fillna('Unknown', inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:33.225241Z","iopub.execute_input":"2024-12-22T22:57:33.225643Z","iopub.status.idle":"2024-12-22T22:57:34.455935Z","shell.execute_reply.started":"2024-12-22T22:57:33.225608Z","shell.execute_reply":"2024-12-22T22:57:34.454866Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2.2 Reduce Memory Usage","metadata":{}},{"cell_type":"code","source":"# Reduce memory usage\ndef reduce_memory_usage(data):\n\n    for col in data.columns:\n        if pd.api.types.is_numeric_dtype(data[col]):\n            if pd.api.types.is_integer_dtype(data[col]):\n                data[col] = pd.to_numeric(data[col], downcast='integer')\n            elif pd.api.types.is_float_dtype(data[col]):\n                data[col] = pd.to_numeric(data[col], downcast='float')\n                \n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:34.456928Z","iopub.execute_input":"2024-12-22T22:57:34.457224Z","iopub.status.idle":"2024-12-22T22:57:34.463127Z","shell.execute_reply.started":"2024-12-22T22:57:34.457198Z","shell.execute_reply":"2024-12-22T22:57:34.462126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = reduce_memory_usage(train)\ntest = reduce_memory_usage(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:34.464287Z","iopub.execute_input":"2024-12-22T22:57:34.464663Z","iopub.status.idle":"2024-12-22T22:57:34.6876Z","shell.execute_reply.started":"2024-12-22T22:57:34.464631Z","shell.execute_reply":"2024-12-22T22:57:34.686464Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2.3 Visualize the Data","metadata":{}},{"cell_type":"code","source":"# Countplot for categoric features\nplt.subplots(3, 4, figsize=(16, 9))\n\nfor i, col in enumerate(categorical_cols, 1):\n   \n    if col != 'Policy Start Date':\n        plt.subplot(3, 4, i)\n        sns.countplot(data=train, x=col, hue=col, dodge=False,\n                    order=train[col].value_counts().index,\n                    )\n        plt.title(f'Countplot of {col}')\n        plt.legend().set_visible(False)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:34.688589Z","iopub.execute_input":"2024-12-22T22:57:34.688894Z","iopub.status.idle":"2024-12-22T22:57:48.733884Z","shell.execute_reply.started":"2024-12-22T22:57:34.68887Z","shell.execute_reply":"2024-12-22T22:57:48.732892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Histogram to check for distributions\nplt.subplots(3, 3, figsize=(15, 9))\n\nfor i, col in enumerate(numerical_cols, 1):\n    \n    plt.subplot(3, 3, i)\n    if train[numerical_cols][col].nunique() <= 10:\n        sns.countplot(data=train, x=col, dodge=False,\n                      palette='tab10',)\n    else:\n        sns.histplot(data=train, x=col, bins=20)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:48.734862Z","iopub.execute_input":"2024-12-22T22:57:48.735138Z","iopub.status.idle":"2024-12-22T22:57:53.272663Z","shell.execute_reply.started":"2024-12-22T22:57:48.735114Z","shell.execute_reply":"2024-12-22T22:57:53.271552Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. LigntGBM with Optuna","metadata":{}},{"cell_type":"markdown","source":"### 3.1 Converting Premium Amount to Log(Amount)\n\n**np.log(x)**\n\n\n- Definition: Computes the natural logarithm (ln(x)).\n- When to Use:\n    - When the input values (x) are significantly larger than 0.\n    - Suitable for datasets without very small positive values or values close to zero.\n\n**np.exp(x):**\n\n- Use it to convert back to the original values.","metadata":{}},{"cell_type":"code","source":"# Apply natural log transformation\ntrain['Premium Amount'] = np.log(train['Premium Amount'])\n# train['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:53.273564Z","iopub.execute_input":"2024-12-22T22:57:53.273945Z","iopub.status.idle":"2024-12-22T22:57:53.282747Z","shell.execute_reply.started":"2024-12-22T22:57:53.273899Z","shell.execute_reply":"2024-12-22T22:57:53.281667Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3.2 Feature Selection & Transformation","metadata":{}},{"cell_type":"code","source":"# Train-test split the train set\nX = train.drop(columns='Premium Amount')\ny = train['Premium Amount']\n\n# from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, \n                                                    random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:53.283741Z","iopub.execute_input":"2024-12-22T22:57:53.284006Z","iopub.status.idle":"2024-12-22T22:57:54.127633Z","shell.execute_reply.started":"2024-12-22T22:57:53.283984Z","shell.execute_reply":"2024-12-22T22:57:54.12678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize and fit scaler on training data\nscaler = StandardScaler()\nX_train_numeric_scaled = scaler.fit_transform(X_train[numerical_cols_excluding_last])\n\n# Transform the test set using the same scaler\nX_test_numeric_scaled = scaler.transform(X_test[numerical_cols_excluding_last])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:54.128635Z","iopub.execute_input":"2024-12-22T22:57:54.128937Z","iopub.status.idle":"2024-12-22T22:57:54.302727Z","shell.execute_reply.started":"2024-12-22T22:57:54.128909Z","shell.execute_reply":"2024-12-22T22:57:54.30191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize and fit encoder on training data\nencoder = OneHotEncoder(handle_unknown='ignore')\nX_train_categorical_encoded = encoder.fit_transform(X_train[categorical_cols]).toarray()\n\n# Transform the test set using the same encoder\nX_test_categorical_encoded = encoder.transform(X_test[categorical_cols]).toarray()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:54.303603Z","iopub.execute_input":"2024-12-22T22:57:54.30387Z","iopub.status.idle":"2024-12-22T22:57:57.659179Z","shell.execute_reply.started":"2024-12-22T22:57:54.303848Z","shell.execute_reply":"2024-12-22T22:57:57.658145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine numeric and categorical features\nX_train_preprocessed = np.hstack((X_train_numeric_scaled, X_train_categorical_encoded))\n\n# Combine numeric and categorical features (check shapes)\nif X_train_preprocessed.shape[0] != y_train.shape[0]:\n    raise ValueError(\"Inconsistent sample sizes after preprocessing!\")\n\nX_test_preprocessed = np.hstack((X_test_numeric_scaled, X_test_categorical_encoded))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:57.663063Z","iopub.execute_input":"2024-12-22T22:57:57.663377Z","iopub.status.idle":"2024-12-22T22:57:57.880578Z","shell.execute_reply.started":"2024-12-22T22:57:57.66335Z","shell.execute_reply":"2024-12-22T22:57:57.879501Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3.3 Build & Tune the Model","metadata":{}},{"cell_type":"code","source":"def objective(trial):\n\n    params = {\n        'objective': 'regression',\n        'metric': 'rmse',\n        'verbosity': -1,\n        'bagging_freq': 1,\n        # Essential parameters with reduced ranges:\n        'n_estimators': trial.suggest_int('n_estimators', 100, 300),  \n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.1, log=True), \n        'num_leaves': trial.suggest_int('num_leaves', 10, 100), \n        'subsample': trial.suggest_float('subsample', 0.5, 1.0),  \n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),  \n        'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 5, 50)  \n    }\n\n    model = lgb.LGBMRegressor(**params, verbose=-1)\n    model.fit(X_train_preprocessed, y_train)\n\n    predictions = model.predict(X_test_preprocessed)\n\n    # Evaluate the model performance\n    # rmse = root_mean_squared_error(y_test, predictions)\n    rmse = mean_squared_error(y_test, predictions, squared=False)\n    \n    return rmse","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:57.882201Z","iopub.execute_input":"2024-12-22T22:57:57.882507Z","iopub.status.idle":"2024-12-22T22:57:57.888541Z","shell.execute_reply.started":"2024-12-22T22:57:57.882477Z","shell.execute_reply":"2024-12-22T22:57:57.887529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study = optuna.create_study(direction='minimize')\nstudy.optimize(objective, n_trials=30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T22:57:57.889364Z","iopub.execute_input":"2024-12-22T22:57:57.88961Z","iopub.status.idle":"2024-12-22T23:04:19.382519Z","shell.execute_reply.started":"2024-12-22T22:57:57.889588Z","shell.execute_reply":"2024-12-22T23:04:19.381445Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"study.best_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:19.383579Z","iopub.execute_input":"2024-12-22T23:04:19.383865Z","iopub.status.idle":"2024-12-22T23:04:19.389884Z","shell.execute_reply.started":"2024-12-22T23:04:19.383839Z","shell.execute_reply":"2024-12-22T23:04:19.389006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_value = study.best_value\nprint('Best RMSE:', best_value)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:19.390873Z","iopub.execute_input":"2024-12-22T23:04:19.39122Z","iopub.status.idle":"2024-12-22T23:04:19.406335Z","shell.execute_reply.started":"2024-12-22T23:04:19.391185Z","shell.execute_reply":"2024-12-22T23:04:19.405398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the model with the best parameters\nbest_params = study.best_params\nfinal_model = lgb.LGBMRegressor(**best_params, objective='regression', \n                                force_col_wise=True, metric='rmse')\n\n# Fit the final model\nfinal_model.fit(X_train_preprocessed, y_train)\n\n# Make predictions\nfinal_predictions = final_model.predict(X_test_preprocessed)\n\n# Optionally calculate RMSE or any other metric on the predictions\nrmse_final = np.sqrt(mean_squared_error(y_test, final_predictions))\nprint(f\"Final RMSE: {rmse_final:.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:19.407451Z","iopub.execute_input":"2024-12-22T23:04:19.407874Z","iopub.status.idle":"2024-12-22T23:04:31.394031Z","shell.execute_reply.started":"2024-12-22T23:04:19.407836Z","shell.execute_reply":"2024-12-22T23:04:31.393028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert predictions back to the original scale\nfinal_predictions = np.exp(final_predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:31.394947Z","iopub.execute_input":"2024-12-22T23:04:31.39533Z","iopub.status.idle":"2024-12-22T23:04:31.401023Z","shell.execute_reply.started":"2024-12-22T23:04:31.395271Z","shell.execute_reply":"2024-12-22T23:04:31.400152Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3.4 Visualize the Model","metadata":{}},{"cell_type":"code","source":"# Optimization History:\nfig = optuna.visualization.plot_optimization_history(study)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:31.401837Z","iopub.execute_input":"2024-12-22T23:04:31.402073Z","iopub.status.idle":"2024-12-22T23:04:32.243364Z","shell.execute_reply.started":"2024-12-22T23:04:31.402053Z","shell.execute_reply":"2024-12-22T23:04:32.242301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Parameter Importance:\nfig = optuna.visualization.plot_param_importances(study)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:32.24443Z","iopub.execute_input":"2024-12-22T23:04:32.244741Z","iopub.status.idle":"2024-12-22T23:04:32.651086Z","shell.execute_reply.started":"2024-12-22T23:04:32.244716Z","shell.execute_reply":"2024-12-22T23:04:32.650158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Parallel Coordinate Plot:\nfig = optuna.visualization.plot_parallel_coordinate(study)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:32.651977Z","iopub.execute_input":"2024-12-22T23:04:32.652242Z","iopub.status.idle":"2024-12-22T23:04:32.702811Z","shell.execute_reply.started":"2024-12-22T23:04:32.652212Z","shell.execute_reply":"2024-12-22T23:04:32.701868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Contour Plot:\nfig = optuna.visualization.plot_contour(study)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:32.703712Z","iopub.execute_input":"2024-12-22T23:04:32.703993Z","iopub.status.idle":"2024-12-22T23:04:33.420907Z","shell.execute_reply.started":"2024-12-22T23:04:32.70397Z","shell.execute_reply":"2024-12-22T23:04:33.419811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Slice Plot:\nfig = optuna.visualization.plot_slice(study)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:33.422033Z","iopub.execute_input":"2024-12-22T23:04:33.422357Z","iopub.status.idle":"2024-12-22T23:04:33.516702Z","shell.execute_reply.started":"2024-12-22T23:04:33.422301Z","shell.execute_reply":"2024-12-22T23:04:33.515658Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3.5 Feature Importances\n\nTo calculate and plot feature importances after training LightGBM model:","metadata":{}},{"cell_type":"code","source":"feature_names = numerical_cols_excluding_last.tolist() + \\\n                list(encoder.get_feature_names_out(categorical_cols))\n\nX_train_preprocessed_df = pd.DataFrame(X_train_preprocessed, \n                                       columns=feature_names)\n# X_train_preprocessed_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:33.517833Z","iopub.execute_input":"2024-12-22T23:04:33.518223Z","iopub.status.idle":"2024-12-22T23:04:33.523047Z","shell.execute_reply.started":"2024-12-22T23:04:33.518186Z","shell.execute_reply":"2024-12-22T23:04:33.522073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Retrieve Feature Importances:\nfeature_importances = pd.DataFrame({\n    'Feature': X_train_preprocessed_df.columns,\n    'Importance': final_model.feature_importances_\n}).sort_values(by='Importance', ascending=False)\n\nprint(feature_importances)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:33.524028Z","iopub.execute_input":"2024-12-22T23:04:33.524425Z","iopub.status.idle":"2024-12-22T23:04:33.543235Z","shell.execute_reply.started":"2024-12-22T23:04:33.52439Z","shell.execute_reply":"2024-12-22T23:04:33.541895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 20))\n\nplt.barh(feature_importances['Feature'], \n         feature_importances['Importance'],)\n\nplt.gca().invert_yaxis()\nplt.title('Feature Importances')\nplt.xlabel('Importance')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:33.544541Z","iopub.execute_input":"2024-12-22T23:04:33.544949Z","iopub.status.idle":"2024-12-22T23:04:34.225102Z","shell.execute_reply.started":"2024-12-22T23:04:33.54491Z","shell.execute_reply":"2024-12-22T23:04:34.224026Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Make Final Predictions & Submit","metadata":{}},{"cell_type":"code","source":"# Transform the test set using the same scaler\ntest_numeric_scaled = scaler.transform(test[numerical_cols_excluding_last])\n\n# Transform the test set using the same encoder\ntest_categorical_encoded = encoder.transform(test[categorical_cols]).toarray()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:34.226361Z","iopub.execute_input":"2024-12-22T23:04:34.226757Z","iopub.status.idle":"2024-12-22T23:04:36.376372Z","shell.execute_reply.started":"2024-12-22T23:04:34.22671Z","shell.execute_reply":"2024-12-22T23:04:36.375407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine numeric and categorical features\ntest_preprocessed = np.hstack((test_numeric_scaled, test_categorical_encoded))\n\n# Combine numeric and categorical features (check shapes)\nif test_preprocessed.shape[0] != test.shape[0]:\n    raise ValueError(\"Inconsistent sample sizes after preprocessing!\")\n\ntest_preprocessed = np.hstack((test_numeric_scaled, test_categorical_encoded))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:36.37723Z","iopub.execute_input":"2024-12-22T23:04:36.377527Z","iopub.status.idle":"2024-12-22T23:04:36.671907Z","shell.execute_reply.started":"2024-12-22T23:04:36.377503Z","shell.execute_reply":"2024-12-22T23:04:36.671069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions on test data\ntest_final_predictions = final_model.predict(test_preprocessed)\n\n# Convert predictions back to the original scale\ntest_final_predictions = np.exp(test_final_predictions)\ntest_final_predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:36.672753Z","iopub.execute_input":"2024-12-22T23:04:36.673109Z","iopub.status.idle":"2024-12-22T23:04:41.280767Z","shell.execute_reply.started":"2024-12-22T23:04:36.673072Z","shell.execute_reply":"2024-12-22T23:04:41.279849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\nsubmission['Premium Amount'] = test_final_predictions\nsubmission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:41.281608Z","iopub.execute_input":"2024-12-22T23:04:41.281968Z","iopub.status.idle":"2024-12-22T23:04:41.559892Z","shell.execute_reply.started":"2024-12-22T23:04:41.281933Z","shell.execute_reply":"2024-12-22T23:04:41.558878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)\nprint(\"Submission File Created Successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:41.560757Z","iopub.execute_input":"2024-12-22T23:04:41.561019Z","iopub.status.idle":"2024-12-22T23:04:43.130645Z","shell.execute_reply.started":"2024-12-22T23:04:41.560998Z","shell.execute_reply":"2024-12-22T23:04:43.129677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\n### Save the Final Model Using joblib for future use\n\nimport joblib\n\njoblib.dump(final_model, 'final_model.pkl')\n\n# Load the Model Later:\nfinal_model = joblib.load('final_model.pkl')\n\n### Using LightGBM Native Save:\nfinal_model.booster_.save_model('final_model.txt')\n\n# Load Native Model:\nimport lightgbm as lgb\nfinal_model = lgb.Booster(model_file='final_model.txt')\n\"\"\";","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T23:04:43.131591Z","iopub.execute_input":"2024-12-22T23:04:43.131843Z","iopub.status.idle":"2024-12-22T23:04:43.136004Z","shell.execute_reply.started":"2024-12-22T23:04:43.131822Z","shell.execute_reply":"2024-12-22T23:04:43.135015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}