{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:17:13.554715Z","iopub.execute_input":"2024-12-18T12:17:13.555191Z","iopub.status.idle":"2024-12-18T12:17:13.967781Z","shell.execute_reply.started":"2024-12-18T12:17:13.555156Z","shell.execute_reply":"2024-12-18T12:17:13.966555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q scikit-learn","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sklearn\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:17:19.003813Z","iopub.execute_input":"2024-12-18T12:17:19.004381Z","iopub.status.idle":"2024-12-18T12:17:19.655152Z","shell.execute_reply.started":"2024-12-18T12:17:19.004342Z","shell.execute_reply":"2024-12-18T12:17:19.654052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# importing necessary libraries\n\n# 1. Data Visualization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport matplotlib.gridspec as gridspec\n\n# Data Manupulation\nimport numpy as np\nimport pandas as pd\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:17:27.315767Z","iopub.execute_input":"2024-12-18T12:17:27.316672Z","iopub.status.idle":"2024-12-18T12:17:27.479396Z","shell.execute_reply.started":"2024-12-18T12:17:27.316625Z","shell.execute_reply":"2024-12-18T12:17:27.478353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest_df =  pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:17:38.157802Z","iopub.execute_input":"2024-12-18T12:17:38.158928Z","iopub.status.idle":"2024-12-18T12:17:45.456719Z","shell.execute_reply.started":"2024-12-18T12:17:38.158883Z","shell.execute_reply":"2024-12-18T12:17:45.45535Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Exploring Data**","metadata":{}},{"cell_type":"code","source":"train_df.columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:17:55.129012Z","iopub.execute_input":"2024-12-18T12:17:55.129394Z","iopub.status.idle":"2024-12-18T12:17:55.13745Z","shell.execute_reply.started":"2024-12-18T12:17:55.129362Z","shell.execute_reply":"2024-12-18T12:17:55.136224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# checking number of rows and columns\nprint(f\"Rows : {train_df.shape[0]} | Columns : {train_df.shape[1]}\")\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:17:58.752693Z","iopub.execute_input":"2024-12-18T12:17:58.75365Z","iopub.status.idle":"2024-12-18T12:17:58.783415Z","shell.execute_reply.started":"2024-12-18T12:17:58.753607Z","shell.execute_reply":"2024-12-18T12:17:58.782404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()\n# by this step we get to know that in data set there are\n# 1. Numerical and categorical columns\n# 2. There are null values in some column ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:18:02.832611Z","iopub.execute_input":"2024-12-18T12:18:02.833061Z","iopub.status.idle":"2024-12-18T12:18:03.491167Z","shell.execute_reply.started":"2024-12-18T12:18:02.833023Z","shell.execute_reply":"2024-12-18T12:18:03.489909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:18:07.158627Z","iopub.execute_input":"2024-12-18T12:18:07.159612Z","iopub.status.idle":"2024-12-18T12:18:07.806213Z","shell.execute_reply.started":"2024-12-18T12:18:07.159567Z","shell.execute_reply":"2024-12-18T12:18:07.805056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:18:11.658891Z","iopub.execute_input":"2024-12-18T12:18:11.659256Z","iopub.status.idle":"2024-12-18T12:18:13.334975Z","shell.execute_reply.started":"2024-12-18T12:18:11.659225Z","shell.execute_reply":"2024-12-18T12:18:13.333686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# partion of numerical and categorical column\n# Save 'ID' column for submission\ntest_id = test_df['id']\n\n# Define the target column\ntarget_column = 'Premium Amount'\n\n# Select categorical and numerical columns(initial)\ncategorical_col =  train_df.select_dtypes(include=['object']).columns\nnumerical_col = train_df.select_dtypes(exclude= ['object']).columns\n\nprint(f\"Categorical Columns : {categorical_col.tolist()}\")\nprint(f\"\\nNumerical Columns : {numerical_col.tolist()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:18:16.916647Z","iopub.execute_input":"2024-12-18T12:18:16.917054Z","iopub.status.idle":"2024-12-18T12:18:17.110555Z","shell.execute_reply.started":"2024-12-18T12:18:16.917019Z","shell.execute_reply":"2024-12-18T12:18:17.109238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.describe().round(2)\n# Analysis\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:18:23.433293Z","iopub.execute_input":"2024-12-18T12:18:23.434125Z","iopub.status.idle":"2024-12-18T12:18:24.152362Z","shell.execute_reply.started":"2024-12-18T12:18:23.434081Z","shell.execute_reply":"2024-12-18T12:18:24.151206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for c in categorical_col:\n    num_unique = train_df[c].nunique()\n    print(f\"'{c}': {num_unique} \")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:18:27.131782Z","iopub.execute_input":"2024-12-18T12:18:27.132168Z","iopub.status.idle":"2024-12-18T12:18:28.08893Z","shell.execute_reply.started":"2024-12-18T12:18:27.132135Z","shell.execute_reply":"2024-12-18T12:18:28.087715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print top 10 unique value counts for each categorical column\nfor c in categorical_col:\n    print(f\"\\nvalue counts in '{c}':\\n{train_df[c].value_counts().head(10)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:18:32.049759Z","iopub.execute_input":"2024-12-18T12:18:32.050436Z","iopub.status.idle":"2024-12-18T12:18:33.506378Z","shell.execute_reply.started":"2024-12-18T12:18:32.050395Z","shell.execute_reply":"2024-12-18T12:18:33.505358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"The mean of columns:\")\nprint(train_df[numerical_col].mean())\n\nprint(\"\\nThe std dev of columns:\")\nprint(train_df[numerical_col].std())\n\nprint(\"\\nThe skewness of columns:\")\nprint(train_df[numerical_col].skew())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:18:39.848896Z","iopub.execute_input":"2024-12-18T12:18:39.849259Z","iopub.status.idle":"2024-12-18T12:18:40.359184Z","shell.execute_reply.started":"2024-12-18T12:18:39.849227Z","shell.execute_reply":"2024-12-18T12:18:40.357798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (15,12))\nplt.title(\"Missing Values\")\nsns.heatmap(train_df.isnull(),cbar=False, cmap=sns.color_palette('magma'), yticklabels=False)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:18:43.963718Z","iopub.execute_input":"2024-12-18T12:18:43.964795Z","iopub.status.idle":"2024-12-18T12:19:07.895056Z","shell.execute_reply.started":"2024-12-18T12:18:43.964717Z","shell.execute_reply":"2024-12-18T12:19:07.893799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorrelation_matrix = train_df[numerical_col].corr()\n\n# Plot the heatmap\nplt.figure(figsize=(12, 8))\nsns.heatmap(correlation_matrix, annot=True, fmt=\".2f\", cmap=\"coolwarm\", cbar=True, linewidths=0.5)\nplt.title(\"Correlation of Numerical Variables\", fontsize=16)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T12:49:28.814428Z","iopub.execute_input":"2024-12-18T12:49:28.815016Z","iopub.status.idle":"2024-12-18T12:49:29.957454Z","shell.execute_reply.started":"2024-12-18T12:49:28.814963Z","shell.execute_reply":"2024-12-18T12:49:29.956202Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing Data","metadata":{}},{"cell_type":"code","source":"def date(df):\n    # converting to date time formate\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    \n    # to identify seasonal trends\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Month'] = df['Policy Start Date'].dt.month\n    df['Month_name'] = df['Policy Start Date'].dt.month_name()\n    df['Day_of_week'] = df['Policy Start Date'].dt.day_name()\n    df['Week'] = df['Policy Start Date'].dt.isocalendar().week\n\n    # sin and cos encode the circular nature of months , days and year.\n    df['Year_sin'] = np.sin(2 * np.pi * df['Year'])\n    df['Year_cos'] = np.cos(2 * np.pi * df['Year'])\n    min_year = df['Year'].min()\n    max_year = df['Year'].max()\n    df['Year_sin'] = np.sin(2 * np.pi * (df['Year'] - min_year) / (max_year - min_year))\n    df['Year_cos'] = np.cos(2 * np.pi * (df['Year'] - min_year) / (max_year - min_year))\n    df['Month_sin'] = np.sin(2 * np.pi * df['Month'] / 12) \n    df['Month_cos'] = np.cos(2 * np.pi * df['Month'] / 12)\n    df['Day_sin'] = np.sin(2 * np.pi * df['Day'] / 31)  \n    df['Day_cos'] = np.cos(2 * np.pi * df['Day'] / 31)\n    # Combines year, month, and day into a single continuous numeric feature that represents time progression ,Helps in identifying patterns over time (e.g., seasonality, trends).\n    df['Group']=(df['Year']-2020)*48+df['Month']*4+df['Day']//7\n    \n    df.drop('Policy Start Date', axis=1, inplace=True)\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:08:03.557958Z","iopub.execute_input":"2024-12-18T13:08:03.559909Z","iopub.status.idle":"2024-12-18T13:08:03.576381Z","shell.execute_reply.started":"2024-12-18T13:08:03.559833Z","shell.execute_reply":"2024-12-18T13:08:03.574967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply the date function to both datasets\ntrain_df = date(train_df)\ntest_df = date(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:08:19.515099Z","iopub.execute_input":"2024-12-18T13:08:19.515544Z","iopub.status.idle":"2024-12-18T13:08:23.718152Z","shell.execute_reply.started":"2024-12-18T13:08:19.515491Z","shell.execute_reply":"2024-12-18T13:08:23.716853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split train data into features and target\nX = train_df.drop(columns=[target_column, 'id', 'Group', 'Year', 'Month', 'Day', 'Week'])\ny = train_df[target_column]\nprint(X.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:30:51.717825Z","iopub.execute_input":"2024-12-18T13:30:51.71877Z","iopub.status.idle":"2024-12-18T13:30:52.195477Z","shell.execute_reply.started":"2024-12-18T13:30:51.718695Z","shell.execute_reply":"2024-12-18T13:30:52.194342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:26:52.321218Z","iopub.execute_input":"2024-12-18T13:26:52.321987Z","iopub.status.idle":"2024-12-18T13:26:52.597296Z","shell.execute_reply.started":"2024-12-18T13:26:52.321945Z","shell.execute_reply":"2024-12-18T13:26:52.596171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Standardizing data\n# numerical data : [age, salary etc]\n# cateegorical data : [geder ,policy type etc]\n\n# replacing numerical value with the median of column\nnum_pipeline = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    #('scaler', StandardScaler())                       # Scale numerical features\n])\n\n# replacing categorical value with unknown\ncat_pipeline = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='constant', fill_value='Unknown')),  # Handle missing values\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))                      # Encode categorical features\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:26:54.568807Z","iopub.execute_input":"2024-12-18T13:26:54.569233Z","iopub.status.idle":"2024-12-18T13:26:54.576149Z","shell.execute_reply.started":"2024-12-18T13:26:54.569194Z","shell.execute_reply":"2024-12-18T13:26:54.57454Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Categorical features\ncategorical_features = [\n    'Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location', \n    'Policy Type', 'Exercise Frequency', 'Property Type', 'Month_name', 'Day_of_week'\n]\n\n# Numerical features\nnumerical_features = [\n    'Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n    'Year_sin', 'Year_cos', 'Month_sin', 'Month_cos', 'Day_sin', 'Day_cos'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:32:18.998668Z","iopub.execute_input":"2024-12-18T13:32:19.000681Z","iopub.status.idle":"2024-12-18T13:32:19.007289Z","shell.execute_reply.started":"2024-12-18T13:32:19.000531Z","shell.execute_reply":"2024-12-18T13:32:19.005833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preprocessor = ColumnTransformer(\n    transformers=[\n        ('num', num_pipeline, numerical_features),\n        ('cat', cat_pipeline, categorical_features)\n    ]\n)\n\n# Preprocess train and test data\nX_processed = preprocessor.fit_transform(X)\ntest_processed = preprocessor.transform(test_df.drop(columns=['id', 'Group', 'Year', 'Month', 'Day', 'Week']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:34:22.26371Z","iopub.execute_input":"2024-12-18T13:34:22.26448Z","iopub.status.idle":"2024-12-18T13:34:33.885029Z","shell.execute_reply.started":"2024-12-18T13:34:22.264428Z","shell.execute_reply":"2024-12-18T13:34:33.883936Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training Model","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:37:38.701271Z","iopub.execute_input":"2024-12-18T13:37:38.702539Z","iopub.status.idle":"2024-12-18T13:37:38.708112Z","shell.execute_reply.started":"2024-12-18T13:37:38.70248Z","shell.execute_reply":"2024-12-18T13:37:38.706823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the data\nX_train, X_val, y_train, y_val = train_test_split(X_processed, y, test_size=0.2, random_state=50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:38:07.909251Z","iopub.execute_input":"2024-12-18T13:38:07.909716Z","iopub.status.idle":"2024-12-18T13:38:08.268541Z","shell.execute_reply.started":"2024-12-18T13:38:07.909679Z","shell.execute_reply":"2024-12-18T13:38:08.267552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\nimport lightgbm as lgb\nfrom sklearn.metrics import mean_squared_log_error ,mean_squared_error, mean_absolute_error, r2_score\nimport torch\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=UserWarning, module=\"seaborn\")\nwarnings.filterwarnings(\"ignore\", category=FutureWarning, module=\"seaborn\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:50:26.903443Z","iopub.execute_input":"2024-12-18T13:50:26.904465Z","iopub.status.idle":"2024-12-18T13:50:26.913372Z","shell.execute_reply.started":"2024-12-18T13:50:26.904419Z","shell.execute_reply":"2024-12-18T13:50:26.912321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define Optuna\ndef objective(trial):\n    param = {\n        \"objective\": \"regression\",\n        \"metric\": \"rmse\",\n        \"boosting_type\": trial.suggest_categorical(\"boosting_type\", [\"gbdt\", \"dart\"]),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 200, 512),\n        \"learning_rate\": trial.suggest_loguniform(\"learning_rate\", 1e-4, 1e-1),\n        \"feature_fraction\": trial.suggest_uniform(\"feature_fraction\", 0.6, 1.0),\n        \"bagging_fraction\": trial.suggest_uniform(\"bagging_fraction\", 0.6, 1.0),\n        \"bagging_freq\": trial.suggest_int(\"bagging_freq\", 5, 12),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 20, 100),\n        \"max_depth\": trial.suggest_int(\"max_depth\", -1, 16),  # -1 means no limit\n        \"lambda_l1\": trial.suggest_loguniform(\"lambda_l1\", 1e-4, 10.0),\n        \"lambda_l2\": trial.suggest_loguniform(\"lambda_l2\", 1e-4, 10.0),\n        \"device_type\": \"cpu\",  # Enable GPU support\n        \"seed\" : 50\n\n    }\n    # Create a LightGBM dataset\n    dtrain = lgb.Dataset(X_train, label=y_train)\n    dval = lgb.Dataset(X_val, label=y_val, reference=dtrain)\n\n    # Train LightGBM model\n    model = lgb.train(\n        param,\n        dtrain,\n        valid_sets=[dval],\n    )\n\n    # Predict on validation set\n    y_val_pred = model.predict(X_val)\n    \n    # Compute RMSLE using sklearn's root_mean_squared_log_error\n    rmsle = mean_squared_log_error(y_val, np.maximum(y_val_pred, 0))\n    return rmsle\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:50:36.77804Z","iopub.execute_input":"2024-12-18T13:50:36.778476Z","iopub.status.idle":"2024-12-18T13:50:36.78807Z","shell.execute_reply.started":"2024-12-18T13:50:36.778438Z","shell.execute_reply":"2024-12-18T13:50:36.786755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run Optuna study\nstudy = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:50:39.302688Z","iopub.execute_input":"2024-12-18T13:50:39.303137Z","iopub.status.idle":"2024-12-18T13:51:03.608948Z","shell.execute_reply.started":"2024-12-18T13:50:39.303098Z","shell.execute_reply":"2024-12-18T13:51:03.607764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize or update the best_params dictionary\nbest_params = {\n    'boosting_type': 'dart',\n    'num_leaves': 384,\n    'learning_rate': 0.024680120465142227,\n    'feature_fraction': 0.9883068358315126,\n    'bagging_fraction': 0.7201712704805496,\n    'bagging_freq': 7,\n    'min_data_in_leaf': 50,\n    'max_depth': 15,\n    'lambda_l1': 0.0011290211269753322,\n    'lambda_l2': 3.056310541294088,\n    'seed': 42\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:51:40.633494Z","iopub.execute_input":"2024-12-18T13:51:40.633916Z","iopub.status.idle":"2024-12-18T13:51:40.639712Z","shell.execute_reply.started":"2024-12-18T13:51:40.63388Z","shell.execute_reply":"2024-12-18T13:51:40.638676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train final model with best parameters\n\nfinal_model = lgb.train(\n    best_params,\n    lgb.Dataset(X_processed, label=y),\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:52:01.122921Z","iopub.execute_input":"2024-12-18T13:52:01.123427Z","iopub.status.idle":"2024-12-18T13:52:53.350105Z","shell.execute_reply.started":"2024-12-18T13:52:01.123385Z","shell.execute_reply":"2024-12-18T13:52:53.348916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Performance Metrics\ny_pred = final_model.predict(X_processed)\n\n# Calcul des métriques\nrmsle = mean_squared_log_error(y, y_pred)\nrmse = np.sqrt(mean_squared_error(y, y_pred))\nmae = mean_absolute_error(y, y_pred)\nr2 = r2_score(y, y_pred)\nmape = np.mean(np.abs((y - y_pred) / y)) * 100\n\n# Display performance metrics\nprint(f\"\\nPerformance Metrics:\\n{'-'*20}\")\nprint(f\"RMSLE: {rmsle:.4f}\")\nprint(f\"MSE: {rmse:.4f}\")\nprint(f\"MAE: {mae:.4f}\")\nprint(f\"R²: {r2:.4f}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:56:15.688183Z","iopub.execute_input":"2024-12-18T13:56:15.688692Z","iopub.status.idle":"2024-12-18T13:56:20.669556Z","shell.execute_reply.started":"2024-12-18T13:56:15.688651Z","shell.execute_reply":"2024-12-18T13:56:20.667957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions on the test set\ntest_predictions = final_model.predict(test_processed, num_iteration=final_model.best_iteration)\n\n# Prepare submission file\nsubmission = pd.DataFrame({'id': test_df['id'], 'Premium Amount': test_predictions})\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-18T13:57:26.008656Z","iopub.execute_input":"2024-12-18T13:57:26.009106Z","iopub.status.idle":"2024-12-18T13:57:30.368713Z","shell.execute_reply.started":"2024-12-18T13:57:26.009069Z","shell.execute_reply":"2024-12-18T13:57:30.367651Z"}},"outputs":[],"execution_count":null}]}