{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --upgrade scikit-learn==1.3.1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T14:52:37.409473Z","iopub.execute_input":"2025-03-02T14:52:37.410115Z","iopub.status.idle":"2025-03-02T14:52:48.675222Z","shell.execute_reply.started":"2025-03-02T14:52:37.410076Z","shell.execute_reply":"2025-03-02T14:52:48.673593Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(__import__('sklearn').__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T14:52:48.676888Z","iopub.execute_input":"2025-03-02T14:52:48.677318Z","iopub.status.idle":"2025-03-02T14:52:49.243015Z","shell.execute_reply.started":"2025-03-02T14:52:48.677281Z","shell.execute_reply":"2025-03-02T14:52:49.241681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.preprocessing import OrdinalEncoder, TargetEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import mean_squared_error\nfrom lightgbm import LGBMRegressor\nimport warnings\n\nwarnings.filterwarnings('ignore')\npd.set_option('display.max_columns', None)\npd.set_option('display.width', 1000)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-02T14:52:49.244125Z","iopub.execute_input":"2025-03-02T14:52:49.24455Z","iopub.status.idle":"2025-03-02T14:52:53.707144Z","shell.execute_reply.started":"2025-03-02T14:52:49.244521Z","shell.execute_reply":"2025-03-02T14:52:53.705953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T14:52:53.708268Z","iopub.execute_input":"2025-03-02T14:52:53.709134Z","iopub.status.idle":"2025-03-02T14:53:05.714604Z","shell.execute_reply.started":"2025-03-02T14:52:53.709083Z","shell.execute_reply":"2025-03-02T14:53:05.71318Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Exploration","metadata":{}},{"cell_type":"code","source":"df_train.isna().sum()[df_train.isna().sum() > 0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:20.137396Z","iopub.execute_input":"2025-03-01T23:40:20.137743Z","iopub.status.idle":"2025-03-01T23:40:21.526681Z","shell.execute_reply.started":"2025-03-01T23:40:20.137695Z","shell.execute_reply":"2025-03-01T23:40:21.525461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.isna().sum()[df_test.isna().sum() > 0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:21.527895Z","iopub.execute_input":"2025-03-01T23:40:21.528362Z","iopub.status.idle":"2025-03-01T23:40:22.456631Z","shell.execute_reply.started":"2025-03-01T23:40:21.528328Z","shell.execute_reply":"2025-03-01T23:40:22.455255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:22.458074Z","iopub.execute_input":"2025-03-01T23:40:22.458423Z","iopub.status.idle":"2025-03-01T23:40:23.179599Z","shell.execute_reply.started":"2025-03-01T23:40:22.458397Z","shell.execute_reply":"2025-03-01T23:40:23.177569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_histogram(df, column, bins=10, title=None, xlabel=None, ylabel='Frequency', figsize=(10, 6), color='skyblue', kde=False, print_mean_median=False):\n    \"\"\"\n    plots a histogram for a specified column in a DataFrame using Seaborn.\n\n    Args:\n        df (pd.DataFrame): the DataFrame containing the data.\n        column (str): the column to plot.\n        bins (int): number of bins for the histogram (default is 10).\n        title (str): title of the plot (default is None).\n        xlabel (str): label for the x-axis (default is None).\n        ylabel (str): label for the y-axis (default is 'Frequency').\n        figsize (tuple): size of the figure (default is (10, 6)).\n        color (str): color of the bars (default is 'skyblue').\n        kde (bool): whether to overlay a Kernel Density Estimate (default is False).\n        print_mean_median (bool): wheter to print mean and median (default is False)\n    \"\"\"\n\n    if print_mean_median:\n        print(f\"mean: {df[column].mean():.2f}\")\n        print(f\"median: {df[column].median():.2f}\")\n        \n    sns.set_style('whitegrid')\n    \n    plt.figure(figsize=figsize)\n    \n    ax = sns.histplot(df[column], bins=bins, color=color, kde=kde, edgecolor='black', alpha=0.8)\n    \n    if title:\n        ax.set_title(title, fontsize=16, fontweight='bold')\n    if xlabel:\n        ax.set_xlabel(xlabel, fontsize=14)\n    ax.set_ylabel(ylabel, fontsize=14)\n    \n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:23.184422Z","iopub.execute_input":"2025-03-01T23:40:23.185127Z","iopub.status.idle":"2025-03-01T23:40:23.195059Z","shell.execute_reply.started":"2025-03-01T23:40:23.185073Z","shell.execute_reply":"2025-03-01T23:40:23.193185Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Age","metadata":{}},{"cell_type":"code","source":"plot_histogram(df_train, 'Age', bins=20, title='Age Distribution', xlabel='Age', kde=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:23.197188Z","iopub.execute_input":"2025-03-01T23:40:23.1976Z","iopub.status.idle":"2025-03-01T23:40:28.772267Z","shell.execute_reply.started":"2025-03-01T23:40:23.197566Z","shell.execute_reply":"2025-03-01T23:40:28.770844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Annual Income', bins=20, title='Annual Income Distribution', xlabel='Annual Income', kde=True, print_mean_median=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:28.773589Z","iopub.execute_input":"2025-03-01T23:40:28.774028Z","iopub.status.idle":"2025-03-01T23:40:34.509709Z","shell.execute_reply.started":"2025-03-01T23:40:28.773987Z","shell.execute_reply":"2025-03-01T23:40:34.508403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Number of Dependents', bins=20, title='Number of Dependents Distribution', xlabel='Number of Dependents', kde=True, print_mean_median=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:34.510974Z","iopub.execute_input":"2025-03-01T23:40:34.511626Z","iopub.status.idle":"2025-03-01T23:40:39.398188Z","shell.execute_reply.started":"2025-03-01T23:40:34.51159Z","shell.execute_reply":"2025-03-01T23:40:39.396854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Health Score', bins=20, title='Health Score Distribution', xlabel='Health Score', kde=True, print_mean_median=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:39.399477Z","iopub.execute_input":"2025-03-01T23:40:39.399936Z","iopub.status.idle":"2025-03-01T23:40:44.856863Z","shell.execute_reply.started":"2025-03-01T23:40:39.399887Z","shell.execute_reply":"2025-03-01T23:40:44.855479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Previous Claims', bins=20, title='Previous Claims Distribution', xlabel='Previous Claims', kde=True, print_mean_median=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:44.858003Z","iopub.execute_input":"2025-03-01T23:40:44.858332Z","iopub.status.idle":"2025-03-01T23:40:49.273888Z","shell.execute_reply.started":"2025-03-01T23:40:44.858303Z","shell.execute_reply":"2025-03-01T23:40:49.272369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Gender', bins=16, title='Gender Distribution', xlabel='Gender')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:49.275127Z","iopub.execute_input":"2025-03-01T23:40:49.275545Z","iopub.status.idle":"2025-03-01T23:40:51.082865Z","shell.execute_reply.started":"2025-03-01T23:40:49.275506Z","shell.execute_reply":"2025-03-01T23:40:51.081576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Marital Status', bins=20, title='Marital Status Distribution', xlabel='Marital Status')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:51.084108Z","iopub.execute_input":"2025-03-01T23:40:51.084448Z","iopub.status.idle":"2025-03-01T23:40:53.007627Z","shell.execute_reply.started":"2025-03-01T23:40:51.084418Z","shell.execute_reply":"2025-03-01T23:40:53.006438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Education Level', bins=20, title='Education Level Distribution', xlabel='Education Level')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:53.008777Z","iopub.execute_input":"2025-03-01T23:40:53.009117Z","iopub.status.idle":"2025-03-01T23:40:54.855171Z","shell.execute_reply.started":"2025-03-01T23:40:53.009087Z","shell.execute_reply":"2025-03-01T23:40:54.853862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Occupation', bins=20, title='Occupation Distribution', xlabel='Occupation')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:54.856163Z","iopub.execute_input":"2025-03-01T23:40:54.85653Z","iopub.status.idle":"2025-03-01T23:40:56.647634Z","shell.execute_reply.started":"2025-03-01T23:40:54.856499Z","shell.execute_reply":"2025-03-01T23:40:56.646261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Location', bins=20, title='Location Distribution', xlabel='Location')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:56.649045Z","iopub.execute_input":"2025-03-01T23:40:56.649394Z","iopub.status.idle":"2025-03-01T23:40:58.484874Z","shell.execute_reply.started":"2025-03-01T23:40:56.649365Z","shell.execute_reply":"2025-03-01T23:40:58.483849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Policy Type', bins=20, title='Policy Type Distribution', xlabel='Policy Type')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:40:58.485804Z","iopub.execute_input":"2025-03-01T23:40:58.486141Z","iopub.status.idle":"2025-03-01T23:41:00.30428Z","shell.execute_reply.started":"2025-03-01T23:40:58.486113Z","shell.execute_reply":"2025-03-01T23:41:00.303025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Customer Feedback', bins=20, title='Customer Feedback Distribution', xlabel='Customer Feedback')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:41:00.305781Z","iopub.execute_input":"2025-03-01T23:41:00.306405Z","iopub.status.idle":"2025-03-01T23:41:02.826249Z","shell.execute_reply.started":"2025-03-01T23:41:00.306355Z","shell.execute_reply":"2025-03-01T23:41:02.824743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Smoking Status', bins=20, title='Smoking Status Distribution', xlabel='Smoking Status')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:41:02.827949Z","iopub.execute_input":"2025-03-01T23:41:02.828417Z","iopub.status.idle":"2025-03-01T23:41:04.655176Z","shell.execute_reply.started":"2025-03-01T23:41:02.828372Z","shell.execute_reply":"2025-03-01T23:41:04.653871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Exercise Frequency', bins=20, title='Exercise Frequency Distribution', xlabel='Exercise Frequency')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:41:04.656449Z","iopub.execute_input":"2025-03-01T23:41:04.656915Z","iopub.status.idle":"2025-03-01T23:41:06.482554Z","shell.execute_reply.started":"2025-03-01T23:41:04.656872Z","shell.execute_reply":"2025-03-01T23:41:06.481375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Property Type', bins=20, title='Property Type Distribution', xlabel='Property Type')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:41:06.486808Z","iopub.execute_input":"2025-03-01T23:41:06.487166Z","iopub.status.idle":"2025-03-01T23:41:08.333625Z","shell.execute_reply.started":"2025-03-01T23:41:06.487138Z","shell.execute_reply":"2025-03-01T23:41:08.332406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Vehicle Age', bins=20, title='Vehicle Age Distribution', xlabel='Vehicle Age', kde=True, print_mean_median=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:41:08.335636Z","iopub.execute_input":"2025-03-01T23:41:08.336279Z","iopub.status.idle":"2025-03-01T23:41:13.900927Z","shell.execute_reply.started":"2025-03-01T23:41:08.336229Z","shell.execute_reply":"2025-03-01T23:41:13.899469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Credit Score', bins=20, title='Credit Score Distribution', xlabel='Credit Score', kde=True, print_mean_median=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:41:13.902257Z","iopub.execute_input":"2025-03-01T23:41:13.902653Z","iopub.status.idle":"2025-03-01T23:41:18.946356Z","shell.execute_reply.started":"2025-03-01T23:41:13.902618Z","shell.execute_reply":"2025-03-01T23:41:18.944409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_histogram(df_train, 'Insurance Duration', bins=20, title='Insurance Duration Distribution', xlabel='Insurance Duration Score', kde=True, print_mean_median=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:41:18.9474Z","iopub.execute_input":"2025-03-01T23:41:18.947819Z","iopub.status.idle":"2025-03-01T23:41:24.501853Z","shell.execute_reply.started":"2025-03-01T23:41:18.94778Z","shell.execute_reply":"2025-03-01T23:41:24.500386Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocesses Data","metadata":{}},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T23:41:24.503244Z","iopub.execute_input":"2025-03-01T23:41:24.503754Z","iopub.status.idle":"2025-03-01T23:41:24.531601Z","shell.execute_reply.started":"2025-03-01T23:41:24.503687Z","shell.execute_reply":"2025-03-01T23:41:24.530422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from time import perf_counter\nfrom functools import wraps\n\ndef timeit(func):\n    @wraps(func)\n    def wrapper(*args, **kwargs):\n        start = perf_counter()\n        result = func(*args, **kwargs)\n        print(f\"{func.__name__} took {perf_counter() - start:.2f} seconds\")\n        return result\n    return wrapper","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T14:53:05.71586Z","iopub.execute_input":"2025-03-02T14:53:05.716191Z","iopub.status.idle":"2025-03-02T14:53:05.722795Z","shell.execute_reply.started":"2025-03-02T14:53:05.716164Z","shell.execute_reply":"2025-03-02T14:53:05.721328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@timeit\ndef preprocess_data(\n    df: pd.DataFrame,\n    const_imputer: SimpleImputer,\n    num_imputer: SimpleImputer,\n    ordinal_encoder: OrdinalEncoder,\n    target_encoder: TargetEncoder,\n    is_test: bool = False\n):\n    \"\"\"\n    preprocess data\n\n    1. impute missing data\n    2. convert gender and smoking status to binary\n    3. feature enginner\n    4. enconding ordinal columns\n    5. encoding category columns\n    \"\"\"\n    if is_test: \n        X = df.drop(['id'], axis=1)\n    else:\n        X, y = df.drop(['id', 'Premium Amount'], axis=1), df['Premium Amount']\n    \n    # 1. impute missing data\n    num_cols = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\n    if is_test:\n        X[num_cols] = num_imputer.transform(X[num_cols])\n    else:\n        X[num_cols] = num_imputer.fit_transform(X[num_cols])\n\n    const_cols = ['Occupation', 'Customer Feedback', 'Marital Status']\n    if is_test:\n        X[const_cols] = const_imputer.transform(X[const_cols])\n    else:\n        X[const_cols] = const_imputer.fit_transform(X[const_cols])\n\n    # 2. convert gender and smoking status to binary\n    X['Gender'] = X['Gender'].map({'Male': 0, 'Female': 1})\n    X['Smoking Status'] = X['Smoking Status'].map({'No': 0, 'Yes': 1})\n\n    # 3. feature enginner\n    X['Policy Start Date'] = pd.to_datetime(X['Policy Start Date'], errors='coerce')\n    X['Year'] = X['Policy Start Date'].dt.year\n    X['Day'] = X['Policy Start Date'].dt.day\n    X['Month'] = X['Policy Start Date'].dt.month\n    X['Month_name'] = X['Policy Start Date'].dt.month_name()\n    X['Day_of_week'] = X['Policy Start Date'].dt.day_name()\n    X['Week'] = X['Policy Start Date'].dt.isocalendar().week\n    X['Year_sin'] = np.sin(2 * np.pi * X['Year'])\n    X['Year_cos'] = np.cos(2 * np.pi * X['Year'])\n    X['Month_sin'] = np.sin(2 * np.pi * X['Month'] / 12)\n    X['Month_cos'] = np.cos(2 * np.pi * X['Month'] / 12)\n    X['Day_sin'] = np.sin(2 * np.pi * X['Day'] / 31)\n    X['Day_cos'] = np.cos(2 * np.pi * X['Day'] / 31)\n    \n    X = X.drop(['Policy Start Date', 'Year', 'Day', 'Month'], axis=1)\n    \n    # 4. encoding ordinal columns\n    if is_test:\n        X[['Education Level', 'Policy Type', 'Customer Feedback', 'Exercise Frequency']] = ordinal_encoder.transform(\n            X[['Education Level', 'Policy Type', 'Customer Feedback', 'Exercise Frequency']]\n        )\n    else:\n        X[['Education Level', 'Policy Type', 'Customer Feedback', 'Exercise Frequency']] = ordinal_encoder.fit_transform(\n            X[['Education Level', 'Policy Type', 'Customer Feedback', 'Exercise Frequency']]\n        )\n\n    # 5. encoding category columns\n    category_cols = X.select_dtypes(include=['object']).columns\n    if is_test:\n        X[category_cols] = target_encoder.transform(X[category_cols])\n    else:\n        X[category_cols] = target_encoder.fit_transform(X[category_cols], y)\n\n    if is_test:\n        return X\n        \n    return X, y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T14:53:14.006299Z","iopub.execute_input":"2025-03-02T14:53:14.006627Z","iopub.status.idle":"2025-03-02T14:53:14.0205Z","shell.execute_reply.started":"2025-03-02T14:53:14.006602Z","shell.execute_reply":"2025-03-02T14:53:14.019118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"const_imputer = SimpleImputer(strategy='constant', fill_value='unknown')\nnum_imputer = SimpleImputer(strategy='median')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T14:53:16.733392Z","iopub.execute_input":"2025-03-02T14:53:16.733831Z","iopub.status.idle":"2025-03-02T14:53:16.738826Z","shell.execute_reply.started":"2025-03-02T14:53:16.733797Z","shell.execute_reply":"2025-03-02T14:53:16.737476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ordinal_map = {\n    'Education Level': [\"High School\", \"Bachelor's\", \"Master's\", \"PhD\"],\n    'Policy Type': ['Basic', 'Comprehensive', 'Premium'],\n    'Customer Feedback': ['unknown', 'Poor', 'Average', 'Good'],\n    'Exercise Frequency': ['Rarely', 'Monthly', 'Weekly', 'Daily'],\n}\n\nordinal_encoder = OrdinalEncoder(categories=[ordinal_map[col] for col in ['Education Level', 'Policy Type', 'Customer Feedback', 'Exercise Frequency']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T14:53:18.594851Z","iopub.execute_input":"2025-03-02T14:53:18.595218Z","iopub.status.idle":"2025-03-02T14:53:18.600806Z","shell.execute_reply.started":"2025-03-02T14:53:18.595189Z","shell.execute_reply":"2025-03-02T14:53:18.599719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_encoder = TargetEncoder(smooth=\"auto\", target_type='continuous', cv=5, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T14:53:20.953869Z","iopub.execute_input":"2025-03-02T14:53:20.954217Z","iopub.status.idle":"2025-03-02T14:53:20.958912Z","shell.execute_reply.started":"2025-03-02T14:53:20.954192Z","shell.execute_reply":"2025-03-02T14:53:20.957804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X, y = preprocess_data(df_train, const_imputer, num_imputer, ordinal_encoder, target_encoder)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T14:53:23.44518Z","iopub.execute_input":"2025-03-02T14:53:23.445696Z","iopub.status.idle":"2025-03-02T14:53:32.330177Z","shell.execute_reply.started":"2025-03-02T14:53:23.445618Z","shell.execute_reply":"2025-03-02T14:53:32.329035Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training, testing and evaluating Model","metadata":{}},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\nprint(f\"shape train: {X_train.shape}, val: {X_val.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T14:54:59.614561Z","iopub.execute_input":"2025-03-02T14:54:59.615022Z","iopub.status.idle":"2025-03-02T14:55:00.365295Z","shell.execute_reply.started":"2025-03-02T14:54:59.614989Z","shell.execute_reply":"2025-03-02T14:55:00.364203Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Lightgbm ","metadata":{}},{"cell_type":"code","source":"lgbm = LGBMRegressor(random_state=42)\nlgbm.fit(X_train, y_train)\n\ny_pred = lgbm.predict(X_val)\n\nmse = mean_squared_error(y_val, y_pred)\nrmse = np.sqrt(mse)\nprint(f\"MSE: {mse:.2f}, RMSE: {rmse:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T14:55:02.289808Z","iopub.execute_input":"2025-03-02T14:55:02.290184Z","iopub.status.idle":"2025-03-02T14:55:09.80461Z","shell.execute_reply.started":"2025-03-02T14:55:02.290157Z","shell.execute_reply":"2025-03-02T14:55:09.803227Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Hypertuning","metadata":{}},{"cell_type":"code","source":"# lgbm = LGBMRegressor(random_state=42)\n\n# param_grid = {\n#     'num_leaves': [31, 50, 70],\n#     'max_depth': [10, 20, 30],\n#     'learning_rate': [0.05, 0.1, 0.15],\n#     'n_estimators': [50, 100, 200]\n# }\n\n# grid_search = GridSearchCV(\n#     estimator=lgbm, \n#     param_grid=param_grid, \n#     cv=5, \n#     scoring='neg_mean_squared_error',\n#     n_jobs=-1\n# )\n\n# grid_search.fit(X_train, y_train)\n# hypertuned_model = grid_search.best_estimator_\n# best_params = grid_search.best_params_\n\n# y_pred = hypertuned_model.predict(X_val)\n\n# mse = mean_squared_error(y_val, y_pred)\n# rmse = np.sqrt(mse)\n\n# print(f\"MSE: {mse:.2f}, RMSE: {rmse:.2f}\")\n# print(\"best paramaters:\", best_params)\n# print(\"=\" * 40)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T14:55:47.243078Z","iopub.execute_input":"2025-03-02T14:55:47.243437Z","iopub.status.idle":"2025-03-02T15:47:11.393151Z","shell.execute_reply.started":"2025-03-02T14:55:47.243406Z","shell.execute_reply":"2025-03-02T15:47:11.39184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hypertuned_xgb = LGBMRegressor(\n    learning_rate=0.05,\n    max_depth=20,\n    n_estimators=200,\n    num_leaves=70,\n    random_state=42\n)\n\nhypertuned_xgb.fit(X_train, y_train)\n\nX = preprocess_data(df_test, const_imputer, num_imputer, ordinal_encoder, target_encoder, is_test=True)\n\ny_pred = hypertuned_xgb.predict(X)\n\ndf_test['Premium Amount'] = y_pred\ndf_test[['id', 'Premium Amount']].to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-02T16:26:59.731597Z","iopub.execute_input":"2025-03-02T16:26:59.733473Z","iopub.status.idle":"2025-03-02T16:27:23.943957Z","shell.execute_reply.started":"2025-03-02T16:26:59.733389Z","shell.execute_reply":"2025-03-02T16:27:23.942484Z"}},"outputs":[],"execution_count":null}]}