{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Import Libraries**\nStart by importing necessary librarie","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport plotly.express as px\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import r2_score, mean_squared_log_error\nimport matplotlib.pyplot as plt\n","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:54:45.082052Z","iopub.execute_input":"2025-01-06T01:54:45.082342Z","iopub.status.idle":"2025-01-06T01:54:45.087313Z","shell.execute_reply.started":"2025-01-06T01:54:45.082322Z","shell.execute_reply":"2025-01-06T01:54:45.086276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport plotly.graph_objects as go\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import FunctionTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom sklearn.experimental import enable_iterative_imputer  # Enable the iterative imputer\nfrom sklearn.impute import IterativeImputer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:54:45.103801Z","iopub.execute_input":"2025-01-06T01:54:45.104065Z","iopub.status.idle":"2025-01-06T01:54:45.109051Z","shell.execute_reply.started":"2025-01-06T01:54:45.104045Z","shell.execute_reply":"2025-01-06T01:54:45.108232Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Load the Dataset**\nLoad the dataset into a DataFrame.","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndata_test= pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:54:45.141186Z","iopub.execute_input":"2025-01-06T01:54:45.141826Z","iopub.status.idle":"2025-01-06T01:54:53.054064Z","shell.execute_reply.started":"2025-01-06T01:54:45.141806Z","shell.execute_reply":"2025-01-06T01:54:53.053214Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Explore the Data**\nCheck the structure, missing values, and general statistics.","metadata":{}},{"cell_type":"code","source":"from IPython.display import display, HTML\n\n# Function to display a title with customized font size and color\ndef print_custom_title(title):\n    style = f\"\"\"\n    <h3 style=\"font-size: 18px; color: black; margin-bottom: 20px;\">{title}</h3>\n    \"\"\"\n    display(HTML(style))\n\n# Shape of the Dataset\nprint_custom_title(\"Shape of the Dataset\")\nprint(f'The Training Dataset has {data.shape[0]} rows and {data.shape[1]} columns.')\nprint(f'The Training Dataset has {data_test.shape[0]} rows and {data_test.shape[1]} columns.')\n\ndisplay(HTML(\"<br><br>\"))\n\n# First 5 Rows\nprint_custom_title(\"First and last 5 Rows\")\ndisplay(data.head())\ndisplay(data.tail())\n\ndisplay(HTML(\"<br><br>\"))\n\n# Summary Statistics\nprint_custom_title(\"Summary Statistics\")\ndisplay(data.describe(include='all'))\n\n# Two blank lines\ndisplay(HTML(\"<br><br>\"))\n\n# Null Values Table\nprint_custom_title(\"Null Values Table\")\nnull_values = data.isnull().sum().to_frame(name='Null Count')\nnull_values['Null Percentage'] = (null_values['Null Count'] / len(data)) * 100\ndisplay(null_values)\n\ndisplay(HTML(\"<br><br>\"))\n\n# Duplicate Rows\nprint_custom_title(\"Duplicate Rows\")\nduplicates = data.duplicated().sum()\nprint(f\"Number of duplicate rows: {duplicates}\")\n\ndisplay(HTML(\"<br><br>\"))\n\n# Data Types Table\nprint_custom_title(\"Data Types Table\")\ndata_types = data.dtypes.to_frame(name='Data Type')\ndisplay(data_types)\n\ndisplay(HTML(\"<br><br>\"))\n\n# Column Names\nprint_custom_title(\"Column Names\")\nprint(data.columns.tolist())\n\ndisplay(HTML(\"<br><br>\"))\n\n# Unique Values\nprint_custom_title(\"Unique Values\")\nunique_values = {col: data[col].nunique() for col in data.columns}\nunique_values_df = pd.DataFrame(list(unique_values.items()), columns=['Column', 'Unique Values'])\ndisplay(unique_values_df)\n\ndisplay(HTML(\"<br><br>\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:54:53.055676Z","iopub.execute_input":"2025-01-06T01:54:53.055936Z","iopub.status.idle":"2025-01-06T01:54:58.041166Z","shell.execute_reply.started":"2025-01-06T01:54:53.055916Z","shell.execute_reply":"2025-01-06T01:54:58.040317Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Observation Report\n\nThe dataset comprises 1,200,000 records with various demographic, financial, and behavioral attributes of individuals. Key observations are summarized below:\n\n#### **Demographic Insights**\n- **Age**: The median age is 41 years, spanning a working-age range of 18–64. \n- **Gender**: Males slightly outnumber females, with a frequency of 602,571.\n- **Marital Status**: Most individuals are Single (395,391), followed by Married and Divorced.\n- **Education Level**: Master's degree holders dominate (303,818).\n- **Location**: Suburban residents are the majority (401,542).\n\n\n#### **Behavioral and Financial Insights**\n- **Annual Income**: The median income is $23,911, with a wide variation, including outliers up to $149,997.\n- **Health Score**: The median health score is 24.58, with a positively skewed distribution up to 58.98.\n- **Credit Score**: Median credit score is 595, ranging from 300 to 849, showing varied financial stability.\n- **Exercise Frequency**: Weekly exercise is the most common pattern (306,179).\n\n\n#### **Insurance-Specific Insights**\n- **Policy Start Date**: Policies range from 2019 to 2024, indicating consistent activity over time.\n- **Insurance Duration**: Policies typically last a median of 5 years, with a range of 1–9 years.\n- **Premium Amount**: Median premium is $872, with significant variation between $20 and $4,999.\n- **Policy Type**: Premium policies are the most common (401,846).\n- **Previous Claims**: Most individuals have 0–1 claims, with a maximum of 9 observed.\n\n\n#### **Lifestyle Insights**\n- **Smoking Status**: A majority are smokers (601,873).\n- **Property Type**: Most participants own a house (400,349).\n- **Customer Feedback**: Average feedback dominates (377,905 responses).","metadata":{}},{"cell_type":"markdown","source":"## Explore Missing Values","metadata":{}},{"cell_type":"code","source":"def calculate_missing_percentage(data):\n    \"\"\"\n    Calculate the percentage of missing values for each column in the dataset.\n    \"\"\"\n    missing_percentage = (data.isnull().sum() / len(data)) * 100\n    return missing_percentage.round(2)\n\ndef plot_missing_percentage(missing_percentage, title=\"Percentage of Missing Values by Column\"):\n    \"\"\"\n    Plot the percentage of missing values for each column as a horizontal bar chart.\n    \"\"\"\n    missing_percentage.plot(kind='barh', figsize=(10, 6), color='skyblue', edgecolor=\"black\", linewidth=1.0)\n    plt.title(title)\n    plt.xlabel('Percentage')\n    plt.ylabel('Columns')\n    plt.show()\n\ndef plot_missing_heatmap(data, title=\"Heatmap of Missing Values\"):\n    \"\"\"\n    Create a heatmap visualization of missing values in the dataset.\n    \"\"\"\n    plt.figure(figsize=(12, 8))\n    sns.heatmap(data.isnull(), cbar=False, cmap='viridis')\n    plt.title(title)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:54:58.042419Z","iopub.execute_input":"2025-01-06T01:54:58.043085Z","iopub.status.idle":"2025-01-06T01:54:58.049753Z","shell.execute_reply.started":"2025-01-06T01:54:58.043044Z","shell.execute_reply":"2025-01-06T01:54:58.048873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Missing Percentage in Training Data\")\nprint(\"------------------------------------\")\ntrain_missing_percentage = calculate_missing_percentage(data)\nprint(train_missing_percentage)\n\nprint(\"\\nMissing Percentage in Testing Data\")\nprint(\"------------------------------------\")\ntest_missing_percentage = calculate_missing_percentage(data_test)\nprint(test_missing_percentage)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:54:58.052366Z","iopub.execute_input":"2025-01-06T01:54:58.052755Z","iopub.status.idle":"2025-01-06T01:54:58.997427Z","shell.execute_reply.started":"2025-01-06T01:54:58.052733Z","shell.execute_reply":"2025-01-06T01:54:58.996528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot for training data\nplot_missing_percentage(train_missing_percentage, title=\"Percentage of Missing Values in Training Data\")\nplot_missing_heatmap(data, title=\"Heatmap of Missing Values in Training Data\")\n\n# Plot for test data\nplot_missing_percentage(test_missing_percentage, title=\"Percentage of Missing Values in Test Data\")\nplot_missing_heatmap(data_test, title=\"Heatmap of Missing Values in Test Data\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:54:58.998522Z","iopub.execute_input":"2025-01-06T01:54:58.998845Z","iopub.status.idle":"2025-01-06T01:55:33.149718Z","shell.execute_reply.started":"2025-01-06T01:54:58.998823Z","shell.execute_reply":"2025-01-06T01:55:33.148861Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Analysis of Missing Data:\n\n#### **Training Data:**\n- **High Missing Values**:\n  - Occupation: 29.84%\n  - Previous Claims: 30.34%\n  - Credit Score: 11.49%\n  - Number of Dependents: 9.14%\n- **Moderate Missing Values**:\n  - Health Score: 6.17%\n  - Customer Feedback: 6.49%\n- **Low Missing Values**:\n  - Age: 1.56%\n  - Marital Status: 1.54%\n  - Annual Income: 3.75%\n\n#### **Testing Data:**\n- The pattern of missing values in the testing data closely mirrors that of the training data.\n- **Highest Missing Values**:\n  - Occupation: 29.89%\n  - Previous Claims: 30.35%\n- Other variables exhibit similar percentages of missing data as in the training dataset.","metadata":{}},{"cell_type":"markdown","source":"## Identify Numerical & Categorical Vars","metadata":{}},{"cell_type":"code","source":"def process_features(data, id_column, date_column, target_column=None):\n    \n    # Remove target and unique identifier from numerical features\n    numerical_features = data.select_dtypes(include=['int64', 'float64']).columns.tolist()\n    if target_column and target_column in numerical_features:\n        numerical_features.remove(target_column)\n    if id_column in numerical_features:\n        numerical_features.remove(id_column)\n\n    # Handle date feature separately\n    if date_column in data.columns:\n        data[date_column] = pd.to_datetime(data[date_column])\n        data['Policy Year'] = data[date_column].dt.year\n        data['Policy Month'] = data[date_column].dt.month\n        data['Policy Day'] = data[date_column].dt.day\n\n        # Add new date-derived features to numerical features\n        numerical_features += ['Policy Year', 'Policy Month', 'Policy Day']\n\n    # Define categorical features\n    categorical_features = data.select_dtypes(include=['object']).columns.tolist()\n\n    return numerical_features, categorical_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:55:33.150754Z","iopub.execute_input":"2025-01-06T01:55:33.151004Z","iopub.status.idle":"2025-01-06T01:55:33.157104Z","shell.execute_reply.started":"2025-01-06T01:55:33.150984Z","shell.execute_reply":"2025-01-06T01:55:33.156281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_features, categorical_features = process_features(\n    data=data, \n    target_column='Premium Amount', \n    id_column='id', \n    date_column='Policy Start Date'\n)\n\nprint(\"Numerical Features:\", numerical_features)\nprint(\"Categorical Features:\", categorical_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:55:33.158215Z","iopub.execute_input":"2025-01-06T01:55:33.158481Z","iopub.status.idle":"2025-01-06T01:55:34.130719Z","shell.execute_reply.started":"2025-01-06T01:55:33.158438Z","shell.execute_reply":"2025-01-06T01:55:34.129838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_features, categorical_features = process_features(\n    data=data_test, \n    id_column='id', \n    date_column='Policy Start Date'\n)\n\nprint(\"Numerical Features:\", numerical_features)\nprint(\"Categorical Features:\", categorical_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:55:34.131896Z","iopub.execute_input":"2025-01-06T01:55:34.132384Z","iopub.status.idle":"2025-01-06T01:55:34.763322Z","shell.execute_reply.started":"2025-01-06T01:55:34.132352Z","shell.execute_reply":"2025-01-06T01:55:34.762492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import plotly.graph_objects as go\n\ndef plot_interactive_pie_chart(data, column):\n    \n    counts = data[column].value_counts()\n    labels = counts.index\n    sizes = counts.values\n\n    # Create the pie chart\n    fig = go.Figure(\n        data=[\n            go.Pie(\n                labels=labels,\n                values=sizes,\n                hole=0.4,  # Creates a donut chart\n                textinfo=\"percent+label\",  # Show percentage and labels\n                marker=dict(\n                    line=dict(color=\"black\", width=2),  # Add a border\n                    colors=px.colors.qualitative.Set3[:len(labels)],  # Attractive colors\n                ),\n                pull=[0.1 if i == sizes.argmax() else 0 for i in range(len(sizes))],  # Highlight the largest slice\n            )\n        ]\n    )\n\n    # Update layout with animations and title\n    fig.update_layout(\n        title={\n            \"text\": f\"Distribution of {column}\",\n            \"y\": 0.9,\n            \"x\": 0.5,\n            \"xanchor\": \"center\",\n            \"yanchor\": \"top\",\n        },\n        template=\"presentation\",\n    )\n\n    # Show the chart\n    fig.show()\n\n# Example usage for all categorical features\nfor column in categorical_features:\n    plot_interactive_pie_chart(data, column)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:55:34.764672Z","iopub.execute_input":"2025-01-06T01:55:34.765024Z","iopub.status.idle":"2025-01-06T01:55:36.043652Z","shell.execute_reply.started":"2025-01-06T01:55:34.764994Z","shell.execute_reply":"2025-01-06T01:55:36.042791Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Feature Engineering**\n\nDrop \"id\" column because it is  irrelevant or redundant columns.\nmake a new feature from 'Policy Start Date' to Year,month and Day variable.\nScale numeric columns (like Annual Income, Credit Score).","metadata":{}},{"cell_type":"code","source":"#Ensure 'Policy Start Date' is in DateTime Format: convert the column into a proper datetime format (if it isn't already).\ndata['Policy Start Date'] = pd.to_datetime(data['Policy Start Date'], errors='coerce')\n\n# Extract Year, Month, and Day: Create new features for the year, month, and day components.\ndata['Policy Start Year'] = data['Policy Start Date'].dt.year\ndata['Policy Start Month'] = data['Policy Start Date'].dt.month\ndata['Policy Start Day'] = data['Policy Start Date'].dt.day\n\ndata = data.drop(['id', 'Policy Start Date'], axis=1, errors='ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:55:36.04573Z","iopub.execute_input":"2025-01-06T01:55:36.045973Z","iopub.status.idle":"2025-01-06T01:55:36.403669Z","shell.execute_reply.started":"2025-01-06T01:55:36.045954Z","shell.execute_reply":"2025-01-06T01:55:36.402964Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Split Dataset into Train and Test**\nSeparate features (X) and target (Premium Amount), then split into training and testing sets.","metadata":{}},{"cell_type":"code","source":"# Define target and features\nX = data.drop(columns=['Premium Amount'])\ny = data['Premium Amount']\n\n# Split into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Identify categorical and numerical columns\ncategorical_columns = X.select_dtypes(include=['object']).columns\nnumerical_columns = X.select_dtypes(include=['int64', 'float64']).columns\n\n# Preprocessing for numerical data\nnum_preprocessor = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')),\n    ('scaler', StandardScaler())\n])\n\n# Preprocessing for categorical data\ncat_preprocessor = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Combine preprocessors\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', num_preprocessor, numerical_columns),\n        ('cat', cat_preprocessor, categorical_columns)\n    ])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:55:36.404635Z","iopub.execute_input":"2025-01-06T01:55:36.404902Z","iopub.status.idle":"2025-01-06T01:55:37.62786Z","shell.execute_reply.started":"2025-01-06T01:55:36.404881Z","shell.execute_reply":"2025-01-06T01:55:37.626854Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Define Regression Models**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression, Ridge, Lasso\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom xgboost import XGBRegressor\n\n# Define models to test\nmodels = {\n    #\"Linear Regression\": LinearRegression(),\n    #\"Ridge Regression\": Ridge(alpha=1.0),\n    #\"Lasso Regression\": Lasso(alpha=0.1),\n    #\"Decision Tree\": DecisionTreeRegressor(max_depth=5, random_state=42),\n    #\"Random Forest\": RandomForestRegressor(n_estimators=100, random_state=42),\n    \"XGBoost\": XGBRegressor(n_estimators=100, learning_rate=0.1, random_state=42)\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:55:37.629031Z","iopub.execute_input":"2025-01-06T01:55:37.629301Z","iopub.status.idle":"2025-01-06T01:55:37.769275Z","shell.execute_reply.started":"2025-01-06T01:55:37.629267Z","shell.execute_reply":"2025-01-06T01:55:37.768693Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Train, Predict, and Evaluate**","metadata":{}},{"cell_type":"code","source":"# Initialize a dictionary to store results\nresults = {}\n\nfor model_name, model in models.items():\n    # Create a pipeline with preprocessing and model\n    pipeline = Pipeline(steps=[('preprocessor', preprocessor),\n                                ('model', model)])\n    \n    # Train the model\n    pipeline.fit(X_train, y_train)\n    \n    # Predict on the test set\n    y_pred = pipeline.predict(X_test)\n    \n    # Calculate R² and RMSLE\n    r2 = r2_score(y_test, y_pred)\n    rmsle = np.sqrt(mean_squared_log_error(y_test, np.maximum(0, y_pred)))  # Ensure no negative values for log\n    \n    # Store results\n    results[model_name] = {'R²': r2, 'RMSLE': rmsle}\n    \n# Display the results\nresults_df = pd.DataFrame(results).T\nprint(results_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:55:37.770124Z","iopub.execute_input":"2025-01-06T01:55:37.770357Z","iopub.status.idle":"2025-01-06T01:55:48.723426Z","shell.execute_reply.started":"2025-01-06T01:55:37.770338Z","shell.execute_reply":"2025-01-06T01:55:48.721683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\n\n\n# Ensure 'Policy Start Date' is in DateTime Format\ntest_data['Policy Start Date'] = pd.to_datetime(test_data['Policy Start Date'], errors='coerce')\n\n# Extract Year, Month, and Day\ntest_data['Policy Start Year'] = test_data['Policy Start Date'].dt.year\ntest_data['Policy Start Month'] = test_data['Policy Start Date'].dt.month\ntest_data['Policy Start Day'] = test_data['Policy Start Date'].dt.day\n\n# Drop unnecessary columns\ntest_ids = test_data['id']  # Preserve the 'id' for final output\ntest_data = test_data.drop(['id', 'Policy Start Date'], axis=1, errors='ignore')\n\n# Identify categorical and numerical columns\ncategorical_columns = test_data.select_dtypes(include=['object']).columns\nnumerical_columns = test_data.select_dtypes(include=['int64', 'float64']).columns\n\n# Preprocessing for numerical data\nnum_preprocessor = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')),\n    ('scaler', StandardScaler())\n])\n\n# Preprocessing for categorical data\ncat_preprocessor = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Combine preprocessors\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', num_preprocessor, numerical_columns),\n        ('cat', cat_preprocessor, categorical_columns)\n    ])\n\n# Define the XGBoost model\nxgb_model = XGBRegressor(n_estimators=100, learning_rate=0.1, random_state=42)\n\n# Create a pipeline with preprocessing and model\npipeline = Pipeline(steps=[('preprocessor', preprocessor),\n                            ('model', xgb_model)])\n\n# Fit the model on the training data\npipeline.fit(X_train, y_train)\n\n# Predict on the test dataset\ntest_predictions = pipeline.predict(test_data)\n\n# Prepare the final output\noutput = pd.DataFrame({'id': test_ids, 'Premium Amount': test_predictions})\n\n# Display the result\noutput.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:55:48.724239Z","iopub.execute_input":"2025-01-06T01:55:48.724509Z","iopub.status.idle":"2025-01-06T01:56:04.883227Z","shell.execute_reply.started":"2025-01-06T01:55:48.724486Z","shell.execute_reply":"2025-01-06T01:56:04.88248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the results to a CSV file in Kaggle working directory\noutput_file_path = '/kaggle/working/predictions.csv'\noutput.to_csv(output_file_path, index=False)\n\nprint(f\"File saved as {output_file_path}. You can download it from the Kaggle output directory.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T01:56:04.883964Z","iopub.execute_input":"2025-01-06T01:56:04.88419Z","iopub.status.idle":"2025-01-06T01:56:05.811259Z","shell.execute_reply.started":"2025-01-06T01:56:04.884172Z","shell.execute_reply":"2025-01-06T01:56:05.810382Z"}},"outputs":[],"execution_count":null}]}