{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"---\n### Upvote, please, if you liked this notebook\n\n---","metadata":{}},{"cell_type":"markdown","source":"# Table of Content\n---\n- <a href='#c1'> 1. Import modules, building classes and read CSV </a>\n\n- <a href='#c2'> 2. Exploratory data analysis </a>\n\n- <a href='#c3'> 3. Data Analysis </a>\n\n- <a href='#c4'> 4. Model building</a>\n\n- <a href='#c5'> 5. Validation</a>\n\n- <a href='#c6'> 6. Results and submissions</a>\n---","metadata":{}},{"cell_type":"markdown","source":"<a id='c1'></a>\n# <div style=\"text-align:center; border-radius:15px 15px; padding:15px; color:#333333; margin:0; ; padding:15px; font-size:100%; font:'Verdana'; background-color:#F5F5F5;border: 1px; overflow:hidden\"><b>1. Import modules, building classes and read CSV</b></div>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n!pip install -q sweetviz\nimport sweetviz as sv \n\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder, LabelEncoder, OrdinalEncoder\nfrom sklearn.model_selection import cross_val_score, KFold\nfrom sklearn.metrics import make_scorer, mean_squared_log_error\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import FunctionTransformer\nfrom sklearn.compose import ColumnTransformer\n\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\n\nimport plotly.express as px\nimport plotly.graph_objects as go\n\nfrom tqdm import tqdm\n\nimport warnings\nimport logging\n\nlogging.basicConfig(filename='warnings.log', level=logging.WARNING)\nlogging.captureWarnings(True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:45:56.782772Z","iopub.execute_input":"2024-12-01T10:45:56.783733Z","iopub.status.idle":"2024-12-01T10:46:12.939721Z","shell.execute_reply.started":"2024-12-01T10:45:56.783684Z","shell.execute_reply":"2024-12-01T10:46:12.938709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:06:29.848105Z","iopub.execute_input":"2024-12-01T11:06:29.848467Z","iopub.status.idle":"2024-12-01T11:06:35.171161Z","shell.execute_reply.started":"2024-12-01T11:06:29.848438Z","shell.execute_reply":"2024-12-01T11:06:35.170421Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bokeh Visualizer Class","metadata":{}},{"cell_type":"code","source":"from bokeh.io import output_notebook\nfrom bokeh.plotting import figure, show, output_notebook\nfrom bokeh.models import ColumnDataSource, FactorRange\nfrom bokeh.transform import factor_cmap, cumsum\nfrom bokeh.palettes import Category20, Spectral6\nfrom scipy.stats import gaussian_kde\noutput_notebook()\n\nclass BokehVisualizer:\n    def __init__(self, dataframe):\n        self.df = dataframe\n        output_notebook()\n\n    def histogram(self, column, title='Histogram', colors=None, hue=None, kde=False):\n        data = self.df\n    \n        if column not in data.columns:\n            raise ValueError(f\"Column '{column}' not found in the dataset.\")\n        if data[column].dropna().empty:\n            raise ValueError(f\"Column '{column}' contains no valid data to plot.\")\n        data[column] = pd.to_numeric(data[column], errors='coerce')\n        data = data.dropna(subset=[column])\n        \n        p = figure(title=title, background_fill_color=\"lightgray\")\n        \n        if hue:\n            if hue not in data.columns:\n                raise ValueError(f\"Hue column '{hue}' not found in the dataset.\")\n            categories = data[hue].unique()\n            if not colors:\n                available_colors = Category20[20] if len(categories) > 10 else Category20[10]\n                colors = available_colors[:len(categories)]\n            color_dict = {cat: colors[i % len(colors)] for i, cat in enumerate(categories)}\n            \n            for cat in categories:\n                subset = data[data[hue] == cat]\n                hist, edges = np.histogram(subset[column], bins=50)\n                p.quad(top=hist, bottom=0, left=edges[:-1], right=edges[1:],\n                       fill_color=color_dict[cat], line_color='black', fill_alpha=0.7, legend_label=str(cat))\n                \n                if kde:\n                    kde_values = gaussian_kde(subset[column].dropna())\n                    x = np.linspace(edges[0], edges[-1], 200)\n                    y = kde_values(x) * len(subset[column]) * (edges[1] - edges[0])  # Scale to match histogram\n                    p.line(x, y, line_width=2, color=color_dict[cat], legend_label=f'{cat} KDE')\n            \n            p.legend.location = \"top_right\"\n        else:\n            hist, edges = np.histogram(data[column], bins=50)\n            p.quad(top=hist, bottom=0, left=edges[:-1], right=edges[1:],\n                   fill_color='#87CEEB', line_color='black', fill_alpha=0.7)\n            \n            if kde:\n                kde_values = gaussian_kde(data[column])\n                x = np.linspace(edges[0], edges[-1], 200)\n                y = kde_values(x) * len(data[column]) * (edges[1] - edges[0])  # Scale to match histogram\n                p.line(x, y, line_width=2, color='blue', legend_label='KDE')\n        \n        p.xaxis.axis_label = column\n        p.yaxis.axis_label = 'Count'\n        p.title.align = 'center'\n        p.title.text_font_size = '20pt'\n        p.title.text_color = 'darkblue'\n        \n        show(p)\n\n    def scatter(self, x_column, y_column, title='Scatter Plot', colors=None, hue=None):\n        data = self.df\n        source = ColumnDataSource(data)\n        p = figure(title=title, background_fill_color=\"lightgray\")\n        if hue:\n            categories = data[hue].unique()\n            if not colors:\n                colors = Category20[len(categories)] if len(categories) <= 20 else Spectral6 * (len(categories) // 6 + 1)\n            mapper = factor_cmap(field_name=hue, palette=colors, factors=categories)\n            p.scatter(x_column, y_column, source=source, color=mapper, legend_field=hue,\n                      size=5, line_color='black', alpha=0.7)\n            p.legend.location = \"top_left\"\n        else:\n            p.scatter(x_column, y_column, source=source, color='blue',\n                      size=5, line_color='black', alpha=0.7)\n        p.xaxis.axis_label = x_column\n        p.yaxis.axis_label = y_column\n        p.title.align = 'center'\n        p.title.text_font_size = '20pt'\n        p.title.text_color = 'darkblue'\n        show(p)\n\n    def box_plot(self, column, title='Box Plot', colors=None, hue=None):\n        data = self.df\n        if hue:\n            categories = data[hue].unique().astype(str)\n            if not colors:\n                colors = Category20[len(categories)] if len(categories) <= 20 else Spectral6 * (len(categories) // 6 + 1)\n            color_dict = {cat: colors[i % len(colors)] for i, cat in enumerate(categories)}\n            p = figure(x_range=categories, title=title, background_fill_color=\"lightgray\")\n    \n            for i, cat in enumerate(categories):\n                subset = data[data[hue].astype(str) == cat]\n                q1 = subset[column].quantile(0.25)\n                q2 = subset[column].quantile(0.5)\n                q3 = subset[column].quantile(0.75)\n                iqr = q3 - q1\n                upper = min(q3 + 1.5 * iqr, subset[column].max())\n                lower = max(q1 - 1.5 * iqr, subset[column].min())\n    \n                p.segment([cat], [upper], [cat], [q3], line_color=\"black\")\n                p.segment([cat], [lower], [cat], [q1], line_color=\"black\")\n                p.vbar([cat], 0.7, q2, q3, fill_color=color_dict[cat], fill_alpha=0.5, line_color=\"black\")\n                p.vbar([cat], 0.7, q1, q2, fill_color=color_dict[cat], fill_alpha=0.5, line_color=\"black\")\n    \n                jittered_x = [cat] * len(subset[column]) \n                p.circle(jittered_x, subset[column], size=5, color=color_dict[cat], alpha=0.6)\n        else:\n            q1 = data[column].quantile(0.25)\n            q2 = data[column].quantile(0.5)\n            q3 = data[column].quantile(0.75)\n            iqr = q3 - q1\n            upper = min(q3 + 1.5 * iqr, data[column].max())\n            lower = max(q1 - 1.5 * iqr, data[column].min())\n            p = figure(title=title, background_fill_color=\"lightgray\")\n    \n            p.segment([1], [upper], [1], [q3], line_color=\"black\")\n            p.segment([1], [lower], [1], [q1], line_color=\"black\")\n            p.vbar([1], 0.7, q2, q3, fill_color='green', fill_alpha=0.5, line_color=\"black\")\n            p.vbar([1], 0.7, q1, q2, fill_color='green', fill_alpha=0.5, line_color=\"black\")\n    \n            jittered_x = [1] * len(data[column])\n            p.circle(jittered_x, data[column], size=5, color='green', alpha=0.6)\n    \n        p.yaxis.axis_label = column\n        p.title.align = 'center'\n        p.title.text_font_size = '20pt'\n        p.title.text_color = 'darkblue'\n        show(p)\n\n    def line_plot(self, x_column, y_column, title='Line Plot', colors=None, hue=None):\n        data = self.df\n        p = figure(title=title, background_fill_color=\"lightgray\", x_axis_type='auto')\n        if hue:\n            categories = data[hue].unique()\n            if not colors:\n                colors = Category20[len(categories)] if len(categories) <= 20 else Spectral6 * (len(categories) // 6 + 1)\n            color_dict = {cat: colors[i % len(colors)] for i, cat in enumerate(categories)}\n            for cat in categories:\n                subset = data[data[hue] == cat]\n                p.line(subset[x_column], subset[y_column], line_width=2,\n                       color=color_dict[cat], legend_label=str(cat))\n            p.legend.location = \"top_left\"\n        else:\n            p.line(data[x_column], data[y_column], line_width=2, color='blue')\n        p.xaxis.axis_label = x_column\n        p.yaxis.axis_label = y_column\n        p.title.align = 'center'\n        p.title.text_font_size = '20pt'\n        p.title.text_color = 'darkblue'\n        show(p)\n\n    def bar_chart(self, x_column, y_column, title='Bar Chart', colors=None, hue=None):\n        data = self.df\n        if hue:\n            categories = data[hue].unique().astype(str)\n            x_factors = data[x_column].astype(str).unique()\n            factors = [(x, cat) for x in x_factors for cat in categories]\n            if not colors:\n                colors = Category20[len(categories)] if len(categories) <= 20 else Spectral6 * (len(categories) // 6 + 1)\n            source = ColumnDataSource(data=dict(\n                x=[(str(x), str(cat)) for x, cat in zip(data[x_column], data[hue])],\n                counts=data[y_column],\n            ))\n            p = figure(x_range=FactorRange(*factors), title=title, background_fill_color=\"lightgray\")\n            p.vbar(x='x', top='counts', width=0.9, source=source,\n                   fill_color=factor_cmap('x', palette=colors, factors=categories, start=1, end=2))\n            p.xaxis.axis_label = x_column\n            p.yaxis.axis_label = y_column\n            p.xaxis.major_label_orientation = 1\n            p.legend.title = hue\n        else:\n            x = data[x_column].astype(str)\n            counts = data[y_column]\n            p = figure(x_range=x.unique(), title=title, background_fill_color=\"lightgray\")\n            p.vbar(x=x, top=counts, width=0.9, color='blue')\n            p.xaxis.axis_label = x_column\n            p.yaxis.axis_label = y_column\n            p.xaxis.major_label_orientation = 1\n        p.title.align = 'center'\n        p.title.text_font_size = '20pt'\n        p.title.text_color = 'darkblue'\n        show(p)\n\n    def pie_chart(self, column, title='Pie Chart'):\n        data = self.df\n        value_counts = data[column].value_counts()\n        total = value_counts.sum()\n        percentages = (value_counts / total) * 100\n    \n        # Filter categories with percentages <= 1 and group them as \"Other\"\n        filtered_value_counts = value_counts[percentages > 1]\n        other_sum = value_counts[percentages <= 1].sum()\n        if other_sum > 0:\n            filtered_value_counts['Other'] = other_sum\n    \n        # Create DataFrame for plotting\n        data = pd.DataFrame({\n            'categories': filtered_value_counts.index.astype(str),\n            'counts': filtered_value_counts.values\n        })\n        data['angle'] = data['counts'] / data['counts'].sum() * 2 * np.pi\n        data['percentage'] = (data['counts'] / data['counts'].sum() * 100).round(2)\n    \n        # Choose colors based on the number of categories\n        num_categories = len(data)\n        if num_categories <= 20:\n            if num_categories <= 2:\n                palette = ['#1f77b4', '#ff7f0e'][:num_categories]\n            elif num_categories <= 3:\n                palette = Category20[3][:num_categories]\n            elif num_categories <= 6:\n                palette = Category20[6][:num_categories]\n            elif num_categories <= 10:\n                palette = Category20[10][:num_categories]\n            else:\n                palette = Category20[20][:num_categories]\n        else:\n            palette = viridis(num_categories)\n    \n        data['color'] = palette\n    \n        # Create the figure\n        p = figure(title=title, background_fill_color=\"lightgray\", tools=\"hover\",\n                   tooltips=\"@categories: @counts (@percentage%)\", x_range=(-0.5, 1.0))\n        \n        # Add a donut chart\n        p.annular_wedge(x=0, y=1, inner_radius=0.2, outer_radius=0.4,\n                        start_angle=cumsum('angle', include_zero=True),\n                        end_angle=cumsum('angle'),\n                        line_color=\"white\", fill_color='color', source=data)\n        \n        # Add percentages in the center\n        p.text(x=[0], y=[1], text=[f\"Total\\n{total}\"], text_align=\"center\",\n               text_font_size=\"12pt\", text_baseline=\"middle\")\n    \n        # Format chart\n        p.title.align = 'center'\n        p.title.text_font_size = '20pt'\n        p.title.text_color = 'darkblue'\n        p.axis.axis_label = None\n        p.axis.visible = False\n        p.grid.grid_line_color = None\n    \n        show(p)\n\nbv = BokehVisualizer(train)","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-01T09:11:48.171544Z","iopub.execute_input":"2024-12-01T09:11:48.171969Z","iopub.status.idle":"2024-12-01T09:11:48.237617Z","shell.execute_reply.started":"2024-12-01T09:11:48.171931Z","shell.execute_reply":"2024-12-01T09:11:48.236344Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c2'></a>\n# <div style=\"text-align:center; border-radius:15px 15px; padding:15px; color:#333333; margin:0; ; padding:15px; font-size:100%; font:'Verdana'; background-color:#F5F5F5;border: 1px; overflow:hidden\"><b> 2. Exploratory Data Analysis </b></div>","metadata":{}},{"cell_type":"markdown","source":"---\n\n## Dataset Description\n\n### **Structure of the Dataset**\n- **Number of Records**: 1,200,000\n- **Number of Columns**: 9\n- **Data Types**:\n  - **Integer (`int64`)**: Includes `id` and other numeric attributes.\n  - **Floating-point (`float64`)**: Includes measures like `Credit Score` and `Annual Income`.\n\n---\n\n### **Columns**\n1. **id**: Unique identifier for each record.\n2. **Age**: Age of the individual (in years).\n   - **Range**: 18–64.\n   - **Mean**: 41.14 years.\n   - **Standard Deviation**: 13.54 years.\n3. **Annual Income**: Annual income of the individual (in the dataset’s monetary unit).\n   - **Range**: 10,000–149,997.\n   - **Mean**: 32,745.22.\n   - **Standard Deviation**: 32,179.51.\n4. **Number of Dependents**: Number of dependents associated with the individual.\n   - **Range**: 0–4.\n   - **Mean**: 2.01.\n   - **Standard Deviation**: 1.42.\n5. **Health Score**: Health score assigned to the individual.\n   - **Range**: 2.01–58.97.\n   - **Mean**: 25.61.\n   - **Standard Deviation**: 12.20.\n6. **Previous Claims**: Number of previous claims made by the individual.\n   - **Range**: 0–9.\n   - **Mean**: 1.00.\n   - **Standard Deviation**: 0.98.\n7. **Vehicle Age**: Age of the individual’s vehicle (in years).\n   - **Range**: 0–19.\n   - **Mean**: 9.57.\n   - **Standard Deviation**: 5.77.\n8. **Credit Score**: Credit score of the individual.\n   - **Range**: 300–849.\n   - **Mean**: 592.92.\n   - **Standard Deviation**: 149.98.\n9. **Insurance Duration**: Duration of the insurance (in years).\n   - **Range**: 1–9.\n   - **Mean**: 5.02.\n   - **Standard Deviation**: 2.59.\n10. **Premium Amount**: Premium amount paid by the individual.\n    - **Range**: 200–4,999.\n    - **Mean**: 1,102.55.\n    - **Standard Deviation**: 864.99.\n\n---","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:56:26.792064Z","iopub.execute_input":"2024-12-01T08:56:26.792747Z","iopub.status.idle":"2024-12-01T08:56:26.834204Z","shell.execute_reply.started":"2024-12-01T08:56:26.792701Z","shell.execute_reply":"2024-12-01T08:56:26.833187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:56:26.835665Z","iopub.execute_input":"2024-12-01T08:56:26.83639Z","iopub.status.idle":"2024-12-01T08:56:27.493683Z","shell.execute_reply.started":"2024-12-01T08:56:26.836339Z","shell.execute_reply":"2024-12-01T08:56:27.492544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:56:27.494872Z","iopub.execute_input":"2024-12-01T08:56:27.495213Z","iopub.status.idle":"2024-12-01T08:56:28.179892Z","shell.execute_reply.started":"2024-12-01T08:56:27.495175Z","shell.execute_reply":"2024-12-01T08:56:28.178721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target = 'Premium Amount'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:47:30.106869Z","iopub.execute_input":"2024-12-01T10:47:30.10769Z","iopub.status.idle":"2024-12-01T10:47:30.111448Z","shell.execute_reply.started":"2024-12-01T10:47:30.107657Z","shell.execute_reply":"2024-12-01T10:47:30.110457Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3'></a>\n# <div style=\"text-align:center; border-radius:15px 15px; padding:15px; color:#333333; margin:0; ; padding:15px; font-size:100%; font:'Verdana'; background-color:#F5F5F5;border: 1px; overflow:hidden\"><b>3. Data Analysis</b></div>","metadata":{}},{"cell_type":"code","source":"missing_nan_count = train.isnull().sum()\n\nmissing_nan_count","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:56:37.610162Z","iopub.execute_input":"2024-12-01T08:56:37.61059Z","iopub.status.idle":"2024-12-01T08:56:38.246295Z","shell.execute_reply.started":"2024-12-01T08:56:37.61055Z","shell.execute_reply":"2024-12-01T08:56:38.245129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols_with_na = train.columns[train.isnull().any()]\nsns.heatmap(train[cols_with_na].isnull(), cbar=False, cmap=\"viridis\")\nplt.title(\"Missing Values Heatmap\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T08:56:38.247905Z","iopub.execute_input":"2024-12-01T08:56:38.248341Z","iopub.status.idle":"2024-12-01T08:56:51.691682Z","shell.execute_reply.started":"2024-12-01T08:56:38.248299Z","shell.execute_reply":"2024-12-01T08:56:51.6906Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.1'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; background: #E8E8E8; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\"><b>3.1. Age</b>\n</div>","metadata":{}},{"cell_type":"code","source":"print(train.Age.describe())\ncorrelation = train.Age.corr(train[target])\nprint(f\"Correlation between age and target: {correlation}\")\nsns.histplot(train.Age, bins=30, kde=True)\nplt.title(\"Hist with KDE of age distr.\")\nplt.xlabel(\"Value\")\nplt.ylabel(\"Freq\")\nplt.show()\nbv.box_plot('Age')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:22:45.915074Z","iopub.execute_input":"2024-12-01T09:22:45.91552Z","iopub.status.idle":"2024-12-01T09:22:54.960287Z","shell.execute_reply.started":"2024-12-01T09:22:45.915467Z","shell.execute_reply":"2024-12-01T09:22:54.959077Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.2'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; background: #E8E8E8; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\"><b>3.2. Gender</b>\n</div>","metadata":{}},{"cell_type":"code","source":"print(train.Gender.describe())\nbv.pie_chart('Gender', title='Pie Chart for Gender value')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:22:24.204183Z","iopub.execute_input":"2024-12-01T09:22:24.204858Z","iopub.status.idle":"2024-12-01T09:22:24.606227Z","shell.execute_reply.started":"2024-12-01T09:22:24.204808Z","shell.execute_reply":"2024-12-01T09:22:24.605129Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.3'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; background: #E8E8E8; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\"><b>3.3. Annual Income</b>\n</div>","metadata":{}},{"cell_type":"code","source":"print(train['Annual Income'].describe())\ncorrelation = train['Annual Income'].corr(train[target])\nprint(f\"Correlation between Annual Income and target: {correlation}\")\nbv.histogram('Annual Income', kde=True, title='Histogram of Annual Income distribution')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:22:05.999247Z","iopub.execute_input":"2024-12-01T09:22:05.999628Z","iopub.status.idle":"2024-12-01T09:22:10.878836Z","shell.execute_reply.started":"2024-12-01T09:22:05.999594Z","shell.execute_reply":"2024-12-01T09:22:10.877733Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.4'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; background: #E8E8E8; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\"><b>3.4. Marital Status</b>\n</div>","metadata":{}},{"cell_type":"code","source":"print(train['Marital Status'].describe())\nbv.pie_chart('Marital Status', title=\"Pie Chart of Marital Status\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:21:44.607685Z","iopub.execute_input":"2024-12-01T09:21:44.608701Z","iopub.status.idle":"2024-12-01T09:21:45.014841Z","shell.execute_reply.started":"2024-12-01T09:21:44.608648Z","shell.execute_reply":"2024-12-01T09:21:45.014042Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.5'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; background: #E8E8E8; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\"><b>3.5. Marital Status</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics and correlation\nprint(train['Number of Dependents'].describe())\ncorrelation = train['Number of Dependents'].corr(train['Premium Amount'])\nprint(f\"Correlation between Number of Dependents and Premium Amount: {correlation}\")\n\n# Visualization\nbv.pie_chart('Number of Dependents', title=\"Pieplot of Number of Dependents\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:28:20.712142Z","iopub.execute_input":"2024-12-01T09:28:20.712508Z","iopub.status.idle":"2024-12-01T09:28:21.007376Z","shell.execute_reply.started":"2024-12-01T09:28:20.712475Z","shell.execute_reply":"2024-12-01T09:28:21.006398Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.6'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.6. Education Level</b>\n</div>\n","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics\nprint(train['Education Level'].describe())\n\n# Visualization\nbv.pie_chart('Education Level', title=\"Pie Chart of Education Level\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:31:57.949547Z","iopub.execute_input":"2024-12-01T09:31:57.950463Z","iopub.status.idle":"2024-12-01T09:31:58.379715Z","shell.execute_reply.started":"2024-12-01T09:31:57.950405Z","shell.execute_reply":"2024-12-01T09:31:58.378769Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.7'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.7. Occupation</b>\n</div>\n","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics\nprint(train['Occupation'].describe())\n\n# Visualization\nbv.pie_chart('Occupation', title=\"Pie Chart of Occupation\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:32:10.994481Z","iopub.execute_input":"2024-12-01T09:32:10.994905Z","iopub.status.idle":"2024-12-01T09:32:11.39698Z","shell.execute_reply.started":"2024-12-01T09:32:10.99487Z","shell.execute_reply":"2024-12-01T09:32:11.395798Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.8'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.8. Health Score</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics and correlation\nprint(train['Health Score'].describe())\ncorrelation = train['Health Score'].corr(train['Premium Amount'])\nprint(f\"Correlation between Health Score and Premium Amount: {correlation}\")\n\n# Visualization\nbv.histogram('Health Score', title=\"Histogram of Health Score with KDE\", kde=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:32:44.541028Z","iopub.execute_input":"2024-12-01T09:32:44.541455Z","iopub.status.idle":"2024-12-01T09:32:49.29001Z","shell.execute_reply.started":"2024-12-01T09:32:44.541389Z","shell.execute_reply":"2024-12-01T09:32:49.288889Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.9'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.9. Location</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics\nprint(train['Location'].describe())\n\n# Visualization\nbv.pie_chart('Location', title=\"Pie Chart of Location\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:33:07.906285Z","iopub.execute_input":"2024-12-01T09:33:07.907045Z","iopub.status.idle":"2024-12-01T09:33:08.353921Z","shell.execute_reply.started":"2024-12-01T09:33:07.906999Z","shell.execute_reply":"2024-12-01T09:33:08.352991Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.10'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.10. Policy Type</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics\nprint(train['Policy Type'].describe())\n\n# Visualization\nbv.pie_chart('Policy Type', title=\"Pie Chart of Policy Type\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T09:33:27.889361Z","iopub.execute_input":"2024-12-01T09:33:27.889781Z","iopub.status.idle":"2024-12-01T09:33:28.344675Z","shell.execute_reply.started":"2024-12-01T09:33:27.889746Z","shell.execute_reply":"2024-12-01T09:33:28.343671Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.11'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.11. Previous Claims</b>\n</div>\n ","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics and correlation\nprint(train['Previous Claims'].describe())\ncorrelation = train['Previous Claims'].corr(train['Premium Amount'])\nprint(f\"Correlation between Previous Claims and Premium Amount: {correlation}\")\n\n# Visualization\nbv.pie_chart('Previous Claims', title=\"Pie chart of Previous Claims\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:01:52.234714Z","iopub.execute_input":"2024-12-01T10:01:52.235111Z","iopub.status.idle":"2024-12-01T10:01:52.58302Z","shell.execute_reply.started":"2024-12-01T10:01:52.235074Z","shell.execute_reply":"2024-12-01T10:01:52.581987Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.12'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.12. Vehicle Age</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics and correlation\nprint(train['Vehicle Age'].describe())\ncorrelation = train['Vehicle Age'].corr(train['Premium Amount'])\nprint(f\"Correlation between Vehicle Age and Premium Amount: {correlation}\")\n\n# Visualization\nbv.histogram('Vehicle Age', title=\"Histogram of Vehicle Age with KDE\", kde=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:02:09.904571Z","iopub.execute_input":"2024-12-01T10:02:09.904958Z","iopub.status.idle":"2024-12-01T10:02:14.580405Z","shell.execute_reply.started":"2024-12-01T10:02:09.904923Z","shell.execute_reply":"2024-12-01T10:02:14.579274Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.13'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.13. Credit Score</b>\n</div>\n","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics and correlation\nprint(train['Credit Score'].describe())\ncorrelation = train['Credit Score'].corr(train['Premium Amount'])\nprint(f\"Correlation between Credit Score and Premium Amount: {correlation}\")\n\n# Visualization\nbv.histogram('Credit Score', title=\"Histogram of Credit Score with KDE\", kde=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:02:34.320954Z","iopub.execute_input":"2024-12-01T10:02:34.321351Z","iopub.status.idle":"2024-12-01T10:02:38.727783Z","shell.execute_reply.started":"2024-12-01T10:02:34.321314Z","shell.execute_reply":"2024-12-01T10:02:38.726713Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.14'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.14. Insurance Duration</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics and correlation\nprint(train['Insurance Duration'].describe())\ncorrelation = train['Insurance Duration'].corr(train['Premium Amount'])\nprint(f\"Correlation between Insurance Duration and Premium Amount: {correlation}\")\n\n# Visualization\nbv.pie_chart('Insurance Duration', title=\"Pie chart of Insurance Duration\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:03:28.488165Z","iopub.execute_input":"2024-12-01T10:03:28.488711Z","iopub.status.idle":"2024-12-01T10:03:28.826644Z","shell.execute_reply.started":"2024-12-01T10:03:28.488669Z","shell.execute_reply":"2024-12-01T10:03:28.825651Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.15'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.15. Policy Start Date</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics\nprint(train['Policy Start Date'].describe())\n# Ensure the column is in datetime format\ntrain['Policy Start Date'] = pd.to_datetime(train['Policy Start Date'], errors='coerce')\n\n# Drop any rows with invalid or missing dates\nvalid_dates = train['Policy Start Date'].dropna()\n\n# Create a histogram of policy start dates by year\nplt.figure(figsize=(12, 6))\nvalid_dates.dt.year.value_counts().sort_index().plot(kind='bar', width=0.8)\nplt.title(\"Distribution of Policies by Start Year\")\nplt.xlabel(\"Year\")\nplt.ylabel(\"Number of Policies\")\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:05:26.104468Z","iopub.execute_input":"2024-12-01T10:05:26.104856Z","iopub.status.idle":"2024-12-01T10:05:27.286544Z","shell.execute_reply.started":"2024-12-01T10:05:26.104814Z","shell.execute_reply":"2024-12-01T10:05:27.285457Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.16'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.16. Customer Feedback</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics\nprint(train['Customer Feedback'].describe())\n\n# Visualization\nbv.pie_chart('Customer Feedback', title=\"Pie Chart of Customer Feedback\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:04:23.508945Z","iopub.execute_input":"2024-12-01T10:04:23.509336Z","iopub.status.idle":"2024-12-01T10:04:23.989601Z","shell.execute_reply.started":"2024-12-01T10:04:23.509299Z","shell.execute_reply":"2024-12-01T10:04:23.988538Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.17'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.17. Smoking Status</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics\nprint(train['Smoking Status'].describe())\n\n# Visualization\nbv.pie_chart('Smoking Status', title=\"Pie Chart of Smoking Status\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:04:35.047084Z","iopub.execute_input":"2024-12-01T10:04:35.048013Z","iopub.status.idle":"2024-12-01T10:04:35.537687Z","shell.execute_reply.started":"2024-12-01T10:04:35.047973Z","shell.execute_reply":"2024-12-01T10:04:35.536651Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.18'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.18. Exercise Frequency</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics\nprint(train['Exercise Frequency'].describe())\n\n# Visualization\nbv.pie_chart('Exercise Frequency', title=\"Pie Chart of Exercise Frequency\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:06:13.903612Z","iopub.execute_input":"2024-12-01T10:06:13.903999Z","iopub.status.idle":"2024-12-01T10:06:14.40564Z","shell.execute_reply.started":"2024-12-01T10:06:13.903961Z","shell.execute_reply":"2024-12-01T10:06:14.404502Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.19'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.19. Property Type</b>\n</div>\n","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics\nprint(train['Property Type'].describe())\n\n# Visualization\nbv.pie_chart('Property Type', title=\"Pie Chart of Property Type\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:06:24.09319Z","iopub.execute_input":"2024-12-01T10:06:24.09361Z","iopub.status.idle":"2024-12-01T10:06:24.605017Z","shell.execute_reply.started":"2024-12-01T10:06:24.093564Z","shell.execute_reply":"2024-12-01T10:06:24.603905Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c3.20'></a>\n<div style=\"text-align:center; border-radius:15px; padding:15px; color:#333333; margin:0; font-size:150%; font:'Verdana'; background: #F5F5F5; border: 1px solid #CCCCCC; box-shadow: 0px 4px 8px rgba(0, 0, 0, 0.2); overflow:hidden;\">\n<b>3.20. Premium Amount</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# Descriptive statistics\nprint(train['Premium Amount'].describe())\n\n# Visualization\nbv.histogram('Premium Amount', title=\"Histogram of Premium Amount with KDE\", kde=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:06:34.804143Z","iopub.execute_input":"2024-12-01T10:06:34.804537Z","iopub.status.idle":"2024-12-01T10:06:40.616493Z","shell.execute_reply.started":"2024-12-01T10:06:34.804501Z","shell.execute_reply":"2024-12-01T10:06:40.61529Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"code","source":"# Compute the correlation matrix for numerical features\ncorrelation_matrix = train.select_dtypes(include=['float64', 'int64']).corr()\n\n# Set up the matplotlib figure\nplt.figure(figsize=(12, 10))\n\n# Create the heatmap using seaborn\nsns.heatmap(correlation_matrix, annot=True, fmt=\".2f\", cmap=\"coolwarm\", square=True, mask=np.triu(np.ones(correlation_matrix.shape)))\n\n# Set heatmap title and labels\nplt.title(\"Correlation Heatmap (Diagonal Matrix)\", fontsize=16)\nplt.xticks(fontsize=10)\nplt.yticks(fontsize=10)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:07:16.681847Z","iopub.execute_input":"2024-12-01T10:07:16.682603Z","iopub.status.idle":"2024-12-01T10:07:17.68912Z","shell.execute_reply.started":"2024-12-01T10:07:16.682562Z","shell.execute_reply":"2024-12-01T10:07:17.688061Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c4'></a>\n# <div style=\"text-align:center; border-radius:15px 15px; padding:15px; color:#333333; margin:0; ; padding:15px; font-size:100%; font:'Verdana'; background-color:#F5F5F5;border: 1px; overflow:hidden\"><b> 4. Model Building </b></div>","metadata":{}},{"cell_type":"markdown","source":"## DataPreprocessor Class","metadata":{}},{"cell_type":"code","source":"class DataPreprocessor:\n    \"\"\"\n    A class for preprocessing data using ColumnTransformer and Pipelines.\n    Automatically handles numerical and categorical features with customizable pipelines.\n\n    Params\n    ----------\n    numerical_columns : list of str\n        List of numerical feature names.\n    one_hot_columns : list of str\n        List of categorical feature names for one-hot encoding.\n    label_columns : list of str\n        List of categorical feature names for label encoding.\n\n    Attributes\n    ----------\n    preprocessor : ColumnTransformer\n        Combines numerical, one-hot, and label pipelines for data transformation.\n\n    Methods\n    -------\n    fit(X)\n        Fits the preprocessor on the provided DataFrame.\n    transform(X)\n        Transforms the DataFrame using the fitted preprocessor.\n    fit_transform(X)\n        Fits the preprocessor and transforms the DataFrame.\n    \"\"\"\n\n    def __init__(self, numerical_columns, one_hot_columns, label_columns):\n        self.numerical_columns = numerical_columns\n        self.one_hot_columns = one_hot_columns\n        self.label_columns = label_columns\n\n        # Define preprocessing pipelines\n        self.numerical_pipeline = Pipeline(steps=[\n            ('imputer', SimpleImputer(strategy='median')),\n            ('scaler', StandardScaler()),\n            ('convert_to_float32', FunctionTransformer(lambda x: x.astype(np.float32)))\n        ])\n\n        self.one_hot_pipeline = Pipeline(steps=[\n            ('imputer', SimpleImputer(strategy='constant', fill_value='missing')),\n            ('one_hot', OneHotEncoder(drop='first', sparse=False, handle_unknown='ignore'))\n        ])\n\n        self.label_pipeline = Pipeline(steps=[\n            ('imputer', SimpleImputer(strategy='constant', fill_value='missing')),\n            ('ordinal', OrdinalEncoder(dtype=np.int32, handle_unknown='use_encoded_value', unknown_value=-1))\n        ])\n\n        # Combine the pipelines into a ColumnTransformer\n        self.preprocessor = ColumnTransformer(\n            transformers=[\n                ('num', self.numerical_pipeline, self.numerical_columns),\n                ('one_hot', self.one_hot_pipeline, self.one_hot_columns),\n                ('label', self.label_pipeline, self.label_columns)\n            ]\n        )\n\n    def fit(self, X):\n        self.preprocessor.fit(X)\n\n    def transform(self, X):\n        return self.preprocessor.transform(X)\n\n    def fit_transform(self, X):\n        return self.preprocessor.fit_transform(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:47:06.062386Z","iopub.execute_input":"2024-12-01T10:47:06.062842Z","iopub.status.idle":"2024-12-01T10:47:06.070935Z","shell.execute_reply.started":"2024-12-01T10:47:06.062809Z","shell.execute_reply":"2024-12-01T10:47:06.070087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop the 'id' column from the dataset (if present)\ntrain = train.drop('id', axis=1)\ntest_ids = test['id']\ntest = test.drop('id', axis=1)\n\n# Separate features and target for the training set\nX_train = train.drop(columns=[target])\ny_train = train[target]\n\n# Test set should only contain features\nX_test = test.copy()\n\n# Automatically determine columns for preprocessing\nnumerical_columns = X_train.select_dtypes(include=['float64', 'int64']).columns.tolist()\ncategorical_columns = X_train.select_dtypes(include=['object']).columns.tolist()\n\n# Further classify categorical columns into one-hot and label encoding based on cardinality\nhigh_cardinality_threshold = 10\none_hot_columns = [col for col in categorical_columns if X_train[col].nunique() <= high_cardinality_threshold]\nlabel_columns = [col for col in categorical_columns if X_train[col].nunique() > high_cardinality_threshold]\n\n# Initialize the preprocessor\npreprocessor = DataPreprocessor(numerical_columns, one_hot_columns, label_columns)\n\n# Fit the preprocessor on the training features\npreprocessor.fit(X_train)\n\n# Transform both training and test datasets\nX_encoded = preprocessor.transform(X_train)  \nX_test_encoded = preprocessor.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:47:54.523045Z","iopub.execute_input":"2024-12-01T10:47:54.523386Z","iopub.status.idle":"2024-12-01T10:48:11.762171Z","shell.execute_reply.started":"2024-12-01T10:47:54.523356Z","shell.execute_reply":"2024-12-01T10:48:11.7612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Custom scorer for RMSLE\ndef rmsle(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, np.maximum(y_pred, 0)))\n\nrmsle_scorer = make_scorer(rmsle, greater_is_better=False)\n\n# XGBoost Regressor with GPU\nmodel_xgb = XGBRegressor(\n    objective='reg:squarederror', # Regression objective\n    tree_method='gpu_hist',       # Use GPU for histogram-based tree building\n    predictor='gpu_predictor',    # Use GPU for predictions\n    eval_metric='rmsle',          # Evaluation metric\n    n_estimators=100,             # Number of trees\n    learning_rate=0.1,            # Learning rate\n    max_depth=6,                  # Maximum depth of trees\n)\n\n# LightGBM Regressor with GPU\nmodel_lgb = LGBMRegressor(\n    objective='regression',       # Regression objective\n    metric='rmsle',               # Evaluation metric\n    n_estimators=100,             # Number of trees\n    learning_rate=0.1,            # Learning rate\n    max_depth=-1,                 # No maximum depth by default\n    device='gpu',                 # Use GPU for training\n    gpu_platform_id=0,            # Platform ID for GPU (adjust if needed)\n    gpu_device_id=0,              # Device ID for GPU\n)\n\n# CatBoost Regressor with GPU\nmodel_cat = CatBoostRegressor(\n    loss_function='RMSE',         # Use RMSE for regression\n    eval_metric='RMSE',           # RMSE as the evaluation metric\n    iterations=100,               # Number of trees\n    learning_rate=0.1,            # Learning rate\n    depth=6,                      # Maximum depth of trees\n    verbose=0,                    # Suppress verbose output\n    task_type='CPU'               # Ensure compatibility with CPU (adjust to 'GPU' if available)\n)\n\n# Suppress verbose output for models\nmodel_xgb.set_params(verbosity=0)\nmodel_lgb.set_params(verbose=-1)\nmodel_cat.set_params(verbose=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:51:10.435964Z","iopub.execute_input":"2024-12-01T10:51:10.4363Z","iopub.status.idle":"2024-12-01T10:51:10.446039Z","shell.execute_reply.started":"2024-12-01T10:51:10.43627Z","shell.execute_reply":"2024-12-01T10:51:10.445185Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c5'></a>\n# <div style=\"text-align:center; border-radius:15px 15px; padding:15px; color:#333333; margin:0; ; padding:15px; font-size:100%; font:'Verdana'; background-color:#F5F5F5;border: 1px; overflow:hidden\"><b>5. Validation</b>\n</div>","metadata":{}},{"cell_type":"code","source":"# K-Fold Cross Validation Loop (No changes needed here)\nkf = KFold(n_splits=10, shuffle=True, random_state=42)\n\noof_preds_xgb = np.zeros(len(X_train))\noof_preds_lgb = np.zeros(len(X_train))\noof_preds_cat = np.zeros(len(X_train))\noof_preds_avg = np.zeros(len(X_train))\n\nfor train_index, valid_index in tqdm(kf.split(X_encoded), total=kf.get_n_splits(), desc=\"KFold Loop\"):\n    # Split data into training and validation\n    X_train_fold, X_valid_fold = X_encoded[train_index], X_encoded[valid_index]\n    y_train_fold, y_valid_fold = y_train.iloc[train_index], y_train.iloc[valid_index]\n\n    # Fit models\n    model_xgb.fit(X_train_fold, y_train_fold)\n    model_lgb.fit(X_train_fold, y_train_fold)\n    model_cat.fit(X_train_fold, y_train_fold)\n\n    # Predict on validation data\n    preds_xgb = np.maximum(model_xgb.predict(X_valid_fold), 0)  # Ensure non-negative predictions\n    preds_lgb = np.maximum(model_lgb.predict(X_valid_fold), 0)\n    preds_cat = np.maximum(model_cat.predict(X_valid_fold), 0)\n\n    # Store OOF predictions\n    oof_preds_xgb[valid_index] = preds_xgb\n    oof_preds_lgb[valid_index] = preds_lgb\n    oof_preds_cat[valid_index] = preds_cat\n\n    # Average predictions\n    preds_avg = (preds_xgb + preds_lgb + preds_cat) / 3\n    oof_preds_avg[valid_index] = preds_avg\n\n# Calculate RMSLE for each model\nscore_xgb = np.sqrt(mean_squared_log_error(y_train, np.maximum(oof_preds_xgb, 0)))\nscore_lgb = np.sqrt(mean_squared_log_error(y_train, np.maximum(oof_preds_lgb, 0)))\nscore_cat = np.sqrt(mean_squared_log_error(y_train, np.maximum(oof_preds_cat, 0)))\nscore_avg = np.sqrt(mean_squared_log_error(y_train, np.maximum(oof_preds_avg, 0)))\n\n# Display scores\nprint(f\"RMSLE XGB: {score_xgb:.4f}\")\nprint(f\"RMSLE LGB: {score_lgb:.4f}\")\nprint(f\"RMSLE CatBoost: {score_cat:.4f}\")\nprint(f\"RMSLE Average: {score_avg:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:55:33.555966Z","iopub.execute_input":"2024-12-01T10:55:33.556791Z","iopub.status.idle":"2024-12-01T10:59:12.833664Z","shell.execute_reply.started":"2024-12-01T10:55:33.556752Z","shell.execute_reply":"2024-12-01T10:59:12.832837Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='c6'></a>\n# <div style=\"text-align:center; border-radius:15px 15px; padding:15px; color:#333333; margin:0; ; padding:15px; font-size:100%; font:'Verdana'; background-color:#F5F5F5;border: 1px; overflow:hidden\"><b>6. Results and submission</b>\n</div>","metadata":{}},{"cell_type":"code","source":"preds_xgb = model_xgb.predict(X_test_encoded)\npreds_lgb = model_lgb.predict(X_test_encoded)\npreds_cat = model_cat.predict(X_test_encoded)\n\n# Prepare submission DataFrame\nsubmit = pd.DataFrame({\n    'id': test_ids, \n    'Premium Amount': preds_lgb.flatten() \n})\n\nsubmit.to_csv(\"../working/submission.csv\", index=False)\n\nprint(submit)\nprint(submit['Premium Amount'].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:06:55.993472Z","iopub.execute_input":"2024-12-01T11:06:55.993825Z","iopub.status.idle":"2024-12-01T11:07:05.280056Z","shell.execute_reply.started":"2024-12-01T11:06:55.993795Z","shell.execute_reply":"2024-12-01T11:07:05.278929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.histplot(submit['Premium Amount'], kde=True, bins=30)\nplt.title(\"Distribution of Predictions\")\nplt.xlabel(\"Prediction\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:07:38.719018Z","iopub.execute_input":"2024-12-01T11:07:38.719781Z","iopub.status.idle":"2024-12-01T11:07:39.424699Z","shell.execute_reply.started":"2024-12-01T11:07:38.719745Z","shell.execute_reply":"2024-12-01T11:07:39.423855Z"}},"outputs":[],"execution_count":null}]}