{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# REGRESSION ANALYSIS","metadata":{}},{"cell_type":"code","source":"!pip install scikit-learn==1.4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:58:52.908863Z","iopub.execute_input":"2025-02-05T17:58:52.909245Z","iopub.status.idle":"2025-02-05T17:59:09.554038Z","shell.execute_reply.started":"2025-02-05T17:58:52.909206Z","shell.execute_reply":"2025-02-05T17:59:09.552717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport math\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_log_error\nfrom time import process_time\nfrom sklearn.impute import SimpleImputer\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:09.556768Z","iopub.execute_input":"2025-02-05T17:59:09.557823Z","iopub.status.idle":"2025-02-05T17:59:11.008094Z","shell.execute_reply.started":"2025-02-05T17:59:09.557764Z","shell.execute_reply":"2025-02-05T17:59:11.00696Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The dataset at our disposal reprsented various customer characteristics and policy details. from insurance company . The goal of this project is to be able to predict the premium Amount based on those characteristics by regression analysis. \n\nWe'll divide this project in differents parts :\n\n- Exploratory Data Analysis\n- Data cleaning\n- Data modeling\n- Data evaluation\n\n**Features description**\n\n    - Age: Age of the insured individual (Numerical)\n    - Gender: Gender of the insured individual (Categorical: Male, Female)\n    - Annual Income: Annual income of the insured individual (Numerical, skewed)\n    - Marital Status: Marital status of the insured individual (Categorical: Single, Married, Divorced)\n    - Number of Dependents: Number of dependents (Numerical, with missing values)\n    - Education Level: Highest education level attained (Categorical: High School, Bachelor's, Master's, PhD)\n    - Occupation: Occupation of the insured individual (Categorical: Employed, Self-Employed, Unemployed)\n    - Health Score: A score representing the health status (Numerical, skewed)\n    - Location: Type of location (Categorical: Urban, Suburban, Rural)\n    - Policy Type: Type of insurance policy (Categorical: Basic, Comprehensive, Premium)\n    - Previous Claims: Number of previous claims made (Numerical, with outliers)\n    - Vehicle Age: Age of the vehicle insured (Numerical)\n    - Credit Score: Credit score of the insured individual (Numerical, with missing values)\n    - Insurance Duration: Duration of the insurance policy (Numerical, in years)\n    - Premium Amount: Target variable representing the insurance premium amount (Numerical, skewed)\n    - Policy Start Date: Start date of the insurance policy (Text, improperly formatted)\n    - Customer Feedback: Short feedback comments from customers (Text)\n    - Smoking Status: Smoking status of the insured individual (Categorical: Yes, No)\n    - Exercise Frequency: Frequency of exercise (Categorical: Daily, Weekly, Monthly, Rarely)\n    - Property Type: Type of property owned (Categorical: House, Apartment, Condo)\n","metadata":{}},{"cell_type":"code","source":"dataset = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv',index_col='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:11.009318Z","iopub.execute_input":"2025-02-05T17:59:11.009824Z","iopub.status.idle":"2025-02-05T17:59:17.418711Z","shell.execute_reply.started":"2025-02-05T17:59:11.009786Z","shell.execute_reply":"2025-02-05T17:59:17.41744Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Description Analysis","metadata":{}},{"cell_type":"markdown","source":"Let's get a look to our dataset","metadata":{}},{"cell_type":"code","source":"dataset.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:17.421052Z","iopub.execute_input":"2025-02-05T17:59:17.421398Z","iopub.status.idle":"2025-02-05T17:59:17.450903Z","shell.execute_reply.started":"2025-02-05T17:59:17.421363Z","shell.execute_reply":"2025-02-05T17:59:17.449676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:17.452309Z","iopub.execute_input":"2025-02-05T17:59:17.452765Z","iopub.status.idle":"2025-02-05T17:59:18.121416Z","shell.execute_reply.started":"2025-02-05T17:59:17.452716Z","shell.execute_reply":"2025-02-05T17:59:18.120131Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From a frist look , we have differents type of variables in our dataset and there are also missing values. \nWe'll start our analysis by correcting the type of some columns such as the date columns ","metadata":{}},{"cell_type":"code","source":"dataset['Policy Start Date']=pd.to_datetime(dataset['Policy Start Date'])\ndataset.info()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:18.123124Z","iopub.execute_input":"2025-02-05T17:59:18.123562Z","iopub.status.idle":"2025-02-05T17:59:19.14596Z","shell.execute_reply.started":"2025-02-05T17:59:18.123501Z","shell.execute_reply":"2025-02-05T17:59:19.144922Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We'll check the descriptive statitics of our dataset to get a better understanding ","metadata":{"execution":{"iopub.status.busy":"2024-12-11T10:52:37.903557Z","iopub.execute_input":"2024-12-11T10:52:37.904097Z","iopub.status.idle":"2024-12-11T10:52:38.562967Z","shell.execute_reply.started":"2024-12-11T10:52:37.904041Z","shell.execute_reply":"2024-12-11T10:52:38.561802Z"},"jupyter":{"outputs_hidden":true},"collapsed":true}},{"cell_type":"code","source":"dataset.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:19.147118Z","iopub.execute_input":"2025-02-05T17:59:19.147406Z","iopub.status.idle":"2025-02-05T17:59:19.889234Z","shell.execute_reply.started":"2025-02-05T17:59:19.147376Z","shell.execute_reply":"2025-02-05T17:59:19.888127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset.describe(include=[object])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:19.890321Z","iopub.execute_input":"2025-02-05T17:59:19.89064Z","iopub.status.idle":"2025-02-05T17:59:21.87678Z","shell.execute_reply.started":"2025-02-05T17:59:19.890575Z","shell.execute_reply":"2025-02-05T17:59:21.875705Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Explanatory Data Analysis","metadata":{}},{"cell_type":"markdown","source":"We'll dive deep in our analysis by doing some visualization of our dataset ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(14,7))\ndataset['Age'].plot.hist(bins=50)\nplt.title('Distributon of the Age')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:21.877956Z","iopub.execute_input":"2025-02-05T17:59:21.878246Z","iopub.status.idle":"2025-02-05T17:59:22.339194Z","shell.execute_reply.started":"2025-02-05T17:59:21.878217Z","shell.execute_reply":"2025-02-05T17:59:22.338046Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We seems to have the dataset well distibuted aound the differents ages","metadata":{"execution":{"iopub.status.busy":"2024-12-11T11:15:44.094909Z","iopub.execute_input":"2024-12-11T11:15:44.095277Z","iopub.status.idle":"2024-12-11T11:15:44.102738Z","shell.execute_reply.started":"2024-12-11T11:15:44.09524Z","shell.execute_reply":"2024-12-11T11:15:44.101219Z"}}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore', category=FutureWarning)\n\nsns.histplot(x=dataset['Annual Income'])\nplt.title('Distribution of Income')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:22.343941Z","iopub.execute_input":"2025-02-05T17:59:22.344299Z","iopub.status.idle":"2025-02-05T17:59:23.336083Z","shell.execute_reply.started":"2025-02-05T17:59:22.344265Z","shell.execute_reply":"2025-02-05T17:59:23.334903Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's check the distribution of the premium amount ","metadata":{}},{"cell_type":"code","source":"sns.histplot(x=dataset['Premium Amount'])\nplt.title('Distribution of Amount')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:23.337487Z","iopub.execute_input":"2025-02-05T17:59:23.337865Z","iopub.status.idle":"2025-02-05T17:59:24.254877Z","shell.execute_reply.started":"2025-02-05T17:59:23.33783Z","shell.execute_reply":"2025-02-05T17:59:24.253753Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Analysis of Categorical Variables**","metadata":{}},{"cell_type":"code","source":"def plot_categorical_variables(df, max_plots_per_row=3, figsize=(15, 10)):\n    \"\"\"\n    Generate pie charts for all categorical variables in a grid layout.\n    \n    Parameters:\n    -----------\n    df : pandas.DataFrame\n        Input DataFrame to analyze\n    max_plots_per_row : int, optional (default=3)\n        Maximum number of plots to display in a single row\n    figsize : tuple, optional (default=(15, 10))\n        Size of the entire figure\n    \n    Returns:\n    --------\n    matplotlib.figure.Figure\n        A figure containing pie charts for all categorical variables\n    \"\"\"\n\n    plt.close('all')\n    # Identify categorical columns\n    categorical_columns = df.select_dtypes(include=['object', 'category']).columns\n    \n    # Calculate number of plots and grid dimensions\n    num_plots = len(categorical_columns)\n    num_rows = math.ceil(num_plots / max_plots_per_row)\n    # Create figure and axes\n    fig, axes = plt.subplots(num_rows, min(max_plots_per_row, num_plots), \n                             figsize=figsize, squeeze=False)\n    \n    # Flatten axes for easier indexing\n    axes = axes.flatten()\n    \n    # Generate pie charts for each categorical variable\n    for i, column in enumerate(categorical_columns):\n        # Get value counts and sort in descending order\n        value_counts = df[column].value_counts()\n        # Create pie chart on the corresponding axis\n        axes[i].pie(value_counts.values, \n                    labels=value_counts.index, \n                    autopct='%1.1f%%',\n                    startangle=90,\n                    wedgeprops={'edgecolor': 'white', 'linewidth': 1},\n                    colors=plt.cm.Pastel1.colors)\n        \n        axes[i].set_title(f'Distribution of {column}', fontsize=12, fontweight='bold')\n        \n    # Remove extra subplots if any\n    for i in range(len(categorical_columns), len(axes)):\n        fig.delaxes(axes[i])\n    \n    # Adjust layout\n    plt.tight_layout()\n\n    plt.show()\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:24.256822Z","iopub.execute_input":"2025-02-05T17:59:24.257245Z","iopub.status.idle":"2025-02-05T17:59:24.26644Z","shell.execute_reply.started":"2025-02-05T17:59:24.257206Z","shell.execute_reply":"2025-02-05T17:59:24.265272Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_categorical_variables(dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:24.267951Z","iopub.execute_input":"2025-02-05T17:59:24.268394Z","iopub.status.idle":"2025-02-05T17:59:26.699031Z","shell.execute_reply.started":"2025-02-05T17:59:24.268342Z","shell.execute_reply":"2025-02-05T17:59:26.697945Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"In those differents variables we can se that the diffents type are equally represented .","metadata":{"execution":{"iopub.status.busy":"2024-12-11T11:44:24.424937Z","iopub.execute_input":"2024-12-11T11:44:24.426176Z","iopub.status.idle":"2024-12-11T11:44:24.43293Z","shell.execute_reply.started":"2024-12-11T11:44:24.426127Z","shell.execute_reply":"2024-12-11T11:44:24.43145Z"}}},{"cell_type":"markdown","source":"### Bivariate Analysis \n\nLet's analyse the relationship between the varaible and the target variable","metadata":{"execution":{"iopub.status.busy":"2024-12-11T11:45:19.640332Z","iopub.execute_input":"2024-12-11T11:45:19.640711Z","iopub.status.idle":"2024-12-11T11:45:19.645693Z","shell.execute_reply.started":"2024-12-11T11:45:19.640679Z","shell.execute_reply":"2024-12-11T11:45:19.64458Z"}}},{"cell_type":"code","source":"#dataset.replace([np.inf, -np.inf], np.nan, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:26.700413Z","iopub.execute_input":"2025-02-05T17:59:26.70077Z","iopub.status.idle":"2025-02-05T17:59:26.705575Z","shell.execute_reply.started":"2025-02-05T17:59:26.700736Z","shell.execute_reply":"2025-02-05T17:59:26.704497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n##plt.figure(figsize=(14,7))\n##sns.pairplot(dataset_copy)\n##plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:26.707054Z","iopub.execute_input":"2025-02-05T17:59:26.707465Z","iopub.status.idle":"2025-02-05T17:59:26.71919Z","shell.execute_reply.started":"2025-02-05T17:59:26.707418Z","shell.execute_reply":"2025-02-05T17:59:26.718081Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"inf_columns = [col for col in dataset.select_dtypes(include=['number']).columns \n               if np.isinf(dataset[col]).any()]\nprint(\"Columns with infinite values:\", inf_columns)","metadata":{"execution":{"iopub.status.busy":"2024-12-17T13:03:42.745332Z","iopub.execute_input":"2024-12-17T13:03:42.745739Z","iopub.status.idle":"2024-12-17T13:03:42.807468Z","shell.execute_reply.started":"2024-12-17T13:03:42.745705Z","shell.execute_reply":"2024-12-17T13:03:42.806105Z"}}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore', category=FutureWarning)\n\nplt.figure(figsize=(14,7))\nsns.heatmap(dataset.dropna().corr(method='pearson',numeric_only=True),annot=True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:26.720506Z","iopub.execute_input":"2025-02-05T17:59:26.720919Z","iopub.status.idle":"2025-02-05T17:59:27.903795Z","shell.execute_reply.started":"2025-02-05T17:59:26.720885Z","shell.execute_reply":"2025-02-05T17:59:27.902737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_copy = dataset.copy()\ndataset_copy['year']=dataset_copy['Policy Start Date'].dt.year","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T18:16:05.943752Z","iopub.execute_input":"2025-02-05T18:16:05.94415Z","iopub.status.idle":"2025-02-05T18:16:06.51376Z","shell.execute_reply.started":"2025-02-05T18:16:05.944115Z","shell.execute_reply":"2025-02-05T18:16:06.51277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.lineplot(data=dataset_copy,x='year',y='Premium Amount')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T18:16:07.545284Z","iopub.execute_input":"2025-02-05T18:16:07.545693Z","iopub.status.idle":"2025-02-05T18:16:18.229265Z","shell.execute_reply.started":"2025-02-05T18:16:07.545653Z","shell.execute_reply":"2025-02-05T18:16:18.228111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_copy['after_covid']=['yes' if x>=2020 else 'no' for x in dataset_copy['year']]\ndataset_copy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T18:16:18.231515Z","iopub.execute_input":"2025-02-05T18:16:18.231985Z","iopub.status.idle":"2025-02-05T18:16:19.572469Z","shell.execute_reply.started":"2025-02-05T18:16:18.231936Z","shell.execute_reply":"2025-02-05T18:16:19.571428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.boxplot(x=dataset_copy['Premium Amount'],y=dataset_copy['Location'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T18:16:19.573754Z","iopub.execute_input":"2025-02-05T18:16:19.574089Z","iopub.status.idle":"2025-02-05T18:16:20.720008Z","shell.execute_reply.started":"2025-02-05T18:16:19.574056Z","shell.execute_reply":"2025-02-05T18:16:20.718933Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Analysing the realation between two categorical variables \n\nH0: There is no relationship between the Occupation and the Education Level \n\nH1 : There is a relationship between the Occupation and the Education Level \n\nThe significan level is 5%","metadata":{}},{"cell_type":"code","source":"contingency = pd.crosstab(dataset['Occupation'], dataset['Education Level'])\ncontingency","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T18:16:20.722663Z","iopub.execute_input":"2025-02-05T18:16:20.723466Z","iopub.status.idle":"2025-02-05T18:16:21.046145Z","shell.execute_reply.started":"2025-02-05T18:16:20.723415Z","shell.execute_reply":"2025-02-05T18:16:21.045001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import scipy.stats as stats\nobservations = np.array([contingency.iloc[0,:],contingency.iloc[1,:],contingency.iloc[2,:]])\nresult = stats.contingency.chi2_contingency(observations, correction=False)\nresult","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T18:16:21.047328Z","iopub.execute_input":"2025-02-05T18:16:21.047679Z","iopub.status.idle":"2025-02-05T18:16:21.05676Z","shell.execute_reply.started":"2025-02-05T18:16:21.047645Z","shell.execute_reply":"2025-02-05T18:16:21.055675Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The p-value is greater thet the signifiance level so we fail to reject the null hypothesis .\nthere seem to have no relationship between the Occupation and the Education Level ","metadata":{"execution":{"iopub.status.busy":"2025-01-24T15:25:06.525094Z","iopub.execute_input":"2025-01-24T15:25:06.525484Z","iopub.status.idle":"2025-01-24T15:25:06.53646Z","shell.execute_reply.started":"2025-01-24T15:25:06.525445Z","shell.execute_reply":"2025-01-24T15:25:06.53396Z"}}},{"cell_type":"code","source":"from statsmodels.stats.outliers_influence import variance_inflation_factor\nX = dataset.iloc[:,1:-1].select_dtypes(include='number').dropna()\nvif = [variance_inflation_factor(X.values, i) for i in range(X.shape[1])]\nvif = zip(X, vif)\nprint(list(vif))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T18:16:21.058318Z","iopub.execute_input":"2025-02-05T18:16:21.058846Z","iopub.status.idle":"2025-02-05T18:16:24.050338Z","shell.execute_reply.started":"2025-02-05T18:16:21.058797Z","shell.execute_reply":"2025-02-05T18:16:24.048247Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We have a strong correlation between Age and Credit Score","metadata":{}},{"cell_type":"markdown","source":"## Data cleaning","metadata":{"execution":{"iopub.status.busy":"2024-12-26T13:01:18.880874Z","iopub.execute_input":"2024-12-26T13:01:18.881257Z","iopub.status.idle":"2024-12-26T13:01:19.083668Z","shell.execute_reply.started":"2024-12-26T13:01:18.881222Z","shell.execute_reply":"2024-12-26T13:01:19.08224Z"}}},{"cell_type":"code","source":"(dataset.isna().sum()/dataset.shape[0])*100","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T18:16:24.052015Z","iopub.execute_input":"2025-02-05T18:16:24.052433Z","iopub.status.idle":"2025-02-05T18:16:24.679481Z","shell.execute_reply.started":"2025-02-05T18:16:24.052387Z","shell.execute_reply":"2025-02-05T18:16:24.678399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install missingno","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T17:59:46.338886Z","iopub.execute_input":"2025-02-05T17:59:46.339167Z","iopub.status.idle":"2025-02-05T17:59:56.542368Z","shell.execute_reply.started":"2025-02-05T17:59:46.339139Z","shell.execute_reply":"2025-02-05T17:59:56.540898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import missingno as msno\nmsno.bar(dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T18:16:24.681011Z","iopub.execute_input":"2025-02-05T18:16:24.681343Z","iopub.status.idle":"2025-02-05T18:16:26.64002Z","shell.execute_reply.started":"2025-02-05T18:16:24.68131Z","shell.execute_reply":"2025-02-05T18:16:26.638958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Feature Engineering ","metadata":{}},{"cell_type":"code","source":"dataset_copy['Annual Income']=np.log(dataset_copy['Annual Income'])\ndataset_copy['Premium Amount']=np.log(dataset_copy['Premium Amount'])\n\ndataset_copy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T18:16:26.641567Z","iopub.execute_input":"2025-02-05T18:16:26.642006Z","iopub.status.idle":"2025-02-05T18:16:27.889759Z","shell.execute_reply.started":"2025-02-05T18:16:26.641959Z","shell.execute_reply":"2025-02-05T18:16:27.888719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ny = dataset_copy['Premium Amount']\ntrain_dataset = dataset_copy.loc[:,~dataset_copy.columns.isin(['Premium Amount','Policy Start Date','id','year'])]\ntrain_X, test_X, train_y, test_y = train_test_split(train_dataset,y , test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:02:40.669326Z","iopub.execute_input":"2025-02-05T19:02:40.670731Z","iopub.status.idle":"2025-02-05T19:02:42.563573Z","shell.execute_reply.started":"2025-02-05T19:02:40.67068Z","shell.execute_reply":"2025-02-05T19:02:42.562483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_columns = [cname for cname in train_dataset.columns if train_dataset[cname].dtype in ['int64', 'float64']]\ncategorical_columns =[cname for cname in train_dataset.columns if train_dataset[cname].dtype=='object']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:02:45.655118Z","iopub.execute_input":"2025-02-05T19:02:45.655511Z","iopub.status.idle":"2025-02-05T19:02:45.662313Z","shell.execute_reply.started":"2025-02-05T19:02:45.655474Z","shell.execute_reply":"2025-02-05T19:02:45.661108Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We'll create a pipeline to handle the transformation of our datset","metadata":{}},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline \nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import FunctionTransformer\n\nlog_transformer = FunctionTransformer(np.log, validate=True)\n# Prepocessing numerical data\nnumerical_transformer = Pipeline(steps=[\n    ('imputer',SimpleImputer(strategy='constant')),\n    ('num_preprocess',MinMaxScaler())])\n\n\n\n#Prepocessing categorical data\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('encoder',OneHotEncoder(handle_unknown='ignore'))])\n\n\npreprocessor = ColumnTransformer(transformers=[\n    ('num',numerical_transformer,numerical_columns),\n    ('cat',categorical_transformer,categorical_columns)])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:04:04.448637Z","iopub.execute_input":"2025-02-05T19:04:04.44907Z","iopub.status.idle":"2025-02-05T19:04:04.457024Z","shell.execute_reply.started":"2025-02-05T19:04:04.44903Z","shell.execute_reply":"2025-02-05T19:04:04.455682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n\npca = PCA()\n\n\npca_pipeline = Pipeline(steps=[('preprocessor',preprocessor),\n                               ('pca',pca)])\n\npca_pipeline.fit(train_X)\n\ncumsum = np.cumsum(pca_pipeline.named_steps['pca'].explained_variance_ratio_)\nd = np.argmax(cumsum >= 0.95) + 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T18:40:08.780345Z","iopub.execute_input":"2025-02-05T18:40:08.780783Z","iopub.status.idle":"2025-02-05T18:40:17.417826Z","shell.execute_reply.started":"2025-02-05T18:40:08.780727Z","shell.execute_reply":"2025-02-05T18:40:17.416595Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(cumsum)\nplt.axhline(y=0.95)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T18:40:17.419502Z","iopub.execute_input":"2025-02-05T18:40:17.420108Z","iopub.status.idle":"2025-02-05T18:40:17.618305Z","shell.execute_reply.started":"2025-02-05T18:40:17.420066Z","shell.execute_reply":"2025-02-05T18:40:17.617023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pca = PCA(n_components = 21)\n\npca_pipeline = Pipeline(steps=[('preprocessor',preprocessor),\n                               ('pca',pca)])\n\nX2D = pca_pipeline.fit_transform(train_X)\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:04:55.89234Z","iopub.execute_input":"2025-02-05T19:04:55.89281Z","iopub.status.idle":"2025-02-05T19:05:09.807066Z","shell.execute_reply.started":"2025-02-05T19:04:55.89276Z","shell.execute_reply":"2025-02-05T19:05:09.805798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost\nfrom sklearn.feature_extraction import DictVectorizer\n\nxgb_reg = xgboost.XGBRegressor(n_estimators=100,random_state=42,early_stopping_rounds=5)\n\n\nstart = process_time()\nmodel_xgb_pca= Pipeline(steps=[\n                        ('model',xgb_reg)])\n\n                                       \nmodel_xgb_pca.fit(X2D, train_y,model__eval_set=[(pca_pipeline.fit_transform(test_X), test_y)])\nend = process_time()\n\ntraining_time = end-start\n\nstart = process_time()\ny_pred_pca = model_xgb_pca.predict(pca_pipeline.fit_transform(test_X))\nend = process_time()\n\npredicting_time = end-start\n\nmsle_pca = mean_squared_log_error(test_y,y_pred_pca)\n\nprint(f' - Root Mean Squared Log Error: {np.exp(np.sqrt(msle_pca))}')\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:06:17.342604Z","iopub.execute_input":"2025-02-05T19:06:17.34301Z","iopub.status.idle":"2025-02-05T19:06:27.478356Z","shell.execute_reply.started":"2025-02-05T19:06:17.342974Z","shell.execute_reply":"2025-02-05T19:06:27.476947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost\nfrom sklearn.feature_extraction import DictVectorizer\n\nxgb_reg = xgboost.XGBRegressor(n_estimators=100,random_state=42,early_stopping_rounds=5)\n\n\nstart = process_time()\nmodel_xgb = Pipeline(steps=[('prepocessor',prepocessor),\n                        ('model',xgb_reg)])\n\n                                       \nmodel_xgb.fit(train_X, train_y,model__eval_set=[(preprocessor.fit_transform(test_X), test_y)])\nend = process_time()\n\ntraining_time = end-start\n\nstart = process_time()\ny_pred = model_xgb.predict(test_X)\nend = process_time()\n\npredicting_time = end-start\n\nmsle = mean_squared_log_error(test_y,y_pred)\n\nprint(f' - Root Mean Squared Log Error: {np.exp(np.sqrt(msle))}')\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T18:55:30.938329Z","iopub.execute_input":"2025-02-05T18:55:30.938813Z","iopub.status.idle":"2025-02-05T18:55:42.479967Z","shell.execute_reply.started":"2025-02-05T18:55:30.938771Z","shell.execute_reply":"2025-02-05T18:55:42.478777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Cross Validation","metadata":{}},{"cell_type":"markdown","source":"from sklearn.model_selection import cross_val_score\n\n Multiply by -1 since sklearn calculates *negative* MAE\nxgb_reg = xgboost.XGBRegressor(n_estimators=100,random_state=42,early_stopping_rounds=5)\n\n##('ohe_onestep', DictVectorizer(sparse=False))\nstart = process_time()\nmodel_xgb_cv = Pipeline(steps=[('prepocessor',prepocessor),\n                        ('model',xgb_reg)])\nscores = cross_val_score(model_xgb_cv, train_X, train_y, cv=5,\n                             scoring='neg_mean_squared_log_error',fit_params={'model__eval_set':[(prepocessor.fit_transform(test_X), test_y)]})\n\nprint(\"MAE scores:\\n\", scores)\nnp.exp(np.sqrt(-np.mean(scores)))","metadata":{"execution":{"iopub.status.busy":"2025-01-29T13:38:07.94234Z","iopub.execute_input":"2025-01-29T13:38:07.942796Z","iopub.status.idle":"2025-01-29T13:38:49.796005Z","shell.execute_reply.started":"2025-01-29T13:38:07.942747Z","shell.execute_reply":"2025-01-29T13:38:49.794822Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}}},{"cell_type":"markdown","source":"model_collections={\n    'linear Regression': LinearRegression(),\n    'Decision Tree': DecisionTreeRegressor(),\n    'Random Forest': RandomForestRegressor(n_estimators=50, random_state=0)\n}\n\nresults = pd.DataFrame(columns=['Model', 'MSLE','Training Time','Prediction time','CV'])\nfor model_name , model in model_collections.items():\n    print('Starting',model_name)\n    start = process_time()\n    pipeline = Pipeline(steps=[('prepocessor',prepocessor),\n                        ('model',model)])\n\n                                       \n    pipeline.fit(train_X, train_y)\n    end = process_time()\n\n    training_time = end-start\n\n    start = process_time()\n    y_pred = pipeline.predict(test_X)\n    end = process_time()\n\n    predicting_time = end-start\n\n    msle = mean_squared_log_error(test_y,y_pred)\n    scores = cross_val_score(pipeline, train_X, train_y,\n                              cv=5,\n                              scoring='neg_root_mean_squared_log_error')\n\n    \n    new_row = pd.DataFrame({'Model': [model_name], 'MSLE': [np.sqrt(msle)],'Training':training_time,'Prediction':predicting_time})\n    results = pd.concat([results, new_row], ignore_index=True)\n\n   \n    print(f'{model_name} -  Mean Squared Error: {np.sqrt(msle)}- Root for cv : {scores.mean()}\\n')\n\n\nresults\n    ","metadata":{}},{"cell_type":"code","source":"\n# Assuming 'pipeline' has already been fitted\nxgb_model = model_xgb.named_steps['model']\n\n# Get the booster (XGBoost model) and feature importance\nbooster = xgb_model.get_booster()\n\n# Get feature importance based on weight (default) - how many times a feature is used in splits\nimportance = booster.get_score(importance_type='weight')\n\n# Alternatively, you can get the importance based on:\n# importance_type='weight'  # Number of times the feature is used in a split\n# importance_type='gain'    # The average gain (contribution) of a feature to the model\n# importance_type='cover'   # The average coverage (number of samples affected) of a feature\n\n# Convert to DataFrame for better visualization\nimportance_df = pd.DataFrame(list(importance.items()), columns=['Feature', 'Importance'])\nimportance_df = importance_df.sort_values(by='Importance', ascending=False)\n\nfeature_names = model_xgb.named_steps['prepocessor'].get_feature_names_out()\n\nfor i in importance_df['Feature']:\n    for index , name in enumerate(feature_names):\n        if int(i.split('f')[1]) == index:\n            importance_df.loc[importance_df['Feature'] == i, 'features_name'] = name\n            break\n\n\nplt.figure(figsize=(14, 17))\nplt.barh(importance_df['features_name'], importance_df['Importance'], color='skyblue')\nplt.xlabel('Feature Importance')\nplt.title('XGBoost Feature Importance')\nplt.gca().invert_yaxis()  # Reverse the order for better readability\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T13:55:23.389861Z","iopub.execute_input":"2025-01-29T13:55:23.390212Z","iopub.status.idle":"2025-01-29T13:55:23.987582Z","shell.execute_reply.started":"2025-01-29T13:55:23.390178Z","shell.execute_reply":"2025-01-29T13:55:23.986525Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submission\n","metadata":{"execution":{"iopub.status.busy":"2024-12-26T16:59:56.852616Z","iopub.execute_input":"2024-12-26T16:59:56.853042Z","iopub.status.idle":"2024-12-26T16:59:56.860853Z","shell.execute_reply.started":"2024-12-26T16:59:56.853008Z","shell.execute_reply":"2024-12-26T16:59:56.859371Z"}}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:07:51.787346Z","iopub.execute_input":"2025-02-05T19:07:51.787794Z","iopub.status.idle":"2025-02-05T19:07:56.219178Z","shell.execute_reply.started":"2025-02-05T19:07:51.787745Z","shell.execute_reply":"2025-02-05T19:07:56.217996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def exp_log_mean(x):\n    return np.expm1(np.log1p(x).mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T13:55:26.679257Z","iopub.execute_input":"2025-01-29T13:55:26.679711Z","iopub.status.idle":"2025-01-29T13:55:26.685254Z","shell.execute_reply.started":"2025-01-29T13:55:26.679662Z","shell.execute_reply":"2025-01-29T13:55:26.684057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test['Annual Income']=np.log(test['Annual Income'])\n\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:07:59.590799Z","iopub.execute_input":"2025-02-05T19:07:59.591206Z","iopub.status.idle":"2025-02-05T19:07:59.626358Z","shell.execute_reply.started":"2025-02-05T19:07:59.591168Z","shell.execute_reply":"2025-02-05T19:07:59.625174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test['Policy Start Date']=pd.to_datetime(test['Policy Start Date'])\ntest['year']=test['Policy Start Date'].dt.year\ntest['after_covid']=['yes' if x>=2020 else 'no' for x in test['year']]\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:08:01.841723Z","iopub.execute_input":"2025-02-05T19:08:01.842124Z","iopub.status.idle":"2025-02-05T19:08:03.058101Z","shell.execute_reply.started":"2025-02-05T19:08:01.842087Z","shell.execute_reply":"2025-02-05T19:08:03.056943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_to = test.loc[:,~test.columns.isin(['Policy Start Date','year','id'])]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:11:47.86965Z","iopub.execute_input":"2025-02-05T19:11:47.870403Z","iopub.status.idle":"2025-02-05T19:11:48.052072Z","shell.execute_reply.started":"2025-02-05T19:11:47.870364Z","shell.execute_reply":"2025-02-05T19:11:48.050887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_to","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:11:51.226606Z","iopub.execute_input":"2025-02-05T19:11:51.227076Z","iopub.status.idle":"2025-02-05T19:11:51.25526Z","shell.execute_reply.started":"2025-02-05T19:11:51.227032Z","shell.execute_reply":"2025-02-05T19:11:51.253908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pca_test= pca_pipeline.fit_transform(test_to)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:11:52.754761Z","iopub.execute_input":"2025-02-05T19:11:52.755155Z","iopub.status.idle":"2025-02-05T19:12:04.255219Z","shell.execute_reply.started":"2025-02-05T19:11:52.75512Z","shell.execute_reply":"2025-02-05T19:12:04.254109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predict = model_xgb_pca.predict(pca_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:12:37.763643Z","iopub.execute_input":"2025-02-05T19:12:37.764454Z","iopub.status.idle":"2025-02-05T19:12:37.830459Z","shell.execute_reply.started":"2025-02-05T19:12:37.764407Z","shell.execute_reply":"2025-02-05T19:12:37.829665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.exp(test_predict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:12:49.651042Z","iopub.execute_input":"2025-02-05T19:12:49.651452Z","iopub.status.idle":"2025-02-05T19:12:49.659408Z","shell.execute_reply.started":"2025-02-05T19:12:49.651414Z","shell.execute_reply":"2025-02-05T19:12:49.658425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({'id':test.id,'Premium Amount':np.exp(test_predict)})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:12:52.394161Z","iopub.execute_input":"2025-02-05T19:12:52.394542Z","iopub.status.idle":"2025-02-05T19:12:52.403493Z","shell.execute_reply.started":"2025-02-05T19:12:52.394506Z","shell.execute_reply":"2025-02-05T19:12:52.402368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-29T13:55:31.222214Z","iopub.execute_input":"2025-01-29T13:55:31.222648Z","iopub.status.idle":"2025-01-29T13:55:31.241498Z","shell.execute_reply.started":"2025-01-29T13:55:31.222583Z","shell.execute_reply":"2025-01-29T13:55:31.240501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-05T19:13:01.813833Z","iopub.execute_input":"2025-02-05T19:13:01.814234Z","iopub.status.idle":"2025-02-05T19:13:03.002559Z","shell.execute_reply.started":"2025-02-05T19:13:01.814194Z","shell.execute_reply":"2025-02-05T19:13:03.001399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}