{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Setting Up\r\nImporting essential libraries and loading data to see data overview.","metadata":{}},{"cell_type":"code","source":"#importing libraires\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport random \nimport missingno as msno\nfrom scipy.stats import shapiro\n\n\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:36:56.42751Z","iopub.execute_input":"2024-12-31T06:36:56.427968Z","iopub.status.idle":"2024-12-31T06:36:59.30929Z","shell.execute_reply.started":"2024-12-31T06:36:56.427878Z","shell.execute_reply":"2024-12-31T06:36:59.308103Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Loading available dataset - Train and Test**","metadata":{}},{"cell_type":"code","source":"#load and check test.csv\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:36:59.312029Z","iopub.execute_input":"2024-12-31T06:36:59.312615Z","iopub.status.idle":"2024-12-31T06:37:03.787271Z","shell.execute_reply.started":"2024-12-31T06:36:59.312565Z","shell.execute_reply":"2024-12-31T06:37:03.78629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#load and check train.csv\ntrain = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:03.788669Z","iopub.execute_input":"2024-12-31T06:37:03.789038Z","iopub.status.idle":"2024-12-31T06:37:10.45519Z","shell.execute_reply.started":"2024-12-31T06:37:03.789005Z","shell.execute_reply":"2024-12-31T06:37:10.453938Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"concat train and test. to perform preprocessing: missing data and label encode","metadata":{}},{"cell_type":"code","source":"#concatting train and test set into \"df\" set for simple parsing and preprocessing \ndf = pd.concat([train, test])\nprint(\"train shape:\", train.shape)\nprint(\"test shape:\", test.shape)\nprint(\"df shape:\", df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:10.456747Z","iopub.execute_input":"2024-12-31T06:37:10.457203Z","iopub.status.idle":"2024-12-31T06:37:10.876913Z","shell.execute_reply.started":"2024-12-31T06:37:10.457156Z","shell.execute_reply":"2024-12-31T06:37:10.875777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#test has a missing column, check to see which \nmissing_cols = set(train.columns) - set(test.columns)\nfor col in missing_cols:\n    print(f\"{col} is missing\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:10.878173Z","iopub.execute_input":"2024-12-31T06:37:10.878483Z","iopub.status.idle":"2024-12-31T06:37:10.88544Z","shell.execute_reply.started":"2024-12-31T06:37:10.878453Z","shell.execute_reply":"2024-12-31T06:37:10.884019Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Premium Amount in test dataset will be the predicted target, thus is missing in test set. ","metadata":{}},{"cell_type":"code","source":"#using df for parsing \ndf.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:10.889468Z","iopub.execute_input":"2024-12-31T06:37:10.889975Z","iopub.status.idle":"2024-12-31T06:37:10.921329Z","shell.execute_reply.started":"2024-12-31T06:37:10.889907Z","shell.execute_reply":"2024-12-31T06:37:10.920135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import missingno as msno\nmsno.matrix(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:10.922939Z","iopub.execute_input":"2024-12-31T06:37:10.923407Z","iopub.status.idle":"2024-12-31T06:37:24.600483Z","shell.execute_reply.started":"2024-12-31T06:37:10.923357Z","shell.execute_reply":"2024-12-31T06:37:24.599238Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Missing data above is likely random, except for Premium Amount. Finding out more about this","metadata":{}},{"cell_type":"code","source":"missing_column = set(train.columns) - set(test.columns)\nfor col in missing_column:\n    print(f\"{col} is missing\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:24.602455Z","iopub.execute_input":"2024-12-31T06:37:24.602885Z","iopub.status.idle":"2024-12-31T06:37:24.609893Z","shell.execute_reply.started":"2024-12-31T06:37:24.602837Z","shell.execute_reply":"2024-12-31T06:37:24.608816Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Check Data Overview","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:24.611372Z","iopub.execute_input":"2024-12-31T06:37:24.611722Z","iopub.status.idle":"2024-12-31T06:37:24.629993Z","shell.execute_reply.started":"2024-12-31T06:37:24.61169Z","shell.execute_reply":"2024-12-31T06:37:24.62864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe().round(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:24.631327Z","iopub.execute_input":"2024-12-31T06:37:24.63163Z","iopub.status.idle":"2024-12-31T06:37:25.92773Z","shell.execute_reply.started":"2024-12-31T06:37:24.6316Z","shell.execute_reply":"2024-12-31T06:37:25.926358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#checking if data is categorical \ncategory_col = df.select_dtypes(include = 'object').columns.tolist()\nfor col in df[category_col].columns:\n    print(f\"{col} is\", df[col].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:25.929259Z","iopub.execute_input":"2024-12-31T06:37:25.92959Z","iopub.status.idle":"2024-12-31T06:37:28.921111Z","shell.execute_reply.started":"2024-12-31T06:37:25.929556Z","shell.execute_reply":"2024-12-31T06:37:28.919952Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Most object columns are not ordial and continuous with certain few unique, thus columns type likely to be category. **Policy Start DAte** is not object, it should be datetime. The rest of object column astype to **category**.","metadata":{}},{"cell_type":"code","source":"#change Policy Start Date to to_datetime\ndf['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\ndf['Policy Year'] = df['Policy Start Date'].dt.year\ndf['Policy Month'] = df['Policy Start Date'].dt.month\ndf['Policy Day'] = df['Policy Start Date'].dt.day\n\ndf = df.drop(columns = ['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:28.922592Z","iopub.execute_input":"2024-12-31T06:37:28.923048Z","iopub.status.idle":"2024-12-31T06:37:30.290435Z","shell.execute_reply.started":"2024-12-31T06:37:28.923006Z","shell.execute_reply":"2024-12-31T06:37:30.289275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"category_col = [col for col in category_col if col !='Policy Start Date']\ndf[category_col] = df[category_col].astype('category')\ndf.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:30.292131Z","iopub.execute_input":"2024-12-31T06:37:30.292482Z","iopub.status.idle":"2024-12-31T06:37:31.85167Z","shell.execute_reply.started":"2024-12-31T06:37:30.292447Z","shell.execute_reply":"2024-12-31T06:37:31.850462Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"code","source":"df.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:31.852777Z","iopub.execute_input":"2024-12-31T06:37:31.853111Z","iopub.status.idle":"2024-12-31T06:37:31.913058Z","shell.execute_reply.started":"2024-12-31T06:37:31.853079Z","shell.execute_reply":"2024-12-31T06:37:31.911854Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Categorical Columns**","metadata":{}},{"cell_type":"code","source":"#checking category columns value counts\nfor col in df[category_col].columns:\n    if df[col].isna().sum() > 0:\n        plt.figure(figsize = (5,5))\n        sns.countplot(data = df, x = col)\n        plt.title(f\"{col} value counts\")\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:31.914443Z","iopub.execute_input":"2024-12-31T06:37:31.91479Z","iopub.status.idle":"2024-12-31T06:37:32.694182Z","shell.execute_reply.started":"2024-12-31T06:37:31.91475Z","shell.execute_reply":"2024-12-31T06:37:32.693078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#checking sum of missing data in category col.\ncategory_missing = [col for col in category_col if df[col].isna().sum() > 0]\ndf[category_missing].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:32.695825Z","iopub.execute_input":"2024-12-31T06:37:32.696186Z","iopub.status.idle":"2024-12-31T06:37:32.731456Z","shell.execute_reply.started":"2024-12-31T06:37:32.69615Z","shell.execute_reply":"2024-12-31T06:37:32.730417Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Sighting from the above count plot, there is no highest or lowest, while most of the values are almost the same we will be using ffill to fillna. ","metadata":{}},{"cell_type":"code","source":"#using ffill to fillna\ndf[category_missing] = df[category_missing].ffill()\ndf[category_col].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:32.73298Z","iopub.execute_input":"2024-12-31T06:37:32.733417Z","iopub.status.idle":"2024-12-31T06:37:32.786457Z","shell.execute_reply.started":"2024-12-31T06:37:32.733367Z","shell.execute_reply":"2024-12-31T06:37:32.785262Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Numerical Columns**","metadata":{}},{"cell_type":"code","source":"#since we already have category columns, now we going to include numeric data\nnumeric_col = df.select_dtypes(include = 'number')\nnumeric_col = [col for col in numeric_col if col != 'Premium Amount' and col != 'id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:32.792004Z","iopub.execute_input":"2024-12-31T06:37:32.792364Z","iopub.status.idle":"2024-12-31T06:37:32.960201Z","shell.execute_reply.started":"2024-12-31T06:37:32.792331Z","shell.execute_reply":"2024-12-31T06:37:32.959048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_col_df = df[numeric_col]\nnum_col_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:32.961709Z","iopub.execute_input":"2024-12-31T06:37:32.962175Z","iopub.status.idle":"2024-12-31T06:37:33.041019Z","shell.execute_reply.started":"2024-12-31T06:37:32.962128Z","shell.execute_reply":"2024-12-31T06:37:33.03991Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Plot Numeric Columns to check how's data doing ","metadata":{}},{"cell_type":"code","source":"def histbox_plot(df):\n    \n    for i,col in enumerate(df.columns):\n        fig, (ax1, ax2) = plt.subplots(1, 2, figsize = (10,5))\n        sns.histplot(df[col], bins = 'auto', kde = True, ax = ax1)\n        sns.boxplot(x = df[col], ax = ax2)\n        ax1.set_title(f\" {col} Histplot\")\n        ax2.set_title(f\" {col} Boxplot\")\n\n    plt.tight_layout()\n    plt.show()\n\nhistbox_plot(num_col_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:33.042529Z","iopub.execute_input":"2024-12-31T06:37:33.043014Z","iopub.status.idle":"2024-12-31T06:39:12.520498Z","shell.execute_reply.started":"2024-12-31T06:37:33.042943Z","shell.execute_reply":"2024-12-31T06:39:12.519331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"histbox_plot(df[['Premium Amount']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:12.52213Z","iopub.execute_input":"2024-12-31T06:39:12.522588Z","iopub.status.idle":"2024-12-31T06:39:19.349279Z","shell.execute_reply.started":"2024-12-31T06:39:12.522536Z","shell.execute_reply":"2024-12-31T06:39:19.348044Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Perform Correlation prior to cleaning**","metadata":{}},{"cell_type":"code","source":"num_df_col = df.select_dtypes(include = ['number'])\ndf_corr = num_df_col.corr()\n\nplt.figure(figsize = (10,10))\nsns.heatmap(df_corr, \n           annot=True, fmt = '.2f',\n           linewidth = 1, cmap = 'inferno')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:19.350924Z","iopub.execute_input":"2024-12-31T06:39:19.351397Z","iopub.status.idle":"2024-12-31T06:39:21.623727Z","shell.execute_reply.started":"2024-12-31T06:39:19.351348Z","shell.execute_reply":"2024-12-31T06:39:21.622708Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Preprocess numeric data**","metadata":{}},{"cell_type":"code","source":"#Filling missing data\n\n#Age Ffill \ndf['Age'] = df['Age'].ffill()\n\n# Annual Income mode\ndf['Annual Income'] = df['Annual Income'].fillna(df['Annual Income'].mode()[0])\n\n# Number of Dependents ffill\ndf['Number of Dependents'] = df['Number of Dependents'].ffill()\n\n# Health Score mean \ndf['Health Score'] = df['Health Score'].fillna(df['Health Score'].mean())\n\n# Previous Claim mode\ndf['Previous Claims'] = df['Previous Claims'].fillna(df['Previous Claims'].mode()[0])\n\n#Credit Score Median \ndf['Credit Score'] = df['Credit Score'].fillna(df['Credit Score'].median())\n\n#Vehicle Age ffill\ndf['Vehicle Age'] = df['Vehicle Age'].ffill()\n\n#Insurance Duration ffill\ndf['Insurance Duration'] = df['Insurance Duration'].ffill()\n\ndf.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:21.625312Z","iopub.execute_input":"2024-12-31T06:39:21.62627Z","iopub.status.idle":"2024-12-31T06:39:21.969175Z","shell.execute_reply.started":"2024-12-31T06:39:21.626221Z","shell.execute_reply":"2024-12-31T06:39:21.967768Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Standard sCaler or mixmaxscaler. for numeric columns. Label encoding categorical. then use all df for corr. then start to split to train test again \n1. Random Forest Reg:\n   - need binning for age and vehicle age\n   - Label Encode cat\n   - no need scaler\n  \n2. Gboost:\n   - need binning for age and vehicle\n   - -Label Encode cat\n   - need scaler","metadata":{"_kg_hide-input":true,"_kg_hide-output":true}},{"cell_type":"markdown","source":"# Binning Age and Vehicle Age ","metadata":{}},{"cell_type":"code","source":"#Bin age and vehicle age\nage_bins = [0, 18, 30, 45, 60, 100]\nage_labels = ['0-17', '18-29', '30-44', '45-59', '60+']\n\ndf['age_binned'] = pd.cut(df['Age'], bins=age_bins, labels=age_labels, right = False)\n\nvage_bins = [0, 4, 8, 13, 20]\nvage_labels = ['0-3', '4-7', '8-12','13+'] \n\ndf['vehicle_age_binned'] = pd.cut(df['Vehicle Age'], bins=vage_bins, labels=vage_labels, right = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:21.970329Z","iopub.execute_input":"2024-12-31T06:39:21.970639Z","iopub.status.idle":"2024-12-31T06:39:22.084729Z","shell.execute_reply.started":"2024-12-31T06:39:21.970601Z","shell.execute_reply":"2024-12-31T06:39:22.083578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.drop(columns = ['Age', 'Vehicle Age'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:22.08626Z","iopub.execute_input":"2024-12-31T06:39:22.086594Z","iopub.status.idle":"2024-12-31T06:39:22.168543Z","shell.execute_reply.started":"2024-12-31T06:39:22.08656Z","shell.execute_reply":"2024-12-31T06:39:22.167625Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Label Encode Categorical columns","metadata":{}},{"cell_type":"code","source":"#Label Encoding all cat\nfrom sklearn.preprocessing import LabelEncoder\n\ncat_col2 = df.select_dtypes(include = 'category').columns.tolist()\nlabel_encoders = {}\nfor col in cat_col2:\n    le = LabelEncoder()\n    df[col] = le.fit_transform(df[col])  \n    label_encoders[col] = le    \n\nfor col,le in label_encoders.items():\n    mapping = dict(zip(le.classes_, range(len(le.classes_))))\n    print(f\"\\n Mapping for {col}: \\n\")\n    print(mapping)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:22.169758Z","iopub.execute_input":"2024-12-31T06:39:22.170084Z","iopub.status.idle":"2024-12-31T06:39:26.455169Z","shell.execute_reply.started":"2024-12-31T06:39:22.170053Z","shell.execute_reply":"2024-12-31T06:39:26.454146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#convert le encoded columns back to category, and also number of dependents\nto_cat = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location', 'Policy Type', \\\n          'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type', 'age_binned', 'vehicle_age_binned', 'Number of Dependents']\ndf[to_cat] = df[to_cat].astype('category')\ndf.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:26.456587Z","iopub.execute_input":"2024-12-31T06:39:26.456984Z","iopub.status.idle":"2024-12-31T06:39:26.818181Z","shell.execute_reply.started":"2024-12-31T06:39:26.45693Z","shell.execute_reply":"2024-12-31T06:39:26.817032Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Correlation parsing after label encoding","metadata":{}},{"cell_type":"code","source":"df_corr2 = df.corr()\nplt.figure(figsize=(10,10))\nsns.heatmap(df_corr2, \n           annot = True, fmt = '.2f',\n           annot_kws = {\"fontsize\":6},\n           linewidth = 1, cmap = \"coolwarm\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:26.819652Z","iopub.execute_input":"2024-12-31T06:39:26.820035Z","iopub.status.idle":"2024-12-31T06:39:31.816605Z","shell.execute_reply.started":"2024-12-31T06:39:26.820001Z","shell.execute_reply":"2024-12-31T06:39:31.815401Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Still no correlation between variables. \n\n**Splitting df into train and test again.**","metadata":{}},{"cell_type":"code","source":"#splitting train and test set based on na in premium amount \ntrain_df = df[df['Premium Amount'].notna()]\ntest_df = df[df['Premium Amount'].isna()]\ntrain_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:31.818312Z","iopub.execute_input":"2024-12-31T06:39:31.818781Z","iopub.status.idle":"2024-12-31T06:39:32.005079Z","shell.execute_reply.started":"2024-12-31T06:39:31.818723Z","shell.execute_reply":"2024-12-31T06:39:32.003888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:32.006572Z","iopub.execute_input":"2024-12-31T06:39:32.006927Z","iopub.status.idle":"2024-12-31T06:39:32.03793Z","shell.execute_reply.started":"2024-12-31T06:39:32.006895Z","shell.execute_reply":"2024-12-31T06:39:32.036518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"train_df:\", train_df.shape, \"test_df:\", test_df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:32.039302Z","iopub.execute_input":"2024-12-31T06:39:32.039825Z","iopub.status.idle":"2024-12-31T06:39:32.046208Z","shell.execute_reply.started":"2024-12-31T06:39:32.039781Z","shell.execute_reply":"2024-12-31T06:39:32.04504Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Selecting Independent Features for Model\n\nBefore splitting into train and test, we need to decide which features to act as independent features. \n\nAccording to Insurance Information Institute(link), there are several factors that will affect Premium Amount, such as :\n**Bold: (Avail. in column)**\n1. Your driving record\n2. How much you use your car \n3. **Location** \n4. Other factors that affect premium price that can vary from one area or state to another \n5. **Age** \n6. **Gender**\n7. The car you drive \n8. **Your credit**\n9. **The type and amount of auto insurance coverage**","metadata":{}},{"cell_type":"markdown","source":"To double confirm we use Recursive Feature Elimination to check importance of columns.","metadata":{}},{"cell_type":"code","source":"#splitting into X and Y\nX = train_df.drop(columns =['Premium Amount'])\ny =train_df['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:32.047603Z","iopub.execute_input":"2024-12-31T06:39:32.047992Z","iopub.status.idle":"2024-12-31T06:39:32.101344Z","shell.execute_reply.started":"2024-12-31T06:39:32.04793Z","shell.execute_reply":"2024-12-31T06:39:32.100065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.feature_selection import RFE\nfrom sklearn.linear_model import LinearRegression\n\n\nmodel = LinearRegression()\nrfe = RFE(model, n_features_to_select=5)\nrfe.fit(X, y)\n\nselected_features = X.columns[rfe.support_]\nprint(\"Selected Features:\", selected_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:32.1028Z","iopub.execute_input":"2024-12-31T06:39:32.103277Z","iopub.status.idle":"2024-12-31T06:39:51.757399Z","shell.execute_reply.started":"2024-12-31T06:39:32.103229Z","shell.execute_reply":"2024-12-31T06:39:51.756415Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Hmm... There are some discrepancies between the published factors and RFE's outcome. \n\nSetting Hypothesis to decide which set of features to be included. There will be no H0(null hypothesis):\n\n* H1: Published factors -> [['Location', 'age_binned', 'Gender', 'Credit Score', 'Policy Type']]\n* H2: RFE selected -> [['Previous Claims', 'Customer Feedback', 'Policy Year', 'age_binned',\n       'vehicle_age_binned']]\n* H3: All features","metadata":{}},{"cell_type":"code","source":"#evaluate which hypothesis set is the best for model development \nfrom sklearn.model_selection import cross_val_score\n\n# use random classifier to evaluate the three sets of hypothesis then to evaluate it using cross validation. cross validation is the best to evaluate mixed type of data. \ntr_copy = train_df.copy()\nx_h1 = tr_copy[['Location', 'age_binned', 'Gender', 'Credit Score', 'Policy Type']]\nx_h2 = tr_copy[['Previous Claims', 'Customer Feedback', 'Policy Year', 'age_binned', 'vehicle_age_binned']]\nx_h3 = tr_copy.drop(columns = ['Premium Amount'])\n\ny_h = tr_copy['Premium Amount']\n\n\n#Using Cross Validation to evaluate\ndef model_eva(x, y, x_name):\n    scores = cross_val_score(model, x, y, scoring='neg_mean_squared_error', cv=5, n_jobs=-1)\n    print(f\"{x_name} score: {-scores.mean():.2f} ± {scores.std():.2f}\") \n\n# Evaluate each hypothesis set\nmodel_eva(x_h1, y_h, \"x_h1\")\nmodel_eva(x_h2, y_h, \"x_h2\")\nmodel_eva(x_h3, y_h, \"x_h3\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:39:51.758496Z","iopub.execute_input":"2024-12-31T06:39:51.760536Z","iopub.status.idle":"2024-12-31T06:40:02.227853Z","shell.execute_reply.started":"2024-12-31T06:39:51.760486Z","shell.execute_reply":"2024-12-31T06:40:02.226219Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As you can see, x_h1 = ['Location', 'age_binned', 'Gender', 'Credit Score', 'Policy Type']] has the best results. Thus we will be using these independent features. \n\n# Model Development \n\nSplitting Train and test set into X and y. ","metadata":{}},{"cell_type":"code","source":"# splitting X and y for train and test set\nX_train = train_df[['Location', 'age_binned', 'Gender', 'Credit Score', 'Policy Type']]\ny_train = train_df['Premium Amount']\n\nX_test = test_df[['Location', 'age_binned', 'Gender', 'Credit Score', 'Policy Type']]\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:40:02.229832Z","iopub.execute_input":"2024-12-31T06:40:02.23037Z","iopub.status.idle":"2024-12-31T06:40:02.248256Z","shell.execute_reply.started":"2024-12-31T06:40:02.230318Z","shell.execute_reply":"2024-12-31T06:40:02.247125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#install random forest reg\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import GridSearchCV\n#create model object\nreg_mod = RandomForestRegressor(n_estimators = 200, \n                                max_depth = 50,\n                                max_features = 'sqrt',\n                                random_state= 36)\n#fit model \nreg_mod.fit(X_train, y_train)\ny_pred = reg_mod.predict(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:40:39.773684Z","iopub.execute_input":"2024-12-31T06:40:39.774125Z","iopub.status.idle":"2024-12-31T06:46:19.295272Z","shell.execute_reply.started":"2024-12-31T06:40:39.774088Z","shell.execute_reply":"2024-12-31T06:46:19.293663Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission ","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame({'id': test_df['id'], 'Premium Amount': y_pred.round(3)})\nsubmission.to_csv('submission.csv', index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:53:48.980564Z","iopub.execute_input":"2024-12-31T06:53:48.98095Z","iopub.status.idle":"2024-12-31T06:53:50.181999Z","shell.execute_reply.started":"2024-12-31T06:53:48.980917Z","shell.execute_reply":"2024-12-31T06:53:50.180429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:53:54.308825Z","iopub.execute_input":"2024-12-31T06:53:54.310248Z","iopub.status.idle":"2024-12-31T06:53:54.327206Z","shell.execute_reply.started":"2024-12-31T06:53:54.310198Z","shell.execute_reply":"2024-12-31T06:53:54.325102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}