{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:55.569824Z","iopub.execute_input":"2026-01-16T17:35:55.571898Z","iopub.status.idle":"2026-01-16T17:35:55.584279Z","shell.execute_reply.started":"2026-01-16T17:35:55.571828Z","shell.execute_reply":"2026-01-16T17:35:55.581775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer \nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import LabelEncoder , OneHotEncoder, MinMaxScaler, StandardScaler, RobustScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import RandomizedSearchCV\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:55.586912Z","iopub.execute_input":"2026-01-16T17:35:55.587308Z","iopub.status.idle":"2026-01-16T17:35:55.614545Z","shell.execute_reply.started":"2026-01-16T17:35:55.587284Z","shell.execute_reply":"2026-01-16T17:35:55.612276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv',nrows=10000)\ndf\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:55.616153Z","iopub.execute_input":"2026-01-16T17:35:55.616448Z","iopub.status.idle":"2026-01-16T17:35:55.712384Z","shell.execute_reply.started":"2026-01-16T17:35:55.616425Z","shell.execute_reply":"2026-01-16T17:35:55.710678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:55.714121Z","iopub.execute_input":"2026-01-16T17:35:55.714373Z","iopub.status.idle":"2026-01-16T17:35:55.750225Z","shell.execute_reply.started":"2026-01-16T17:35:55.714353Z","shell.execute_reply":"2026-01-16T17:35:55.748388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:55.752321Z","iopub.execute_input":"2026-01-16T17:35:55.753789Z","iopub.status.idle":"2026-01-16T17:35:55.771799Z","shell.execute_reply.started":"2026-01-16T17:35:55.753706Z","shell.execute_reply":"2026-01-16T17:35:55.769554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:55.774495Z","iopub.execute_input":"2026-01-16T17:35:55.775868Z","iopub.status.idle":"2026-01-16T17:35:55.804091Z","shell.execute_reply.started":"2026-01-16T17:35:55.775788Z","shell.execute_reply":"2026-01-16T17:35:55.802805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_cols = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Credit Score', 'Previous Claims', 'Premium Amount', 'Insurance Duration', 'Vehicle Age']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:55.805196Z","iopub.execute_input":"2026-01-16T17:35:55.805705Z","iopub.status.idle":"2026-01-16T17:35:55.828751Z","shell.execute_reply.started":"2026-01-16T17:35:55.805651Z","shell.execute_reply":"2026-01-16T17:35:55.826648Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_cols = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location', 'Policy Type', 'Smoking Status', 'Exercise Frequency', 'Property Type', 'Customer Feedback']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:55.830436Z","iopub.execute_input":"2026-01-16T17:35:55.830762Z","iopub.status.idle":"2026-01-16T17:35:55.862029Z","shell.execute_reply.started":"2026-01-16T17:35:55.83073Z","shell.execute_reply":"2026-01-16T17:35:55.860435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in categorical_cols:\n    missing_percent = df[col].isnull().mean()\n    if missing_percent < 0.1:\n        df[col].fillna(df[col].mode()[0], inplace=True)\n    else:\n        df[col].fillna('Missing', inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:55.864006Z","iopub.execute_input":"2026-01-16T17:35:55.864359Z","iopub.status.idle":"2026-01-16T17:35:55.913433Z","shell.execute_reply.started":"2026-01-16T17:35:55.864337Z","shell.execute_reply":"2026-01-16T17:35:55.912106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:55.915086Z","iopub.execute_input":"2026-01-16T17:35:55.916551Z","iopub.status.idle":"2026-01-16T17:35:55.947252Z","shell.execute_reply.started":"2026-01-16T17:35:55.916484Z","shell.execute_reply":"2026-01-16T17:35:55.946415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:55.948185Z","iopub.execute_input":"2026-01-16T17:35:55.94852Z","iopub.status.idle":"2026-01-16T17:35:55.986935Z","shell.execute_reply.started":"2026-01-16T17:35:55.948493Z","shell.execute_reply":"2026-01-16T17:35:55.985613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"count_age = df['Age'].value_counts()\ncount_age","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:55.990111Z","iopub.execute_input":"2026-01-16T17:35:55.990424Z","iopub.status.idle":"2026-01-16T17:35:56.022808Z","shell.execute_reply.started":"2026-01-16T17:35:55.990396Z","shell.execute_reply":"2026-01-16T17:35:56.020266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"count_age_na = df['Age'].isnull().sum()\ncount_age_na","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.024403Z","iopub.execute_input":"2026-01-16T17:35:56.024685Z","iopub.status.idle":"2026-01-16T17:35:56.053237Z","shell.execute_reply.started":"2026-01-16T17:35:56.024666Z","shell.execute_reply":"2026-01-16T17:35:56.051544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"( count_age_na/1200000 ) * 100 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.054576Z","iopub.execute_input":"2026-01-16T17:35:56.055495Z","iopub.status.idle":"2026-01-16T17:35:56.088667Z","shell.execute_reply.started":"2026-01-16T17:35:56.055432Z","shell.execute_reply":"2026-01-16T17:35:56.086497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.090368Z","iopub.execute_input":"2026-01-16T17:35:56.090664Z","iopub.status.idle":"2026-01-16T17:35:56.11658Z","shell.execute_reply.started":"2026-01-16T17:35:56.090639Z","shell.execute_reply":"2026-01-16T17:35:56.11465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in df.columns :\n    na = (df[i].isnull().sum() / df.shape[0] *100)\n    print(f\"porsentage missing in{i}  in is : {na} %\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.118607Z","iopub.execute_input":"2026-01-16T17:35:56.119064Z","iopub.status.idle":"2026-01-16T17:35:56.15957Z","shell.execute_reply.started":"2026-01-16T17:35:56.118997Z","shell.execute_reply":"2026-01-16T17:35:56.157625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df. info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.161175Z","iopub.execute_input":"2026-01-16T17:35:56.161564Z","iopub.status.idle":"2026-01-16T17:35:56.203407Z","shell.execute_reply.started":"2026-01-16T17:35:56.161533Z","shell.execute_reply":"2026-01-16T17:35:56.201664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat = df.select_dtypes('object').columns\ncat","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.206926Z","iopub.execute_input":"2026-01-16T17:35:56.2073Z","iopub.status.idle":"2026-01-16T17:35:56.24905Z","shell.execute_reply.started":"2026-01-16T17:35:56.207277Z","shell.execute_reply":"2026-01-16T17:35:56.246682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in cat:\n    df[i]= df[i].astype('category')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.251159Z","iopub.execute_input":"2026-01-16T17:35:56.251706Z","iopub.status.idle":"2026-01-16T17:35:56.307608Z","shell.execute_reply.started":"2026-01-16T17:35:56.25168Z","shell.execute_reply":"2026-01-16T17:35:56.306804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.308914Z","iopub.execute_input":"2026-01-16T17:35:56.309214Z","iopub.status.idle":"2026-01-16T17:35:56.344571Z","shell.execute_reply.started":"2026-01-16T17:35:56.309188Z","shell.execute_reply":"2026-01-16T17:35:56.343799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in df.columns :\n    print(df[i].value_counts())\n    print(\"______________________\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.345834Z","iopub.execute_input":"2026-01-16T17:35:56.346139Z","iopub.status.idle":"2026-01-16T17:35:56.393833Z","shell.execute_reply.started":"2026-01-16T17:35:56.346111Z","shell.execute_reply":"2026-01-16T17:35:56.392445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Policy Start Date'] =df['Policy Start Date'].astype('datetime64')\ndf.info()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.395074Z","iopub.execute_input":"2026-01-16T17:35:56.395386Z","iopub.status.idle":"2026-01-16T17:35:56.421392Z","shell.execute_reply.started":"2026-01-16T17:35:56.395354Z","shell.execute_reply":"2026-01-16T17:35:56.419585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Number of Dependents']= df ['Number of Dependents'].astype('category')\ndf['Previous Claims'] =df['Previous Claims'].astype('category')\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.423628Z","iopub.execute_input":"2026-01-16T17:35:56.425141Z","iopub.status.idle":"2026-01-16T17:35:56.476481Z","shell.execute_reply.started":"2026-01-16T17:35:56.424997Z","shell.execute_reply":"2026-01-16T17:35:56.474621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.481508Z","iopub.execute_input":"2026-01-16T17:35:56.481892Z","iopub.status.idle":"2026-01-16T17:35:56.50157Z","shell.execute_reply.started":"2026-01-16T17:35:56.481868Z","shell.execute_reply":"2026-01-16T17:35:56.49974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def null_percentage (col) :\n    col_nun = df[col].isnull().sum()\n    na_per = (col_nun / df.shape[0] ) * 100\n    return na_per\nfor i in df.columns :\n    if null_percentage(i) > 0 :\n        print(f\"{i} null : {null_percentage(i).round(2)} %\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.503272Z","iopub.execute_input":"2026-01-16T17:35:56.503627Z","iopub.status.idle":"2026-01-16T17:35:56.532609Z","shell.execute_reply.started":"2026-01-16T17:35:56.503597Z","shell.execute_reply":"2026-01-16T17:35:56.530564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" def null_percentage (col) :\n    col_nun = df[col].isnull().sum()\n    na_per = (col_nun / df.shape[0] ) * 100\n    return na_per\nfor i in df.columns :\n    if null_percentage(i) > 0 :\n        print(f\"{i} null : {null_percentage(i).round(2)} \")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.534514Z","iopub.execute_input":"2026-01-16T17:35:56.534977Z","iopub.status.idle":"2026-01-16T17:35:56.576593Z","shell.execute_reply.started":"2026-01-16T17:35:56.534922Z","shell.execute_reply":"2026-01-16T17:35:56.574955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.577887Z","iopub.execute_input":"2026-01-16T17:35:56.578255Z","iopub.status.idle":"2026-01-16T17:35:56.620614Z","shell.execute_reply.started":"2026-01-16T17:35:56.578227Z","shell.execute_reply":"2026-01-16T17:35:56.619735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.drop('id' , axis=1)\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.621761Z","iopub.execute_input":"2026-01-16T17:35:56.622104Z","iopub.status.idle":"2026-01-16T17:35:56.684481Z","shell.execute_reply.started":"2026-01-16T17:35:56.622075Z","shell.execute_reply":"2026-01-16T17:35:56.681598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.686607Z","iopub.execute_input":"2026-01-16T17:35:56.686997Z","iopub.status.idle":"2026-01-16T17:35:56.730806Z","shell.execute_reply.started":"2026-01-16T17:35:56.686972Z","shell.execute_reply":"2026-01-16T17:35:56.729525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.73193Z","iopub.execute_input":"2026-01-16T17:35:56.73222Z","iopub.status.idle":"2026-01-16T17:35:56.748961Z","shell.execute_reply.started":"2026-01-16T17:35:56.732197Z","shell.execute_reply":"2026-01-16T17:35:56.748037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'], errors='coerce')\n\ndf['Policy_Start_Year'] = df['Policy Start Date'].dt.year\ndf['Policy_Start_Month'] = df['Policy Start Date'].dt.month\ndf['Policy_Start_Day'] = df['Policy Start Date'].dt.day\ndf['Policy_Start_Weekday'] = df['Policy Start Date'].dt.weekday","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.750031Z","iopub.execute_input":"2026-01-16T17:35:56.750277Z","iopub.status.idle":"2026-01-16T17:35:56.79384Z","shell.execute_reply.started":"2026-01-16T17:35:56.750256Z","shell.execute_reply":"2026-01-16T17:35:56.791911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nnumeric_cols = df.select_dtypes(include=['float64', 'int64']).columns\n\nplt.figure(figsize=(10, 8))\nsns.heatmap(df[numeric_cols].corr(),\n    annot=True,\n    cmap='coolwarm',\n    fmt='.2f',\n    linewidths=0.5\n)\n\nplt.title('Correlation heatmap for all numerical features')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:56.796276Z","iopub.execute_input":"2026-01-16T17:35:56.798381Z","iopub.status.idle":"2026-01-16T17:35:57.780418Z","shell.execute_reply.started":"2026-01-16T17:35:56.798337Z","shell.execute_reply":"2026-01-16T17:35:57.778587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.get_dummies(df, columns=['Gender', 'Marital Status', 'Education Level', 'Location', 'Policy Type', 'Smoking Status', 'Exercise Frequency', 'Property Type'], drop_first=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:57.782339Z","iopub.execute_input":"2026-01-16T17:35:57.782722Z","iopub.status.idle":"2026-01-16T17:35:57.805618Z","shell.execute_reply.started":"2026-01-16T17:35:57.782697Z","shell.execute_reply":"2026-01-16T17:35:57.804567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"non_numeric_cols = X_train.select_dtypes(include=['object', 'category']).columns\nprint(non_numeric_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:57.806598Z","iopub.execute_input":"2026-01-16T17:35:57.807316Z","iopub.status.idle":"2026-01-16T17:35:57.826792Z","shell.execute_reply.started":"2026-01-16T17:35:57.807274Z","shell.execute_reply":"2026-01-16T17:35:57.82543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"datetime_cols = X_train.select_dtypes(include=['datetime64[ns]']).columns\nfor col in datetime_cols:\n    for df_ in [X_train, X_test]:\n        df_[col + '_year'] = df_[col].dt.year\n        df_[col + '_month'] = df_[col].dt.month\n        df_[col + '_day'] = df_[col].dt.day\n        df_[col + '_weekday'] = df_[col].dt.weekday\n    X_train.drop(columns=[col], inplace=True)\n    X_test.drop(columns=[col], inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:57.828084Z","iopub.execute_input":"2026-01-16T17:35:57.828343Z","iopub.status.idle":"2026-01-16T17:35:57.855147Z","shell.execute_reply.started":"2026-01-16T17:35:57.828318Z","shell.execute_reply":"2026-01-16T17:35:57.853692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bool_cols = X_train.select_dtypes(include=['bool', 'boolean']).columns\nX_train[bool_cols] = X_train[bool_cols].astype(int)\nX_test[bool_cols] = X_test[bool_cols].astype(int)\nint_cols = X_train.select_dtypes(include=['Int32', 'Int64']).columns\nX_train[int_cols] = X_train[int_cols].astype('int64')\nX_test[int_cols] = X_test[int_cols].astype('int64')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:57.856206Z","iopub.execute_input":"2026-01-16T17:35:57.856539Z","iopub.status.idle":"2026-01-16T17:35:57.907835Z","shell.execute_reply.started":"2026-01-16T17:35:57.856509Z","shell.execute_reply":"2026-01-16T17:35:57.905968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"int_cols = X_train.select_dtypes(include=['Int32', 'Int64']).columns\nX_train[int_cols] = X_train[int_cols].astype('int64')\nX_test[int_cols] = X_test[int_cols].astype('int64')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:57.910094Z","iopub.execute_input":"2026-01-16T17:35:57.910829Z","iopub.status.idle":"2026-01-16T17:35:57.940206Z","shell.execute_reply.started":"2026-01-16T17:35:57.910772Z","shell.execute_reply":"2026-01-16T17:35:57.938084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in df.select_dtypes(include=['object', 'category']).columns:\n    df[col] = pd.factorize(df[col])[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:57.942512Z","iopub.execute_input":"2026-01-16T17:35:57.943683Z","iopub.status.idle":"2026-01-16T17:35:57.966833Z","shell.execute_reply.started":"2026-01-16T17:35:57.943623Z","shell.execute_reply":"2026-01-16T17:35:57.964723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in non_numeric_cols:\n    le = LabelEncoder()\n    X_train[col] = le.fit_transform(X_train[col])\n    X_test[col] = le.transform(X_test[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:57.968831Z","iopub.execute_input":"2026-01-16T17:35:57.969271Z","iopub.status.idle":"2026-01-16T17:35:58.002123Z","shell.execute_reply.started":"2026-01-16T17:35:57.969246Z","shell.execute_reply":"2026-01-16T17:35:58.000092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df.drop('Premium Amount', axis=1)  \ny = df['Premium Amount']  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:58.006381Z","iopub.execute_input":"2026-01-16T17:35:58.007063Z","iopub.status.idle":"2026-01-16T17:35:58.034732Z","shell.execute_reply.started":"2026-01-16T17:35:58.007001Z","shell.execute_reply":"2026-01-16T17:35:58.033777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in X_train.select_dtypes(include=['object', 'category']).columns:\n    X_train[col], _ = pd.factorize(X_train[col])\n    X_test[col], _ = pd.factorize(X_test[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:58.035735Z","iopub.execute_input":"2026-01-16T17:35:58.035971Z","iopub.status.idle":"2026-01-16T17:35:58.070729Z","shell.execute_reply.started":"2026-01-16T17:35:58.03595Z","shell.execute_reply":"2026-01-16T17:35:58.068304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler()\nnumeric_cols = X.select_dtypes(include=['int64', 'float64']).columns\n\nX_train[numeric_cols] = scaler.fit_transform(X_train[numeric_cols])\nX_test[numeric_cols] = scaler.transform(X_test[numeric_cols])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:58.073173Z","iopub.execute_input":"2026-01-16T17:35:58.073821Z","iopub.status.idle":"2026-01-16T17:35:58.127656Z","shell.execute_reply.started":"2026-01-16T17:35:58.073776Z","shell.execute_reply":"2026-01-16T17:35:58.122899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = RandomForestRegressor(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)\ny_pred = model.predict(X_test)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:35:58.129977Z","iopub.execute_input":"2026-01-16T17:35:58.131013Z","iopub.status.idle":"2026-01-16T17:36:09.489136Z","shell.execute_reply.started":"2026-01-16T17:35:58.130735Z","shell.execute_reply":"2026-01-16T17:36:09.487385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mse = mean_squared_error(y_test, y_pred)\nrmse = np.sqrt(mse)\nmae = mean_absolute_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\nprint(\"RandomForest Regression Evaluation Report\")\nprint(\"----------------------------------------\")\nprint(f\"MSE{mse:.2f}\")\nprint(f\"RMSE: {rmse:.2f}\")\nprint(f\"MAE: {mae:.2f}\")\nprint(f\"R² Score: {r2:.2f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:36:09.490889Z","iopub.execute_input":"2026-01-16T17:36:09.491373Z","iopub.status.idle":"2026-01-16T17:36:09.504396Z","shell.execute_reply.started":"2026-01-16T17:36:09.491323Z","shell.execute_reply":"2026-01-16T17:36:09.503292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_grid = {\n    'n_estimators': [100, 300, 500],\n    'max_depth': [None, 10, 20, 30],\n    'min_samples_split': [2, 5, 10],\n    'min_samples_leaf': [1, 2, 4],\n    'max_features': ['auto', 'sqrt', 'log2']}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:36:09.505609Z","iopub.execute_input":"2026-01-16T17:36:09.505941Z","iopub.status.idle":"2026-01-16T17:36:09.534471Z","shell.execute_reply.started":"2026-01-16T17:36:09.505912Z","shell.execute_reply":"2026-01-16T17:36:09.532696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rf = RandomForestRegressor(random_state=42)\nparam_dist = {\n    'n_estimators': [50, 100, 150],\n    'max_depth': [None, 10, 20, 30],\n    'min_samples_split': [2, 5, 10],\n    'min_samples_leaf': [1, 2, 4]}\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:36:09.536127Z","iopub.execute_input":"2026-01-16T17:36:09.536539Z","iopub.status.idle":"2026-01-16T17:36:09.562805Z","shell.execute_reply.started":"2026-01-16T17:36:09.536495Z","shell.execute_reply":"2026-01-16T17:36:09.560856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_search = RandomizedSearchCV(\n    estimator=rf,\n    param_distributions=param_dist,\n    n_iter=3,          # Only test 10 random combinations\n    cv=3,\n    n_jobs=-1,\n    scoring='r2',\n    verbose=1,\n    random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:36:09.564726Z","iopub.execute_input":"2026-01-16T17:36:09.565108Z","iopub.status.idle":"2026-01-16T17:36:09.596629Z","shell.execute_reply.started":"2026-01-16T17:36:09.565083Z","shell.execute_reply":"2026-01-16T17:36:09.594781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_search.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:36:09.597812Z","iopub.execute_input":"2026-01-16T17:36:09.598141Z","iopub.status.idle":"2026-01-16T17:36:45.910805Z","shell.execute_reply.started":"2026-01-16T17:36:09.598111Z","shell.execute_reply":"2026-01-16T17:36:45.909252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_model = random_search.best_estimator_\ny_pred = best_model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:36:45.912069Z","iopub.execute_input":"2026-01-16T17:36:45.912402Z","iopub.status.idle":"2026-01-16T17:36:45.986853Z","shell.execute_reply.started":"2026-01-16T17:36:45.912373Z","shell.execute_reply":"2026-01-16T17:36:45.984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rmse = np.sqrt(mean_squared_error(y_test, y_pred))\nr2 = r2_score(y_test, y_pred)\nprint(\"Best parameters:\", random_search.best_params_)\nprint(f\"RMSE: {rmse:.2f}\")\nprint(f\"R² Score: {r2:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-16T17:36:45.989271Z","iopub.execute_input":"2026-01-16T17:36:45.991818Z","iopub.status.idle":"2026-01-16T17:36:46.00286Z","shell.execute_reply.started":"2026-01-16T17:36:45.991748Z","shell.execute_reply":"2026-01-16T17:36:46.001438Z"}},"outputs":[],"execution_count":null}]}