{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport sklearn \nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.decomposition import PCA\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:22.985741Z","iopub.execute_input":"2025-01-07T04:48:22.986278Z","iopub.status.idle":"2025-01-07T04:48:25.851719Z","shell.execute_reply.started":"2025-01-07T04:48:22.986234Z","shell.execute_reply":"2025-01-07T04:48:25.850546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data=pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ndata.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:25.853435Z","iopub.execute_input":"2025-01-07T04:48:25.853973Z","iopub.status.idle":"2025-01-07T04:48:32.702757Z","shell.execute_reply.started":"2025-01-07T04:48:25.853929Z","shell.execute_reply":"2025-01-07T04:48:32.701494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test=pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:32.704879Z","iopub.execute_input":"2025-01-07T04:48:32.70522Z","iopub.status.idle":"2025-01-07T04:48:36.923214Z","shell.execute_reply.started":"2025-01-07T04:48:32.70519Z","shell.execute_reply":"2025-01-07T04:48:36.922147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:36.924715Z","iopub.execute_input":"2025-01-07T04:48:36.925031Z","iopub.status.idle":"2025-01-07T04:48:37.635142Z","shell.execute_reply.started":"2025-01-07T04:48:36.925006Z","shell.execute_reply":"2025-01-07T04:48:37.634187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"t=data.isnull().sum()\nprint(t)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:37.636068Z","iopub.execute_input":"2025-01-07T04:48:37.636318Z","iopub.status.idle":"2025-01-07T04:48:38.263799Z","shell.execute_reply.started":"2025-01-07T04:48:37.636297Z","shell.execute_reply":"2025-01-07T04:48:38.262439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def outliers_removal(df,columnName):\n    q1=df[columnName].quantile(0.25)\n    q3=df[columnName].quantile(0.75)\n    IQR=q3-q1\n    lower_bound=q1-1.5*IQR\n    higher_bound=q3+1.5*IQR\n    df_cleaned = df[(df[columnName] >= lower_bound) & (df[columnName] <= higher_bound)]\n    return df_cleaned\ndata=outliers_removal(data,'Previous Claims')\ntest=outliers_removal(test,'Previous Claims')\ndata=outliers_removal(data,'Credit Score')\ntest=outliers_removal(data,'Credit Score')\nprint(data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:38.264844Z","iopub.execute_input":"2025-01-07T04:48:38.265162Z","iopub.status.idle":"2025-01-07T04:48:39.173068Z","shell.execute_reply.started":"2025-01-07T04:48:38.265136Z","shell.execute_reply":"2025-01-07T04:48:39.171846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"t=data.isnull().sum()\nprint(t)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:39.174107Z","iopub.execute_input":"2025-01-07T04:48:39.17448Z","iopub.status.idle":"2025-01-07T04:48:39.563787Z","shell.execute_reply.started":"2025-01-07T04:48:39.174448Z","shell.execute_reply":"2025-01-07T04:48:39.562584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:39.567254Z","iopub.execute_input":"2025-01-07T04:48:39.567629Z","iopub.status.idle":"2025-01-07T04:48:39.960371Z","shell.execute_reply.started":"2025-01-07T04:48:39.567597Z","shell.execute_reply":"2025-01-07T04:48:39.959107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_iunique=[\"Marital Status\",\"Number of Dependents\",\"Occupation\",\"Customer Feedback\",\"Customer Feedback\"]\ndrive_unique=[]\nfor i in categorical_iunique:\n    en=list(data[i].unique())\n    drive_unique.append(en)\ndataframe_unique=pd.DataFrame(drive_unique)\nprint(dataframe_unique)\nprint(data[\"Number of Dependents\"].mode()[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:39.961947Z","iopub.execute_input":"2025-01-07T04:48:39.962268Z","iopub.status.idle":"2025-01-07T04:48:40.124848Z","shell.execute_reply.started":"2025-01-07T04:48:39.962239Z","shell.execute_reply":"2025-01-07T04:48:40.123683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def input_preprocessing(data):\n    if \"Age\" in data.columns:\n        if not data[\"Age\"].mode().empty:\n            data[\"Age\"].fillna(data[\"Age\"].mode()[0], inplace=True)\n    if \"Occupation\" in data.columns:\n        data[\"Occupation\"].fillna(\"eks\", inplace=True)\n    if \"Marital Status\" in data.columns:\n        data[\"Marital Status\"].fillna(\"Single\", inplace=True)\n    if \"Number of Dependents\" in data.columns:\n        if not data[\"Number of Dependents\"].mode().empty:\n            data[\"Number of Dependents\"].fillna(data[\"Number of Dependents\"].mode()[0], inplace=True)\n    if \"Annual Income\" in data.columns:\n        if not data[\"Annual Income\"].isnull().all():\n            data[\"Annual Income\"].fillna(data[\"Annual Income\"].mean(), inplace=True)\n    if \"Health Score\" in data.columns:\n        if not data[\"Health Score\"].isnull().all():\n            data[\"Health Score\"].fillna(data[\"Health Score\"].mean(), inplace=True)\n    if \"Customer Feedback\" in data.columns:\n        if not data[\"Customer Feedback\"].mode().empty:\n            data[\"Customer Feedback\"].fillna(data[\"Customer Feedback\"].mode()[0], inplace=True)\n\n# Apply preprocessing to both datasets\ninput_preprocessing(data)\ninput_preprocessing(test)\n\n# Check for remaining null values\nprint(data.isnull().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:40.125888Z","iopub.execute_input":"2025-01-07T04:48:40.126177Z","iopub.status.idle":"2025-01-07T04:48:41.141516Z","shell.execute_reply.started":"2025-01-07T04:48:40.126151Z","shell.execute_reply":"2025-01-07T04:48:41.140203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(data.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:41.14254Z","iopub.execute_input":"2025-01-07T04:48:41.142845Z","iopub.status.idle":"2025-01-07T04:48:41.536268Z","shell.execute_reply.started":"2025-01-07T04:48:41.142819Z","shell.execute_reply":"2025-01-07T04:48:41.53502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data=data.dropna()\nprint(data.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:41.537193Z","iopub.execute_input":"2025-01-07T04:48:41.537558Z","iopub.status.idle":"2025-01-07T04:48:42.473581Z","shell.execute_reply.started":"2025-01-07T04:48:41.53753Z","shell.execute_reply":"2025-01-07T04:48:42.472353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def date(df):\n\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Month'] = df['Policy Start Date'].dt.month\n\n    df['Year_sin'] = np.sin(2 * np.pi * df['Year'])\n    df['Year_cos'] = np.cos(2 * np.pi * df['Year'])\n    df['Month_sin'] = np.sin(2 * np.pi * df['Month'] / 12) \n    df['Month_cos'] = np.cos(2 * np.pi * df['Month'] / 12)\n    df['Day_sin'] = np.sin(2 * np.pi * df['Day'] / 31)  \n    df['Day_cos'] = np.cos(2 * np.pi * df['Day'] / 31)\n    df['Group']=(df['Year']-2020)*48+df['Month']*4+df['Day']//7\n    \n    df.drop('Policy Start Date', axis=1, inplace=True)\n\n    return df\n\ndata=date(data)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:42.474671Z","iopub.execute_input":"2025-01-07T04:48:42.474954Z","iopub.status.idle":"2025-01-07T04:48:43.175932Z","shell.execute_reply.started":"2025-01-07T04:48:42.474931Z","shell.execute_reply":"2025-01-07T04:48:43.174603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#implement pca to reduction and get correlation each dimension\nfrom sklearn.preprocessing import LabelEncoder\nencoder=LabelEncoder()\ndef encode_all(data,k):\n    for i in k:\n        data[i]=encoder.fit_transform(data[i])\ncategorical_iunique=[\"Marital Status\",\"Location\",\"Occupation\",\"Smoking Status\",\"Policy Type\",\"Exercise Frequency\",\"Property Type\",\"Customer Feedback\",\"Gender\",\"Education Level\"]\nencode_all(data,categorical_iunique)\nencode_all(test,categorical_iunique)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:43.177242Z","iopub.execute_input":"2025-01-07T04:48:43.177616Z","iopub.status.idle":"2025-01-07T04:48:45.823306Z","shell.execute_reply.started":"2025-01-07T04:48:43.177582Z","shell.execute_reply":"2025-01-07T04:48:45.822253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:45.824466Z","iopub.execute_input":"2025-01-07T04:48:45.824805Z","iopub.status.idle":"2025-01-07T04:48:45.854166Z","shell.execute_reply.started":"2025-01-07T04:48:45.824776Z","shell.execute_reply":"2025-01-07T04:48:45.852744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nskalarisasi=StandardScaler()\nscaled_data=skalarisasi.fit_transform(data)\n# columns_with_premium = data.apply(lambda col: col.astype(str).str.contains('Premium')).any()\n# print(columns_with_premium)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:45.855394Z","iopub.execute_input":"2025-01-07T04:48:45.855835Z","iopub.status.idle":"2025-01-07T04:48:46.296654Z","shell.execute_reply.started":"2025-01-07T04:48:45.855795Z","shell.execute_reply":"2025-01-07T04:48:46.295446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pca = PCA(n_components=2)\npca_result = pca.fit_transform(scaled_data)\n\n# Hasil PC1 dan PC2\ndf_pca = pd.DataFrame(pca_result, columns=['PC1', 'PC2'])\nprint(df_pca)\n\n# Visualisasi PC1 dan PC2\nplt.scatter(df_pca['PC1'], df_pca['PC2'])\nplt.xlabel('PC1')\nplt.ylabel('PC2')\nplt.title('Visualisasi PCA')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:46.297651Z","iopub.execute_input":"2025-01-07T04:48:46.298075Z","iopub.status.idle":"2025-01-07T04:48:51.808899Z","shell.execute_reply.started":"2025-01-07T04:48:46.298037Z","shell.execute_reply":"2025-01-07T04:48:51.807721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"explained_variance = pca.explained_variance_ratio_\nplt.bar([f'PC{i+1}' for i in range(len(explained_variance))], explained_variance)\nplt.xlabel('Principal Components')\nplt.ylabel('Explained Variance Ratio')\nplt.title('Explained Variance per Komponen Utama')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:51.809953Z","iopub.execute_input":"2025-01-07T04:48:51.810272Z","iopub.status.idle":"2025-01-07T04:48:52.004646Z","shell.execute_reply.started":"2025-01-07T04:48:51.810235Z","shell.execute_reply":"2025-01-07T04:48:52.003266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"loadings = pd.DataFrame(pca.components_.T, columns=['PC1', 'PC2'], index=data.columns)\n\n# 4. Menampilkan kontribusi kolom terhadap PC1\nprint(\"Kontribusi Kolom terhadap PC1:\")\nprint(loadings['PC1'])\n\n# 5. Menampilkan komponen utama pertama (nilai PC1 untuk setiap observasi)\ndf_pca = pd.DataFrame(pca_result, columns=['PC1', 'PC2'])\nprint(\"\\nNilai PC1 untuk setiap observasi:\")\nprint(df_pca['PC1'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:52.005957Z","iopub.execute_input":"2025-01-07T04:48:52.006392Z","iopub.status.idle":"2025-01-07T04:48:52.01719Z","shell.execute_reply.started":"2025-01-07T04:48:52.006349Z","shell.execute_reply":"2025-01-07T04:48:52.016002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\ngood_pca=[\"Education Level\",\"Occupation\",\"Policy Type\",\"Previous Claims\",\"Smoking Status\",\"Exercise Frequency\",\"Property Type\",\"Year\"]\ntarget=\"Premium Amount\"\nX=data[good_pca]\ny=data[target]\nrfc=RandomForestRegressor()\nrfc.fit(X,y)\nimportances = rfc.feature_importances_\nfeature_names = good_pca\n\n# Visualisasi Feature Importance\nplt.figure(figsize=(10, 6))\nplt.barh(feature_names, importances)\nplt.xlabel('Feature Importance')\nplt.ylabel('Features')\nplt.title('Feature Importance dari Random Forest')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-07T04:48:52.018886Z","iopub.execute_input":"2025-01-07T04:48:52.019387Z","iopub.status.idle":"2025-01-07T04:51:03.241053Z","shell.execute_reply.started":"2025-01-07T04:48:52.01935Z","shell.execute_reply":"2025-01-07T04:51:03.239785Z"}},"outputs":[],"execution_count":null}]}