{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":31236,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:25:59.130693Z","iopub.execute_input":"2026-01-15T09:25:59.13126Z","iopub.status.idle":"2026-01-15T09:25:59.398751Z","shell.execute_reply.started":"2026-01-15T09:25:59.131234Z","shell.execute_reply":"2026-01-15T09:25:59.39812Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h2 style=\"\n    font-size: 32px;\n    font-weight: 900;\n    color: #ffffff;\n    padding: 14px 22px;\n    background: linear-gradient(135deg, #ff8800, #ff5e62);\n    border-radius: 10px;\n    letter-spacing: 1px;\n    box-shadow: 0 6px 18px rgba(255, 94, 98, 0.35);\n    text-transform: uppercase;\n\">\nImport All Libraries\n</h2>\n","metadata":{}},{"cell_type":"code","source":"import seaborn as sns \nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import root_mean_squared_log_error,root_mean_squared_error\nfrom sklearn.preprocessing import LabelEncoder,StandardScaler\n\nfrom sklearn.model_selection import KFold,StratifiedKFold\nfrom lightgbm import LGBMRegressor \nimport optuna\nimport lightgbm as lgb\n\nimport warnings\nwarnings.filterwarnings(\"ignore\",category = FutureWarning)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:25:59.400017Z","iopub.execute_input":"2026-01-15T09:25:59.400472Z","iopub.status.idle":"2026-01-15T09:26:09.027876Z","shell.execute_reply.started":"2026-01-15T09:25:59.400439Z","shell.execute_reply":"2026-01-15T09:26:09.027256Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h2 style=\"\n    font-size: 32px;\n    font-weight: 900;\n    color: #ffffff;\n    padding: 14px 22px;\n    background: linear-gradient(135deg, #ff8800, #ff5e62);\n    border-radius: 10px;\n    letter-spacing: 1px;\n    box-shadow: 0 6px 18px rgba(255, 94, 98, 0.35);\n    text-transform: uppercase;\n\">\nTaking Info. About The Dataset\n</h2>\n","metadata":{}},{"cell_type":"code","source":"train_d = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest_d = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nsubmission = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:26:09.029162Z","iopub.execute_input":"2026-01-15T09:26:09.029839Z","iopub.status.idle":"2026-01-15T09:26:16.854644Z","shell.execute_reply.started":"2026-01-15T09:26:09.029816Z","shell.execute_reply":"2026-01-15T09:26:16.854041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_d.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:26:16.855584Z","iopub.execute_input":"2026-01-15T09:26:16.856095Z","iopub.status.idle":"2026-01-15T09:26:16.900865Z","shell.execute_reply.started":"2026-01-15T09:26:16.856069Z","shell.execute_reply":"2026-01-15T09:26:16.900202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_d.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:26:16.902241Z","iopub.execute_input":"2026-01-15T09:26:16.902501Z","iopub.status.idle":"2026-01-15T09:26:16.918991Z","shell.execute_reply.started":"2026-01-15T09:26:16.902479Z","shell.execute_reply":"2026-01-15T09:26:16.918486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_d.isnull().sum().sort_values(ascending = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:26:16.919755Z","iopub.execute_input":"2026-01-15T09:26:16.919983Z","iopub.status.idle":"2026-01-15T09:26:17.526426Z","shell.execute_reply.started":"2026-01-15T09:26:16.919964Z","shell.execute_reply":"2026-01-15T09:26:17.525816Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Our Dataset Has Many Null Values","metadata":{}},{"cell_type":"code","source":"train_d.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:26:17.527347Z","iopub.execute_input":"2026-01-15T09:26:17.527651Z","iopub.status.idle":"2026-01-15T09:26:17.532517Z","shell.execute_reply.started":"2026-01-15T09:26:17.52762Z","shell.execute_reply":"2026-01-15T09:26:17.531768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_d.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:26:17.533566Z","iopub.execute_input":"2026-01-15T09:26:17.533818Z","iopub.status.idle":"2026-01-15T09:26:17.55269Z","shell.execute_reply.started":"2026-01-15T09:26:17.533795Z","shell.execute_reply":"2026-01-15T09:26:17.552037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_d.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:26:17.55351Z","iopub.execute_input":"2026-01-15T09:26:17.553799Z","iopub.status.idle":"2026-01-15T09:26:18.183129Z","shell.execute_reply.started":"2026-01-15T09:26:17.553773Z","shell.execute_reply":"2026-01-15T09:26:18.182524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = train_d.select_dtypes('object').columns\nnum_cols = train_d.select_dtypes(include = ['float64','int64']).columns\nprint(\"Categorical Columns Are: \")\nfor i,col in enumerate(cat_cols,1):\n    print(i,col)\nprint(\"\\n\")\n\nprint(\"Numerical Columns Are: \")\nfor i,col in enumerate(num_cols,1):\n    print(i,col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:26:18.183947Z","iopub.execute_input":"2026-01-15T09:26:18.1842Z","iopub.status.idle":"2026-01-15T09:26:18.374117Z","shell.execute_reply.started":"2026-01-15T09:26:18.184166Z","shell.execute_reply":"2026-01-15T09:26:18.373387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_d.describe().style.set_properties(**{\n    'backgroound-color' : '#f2f2f2',\n    'color' : 'white'\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:26:18.376229Z","iopub.execute_input":"2026-01-15T09:26:18.376817Z","iopub.status.idle":"2026-01-15T09:26:18.88083Z","shell.execute_reply.started":"2026-01-15T09:26:18.376795Z","shell.execute_reply":"2026-01-15T09:26:18.880203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Unique Values \nprint(\"Printing Unique Values Of All Columns\")\nfor col in train_d:\n    print(\"-\"*60)\n    unq = train_d[col].unique()\n    if len(unq) < 15:\n        print(f\"{col} : {unq}\\n\")\n    else:\n        print(f\"{col} : {unq[:5]} .... {col} Have So much Unique Values \\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:26:18.881644Z","iopub.execute_input":"2026-01-15T09:26:18.881919Z","iopub.status.idle":"2026-01-15T09:26:19.803944Z","shell.execute_reply.started":"2026-01-15T09:26:18.881891Z","shell.execute_reply":"2026-01-15T09:26:19.803344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_d.isnull().sum() / len(train_d) * 100","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:26:19.804856Z","iopub.execute_input":"2026-01-15T09:26:19.805107Z","iopub.status.idle":"2026-01-15T09:26:20.396192Z","shell.execute_reply.started":"2026-01-15T09:26:19.805085Z","shell.execute_reply":"2026-01-15T09:26:20.395468Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h2 style=\"\n    font-size: 32px;\n    font-weight: 900;\n    color: #ffffff;\n    padding: 14px 22px;\n    background: linear-gradient(135deg, #ff8800, #ff5e62);\n    border-radius: 10px;\n    letter-spacing: 1px;\n    box-shadow: 0 6px 18px rgba(255, 94, 98, 0.35);\n    text-transform: uppercase;\n\">\nEDA\n</h2>\n","metadata":{}},{"cell_type":"code","source":"sns.set_style('darkgrid')\nplt.figure(figsize = (14,len(num_cols)*3))\n\nfor idx,col in enumerate(num_cols,1):\n    plt.subplot(len(num_cols),2,2*idx-1)\n    sns.histplot(train_d[col],kde = True,bins = 20,color = 'darkorange',edgecolor ='black')\n    plt.title(f\"Distribution of {col}\")\n\n    plt.subplot(len(num_cols),2,2*idx)\n    sns.boxplot(x = train_d[col],color = 'lavender')\n    plt.title(f\"Boxplot of {col}\")\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:26:20.397364Z","iopub.execute_input":"2026-01-15T09:26:20.397676Z","iopub.status.idle":"2026-01-15T09:27:21.131476Z","shell.execute_reply.started":"2026-01-15T09:26:20.39765Z","shell.execute_reply":"2026-01-15T09:27:21.130672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# correlation matrix\ncorrelation_matrix = train_d[num_cols].corr()\n\nplt.figure(figsize = (12,8))\n\nsns.heatmap(correlation_matrix,annot = True,fmt = \".2f\",cmap = 'coolwarm',cbar = True,linewidth = 0.5)\nplt.title(\"correlation Heatmap of Numerical Variable\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:27:21.132688Z","iopub.execute_input":"2026-01-15T09:27:21.133251Z","iopub.status.idle":"2026-01-15T09:27:21.87198Z","shell.execute_reply.started":"2026-01-15T09:27:21.133217Z","shell.execute_reply":"2026-01-15T09:27:21.871344Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h2 style=\"\n    font-size: 32px;\n    font-weight: 900;\n    color: #ffffff;\n    padding: 14px 22px;\n    background: linear-gradient(135deg, #ff8800, #ff5e62);\n    border-radius: 10px;\n    letter-spacing: 1px;\n    box-shadow: 0 6px 18px rgba(255, 94, 98, 0.35);\n    text-transform: uppercase;\n\">\nFeature Engineerig\n</h2>\n","metadata":{}},{"cell_type":"code","source":"# Applying target encoding \ndef target_encoding(train, predict, n_splits=8):\n    train = train.copy()\n    predict = predict.copy()\n\n    kf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\n\n    mean_features_train = {}\n    mean_features_test = {}\n\n    # Compute global mean once\n    target_global = train[target].mean()\n\n    for col in cols:\n\n        # K-FOLD TARGET ENCODING\n        oof = np.zeros(len(train))\n        for tr_idx, val_idx in kf.split(train):\n            tr_fold = train.iloc[tr_idx]\n\n            # Mean for each category in this fold\n            fold_map = tr_fold.groupby(col)[target].mean()\n\n            # Map on validation fold\n            oof[val_idx] = train[col].iloc[val_idx].map(fold_map).fillna(target_global)\n\n        mean_features_train[f\"mean_{col}\"] = oof\n\n        # Apply encoding to the prediction set\n        global_map = train.groupby(col)[target].mean()\n        mean_features_test[f\"mean_{col}\"] = (\n            predict[col].map(global_map).fillna(target_global)\n        )\n\n\n    train = pd.concat([train, pd.DataFrame(mean_features_train)], axis=1)\n    predict = pd.concat([predict, pd.DataFrame(mean_features_test)], axis=1)\n\n    return train, predict\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:27:21.872942Z","iopub.execute_input":"2026-01-15T09:27:21.873227Z","iopub.status.idle":"2026-01-15T09:27:21.879381Z","shell.execute_reply.started":"2026-01-15T09:27:21.873198Z","shell.execute_reply":"2026-01-15T09:27:21.878852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Applying target encoding\ntarget = 'Premium Amount'\ncols = (\n    train_d.drop(columns=['id',target],errors = 'ignore')\n    .columns\n    .tolist()\n)\n\ntrain_d,test_d = target_encoding(train_d,test_d,n_splits = 10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:27:21.879999Z","iopub.execute_input":"2026-01-15T09:27:21.880235Z","iopub.status.idle":"2026-01-15T09:28:25.759242Z","shell.execute_reply.started":"2026-01-15T09:27:21.880215Z","shell.execute_reply":"2026-01-15T09:28:25.758641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_d.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:30:44.798963Z","iopub.execute_input":"2026-01-15T09:30:44.799635Z","iopub.status.idle":"2026-01-15T09:30:44.818301Z","shell.execute_reply.started":"2026-01-15T09:30:44.799607Z","shell.execute_reply":"2026-01-15T09:30:44.8175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_d['Occupation'] = train_d['Occupation'].fillna('Unknown')\ntest_d['Occupation'] =test_d['Occupation'].fillna('Unknown')\n\ntrain_d['Previous Claims'] = train_d['Previous Claims'].fillna(-1)\ntest_d['Previous Claims'] = test_d['Previous Claims'].fillna(-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:30:57.466485Z","iopub.execute_input":"2026-01-15T09:30:57.467054Z","iopub.status.idle":"2026-01-15T09:30:57.652958Z","shell.execute_reply.started":"2026-01-15T09:30:57.467028Z","shell.execute_reply":"2026-01-15T09:30:57.6522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in ['Number of Dependents','Credit Score','Health Score']:\n\n    med = train_d[col].median()\n    train_d[col] = train_d[col].fillna(med)\n    test_d[col]  = test_d[col].fillna(med)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:30:57.699749Z","iopub.execute_input":"2026-01-15T09:30:57.700392Z","iopub.status.idle":"2026-01-15T09:30:57.781715Z","shell.execute_reply.started":"2026-01-15T09:30:57.700364Z","shell.execute_reply":"2026-01-15T09:30:57.780951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_col = ['Annual Income','Vehicle Age','Insurance Duration','Age']\nfor col in numerical_col:\n    med = train_d[col].median()\n    train_d[col] = train_d[col].fillna(med)\n    test_d[col] = test_d[col].fillna(med)\n\ncategorical_col = ['Marital Status','Customer Feedback']\n\nfor col in categorical_col:\n    med = train_d[col].mode()[0]\n    train_d[col] = train_d[col].fillna(med)\n    test_d[col] = test_d[col].fillna(med)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:30:57.87631Z","iopub.execute_input":"2026-01-15T09:30:57.876859Z","iopub.status.idle":"2026-01-15T09:30:58.380968Z","shell.execute_reply.started":"2026-01-15T09:30:57.87683Z","shell.execute_reply":"2026-01-15T09:30:58.38021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_d.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:31:00.025783Z","iopub.execute_input":"2026-01-15T09:31:00.026484Z","iopub.status.idle":"2026-01-15T09:31:00.044403Z","shell.execute_reply.started":"2026-01-15T09:31:00.026457Z","shell.execute_reply":"2026-01-15T09:31:00.043806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Date And Time \ndef date(data,ref):\n\n    data['policy_date'] = data['Policy Start Date'].dt.day\n    data['policy_month']= data['Policy Start Date'].dt.month\n    data['policy_year'] = data['Policy Start Date'].dt.year\n\n    data['month_name'] = data['Policy Start Date'].dt.month_name()\n    data['day_of_week'] = data['Policy Start Date'].dt.day_name()\n    data['week'] = data['Policy Start Date'].dt.isocalendar().week.astype(int)\n\n    data['policy_age_days'] = (ref - data['Policy Start Date']).dt.days\n\n    return data\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:31:00.276825Z","iopub.execute_input":"2026-01-15T09:31:00.277376Z","iopub.status.idle":"2026-01-15T09:31:00.28184Z","shell.execute_reply.started":"2026-01-15T09:31:00.27735Z","shell.execute_reply":"2026-01-15T09:31:00.281073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_d['Policy Start Date'] = pd.to_datetime(train_d['Policy Start Date'])\ntest_d['Policy Start Date'] = pd.to_datetime(test_d['Policy Start Date'])\nref = train_d['Policy Start Date'].max()\n\ntrain_d = date(train_d, ref)\ntest_d  = date(test_d, ref)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:31:00.484376Z","iopub.execute_input":"2026-01-15T09:31:00.484946Z","iopub.status.idle":"2026-01-15T09:31:02.435067Z","shell.execute_reply.started":"2026-01-15T09:31:00.484922Z","shell.execute_reply":"2026-01-15T09:31:02.434493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_d.drop('Policy Start Date',axis = 1,inplace = True)\ntest_d.drop('Policy Start Date',axis = 1,inplace = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:31:03.170191Z","iopub.execute_input":"2026-01-15T09:31:03.170505Z","iopub.status.idle":"2026-01-15T09:31:03.71322Z","shell.execute_reply.started":"2026-01-15T09:31:03.170481Z","shell.execute_reply":"2026-01-15T09:31:03.712644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Now applying label Encoder\ncategorical = train_d.select_dtypes('object').columns\n\nfor col in categorical:\n    train_d[col] = train_d[col].astype('category')\n    test_d[col] = test_d[col].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:31:05.052452Z","iopub.execute_input":"2026-01-15T09:31:05.052765Z","iopub.status.idle":"2026-01-15T09:31:07.091414Z","shell.execute_reply.started":"2026-01-15T09:31:05.052742Z","shell.execute_reply":"2026-01-15T09:31:07.090591Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h2 style=\"\n    font-size: 32px;\n    font-weight: 900;\n    color: #ffffff;\n    padding: 14px 22px;\n    background: linear-gradient(135deg, #ff8800, #ff5e62);\n    border-radius: 10px;\n    letter-spacing: 1px;\n    box-shadow: 0 6px 18px rgba(255, 94, 98, 0.35);\n    text-transform: uppercase;\n\">\nTraining Our Model \n</h2>\n","metadata":{}},{"cell_type":"code","source":"test_d = test_d.drop('id',axis = 1)\ntrain_d = train_d.drop('id',axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:31:07.092752Z","iopub.execute_input":"2026-01-15T09:31:07.093331Z","iopub.status.idle":"2026-01-15T09:31:07.271418Z","shell.execute_reply.started":"2026-01-15T09:31:07.093297Z","shell.execute_reply":"2026-01-15T09:31:07.270647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x,y = train_d.drop('Premium Amount',axis = 1), train_d['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:31:08.901456Z","iopub.execute_input":"2026-01-15T09:31:08.902022Z","iopub.status.idle":"2026-01-15T09:31:09.00989Z","shell.execute_reply.started":"2026-01-15T09:31:08.901994Z","shell.execute_reply":"2026-01-15T09:31:09.009119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = np.log1p(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:31:10.343205Z","iopub.execute_input":"2026-01-15T09:31:10.343898Z","iopub.status.idle":"2026-01-15T09:31:10.353332Z","shell.execute_reply.started":"2026-01-15T09:31:10.343868Z","shell.execute_reply":"2026-01-15T09:31:10.352565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_params = {'n_estimators': 4268,\n 'learning_rate': 0.0671671420012647,\n 'num_leaves': 98,\n 'max_depth': 12,\n 'min_child_samples': 45,\n 'bagging_freq': 7,\n 'subsample_freq' : 1,\n 'verbosity' : -1,\n 'verbose' : -1,\n 'subsample': 0.8587749845586551,\n 'colsample_bytree': 0.8229400801768992,\n 'reg_alpha': 2.854486192155767,\n 'reg_lambda': 1.9333719080554501}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:32:37.57775Z","iopub.execute_input":"2026-01-15T09:32:37.578358Z","iopub.status.idle":"2026-01-15T09:32:37.582061Z","shell.execute_reply.started":"2026-01-15T09:32:37.578325Z","shell.execute_reply":"2026-01-15T09:32:37.581412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# optuna parameter + CV \n\noof_preds = np.zeros(len(train_d))\nlgb_preds = np.zeros(len(test_d))\nfold_rmsle = []\n\nn_splits = 5\nsk = KFold(n_splits = n_splits,shuffle = True,random_state = 42)\n\nfor fold,(tr_idx,val_idx) in enumerate (sk.split(x,y),1):\n    x_train,y_train = x.iloc[tr_idx], y.iloc[tr_idx]\n    x_val,y_val = x.iloc[val_idx],y.iloc[val_idx]\n\n    model = LGBMRegressor(\n        **best_params,\n        objective = 'regression',\n        device = 'gpu',\n        random_state = 42)\n\n\n    model.fit(\n        x_train,y_train,\n        eval_set = [(x_val,y_val)],\n        eval_metric = 'rmse',\n        callbacks = [lgb.early_stopping(300)]\n        \n    )\n\n    val_preds = model.predict(x_val)\n    oof_preds[val_idx] = val_preds\n    lgb_preds += model.predict(test_d)/n_splits\n\n    y_preds_log = np.expm1(val_preds)\n    y_true = np.expm1(y_val)\n    \n    rmsle = root_mean_squared_log_error(y_true,y_preds_log)\n    fold_rmsle.append(rmsle)\n    print(f\"FOLD: {fold} RMSLE : {rmsle:.5f}\")\n\noverall_rmse = root_mean_squared_log_error(\n    np.expm1(y),\n    np.expm1(oof_preds))\nprint(\"FOLD RMSLE: \",[round(s,4) for s in fold_rmsle])\nprint(f\"Overall RMSLE: {overall_rmse:.4f}\" )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:32:38.279379Z","iopub.execute_input":"2026-01-15T09:32:38.28009Z","iopub.status.idle":"2026-01-15T09:35:25.560964Z","shell.execute_reply.started":"2026-01-15T09:32:38.280061Z","shell.execute_reply":"2026-01-15T09:35:25.560188Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h2 style=\"\n    font-size: 32px;\n    font-weight: 900;\n    color: #ffffff;\n    padding: 14px 22px;\n    background: linear-gradient(135deg, #ff8800, #ff5e62);\n    border-radius: 10px;\n    letter-spacing: 1px;\n    box-shadow: 0 6px 18px rgba(255, 94, 98, 0.35);\n    text-transform: uppercase;\n\">\nSubmission\n</h2>\n","metadata":{}},{"cell_type":"code","source":"submission['Premium Amount'] = np.expm1(lgb_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:35:40.091852Z","iopub.execute_input":"2026-01-15T09:35:40.092166Z","iopub.status.idle":"2026-01-15T09:35:40.09979Z","shell.execute_reply.started":"2026-01-15T09:35:40.092138Z","shell.execute_reply":"2026-01-15T09:35:40.098915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:35:40.53098Z","iopub.execute_input":"2026-01-15T09:35:40.531591Z","iopub.status.idle":"2026-01-15T09:35:40.538311Z","shell.execute_reply.started":"2026-01-15T09:35:40.531564Z","shell.execute_reply":"2026-01-15T09:35:40.537708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv',index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-15T09:35:41.038829Z","iopub.execute_input":"2026-01-15T09:35:41.039316Z","iopub.status.idle":"2026-01-15T09:35:42.465056Z","shell.execute_reply.started":"2026-01-15T09:35:41.039292Z","shell.execute_reply":"2026-01-15T09:35:42.464333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}