{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":9178166,"sourceType":"datasetVersion","datasetId":5547076}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.simplefilter('ignore')\n\nimport pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error # Import for RMSLE calculation\nfrom itertools import combinations\nimport gc # Import the garbage collector\n\n# --- 1. Data Loading and Initial Preparation ---\n\n# Load the datasets\n# Make sure to adjust the path if you are not running this in a Kaggle environment\ntry:\n    train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\n    test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nexcept FileNotFoundError:\n    print(\"CSV files not found. Creating dummy data for demonstration.\")\n    # Create a dummy dataset if the original files are not available\n    def create_dummy_data(n_samples, name):\n        data = {}\n        data['id'] = range(n_samples)\n        CATS_dummy = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location', 'Policy Type', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']\n        NUMS_dummy = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\n        for col in CATS_dummy:\n            data[col] = np.random.choice([f'{col}_A', f'{col}_B', f'{col}_C'], n_samples)\n        for col in NUMS_dummy:\n            data[col] = np.random.randint(0, 100, n_samples)\n        data['Policy Start Date'] = pd.to_datetime(pd.Timestamp('2023-01-01') + pd.to_timedelta(np.random.randint(0, 365*2, n_samples), 'D'))\n        if name == 'train':\n            data['Premium Amount'] = np.random.rand(n_samples) * 1000 + 50\n        return pd.DataFrame(data)\n    train = create_dummy_data(10000, 'train') # Increased data size for testing\n    test = create_dummy_data(5000, 'test')\n\n\nprint('Train Shape', train.shape)\nprint('Test Shape', test.shape)\n\n# Define the target variable\nTARGET = 'Premium Amount'\n\n# --- [MODIFIED] Log-transform the target variable for RMSLE optimization ---\nprint(f\"\\nApplying log1p transformation to the target variable: {TARGET}\")\ntrain[TARGET] = np.log1p(train[TARGET])\n\n\n# Function to decompose the date feature\ndef date_feat(df):\n  \"\"\"Decomposes the 'Policy Start Date' into time-based features.\"\"\"\n  df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n  df['Year'] = df['Policy Start Date'].dt.year\n  df['Month'] = df['Policy Start Date'].dt.month\n  df['Day'] = df['Policy Start Date'].dt.day\n  df['Dayofweek'] = df['Policy Start Date'].dt.dayofweek\n  return df\n\n# Apply the date feature decomposition\ntrain = date_feat(train)\ntest = date_feat(test)\n\n# Identify categorical and numerical features\nCATS = train.select_dtypes(include='object').columns.tolist()\nNUMS = [col for col in train.select_dtypes(include='number').columns.tolist() if col not in ['id', TARGET]]\nprint(len(CATS), 'Categoricals:',CATS, '\\n')\nprint(len(NUMS), 'Numericals:',NUMS, '\\n')\nprint('-->', len(NUMS+CATS), 'Features')\n\n\n# --- 2. Encoding Functions ---\n\ndef target_encode(train_df, valid_df, col, target=TARGET, kfold=5, smooth=20):\n    col_name = '_'.join(col) if isinstance(col, list) else col\n    \n    # --- Create OOF predictions for the training part ---\n    # Note: This implementation of target encoding is for the full dataset,\n    # not within the CV fold's training set. A true OOF TE would be more complex.\n    train_df['kfold'] = ((train_df.index) % kfold)\n    oof_preds = pd.Series(index=train_df.index, dtype=float)\n    \n    for i in range(kfold):\n        train_fold = train_df[train_df['kfold'] != i]\n        val_fold = train_df[train_df['kfold'] == i]\n        \n        global_mean = train_fold[target].mean()\n        agg = train_fold.groupby(col)[target].agg(['mean', 'count'])\n        smoothed_mean = (agg['mean'] * agg['count'] + global_mean * smooth) / (agg['count'] + smooth)\n        \n        oof_preds.loc[val_fold.index] = val_fold[col].map(smoothed_mean).fillna(global_mean)\n        \n    train_df[f'TE_{col_name}'] = oof_preds\n    train_df = train_df.drop('kfold', axis=1)\n\n    # --- Create mapping for the validation/test set ---\n    global_mean_full = train_df[target].mean()\n    agg_full = train_df.groupby(col)[target].agg(['mean', 'count'])\n    smoothed_mean_full = (agg_full['mean'] * agg_full['count'] + global_mean_full * smooth) / (agg_full['count'] + smooth)\n    \n    valid_df[f'TE_{col_name}'] = valid_df[col].map(smoothed_mean_full).fillna(global_mean_full)\n    \n    return train_df, valid_df\n\ndef count_encode(train_df, valid_df, col):\n    col_name = '_'.join(col) if isinstance(col, list) else col\n    counts = train_df[col].value_counts()\n    train_df[f'CE_{col_name}'] = train_df[col].map(counts).fillna(0)\n    valid_df[f'CE_{col_name}'] = valid_df[col].map(counts).fillna(0)\n    return train_df, valid_df\n\n# --- 3 & 4. Memory-Efficient Feature Engineering and CV ---\n\n# KFold setup\nNFOLDS = 5\nkf = KFold(n_splits=NFOLDS, shuffle=True, random_state=42)\n\n# Placeholders for predictions (will store log-transformed values)\noof_preds = np.zeros(train.shape[0])\ntest_preds = np.zeros(test.shape[0])\nbest_iterations = []\n\n# LGBM parameters\nlgbm_params = {\n    'objective': 'regression_l1', 'metric': 'rmse', 'n_estimators': 2000,\n    'learning_rate': 0.01, 'feature_fraction': 0.8, 'bagging_fraction': 0.8,\n    'bagging_freq': 1, 'lambda_l1': 0.1, 'lambda_l2': 0.1,\n    'num_leaves': 31, 'verbose': -1, 'n_jobs': -1, 'seed': 42,\n    'boosting_type': 'gbdt',\n}\n\nfeatures_to_combine = NUMS + CATS\n\n# --- Start CV Loop ---\nprint(f\"\\nStarting {NFOLDS}-Fold CV with on-the-fly feature engineering...\")\nfor fold, (train_index, val_index) in enumerate(kf.split(train, train[TARGET])):\n    print(f\"===== FOLD {fold+1} =====\")\n    \n    X_train, X_val = train.iloc[train_index].copy(), train.iloc[val_index].copy()\n    y_train, y_val = train[TARGET].iloc[train_index], train[TARGET].iloc[val_index]\n    test_fold = test.copy()\n\n    encoded_cols = []\n    \n    # On-the-fly feature generation and encoding\n    print(\"Generating and encoding features for this fold...\")\n    for col1, col2 in combinations(features_to_combine, 2):\n        feature_name = f'{col1}_{col2}'\n        \n        # Create temporary interaction feature\n        X_train[feature_name] = X_train[col1].astype(str) + '_' + X_train[col2].astype(str)\n        X_val[feature_name] = X_val[col1].astype(str) + '_' + X_val[col2].astype(str)\n        test_fold[feature_name] = test_fold[col1].astype(str) + '_' + test_fold[col2].astype(str)\n        \n        # Apply encoding\n        X_train, X_val = count_encode(X_train, X_val, feature_name)\n        _, test_fold = count_encode(X_train, test_fold, feature_name)\n        \n        X_train, X_val = target_encode(X_train, X_val, feature_name, target=TARGET)\n        _, test_fold = target_encode(X_train, test_fold, feature_name, target=TARGET)\n\n        # Drop the temporary interaction column to save memory\n        X_train = X_train.drop(feature_name, axis=1)\n        X_val = X_val.drop(feature_name, axis=1)\n        test_fold = test_fold.drop(feature_name, axis=1)\n        \n        # Keep track of the newly created encoded columns\n        encoded_cols.extend([f'CE_{feature_name}', f'TE_{feature_name}'])\n        \n    print(f\"Finished encoding. Total encoded features: {len(encoded_cols)}\")\n    \n    model_features = NUMS + encoded_cols\n    \n    model = lgb.LGBMRegressor(**lgbm_params)\n    model.fit(X_train[model_features], y_train,\n              eval_set=[(X_val[model_features], y_val)],\n              eval_metric='rmse',\n              callbacks=[lgb.early_stopping(100, verbose=False)])\n    \n    val_preds = model.predict(X_val[model_features])\n    oof_preds[val_index] = val_preds\n    test_preds += model.predict(test_fold[model_features]) / NFOLDS\n    best_iterations.append(model.best_iteration_)\n    \n    # --- [MODIFIED] Calculate and display RMSLE for the fold ---\n    # RMSE on log-transformed values is equivalent to RMSLE on original values\n    fold_rmsle = np.sqrt(mean_squared_error(y_val, val_preds))\n    print(f\"Fold {fold+1} RMSLE: {fold_rmsle:.5f}\")\n    print(f\"Fold {fold+1} Best Iteration: {model.best_iteration_}\")\n    \n    # Clean up memory\n    del X_train, X_val, test_fold, model\n    gc.collect()\n\n\n# --- 5. Final Model Training and Prediction ---\n\nfinal_iterations = int(np.mean(best_iterations) * 1.2)\nprint(f\"\\nAverage best iteration: {np.mean(best_iterations):.0f}\")\nprint(f\"Training final model with {final_iterations} iterations...\")\n\n# We need to re-create the encoded features on the full dataset for the final model\ntrain_full = train.copy()\ntest_full = test.copy()\nencoded_cols_final = []\n\nprint(\"Generating and encoding features for the final model...\")\nfor col1, col2 in combinations(features_to_combine, 2):\n    feature_name = f'{col1}_{col2}'\n\n    # Create temporary interaction feature\n    train_full[feature_name] = train_full[col1].astype(str) + '_' + train_full[col2].astype(str)\n    test_full[feature_name] = test_full[col1].astype(str) + '_' + test_full[col2].astype(str)\n\n    # Apply encoding (using the whole training data)\n    train_full, test_full = count_encode(train_full, test_full, feature_name)\n    train_full, test_full = target_encode(train_full, test_full, feature_name, target=TARGET)\n\n    # Drop temporary column\n    train_full = train_full.drop(feature_name, axis=1)\n    test_full = test_full.drop(feature_name, axis=1)\n\n    encoded_cols_final.extend([f'CE_{feature_name}', f'TE_{feature_name}'])\n\nprint(f\"Finished encoding. Total encoded features: {len(encoded_cols_final)}\")\n\nfinal_model_features = NUMS + encoded_cols_final\nlgbm_params['n_estimators'] = final_iterations\n\nfinal_model = lgb.LGBMRegressor(**lgbm_params)\nfinal_model.fit(train_full[final_model_features], train_full[TARGET])\nfinal_test_predictions = final_model.predict(test_full[final_model_features])\n\n\n# --- 6. Save Results ---\n\n# --- [MODIFIED] Inverse transform predictions before saving ---\n# Apply expm1 to convert log-predictions back to the original scale\noof_df = pd.DataFrame({'id': train['id'], TARGET: np.expm1(oof_preds)})\noof_filename = 'oof_2way_te_lgbm_rmsle.csv'\noof_df.to_csv(oof_filename, index=False)\nprint(f\"\\nOOF predictions saved to {oof_filename}\")\n\ntest_df = pd.DataFrame({'id': test['id'], TARGET: np.expm1(final_test_predictions)})\ntest_filename = 'test_2way_te_lgbm_rmsle.csv'\ntest_df.to_csv(test_filename, index=False)\nprint(f\"Test predictions saved to {test_filename}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-19T00:18:55.011724Z","iopub.execute_input":"2025-08-19T00:18:55.012061Z","execution_failed":"2025-08-19T00:25:59.155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}