{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":9178166,"sourceType":"datasetVersion","datasetId":5547076}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.simplefilter('ignore')\n\nimport pandas as pd\nimport numpy as np\nimport xgboost as xgb\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error\nimport gc\n\n# --- 1. Data Loading and Initial Preparation ---\n\n# Load the datasets\n# Make sure to adjust the path if you are not running this in a Kaggle environment\ntry:\n    train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\n    test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nexcept FileNotFoundError:\n    print(\"CSV files not found. Creating dummy data for demonstration.\")\n    # Create a dummy dataset if the original files are not available\n    def create_dummy_data(n_samples, name):\n        data = {}\n        data['id'] = range(n_samples)\n        CATS_dummy = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location', 'Policy Type', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']\n        NUMS_dummy = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\n        for col in CATS_dummy:\n            data[col] = np.random.choice([f'{col}_A', f'{col}_B', f'{col}_C'], n_samples)\n        for col in NUMS_dummy:\n            data[col] = np.random.randint(0, 100, n_samples)\n        data['Policy Start Date'] = pd.to_datetime(pd.Timestamp('2023-01-01') + pd.to_timedelta(np.random.randint(0, 365*2, n_samples), 'D'))\n        if name == 'train':\n            data['Premium Amount'] = np.random.rand(n_samples) * 1000 + 50\n        return pd.DataFrame(data)\n    train = create_dummy_data(1000, 'train')\n    test = create_dummy_data(500, 'test')\n\nprint('Train Shape', train.shape)\nprint('Test Shape', test.shape)\n\n# Define the target variable\nTARGET = 'Premium Amount'\n\n# Apply log1p transformation for RMSLE optimization\nprint(f\"\\nApplying log1p transformation to the target variable: {TARGET}\")\ntrain[TARGET] = np.log1p(train[TARGET])\n\n# Function to decompose the date feature\ndef date_feat(df):\n  \"\"\"Decomposes the 'Policy Start Date' into time-based features.\"\"\"\n  df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n  df['Year'] = df['Policy Start Date'].dt.year\n  df['Month'] = df['Policy Start Date'].dt.month\n  df['Day'] = df['Policy Start Date'].dt.day\n  df['Dayofweek'] = df['Policy Start Date'].dt.dayofweek\n  return df\n\n# Apply the date feature decomposition\ntrain = date_feat(train)\ntest = date_feat(test)\n\n# --- 2. Feature and Model Preparation ---\n\n# Identify categorical and numerical features\nCATS = train.select_dtypes(include='object').columns.tolist()\nNUMS = [col for col in train.select_dtypes(include='number').columns.tolist() if col not in ['id', TARGET]]\n\n# --- [MODIFIED] Convert object columns to 'category' dtype for XGBoost ---\nprint(\"\\nConverting object columns to 'category' dtype...\")\nfor col in CATS:\n    train[col] = train[col].astype('category')\n    test[col] = test[col].astype('category')\n\n# The feature set now uses the original column names\nmodel_features = NUMS + CATS\nprint(f\"Using {len(model_features)} features, including native categorical features.\")\n\n\n# --- 3. Cross-Validation, Modeling, and Prediction ---\n\n# KFold setup\nNFOLDS = 5\nkf = KFold(n_splits=NFOLDS, shuffle=True, random_state=42)\n\n# Placeholders for predictions\noof_preds = np.zeros(train.shape[0])\ntest_preds = np.zeros(test.shape[0])\n\n# XGBoost parameters\nxgb_params = {\n    'objective': 'reg:squarederror',\n    'eval_metric': 'rmse',\n    'n_estimators': 2000,\n    'learning_rate': 0.01,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'max_depth': 5,\n    'gamma': 0.1,\n    'n_jobs': -1,\n    'seed': 42,\n    'tree_method': 'hist',\n    'early_stopping_rounds': 100,\n    'enable_categorical': True  # --- [MODIFIED] Enable native categorical feature support ---\n}\n\n# --- Start CV Loop ---\nprint(f\"\\nStarting {NFOLDS}-Fold CV for the XGBoost model...\")\nfor fold, (train_index, val_index) in enumerate(kf.split(train, train[TARGET])):\n    print(f\"===== FOLD {fold+1} =====\")\n    \n    X_train, X_val = train.iloc[train_index], train.iloc[val_index]\n    y_train, y_val = train[TARGET].iloc[train_index], train[TARGET].iloc[val_index]\n    \n    model = xgb.XGBRegressor(**xgb_params)\n    # Pass the feature-prepared dataframes directly\n    model.fit(X_train[model_features], y_train,\n              eval_set=[(X_val[model_features], y_val)],\n              verbose=False)\n    \n    val_preds = model.predict(X_val[model_features])\n    oof_preds[val_index] = val_preds\n    test_preds += model.predict(test[model_features]) / NFOLDS\n    \n    # Calculate and display RMSLE for the fold\n    fold_rmsle = np.sqrt(mean_squared_error(y_val, val_preds))\n    print(f\"Fold {fold+1} RMSLE: {fold_rmsle:.5f}\")\n    \n    del X_train, X_val, y_train, y_val, model\n    gc.collect()\n\n# --- 4. Save Results ---\n\n# Apply expm1 to convert log-predictions back to the original scale\noof_df = pd.DataFrame({'id': train['id'], TARGET: np.expm1(oof_preds)})\noof_filename = 'oof_basic_xgb.csv'\noof_df.to_csv(oof_filename, index=False)\nprint(f\"\\nOOF predictions for the basic XGBoost model saved to {oof_filename}\")\n\ntest_df = pd.DataFrame({'id': test['id'], TARGET: np.expm1(test_preds)})\ntest_filename = 'test_basic_xgb.csv'\ntest_df.to_csv(test_filename, index=False)\nprint(f\"Test predictions for the basic XGBoost model saved to {test_filename}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}