{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport numpy as np\nimport plotly.express as px\nfrom IPython.display import display, HTML\nimport warnings\nfrom colorama import Fore, Style\n\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, QuantileTransformer, OneHotEncoder, LabelEncoder\nfrom sklearn.metrics import (\n    accuracy_score,\n    classification_report,\n    precision_score,\n    recall_score\n)\nfrom scipy.stats import randint\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_log_error\nimport optuna\n\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T07:03:29.054346Z","iopub.execute_input":"2024-12-08T07:03:29.056336Z","iopub.status.idle":"2024-12-08T07:03:32.398039Z","shell.execute_reply.started":"2024-12-08T07:03:29.056269Z","shell.execute_reply":"2024-12-08T07:03:32.397338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv') # train dataset\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv') # test dataset\nsubmission  = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T07:03:32.399817Z","iopub.execute_input":"2024-12-08T07:03:32.400624Z","iopub.status.idle":"2024-12-08T07:03:40.745452Z","shell.execute_reply.started":"2024-12-08T07:03:32.40058Z","shell.execute_reply":"2024-12-08T07:03:40.744442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target = 'Premium Amount'\nid_col = 'id'\n\nX = train_df.drop(columns=[target, id_col])\ny = train_df[target]\ntest_df = test_df.drop(columns=[id_col])\n\ncategorical_features = X.select_dtypes(include='object').columns.tolist()\nnumerical_features = X.select_dtypes(exclude='object').columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T07:03:40.746458Z","iopub.execute_input":"2024-12-08T07:03:40.746725Z","iopub.status.idle":"2024-12-08T07:03:41.18903Z","shell.execute_reply.started":"2024-12-08T07:03:40.746699Z","shell.execute_reply":"2024-12-08T07:03:41.188281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def styled_heading(text):\n    return f\"\"\"\n    <div style=\"background-color:#f9f9f9; padding:20px; border:1px solid #e0e0e0; border-radius:8px; margin-bottom:20px; box-shadow: 0px 4px 6px rgba(0, 0, 0, 0.1);\">\n        <h1 style=\"color:#4CAF50; text-align:center; font-size:28px; font-weight:bold; margin:0;\">{text}</h1>\n    </div>\n    \"\"\"\n\ndef print_error(message):\n    error_style = \"\"\"\n    <div style=\"color:#ff4d4d; background-color:#ffe6e6; border:1px solid #ffcccc; padding:15px; border-radius:5px; margin:20px 0;\">\n        <strong>Error:</strong> {message}\n    </div>\n    \"\"\"\n    display(HTML(error_style.format(message=message)))\n\n# Helper function to generate colored horizontal line\ndef colored_line(color='#d9d9d9'):\n    return f\"\"\"\n    <hr style=\"border: none; height: 2px; background-color: {color}; margin: 20px 0;\">\n    \"\"\"\n\ndef print_dataset_analysis(train_dataset, test_dataset, n_top=5, heading_color='#4CAF50', line_color='#d9d9d9'):\n    try:\n        # Printing top values\n        train_heading = styled_heading(f\"Top {n_top} rows of Training Dataset\")\n        test_heading = styled_heading(f\"Top {n_top} rows of Test Dataset\")\n\n        display(HTML(colored_line(line_color)))\n        display(HTML(train_heading))\n        display(HTML(train_dataset.head(n_top).to_html(index=False, border=0, classes='table table-hover')))\n\n        display(HTML(colored_line(line_color)))\n        display(HTML(test_heading))\n        display(HTML(test_dataset.head(n_top).to_html(index=False, border=0, classes='table table-hover')))\n        \n        # Printing dataset summary\n        summary_heading = styled_heading(\"Summary of Training Dataset\")\n        display(HTML(colored_line(line_color)))\n        display(HTML(summary_heading))\n        display(HTML(train_dataset.describe().to_html(border=0, classes='table table-striped')))\n\n        # Printing null values\n        null_heading = styled_heading(\"Null Values in Datasets\")\n        train_null_count = train_dataset.isnull().sum()\n        test_null_count = test_dataset.isnull().sum()\n\n        display(HTML(colored_line(line_color)))\n        display(HTML(null_heading))\n        \n        display(HTML(\"<h3>Training Dataset:</h3>\"))\n        if train_null_count.sum() == 0:\n            display(HTML(\"<p style='color:#4CAF50;'>No null values in the training dataset.</p>\"))\n        else:\n            display(HTML(train_null_count[train_null_count > 0].to_frame().to_html(border=0, classes='table table-bordered')))\n\n        display(HTML(\"<h3>Test Dataset:</h3>\"))\n        if test_null_count.sum() == 0:\n            display(HTML(\"<p style='color:#4CAF50;'>No null values in the test dataset.</p>\"))\n        else:\n            display(HTML(test_null_count[test_null_count > 0].to_frame().to_html(border=0, classes='table table-bordered')))\n\n        # Printing duplicate values\n        duplicate_heading = styled_heading(\"Duplicate Values in Datasets\")\n        train_duplicates = train_dataset.duplicated().sum()\n        test_duplicates = test_dataset.duplicated().sum()\n\n        display(HTML(colored_line(line_color)))\n        display(HTML(duplicate_heading))\n        display(HTML(f\"<p><strong>Training Dataset:</strong> {train_duplicates} duplicate rows</p>\"))\n        display(HTML(f\"<p><strong>Test Dataset:</strong> {test_duplicates} duplicate rows</p>\"))\n\n        # Printing number of rows and columns\n        shape_heading = styled_heading(\"Number of Rows and Columns\")\n        display(HTML(colored_line(line_color)))\n        display(HTML(shape_heading))\n        display(HTML(f\"<p><strong>Training Dataset:</strong> Rows: {train_dataset.shape[0]}, Columns: {train_dataset.shape[1]}</p>\"))\n        display(HTML(f\"<p><strong>Test Dataset:</strong> Rows: {test_dataset.shape[0]}, Columns: {test_dataset.shape[1]}</p>\"))\n\n    except Exception as e:\n        print_error(str(e))\n\ndef print_unique_values(dataset, heading_color='#4CAF50', line_color='#d9d9d9'):\n    try:\n        unique_values_heading = styled_heading(\"Unique Values in Dataset\")\n        \n        display(HTML(colored_line(line_color)))\n        display(HTML(unique_values_heading))\n        \n        unique_values_table = \"\"\"\n        <table style=\"width:100%; border-collapse:collapse; text-align:left; border:1px solid #e0e0e0;\">\n            <thead>\n                <tr style=\"background-color:#f2f2f2; border-bottom:2px solid #d9d9d9;\">\n                    <th style=\"padding:8px;\">Column Name</th>\n                    <th style=\"padding:8px;\">Data Type</th>\n                    <th style=\"padding:8px;\">Unique Values</th>\n                </tr>\n            </thead>\n            <tbody>\n        \"\"\"\n        for column in dataset.columns:\n            unique_values = dataset[column].unique()[:5]  # Taking up to 5 unique values for preview\n            unique_values_str = ', '.join(map(str, unique_values))\n            data_type = dataset[column].dtype\n            unique_values_table += f\"\"\"\n            <tr>\n                <td style=\"padding:8px;\">{column}</td>\n                <td style=\"padding:8px;\">{data_type}</td>\n                <td style=\"padding:8px;\">{unique_values_str}</td>\n            </tr>\n            \"\"\"\n\n        unique_values_table += \"</tbody></table>\"\n        display(HTML(unique_values_table))\n    \n    except Exception as e:\n        print_error(str(e))\n","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-08T07:03:41.19038Z","iopub.execute_input":"2024-12-08T07:03:41.190721Z","iopub.status.idle":"2024-12-08T07:03:41.204488Z","shell.execute_reply.started":"2024-12-08T07:03:41.190692Z","shell.execute_reply":"2024-12-08T07:03:41.203636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_dataset_analysis(train_df, test_df)\nprint_unique_values(train_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T07:03:41.206922Z","iopub.execute_input":"2024-12-08T07:03:41.207506Z","iopub.status.idle":"2024-12-08T07:03:45.718684Z","shell.execute_reply.started":"2024-12-08T07:03:41.207468Z","shell.execute_reply":"2024-12-08T07:03:45.717781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_numeric_distribution(df_train):\n    # Define all numeric columns to plot\n    numeric_columns = [\n        'Age',  'Annual Income',\n         'Number of Dependents',\n       'Health Score', 'Vehicle Age','Credit Score'\n    ]\n\n    # Adjust layout parameters\n    background_color = '#f9f9f9'\n    sns.set_style(\"whitegrid\", {\"axes.facecolor\": background_color})\n    \n    num_plots = len(numeric_columns)\n    num_rows = (num_plots + 1) // 2  # Adjust number of rows for subplot layout\n    num_cols = 2  # Set number of columns for subplot layout\n    \n    # Create subplots\n    fig, axs = plt.subplots(num_rows, num_cols, figsize=(15, 5 * num_rows))\n    axs = axs.flatten()\n    \n    # Plot each numeric column\n    for i, col in enumerate(numeric_columns):\n        # Histogram with KDE plot\n        p = sns.histplot(df_train[col], bins=20, kde=True, ax=axs[i], palette='magma', edgecolor='black', linewidth=1)\n        p.set_title(f\"Distribution of {col}\")\n        p.set_xlabel(col)\n        p.set_ylabel(\"Frequency\")\n        p.grid(True, linestyle='--', linewidth=0.5)\n    \n    # Hide any unused subplots\n    for j in range(i + 1, len(axs)):\n        fig.delaxes(axs[j])\n    \n    plt.tight_layout()\n    plt.show()\n\n# Usage example:\nplot_numeric_distribution(train_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T07:03:45.719792Z","iopub.execute_input":"2024-12-08T07:03:45.720142Z","iopub.status.idle":"2024-12-08T07:04:12.959199Z","shell.execute_reply.started":"2024-12-08T07:03:45.720113Z","shell.execute_reply":"2024-12-08T07:04:12.958407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X[categorical_features] = X[categorical_features].fillna('missing')\nX[numerical_features] = X[numerical_features].fillna(X[numerical_features].median())\n\ntest_df[categorical_features] = test_df[categorical_features].fillna('missing')\ntest_df[numerical_features] = test_df[numerical_features].fillna(test_df[numerical_features].median())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T07:04:12.96047Z","iopub.execute_input":"2024-12-08T07:04:12.961Z","iopub.status.idle":"2024-12-08T07:04:15.62932Z","shell.execute_reply.started":"2024-12-08T07:04:12.96096Z","shell.execute_reply":"2024-12-08T07:04:15.628591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_model():\n    return CatBoostRegressor(\n        iterations=5000,\n        learning_rate=0.04591420079381261,\n        depth=10,\n        loss_function=\"RMSE\",\n        eval_metric=\"RMSE\",\n        cat_features=categorical_features,\n        random_seed=42,\n        task_type=\"GPU\",\n        devices=\"0,1\",\n        verbose=100\n    )\n\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\nmodels = []\ncv_scores = []\n\nfor train_idx, valid_idx in kf.split(X):\n    train_data = Pool(X.iloc[train_idx], np.log1p(y.iloc[train_idx]), cat_features=categorical_features)\n    valid_data = Pool(X.iloc[valid_idx], np.log1p(y.iloc[valid_idx]), cat_features=categorical_features)\n    \n    model = create_model()\n    model.fit(train_data, eval_set=valid_data, early_stopping_rounds=200, verbose=100)\n    models.append(model)\n    \n    valid_preds = model.predict(valid_data)\n    rmsle = np.sqrt(((np.log1p(np.maximum(0, np.expm1(valid_preds))) - np.log1p(np.expm1(y.iloc[valid_idx])))**2).mean())\n    cv_scores.append(rmsle)\n\nprint(f\"Cross-Validation RMSLE Scores: {cv_scores}\")\nprint(f\"Mean CV RMSLE: {np.mean(cv_scores)}\")\n\ntest_predictions = np.zeros(len(test_df))\nfor model in models:\n    test_predictions += np.maximum(0, np.expm1(model.predict(test_df))) / len(models)\n\ntest_predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T07:04:15.630291Z","iopub.execute_input":"2024-12-08T07:04:15.630522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nsubmission[target] = test_predictions\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}