{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Here I provided a class to handle the missing and categorical values and create a pipeline with your wishes model","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, RandomizedSearchCV\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler, PowerTransformer\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor, AdaBoostRegressor\nfrom sklearn.linear_model import LinearRegression, Ridge, Lasso\nfrom sklearn.svm import SVR\nfrom sklearn.feature_selection import SelectKBest, f_regression\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\n\nclass RegressionPipeline:\n    def __init__(self, model=None):\n        self.pipeline = None\n        self.preprocessor = None\n        self.model = model if model else RandomForestRegressor(random_state=42)\n\n    def handle_missing_values(self, strategy='mean', knn_neighbors=5):\n        \"\"\"\n        Handle missing values with a given strategy.\n        Options: 'mean', 'median', 'most_frequent', 'constant', or 'knn'.\n        \"\"\"\n        if strategy == 'knn':\n            self.numeric_imputer = KNNImputer(n_neighbors=knn_neighbors)\n        else:\n            self.numeric_imputer = SimpleImputer(strategy=strategy)\n        self.categorical_imputer = SimpleImputer(strategy='most_frequent')\n\n    def handle_categorical_data(self, strategy='onehot'):\n        \"\"\"\n        Encode categorical data.\n        Options: 'onehot' (default).\n        \"\"\"\n        if strategy == 'onehot':\n            self.categorical_encoder = OneHotEncoder(handle_unknown='ignore')\n\n    def preprocess_data(self, X, feature_selection=False, k_best=10):\n        \"\"\"\n        Create preprocessing pipeline for numerical and categorical features.\n        Optionally apply feature selection.\n        \"\"\"\n        numeric_features = X.select_dtypes(include=['int64', 'float64']).columns\n        categorical_features = X.select_dtypes(include=['object', 'category']).columns\n\n        # Define transformers\n        numeric_transformer = Pipeline(steps=[\n            ('imputer', self.numeric_imputer),\n            ('scaler', StandardScaler()),\n            ('power_transform', PowerTransformer(method='yeo-johnson'))\n        ])\n\n        categorical_transformer = Pipeline(steps=[\n            ('imputer', self.categorical_imputer),\n            ('encoder', self.categorical_encoder)\n        ])\n\n        # Combine transformers into a preprocessor\n        self.preprocessor = ColumnTransformer(\n            transformers=[\n                ('num', numeric_transformer, numeric_features),\n                ('cat', categorical_transformer, categorical_features)\n            ]\n        )\n\n        if feature_selection:\n            self.preprocessor = Pipeline(steps=[\n                ('preprocessor', self.preprocessor),\n                ('feature_selection', SelectKBest(score_func=f_regression, k=k_best))\n            ])\n\n    def handle_outliers(self, X, method='zscore', threshold=3):\n        \"\"\"\n        Handle outliers in the dataset.\n        Options: 'zscore' or 'iqr'.\n        \"\"\"\n        if method == 'zscore':\n            numeric_features = X.select_dtypes(include=['int64', 'float64']).columns\n            z_scores = np.abs((X[numeric_features] - X[numeric_features].mean()) / X[numeric_features].std())\n            X = X[(z_scores < threshold).all(axis=1)]\n        elif method == 'iqr':\n            numeric_features = X.select_dtypes(include=['int64', 'float64']).columns\n            Q1 = X[numeric_features].quantile(0.25)\n            Q3 = X[numeric_features].quantile(0.75)\n            IQR = Q3 - Q1\n            X = X[~((X[numeric_features] < (Q1 - 1.5 * IQR)) | (X[numeric_features] > (Q3 + 1.5 * IQR))).any(axis=1)]\n        return X\n\n    def reduce_cardinality(self, X, column, top_n=100):\n        \"\"\"\n        Reduce cardinality of a high-cardinality categorical column.\n        \"\"\"\n        if column in X.columns:\n            top_values = X[column].value_counts().nlargest(top_n).index\n            X[column] = X[column].where(X[column].isin(top_values), 'Other')\n        return X\n\n    def process_date_column(self, X, column):\n        \"\"\"\n        Convert a date column into numeric features (year and month).\n        \"\"\"\n        if column in X.columns:\n            X[column] = pd.to_datetime(X[column], errors='coerce')\n            X[f'{column}_Year'] = X[column].dt.year\n            X[f'{column}_Month'] = X[column].dt.month\n            X.drop(columns=[column], inplace=True)\n        return X\n\n    def tune_hyperparameters(self, X_train, y_train, param_distributions, n_iter=50, cv=3, scoring='neg_mean_squared_error', random_state=42):\n        \"\"\"\n        Optimize hyperparameters for CatBoostRegressor using RandomizedSearchCV.\n        \"\"\"\n        search = RandomizedSearchCV(\n            estimator=self.model,\n            param_distributions=param_distributions,\n            n_iter=n_iter,\n            scoring=scoring,\n            cv=cv,\n            random_state=random_state,\n            n_jobs=-1,\n            verbose=1\n        )\n        search.fit(X_train, y_train)\n        self.model = search.best_estimator_\n        print(\"Best Parameters:\", search.best_params_)\n        print(\"Best Score:\", search.best_score_)\n        return search.best_params_\n\n    def create_pipeline(self):\n        \"\"\"\n        Create a complete pipeline with preprocessing and model.\n        \"\"\"\n        self.pipeline = Pipeline(steps=[\n            ('preprocessor', self.preprocessor),\n            ('model', self.model)\n        ])\n\n    def fit(self, X_train, y_train):\n        \"\"\"\n        Fit the pipeline to the training data.\n        \"\"\"\n        self.pipeline.fit(X_train, y_train)\n\n    def predict(self, X):\n        \"\"\"\n        Make predictions on new data.\n        \"\"\"\n        return self.pipeline.predict(X)\n\n    def evaluate(self, X_val, y_val):\n        \"\"\"\n        Evaluate the model using RMSLE.\n        \"\"\"\n        predictions = self.predict(X_val)\n        return self.rmsle(y_val, predictions)\n\n    @staticmethod\n    def rmsle(y_true, y_pred):\n        \"\"\"\n        Compute the Root Mean Squared Logarithmic Error (RMSLE).\n        \"\"\"\n        y_true = np.clip(y_true, 0, None)  # Ensure no negative values\n        y_pred = np.clip(y_pred, 0, None)\n        return np.sqrt(np.mean(np.square(np.log1p(y_true) - np.log1p(y_pred))))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# If you want to do some EDA here is the code below:","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nclass DataEDA:\n    @staticmethod\n    def describe_data(df):\n        print(\"Dataset Information:\")\n        print(df.info())\n        print(\"\\nSummary Statistics:\")\n        print(df.describe())\n        print(\"\\nMissing Values:\")\n        print(df.isnull().sum())\n    \n    @staticmethod\n    def visualize_missing_data(df):\n        missing = df.isnull().mean()\n        missing = missing[missing > 0]\n        if not missing.empty:\n            plt.figure(figsize=(10, 5))\n            missing.sort_values().plot(kind='bar')\n            plt.title(\"Missing Data Proportion\")\n            plt.show()\n        else:\n            print(\"No missing values to visualize.\")\n    \n    @staticmethod\n    def plot_distributions(df):\n        numeric_features = df.select_dtypes(include=['int64', 'float64']).columns\n        for col in numeric_features:\n            plt.figure(figsize=(10, 5))\n            sns.histplot(df[col], kde=True, bins=30)\n            plt.title(f\"Distribution of {col}\")\n            plt.show()\n    \n    @staticmethod\n    def visualize_correlations(df, target=None):\n        plt.figure(figsize=(12, 8))\n        correlation_matrix = df.corr()\n        sns.heatmap(correlation_matrix, annot=True, fmt='.2f', cmap='coolwarm')\n        plt.title(\"Feature Correlation Matrix\")\n        plt.show()\n        if target and target in df.columns:\n            plt.figure(figsize=(8, 5))\n            sns.barplot(x=correlation_matrix[target].abs().sort_values(ascending=False).index, \n                        y=correlation_matrix[target].abs().sort_values(ascending=False).values)\n            plt.xticks(rotation=45, ha='right')\n            plt.title(f\"Correlation with {target}\")\n            plt.show()\n\n# Usage Example\n# eda = DataEDA()\n# eda.describe_data(train_df)\n# eda.visualize_missing_data(train_df)\n# eda.plot_distributions(train_df)\n# eda.visualize_correlations(train_df, target='Premium Amount')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"find your file path for the dataset","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\n\n# Define file paths\ntrain_path = r\"location for your train dataset\\train.csv\"\ntest_path = r\"C:location for your test dataset\\test.csv\"\nsubmission_path = r\"location for your submission dataset\\submission.csv\"\n\n# Load the datasets\ntrain_df = pd.read_csv(train_path)\ntest_df = pd.read_csv(test_path)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Do some EDA here","metadata":{}},{"cell_type":"code","source":"eda = DataEDA()\neda.describe_data(train_df)\neda.visualize_missing_data(train_df)\neda.plot_distributions(train_df)\neda.visualize_correlations(train_df, target='Premium Amount')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"split the data into X and y","metadata":{}},{"cell_type":"code","source":"# Separate features and target\nX = train_df.drop(columns=[\"id\", \"Premium Amount\"])  # Adjust 'Premium Amount' if the column name is different\ny = train_df[\"Premium Amount\"]\n\n# Split data into train and validation sets\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Preprocess the test dataset\nX_test = test_df.drop(columns=[\"id\"])  # Drop ID column for testing","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Do the calculations with the Catboost, it took less than 10 min with my laptop","metadata":{}},{"cell_type":"code","source":"pipeline = RegressionPipeline(model=CatBoostRegressor(verbose=0, random_state=42))\n# Reduce cardinality for 'Location' column\nX_train = pipeline.reduce_cardinality(X_train, column='Location', top_n=100)\nX_val = pipeline.reduce_cardinality(X_val, column='Location', top_n=100)\nX_test = pipeline.reduce_cardinality(X_test, column='Location', top_n=100)\n\n# Process 'Policy Start Date'\nX_train = pipeline.process_date_column(X_train, column='Policy Start Date')\nX_val = pipeline.process_date_column(X_val, column='Policy Start Date')\nX_test = pipeline.process_date_column(X_test, column='Policy Start Date')\n\n\n\n# Configure the pipeline steps\npipeline.handle_missing_values(strategy='mean')\npipeline.handle_categorical_data(strategy='onehot')\npipeline.preprocess_data(X_train)\n\n# Create the complete pipeline\npipeline.create_pipeline()\n\n# Fit and evaluate the model\npipeline.fit(X_train, y_train)\nrmsle_score = pipeline.evaluate(X_val, y_val)\nprint(\"RMSLE:\", rmsle_score)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"submit your submission","metadata":{}},{"cell_type":"code","source":"# Prepare test data for submission\nX_test = test_df.drop(columns=[\"id\"])\npredictions = pipeline.predict(X_test)\nsubmission = pd.DataFrame({\n    \"id\": test_df[\"id\"],\n    \"Premium Amount\": predictions\n})\nsubmission.to_csv(r\"C:location\\submission.csv\", index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}