{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Regression with an Insurance Dataset","metadata":{}},{"cell_type":"markdown","source":"## Introduction","metadata":{}},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom scipy import stats\n\n# File paths\nTRAIN_PATH = '/kaggle/input/playground-series-s4e12/train.csv'\nTEST_PATH = '/kaggle/input/playground-series-s4e12/test.csv'\nSUBMISSION_PATH = '/kaggle/input/playground-series-s4e12/sample_submission.csv'\n\ndef load_and_examine_data():\n    # Read training data\n    train_df = pd.read_csv(TRAIN_PATH)\n    test_df = pd.read_csv(TEST_PATH)\n    \n    print(\"\\n=== Dataset Shape ===\")\n    print(f\"Training set shape: {train_df.shape}\")\n    print(f\"Test set shape: {test_df.shape}\")\n    \n    print(\"\\n=== First Few Rows ===\")\n    print(train_df.head())\n    \n    print(\"\\n=== Data Info ===\")\n    print(train_df.info())\n    \n    return train_df, test_df\n\ntrain_df, test_df = load_and_examine_data()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-15T23:18:44.232885Z","iopub.execute_input":"2024-12-15T23:18:44.233258Z","iopub.status.idle":"2024-12-15T23:18:53.230902Z","shell.execute_reply.started":"2024-12-15T23:18:44.233224Z","shell.execute_reply":"2024-12-15T23:18:53.229717Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Preprocessing","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom datetime import datetime\n\ndef preprocess_data(train_df, test_df):\n    # Make copies to avoid modifying original data\n    train = train_df.copy()\n    test = test_df.copy()\n    \n    # Process datetime\n    for df in [train, test]:\n        df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n        df['Policy_Year'] = df['Policy Start Date'].dt.year\n        df['Policy_Month'] = df['Policy Start Date'].dt.month\n        df['Policy_Day'] = df['Policy Start Date'].dt.day\n        df.drop('Policy Start Date', axis=1, inplace=True)\n    \n    # Handle numerical missing values\n    numerical_cols = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n                     'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration']\n    \n    # Store medians for test set imputation\n    medians = {}\n    \n    for col in numerical_cols:\n        # Calculate median for each column\n        median_val = train[col].median()\n        medians[col] = median_val\n        \n        # Impute missing values without using inplace\n        train = train.assign(**{col: train[col].fillna(median_val)})\n        test = test.assign(**{col: test[col].fillna(median_val)})\n    \n    # Handle categorical missing values\n    categorical_cols = ['Gender', 'Marital Status', 'Education Level', 'Occupation', \n                       'Location', 'Policy Type', 'Customer Feedback', 'Smoking Status',\n                       'Exercise Frequency', 'Property Type']\n    \n    # Store modes for test set imputation\n    modes = {}\n    \n    for col in categorical_cols:\n        # Calculate mode for each column\n        mode_val = train[col].mode()[0]\n        modes[col] = mode_val\n        \n        # Impute missing values without using inplace\n        train = train.assign(**{col: train[col].fillna('Missing')})\n        test = test.assign(**{col: test[col].fillna('Missing')})\n    \n    # Create binary features for categorical variables\n    for col in categorical_cols:\n        # Get dummy variables\n        dummies = pd.get_dummies(pd.concat([train[col], test[col]]), prefix=col, dummy_na=False)\n        \n        # Split the dummy variables back into train and test\n        train_dummies = dummies[:len(train)]\n        test_dummies = dummies[len(train):]\n        \n        # Add dummy variables to datasets\n        train = pd.concat([train, train_dummies], axis=1)\n        test = pd.concat([test, test_dummies], axis=1)\n        \n        # Drop original categorical columns\n        train = train.drop(col, axis=1)\n        test = test.drop(col, axis=1)\n    \n    # Drop ID column as it's not useful for modeling\n    train_ids = train['id'].copy()\n    test_ids = test['id'].copy()\n    train = train.drop('id', axis=1)\n    test = test.drop('id', axis=1)\n    \n    print(\"=== Missing Values After Preprocessing ===\")\n    print(\"\\nTraining set:\")\n    print(train.isnull().sum().sum())\n    print(\"\\nTest set:\")\n    print(test.isnull().sum().sum())\n    \n    print(\"\\n=== Final Dataset Shapes ===\")\n    print(f\"Training set shape: {train.shape}\")\n    print(f\"Test set shape: {test.shape}\")\n    \n    return train, test, train_ids, test_ids, medians, modes\n\n# Load the data\ndef load_and_preprocess():\n    train_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\n    test_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\n    \n    # Preprocess the data\n    train_processed, test_processed, train_ids, test_ids, medians, modes = preprocess_data(train_df, test_df)\n    \n    # Save preprocessing parameters\n    preprocessing_params = {\n        'medians': medians,\n        'modes': modes\n    }\n    \n    return train_processed, test_processed, train_ids, test_ids, preprocessing_params\n\nif __name__ == \"__main__\":\n    train_processed, test_processed, train_ids, test_ids, preprocessing_params = load_and_preprocess()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T23:18:53.233035Z","iopub.execute_input":"2024-12-15T23:18:53.233435Z","iopub.status.idle":"2024-12-15T23:19:23.585832Z","shell.execute_reply.started":"2024-12-15T23:18:53.23339Z","shell.execute_reply":"2024-12-15T23:19:23.584689Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Relationship Analysis","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom scipy import stats\n\ndef analyze_target_distribution(df):\n    \"\"\"Analyze the distribution of the target variable\"\"\"\n    plt.figure(figsize=(15, 5))\n    \n    # Original distribution\n    plt.subplot(1, 2, 1)\n    sns.histplot(df['Premium Amount'], bins=50)\n    plt.title('Premium Amount Distribution')\n    plt.xlabel('Premium Amount')\n    \n    # Log-transformed distribution (since we're using RMSLE)\n    plt.subplot(1, 2, 2)\n    sns.histplot(np.log1p(df['Premium Amount']), bins=50)\n    plt.title('Log Premium Amount Distribution')\n    plt.xlabel('Log(Premium Amount + 1)')\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # Print summary statistics\n    print(\"\\n=== Target Variable Statistics ===\")\n    print(df['Premium Amount'].describe())\n    print(f\"\\nSkewness: {df['Premium Amount'].skew():.2f}\")\n    print(f\"Kurtosis: {df['Premium Amount'].kurtosis():.2f}\")\n\ndef analyze_numeric_relationships(df, target='Premium Amount'):\n    \"\"\"Analyze relationships between numeric features and target\"\"\"\n    # Select numeric columns excluding the target\n    numeric_cols = df.select_dtypes(include=['float64', 'int64']).columns\n    numeric_cols = [col for col in numeric_cols if col != target]\n    \n    # Calculate correlations\n    correlations = df[numeric_cols].corrwith(df[target]).sort_values(ascending=False)\n    \n    # Plot correlations\n    plt.figure(figsize=(12, 6))\n    correlations.plot(kind='bar')\n    plt.title('Feature Correlations with Premium Amount')\n    plt.xticks(rotation=45, ha='right')\n    plt.tight_layout()\n    plt.show()\n    \n    print(\"\\n=== Top Feature Correlations ===\")\n    print(correlations)\n    \n    # Scatter plots for top features\n    top_features = correlations.abs().nlargest(5).index\n    plt.figure(figsize=(15, 10))\n    for i, feature in enumerate(top_features, 1):\n        plt.subplot(2, 3, i)\n        plt.scatter(df[feature], df[target], alpha=0.5)\n        plt.xlabel(feature)\n        plt.ylabel(target)\n        plt.title(f'Premium Amount vs {feature}')\n    plt.tight_layout()\n    plt.show()\n\ndef analyze_categorical_impact(df, target='Premium Amount'):\n    \"\"\"Analyze impact of categorical features on target\"\"\"\n    # Find binary (one-hot encoded) columns\n    binary_cols = [col for col in df.columns if df[col].nunique() == 2 and col != target]\n    \n    # Calculate mean target value for each category\n    impact_dict = {}\n    for col in binary_cols:\n        impact = df.groupby(col)[target].agg(['mean', 'count', 'std']).round(2)\n        impact_dict[col] = impact\n    \n    # Sort features by their impact on target\n    feature_impacts = {col: abs(impact['mean'].diff()).max() \n                      for col, impact in impact_dict.items()}\n    top_features = dict(sorted(feature_impacts.items(), \n                             key=lambda x: x[1], reverse=True)[:10])\n    \n    # Plot top categorical features\n    plt.figure(figsize=(15, 6))\n    plt.bar(top_features.keys(), top_features.values())\n    plt.xticks(rotation=45, ha='right')\n    plt.title('Top 10 Categorical Features by Impact on Premium Amount')\n    plt.tight_layout()\n    plt.show()\n    \n    print(\"\\n=== Top Categorical Feature Impacts ===\")\n    for feature in top_features.keys():\n        print(f\"\\n{feature}:\")\n        print(impact_dict[feature])\n\ndef main():\n    # Load preprocessed data\n    train_df = train_processed  # Using the previously processed data\n    \n    # Analyze target distribution\n    analyze_target_distribution(train_df)\n    \n    # Analyze numeric relationships\n    analyze_numeric_relationships(train_df)\n    \n    # Analyze categorical impact\n    analyze_categorical_impact(train_df)\n    \n    return train_df\n\nif __name__ == \"__main__\":\n    train_df = main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T23:22:25.81503Z","iopub.execute_input":"2024-12-15T23:22:25.817125Z","iopub.status.idle":"2024-12-15T23:22:48.008138Z","shell.execute_reply.started":"2024-12-15T23:22:25.81707Z","shell.execute_reply":"2024-12-15T23:22:48.006835Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Engineering Pipeline","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.preprocessing import StandardScaler\n\ndef create_interaction_features(df):\n    \"\"\"Create interaction features between important numeric columns\"\"\"\n    numeric_cols = ['Age', 'Annual Income', 'Health Score', 'Previous Claims', \n                   'Credit Score', 'Insurance Duration', 'Vehicle Age']\n    \n    for i, col1 in enumerate(numeric_cols):\n        for col2 in numeric_cols[i+1:]:\n            if col1 in df.columns and col2 in df.columns:\n                df[f'{col1}_{col2}_interaction'] = df[col1] * df[col2]\n    \n    return df\n\ndef create_ratio_features(df):\n    \"\"\"Create meaningful ratio features\"\"\"\n    # Income related ratios\n    if 'Annual Income' in df.columns:\n        df['Income_per_Dependent'] = df['Annual Income'] / (df['Number of Dependents'] + 1)\n        df['Income_per_Age'] = df['Annual Income'] / (df['Age'] + 1)\n    \n    # Health related ratios\n    if 'Health Score' in df.columns:\n        df['Health_Age_Ratio'] = df['Health Score'] / (df['Age'] + 1)\n    \n    # Insurance related ratios\n    if 'Previous Claims' in df.columns and 'Insurance Duration' in df.columns:\n        df['Claims_per_Duration'] = df['Previous Claims'] / (df['Insurance Duration'] + 1)\n        \n    # Credit related ratios\n    if 'Credit Score' in df.columns and 'Annual Income' in df.columns:\n        df['Credit_Income_Ratio'] = df['Credit Score'] / (df['Annual Income'] + 1)\n    \n    return df\n\ndef create_binned_features(df, train_mode=True, bin_edges=None):\n    \"\"\"Create binned versions of numeric features\"\"\"\n    bin_edges_dict = {}\n    \n    # Age bins\n    if 'Age' in df.columns:\n        if train_mode:\n            df['Age_Bin'], bin_edges_dict['Age'] = pd.qcut(df['Age'], q=5, \n                                                          labels=['Very Young', 'Young', 'Middle', 'Senior', 'Elderly'],\n                                                          retbins=True)\n        else:\n            df['Age_Bin'] = pd.cut(df['Age'], bins=bin_edges['Age'], \n                                  labels=['Very Young', 'Young', 'Middle', 'Senior', 'Elderly'])\n    \n    # Income bins\n    if 'Annual Income' in df.columns:\n        if train_mode:\n            df['Income_Bin'], bin_edges_dict['Income'] = pd.qcut(df['Annual Income'], q=5,\n                                                                labels=['Low', 'Lower Middle', 'Middle', 'Upper Middle', 'High'],\n                                                                retbins=True)\n        else:\n            df['Income_Bin'] = pd.cut(df['Annual Income'], bins=bin_edges['Income'],\n                                    labels=['Low', 'Lower Middle', 'Middle', 'Upper Middle', 'High'])\n    \n    # Health Score bins\n    if 'Health Score' in df.columns:\n        if train_mode:\n            df['Health_Bin'], bin_edges_dict['Health'] = pd.qcut(df['Health Score'], q=5,\n                                                                labels=['Poor', 'Fair', 'Good', 'Very Good', 'Excellent'],\n                                                                retbins=True)\n        else:\n            df['Health_Bin'] = pd.cut(df['Health Score'], bins=bin_edges['Health'],\n                                    labels=['Poor', 'Fair', 'Good', 'Very Good', 'Excellent'])\n    \n    # Convert to dummy variables\n    bin_columns = [col for col in ['Age_Bin', 'Income_Bin', 'Health_Bin'] if col in df.columns]\n    if bin_columns:\n        df = pd.get_dummies(df, columns=bin_columns)\n    \n    if train_mode:\n        return df, bin_edges_dict\n    return df\n\ndef create_complex_features(df):\n    \"\"\"Create more complex feature combinations\"\"\"\n    # Standardize numeric features for complex calculations\n    scaler = StandardScaler()\n    numeric_cols = ['Age', 'Health Score', 'Previous Claims', 'Credit Score', 'Annual Income']\n    temp_df = df.copy()\n    \n    for col in numeric_cols:\n        if col in df.columns:\n            temp_df[col] = scaler.fit_transform(df[[col]])\n    \n    # Risk score (using standardized values)\n    if all(col in df.columns for col in ['Age', 'Previous Claims', 'Health Score']):\n        df['Risk_Score'] = (\n            temp_df['Age'] * 0.3 +\n            temp_df['Previous Claims'] * 0.4 +\n            (-temp_df['Health Score']) * 0.3  # Negative because lower health score means higher risk\n        )\n    \n    # Customer profile score (using standardized values)\n    if all(col in df.columns for col in ['Credit Score', 'Annual Income', 'Previous Claims']):\n        df['Customer_Profile_Score'] = (\n            temp_df['Credit Score'] * 0.4 +\n            temp_df['Annual Income'] * 0.4 +\n            (-temp_df['Previous Claims']) * 0.2  # Negative because more claims means lower score\n        )\n    \n    return df\n\ndef create_seasonal_features(df):\n    \"\"\"Create seasonal features from Policy_Month\"\"\"\n    if 'Policy_Month' in df.columns:\n        # Create season mapping\n        season_mapping = {\n            1: 'Winter', 2: 'Winter', 3: 'Spring',\n            4: 'Spring', 5: 'Spring', 6: 'Summer',\n            7: 'Summer', 8: 'Summer', 9: 'Fall',\n            10: 'Fall', 11: 'Fall', 12: 'Winter'\n        }\n        \n        # Map months to seasons\n        df['Season'] = df['Policy_Month'].map(season_mapping)\n        \n        # Create dummy variables for seasons\n        df = pd.get_dummies(df, columns=['Season'], prefix='Season')\n    \n    return df\n\ndef feature_engineering_pipeline(train_df, test_df):\n    \"\"\"Complete feature engineering pipeline\"\"\"\n    print(\"Starting feature engineering pipeline...\")\n    \n    # Make copies to avoid modifying original data\n    train = train_df.copy()\n    test = test_df.copy()\n    \n    # Store target variable\n    if 'Premium Amount' in train.columns:\n        target = train['Premium Amount']\n        train.drop('Premium Amount', axis=1, inplace=True)\n    \n    # Create features for training set first\n    print(\"Processing training set...\")\n    train = create_interaction_features(train)\n    train = create_ratio_features(train)\n    train, bin_edges = create_binned_features(train, train_mode=True)\n    train = create_complex_features(train)\n    train = create_seasonal_features(train)\n    \n    # Create features for test set using training set parameters\n    print(\"Processing test set...\")\n    test = create_interaction_features(test)\n    test = create_ratio_features(test)\n    test = create_binned_features(test, train_mode=False, bin_edges=bin_edges)\n    test = create_complex_features(test)\n    test = create_seasonal_features(test)\n    \n    # Add target back to training set\n    train['Premium Amount'] = target\n    \n    print(\"\\nFeature engineering completed.\")\n    print(f\"Training set shape: {train.shape}\")\n    print(f\"Test set shape: {test.shape}\")\n    \n    # Print new features\n    print(\"\\nNew features created:\")\n    new_features = sorted(set(train.columns) - set(train_df.columns))\n    print(\"\\n\".join(new_features))\n    \n    return train, test\n\n# Run the feature engineering pipeline\ntrain_engineered, test_engineered = feature_engineering_pipeline(train_processed, test_processed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T23:29:42.40919Z","iopub.execute_input":"2024-12-15T23:29:42.409679Z","iopub.status.idle":"2024-12-15T23:29:45.835044Z","shell.execute_reply.started":"2024-12-15T23:29:42.409639Z","shell.execute_reply":"2024-12-15T23:29:45.833834Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Execution","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, callbacks, regularizers\nimport matplotlib.pyplot as plt\n\ndef prepare_data_for_nn(train_df, test_df):\n    \"\"\"Prepare data for neural network\"\"\"\n    train_ids = train_df.index if 'id' not in train_df.columns else train_df['id']\n    test_ids = test_df.index if 'id' not in test_df.columns else test_df['id']\n    \n    if 'id' in train_df.columns:\n        train_df = train_df.drop('id', axis=1)\n    if 'id' in test_df.columns:\n        test_df = test_df.drop('id', axis=1)\n    \n    X = train_df.drop(['Premium Amount'], axis=1)\n    y = train_df['Premium Amount']\n    X_test = test_df\n    \n    X_train, X_val, y_train, y_val = train_test_split(\n        X, y, test_size=0.2, random_state=42\n    )\n    \n    scaler = StandardScaler()\n    X_train_scaled = scaler.fit_transform(X_train)\n    X_val_scaled = scaler.transform(X_val)\n    X_test_scaled = scaler.transform(X_test)\n    \n    y_train_log = np.log1p(y_train)\n    y_val_log = np.log1p(y_val)\n    \n    return (X_train_scaled, X_val_scaled, X_test_scaled, \n            y_train_log, y_val_log, scaler, train_ids, test_ids)\n\ndef squeeze_excite_block(input_tensor, ratio=16):\n    \"\"\"Create squeeze and excite block\"\"\"\n    filters = input_tensor.shape[-1]\n    se = layers.Reshape((1, filters))(input_tensor)\n    se = layers.GlobalAveragePooling1D()(se)\n    se = layers.Dense(max(filters // ratio, 1), activation='selu')(se)\n    se = layers.Dense(filters, activation='sigmoid')(se)\n    se = layers.Reshape((filters,))(se)\n    return layers.multiply([input_tensor, se])\n\ndef create_residual_block(x, units, dropout_rate=0.3):\n    \"\"\"Create residual block with squeeze-excite\"\"\"\n    # Main path with increased regularization\n    y = layers.Dense(\n        units,\n        activation='selu',\n        kernel_regularizer=regularizers.l2(5e-4)\n    )(x)\n    y = layers.BatchNormalization()(y)\n    y = layers.Dropout(dropout_rate)(y)\n    \n    y = layers.Dense(\n        units,\n        activation='selu',\n        kernel_regularizer=regularizers.l2(5e-4)\n    )(y)\n    y = layers.BatchNormalization()(y)\n    \n    # Add squeeze-excite block\n    y = squeeze_excite_block(y)\n    \n    # Transform input if needed\n    if x.shape[-1] != units:\n        x = layers.Dense(units)(x)\n    \n    # Add skip connection\n    out = layers.Add()([x, y])\n    out = layers.Activation('selu')(out)\n    out = layers.Dropout(dropout_rate)(out)\n    \n    return out\n\ndef create_improved_model(input_dim):\n    \"\"\"Create improved neural network model\"\"\"\n    inputs = layers.Input(shape=(input_dim,))\n    \n    # Initial dense layer with increased width and dropout\n    x = layers.Dense(\n        768,\n        activation='selu',\n        kernel_regularizer=regularizers.l2(5e-4)\n    )(inputs)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.6)(x)\n    \n    # Residual blocks with decreasing units\n    x = create_residual_block(x, 512, dropout_rate=0.5)\n    x = create_residual_block(x, 256, dropout_rate=0.4)\n    x = create_residual_block(x, 128, dropout_rate=0.3)\n    \n    # Extra processing before output\n    x = layers.Dense(64, activation='selu',\n                    kernel_regularizer=regularizers.l2(5e-4))(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.2)(x)\n    \n    outputs = layers.Dense(1)(x)\n    \n    model = models.Model(inputs=inputs, outputs=outputs)\n    return model\n\ndef rmsle(y_true, y_pred):\n    \"\"\"Custom RMSLE metric\"\"\"\n    return tf.sqrt(tf.reduce_mean(tf.square(y_pred - y_true)))\n\ndef train_improved_model(model, X_train, X_val, y_train, y_val):\n    \"\"\"Train model with improved settings\"\"\"\n    callbacks_list = [\n        # Early stopping\n        callbacks.EarlyStopping(\n            monitor='val_loss',\n            patience=25,\n            restore_best_weights=True,\n            min_delta=1e-5\n        ),\n        \n        # Model checkpoint\n        callbacks.ModelCheckpoint(\n            'best_model.keras',\n            monitor='val_loss',\n            save_best_only=True,\n            mode='min',\n            verbose=1\n        ),\n        \n        # Learning rate reduction\n        callbacks.ReduceLROnPlateau(\n            monitor='val_loss',\n            factor=0.2,\n            patience=7,\n            min_lr=1e-6,\n            verbose=1\n        ),\n        \n        # Cosine decay with warmup\n        callbacks.LearningRateScheduler(\n            lambda epoch: 0.0005 * (1 + np.cos((epoch * np.pi) / 300)),\n            verbose=0\n        )\n    ]\n    \n    # Use constant initial learning rate\n    optimizer = tf.keras.optimizers.Adam(\n        learning_rate=0.0005,  # Fixed initial learning rate\n        beta_1=0.9,\n        beta_2=0.999,\n        epsilon=1e-7\n    )\n    \n    model.compile(\n        optimizer=optimizer,\n        loss='mse',\n        metrics=[rmsle]\n    )\n    \n    history = model.fit(\n        X_train, y_train,\n        epochs=300,\n        batch_size=1024,\n        validation_data=(X_val, y_val),\n        callbacks=callbacks_list,\n        verbose=1\n    )\n    \n    return model, history\n\ndef plot_training_history(history):\n    \"\"\"Plot training history\"\"\"\n    plt.figure(figsize=(12, 4))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(history.history['loss'], label='Training Loss')\n    plt.plot(history.history['val_loss'], label='Validation Loss')\n    plt.title('Model Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    plt.subplot(1, 2, 2)\n    plt.plot(history.history['rmsle'], label='Training RMSLE')\n    plt.plot(history.history['val_rmsle'], label='Validation RMSLE')\n    plt.title('Model RMSLE')\n    plt.xlabel('Epoch')\n    plt.ylabel('RMSLE')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.show()\n\ndef make_predictions(model, X_test_scaled):\n    \"\"\"Make predictions\"\"\"\n    predictions_log = model.predict(X_test_scaled, verbose=1)\n    predictions = np.expm1(predictions_log)\n    return predictions\n\ndef create_submission_file(predictions, test_ids):\n    \"\"\"Create submission file\"\"\"\n    submission_df = pd.DataFrame({\n        'id': range(1200000, 2000000),\n        'Premium Amount': predictions.flatten()\n    })\n    return submission_df\n\ndef main():\n    print(\"Preparing data...\")\n    (X_train_scaled, X_val_scaled, X_test_scaled,\n     y_train_log, y_val_log, scaler, train_ids, test_ids) = prepare_data_for_nn(train_engineered, test_engineered)\n    \n    print(\"\\nCreating improved model...\")\n    model = create_improved_model(X_train_scaled.shape[1])\n    model.summary()\n    \n    print(\"\\nTraining model...\")\n    model, history = train_improved_model(model, X_train_scaled, X_val_scaled, \n                                        y_train_log, y_val_log)\n    \n    plot_training_history(history)\n    \n    print(\"\\nMaking predictions...\")\n    predictions = make_predictions(model, X_test_scaled)\n    \n    submission_df = create_submission_file(predictions, test_ids)\n    submission_df.to_csv('submission.csv', index=False)\n    print(\"\\nSubmission file created!\")\n    \n    return model, history, submission_df\n\n# Run the pipeline\nmodel, history, submission_df = main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T03:38:32.980658Z","iopub.execute_input":"2024-12-16T03:38:32.981421Z","iopub.status.idle":"2024-12-16T05:49:51.972215Z","shell.execute_reply.started":"2024-12-16T03:38:32.981368Z","shell.execute_reply":"2024-12-16T05:49:51.968176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}