{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from autogluon.tabular import TabularPredictor\n\n# Define the path to your dataset\ntrain_path = \"/kaggle/input/playground-series-s4e12/train.csv\"\ntest_path = \"/kaggle/input/playground-series-s4e12/test.csv\"\n\n# Specify the models and hyperparameters for GPU usage\ncustom_hyperparameters = {\n    'GBM': {  # LightGBM with GPU\n        'num_boost_round': 100,\n        'learning_rate': 0.05,\n        'tree_method': 'gpu_hist',  # Enable GPU\n        'predictor': 'gpu_predictor',\n    },\n    'CAT': {  # CatBoost with GPU\n        'task_type': 'GPU',\n        'iterations': 1000,\n        'learning_rate': 0.1,\n    },\n    'XGB': {  # XGBoost with GPU\n        'n_estimators': 500,\n        'max_depth': 6,\n        'tree_method': 'gpu_hist',  # Enable GPU\n    },\n    'NN_TORCH': {  # Neural Networks with GPU\n        'use_gpu': True,  # Enable GPU\n        'num_epochs': 20,\n    },\n}\n\n# Train the TabularPredictor\npredictor = TabularPredictor(label=\"Premium Amount\", problem_type=\"regression\").fit(\n    train_data=train_path,\n    time_limit=600,  # 1-hour time limit\n    presets='high_quality',  # Optimize for high accuracy\n)\n\n# Make predictions\npredictions = predictor.predict(test_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T09:10:57.266156Z","iopub.execute_input":"2024-12-02T09:10:57.2665Z","iopub.status.idle":"2024-12-02T09:22:58.43785Z","shell.execute_reply.started":"2024-12-02T09:10:57.266469Z","shell.execute_reply":"2024-12-02T09:22:58.436986Z"}},"outputs":[{"name":"stderr","text":"No path specified. Models will be saved in: \"AutogluonModels/ag-20241202_091100\"\nVerbosity: 2 (Standard Logging)\n=================== System Info ===================\nAutoGluon Version:  1.2\nPython Version:     3.10.14\nOperating System:   Linux\nPlatform Machine:   x86_64\nPlatform Version:   #1 SMP PREEMPT_DYNAMIC Sun Nov 10 10:07:59 UTC 2024\nCPU Count:          4\nMemory Avail:       30.16 GB / 31.35 GB (96.2%)\nDisk Space Avail:   19.50 GB / 19.52 GB (99.9%)\n===================================================\nPresets specified: ['high_quality']\nLoaded data from: /kaggle/input/playground-series-s4e12/train.csv | Columns = 21 / 21 | Rows = 1200000 -> 1200000\nSetting dynamic_stacking from 'auto' to True. Reason: Enable dynamic_stacking when use_bag_holdout is disabled. (use_bag_holdout=False)\nStack configuration (auto_stack=True): num_stack_levels=1, num_bag_folds=8, num_bag_sets=1\nNote: `save_bag_folds=False`! This will greatly reduce peak disk usage during fit (by ~8x), but runs the risk of an out-of-memory error during model refit if memory is small relative to the data size.\n\tYou can avoid this risk by setting `save_bag_folds=True`.\nDyStack is enabled (dynamic_stacking=True). AutoGluon will try to determine whether the input data is affected by stacked overfitting and enable or disable stacking as a consequence.\n\tThis is used to identify the optimal `num_stack_levels` value. Copies of AutoGluon will be fit on subsets of the data. Then holdout validation data is used to detect stacked overfitting.\n\tRunning DyStack for up to 150s of the 600s of remaining time (25%).\n2024-12-02 09:11:07,752\tINFO util.py:124 -- Outdated packages:\n  ipywidgets==7.7.1 found, needs ipywidgets>=8\nRun `pip install -U ipywidgets`, then restart the notebook server for rich notebook output.\n\tRunning DyStack sub-fit in a ray process to avoid memory leakage. Enabling ray logging (enable_ray_logging=True). Specify `ds_args={'enable_ray_logging': False}` if you experience logging issues.\n2024-12-02 09:11:11,729\tINFO worker.py:1744 -- Started a local Ray instance. View the dashboard at \u001b[1m\u001b[32m127.0.0.1:8265 \u001b[39m\u001b[22m\n\t\tContext path: \"/kaggle/working/AutogluonModels/ag-20241202_091100/ds_sub_fit/sub_fit_ho\"\n\u001b[36m(_dystack pid=258)\u001b[0m Running DyStack sub-fit ...\n\u001b[36m(_dystack pid=258)\u001b[0m Beginning AutoGluon training ... Time limit = 144s\n\u001b[36m(_dystack pid=258)\u001b[0m AutoGluon will save models to \"/kaggle/working/AutogluonModels/ag-20241202_091100/ds_sub_fit/sub_fit_ho\"\n\u001b[36m(_dystack pid=258)\u001b[0m Train Data Rows:    1066666\n\u001b[36m(_dystack pid=258)\u001b[0m Train Data Columns: 20\n\u001b[36m(_dystack pid=258)\u001b[0m Label Column:       Premium Amount\n\u001b[36m(_dystack pid=258)\u001b[0m Problem Type:       regression\n\u001b[36m(_dystack pid=258)\u001b[0m Preprocessing data ...\n\u001b[36m(_dystack pid=258)\u001b[0m Using Feature Generators to preprocess the data ...\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting AutoMLPipelineFeatureGenerator...\n\u001b[36m(_dystack pid=258)\u001b[0m \tAvailable Memory:                    30134.00 MB\n\u001b[36m(_dystack pid=258)\u001b[0m \tTrain Data (Original)  Memory Usage: 789.87 MB (2.6% of available memory)\n\u001b[36m(_dystack pid=258)\u001b[0m \tInferring data type of each feature based on column values. Set feature_metadata_in to manually specify special dtypes of the features.\n\u001b[36m(_dystack pid=258)\u001b[0m \tStage 1 Generators:\n\u001b[36m(_dystack pid=258)\u001b[0m \t\tFitting AsTypeFeatureGenerator...\n\u001b[36m(_dystack pid=258)\u001b[0m \t\t\tNote: Converting 2 features to boolean dtype as they only contain 2 unique values.\n\u001b[36m(_dystack pid=258)\u001b[0m \tStage 2 Generators:\n\u001b[36m(_dystack pid=258)\u001b[0m \t\tFitting FillNaFeatureGenerator...\n\u001b[36m(_dystack pid=258)\u001b[0m \tStage 3 Generators:\n\u001b[36m(_dystack pid=258)\u001b[0m \t\tFitting IdentityFeatureGenerator...\n\u001b[36m(_dystack pid=258)\u001b[0m \t\tFitting CategoryFeatureGenerator...\n\u001b[36m(_dystack pid=258)\u001b[0m \t\t\tFitting CategoryMemoryMinimizeFeatureGenerator...\n\u001b[36m(_dystack pid=258)\u001b[0m \t\tFitting DatetimeFeatureGenerator...\n\u001b[36m(_dystack pid=258)\u001b[0m \tStage 4 Generators:\n\u001b[36m(_dystack pid=258)\u001b[0m \t\tFitting DropUniqueFeatureGenerator...\n\u001b[36m(_dystack pid=258)\u001b[0m \tStage 5 Generators:\n\u001b[36m(_dystack pid=258)\u001b[0m \t\tFitting DropDuplicatesFeatureGenerator...\n\u001b[36m(_dystack pid=258)\u001b[0m \tTypes of features in original data (raw dtype, special dtypes):\n\u001b[36m(_dystack pid=258)\u001b[0m \t\t('float', [])                      :  8 | ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', ...]\n\u001b[36m(_dystack pid=258)\u001b[0m \t\t('int', [])                        :  1 | ['id']\n\u001b[36m(_dystack pid=258)\u001b[0m \t\t('object', [])                     : 10 | ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location', ...]\n\u001b[36m(_dystack pid=258)\u001b[0m \t\t('object', ['datetime_as_object']) :  1 | ['Policy Start Date']\n\u001b[36m(_dystack pid=258)\u001b[0m \tTypes of features in processed data (raw dtype, special dtypes):\n\u001b[36m(_dystack pid=258)\u001b[0m \t\t('category', [])             : 8 | ['Marital Status', 'Education Level', 'Occupation', 'Location', 'Policy Type', ...]\n\u001b[36m(_dystack pid=258)\u001b[0m \t\t('float', [])                : 8 | ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', ...]\n\u001b[36m(_dystack pid=258)\u001b[0m \t\t('int', [])                  : 1 | ['id']\n\u001b[36m(_dystack pid=258)\u001b[0m \t\t('int', ['bool'])            : 2 | ['Gender', 'Smoking Status']\n\u001b[36m(_dystack pid=258)\u001b[0m \t\t('int', ['datetime_as_int']) : 5 | ['Policy Start Date', 'Policy Start Date.year', 'Policy Start Date.month', 'Policy Start Date.day', 'Policy Start Date.dayofweek']\n\u001b[36m(_dystack pid=258)\u001b[0m \t8.2s = Fit runtime\n\u001b[36m(_dystack pid=258)\u001b[0m \t20 features in original data used to generate 24 features in processed data.\n\u001b[36m(_dystack pid=258)\u001b[0m \tTrain Data (Processed) Memory Usage: 124.11 MB (0.4% of available memory)\n\u001b[36m(_dystack pid=258)\u001b[0m Data preprocessing and feature engineering runtime = 8.71s ...\n\u001b[36m(_dystack pid=258)\u001b[0m AutoGluon will gauge predictive performance using evaluation metric: 'root_mean_squared_error'\n\u001b[36m(_dystack pid=258)\u001b[0m \tThis metric's sign has been flipped to adhere to being higher_is_better. The metric score can be multiplied by -1 to get the metric value.\n\u001b[36m(_dystack pid=258)\u001b[0m \tTo change this, specify the eval_metric parameter of Predictor()\n\u001b[36m(_dystack pid=258)\u001b[0m Large model count detected (112 configs) ... Only displaying the first 3 models of each family. To see all, set `verbosity=3`.\n\u001b[36m(_dystack pid=258)\u001b[0m User-specified model hyperparameters to be fit:\n\u001b[36m(_dystack pid=258)\u001b[0m {\n\u001b[36m(_dystack pid=258)\u001b[0m \t'NN_TORCH': [{}, {'activation': 'elu', 'dropout_prob': 0.10077639529843717, 'hidden_size': 108, 'learning_rate': 0.002735937344002146, 'num_layers': 4, 'use_batchnorm': True, 'weight_decay': 1.356433327634438e-12, 'ag_args': {'name_suffix': '_r79', 'priority': -2}}, {'activation': 'elu', 'dropout_prob': 0.11897478034205347, 'hidden_size': 213, 'learning_rate': 0.0010474382260641949, 'num_layers': 4, 'use_batchnorm': False, 'weight_decay': 5.594471067786272e-10, 'ag_args': {'name_suffix': '_r22', 'priority': -7}}],\n\u001b[36m(_dystack pid=258)\u001b[0m \t'GBM': [{'extra_trees': True, 'ag_args': {'name_suffix': 'XT'}}, {}, {'learning_rate': 0.03, 'num_leaves': 128, 'feature_fraction': 0.9, 'min_data_in_leaf': 3, 'ag_args': {'name_suffix': 'Large', 'priority': 0, 'hyperparameter_tune_kwargs': None}}],\n\u001b[36m(_dystack pid=258)\u001b[0m \t'CAT': [{}, {'depth': 6, 'grow_policy': 'SymmetricTree', 'l2_leaf_reg': 2.1542798306067823, 'learning_rate': 0.06864209415792857, 'max_ctr_complexity': 4, 'one_hot_max_size': 10, 'ag_args': {'name_suffix': '_r177', 'priority': -1}}, {'depth': 8, 'grow_policy': 'Depthwise', 'l2_leaf_reg': 2.7997999596449104, 'learning_rate': 0.031375015734637225, 'max_ctr_complexity': 2, 'one_hot_max_size': 3, 'ag_args': {'name_suffix': '_r9', 'priority': -5}}],\n\u001b[36m(_dystack pid=258)\u001b[0m \t'XGB': [{}, {'colsample_bytree': 0.6917311125174739, 'enable_categorical': False, 'learning_rate': 0.018063876087523967, 'max_depth': 10, 'min_child_weight': 0.6028633586934382, 'ag_args': {'name_suffix': '_r33', 'priority': -8}}, {'colsample_bytree': 0.6628423832084077, 'enable_categorical': False, 'learning_rate': 0.08775715546881824, 'max_depth': 5, 'min_child_weight': 0.6294123374222513, 'ag_args': {'name_suffix': '_r89', 'priority': -16}}],\n\u001b[36m(_dystack pid=258)\u001b[0m \t'FASTAI': [{}, {'bs': 256, 'emb_drop': 0.5411770367537934, 'epochs': 43, 'layers': [800, 400], 'lr': 0.01519848858318159, 'ps': 0.23782946566604385, 'ag_args': {'name_suffix': '_r191', 'priority': -4}}, {'bs': 2048, 'emb_drop': 0.05070411322605811, 'epochs': 29, 'layers': [200, 100], 'lr': 0.08974235041576624, 'ps': 0.10393466140748028, 'ag_args': {'name_suffix': '_r102', 'priority': -11}}],\n\u001b[36m(_dystack pid=258)\u001b[0m \t'RF': [{'criterion': 'gini', 'ag_args': {'name_suffix': 'Gini', 'problem_types': ['binary', 'multiclass']}}, {'criterion': 'entropy', 'ag_args': {'name_suffix': 'Entr', 'problem_types': ['binary', 'multiclass']}}, {'criterion': 'squared_error', 'ag_args': {'name_suffix': 'MSE', 'problem_types': ['regression', 'quantile']}}],\n\u001b[36m(_dystack pid=258)\u001b[0m \t'XT': [{'criterion': 'gini', 'ag_args': {'name_suffix': 'Gini', 'problem_types': ['binary', 'multiclass']}}, {'criterion': 'entropy', 'ag_args': {'name_suffix': 'Entr', 'problem_types': ['binary', 'multiclass']}}, {'criterion': 'squared_error', 'ag_args': {'name_suffix': 'MSE', 'problem_types': ['regression', 'quantile']}}],\n\u001b[36m(_dystack pid=258)\u001b[0m \t'KNN': [{'weights': 'uniform', 'ag_args': {'name_suffix': 'Unif'}}, {'weights': 'distance', 'ag_args': {'name_suffix': 'Dist'}}],\n\u001b[36m(_dystack pid=258)\u001b[0m }\n\u001b[36m(_dystack pid=258)\u001b[0m AutoGluon will fit 2 stack levels (L1 to L2) ...\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting 108 L1 models, fit_strategy=\"sequential\" ...\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting model: KNeighborsUnif_BAG_L1 ... Training model for up to 90.28s of the 135.44s of remaining time.\n\u001b[36m(_dystack pid=258)\u001b[0m \t-944.9716\t = Validation score   (-root_mean_squared_error)\n\u001b[36m(_dystack pid=258)\u001b[0m \t2.61s\t = Training   runtime\n\u001b[36m(_dystack pid=258)\u001b[0m \t6.18s\t = Validation runtime\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting model: KNeighborsDist_BAG_L1 ... Training model for up to 80.80s of the 125.96s of remaining time.\n\u001b[36m(_dystack pid=258)\u001b[0m \t-968.4786\t = Validation score   (-root_mean_squared_error)\n\u001b[36m(_dystack pid=258)\u001b[0m \t2.04s\t = Training   runtime\n\u001b[36m(_dystack pid=258)\u001b[0m \t6.56s\t = Validation runtime\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting model: LightGBMXT_BAG_L1 ... Training model for up to 71.52s of the 116.68s of remaining time.\n\u001b[36m(_dystack pid=258)\u001b[0m \tFitting 8 child models (S1F1 - S1F8) | Fitting with ParallelLocalFoldFittingStrategy (4 workers, per: cpus=1, gpus=0, memory=2.90%)\n\u001b[36m(_ray_fit pid=640)\u001b[0m \tRan out of time, early stopping on iteration 161. Best iteration is:\n\u001b[36m(_ray_fit pid=640)\u001b[0m \t[161]\tvalid_set's rmse: 844.836\n\u001b[36m(_ray_fit pid=791)\u001b[0m \tRan out of time, early stopping on iteration 157. Best iteration is:\u001b[32m [repeated 4x across cluster] (Ray deduplicates logs by default. Set RAY_DEDUP_LOGS=0 to disable log deduplication, or see https://docs.ray.io/en/master/ray-observability/user-guides/configure-logging.html#log-deduplication for more options.)\u001b[0m\n\u001b[36m(_ray_fit pid=791)\u001b[0m \t[157]\tvalid_set's rmse: 843.592\u001b[32m [repeated 4x across cluster]\u001b[0m\n\u001b[36m(_dystack pid=258)\u001b[0m \t-841.2996\t = Validation score   (-root_mean_squared_error)\n\u001b[36m(_dystack pid=258)\u001b[0m \t69.54s\t = Training   runtime\n\u001b[36m(_dystack pid=258)\u001b[0m \t17.2s\t = Validation runtime\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting model: WeightedEnsemble_L2 ... Training model for up to 135.45s of the 41.29s of remaining time.\n\u001b[36m(_dystack pid=258)\u001b[0m \tEnsemble Weights: {'LightGBMXT_BAG_L1': 1.0}\n\u001b[36m(_dystack pid=258)\u001b[0m \t-841.2996\t = Validation score   (-root_mean_squared_error)\n\u001b[36m(_dystack pid=258)\u001b[0m \t0.77s\t = Training   runtime\n\u001b[36m(_dystack pid=258)\u001b[0m \t0.03s\t = Validation runtime\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting 106 L2 models, fit_strategy=\"sequential\" ...\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting model: LightGBMXT_BAG_L2 ... Training model for up to 40.43s of the 40.34s of remaining time.\n\u001b[36m(_dystack pid=258)\u001b[0m \tFitting 8 child models (S1F1 - S1F8) | Fitting with ParallelLocalFoldFittingStrategy (4 workers, per: cpus=1, gpus=0, memory=3.19%)\n\u001b[36m(_ray_fit pid=1084)\u001b[0m \tRan out of time, early stopping on iteration 60. Best iteration is:\u001b[32m [repeated 4x across cluster]\u001b[0m\n\u001b[36m(_ray_fit pid=1084)\u001b[0m \t[60]\tvalid_set's rmse: 839.082\u001b[32m [repeated 4x across cluster]\u001b[0m\n\u001b[36m(_ray_fit pid=1205)\u001b[0m \tRan out of time, early stopping on iteration 61. Best iteration is:\u001b[32m [repeated 4x across cluster]\u001b[0m\n\u001b[36m(_ray_fit pid=1205)\u001b[0m \t[61]\tvalid_set's rmse: 838.701\u001b[32m [repeated 4x across cluster]\u001b[0m\n\u001b[36m(_dystack pid=258)\u001b[0m \t-839.9859\t = Validation score   (-root_mean_squared_error)\n\u001b[36m(_dystack pid=258)\u001b[0m \t42.08s\t = Training   runtime\n\u001b[36m(_dystack pid=258)\u001b[0m \t6.25s\t = Validation runtime\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting model: WeightedEnsemble_L3 ... Training model for up to 135.45s of the -6.25s of remaining time.\n\u001b[36m(_dystack pid=258)\u001b[0m \tEnsemble Weights: {'LightGBMXT_BAG_L2': 1.0}\n\u001b[36m(_dystack pid=258)\u001b[0m \t-839.9859\t = Validation score   (-root_mean_squared_error)\n\u001b[36m(_dystack pid=258)\u001b[0m \t0.94s\t = Training   runtime\n\u001b[36m(_dystack pid=258)\u001b[0m \t0.02s\t = Validation runtime\n\u001b[36m(_dystack pid=258)\u001b[0m AutoGluon training complete, total runtime = 151.7s ... Best model: WeightedEnsemble_L3 | Estimated inference throughput: 5322.2 rows/s (133334 batch size)\n\u001b[36m(_dystack pid=258)\u001b[0m Automatically performing refit_full as a post-fit operation (due to `.fit(..., refit_full=True)`\n\u001b[36m(_dystack pid=258)\u001b[0m Refitting models via `predictor.refit_full` using all of the data (combined train and validation)...\n\u001b[36m(_dystack pid=258)\u001b[0m \tModels trained in this way will have the suffix \"_FULL\" and have NaN validation score.\n\u001b[36m(_dystack pid=258)\u001b[0m \tThis process is not bound by time_limit, but should take less time than the original `predictor.fit` call.\n\u001b[36m(_dystack pid=258)\u001b[0m \tTo learn more, refer to the `.refit_full` method docstring which explains how \"_FULL\" models differ from normal models.\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting model: KNeighborsUnif_BAG_L1_FULL | Skipping fit via cloning parent ...\n\u001b[36m(_dystack pid=258)\u001b[0m \t2.61s\t = Training   runtime\n\u001b[36m(_dystack pid=258)\u001b[0m \t6.18s\t = Validation runtime\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting model: KNeighborsDist_BAG_L1_FULL | Skipping fit via cloning parent ...\n\u001b[36m(_dystack pid=258)\u001b[0m \t2.04s\t = Training   runtime\n\u001b[36m(_dystack pid=258)\u001b[0m \t6.56s\t = Validation runtime\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting 1 L1 models, fit_strategy=\"sequential\" ...\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting model: LightGBMXT_BAG_L1_FULL ...\n\u001b[36m(_dystack pid=258)\u001b[0m \t11.1s\t = Training   runtime\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting model: WeightedEnsemble_L2_FULL | Skipping fit via cloning parent ...\n\u001b[36m(_dystack pid=258)\u001b[0m \tEnsemble Weights: {'LightGBMXT_BAG_L1': 1.0}\n\u001b[36m(_dystack pid=258)\u001b[0m \t0.77s\t = Training   runtime\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting 1 L2 models, fit_strategy=\"sequential\" ...\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting model: LightGBMXT_BAG_L2_FULL ...\n\u001b[36m(_ray_fit pid=1211)\u001b[0m \tRan out of time, early stopping on iteration 60. Best iteration is:\u001b[32m [repeated 3x across cluster]\u001b[0m\n\u001b[36m(_ray_fit pid=1211)\u001b[0m \t[60]\tvalid_set's rmse: 837.594\u001b[32m [repeated 3x across cluster]\u001b[0m\n\u001b[36m(_dystack pid=258)\u001b[0m \t5.42s\t = Training   runtime\n\u001b[36m(_dystack pid=258)\u001b[0m Fitting model: WeightedEnsemble_L3_FULL | Skipping fit via cloning parent ...\n\u001b[36m(_dystack pid=258)\u001b[0m \tEnsemble Weights: {'LightGBMXT_BAG_L2': 1.0}\n\u001b[36m(_dystack pid=258)\u001b[0m \t0.94s\t = Training   runtime\n\u001b[36m(_dystack pid=258)\u001b[0m Updated best model to \"LightGBMXT_BAG_L2_FULL\" (Previously \"WeightedEnsemble_L3\"). AutoGluon will default to using \"LightGBMXT_BAG_L2_FULL\" for predict() and predict_proba().\n\u001b[36m(_dystack pid=258)\u001b[0m Refit complete, total runtime = 18.72s ... Best model: \"LightGBMXT_BAG_L2_FULL\"\n\u001b[36m(_dystack pid=258)\u001b[0m TabularPredictor saved. To load, use: predictor = TabularPredictor.load(\"/kaggle/working/AutogluonModels/ag-20241202_091100/ds_sub_fit/sub_fit_ho\")\n\u001b[36m(_dystack pid=258)\u001b[0m Deleting DyStack predictor artifacts (clean_up_fits=True) ...\nLeaderboard on holdout data (DyStack):\n                        model  score_holdout   score_val              eval_metric  pred_time_test  pred_time_val   fit_time  pred_time_test_marginal  pred_time_val_marginal  fit_time_marginal  stack_level  can_infer  fit_order\n0      LightGBMXT_BAG_L2_FULL    -838.835756 -839.985909  root_mean_squared_error        2.496820            NaN  21.168870                 0.200756                     NaN           5.418919            2       True          5\n1    WeightedEnsemble_L3_FULL    -838.835756 -839.985909  root_mean_squared_error        2.500174            NaN  22.106966                 0.003353                     NaN           0.938096            3       True          6\n2      LightGBMXT_BAG_L1_FULL    -839.957107 -841.299644  root_mean_squared_error        0.476803            NaN  11.102288                 0.476803                     NaN          11.102288            1       True          3\n3    WeightedEnsemble_L2_FULL    -839.957107 -841.299644  root_mean_squared_error        0.480121            NaN  11.872379                 0.003318                     NaN           0.770091            2       True          4\n4  KNeighborsUnif_BAG_L1_FULL    -940.960723 -944.971585  root_mean_squared_error        0.910275        6.17894   2.611096                 0.910275                 6.17894           2.611096            1       True          1\n5  KNeighborsDist_BAG_L1_FULL    -964.219833 -968.478614  root_mean_squared_error        0.908986        6.56189   2.036567                 0.908986                 6.56189           2.036567            1       True          2\n\t1\t = Optimal   num_stack_levels (Stacked Overfitting Occurred: False)\n\t183s\t = DyStack   runtime |\t417s\t = Remaining runtime\nStarting main fit with num_stack_levels=1.\n\tFor future fit calls on this dataset, you can skip DyStack to save time: `predictor.fit(..., dynamic_stacking=False, num_stack_levels=1)`\nBeginning AutoGluon training ... Time limit = 417s\nAutoGluon will save models to \"/kaggle/working/AutogluonModels/ag-20241202_091100\"\nTrain Data Rows:    1200000\nTrain Data Columns: 20\nLabel Column:       Premium Amount\nProblem Type:       regression\nPreprocessing data ...\nUsing Feature Generators to preprocess the data ...\nFitting AutoMLPipelineFeatureGenerator...\n\tAvailable Memory:                    30499.09 MB\n\tTrain Data (Original)  Memory Usage: 888.65 MB (2.9% of available memory)\n\tInferring data type of each feature based on column values. Set feature_metadata_in to manually specify special dtypes of the features.\n\tStage 1 Generators:\n\t\tFitting AsTypeFeatureGenerator...\n\t\t\tNote: Converting 2 features to boolean dtype as they only contain 2 unique values.\n\tStage 2 Generators:\n\t\tFitting FillNaFeatureGenerator...\n\tStage 3 Generators:\n\t\tFitting IdentityFeatureGenerator...\n\t\tFitting CategoryFeatureGenerator...\n\t\t\tFitting CategoryMemoryMinimizeFeatureGenerator...\n\t\tFitting DatetimeFeatureGenerator...\n\tStage 4 Generators:\n\t\tFitting DropUniqueFeatureGenerator...\n\tStage 5 Generators:\n\t\tFitting DropDuplicatesFeatureGenerator...\n\tTypes of features in original data (raw dtype, special dtypes):\n\t\t('float', [])                      :  8 | ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', ...]\n\t\t('int', [])                        :  1 | ['id']\n\t\t('object', [])                     : 10 | ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location', ...]\n\t\t('object', ['datetime_as_object']) :  1 | ['Policy Start Date']\n\tTypes of features in processed data (raw dtype, special dtypes):\n\t\t('category', [])             : 8 | ['Marital Status', 'Education Level', 'Occupation', 'Location', 'Policy Type', ...]\n\t\t('float', [])                : 8 | ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Previous Claims', ...]\n\t\t('int', [])                  : 1 | ['id']\n\t\t('int', ['bool'])            : 2 | ['Gender', 'Smoking Status']\n\t\t('int', ['datetime_as_int']) : 5 | ['Policy Start Date', 'Policy Start Date.year', 'Policy Start Date.month', 'Policy Start Date.day', 'Policy Start Date.dayofweek']\n\t8.5s = Fit runtime\n\t20 features in original data used to generate 24 features in processed data.\n\tTrain Data (Processed) Memory Usage: 139.62 MB (0.5% of available memory)\nData preprocessing and feature engineering runtime = 8.97s ...\nAutoGluon will gauge predictive performance using evaluation metric: 'root_mean_squared_error'\n\tThis metric's sign has been flipped to adhere to being higher_is_better. The metric score can be multiplied by -1 to get the metric value.\n\tTo change this, specify the eval_metric parameter of Predictor()\nLarge model count detected (112 configs) ... Only displaying the first 3 models of each family. To see all, set `verbosity=3`.\nUser-specified model hyperparameters to be fit:\n{\n\t'NN_TORCH': [{}, {'activation': 'elu', 'dropout_prob': 0.10077639529843717, 'hidden_size': 108, 'learning_rate': 0.002735937344002146, 'num_layers': 4, 'use_batchnorm': True, 'weight_decay': 1.356433327634438e-12, 'ag_args': {'name_suffix': '_r79', 'priority': -2}}, {'activation': 'elu', 'dropout_prob': 0.11897478034205347, 'hidden_size': 213, 'learning_rate': 0.0010474382260641949, 'num_layers': 4, 'use_batchnorm': False, 'weight_decay': 5.594471067786272e-10, 'ag_args': {'name_suffix': '_r22', 'priority': -7}}],\n\t'GBM': [{'extra_trees': True, 'ag_args': {'name_suffix': 'XT'}}, {}, {'learning_rate': 0.03, 'num_leaves': 128, 'feature_fraction': 0.9, 'min_data_in_leaf': 3, 'ag_args': {'name_suffix': 'Large', 'priority': 0, 'hyperparameter_tune_kwargs': None}}],\n\t'CAT': [{}, {'depth': 6, 'grow_policy': 'SymmetricTree', 'l2_leaf_reg': 2.1542798306067823, 'learning_rate': 0.06864209415792857, 'max_ctr_complexity': 4, 'one_hot_max_size': 10, 'ag_args': {'name_suffix': '_r177', 'priority': -1}}, {'depth': 8, 'grow_policy': 'Depthwise', 'l2_leaf_reg': 2.7997999596449104, 'learning_rate': 0.031375015734637225, 'max_ctr_complexity': 2, 'one_hot_max_size': 3, 'ag_args': {'name_suffix': '_r9', 'priority': -5}}],\n\t'XGB': [{}, {'colsample_bytree': 0.6917311125174739, 'enable_categorical': False, 'learning_rate': 0.018063876087523967, 'max_depth': 10, 'min_child_weight': 0.6028633586934382, 'ag_args': {'name_suffix': '_r33', 'priority': -8}}, {'colsample_bytree': 0.6628423832084077, 'enable_categorical': False, 'learning_rate': 0.08775715546881824, 'max_depth': 5, 'min_child_weight': 0.6294123374222513, 'ag_args': {'name_suffix': '_r89', 'priority': -16}}],\n\t'FASTAI': [{}, {'bs': 256, 'emb_drop': 0.5411770367537934, 'epochs': 43, 'layers': [800, 400], 'lr': 0.01519848858318159, 'ps': 0.23782946566604385, 'ag_args': {'name_suffix': '_r191', 'priority': -4}}, {'bs': 2048, 'emb_drop': 0.05070411322605811, 'epochs': 29, 'layers': [200, 100], 'lr': 0.08974235041576624, 'ps': 0.10393466140748028, 'ag_args': {'name_suffix': '_r102', 'priority': -11}}],\n\t'RF': [{'criterion': 'gini', 'ag_args': {'name_suffix': 'Gini', 'problem_types': ['binary', 'multiclass']}}, {'criterion': 'entropy', 'ag_args': {'name_suffix': 'Entr', 'problem_types': ['binary', 'multiclass']}}, {'criterion': 'squared_error', 'ag_args': {'name_suffix': 'MSE', 'problem_types': ['regression', 'quantile']}}],\n\t'XT': [{'criterion': 'gini', 'ag_args': {'name_suffix': 'Gini', 'problem_types': ['binary', 'multiclass']}}, {'criterion': 'entropy', 'ag_args': {'name_suffix': 'Entr', 'problem_types': ['binary', 'multiclass']}}, {'criterion': 'squared_error', 'ag_args': {'name_suffix': 'MSE', 'problem_types': ['regression', 'quantile']}}],\n\t'KNN': [{'weights': 'uniform', 'ag_args': {'name_suffix': 'Unif'}}, {'weights': 'distance', 'ag_args': {'name_suffix': 'Dist'}}],\n}\nAutoGluon will fit 2 stack levels (L1 to L2) ...\nFitting 108 L1 models, fit_strategy=\"sequential\" ...\nFitting model: KNeighborsUnif_BAG_L1 ... Training model for up to 271.80s of the 407.80s of remaining time.\n\t-944.4065\t = Validation score   (-root_mean_squared_error)\n\t2.29s\t = Training   runtime\n\t7.0s\t = Validation runtime\nFitting model: KNeighborsDist_BAG_L1 ... Training model for up to 261.76s of the 397.76s of remaining time.\n\t-967.7972\t = Validation score   (-root_mean_squared_error)\n\t1.94s\t = Training   runtime\n\t7.65s\t = Validation runtime\nFitting model: LightGBMXT_BAG_L1 ... Training model for up to 251.42s of the 387.41s of remaining time.\n\tFitting 8 child models (S1F1 - S1F8) | Fitting with ParallelLocalFoldFittingStrategy (4 workers, per: cpus=1, gpus=0, memory=2.99%)\n\t-838.5954\t = Validation score   (-root_mean_squared_error)\n\t224.61s\t = Training   runtime\n\t103.28s\t = Validation runtime\nFitting model: LightGBM_BAG_L1 ... Training model for up to 10.65s of the 146.64s of remaining time.\n\tFitting 8 child models (S1F1 - S1F8) | Fitting with ParallelLocalFoldFittingStrategy (4 workers, per: cpus=1, gpus=0, memory=2.99%)\n\t-862.3566\t = Validation score   (-root_mean_squared_error)\n\t19.07s\t = Training   runtime\n\t0.96s\t = Validation runtime\nFitting model: WeightedEnsemble_L2 ... Training model for up to 360.00s of the 123.86s of remaining time.\n\tEnsemble Weights: {'LightGBMXT_BAG_L1': 1.0}\n\t-838.5954\t = Validation score   (-root_mean_squared_error)\n\t0.8s\t = Training   runtime\n\t0.03s\t = Validation runtime\nFitting 106 L2 models, fit_strategy=\"sequential\" ...\nFitting model: LightGBMXT_BAG_L2 ... Training model for up to 122.97s of the 122.87s of remaining time.\n\tFitting 8 child models (S1F1 - S1F8) | Fitting with ParallelLocalFoldFittingStrategy (4 workers, per: cpus=1, gpus=0, memory=3.40%)\n\t-835.6539\t = Validation score   (-root_mean_squared_error)\n\t111.36s\t = Training   runtime\n\t25.36s\t = Validation runtime\nFitting model: LightGBM_BAG_L2 ... Training model for up to 4.76s of the 4.66s of remaining time.\n\tFitting 8 child models (S1F1 - S1F8) | Fitting with ParallelLocalFoldFittingStrategy (4 workers, per: cpus=1, gpus=0, memory=3.40%)\n\t-862.2032\t = Validation score   (-root_mean_squared_error)\n\t20.94s\t = Training   runtime\n\t1.0s\t = Validation runtime\nFitting model: WeightedEnsemble_L3 ... Training model for up to 360.00s of the -20.95s of remaining time.\n\tEnsemble Weights: {'LightGBMXT_BAG_L2': 1.0}\n\t-835.6539\t = Validation score   (-root_mean_squared_error)\n\t1.16s\t = Training   runtime\n\t0.03s\t = Validation runtime\nAutoGluon training complete, total runtime = 439.25s ... Best model: WeightedEnsemble_L3 | Estimated inference throughput: 1141.3 rows/s (150000 batch size)\nAutomatically performing refit_full as a post-fit operation (due to `.fit(..., refit_full=True)`\nRefitting models via `predictor.refit_full` using all of the data (combined train and validation)...\n\tModels trained in this way will have the suffix \"_FULL\" and have NaN validation score.\n\tThis process is not bound by time_limit, but should take less time than the original `predictor.fit` call.\n\tTo learn more, refer to the `.refit_full` method docstring which explains how \"_FULL\" models differ from normal models.\nFitting model: KNeighborsUnif_BAG_L1_FULL | Skipping fit via cloning parent ...\n\t2.29s\t = Training   runtime\n\t7.0s\t = Validation runtime\nFitting model: KNeighborsDist_BAG_L1_FULL | Skipping fit via cloning parent ...\n\t1.94s\t = Training   runtime\n\t7.65s\t = Validation runtime\nFitting 1 L1 models, fit_strategy=\"sequential\" ...\nFitting model: LightGBMXT_BAG_L1_FULL ...\n\t34.64s\t = Training   runtime\nFitting 1 L1 models, fit_strategy=\"sequential\" ...\nFitting model: LightGBM_BAG_L1_FULL ...\n\t1.87s\t = Training   runtime\nFitting model: WeightedEnsemble_L2_FULL | Skipping fit via cloning parent ...\n\tEnsemble Weights: {'LightGBMXT_BAG_L1': 1.0}\n\t0.8s\t = Training   runtime\nFitting 1 L2 models, fit_strategy=\"sequential\" ...\nFitting model: LightGBMXT_BAG_L2_FULL ...\n\t15.07s\t = Training   runtime\nFitting 1 L2 models, fit_strategy=\"sequential\" ...\nFitting model: LightGBM_BAG_L2_FULL ...\n\t2.47s\t = Training   runtime\nFitting model: WeightedEnsemble_L3_FULL | Skipping fit via cloning parent ...\n\tEnsemble Weights: {'LightGBMXT_BAG_L2': 1.0}\n\t1.16s\t = Training   runtime\nUpdated best model to \"LightGBMXT_BAG_L2_FULL\" (Previously \"WeightedEnsemble_L3\"). AutoGluon will default to using \"LightGBMXT_BAG_L2_FULL\" for predict() and predict_proba().\nRefit complete, total runtime = 57.42s ... Best model: \"LightGBMXT_BAG_L2_FULL\"\nTabularPredictor saved. To load, use: predictor = TabularPredictor.load(\"/kaggle/working/AutogluonModels/ag-20241202_091100\")\nLoaded data from: /kaggle/input/playground-series-s4e12/test.csv | Columns = 20 / 20 | Rows = 800000 -> 800000\n","output_type":"stream"}],"execution_count":2},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T09:24:07.242383Z","iopub.execute_input":"2024-12-02T09:24:07.243084Z","iopub.status.idle":"2024-12-02T09:24:07.251881Z","shell.execute_reply.started":"2024-12-02T09:24:07.24305Z","shell.execute_reply":"2024-12-02T09:24:07.250977Z"}},"outputs":[{"execution_count":3,"output_type":"execute_result","data":{"text/plain":"0         1179.897827\n1         1119.586670\n2         1050.451294\n3         1093.011963\n4         1018.767639\n             ...     \n799995    1214.829712\n799996    1488.247925\n799997    1112.045532\n799998    1130.984985\n799999    1052.085571\nName: Premium Amount, Length: 800000, dtype: float32"},"metadata":{}}],"execution_count":3},{"cell_type":"code","source":"import pandas as pd\n\nsub_df = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\nsub_df['Premium Amount'] = predictions\nsub_df.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T09:24:55.748273Z","iopub.execute_input":"2024-12-02T09:24:55.749305Z","iopub.status.idle":"2024-12-02T09:24:55.917269Z","shell.execute_reply.started":"2024-12-02T09:24:55.749256Z","shell.execute_reply":"2024-12-02T09:24:55.916134Z"}},"outputs":[{"execution_count":5,"output_type":"execute_result","data":{"text/plain":"        id  Premium Amount\n0  1200000     1179.897827\n1  1200001     1119.586670\n2  1200002     1050.451294\n3  1200003     1093.011963\n4  1200004     1018.767639\n5  1200005     1138.434692\n6  1200006     1183.934082\n7  1200007     1051.517456\n8  1200008      322.907013\n9  1200009     1086.429443","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>id</th>\n      <th>Premium Amount</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>1200000</td>\n      <td>1179.897827</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>1200001</td>\n      <td>1119.586670</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>1200002</td>\n      <td>1050.451294</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>1200003</td>\n      <td>1093.011963</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>1200004</td>\n      <td>1018.767639</td>\n    </tr>\n    <tr>\n      <th>5</th>\n      <td>1200005</td>\n      <td>1138.434692</td>\n    </tr>\n    <tr>\n      <th>6</th>\n      <td>1200006</td>\n      <td>1183.934082</td>\n    </tr>\n    <tr>\n      <th>7</th>\n      <td>1200007</td>\n      <td>1051.517456</td>\n    </tr>\n    <tr>\n      <th>8</th>\n      <td>1200008</td>\n      <td>322.907013</td>\n    </tr>\n    <tr>\n      <th>9</th>\n      <td>1200009</td>\n      <td>1086.429443</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}],"execution_count":5},{"cell_type":"code","source":"sub_df.to_csv(\"autogluon-v1.csv\",index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T09:25:47.457161Z","iopub.execute_input":"2024-12-02T09:25:47.457477Z","iopub.status.idle":"2024-12-02T09:25:48.460962Z","shell.execute_reply.started":"2024-12-02T09:25:47.45745Z","shell.execute_reply":"2024-12-02T09:25:48.460264Z"}},"outputs":[],"execution_count":7}]}