{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch, numpy as np, pandas as pd, matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom fastai.tabular.core import *\nfrom fastai.tabular.all import *","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:44:24.483144Z","iopub.execute_input":"2024-12-26T00:44:24.483607Z","iopub.status.idle":"2024-12-26T00:44:26.627843Z","shell.execute_reply.started":"2024-12-26T00:44:24.48357Z","shell.execute_reply":"2024-12-26T00:44:26.626794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sklearn\nprint(F'{sklearn.__version__=}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:44:26.629051Z","iopub.execute_input":"2024-12-26T00:44:26.629653Z","iopub.status.idle":"2024-12-26T00:44:26.634782Z","shell.execute_reply.started":"2024-12-26T00:44:26.629628Z","shell.execute_reply":"2024-12-26T00:44:26.63388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T01:16:13.719228Z","iopub.execute_input":"2024-12-26T01:16:13.719578Z","iopub.status.idle":"2024-12-26T01:16:13.72396Z","shell.execute_reply.started":"2024-12-26T01:16:13.719552Z","shell.execute_reply":"2024-12-26T01:16:13.722756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ntorch.cuda.empty_cache()\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T23:36:38.92165Z","iopub.execute_input":"2024-12-24T23:36:38.921934Z","iopub.status.idle":"2024-12-24T23:36:39.028895Z","shell.execute_reply.started":"2024-12-24T23:36:38.921913Z","shell.execute_reply":"2024-12-24T23:36:39.028075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = Path('/kaggle/input/playground-series-s4e12')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:44:29.181811Z","iopub.execute_input":"2024-12-26T00:44:29.18224Z","iopub.status.idle":"2024-12-26T00:44:29.186474Z","shell.execute_reply.started":"2024-12-26T00:44:29.182206Z","shell.execute_reply":"2024-12-26T00:44:29.185472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(path/'train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:44:29.50763Z","iopub.execute_input":"2024-12-26T00:44:29.507929Z","iopub.status.idle":"2024-12-26T00:44:33.111685Z","shell.execute_reply.started":"2024-12-26T00:44:29.507904Z","shell.execute_reply":"2024-12-26T00:44:33.110748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T19:20:59.101941Z","iopub.execute_input":"2024-12-24T19:20:59.102245Z","iopub.status.idle":"2024-12-24T19:20:59.474752Z","shell.execute_reply.started":"2024-12-24T19:20:59.102219Z","shell.execute_reply":"2024-12-24T19:20:59.473872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Policy Start Date'].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T19:20:59.476031Z","iopub.execute_input":"2024-12-24T19:20:59.476265Z","iopub.status.idle":"2024-12-24T19:20:59.485545Z","shell.execute_reply.started":"2024-12-24T19:20:59.476243Z","shell.execute_reply":"2024-12-24T19:20:59.484743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Month'] = df['Policy Start Date'].dt.month.astype(float)\ndf['Day'] = df['Policy Start Date'].dt.day\ndf['Week']  = df['Policy Start Date'].dt.isocalendar().week\ndf['Weekday'] = df['Policy Start Date'].dt.weekday.astype(float)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T19:20:59.486532Z","iopub.execute_input":"2024-12-24T19:20:59.486821Z","iopub.status.idle":"2024-12-24T19:20:59.732457Z","shell.execute_reply.started":"2024-12-24T19:20:59.486786Z","shell.execute_reply":"2024-12-24T19:20:59.731561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T19:20:59.73341Z","iopub.execute_input":"2024-12-24T19:20:59.733732Z","iopub.status.idle":"2024-12-24T19:20:59.740898Z","shell.execute_reply.started":"2024-12-24T19:20:59.733699Z","shell.execute_reply":"2024-12-24T19:20:59.740039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['Income-Age Ratio'] = df['Annual Income'] / df['Age']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T18:52:09.803059Z","iopub.execute_input":"2024-12-24T18:52:09.803371Z","iopub.status.idle":"2024-12-24T18:52:09.810877Z","shell.execute_reply.started":"2024-12-24T18:52:09.803347Z","shell.execute_reply":"2024-12-24T18:52:09.810022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in df:\n    if(df[col].isna().sum() > 0): df[col + '_na'] = df[col].isna().astype(float)\nmodes = df.mode().iloc[0]\ndf.fillna(modes, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T18:52:12.28731Z","iopub.execute_input":"2024-12-24T18:52:12.28761Z","iopub.status.idle":"2024-12-24T18:52:16.53654Z","shell.execute_reply.started":"2024-12-24T18:52:12.287584Z","shell.execute_reply":"2024-12-24T18:52:16.535829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in cats: print(f'{col}: {df[col].unique()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T18:52:25.257511Z","iopub.execute_input":"2024-12-24T18:52:25.257811Z","iopub.status.idle":"2024-12-24T18:52:25.797489Z","shell.execute_reply.started":"2024-12-24T18:52:25.257786Z","shell.execute_reply":"2024-12-24T18:52:25.796767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conts,cats = cont_cat_split(df)\n#conts = conts[1:-1]\nconts,cats","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T18:58:55.718744Z","iopub.execute_input":"2024-12-24T18:58:55.719055Z","iopub.status.idle":"2024-12-24T18:58:55.833873Z","shell.execute_reply.started":"2024-12-24T18:58:55.719029Z","shell.execute_reply":"2024-12-24T18:58:55.832988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T18:59:01.556431Z","iopub.execute_input":"2024-12-24T18:59:01.556753Z","iopub.status.idle":"2024-12-24T18:59:01.56231Z","shell.execute_reply.started":"2024-12-24T18:59:01.556727Z","shell.execute_reply":"2024-12-24T18:59:01.561359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in cats:\n    df[col] = pd.Categorical(df[col])\n    df[col] = df[col].cat.codes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T18:53:17.839328Z","iopub.execute_input":"2024-12-24T18:53:17.839608Z","iopub.status.idle":"2024-12-24T18:53:17.972691Z","shell.execute_reply.started":"2024-12-24T18:53:17.839586Z","shell.execute_reply":"2024-12-24T18:53:17.971984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def proc_data(df):\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['Month'] = df['Policy Start Date'].dt.month.astype(float)\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Week']  = df['Policy Start Date'].dt.isocalendar().week\n    df['Weekday'] = df['Policy Start Date'].dt.weekday.astype(float)\n    df['Income-Age Ratio'] = df['Annual Income'] / df['Age']\n    for col in df:\n        if(df[col].isna().sum() > 0): df[col + '_na'] = df[col].isna().astype(float)\n    df = df.fillna(df.mode().iloc[0])\n    conts,cats = cont_cat_split(df)\n    for col in cats:\n        df[col] = pd.Categorical(df[col])\n        df[col] = df[col].cat.codes\n    indeps = df.drop(['id', 'Policy Start Date'], axis=1)\n    deps = None\n    if 'Premium Amount' in df.columns:\n        indeps = indeps.drop('Premium Amount', axis=1)\n        deps = df['Premium Amount']\n    return indeps,deps","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:44:40.086401Z","iopub.execute_input":"2024-12-26T00:44:40.086738Z","iopub.status.idle":"2024-12-26T00:44:40.093315Z","shell.execute_reply.started":"2024-12-26T00:44:40.086713Z","shell.execute_reply":"2024-12-26T00:44:40.092278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"indeps,deps = proc_data(df)\nindeps.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:44:40.61665Z","iopub.execute_input":"2024-12-26T00:44:40.617137Z","iopub.status.idle":"2024-12-26T00:44:47.615264Z","shell.execute_reply.started":"2024-12-26T00:44:40.617074Z","shell.execute_reply":"2024-12-26T00:44:47.614244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nrf = RandomForestRegressor(5, min_samples_leaf=1000, n_jobs=-1).fit(indeps, deps)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T19:21:53.196077Z","iopub.execute_input":"2024-12-24T19:21:53.19636Z","iopub.status.idle":"2024-12-24T19:22:24.315942Z","shell.execute_reply.started":"2024-12-24T19:21:53.196337Z","shell.execute_reply":"2024-12-24T19:22:24.31508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.DataFrame(dict(cols=indeps.columns, importance=rf.feature_importances_)).plot('cols', 'importance', 'barh')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T19:22:24.317234Z","iopub.execute_input":"2024-12-24T19:22:24.317559Z","iopub.status.idle":"2024-12-24T19:22:24.930279Z","shell.execute_reply.started":"2024-12-24T19:22:24.317528Z","shell.execute_reply":"2024-12-24T19:22:24.929296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_log_error\ndef rmsle(preds, targs):\n    #preds = preds.cpu().numpy() if preds.is_cuda else preds.numpy()\n    #targs = targs.cpu().numpy() if targs.is_cuda else targs.numpy()\n    preds = np.maximum(0, preds)\n    targs = np.maximum(0, targs)\n    return np.sqrt(mean_squared_log_error(targs, preds))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:44:47.616627Z","iopub.execute_input":"2024-12-26T00:44:47.616941Z","iopub.status.idle":"2024-12-26T00:44:47.621354Z","shell.execute_reply.started":"2024-12-26T00:44:47.616909Z","shell.execute_reply":"2024-12-26T00:44:47.620189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from fastai.data.transforms import RandomSplitter\ntrain_split,valid_split = RandomSplitter(valid_pct=0.2)(df)\ntrain_indeps,train_deps = indeps.iloc[train_split],deps.iloc[train_split]\nvalid_indeps,valid_deps = indeps.iloc[valid_split],deps.iloc[valid_split]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:44:47.622931Z","iopub.execute_input":"2024-12-26T00:44:47.623244Z","iopub.status.idle":"2024-12-26T00:44:48.258762Z","shell.execute_reply.started":"2024-12-26T00:44:47.623216Z","shell.execute_reply":"2024-12-26T00:44:48.25778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"start_time = time.time()\n\nxgbrf = xgb.XGBRFRegressor(n_estimators=10, max_depth=6, learning_rate=0.1, n_jobs=-1)\nxgbrf.fit(train_indeps, train_deps)\n\npreds = xgbrf.predict(valid_indeps)\n\nrmsle_score = rmsle(valid_deps, preds)\nprint(f'RMSLE: {rmsle_score}')\n\nend_time = time.time()\nprint(end_time - start_time)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T23:24:22.75119Z","iopub.execute_input":"2024-12-24T23:24:22.75162Z","execution_failed":"2024-12-24T23:24:26.666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {\n    #'device': 'cuda',\n    #'predictor': 'gpu_predictor',\n    'n_estimators' : 10,\n    'max_depth': 3, \n    'learning_rate': 0.1,\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T00:18:14.832022Z","iopub.execute_input":"2024-12-25T00:18:14.832383Z","iopub.status.idle":"2024-12-25T00:18:14.836004Z","shell.execute_reply.started":"2024-12-25T00:18:14.832351Z","shell.execute_reply":"2024-12-25T00:18:14.835125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nlosses = {}\nfor depth in range(params['max_depth'], 30, 3):\n    start_time = time.time()\n    params['max_depth'] = depth\n    xgbrf = xgb.XGBRFRegressor(**params, n_jobs=-1)\n    xgbrf.fit(train_indeps, train_deps)\n    preds = xgbrf.predict(valid_indeps)\n    #preds = xgbrf.get_booster().inplace_predict(valid_indeps)\n    losses[depth] = rmsle(valid_deps, preds)\n    end_time = time.time()\n    print(f'tested depth: {depth:<4} elapsed time: {(end_time - start_time):.2f}s')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T00:02:49.398046Z","iopub.execute_input":"2024-12-25T00:02:49.398433Z","execution_failed":"2024-12-25T00:03:00.54Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimal_depth = min(losses, key=losses.get)\nparams['max_depth'] = optimal_depth\noptimal_depth, losses[optimal_depth]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T20:00:27.206958Z","iopub.execute_input":"2024-12-24T20:00:27.207306Z","iopub.status.idle":"2024-12-24T20:00:27.21306Z","shell.execute_reply.started":"2024-12-24T20:00:27.207273Z","shell.execute_reply":"2024-12-24T20:00:27.212249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def graph_params(losses, x_label='', log=False):\n    x, y = losses.keys(), losses.values()\n    plt.xlabel(x_label)\n    if log: plt.xscale('log')\n    plt.ylabel('RMSLE')\n    plt.plot(x, y)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T01:15:03.426496Z","iopub.execute_input":"2024-12-26T01:15:03.426786Z","iopub.status.idle":"2024-12-26T01:15:03.431562Z","shell.execute_reply.started":"2024-12-26T01:15:03.426765Z","shell.execute_reply":"2024-12-26T01:15:03.430262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"graph_params(losses, 'Depth of Random Forest')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T20:01:57.058486Z","iopub.execute_input":"2024-12-24T20:01:57.058786Z","iopub.status.idle":"2024-12-24T20:01:57.258474Z","shell.execute_reply.started":"2024-12-24T20:01:57.058763Z","shell.execute_reply":"2024-12-24T20:01:57.257466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"losses = {}\nfor depth in range(0, 30, 5):\n    xgbrf = xgb.XGBRFRegressor(n_estimators=1, max_depth=depth, learning_rate=0.1, n_jobs=-1)\n    xgbrf.fit(train_indeps, train_deps)\n    preds = xgbrf.predict(valid_indeps)\n    losses[depth] = rmsle(valid_deps, preds)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_ranges = {\n    'alpha': np.logspace(-3, 3, num=10),\n    'colsample_bytree': np.linspace(0.5, 1, num=10),\n    'gamma': np.logspace(0, 1, num=10),\n    'lambda': np.logspace(-3, 3, num=10),\n    'learning_rate': np.logspace(-0.2, 0.1, num=30),\n    'max_bins': np.logspace(1, 5, num=10),\n    'max_depth': np.linspace(1, 20, num=20, dtype=int),\n    'min_child_weight': np.linspace(1, 10, num=10),\n    #'n_estimators': np.logspace(1, 2, num=10, dtype=int)\n    'subsample': np.linspace(0.5, 1, num=10)\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T01:29:08.69909Z","iopub.execute_input":"2024-12-26T01:29:08.699542Z","iopub.status.idle":"2024-12-26T01:29:08.706495Z","shell.execute_reply.started":"2024-12-26T01:29:08.699506Z","shell.execute_reply":"2024-12-26T01:29:08.705583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"graph_settings = {\n    'alpha': {\n        'x_label': 'L1 Regularization',\n        'log': True\n    },\n    'colsample_bytree': {\n        'x_label': 'Fraction of Column Samples in Training Set'\n    },\n    'gamma': {\n        'x_label': 'Minimum Loss Required for Split',\n        'log': True\n    },\n    'lambda': {\n        'x_label': 'L2 Regularization',\n        'log': True\n    },\n    'learning_rate': {\n        'x_label': 'Learning Rate',\n        'log': True\n    },\n    'max_bins': {\n        'x_label': 'Maximum Bins',\n        'log': True\n    },\n    'max_depth': {\n        'x_label': 'Maximum Depth of Trees'\n    },\n    'min_child_weight': {\n        'x_label': 'Minimum Sum of Instance Weights in Child Nodes'\n    },\n    'n_estimators': {\n        'x_label': 'Number of Trees'\n    },\n    'subsample': {\n        'x_label': 'Fraction of Row Samples in Training Set'\n    }\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T01:29:09.401068Z","iopub.execute_input":"2024-12-26T01:29:09.401544Z","iopub.status.idle":"2024-12-26T01:29:09.408681Z","shell.execute_reply.started":"2024-12-26T01:29:09.401506Z","shell.execute_reply":"2024-12-26T01:29:09.406857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\ndef optimize_params(param_ranges, optimal_params={}, graph_settings=None, print_log=True, print_graphs=True):\n    if graph_settings == None: print_graphs = False\n    for param,vals in param_ranges.items():\n        if param not in optimal_params.keys():\n            losses = {}\n            for val in vals:\n                start_time = time.time()\n                optimal_params[param] = val\n                xgbrf = xgb.XGBRFRegressor(**optimal_params, n_jobs=-1)\n                xgbrf.fit(train_indeps, train_deps)\n                preds = xgbrf.predict(valid_indeps)\n                losses[val] = rmsle(valid_deps, preds)\n                end_time = time.time()\n                print(f'tested {param}: {val:<10.2e} rmsle: {losses[val]:<8.4f} elapsed time: {(end_time - start_time):.2f}s')\n            optimal_params[param] = min(losses, key=losses.get)\n            print(f'optimal value for {param} is {optimal_params[param]:.2f}')\n            if print_graphs: graph_params(losses, **graph_settings[param])\n    return optimal_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T01:29:18.109291Z","iopub.execute_input":"2024-12-26T01:29:18.109749Z","iopub.status.idle":"2024-12-26T01:29:18.119018Z","shell.execute_reply.started":"2024-12-26T01:29:18.109714Z","shell.execute_reply":"2024-12-26T01:29:18.118026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimal_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T01:29:27.932787Z","iopub.execute_input":"2024-12-26T01:29:27.93323Z","iopub.status.idle":"2024-12-26T01:29:27.939649Z","shell.execute_reply.started":"2024-12-26T01:29:27.933191Z","shell.execute_reply":"2024-12-26T01:29:27.938645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fixed_params = {\n    'n_estimators': 10,\n    'max_depth': 12,\n    'alpha': 0,\n    'colsample_bytree': 1,\n    'gamma': 4.5,\n    'learning_rate': 1.09,\n    'lambda': 0,\n    'max_bins': 256,\n    'min_child_weight': 1,\n    'subsample': 0.72\n}\noptimal_params = optimize_params(param_ranges, optimal_params=fixed_params, graph_settings=graph_settings)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimal_params['n_estimators'] = 200\noptimal_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:44:00.403541Z","iopub.execute_input":"2024-12-26T00:44:00.403936Z","iopub.status.idle":"2024-12-26T00:44:00.411046Z","shell.execute_reply.started":"2024-12-26T00:44:00.403905Z","shell.execute_reply":"2024-12-26T00:44:00.410126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgbrf = xgb.XGBRFRegressor(**optimal_params, n_jobs=-1)\nxgbrf.fit(indeps, deps)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:44:49.072194Z","iopub.execute_input":"2024-12-26T00:44:49.0725Z","iopub.status.idle":"2024-12-26T00:46:29.210332Z","shell.execute_reply.started":"2024-12-26T00:44:49.07248Z","shell.execute_reply":"2024-12-26T00:46:29.209339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb.plot_importance(xgbrf, importance_type='weight') # weight, gain, cover\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:48:25.349192Z","iopub.execute_input":"2024-12-26T00:48:25.34957Z","iopub.status.idle":"2024-12-26T00:48:25.897296Z","shell.execute_reply.started":"2024-12-26T00:48:25.349544Z","shell.execute_reply":"2024-12-26T00:48:25.895945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_csv(path/'test.csv')\ntest_indeps,_ = proc_data(test_df)\nsubmit_df = pd.read_csv(path/'sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:49:13.033105Z","iopub.execute_input":"2024-12-26T00:49:13.033453Z","iopub.status.idle":"2024-12-26T00:49:20.667845Z","shell.execute_reply.started":"2024-12-26T00:49:13.033429Z","shell.execute_reply":"2024-12-26T00:49:20.666755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_indeps.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:49:20.669312Z","iopub.execute_input":"2024-12-26T00:49:20.669721Z","iopub.status.idle":"2024-12-26T00:49:20.698323Z","shell.execute_reply.started":"2024-12-26T00:49:20.669681Z","shell.execute_reply":"2024-12-26T00:49:20.697358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds = xgbrf.predict(test_indeps)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:49:21.979102Z","iopub.execute_input":"2024-12-26T00:49:21.979459Z","iopub.status.idle":"2024-12-26T00:49:28.211451Z","shell.execute_reply.started":"2024-12-26T00:49:21.979431Z","shell.execute_reply":"2024-12-26T00:49:28.209553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit_df['Premium Amount'] = preds\nsubmit_df.to_csv(f'insurance_v6.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T00:49:32.092735Z","iopub.execute_input":"2024-12-26T00:49:32.09304Z","iopub.status.idle":"2024-12-26T00:49:33.055745Z","shell.execute_reply.started":"2024-12-26T00:49:32.093016Z","shell.execute_reply":"2024-12-26T00:49:33.054848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T01:08:11.435419Z","iopub.execute_input":"2024-12-26T01:08:11.435751Z","iopub.status.idle":"2024-12-26T01:08:14.351494Z","shell.execute_reply.started":"2024-12-26T01:08:11.43573Z","shell.execute_reply":"2024-12-26T01:08:14.350574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_ranges = {\n    'colsample_bytree': np.linspace(0.5, 1, num=10),\n    'learning_rate': np.logspace(-0.5, 0.1, num=30),\n    'max_bins': np.logspace(1, 5, num=10, dtype=int),\n    'max_depth': np.linspace(1, 20, num=20, dtype=int),\n    'min_data_in_leaf': np.linspace(5, 50, num=10, dtype=int),\n    'min_split_gain': np.linspace(0, 1, num=10),\n    'num_leaves': np.linspace(2, 40, num=20, dtype=int), # 2^max_depth\n    'n_estimators': np.logspace(1, 3, num=10, dtype=int),\n    'reg_alpha': np.logspace(-3, 3, num=10),\n    'reg_lambda': np.logspace(-3, 3, num=10),\n    'subsample': np.linspace(0.5, 1, num=10)\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:27:55.979109Z","iopub.execute_input":"2024-12-26T02:27:55.979443Z","iopub.status.idle":"2024-12-26T02:27:55.985901Z","shell.execute_reply.started":"2024-12-26T02:27:55.979421Z","shell.execute_reply":"2024-12-26T02:27:55.984725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"graph_settings = {\n    'alpha': {\n        'x_label': 'L1 Regularization',\n        'log': True\n    },\n    'colsample_bytree': {\n        'x_label': 'Fraction of Column Samples in Training Set'\n    },\n    'gamma': {\n        'x_label': 'Minimum Loss Required for Split',\n        'log': True\n    },\n    'lambda': {\n        'x_label': 'L2 Regularization',\n        'log': True\n    },\n    'learning_rate': {\n        'x_label': 'Learning Rate',\n        'log': True\n    },\n    'max_bins': {\n        'x_label': 'Maximum Bins',\n        'log': True\n    },\n    'max_depth': {\n        'x_label': 'Maximum Depth of Trees'\n    },\n    'min_child_weight': {\n        'x_label': 'Minimum Sum of Instance Weights in Child Nodes'\n    },\n    'min_data_in_leaf': {\n        'x_label': 'Minimum Data in Leaf'\n    },\n    'min_split_gain': {\n        'x_label': 'Minimum Sum of Instance Weights in Child Nodes'\n    },\n    'num_leaves': {\n        'x_label': 'Number of Leaves'\n    },\n    'n_estimators': {\n        'x_label': 'Number of Trees'\n    },\n    'reg_alpha': {\n        'x_label': 'L1 Regularization',\n        'log': True\n    },\n    'reg_lambda': {\n        'x_label': 'L2 Regularization',\n        'log': True\n    },\n    'subsample': {\n        'x_label': 'Fraction of Row Samples in Training Set'\n    }\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:33:45.503035Z","iopub.execute_input":"2024-12-26T02:33:45.503337Z","iopub.status.idle":"2024-12-26T02:33:45.508391Z","shell.execute_reply.started":"2024-12-26T02:33:45.503317Z","shell.execute_reply":"2024-12-26T02:33:45.507193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_model(arch, optimal_params):\n    if(arch == 'LGB'): return lgb.LGBMRegressor(**optimal_params, n_jobs=-1, verbose=-1)\n    if(arch == 'XGB'): return xgb.XGBRFRegressor(**optimal_params, n_jobs=-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:33:47.360271Z","iopub.execute_input":"2024-12-26T02:33:47.360669Z","iopub.status.idle":"2024-12-26T02:33:47.365789Z","shell.execute_reply.started":"2024-12-26T02:33:47.360639Z","shell.execute_reply":"2024-12-26T02:33:47.364633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def optimize_params(param_ranges, arch, optimal_params={}, graph_settings=None, print_log=True, print_graphs=True):\n    if graph_settings == None: print_graphs = False\n    for param,vals in param_ranges.items():\n        if param not in optimal_params.keys():\n            losses = {}\n            for val in vals:\n                start_time = time.time()\n                optimal_params[param] = val\n                model = get_model(arch=arch, optimal_params=optimal_params)\n                model.fit(train_indeps, train_deps)\n                preds = model.predict(valid_indeps)\n                losses[val] = rmsle(valid_deps, preds)\n                end_time = time.time()\n                print(f'tested {param}: {val:<10.2e} rmsle: {losses[val]:<8.4f} elapsed time: {(end_time - start_time):.2f}s')\n            optimal_params[param] = min(losses, key=losses.get)\n            print(f'optimal value for {param} is {optimal_params[param]:.2f}')\n            if print_graphs: graph_params(losses, **graph_settings[param])\n    return optimal_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:33:47.698974Z","iopub.execute_input":"2024-12-26T02:33:47.699324Z","iopub.status.idle":"2024-12-26T02:33:47.706936Z","shell.execute_reply.started":"2024-12-26T02:33:47.699292Z","shell.execute_reply":"2024-12-26T02:33:47.705827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fixed_params = {\n    'n_estimators': 10,\n    'colsample_bytree': 1,\n    'learning_rate': 1.20,\n    'max_bins': 35938,\n    'max_depth': 13 \n}\noptimize_params(param_ranges, arch='LGB', optimal_params=fixed_params, graph_settings=graph_settings)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T02:33:49.774622Z","iopub.execute_input":"2024-12-26T02:33:49.774902Z","iopub.status.idle":"2024-12-26T02:39:04.410474Z","shell.execute_reply.started":"2024-12-26T02:33:49.774883Z","shell.execute_reply":"2024-12-26T02:39:04.409501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimal_lgb = {'n_estimators': 10,\n 'colsample_bytree': 1,\n 'learning_rate': 1.2,\n 'max_bins': 35938,\n 'max_depth': 13, # everything above test more\n 'min_data_in_leaf': 30, # test more\n 'num_leaves': 16, # test more\n 'subsample': 1\n}","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}