{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":10064298,"sourceType":"datasetVersion","datasetId":6202384}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Adversarial Validation","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"## Part 1 - Train Data v/s Test Data","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport xgboost as xgb\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T06:47:26.74832Z","iopub.execute_input":"2024-12-01T06:47:26.74877Z","iopub.status.idle":"2024-12-01T06:47:27.994002Z","shell.execute_reply.started":"2024-12-01T06:47:26.748708Z","shell.execute_reply":"2024-12-01T06:47:27.992851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    train_filepath = '/kaggle/input/playground-series-s4e12/train.csv'\n    test_filepath = '/kaggle/input/playground-series-s4e12/test.csv'\n    original_filepath = '/kaggle/input/s4e12-original-data/Insurance Premium Prediction Dataset.csv'\n\n    target = 'Premium Amount'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:04:42.185511Z","iopub.execute_input":"2024-12-01T07:04:42.186122Z","iopub.status.idle":"2024-12-01T07:04:42.193014Z","shell.execute_reply.started":"2024-12-01T07:04:42.186078Z","shell.execute_reply":"2024-12-01T07:04:42.191596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(CFG.train_filepath, index_col=0)\ntest = pd.read_csv(CFG.test_filepath, index_col=0)\n\ntrain = train.drop([CFG.target], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:04:42.410364Z","iopub.execute_input":"2024-12-01T07:04:42.410825Z","iopub.status.idle":"2024-12-01T07:04:51.992863Z","shell.execute_reply.started":"2024-12-01T07:04:42.41078Z","shell.execute_reply":"2024-12-01T07:04:51.991558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['is_test'] = 0\ntest['is_test'] = 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:04:51.995057Z","iopub.execute_input":"2024-12-01T07:04:51.995531Z","iopub.status.idle":"2024-12-01T07:04:52.004044Z","shell.execute_reply.started":"2024-12-01T07:04:51.995476Z","shell.execute_reply":"2024-12-01T07:04:52.002987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = pd.concat([train, test], axis=0).sample(frac=1.0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:04:52.005278Z","iopub.execute_input":"2024-12-01T07:04:52.005658Z","iopub.status.idle":"2024-12-01T07:04:54.999982Z","shell.execute_reply.started":"2024-12-01T07:04:52.005622Z","shell.execute_reply":"2024-12-01T07:04:54.998537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_numeric = data.select_dtypes(exclude=['object']).copy()\ndf_obj = data.select_dtypes(include=['object']).copy()\n\nfor col in df_obj:\n    df_obj[col] = pd.factorize(df_obj[col])[0]\n\ndata = pd.concat([df_numeric, df_obj], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:04:55.002478Z","iopub.execute_input":"2024-12-01T07:04:55.00286Z","iopub.status.idle":"2024-12-01T07:04:59.88439Z","shell.execute_reply.started":"2024-12-01T07:04:55.002825Z","shell.execute_reply":"2024-12-01T07:04:59.883087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = data.drop(['is_test'], axis=1)\ny = data['is_test']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:04:59.886055Z","iopub.execute_input":"2024-12-01T07:04:59.886499Z","iopub.status.idle":"2024-12-01T07:05:00.054415Z","shell.execute_reply.started":"2024-12-01T07:04:59.886454Z","shell.execute_reply":"2024-12-01T07:05:00.053194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5,\n                      shuffle=True,\n                      random_state=42)\n\nxgb_params = {\n    'learning_rate': 0.05, \n    'max_depth': 4, \n    'subsample': 0.9,\n    'colsample_bytree': 0.9,\n    'objective': 'binary:logistic',\n    'n_estimators': 100, \n    'gamma': 1, \n    'min_child_weight': 4,\n    'verbosity': 0, \n    'enable_categorical': True,\n    'eval_metric': 'logloss', \n    'early_stopping_rounds': 10\n}\n\nclf = xgb.XGBClassifier(**xgb_params, seed=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:05:00.055984Z","iopub.execute_input":"2024-12-01T07:05:00.056361Z","iopub.status.idle":"2024-12-01T07:05:00.064487Z","shell.execute_reply.started":"2024-12-01T07:05:00.056324Z","shell.execute_reply":"2024-12-01T07:05:00.063032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"aucs = list()\n\nfor i, (train_idx, test_idx) in enumerate(skf.split(X, y)):\n    x0, x1 = X.iloc[train_idx], X.iloc[test_idx]\n    y0, y1 = y.iloc[train_idx], y.iloc[test_idx]\n    \n    clf.fit(x0, y0, eval_set=[(x1, y1)], verbose=False)\n    oof_preds = clf.predict_proba(x1)[:, 1]\n    auc = roc_auc_score(y1, oof_preds)\n    aucs.append(auc)\n    print(f'Fold {i}: AUC-ROC score = {auc}')\n\nprint(f'\\nAverage ROC-AUC score is {np.mean(aucs)} ± {np.std(aucs)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:05:00.066174Z","iopub.execute_input":"2024-12-01T07:05:00.066695Z","iopub.status.idle":"2024-12-01T07:05:34.186864Z","shell.execute_reply.started":"2024-12-01T07:05:00.066645Z","shell.execute_reply":"2024-12-01T07:05:34.185652Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> This is very much expected in a Playground Series Competition with synthetic data.","metadata":{}},{"cell_type":"markdown","source":"## Part 2 - Competition Data v/s Original Data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(CFG.train_filepath, index_col=0)\ntest = pd.read_csv(CFG.test_filepath, index_col=0)\noriginal = pd.read_csv(CFG.original_filepath)\n\ntrain = train.drop(CFG.target, axis=1)\noriginal = original.drop(CFG.target, axis=1)\n\ndata = pd.concat([train, test], axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:09:10.196883Z","iopub.execute_input":"2024-12-01T07:09:10.198009Z","iopub.status.idle":"2024-12-01T07:09:20.371695Z","shell.execute_reply.started":"2024-12-01T07:09:10.19796Z","shell.execute_reply":"2024-12-01T07:09:20.370373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data['is_org'] = 0\noriginal['is_org'] = 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:09:22.028936Z","iopub.execute_input":"2024-12-01T07:09:22.029393Z","iopub.status.idle":"2024-12-01T07:09:22.038981Z","shell.execute_reply.started":"2024-12-01T07:09:22.029354Z","shell.execute_reply":"2024-12-01T07:09:22.037804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"combined_data = pd.concat([data, original], axis=0).sample(frac=1.0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:09:29.541231Z","iopub.execute_input":"2024-12-01T07:09:29.541692Z","iopub.status.idle":"2024-12-01T07:09:33.686727Z","shell.execute_reply.started":"2024-12-01T07:09:29.541652Z","shell.execute_reply":"2024-12-01T07:09:33.685594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_numeric = combined_data.select_dtypes(exclude=['object']).copy()\ndf_obj = combined_data.select_dtypes(include=['object']).copy()\n\nfor col in df_obj:\n    df_obj[col] = pd.factorize(df_obj[col])[0]\n\ncombined_data = pd.concat([df_numeric, df_obj], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:09:35.461629Z","iopub.execute_input":"2024-12-01T07:09:35.462694Z","iopub.status.idle":"2024-12-01T07:09:40.673026Z","shell.execute_reply.started":"2024-12-01T07:09:35.462649Z","shell.execute_reply":"2024-12-01T07:09:40.671876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = combined_data.drop(['is_org'], axis=1)\ny = combined_data['is_org']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:09:40.675112Z","iopub.execute_input":"2024-12-01T07:09:40.675555Z","iopub.status.idle":"2024-12-01T07:09:40.7974Z","shell.execute_reply.started":"2024-12-01T07:09:40.675519Z","shell.execute_reply":"2024-12-01T07:09:40.796028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5,\n                      shuffle=True,\n                      random_state=42)\n\n\nxgb_params = {\n    'learning_rate': 0.05, \n    'max_depth': 4, \n    'subsample': 0.9,\n    'colsample_bytree': 0.9,\n    'objective': 'binary:logistic',\n    'n_estimators': 100, \n    'gamma': 1, \n    'min_child_weight': 4,\n    'verbosity': 0, \n    'enable_categorical': True,\n    'eval_metric': 'logloss', \n    'early_stopping_rounds': 10\n}\n\nclf = xgb.XGBClassifier(**xgb_params, seed=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:10:29.195602Z","iopub.execute_input":"2024-12-01T07:10:29.196355Z","iopub.status.idle":"2024-12-01T07:10:29.205014Z","shell.execute_reply.started":"2024-12-01T07:10:29.196294Z","shell.execute_reply":"2024-12-01T07:10:29.203461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"aucs = list()\n\nfor i, (train_idx, test_idx) in enumerate(skf.split(X, y)):\n    x0, x1 = X.iloc[train_idx], X.iloc[test_idx]\n    y0, y1 = y.iloc[train_idx], y.iloc[test_idx]\n    \n    clf.fit(x0, y0, eval_set=[(x1, y1)], verbose=False)\n    oof_preds = clf.predict_proba(x1)[:, 1]\n    auc = roc_auc_score(y1, oof_preds)\n    aucs.append(auc)\n    print(f'Fold {i}: AUC-ROC score = {auc}')\n\nprint(f'\\nAverage ROC-AUC score is {np.mean(aucs)} ± {np.std(aucs)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T07:10:37.49518Z","iopub.execute_input":"2024-12-01T07:10:37.49566Z","iopub.status.idle":"2024-12-01T07:12:41.613986Z","shell.execute_reply.started":"2024-12-01T07:10:37.495619Z","shell.execute_reply":"2024-12-01T07:12:41.612847Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> A 0.72 ROC-AUC suggests a noticeable difference in the distributions between the competition data and the original data. Additional experiments, both including and excluding the original data during modeling, are necessary to better understand its impact on our results. Notably, this ROC-AUC is higher than what was observed in the previous two Playground Series competitions.\n\n\n\n\n\n\n","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}